langfuse-haystack 3.1.0__tar.gz → 3.2.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {langfuse_haystack-3.1.0 → langfuse_haystack-3.2.0}/CHANGELOG.md +19 -0
- {langfuse_haystack-3.1.0 → langfuse_haystack-3.2.0}/PKG-INFO +1 -1
- {langfuse_haystack-3.1.0 → langfuse_haystack-3.2.0}/pyproject.toml +1 -1
- {langfuse_haystack-3.1.0 → langfuse_haystack-3.2.0}/src/haystack_integrations/tracing/langfuse/tracer.py +56 -6
- {langfuse_haystack-3.1.0 → langfuse_haystack-3.2.0}/tests/test_tracer.py +41 -5
- {langfuse_haystack-3.1.0 → langfuse_haystack-3.2.0}/.gitignore +0 -0
- {langfuse_haystack-3.1.0 → langfuse_haystack-3.2.0}/LICENSE.txt +0 -0
- {langfuse_haystack-3.1.0 → langfuse_haystack-3.2.0}/README.md +0 -0
- {langfuse_haystack-3.1.0 → langfuse_haystack-3.2.0}/example/basic_rag.py +0 -0
- {langfuse_haystack-3.1.0 → langfuse_haystack-3.2.0}/example/chat.py +0 -0
- {langfuse_haystack-3.1.0 → langfuse_haystack-3.2.0}/example/requirements.txt +0 -0
- {langfuse_haystack-3.1.0 → langfuse_haystack-3.2.0}/pydoc/config.yml +0 -0
- {langfuse_haystack-3.1.0 → langfuse_haystack-3.2.0}/pydoc/config_docusaurus.yml +0 -0
- {langfuse_haystack-3.1.0 → langfuse_haystack-3.2.0}/src/haystack_integrations/components/connectors/__init__.py +0 -0
- {langfuse_haystack-3.1.0 → langfuse_haystack-3.2.0}/src/haystack_integrations/components/connectors/langfuse/__init__.py +0 -0
- {langfuse_haystack-3.1.0 → langfuse_haystack-3.2.0}/src/haystack_integrations/components/connectors/langfuse/langfuse_connector.py +0 -0
- {langfuse_haystack-3.1.0 → langfuse_haystack-3.2.0}/src/haystack_integrations/components/connectors/py.typed +0 -0
- {langfuse_haystack-3.1.0 → langfuse_haystack-3.2.0}/src/haystack_integrations/tracing/langfuse/__init__.py +0 -0
- {langfuse_haystack-3.1.0 → langfuse_haystack-3.2.0}/src/haystack_integrations/tracing/py.typed +0 -0
- {langfuse_haystack-3.1.0 → langfuse_haystack-3.2.0}/tests/__init__.py +0 -0
- {langfuse_haystack-3.1.0 → langfuse_haystack-3.2.0}/tests/test_langfuse_connector.py +0 -0
- {langfuse_haystack-3.1.0 → langfuse_haystack-3.2.0}/tests/test_tracing.py +0 -0
|
@@ -1,5 +1,24 @@
|
|
|
1
1
|
# Changelog
|
|
2
2
|
|
|
3
|
+
## [integrations/langfuse-v3.1.0] - 2025-10-24
|
|
4
|
+
|
|
5
|
+
### 🐛 Bug Fixes
|
|
6
|
+
|
|
7
|
+
- Langfuse - add py.typed; fix testing with lowest deps (#2458)
|
|
8
|
+
|
|
9
|
+
### 📚 Documentation
|
|
10
|
+
|
|
11
|
+
- Add pydoc configurations for Docusaurus (#2411)
|
|
12
|
+
|
|
13
|
+
### ⚙️ CI
|
|
14
|
+
|
|
15
|
+
- Install dependencies in the `test` environment when testing with lowest direct dependencies and Haystack main (#2418)
|
|
16
|
+
|
|
17
|
+
### 🧹 Chores
|
|
18
|
+
|
|
19
|
+
- Remove ruff exclude and fix linting in Langfuse integration (#2257)
|
|
20
|
+
|
|
21
|
+
|
|
3
22
|
## [integrations/langfuse-v3.0.0] - 2025-09-19
|
|
4
23
|
|
|
5
24
|
### 🌀 Miscellaneous
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: langfuse-haystack
|
|
3
|
-
Version: 3.
|
|
3
|
+
Version: 3.2.0
|
|
4
4
|
Summary: Langfuse integration for Haystack
|
|
5
5
|
Project-URL: Documentation, https://github.com/deepset-ai/haystack-core-integrations/tree/main/integrations/langfuse#readme
|
|
6
6
|
Project-URL: Issues, https://github.com/deepset-ai/haystack-core-integrations/issues
|
|
@@ -67,7 +67,7 @@ dependencies = [
|
|
|
67
67
|
unit = 'pytest -m "not integration" {args:tests}'
|
|
68
68
|
integration = 'pytest -m "integration" {args:tests}'
|
|
69
69
|
all = 'pytest {args:tests}'
|
|
70
|
-
cov-retry = '
|
|
70
|
+
cov-retry = 'pytest --cov=haystack_integrations --reruns 3 --reruns-delay 30 -x {args:tests}'
|
|
71
71
|
|
|
72
72
|
types = "mypy -p haystack_integrations.components.connectors.langfuse -p haystack_integrations.tracing.langfuse {args}"
|
|
73
73
|
|
|
@@ -10,7 +10,7 @@ from contextlib import AbstractContextManager
|
|
|
10
10
|
from contextvars import ContextVar
|
|
11
11
|
from dataclasses import dataclass
|
|
12
12
|
from datetime import datetime
|
|
13
|
-
from typing import Any, Dict, Iterator, List, Optional
|
|
13
|
+
from typing import Any, Dict, Iterator, List, Literal, Optional
|
|
14
14
|
|
|
15
15
|
from haystack import default_from_dict, default_to_dict, logging
|
|
16
16
|
from haystack.dataclasses import ChatMessage
|
|
@@ -256,6 +256,46 @@ class SpanHandler(ABC):
|
|
|
256
256
|
return default_to_dict(self)
|
|
257
257
|
|
|
258
258
|
|
|
259
|
+
def _sanitize_usage_data(usage: Dict[str, Any]) -> Dict[str, Any]:
|
|
260
|
+
"""
|
|
261
|
+
Sanitize usage data for Langfuse by flattening to a single-level dictionary.
|
|
262
|
+
|
|
263
|
+
Langfuse's usage_details must be a flat dictionary with only numeric values. This function:
|
|
264
|
+
- Flattens nested dictionaries using dot notation (e.g., cache_creation.input_tokens)
|
|
265
|
+
- Keeps int and float values
|
|
266
|
+
- Skips None, boolean, string, and other non-numeric types
|
|
267
|
+
|
|
268
|
+
:param usage: Raw usage dictionary from the provider.
|
|
269
|
+
:returns: Flat dictionary with only numeric values (int or float).
|
|
270
|
+
"""
|
|
271
|
+
if not isinstance(usage, dict):
|
|
272
|
+
return {}
|
|
273
|
+
|
|
274
|
+
sanitized: Dict[str, Any] = {}
|
|
275
|
+
|
|
276
|
+
def _flatten(data: Dict[str, Any], prefix: str = "") -> None:
|
|
277
|
+
"""Recursively flatten nested dictionaries."""
|
|
278
|
+
for key, value in data.items():
|
|
279
|
+
full_key = f"{prefix}.{key}" if prefix else key
|
|
280
|
+
|
|
281
|
+
if value is None:
|
|
282
|
+
# Skip None values (e.g., Anthropic's server_tool_use)
|
|
283
|
+
continue
|
|
284
|
+
elif isinstance(value, bool):
|
|
285
|
+
# Skip boolean values
|
|
286
|
+
continue
|
|
287
|
+
elif isinstance(value, (int, float)):
|
|
288
|
+
# Keep numeric values
|
|
289
|
+
sanitized[full_key] = value
|
|
290
|
+
elif isinstance(value, dict):
|
|
291
|
+
# Recursively flatten nested dicts
|
|
292
|
+
_flatten(value, full_key)
|
|
293
|
+
# Skip strings and other non-numeric types (e.g., Anthropic's service_tier)
|
|
294
|
+
|
|
295
|
+
_flatten(usage)
|
|
296
|
+
return sanitized
|
|
297
|
+
|
|
298
|
+
|
|
259
299
|
class DefaultSpanHandler(SpanHandler):
|
|
260
300
|
"""DefaultSpanHandler provides the default Langfuse tracing behavior for Haystack."""
|
|
261
301
|
|
|
@@ -271,10 +311,12 @@ class DefaultSpanHandler(SpanHandler):
|
|
|
271
311
|
# Get external tracing context for root trace creation (correlation metadata)
|
|
272
312
|
tracing_ctx = tracing_context_var.get({})
|
|
273
313
|
if not context.parent_span:
|
|
314
|
+
root_span_type: Literal["agent", "span"] = (
|
|
315
|
+
"agent" if context.operation_name == "haystack.agent.run" else "span"
|
|
316
|
+
)
|
|
274
317
|
# Create a new trace when there's no parent span
|
|
275
|
-
span_context_manager = self.tracer.
|
|
276
|
-
name=context.trace_name,
|
|
277
|
-
version=tracing_ctx.get("version"),
|
|
318
|
+
span_context_manager = self.tracer.start_as_current_observation(
|
|
319
|
+
name=context.trace_name, version=tracing_ctx.get("version"), as_type=root_span_type
|
|
278
320
|
)
|
|
279
321
|
|
|
280
322
|
# Create LangfuseSpan which will handle entering the context manager
|
|
@@ -298,6 +340,10 @@ class DefaultSpanHandler(SpanHandler):
|
|
|
298
340
|
span._span.update_trace(**trace_attrs)
|
|
299
341
|
|
|
300
342
|
return span
|
|
343
|
+
elif context.component_type == "ToolInvoker":
|
|
344
|
+
return LangfuseSpan(self.tracer.start_as_current_observation(name=context.name, as_type="tool"))
|
|
345
|
+
elif context.operation_name == "haystack.agent.run":
|
|
346
|
+
return LangfuseSpan(self.tracer.start_as_current_observation(name=context.name, as_type="agent"))
|
|
301
347
|
elif context.component_type in _ALL_SUPPORTED_GENERATORS:
|
|
302
348
|
return LangfuseSpan(self.tracer.start_as_current_observation(name=context.name, as_type="generation"))
|
|
303
349
|
else:
|
|
@@ -328,7 +374,9 @@ class DefaultSpanHandler(SpanHandler):
|
|
|
328
374
|
if component_type in _SUPPORTED_GENERATORS:
|
|
329
375
|
meta = span.get_data().get(_COMPONENT_OUTPUT_KEY, {}).get("meta")
|
|
330
376
|
if meta:
|
|
331
|
-
|
|
377
|
+
usage = meta[0].get("usage")
|
|
378
|
+
sanitized_usage = _sanitize_usage_data(usage) if usage else None
|
|
379
|
+
span.raw_span().update(usage_details=sanitized_usage, model=meta[0].get("model"))
|
|
332
380
|
|
|
333
381
|
if component_type in _SUPPORTED_CHAT_GENERATORS:
|
|
334
382
|
replies = span.get_data().get(_COMPONENT_OUTPUT_KEY, {}).get("replies")
|
|
@@ -341,8 +389,10 @@ class DefaultSpanHandler(SpanHandler):
|
|
|
341
389
|
except ValueError:
|
|
342
390
|
logger.error(f"Failed to parse completion_start_time: {completion_start_time}")
|
|
343
391
|
completion_start_time = None
|
|
392
|
+
usage = meta.get("usage")
|
|
393
|
+
sanitized_usage = _sanitize_usage_data(usage) if usage else None
|
|
344
394
|
span.raw_span().update(
|
|
345
|
-
|
|
395
|
+
usage_details=sanitized_usage,
|
|
346
396
|
model=meta.get("model"),
|
|
347
397
|
completion_start_time=completion_start_time,
|
|
348
398
|
)
|
|
@@ -18,6 +18,7 @@ from haystack_integrations.tracing.langfuse.tracer import (
|
|
|
18
18
|
LangfuseSpan,
|
|
19
19
|
LangfuseTracer,
|
|
20
20
|
SpanContext,
|
|
21
|
+
_sanitize_usage_data,
|
|
21
22
|
)
|
|
22
23
|
|
|
23
24
|
|
|
@@ -190,6 +191,41 @@ class TestSpanContext:
|
|
|
190
191
|
)
|
|
191
192
|
|
|
192
193
|
|
|
194
|
+
class TestSanitizeUsageData:
|
|
195
|
+
def test_anthropic_usage_flattens_and_filters(self):
|
|
196
|
+
"""Test Anthropic's nested dict with None and strings gets flattened and filtered"""
|
|
197
|
+
usage = {
|
|
198
|
+
"cache_creation": {"ephemeral_1h_input_tokens": 0, "ephemeral_5m_input_tokens": 0},
|
|
199
|
+
"cache_creation_input_tokens": 0,
|
|
200
|
+
"cache_read_input_tokens": 0,
|
|
201
|
+
"server_tool_use": None, # Should be filtered
|
|
202
|
+
"service_tier": "standard", # Should be filtered
|
|
203
|
+
"prompt_tokens": 25,
|
|
204
|
+
"completion_tokens": 449,
|
|
205
|
+
}
|
|
206
|
+
result = _sanitize_usage_data(usage)
|
|
207
|
+
assert result == {
|
|
208
|
+
"cache_creation.ephemeral_1h_input_tokens": 0,
|
|
209
|
+
"cache_creation.ephemeral_5m_input_tokens": 0,
|
|
210
|
+
"cache_creation_input_tokens": 0,
|
|
211
|
+
"cache_read_input_tokens": 0,
|
|
212
|
+
"prompt_tokens": 25,
|
|
213
|
+
"completion_tokens": 449,
|
|
214
|
+
}
|
|
215
|
+
|
|
216
|
+
def test_openai_usage_preserved(self):
|
|
217
|
+
"""Test OpenAI/Cohere flat dict with only numeric values works unchanged"""
|
|
218
|
+
usage = {"prompt_tokens": 29, "completion_tokens": 267, "total_tokens": 296}
|
|
219
|
+
result = _sanitize_usage_data(usage)
|
|
220
|
+
assert result == {"prompt_tokens": 29, "completion_tokens": 267, "total_tokens": 296}
|
|
221
|
+
|
|
222
|
+
def test_empty_and_invalid_input(self):
|
|
223
|
+
"""Test edge cases return empty dict"""
|
|
224
|
+
assert _sanitize_usage_data({}) == {}
|
|
225
|
+
assert _sanitize_usage_data(None) == {}
|
|
226
|
+
assert _sanitize_usage_data({"only_strings": "value", "only_none": None}) == {}
|
|
227
|
+
|
|
228
|
+
|
|
193
229
|
class TestDefaultSpanHandler:
|
|
194
230
|
def test_handle_generator(self):
|
|
195
231
|
mock_span = Mock()
|
|
@@ -203,7 +239,7 @@ class TestDefaultSpanHandler:
|
|
|
203
239
|
handler.handle(mock_span, component_type="OpenAIGenerator")
|
|
204
240
|
|
|
205
241
|
assert mock_span.update.call_count == 1
|
|
206
|
-
assert mock_span.update.call_args_list[0][1] == {"
|
|
242
|
+
assert mock_span.update.call_args_list[0][1] == {"usage_details": None, "model": "test_model"}
|
|
207
243
|
|
|
208
244
|
def test_handle_chat_generator(self):
|
|
209
245
|
mock_span = Mock()
|
|
@@ -225,7 +261,7 @@ class TestDefaultSpanHandler:
|
|
|
225
261
|
|
|
226
262
|
assert mock_span.update.call_count == 1
|
|
227
263
|
assert mock_span.update.call_args_list[0][1] == {
|
|
228
|
-
"
|
|
264
|
+
"usage_details": None,
|
|
229
265
|
"model": "test_model",
|
|
230
266
|
"completion_start_time": datetime.datetime( # noqa: DTZ001
|
|
231
267
|
2021, 7, 27, 16, 2, 8, 12345
|
|
@@ -254,7 +290,7 @@ class TestDefaultSpanHandler:
|
|
|
254
290
|
|
|
255
291
|
assert mock_span.update.call_count == 1
|
|
256
292
|
assert mock_span.update.call_args_list[0][1] == {
|
|
257
|
-
"
|
|
293
|
+
"usage_details": None,
|
|
258
294
|
"model": "test_model",
|
|
259
295
|
"completion_start_time": None,
|
|
260
296
|
}
|
|
@@ -348,7 +384,7 @@ class TestLangfuseTracer:
|
|
|
348
384
|
}
|
|
349
385
|
with tracer.trace(operation_name="operation_name", tags=tags) as span:
|
|
350
386
|
...
|
|
351
|
-
assert span.raw_span()._data["
|
|
387
|
+
assert span.raw_span()._data["usage_details"] is None
|
|
352
388
|
assert span.raw_span()._data["model"] == "test_model"
|
|
353
389
|
assert span.raw_span()._data["completion_start_time"] == datetime.datetime(2021, 7, 27, 16, 2, 8, 12345) # noqa: DTZ001
|
|
354
390
|
|
|
@@ -415,7 +451,7 @@ class TestLangfuseTracer:
|
|
|
415
451
|
}
|
|
416
452
|
with tracer.trace(operation_name="operation_name", tags=tags) as span:
|
|
417
453
|
...
|
|
418
|
-
assert span.raw_span()._data["
|
|
454
|
+
assert span.raw_span()._data["usage_details"] is None
|
|
419
455
|
assert span.raw_span()._data["model"] == "test_model"
|
|
420
456
|
assert span.raw_span()._data["completion_start_time"] is None
|
|
421
457
|
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{langfuse_haystack-3.1.0 → langfuse_haystack-3.2.0}/src/haystack_integrations/tracing/py.typed
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|