langfuse-haystack 5.2.0__tar.gz → 7.0.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {langfuse_haystack-5.2.0 → langfuse_haystack-7.0.0}/CHANGELOG.md +51 -8
- {langfuse_haystack-5.2.0 → langfuse_haystack-7.0.0}/LICENSE.txt +1 -1
- {langfuse_haystack-5.2.0 → langfuse_haystack-7.0.0}/PKG-INFO +3 -3
- {langfuse_haystack-5.2.0 → langfuse_haystack-7.0.0}/pyproject.toml +2 -1
- {langfuse_haystack-5.2.0 → langfuse_haystack-7.0.0}/src/haystack_integrations/components/connectors/langfuse/langfuse_connector.py +20 -8
- {langfuse_haystack-5.2.0 → langfuse_haystack-7.0.0}/src/haystack_integrations/tracing/langfuse/tracer.py +191 -88
- {langfuse_haystack-5.2.0 → langfuse_haystack-7.0.0}/tests/test_langfuse_connector.py +78 -38
- {langfuse_haystack-5.2.0 → langfuse_haystack-7.0.0}/tests/test_tracer.py +199 -130
- {langfuse_haystack-5.2.0 → langfuse_haystack-7.0.0}/.gitignore +0 -0
- {langfuse_haystack-5.2.0 → langfuse_haystack-7.0.0}/README.md +0 -0
- {langfuse_haystack-5.2.0 → langfuse_haystack-7.0.0}/example/basic_rag.py +0 -0
- {langfuse_haystack-5.2.0 → langfuse_haystack-7.0.0}/example/chat.py +0 -0
- {langfuse_haystack-5.2.0 → langfuse_haystack-7.0.0}/example/requirements.txt +0 -0
- {langfuse_haystack-5.2.0 → langfuse_haystack-7.0.0}/pydoc/config_docusaurus.yml +0 -0
- {langfuse_haystack-5.2.0 → langfuse_haystack-7.0.0}/src/haystack_integrations/components/connectors/langfuse/__init__.py +0 -0
- {langfuse_haystack-5.2.0 → langfuse_haystack-7.0.0}/src/haystack_integrations/components/connectors/py.typed +0 -0
- {langfuse_haystack-5.2.0 → langfuse_haystack-7.0.0}/src/haystack_integrations/tracing/langfuse/__init__.py +0 -0
- {langfuse_haystack-5.2.0 → langfuse_haystack-7.0.0}/src/haystack_integrations/tracing/py.typed +0 -0
- {langfuse_haystack-5.2.0 → langfuse_haystack-7.0.0}/tests/__init__.py +0 -0
- {langfuse_haystack-5.2.0 → langfuse_haystack-7.0.0}/tests/conftest.py +0 -0
- {langfuse_haystack-5.2.0 → langfuse_haystack-7.0.0}/tests/test_tracing.py +0 -0
|
@@ -1,5 +1,28 @@
|
|
|
1
1
|
# Changelog
|
|
2
2
|
|
|
3
|
+
## [integrations/langfuse-v6.0.0] - 2026-09-11
|
|
4
|
+
|
|
5
|
+
### 🐛 Bug Fixes
|
|
6
|
+
|
|
7
|
+
- Fix new issues raised by ruff 0.16.0 (#3670)
|
|
8
|
+
- Standardize license files (#3771)
|
|
9
|
+
|
|
10
|
+
### 🚜 Refactor
|
|
11
|
+
|
|
12
|
+
- [**breaking**] Langfuse connector - move tracer creation at warm_up (#3949)
|
|
13
|
+
|
|
14
|
+
### ⚙️ CI
|
|
15
|
+
|
|
16
|
+
- Improve changelog generation; fix existing changelogs (#3883)
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
## [integrations/langfuse-v5.2.0] - 2026-07-15
|
|
20
|
+
|
|
21
|
+
### 🐛 Bug Fixes
|
|
22
|
+
|
|
23
|
+
- *(langfuse)* Replace noisy ToolInvoker input with focused tool call arguments (#3520)
|
|
24
|
+
|
|
25
|
+
|
|
3
26
|
## [integrations/langfuse-v5.1.0] - 2026-07-08
|
|
4
27
|
|
|
5
28
|
### 🐛 Bug Fixes
|
|
@@ -184,14 +207,19 @@
|
|
|
184
207
|
|
|
185
208
|
- Properly cleanup Langfuse tracing context after pipeline run failures (#1999)
|
|
186
209
|
|
|
210
|
+
### 🧹 Chores
|
|
211
|
+
|
|
212
|
+
- Remove black (#1985)
|
|
213
|
+
|
|
214
|
+
|
|
215
|
+
## [integrations/langfuse-v2.1.0] - 2025-06-16
|
|
216
|
+
|
|
187
217
|
|
|
188
218
|
### 🧹 Chores
|
|
189
219
|
|
|
190
220
|
- Pin langfuse<3.0.0 (#1904)
|
|
191
221
|
- Align core-integrations Hatch scripts (#1898)
|
|
192
222
|
- Update md files for new hatch scripts (#1911)
|
|
193
|
-
- Remove black (#1985)
|
|
194
|
-
|
|
195
223
|
|
|
196
224
|
## [integrations/langfuse-v2.0.1] - 2025-06-02
|
|
197
225
|
|
|
@@ -248,10 +276,15 @@
|
|
|
248
276
|
|
|
249
277
|
### 🚀 Features
|
|
250
278
|
|
|
251
|
-
- Adapt Ollama metadata to OpenAI format; support Ollama in Langfuse (#1577)
|
|
252
279
|
- Unify traces of sub-pipelines within pipelines with Langfuse (#1624)
|
|
253
280
|
|
|
254
281
|
|
|
282
|
+
## [integrations/langfuse-v0.10.0] - 2025-04-04
|
|
283
|
+
|
|
284
|
+
### 🚀 Features
|
|
285
|
+
|
|
286
|
+
- Adapt Ollama metadata to OpenAI format; support Ollama in Langfuse (#1577)
|
|
287
|
+
|
|
255
288
|
|
|
256
289
|
## [integrations/langfuse-v0.9.0] - 2025-04-04
|
|
257
290
|
|
|
@@ -311,6 +344,12 @@
|
|
|
311
344
|
|
|
312
345
|
## [integrations/langfuse-v0.6.2] - 2025-01-02
|
|
313
346
|
|
|
347
|
+
### 🌀 Miscellaneous
|
|
348
|
+
|
|
349
|
+
- Fix messages conversion to OpenAI format (#1272)
|
|
350
|
+
|
|
351
|
+
## [integrations/langfuse-v0.6.1] - 2024-12-11
|
|
352
|
+
|
|
314
353
|
### 🚀 Features
|
|
315
354
|
|
|
316
355
|
- Warn if LangfuseTracer initialized without tracing enabled (#1231)
|
|
@@ -322,7 +361,6 @@
|
|
|
322
361
|
### 🌀 Miscellaneous
|
|
323
362
|
|
|
324
363
|
- Chore: Fix tracing_context_var lint errors (#1220)
|
|
325
|
-
- Fix messages conversion to OpenAI format (#1272)
|
|
326
364
|
|
|
327
365
|
## [integrations/langfuse-v0.6.0] - 2024-11-18
|
|
328
366
|
|
|
@@ -355,6 +393,9 @@
|
|
|
355
393
|
|
|
356
394
|
- Langfuse - support generation span for more LLMs (#1087)
|
|
357
395
|
|
|
396
|
+
|
|
397
|
+
## [integrations/langfuse-v0.3.0] - 2024-09-11
|
|
398
|
+
|
|
358
399
|
### 🚜 Refactor
|
|
359
400
|
|
|
360
401
|
- Remove usage of deprecated `ChatMessage.to_openai_format` (#1001)
|
|
@@ -390,10 +431,6 @@
|
|
|
390
431
|
|
|
391
432
|
## [integrations/langfuse-v0.1.0] - 2024-06-13
|
|
392
433
|
|
|
393
|
-
### 🚀 Features
|
|
394
|
-
|
|
395
|
-
- Langfuse integration (#686)
|
|
396
|
-
|
|
397
434
|
### 🐛 Bug Fixes
|
|
398
435
|
|
|
399
436
|
- Performance optimizations and value error when streaming in langfuse (#798)
|
|
@@ -407,4 +444,10 @@
|
|
|
407
444
|
- Chore: change the pydoc renderer class (#718)
|
|
408
445
|
- Docs: add missing api references (#728)
|
|
409
446
|
|
|
447
|
+
## [integrations/langfuse-v0.0.4] - 2024-05-02
|
|
448
|
+
|
|
449
|
+
### 🚀 Features
|
|
450
|
+
|
|
451
|
+
- Langfuse integration (#686)
|
|
452
|
+
|
|
410
453
|
<!-- generated by git-cliff -->
|
|
@@ -58,7 +58,7 @@ APPENDIX: How to apply the Apache License to your work.
|
|
|
58
58
|
|
|
59
59
|
To apply the Apache License to your work, attach the following boilerplate notice, with the fields enclosed by brackets "[]" replaced with your own identifying information. (Don't include the brackets!) The text should be enclosed in the appropriate comment syntax for the file format. We also recommend that a file or class name and description of purpose be included on the same "printed page" as the copyright notice for easier identification within third-party archives.
|
|
60
60
|
|
|
61
|
-
Copyright
|
|
61
|
+
Copyright 2023-present deepset GmbH
|
|
62
62
|
|
|
63
63
|
Licensed under the Apache License, Version 2.0 (the "License");
|
|
64
64
|
you may not use this file except in compliance with the License.
|
|
@@ -1,6 +1,6 @@
|
|
|
1
|
-
Metadata-Version: 2.
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
2
|
Name: langfuse-haystack
|
|
3
|
-
Version:
|
|
3
|
+
Version: 7.0.0
|
|
4
4
|
Summary: Langfuse integration for Haystack
|
|
5
5
|
Project-URL: Documentation, https://github.com/deepset-ai/haystack-core-integrations/tree/main/integrations/langfuse#readme
|
|
6
6
|
Project-URL: Issues, https://github.com/deepset-ai/haystack-core-integrations/issues
|
|
@@ -17,7 +17,7 @@ Classifier: Programming Language :: Python :: 3.13
|
|
|
17
17
|
Classifier: Programming Language :: Python :: Implementation :: CPython
|
|
18
18
|
Classifier: Programming Language :: Python :: Implementation :: PyPy
|
|
19
19
|
Requires-Python: >=3.10
|
|
20
|
-
Requires-Dist: haystack-ai>=
|
|
20
|
+
Requires-Dist: haystack-ai>=3.0.0
|
|
21
21
|
Requires-Dist: langfuse>=4.0.0
|
|
22
22
|
Description-Content-Type: text/markdown
|
|
23
23
|
|
|
@@ -21,7 +21,7 @@ classifiers = [
|
|
|
21
21
|
"Programming Language :: Python :: Implementation :: CPython",
|
|
22
22
|
"Programming Language :: Python :: Implementation :: PyPy",
|
|
23
23
|
]
|
|
24
|
-
dependencies = ["haystack-ai>=
|
|
24
|
+
dependencies = ["haystack-ai>=3.0.0", "langfuse>=4.0.0"]
|
|
25
25
|
|
|
26
26
|
[project.urls]
|
|
27
27
|
Documentation = "https://github.com/deepset-ai/haystack-core-integrations/tree/main/integrations/langfuse#readme"
|
|
@@ -130,6 +130,7 @@ ignore = [
|
|
|
130
130
|
"PLR0912",
|
|
131
131
|
"PLR0913",
|
|
132
132
|
"PLR0915",
|
|
133
|
+
"PLR0917",
|
|
133
134
|
# Asserts
|
|
134
135
|
"S101",
|
|
135
136
|
# Allow `Any` - used legitimately for dynamic types and SDK boundaries
|
|
@@ -157,18 +157,28 @@ class LangfuseConnector:
|
|
|
157
157
|
self.span_handler = span_handler
|
|
158
158
|
self.host = host
|
|
159
159
|
self.langfuse_client_kwargs = langfuse_client_kwargs
|
|
160
|
+
self._httpx_client = httpx_client
|
|
161
|
+
self.tracer: LangfuseTracer | None = None
|
|
162
|
+
|
|
163
|
+
def warm_up(self) -> None:
|
|
164
|
+
"""
|
|
165
|
+
Initialize the Langfuse client and enable tracing once.
|
|
166
|
+
"""
|
|
167
|
+
if self.tracer is not None:
|
|
168
|
+
return
|
|
169
|
+
|
|
160
170
|
resolved_langfuse_client_kwargs = {
|
|
161
|
-
"secret_key": secret_key.resolve_value() if secret_key else None,
|
|
162
|
-
"public_key": public_key.resolve_value() if public_key else None,
|
|
163
|
-
"httpx_client":
|
|
164
|
-
"host": host,
|
|
165
|
-
**(langfuse_client_kwargs or {}),
|
|
171
|
+
"secret_key": self.secret_key.resolve_value() if self.secret_key else None,
|
|
172
|
+
"public_key": self.public_key.resolve_value() if self.public_key else None,
|
|
173
|
+
"httpx_client": self._httpx_client,
|
|
174
|
+
"host": self.host,
|
|
175
|
+
**(self.langfuse_client_kwargs or {}),
|
|
166
176
|
}
|
|
167
177
|
self.tracer = LangfuseTracer(
|
|
168
178
|
tracer=Langfuse(**resolved_langfuse_client_kwargs),
|
|
169
|
-
name=name,
|
|
170
|
-
public=public,
|
|
171
|
-
span_handler=span_handler,
|
|
179
|
+
name=self.name,
|
|
180
|
+
public=self.public,
|
|
181
|
+
span_handler=self.span_handler,
|
|
172
182
|
)
|
|
173
183
|
tracing.enable_tracing(self.tracer)
|
|
174
184
|
|
|
@@ -186,6 +196,8 @@ class LangfuseConnector:
|
|
|
186
196
|
- `trace_url`: The URL to the tracing data.
|
|
187
197
|
- `trace_id`: The ID of the trace.
|
|
188
198
|
"""
|
|
199
|
+
self.warm_up()
|
|
200
|
+
assert self.tracer is not None
|
|
189
201
|
logger.debug(
|
|
190
202
|
"Langfuse tracer invoked with the following context: '{invocation_context}'",
|
|
191
203
|
invocation_context=invocation_context,
|
|
@@ -6,8 +6,7 @@ import contextlib
|
|
|
6
6
|
import os
|
|
7
7
|
import sys
|
|
8
8
|
from abc import ABC, abstractmethod
|
|
9
|
-
from collections import
|
|
10
|
-
from collections.abc import Iterator
|
|
9
|
+
from collections.abc import Iterator, Sequence
|
|
11
10
|
from contextlib import AbstractContextManager
|
|
12
11
|
from contextvars import ContextVar
|
|
13
12
|
from dataclasses import dataclass
|
|
@@ -15,14 +14,15 @@ from datetime import datetime
|
|
|
15
14
|
from typing import Any, Literal, cast
|
|
16
15
|
|
|
17
16
|
from haystack import default_from_dict, default_to_dict, logging
|
|
18
|
-
from haystack.dataclasses import ChatMessage
|
|
17
|
+
from haystack.dataclasses import ChatMessage, FileContent, ImageContent, TextContent
|
|
18
|
+
from haystack.tools import flatten_tools_or_toolsets
|
|
19
19
|
from haystack.tracing import Span, Tracer
|
|
20
20
|
from haystack.tracing import tracer as proxy_tracer
|
|
21
21
|
from haystack.tracing import utils as tracing_utils
|
|
22
22
|
|
|
23
23
|
import langfuse
|
|
24
|
+
from langfuse import LangfuseAgent, LangfuseGeneration, LangfuseTool, propagate_attributes
|
|
24
25
|
from langfuse import LangfuseSpan as LangfuseClientSpan
|
|
25
|
-
from langfuse import propagate_attributes
|
|
26
26
|
from langfuse.types import TraceContext
|
|
27
27
|
|
|
28
28
|
logger = logging.getLogger(__name__)
|
|
@@ -39,9 +39,16 @@ _COMPONENT_NAME_KEY = "haystack.component.name"
|
|
|
39
39
|
_COMPONENT_TYPE_KEY = "haystack.component.type"
|
|
40
40
|
_COMPONENT_OUTPUT_KEY = "haystack.component.output"
|
|
41
41
|
_COMPONENT_INPUT_KEY = "haystack.component.input"
|
|
42
|
+
_AGENT_STEP_OPERATION = "haystack.agent.step"
|
|
43
|
+
_AGENT_STEP_LLM_OPERATION = "haystack.agent.step.llm"
|
|
44
|
+
_AGENT_STEP_TOOL_OPERATION = "haystack.agent.step.tool"
|
|
45
|
+
_AGENT_STEP_KEY = "haystack.agent.step"
|
|
46
|
+
_AGENT_STEP_LLM_INPUT_KEY = "haystack.agent.step.llm.input"
|
|
47
|
+
_AGENT_STEP_LLM_OUTPUT_KEY = "haystack.agent.step.llm.output"
|
|
48
|
+
_TOOL_NAME_KEY = "haystack.tool.name"
|
|
42
49
|
|
|
43
50
|
# Type alias for observation span types
|
|
44
|
-
ObservationSpanType = Literal["tool", "agent", "retriever", "embedding", "generation"]
|
|
51
|
+
ObservationSpanType = Literal["tool", "agent", "chain", "retriever", "embedding", "generation"]
|
|
45
52
|
|
|
46
53
|
# External session metadata for trace correlation (Haystack system)
|
|
47
54
|
# Stores trace_id, user_id, session_id, tags, version for root trace creation
|
|
@@ -88,27 +95,19 @@ class LangfuseSpan(Span):
|
|
|
88
95
|
"""
|
|
89
96
|
if not proxy_tracer.is_content_tracing_enabled:
|
|
90
97
|
return
|
|
98
|
+
# Only generation and agent observations carry chat messages, other spans like tool calls get a coerced value
|
|
99
|
+
is_chat = isinstance(self._span, (LangfuseGeneration, LangfuseAgent))
|
|
91
100
|
if key.endswith(".input"):
|
|
92
|
-
|
|
93
|
-
|
|
94
|
-
|
|
95
|
-
self._span.update(input={"messages": messages, "generation_kwargs": gen_kwargs})
|
|
96
|
-
else:
|
|
97
|
-
self._span.update(input=messages)
|
|
98
|
-
else:
|
|
99
|
-
coerced_value = tracing_utils.coerce_tag_value(value)
|
|
100
|
-
self._span.update(input=coerced_value)
|
|
101
|
+
self._span.update(
|
|
102
|
+
input=_format_chat_input(value=value) if is_chat else tracing_utils.coerce_tag_value(value)
|
|
103
|
+
)
|
|
101
104
|
elif key.endswith(".output"):
|
|
102
|
-
if
|
|
103
|
-
|
|
104
|
-
|
|
105
|
-
|
|
106
|
-
else:
|
|
107
|
-
replies = replies_list
|
|
108
|
-
self._span.update(output=replies)
|
|
105
|
+
if is_chat:
|
|
106
|
+
self._span.update(output=_format_chat_output(value=value))
|
|
107
|
+
elif isinstance(self._span, LangfuseTool):
|
|
108
|
+
self._span.update(output=_format_tool_output(value=value))
|
|
109
109
|
else:
|
|
110
|
-
|
|
111
|
-
self._span.update(output=coerced_value)
|
|
110
|
+
self._span.update(output=tracing_utils.coerce_tag_value(value))
|
|
112
111
|
|
|
113
112
|
self._data[key] = value
|
|
114
113
|
|
|
@@ -276,6 +275,150 @@ def _sanitize_usage_data(usage: dict[str, Any]) -> dict[str, Any]:
|
|
|
276
275
|
return sanitized
|
|
277
276
|
|
|
278
277
|
|
|
278
|
+
def _to_openai_content_parts(parts: Sequence[TextContent | ImageContent | FileContent]) -> list[dict[str, Any]]:
|
|
279
|
+
"""
|
|
280
|
+
Convert content parts to the `text`, `image_url` and `file` parts of OpenAI user messages.
|
|
281
|
+
|
|
282
|
+
:param parts: The content parts, e.g. the result of a tool.
|
|
283
|
+
:returns: The content parts in OpenAI format.
|
|
284
|
+
"""
|
|
285
|
+
content: list[dict[str, Any]] = []
|
|
286
|
+
for part in parts:
|
|
287
|
+
if isinstance(part, TextContent):
|
|
288
|
+
content.append({"type": "text", "text": part.text})
|
|
289
|
+
elif isinstance(part, ImageContent):
|
|
290
|
+
image_url = f"data:{part.mime_type or 'image/jpeg'};base64,{part.base64_image}"
|
|
291
|
+
content.append({"type": "image_url", "image_url": {"url": image_url}})
|
|
292
|
+
elif isinstance(part, FileContent):
|
|
293
|
+
file_data = f"data:{part.mime_type or 'application/pdf'};base64,{part.base64_data}"
|
|
294
|
+
content.append({"type": "file", "file": {"file_data": file_data, "filename": part.filename}})
|
|
295
|
+
return content
|
|
296
|
+
|
|
297
|
+
|
|
298
|
+
def _to_openai_message(message: ChatMessage) -> dict[str, Any]:
|
|
299
|
+
"""
|
|
300
|
+
Convert a ChatMessage to OpenAI Chat Completions format for Langfuse.
|
|
301
|
+
|
|
302
|
+
Tool results made of content parts get `text`, `image_url` and `file` parts, as used in OpenAI user messages.
|
|
303
|
+
|
|
304
|
+
:param message: The ChatMessage to convert.
|
|
305
|
+
:returns: The message in OpenAI format.
|
|
306
|
+
:raises ValueError: If the message has no OpenAI format, e.g. because it has no content.
|
|
307
|
+
"""
|
|
308
|
+
result = message.tool_call_result
|
|
309
|
+
if result is None or isinstance(result.result, str):
|
|
310
|
+
return message.to_openai_dict_format(require_tool_call_ids=False)
|
|
311
|
+
|
|
312
|
+
openai_message: dict[str, Any] = {"role": "tool", "content": _to_openai_content_parts(parts=result.result)}
|
|
313
|
+
if result.origin.id is not None:
|
|
314
|
+
openai_message["tool_call_id"] = result.origin.id
|
|
315
|
+
return openai_message
|
|
316
|
+
|
|
317
|
+
|
|
318
|
+
def _format_tool_output(value: Any) -> Any:
|
|
319
|
+
"""
|
|
320
|
+
Format the result of a tool call for Langfuse.
|
|
321
|
+
|
|
322
|
+
:param value: The tool result.
|
|
323
|
+
:returns: The content parts in OpenAI format if the result is a list of content parts, e.g. text and images.
|
|
324
|
+
Any other result is returned as a coerced tag value.
|
|
325
|
+
"""
|
|
326
|
+
if (
|
|
327
|
+
isinstance(value, list)
|
|
328
|
+
and value
|
|
329
|
+
and all(isinstance(part, (TextContent, ImageContent, FileContent)) for part in value)
|
|
330
|
+
):
|
|
331
|
+
return _to_openai_content_parts(parts=value)
|
|
332
|
+
return tracing_utils.coerce_tag_value(value)
|
|
333
|
+
|
|
334
|
+
|
|
335
|
+
def _format_chat_input(value: Any) -> Any:
|
|
336
|
+
"""
|
|
337
|
+
Format the input of a generation or agent span for Langfuse.
|
|
338
|
+
|
|
339
|
+
:param value: The traced input, e.g. the inputs of a ChatGenerator.
|
|
340
|
+
:returns: The messages in OpenAI format. If `generation_kwargs` or `tools` are set, a dictionary with the
|
|
341
|
+
`messages` and the `generation_kwargs` and OpenAI tool definitions is returned instead.
|
|
342
|
+
Messages that have no OpenAI format are returned as a coerced tag value.
|
|
343
|
+
Inputs without `messages` are returned as a coerced tag value.
|
|
344
|
+
"""
|
|
345
|
+
if "messages" not in value:
|
|
346
|
+
return tracing_utils.coerce_tag_value(value)
|
|
347
|
+
messages: Any
|
|
348
|
+
try:
|
|
349
|
+
messages = [_to_openai_message(message=m) for m in (value.get("messages") or [])]
|
|
350
|
+
except ValueError:
|
|
351
|
+
messages = tracing_utils.coerce_tag_value(value.get("messages"))
|
|
352
|
+
|
|
353
|
+
formatted: dict[str, Any] = {"messages": messages}
|
|
354
|
+
if isinstance(gen_kwargs := value.get("generation_kwargs"), dict):
|
|
355
|
+
formatted["generation_kwargs"] = gen_kwargs
|
|
356
|
+
# Langfuse shows the tools of `{"messages": ..., "tools": ...}` inputs next to the messages
|
|
357
|
+
if tools := value.get("tools"):
|
|
358
|
+
formatted["tools"] = [{"type": "function", "function": t.tool_spec} for t in flatten_tools_or_toolsets(tools)]
|
|
359
|
+
return formatted if len(formatted) > 1 else messages
|
|
360
|
+
|
|
361
|
+
|
|
362
|
+
def _format_chat_output(value: Any) -> Any:
|
|
363
|
+
"""
|
|
364
|
+
Format the output of a generation or agent span for Langfuse.
|
|
365
|
+
|
|
366
|
+
:param value: The traced output, e.g. the outputs of a ChatGenerator.
|
|
367
|
+
:returns: The replies in OpenAI format. String replies are returned as they are.
|
|
368
|
+
Replies that have no OpenAI format are returned as a coerced tag value.
|
|
369
|
+
Outputs without `replies` are returned as a coerced tag value.
|
|
370
|
+
"""
|
|
371
|
+
if "replies" not in value:
|
|
372
|
+
return tracing_utils.coerce_tag_value(value)
|
|
373
|
+
replies = value.get("replies") or []
|
|
374
|
+
# Generators that aren't ChatGenerators return string replies
|
|
375
|
+
if not all(isinstance(r, ChatMessage) for r in replies):
|
|
376
|
+
return replies
|
|
377
|
+
try:
|
|
378
|
+
return [_to_openai_message(message=m) for m in replies]
|
|
379
|
+
except ValueError:
|
|
380
|
+
return tracing_utils.coerce_tag_value(replies)
|
|
381
|
+
|
|
382
|
+
|
|
383
|
+
def _update_generation_details(
|
|
384
|
+
span: LangfuseSpan, chat_generator_inputs: dict[str, Any], chat_generator_output: dict[str, Any]
|
|
385
|
+
) -> None:
|
|
386
|
+
"""
|
|
387
|
+
Add the model details of a ChatGenerator call to the span.
|
|
388
|
+
|
|
389
|
+
The model, token usage and completion start time come from the first reply, the model parameters from the
|
|
390
|
+
`generation_kwargs` passed to the ChatGenerator.
|
|
391
|
+
|
|
392
|
+
:param span: The generation span.
|
|
393
|
+
:param chat_generator_inputs: The inputs of the ChatGenerator.
|
|
394
|
+
:param chat_generator_output: The outputs of the ChatGenerator.
|
|
395
|
+
"""
|
|
396
|
+
update_kwargs: dict[str, Any] = {}
|
|
397
|
+
if replies := chat_generator_output.get("replies"):
|
|
398
|
+
meta = replies[0].meta
|
|
399
|
+
completion_start_time = meta.get("completion_start_time")
|
|
400
|
+
if completion_start_time:
|
|
401
|
+
try:
|
|
402
|
+
completion_start_time = datetime.fromisoformat(completion_start_time)
|
|
403
|
+
except ValueError:
|
|
404
|
+
logger.error(f"Failed to parse completion_start_time: {completion_start_time}")
|
|
405
|
+
completion_start_time = None
|
|
406
|
+
usage = meta.get("usage")
|
|
407
|
+
update_kwargs["usage_details"] = _sanitize_usage_data(usage=usage) if usage else None
|
|
408
|
+
update_kwargs["model"] = meta.get("model")
|
|
409
|
+
update_kwargs["completion_start_time"] = completion_start_time
|
|
410
|
+
if generation_kwargs := chat_generator_inputs.get("generation_kwargs"):
|
|
411
|
+
# Langfuse model parameters only take primitive values, so nested values like `response_format` are coerced
|
|
412
|
+
update_kwargs["model_parameters"] = {
|
|
413
|
+
key: value
|
|
414
|
+
if value is None or isinstance(value, tracing_utils.PRIMITIVE_TYPES)
|
|
415
|
+
else tracing_utils.coerce_tag_value(value)
|
|
416
|
+
for key, value in generation_kwargs.items()
|
|
417
|
+
}
|
|
418
|
+
if update_kwargs:
|
|
419
|
+
span.raw_span().update(**update_kwargs)
|
|
420
|
+
|
|
421
|
+
|
|
279
422
|
class DefaultSpanHandler(SpanHandler):
|
|
280
423
|
"""DefaultSpanHandler provides the default Langfuse tracing behavior for Haystack."""
|
|
281
424
|
|
|
@@ -313,9 +456,19 @@ class DefaultSpanHandler(SpanHandler):
|
|
|
313
456
|
return span
|
|
314
457
|
|
|
315
458
|
span_type = None
|
|
459
|
+
name = context.name
|
|
316
460
|
|
|
317
|
-
|
|
461
|
+
# Agent spans carry no component tags, so they're matched by operation name
|
|
462
|
+
if context.operation_name == _AGENT_STEP_OPERATION:
|
|
463
|
+
span_type = "chain"
|
|
464
|
+
name = f"agent step {context.tags.get(_AGENT_STEP_KEY)}"
|
|
465
|
+
elif context.operation_name == _AGENT_STEP_LLM_OPERATION:
|
|
466
|
+
span_type = "generation"
|
|
467
|
+
name = "llm"
|
|
468
|
+
elif context.operation_name == _AGENT_STEP_TOOL_OPERATION:
|
|
318
469
|
span_type = "tool"
|
|
470
|
+
tool_name = context.tags.get(_TOOL_NAME_KEY)
|
|
471
|
+
name = f"tool - {tool_name}" if tool_name else "tool"
|
|
319
472
|
elif context.operation_name == "haystack.agent.run":
|
|
320
473
|
span_type = "agent"
|
|
321
474
|
elif context.component_type and context.component_type.endswith("Retriever"):
|
|
@@ -327,12 +480,10 @@ class DefaultSpanHandler(SpanHandler):
|
|
|
327
480
|
|
|
328
481
|
if span_type:
|
|
329
482
|
return LangfuseSpan(
|
|
330
|
-
self.tracer.start_as_current_observation(
|
|
331
|
-
name=context.name, as_type=cast(ObservationSpanType, span_type)
|
|
332
|
-
)
|
|
483
|
+
self.tracer.start_as_current_observation(name=name, as_type=cast(ObservationSpanType, span_type))
|
|
333
484
|
)
|
|
334
485
|
else:
|
|
335
|
-
return LangfuseSpan(self.tracer.start_as_current_observation(name=
|
|
486
|
+
return LangfuseSpan(self.tracer.start_as_current_observation(name=name))
|
|
336
487
|
|
|
337
488
|
def handle(self, span: LangfuseSpan, component_type: str | None) -> None:
|
|
338
489
|
"""Process and enrich a span after component execution."""
|
|
@@ -342,66 +493,18 @@ class DefaultSpanHandler(SpanHandler):
|
|
|
342
493
|
coerced_input = tracing_utils.coerce_tag_value(span.get_data().get(_PIPELINE_INPUT_KEY))
|
|
343
494
|
coerced_output = tracing_utils.coerce_tag_value(span.get_data().get(_PIPELINE_OUTPUT_KEY))
|
|
344
495
|
span.raw_span().update(input=coerced_input, output=coerced_output)
|
|
345
|
-
|
|
346
|
-
|
|
347
|
-
|
|
348
|
-
|
|
349
|
-
|
|
350
|
-
|
|
351
|
-
|
|
352
|
-
|
|
353
|
-
|
|
354
|
-
|
|
355
|
-
|
|
356
|
-
|
|
357
|
-
# Fallback to "ToolInvoker" if we can't retrieve component name
|
|
358
|
-
tool_invoker_name = span.get_data().get(_COMPONENT_NAME_KEY, "ToolInvoker")
|
|
359
|
-
tool_counts = Counter(tool_names) # how many times each tool was called
|
|
360
|
-
formatted_names = [f"{name} (x{count})" if count > 1 else name for name, count in tool_counts.items()]
|
|
361
|
-
span.raw_span().update(name=f"{tool_invoker_name} - {sorted(formatted_names)}")
|
|
362
|
-
|
|
363
|
-
if tool_calls_input and proxy_tracer.is_content_tracing_enabled:
|
|
364
|
-
# Replace the noisy full message history with just the tool call arguments
|
|
365
|
-
span.raw_span().update(input=tool_calls_input)
|
|
366
|
-
|
|
367
|
-
output_messages = span.get_data().get(_COMPONENT_OUTPUT_KEY, {}).get("tool_messages", [])
|
|
368
|
-
tool_results: list[dict[str, Any]] = []
|
|
369
|
-
for message in output_messages:
|
|
370
|
-
if isinstance(message, ChatMessage) and message.tool_call_results:
|
|
371
|
-
for tcr in message.tool_call_results:
|
|
372
|
-
origin = tcr.origin
|
|
373
|
-
# Keys `name`, `arguments` and `id` let Langfuse detect these as tool
|
|
374
|
-
# calls at ingestion and populate the Tool Call Name filter in the UI.
|
|
375
|
-
tool_results.append(
|
|
376
|
-
{
|
|
377
|
-
"id": origin.id if origin else None,
|
|
378
|
-
"name": origin.tool_name if origin else None,
|
|
379
|
-
"arguments": origin.arguments if origin else None,
|
|
380
|
-
"result": tcr.result,
|
|
381
|
-
"error": tcr.error,
|
|
382
|
-
}
|
|
383
|
-
)
|
|
384
|
-
if tool_results:
|
|
385
|
-
span.raw_span().update(output=tool_results)
|
|
386
|
-
|
|
387
|
-
if component_type and component_type.endswith("ChatGenerator"):
|
|
388
|
-
replies = span.get_data().get(_COMPONENT_OUTPUT_KEY, {}).get("replies")
|
|
389
|
-
if replies:
|
|
390
|
-
meta = replies[0].meta
|
|
391
|
-
completion_start_time = meta.get("completion_start_time")
|
|
392
|
-
if completion_start_time:
|
|
393
|
-
try:
|
|
394
|
-
completion_start_time = datetime.fromisoformat(completion_start_time)
|
|
395
|
-
except ValueError:
|
|
396
|
-
logger.error(f"Failed to parse completion_start_time: {completion_start_time}")
|
|
397
|
-
completion_start_time = None
|
|
398
|
-
usage = meta.get("usage")
|
|
399
|
-
sanitized_usage = _sanitize_usage_data(usage) if usage else None
|
|
400
|
-
span.raw_span().update(
|
|
401
|
-
usage_details=sanitized_usage,
|
|
402
|
-
model=meta.get("model"),
|
|
403
|
-
completion_start_time=completion_start_time,
|
|
404
|
-
)
|
|
496
|
+
if _AGENT_STEP_LLM_OUTPUT_KEY in span.get_data():
|
|
497
|
+
_update_generation_details(
|
|
498
|
+
span=span,
|
|
499
|
+
chat_generator_inputs=span.get_data().get(_AGENT_STEP_LLM_INPUT_KEY, {}),
|
|
500
|
+
chat_generator_output=span.get_data()[_AGENT_STEP_LLM_OUTPUT_KEY],
|
|
501
|
+
)
|
|
502
|
+
elif component_type and component_type.endswith("ChatGenerator"):
|
|
503
|
+
_update_generation_details(
|
|
504
|
+
span=span,
|
|
505
|
+
chat_generator_inputs=span.get_data().get(_COMPONENT_INPUT_KEY, {}),
|
|
506
|
+
chat_generator_output=span.get_data().get(_COMPONENT_OUTPUT_KEY, {}),
|
|
507
|
+
)
|
|
405
508
|
elif component_type and component_type.endswith("Generator"):
|
|
406
509
|
meta = span.get_data().get(_COMPONENT_OUTPUT_KEY, {}).get("meta")
|
|
407
510
|
if meta:
|
|
@@ -6,48 +6,53 @@ import os
|
|
|
6
6
|
|
|
7
7
|
os.environ["HAYSTACK_CONTENT_TRACING_ENABLED"] = "true"
|
|
8
8
|
|
|
9
|
-
from unittest.mock import Mock
|
|
9
|
+
from unittest.mock import Mock, patch
|
|
10
10
|
|
|
11
|
-
|
|
11
|
+
import httpx
|
|
12
|
+
import pytest
|
|
13
|
+
from haystack import Pipeline, tracing
|
|
12
14
|
from haystack.components.builders import ChatPromptBuilder
|
|
13
15
|
from haystack.components.generators.chat import OpenAIChatGenerator
|
|
16
|
+
from haystack.tracing.tracer import NullTracer
|
|
14
17
|
from haystack.utils import Secret
|
|
15
18
|
|
|
16
19
|
from haystack_integrations.components.connectors.langfuse import LangfuseConnector
|
|
17
20
|
from haystack_integrations.tracing.langfuse import DefaultSpanHandler
|
|
18
21
|
|
|
19
22
|
|
|
23
|
+
@pytest.fixture
|
|
24
|
+
def mock_langfuse(monkeypatch):
|
|
25
|
+
monkeypatch.setattr(tracing.tracer, "actual_tracer", NullTracer())
|
|
26
|
+
with patch("haystack_integrations.components.connectors.langfuse.langfuse_connector.Langfuse") as constructor:
|
|
27
|
+
constructor.return_value.get_trace_url.return_value = "https://example.com/trace"
|
|
28
|
+
constructor.return_value.get_current_trace_id.return_value = "12345"
|
|
29
|
+
yield constructor
|
|
30
|
+
|
|
31
|
+
|
|
20
32
|
class CustomSpanHandler(DefaultSpanHandler):
|
|
21
33
|
def handle(self, span, component_type=None):
|
|
22
34
|
pass
|
|
23
35
|
|
|
24
36
|
|
|
25
|
-
class
|
|
26
|
-
def test_run(self,
|
|
27
|
-
monkeypatch.setenv("LANGFUSE_SECRET_KEY", "secret")
|
|
28
|
-
monkeypatch.setenv("LANGFUSE_PUBLIC_KEY", "public")
|
|
29
|
-
|
|
37
|
+
class TestRun:
|
|
38
|
+
def test_run(self, mock_langfuse):
|
|
30
39
|
langfuse_connector = LangfuseConnector(
|
|
31
40
|
name="Chat example - OpenAI",
|
|
32
41
|
public=True,
|
|
33
|
-
secret_key=Secret.
|
|
34
|
-
public_key=Secret.
|
|
42
|
+
secret_key=Secret.from_token("secret"),
|
|
43
|
+
public_key=Secret.from_token("public"),
|
|
35
44
|
)
|
|
36
45
|
|
|
37
|
-
mock_tracer = Mock()
|
|
38
|
-
mock_tracer.get_trace_url.return_value = "https://example.com/trace"
|
|
39
|
-
mock_tracer.get_trace_id.return_value = "12345"
|
|
40
|
-
langfuse_connector.tracer = mock_tracer
|
|
41
|
-
|
|
42
46
|
response = langfuse_connector.run(invocation_context={"some_key": "some_value"})
|
|
43
47
|
assert response["name"] == "Chat example - OpenAI"
|
|
44
48
|
assert response["trace_url"] == "https://example.com/trace"
|
|
45
49
|
assert response["trace_id"] == "12345"
|
|
50
|
+
mock_langfuse.assert_called_once()
|
|
51
|
+
assert tracing.tracer.actual_tracer is langfuse_connector.tracer
|
|
46
52
|
|
|
47
|
-
def test_to_dict(self, monkeypatch):
|
|
48
|
-
monkeypatch.setenv("LANGFUSE_SECRET_KEY", "secret")
|
|
49
|
-
monkeypatch.setenv("LANGFUSE_PUBLIC_KEY", "public")
|
|
50
53
|
|
|
54
|
+
class TestSerialization:
|
|
55
|
+
def test_to_dict(self):
|
|
51
56
|
langfuse_connector = LangfuseConnector(name="Chat example - OpenAI")
|
|
52
57
|
serialized = langfuse_connector.to_dict()
|
|
53
58
|
|
|
@@ -72,10 +77,7 @@ class TestLangfuseConnector:
|
|
|
72
77
|
},
|
|
73
78
|
}
|
|
74
79
|
|
|
75
|
-
def test_to_dict_with_params(self
|
|
76
|
-
monkeypatch.setenv("LANGFUSE_SECRET_KEY", "secret")
|
|
77
|
-
monkeypatch.setenv("LANGFUSE_PUBLIC_KEY", "public")
|
|
78
|
-
|
|
80
|
+
def test_to_dict_with_params(self):
|
|
79
81
|
langfuse_connector = LangfuseConnector(
|
|
80
82
|
name="Chat example - OpenAI",
|
|
81
83
|
public=True,
|
|
@@ -111,10 +113,7 @@ class TestLangfuseConnector:
|
|
|
111
113
|
},
|
|
112
114
|
}
|
|
113
115
|
|
|
114
|
-
def test_from_dict(self
|
|
115
|
-
monkeypatch.setenv("LANGFUSE_SECRET_KEY", "secret")
|
|
116
|
-
monkeypatch.setenv("LANGFUSE_PUBLIC_KEY", "public")
|
|
117
|
-
|
|
116
|
+
def test_from_dict(self):
|
|
118
117
|
data = {
|
|
119
118
|
"type": "haystack_integrations.components.connectors.langfuse.langfuse_connector.LangfuseConnector",
|
|
120
119
|
"init_parameters": {
|
|
@@ -144,10 +143,7 @@ class TestLangfuseConnector:
|
|
|
144
143
|
assert langfuse_connector.host is None
|
|
145
144
|
assert langfuse_connector.langfuse_client_kwargs is None
|
|
146
145
|
|
|
147
|
-
def test_from_dict_without_span_handler(self
|
|
148
|
-
monkeypatch.setenv("LANGFUSE_SECRET_KEY", "secret")
|
|
149
|
-
monkeypatch.setenv("LANGFUSE_PUBLIC_KEY", "public")
|
|
150
|
-
|
|
146
|
+
def test_from_dict_without_span_handler(self):
|
|
151
147
|
# All keys that would point to None (span_handler, host, langfuse_client_kwargs) are intentionally absent
|
|
152
148
|
data = {
|
|
153
149
|
"type": "haystack_integrations.components.connectors.langfuse.langfuse_connector.LangfuseConnector",
|
|
@@ -175,10 +171,7 @@ class TestLangfuseConnector:
|
|
|
175
171
|
assert langfuse_connector.host is None
|
|
176
172
|
assert langfuse_connector.langfuse_client_kwargs is None
|
|
177
173
|
|
|
178
|
-
def test_from_dict_with_params(self
|
|
179
|
-
monkeypatch.setenv("LANGFUSE_SECRET_KEY", "secret")
|
|
180
|
-
monkeypatch.setenv("LANGFUSE_PUBLIC_KEY", "public")
|
|
181
|
-
|
|
174
|
+
def test_from_dict_with_params(self):
|
|
182
175
|
data = {
|
|
183
176
|
"type": "haystack_integrations.components.connectors.langfuse.langfuse_connector.LangfuseConnector",
|
|
184
177
|
"init_parameters": {
|
|
@@ -212,10 +205,8 @@ class TestLangfuseConnector:
|
|
|
212
205
|
assert langfuse_connector.host == "https://example.com"
|
|
213
206
|
assert langfuse_connector.langfuse_client_kwargs == {"timeout": 30.0}
|
|
214
207
|
|
|
215
|
-
def test_from_dict_with_legacy_span_handler_format(self
|
|
208
|
+
def test_from_dict_with_legacy_span_handler_format(self):
|
|
216
209
|
# Pipelines serialized before this fix wrap span_handler as {"type": ..., "data": {...}}
|
|
217
|
-
monkeypatch.setenv("LANGFUSE_SECRET_KEY", "secret")
|
|
218
|
-
monkeypatch.setenv("LANGFUSE_PUBLIC_KEY", "public")
|
|
219
210
|
|
|
220
211
|
data = {
|
|
221
212
|
"type": "haystack_integrations.components.connectors.langfuse.langfuse_connector.LangfuseConnector",
|
|
@@ -249,8 +240,6 @@ class TestLangfuseConnector:
|
|
|
249
240
|
|
|
250
241
|
def test_pipeline_serialization(self, monkeypatch):
|
|
251
242
|
# Set test env vars
|
|
252
|
-
monkeypatch.setenv("LANGFUSE_SECRET_KEY", "secret")
|
|
253
|
-
monkeypatch.setenv("LANGFUSE_PUBLIC_KEY", "public")
|
|
254
243
|
monkeypatch.setenv("OPENAI_API_KEY", "openai_api_key")
|
|
255
244
|
|
|
256
245
|
# Create pipeline with OpenAI LLM
|
|
@@ -285,3 +274,54 @@ class TestLangfuseConnector:
|
|
|
285
274
|
|
|
286
275
|
# Verify pipeline is the same
|
|
287
276
|
assert new_pipe == pipe
|
|
277
|
+
|
|
278
|
+
|
|
279
|
+
class TestComponentLifecycle:
|
|
280
|
+
@pytest.mark.parametrize("key", ["LANGFUSE_SECRET_KEY", "LANGFUSE_PUBLIC_KEY"])
|
|
281
|
+
def test_key_resolved_at_warm_up_not_init(self, key, monkeypatch, mock_langfuse):
|
|
282
|
+
monkeypatch.setenv("LANGFUSE_SECRET_KEY", "secret")
|
|
283
|
+
monkeypatch.setenv("LANGFUSE_PUBLIC_KEY", "public")
|
|
284
|
+
monkeypatch.delenv(key)
|
|
285
|
+
connector = LangfuseConnector("test")
|
|
286
|
+
assert connector.tracer is None
|
|
287
|
+
mock_langfuse.assert_not_called()
|
|
288
|
+
assert not tracing.is_tracing_enabled()
|
|
289
|
+
|
|
290
|
+
with pytest.raises(ValueError, match=key):
|
|
291
|
+
connector.warm_up()
|
|
292
|
+
assert connector.tracer is None
|
|
293
|
+
mock_langfuse.assert_not_called()
|
|
294
|
+
|
|
295
|
+
monkeypatch.setenv(key, "available-now")
|
|
296
|
+
connector.warm_up()
|
|
297
|
+
assert connector.tracer is not None
|
|
298
|
+
mock_langfuse.assert_called_once()
|
|
299
|
+
|
|
300
|
+
def test_warm_up_passes_configuration_and_is_idempotent(self, mock_langfuse):
|
|
301
|
+
client = Mock(spec=httpx.Client)
|
|
302
|
+
handler = CustomSpanHandler()
|
|
303
|
+
connector = LangfuseConnector(
|
|
304
|
+
"configured",
|
|
305
|
+
public=True,
|
|
306
|
+
public_key=Secret.from_token("public"),
|
|
307
|
+
secret_key=Secret.from_token("secret"),
|
|
308
|
+
httpx_client=client,
|
|
309
|
+
span_handler=handler,
|
|
310
|
+
host="https://example.com",
|
|
311
|
+
langfuse_client_kwargs={"timeout": 30, "host": "https://override.example.com"},
|
|
312
|
+
)
|
|
313
|
+
connector.warm_up()
|
|
314
|
+
first_tracer = connector.tracer
|
|
315
|
+
connector.warm_up()
|
|
316
|
+
assert connector.tracer is first_tracer
|
|
317
|
+
mock_langfuse.assert_called_once_with(
|
|
318
|
+
public_key="public",
|
|
319
|
+
secret_key="secret",
|
|
320
|
+
httpx_client=client,
|
|
321
|
+
host="https://override.example.com",
|
|
322
|
+
timeout=30,
|
|
323
|
+
)
|
|
324
|
+
assert handler.tracer is mock_langfuse.return_value
|
|
325
|
+
assert connector.tracer is not None
|
|
326
|
+
assert connector.tracer._name == "configured"
|
|
327
|
+
assert connector.tracer._public is True
|
|
@@ -3,13 +3,18 @@
|
|
|
3
3
|
# SPDX-License-Identifier: Apache-2.0
|
|
4
4
|
|
|
5
5
|
import asyncio
|
|
6
|
+
import base64
|
|
6
7
|
import datetime
|
|
7
8
|
import logging
|
|
8
9
|
import sys
|
|
9
10
|
from unittest.mock import MagicMock, Mock, patch
|
|
10
11
|
|
|
11
12
|
import pytest
|
|
12
|
-
from haystack.dataclasses import ChatMessage, ToolCall
|
|
13
|
+
from haystack.dataclasses import ChatMessage, ChatRole, FileContent, ImageContent, TextContent, ToolCall
|
|
14
|
+
from haystack.tools import Tool, Toolset
|
|
15
|
+
from haystack.tracing import utils as tracing_utils
|
|
16
|
+
from langfuse import LangfuseAgent, LangfuseGeneration, LangfuseTool
|
|
17
|
+
from langfuse import LangfuseSpan as LangfuseClientSpan
|
|
13
18
|
|
|
14
19
|
from haystack_integrations.tracing.langfuse.tracer import (
|
|
15
20
|
_COMPONENT_OUTPUT_KEY,
|
|
@@ -32,8 +37,8 @@ def mock_get_client():
|
|
|
32
37
|
class MockContextManager:
|
|
33
38
|
"""Mock context manager that simulates Langfuse v4 context managers"""
|
|
34
39
|
|
|
35
|
-
def __init__(self, name="mock_span"):
|
|
36
|
-
self._span = MockSpan(name)
|
|
40
|
+
def __init__(self, name="mock_span", span=None):
|
|
41
|
+
self._span = span or MockSpan(name)
|
|
37
42
|
|
|
38
43
|
def __enter__(self):
|
|
39
44
|
return self._span
|
|
@@ -139,8 +144,9 @@ class TestLangfuseSpan:
|
|
|
139
144
|
mock_context_manager._span.update.assert_called_with(output="output_value")
|
|
140
145
|
|
|
141
146
|
# set_content_tag method can update input and output of the span object with messages/replies
|
|
142
|
-
|
|
143
|
-
|
|
147
|
+
@pytest.mark.parametrize("observation_class", [LangfuseGeneration, LangfuseAgent])
|
|
148
|
+
def test_set_content_tag_updates_input_and_output_with_messages(self, observation_class):
|
|
149
|
+
mock_context_manager = MockContextManager(span=Mock(spec=observation_class))
|
|
144
150
|
span = LangfuseSpan(mock_context_manager)
|
|
145
151
|
|
|
146
152
|
with patch("haystack_integrations.tracing.langfuse.tracer.proxy_tracer.is_content_tracing_enabled", True):
|
|
@@ -180,8 +186,40 @@ class TestLangfuseSpan:
|
|
|
180
186
|
# check we handle properly string list replies
|
|
181
187
|
assert mock_context_manager._span.update.call_args_list[0][1] == {"output": ["reply1", "reply2"]}
|
|
182
188
|
|
|
189
|
+
def test_set_content_tag_input_with_tools(self):
|
|
190
|
+
mock_context_manager = MockContextManager(span=Mock(spec=LangfuseGeneration))
|
|
191
|
+
span = LangfuseSpan(mock_context_manager)
|
|
192
|
+
weather = Tool(
|
|
193
|
+
name="weather",
|
|
194
|
+
description="Get the weather",
|
|
195
|
+
parameters={"type": "object", "properties": {"city": {"type": "string"}}},
|
|
196
|
+
function=lambda city: city,
|
|
197
|
+
)
|
|
198
|
+
|
|
199
|
+
with patch("haystack_integrations.tracing.langfuse.tracer.proxy_tracer.is_content_tracing_enabled", True):
|
|
200
|
+
span.set_content_tag(
|
|
201
|
+
"haystack.agent.step.llm.input",
|
|
202
|
+
{"messages": [ChatMessage.from_user("message")], "tools": Toolset([weather])},
|
|
203
|
+
)
|
|
204
|
+
|
|
205
|
+
mock_context_manager._span.update.assert_called_once_with(
|
|
206
|
+
input={
|
|
207
|
+
"messages": [{"role": "user", "content": "message"}],
|
|
208
|
+
"tools": [
|
|
209
|
+
{
|
|
210
|
+
"type": "function",
|
|
211
|
+
"function": {
|
|
212
|
+
"name": "weather",
|
|
213
|
+
"description": "Get the weather",
|
|
214
|
+
"parameters": {"type": "object", "properties": {"city": {"type": "string"}}},
|
|
215
|
+
},
|
|
216
|
+
}
|
|
217
|
+
],
|
|
218
|
+
}
|
|
219
|
+
)
|
|
220
|
+
|
|
183
221
|
def test_set_content_tag_messages_none_does_not_raise(self):
|
|
184
|
-
mock_context_manager = MockContextManager()
|
|
222
|
+
mock_context_manager = MockContextManager(span=Mock(spec=LangfuseGeneration))
|
|
185
223
|
span = LangfuseSpan(mock_context_manager)
|
|
186
224
|
|
|
187
225
|
with patch("haystack_integrations.tracing.langfuse.tracer.proxy_tracer.is_content_tracing_enabled", True):
|
|
@@ -190,7 +228,7 @@ class TestLangfuseSpan:
|
|
|
190
228
|
assert mock_context_manager._span.update.call_args_list[0][1] == {"input": []}
|
|
191
229
|
|
|
192
230
|
def test_set_content_tag_replies_none_does_not_raise(self):
|
|
193
|
-
mock_context_manager = MockContextManager()
|
|
231
|
+
mock_context_manager = MockContextManager(span=Mock(spec=LangfuseGeneration))
|
|
194
232
|
span = LangfuseSpan(mock_context_manager)
|
|
195
233
|
|
|
196
234
|
with patch("haystack_integrations.tracing.langfuse.tracer.proxy_tracer.is_content_tracing_enabled", True):
|
|
@@ -198,6 +236,91 @@ class TestLangfuseSpan:
|
|
|
198
236
|
assert mock_context_manager._span.update.call_count == 1
|
|
199
237
|
assert mock_context_manager._span.update.call_args_list[0][1] == {"output": []}
|
|
200
238
|
|
|
239
|
+
@pytest.mark.parametrize(
|
|
240
|
+
"key,value,expected",
|
|
241
|
+
[
|
|
242
|
+
("haystack.agent.step.tool.input", {"messages": ["hi", "there"]}, '{"messages": ["hi", "there"]}'),
|
|
243
|
+
("haystack.agent.step.tool.output", None, ""),
|
|
244
|
+
("haystack.agent.step.tool.output", 42, 42),
|
|
245
|
+
("haystack.agent.step.tool.output", "No replies found", "No replies found"),
|
|
246
|
+
("haystack.agent.step.tool.output", {"replies": 5}, '{"replies": 5}'),
|
|
247
|
+
("haystack.component.input", {"messages": ["hi", "there"]}, '{"messages": ["hi", "there"]}'),
|
|
248
|
+
],
|
|
249
|
+
)
|
|
250
|
+
def test_set_content_tag_non_chat_span_coerces_value(self, key, value, expected):
|
|
251
|
+
span_spec = LangfuseTool if key.startswith("haystack.agent.step.tool") else LangfuseClientSpan
|
|
252
|
+
mock_context_manager = MockContextManager(span=Mock(spec=span_spec))
|
|
253
|
+
span = LangfuseSpan(mock_context_manager)
|
|
254
|
+
|
|
255
|
+
with patch("haystack_integrations.tracing.langfuse.tracer.proxy_tracer.is_content_tracing_enabled", True):
|
|
256
|
+
span.set_content_tag(key, value)
|
|
257
|
+
|
|
258
|
+
field = "input" if key.endswith(".input") else "output"
|
|
259
|
+
mock_context_manager._span.update.assert_called_once_with(**{field: expected})
|
|
260
|
+
|
|
261
|
+
def test_set_content_tag_tool_result_with_image_and_file(self):
|
|
262
|
+
mock_context_manager = MockContextManager(span=Mock(spec=LangfuseGeneration))
|
|
263
|
+
span = LangfuseSpan(mock_context_manager)
|
|
264
|
+
png = base64.b64encode(b"\x89PNG\r\n\x1a\n").decode()
|
|
265
|
+
pdf = base64.b64encode(b"%PDF-1.4").decode()
|
|
266
|
+
tool_message = ChatMessage.from_tool(
|
|
267
|
+
tool_result=[
|
|
268
|
+
TextContent("chart"),
|
|
269
|
+
ImageContent(base64_image=png, mime_type="image/png"),
|
|
270
|
+
FileContent(base64_data=pdf, mime_type="application/pdf", filename="report.pdf"),
|
|
271
|
+
],
|
|
272
|
+
origin=ToolCall(tool_name="plot", arguments={}, id="call_1"),
|
|
273
|
+
)
|
|
274
|
+
|
|
275
|
+
with patch("haystack_integrations.tracing.langfuse.tracer.proxy_tracer.is_content_tracing_enabled", True):
|
|
276
|
+
span.set_content_tag("haystack.agent.step.llm.input", {"messages": [tool_message]})
|
|
277
|
+
|
|
278
|
+
mock_context_manager._span.update.assert_called_once_with(
|
|
279
|
+
input=[
|
|
280
|
+
{
|
|
281
|
+
"role": "tool",
|
|
282
|
+
"content": [
|
|
283
|
+
{"type": "text", "text": "chart"},
|
|
284
|
+
{"type": "image_url", "image_url": {"url": f"data:image/png;base64,{png}"}},
|
|
285
|
+
{
|
|
286
|
+
"type": "file",
|
|
287
|
+
"file": {"file_data": f"data:application/pdf;base64,{pdf}", "filename": "report.pdf"},
|
|
288
|
+
},
|
|
289
|
+
],
|
|
290
|
+
"tool_call_id": "call_1",
|
|
291
|
+
}
|
|
292
|
+
]
|
|
293
|
+
)
|
|
294
|
+
|
|
295
|
+
def test_set_content_tag_tool_output_with_image(self):
|
|
296
|
+
mock_context_manager = MockContextManager(span=Mock(spec=LangfuseTool))
|
|
297
|
+
span = LangfuseSpan(mock_context_manager)
|
|
298
|
+
png = base64.b64encode(b"\x89PNG\r\n\x1a\n").decode()
|
|
299
|
+
|
|
300
|
+
with patch("haystack_integrations.tracing.langfuse.tracer.proxy_tracer.is_content_tracing_enabled", True):
|
|
301
|
+
span.set_content_tag(
|
|
302
|
+
"haystack.agent.step.tool.output",
|
|
303
|
+
[TextContent("chart"), ImageContent(base64_image=png, mime_type="image/png")],
|
|
304
|
+
)
|
|
305
|
+
|
|
306
|
+
mock_context_manager._span.update.assert_called_once_with(
|
|
307
|
+
output=[
|
|
308
|
+
{"type": "text", "text": "chart"},
|
|
309
|
+
{"type": "image_url", "image_url": {"url": f"data:image/png;base64,{png}"}},
|
|
310
|
+
]
|
|
311
|
+
)
|
|
312
|
+
|
|
313
|
+
def test_set_content_tag_messages_without_openai_format_are_coerced(self):
|
|
314
|
+
mock_context_manager = MockContextManager(span=Mock(spec=LangfuseGeneration))
|
|
315
|
+
span = LangfuseSpan(mock_context_manager)
|
|
316
|
+
# A user message without content has no OpenAI format
|
|
317
|
+
messages = [ChatMessage.from_user("hi"), ChatMessage(_role=ChatRole.USER, _content=[])]
|
|
318
|
+
|
|
319
|
+
with patch("haystack_integrations.tracing.langfuse.tracer.proxy_tracer.is_content_tracing_enabled", True):
|
|
320
|
+
span.set_content_tag("haystack.agent.step.llm.input", {"messages": messages})
|
|
321
|
+
|
|
322
|
+
mock_context_manager._span.update.assert_called_once_with(input=tracing_utils.coerce_tag_value(messages))
|
|
323
|
+
|
|
201
324
|
|
|
202
325
|
class TestSpanContext:
|
|
203
326
|
def test_post_init(self):
|
|
@@ -344,6 +467,48 @@ class TestDefaultSpanHandler:
|
|
|
344
467
|
),
|
|
345
468
|
}
|
|
346
469
|
|
|
470
|
+
def test_handle_agent_step_llm(self):
|
|
471
|
+
mock_span = Mock()
|
|
472
|
+
mock_span.raw_span.return_value = mock_span
|
|
473
|
+
mock_span.get_data.return_value = {
|
|
474
|
+
"haystack.agent.step.llm.output": {
|
|
475
|
+
"replies": [
|
|
476
|
+
ChatMessage.from_assistant(
|
|
477
|
+
"This the LLM's response",
|
|
478
|
+
meta={"model": "test_model", "usage": {"prompt_tokens": 10, "completion_tokens": 5}},
|
|
479
|
+
)
|
|
480
|
+
]
|
|
481
|
+
},
|
|
482
|
+
}
|
|
483
|
+
|
|
484
|
+
handler = DefaultSpanHandler()
|
|
485
|
+
handler.handle(mock_span, component_type=None)
|
|
486
|
+
|
|
487
|
+
mock_span.update.assert_called_once_with(
|
|
488
|
+
usage_details={"input_tokens": 10, "output_tokens": 5}, model="test_model", completion_start_time=None
|
|
489
|
+
)
|
|
490
|
+
|
|
491
|
+
def test_handle_agent_step_llm_model_parameters(self):
|
|
492
|
+
mock_span = Mock()
|
|
493
|
+
mock_span.raw_span.return_value = mock_span
|
|
494
|
+
mock_span.get_data.return_value = {
|
|
495
|
+
"haystack.agent.step.llm.input": {
|
|
496
|
+
"messages": [ChatMessage.from_user("message")],
|
|
497
|
+
"generation_kwargs": {"temperature": 0.2, "response_format": {"type": "json_object"}},
|
|
498
|
+
},
|
|
499
|
+
"haystack.agent.step.llm.output": {"replies": [ChatMessage.from_assistant("reply")]},
|
|
500
|
+
}
|
|
501
|
+
|
|
502
|
+
handler = DefaultSpanHandler()
|
|
503
|
+
handler.handle(mock_span, component_type=None)
|
|
504
|
+
|
|
505
|
+
mock_span.update.assert_called_once_with(
|
|
506
|
+
usage_details=None,
|
|
507
|
+
model=None,
|
|
508
|
+
completion_start_time=None,
|
|
509
|
+
model_parameters={"temperature": 0.2, "response_format": '{"type": "json_object"}'},
|
|
510
|
+
)
|
|
511
|
+
|
|
347
512
|
def test_handle_bad_completion_start_time(self, caplog):
|
|
348
513
|
mock_span = Mock()
|
|
349
514
|
mock_span.raw_span.return_value = mock_span
|
|
@@ -463,6 +628,33 @@ class TestDefaultSpanHandler:
|
|
|
463
628
|
name="SentenceTransformersDocumentEmbedder", as_type="embedding"
|
|
464
629
|
)
|
|
465
630
|
|
|
631
|
+
@pytest.mark.parametrize(
|
|
632
|
+
"operation_name,tags,expected_name,expected_type",
|
|
633
|
+
[
|
|
634
|
+
("haystack.agent.step", {"haystack.agent.step": 1}, "agent step 1", "chain"),
|
|
635
|
+
("haystack.agent.step.llm", {}, "llm", "generation"),
|
|
636
|
+
("haystack.agent.step.tool", {"haystack.tool.name": "weather_tool"}, "tool - weather_tool", "tool"),
|
|
637
|
+
],
|
|
638
|
+
)
|
|
639
|
+
def test_create_span_agent_operations(self, operation_name, tags, expected_name, expected_type):
|
|
640
|
+
mock_client = Mock()
|
|
641
|
+
mock_client.start_as_current_observation = Mock(return_value=MockContextManager())
|
|
642
|
+
|
|
643
|
+
handler = DefaultSpanHandler()
|
|
644
|
+
handler.init_tracer(mock_client)
|
|
645
|
+
|
|
646
|
+
context = SpanContext(
|
|
647
|
+
name=operation_name,
|
|
648
|
+
operation_name=operation_name,
|
|
649
|
+
component_type=None,
|
|
650
|
+
tags=tags,
|
|
651
|
+
parent_span=LangfuseSpan(mock_client.start_as_current_observation()),
|
|
652
|
+
)
|
|
653
|
+
mock_client.start_as_current_observation.reset_mock()
|
|
654
|
+
|
|
655
|
+
handler.create_span(context)
|
|
656
|
+
mock_client.start_as_current_observation.assert_called_once_with(name=expected_name, as_type=expected_type)
|
|
657
|
+
|
|
466
658
|
def test_create_span_non_component(self):
|
|
467
659
|
"""Test that non-matching components create default span type."""
|
|
468
660
|
mock_client = Mock()
|
|
@@ -698,129 +890,6 @@ class TestLangfuseTracer:
|
|
|
698
890
|
assert span.raw_span()._data["model"] == "test_model"
|
|
699
891
|
assert span.raw_span()._data["completion_start_time"] == datetime.datetime(2021, 7, 27, 16, 2, 8, 12345) # noqa: DTZ001
|
|
700
892
|
|
|
701
|
-
def test_handle_tool_invoker(self):
|
|
702
|
-
"""
|
|
703
|
-
Test that the ToolInvoker span name is updated correctly with the tool names invoked for better UI/UX
|
|
704
|
-
"""
|
|
705
|
-
mock_span = Mock()
|
|
706
|
-
mock_span.raw_span.return_value = mock_span
|
|
707
|
-
|
|
708
|
-
# Simulate data for the ToolInvoker component
|
|
709
|
-
span_data = {
|
|
710
|
-
"haystack.component.name": "tool_invoker",
|
|
711
|
-
"haystack.component.type": "ToolInvoker",
|
|
712
|
-
"haystack.component.input": {
|
|
713
|
-
"messages": [
|
|
714
|
-
# Create a chat message with tool calls
|
|
715
|
-
ChatMessage.from_assistant(
|
|
716
|
-
text="Calling tools",
|
|
717
|
-
tool_calls=[
|
|
718
|
-
ToolCall(tool_name="search_tool", arguments={"query": "test"}),
|
|
719
|
-
ToolCall(tool_name="search_tool", arguments={"query": "another test"}),
|
|
720
|
-
ToolCall(tool_name="weather_tool", arguments={"location": "Berlin"}),
|
|
721
|
-
],
|
|
722
|
-
)
|
|
723
|
-
]
|
|
724
|
-
},
|
|
725
|
-
}
|
|
726
|
-
|
|
727
|
-
mock_span.get_data.return_value = span_data
|
|
728
|
-
|
|
729
|
-
handler = DefaultSpanHandler()
|
|
730
|
-
handler.handle(mock_span, component_type="ToolInvoker")
|
|
731
|
-
|
|
732
|
-
assert mock_span.update.call_count >= 1
|
|
733
|
-
name_update_call = None
|
|
734
|
-
for call in mock_span.update.call_args_list:
|
|
735
|
-
if "name" in call[1]:
|
|
736
|
-
name_update_call = call
|
|
737
|
-
break
|
|
738
|
-
|
|
739
|
-
assert name_update_call is not None, "No call to update the span name was made"
|
|
740
|
-
updated_name = name_update_call[1]["name"]
|
|
741
|
-
|
|
742
|
-
# verify the format of the updated span name to be: `original_component_name - [list_of_tool_names]`
|
|
743
|
-
assert updated_name != "tool_invoker", "Expected 'tool_invoker` to be upddated with tool names"
|
|
744
|
-
assert " - " in updated_name, f"Expected ' - ' in {updated_name}"
|
|
745
|
-
assert "[" in updated_name, f"Expected '[' in {updated_name}"
|
|
746
|
-
assert "]" in updated_name, f"Expected ']' in {updated_name}"
|
|
747
|
-
assert "tool_invoker" in updated_name, f"Expected 'tool_invoker' in {updated_name}"
|
|
748
|
-
assert "search_tool (x2)" in updated_name, f"Expected 'search_tool (x2)' in {updated_name}"
|
|
749
|
-
assert "weather_tool" in updated_name, f"Expected 'weather_tool' in {updated_name}"
|
|
750
|
-
|
|
751
|
-
def test_handle_tool_invoker_input_output_with_content_tracing(self):
|
|
752
|
-
"""
|
|
753
|
-
Test that ToolInvoker spans replace the noisy full-message input with just tool call
|
|
754
|
-
arguments, and populate output with tool results, when content tracing is enabled.
|
|
755
|
-
"""
|
|
756
|
-
mock_span = Mock()
|
|
757
|
-
mock_span.raw_span.return_value = mock_span
|
|
758
|
-
|
|
759
|
-
tool_call = ToolCall(id="call_123", tool_name="search_tool", arguments={"query": "RAG pipelines"})
|
|
760
|
-
|
|
761
|
-
span_data = {
|
|
762
|
-
"haystack.component.name": "tool_invoker",
|
|
763
|
-
"haystack.component.type": "ToolInvoker",
|
|
764
|
-
"haystack.component.input": {
|
|
765
|
-
"messages": [
|
|
766
|
-
ChatMessage.from_user("what is RAG?"),
|
|
767
|
-
ChatMessage.from_assistant(text="Calling search", tool_calls=[tool_call]),
|
|
768
|
-
]
|
|
769
|
-
},
|
|
770
|
-
"haystack.component.output": {
|
|
771
|
-
"tool_messages": [ChatMessage.from_tool("RAG stands for Retrieval-Augmented Generation", tool_call)]
|
|
772
|
-
},
|
|
773
|
-
}
|
|
774
|
-
mock_span.get_data.return_value = span_data
|
|
775
|
-
|
|
776
|
-
handler = DefaultSpanHandler()
|
|
777
|
-
with patch("haystack_integrations.tracing.langfuse.tracer.proxy_tracer.is_content_tracing_enabled", True):
|
|
778
|
-
handler.handle(mock_span, component_type="ToolInvoker")
|
|
779
|
-
|
|
780
|
-
update_calls = {k: v for call in mock_span.update.call_args_list for k, v in call[1].items()}
|
|
781
|
-
|
|
782
|
-
# input should be just the tool call arguments, not the full message list
|
|
783
|
-
assert update_calls["input"] == [{"tool_name": "search_tool", "arguments": {"query": "RAG pipelines"}}]
|
|
784
|
-
# output carries id/name/arguments so Langfuse detects the tool call and populates its filter
|
|
785
|
-
assert update_calls["output"] == [
|
|
786
|
-
{
|
|
787
|
-
"id": "call_123",
|
|
788
|
-
"name": "search_tool",
|
|
789
|
-
"arguments": {"query": "RAG pipelines"},
|
|
790
|
-
"result": "RAG stands for Retrieval-Augmented Generation",
|
|
791
|
-
"error": False,
|
|
792
|
-
}
|
|
793
|
-
]
|
|
794
|
-
|
|
795
|
-
def test_handle_tool_invoker_no_content_tracing(self):
|
|
796
|
-
"""
|
|
797
|
-
Test that ToolInvoker input/output is NOT updated when content tracing is disabled.
|
|
798
|
-
The span name update (tool names) should still happen.
|
|
799
|
-
"""
|
|
800
|
-
mock_span = Mock()
|
|
801
|
-
mock_span.raw_span.return_value = mock_span
|
|
802
|
-
|
|
803
|
-
tool_call = ToolCall(tool_name="weather_tool", arguments={"location": "Tokyo"})
|
|
804
|
-
|
|
805
|
-
span_data = {
|
|
806
|
-
"haystack.component.name": "tool_invoker",
|
|
807
|
-
"haystack.component.type": "ToolInvoker",
|
|
808
|
-
"haystack.component.input": {"messages": [ChatMessage.from_assistant(text="", tool_calls=[tool_call])]},
|
|
809
|
-
"haystack.component.output": {"tool_messages": [ChatMessage.from_tool("Sunny, 28°C", tool_call)]},
|
|
810
|
-
}
|
|
811
|
-
mock_span.get_data.return_value = span_data
|
|
812
|
-
|
|
813
|
-
handler = DefaultSpanHandler()
|
|
814
|
-
with patch("haystack_integrations.tracing.langfuse.tracer.proxy_tracer.is_content_tracing_enabled", False):
|
|
815
|
-
handler.handle(mock_span, component_type="ToolInvoker")
|
|
816
|
-
|
|
817
|
-
update_kwargs_keys = {k for call in mock_span.update.call_args_list for k in call[1]}
|
|
818
|
-
# name should still be updated
|
|
819
|
-
assert "name" in update_kwargs_keys
|
|
820
|
-
# input and output must NOT be set when content tracing is off
|
|
821
|
-
assert "input" not in update_kwargs_keys
|
|
822
|
-
assert "output" not in update_kwargs_keys
|
|
823
|
-
|
|
824
893
|
def test_trace_generation_invalid_start_time(self):
|
|
825
894
|
with patch("haystack_integrations.tracing.langfuse.tracer.langfuse.get_client"):
|
|
826
895
|
tracer = LangfuseTracer(tracer=MockLangfuseClient(), name="Haystack", public=False)
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{langfuse_haystack-5.2.0 → langfuse_haystack-7.0.0}/src/haystack_integrations/tracing/py.typed
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|