langfuse-haystack 5.2.0__tar.gz → 7.0.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (21) hide show
  1. {langfuse_haystack-5.2.0 → langfuse_haystack-7.0.0}/CHANGELOG.md +51 -8
  2. {langfuse_haystack-5.2.0 → langfuse_haystack-7.0.0}/LICENSE.txt +1 -1
  3. {langfuse_haystack-5.2.0 → langfuse_haystack-7.0.0}/PKG-INFO +3 -3
  4. {langfuse_haystack-5.2.0 → langfuse_haystack-7.0.0}/pyproject.toml +2 -1
  5. {langfuse_haystack-5.2.0 → langfuse_haystack-7.0.0}/src/haystack_integrations/components/connectors/langfuse/langfuse_connector.py +20 -8
  6. {langfuse_haystack-5.2.0 → langfuse_haystack-7.0.0}/src/haystack_integrations/tracing/langfuse/tracer.py +191 -88
  7. {langfuse_haystack-5.2.0 → langfuse_haystack-7.0.0}/tests/test_langfuse_connector.py +78 -38
  8. {langfuse_haystack-5.2.0 → langfuse_haystack-7.0.0}/tests/test_tracer.py +199 -130
  9. {langfuse_haystack-5.2.0 → langfuse_haystack-7.0.0}/.gitignore +0 -0
  10. {langfuse_haystack-5.2.0 → langfuse_haystack-7.0.0}/README.md +0 -0
  11. {langfuse_haystack-5.2.0 → langfuse_haystack-7.0.0}/example/basic_rag.py +0 -0
  12. {langfuse_haystack-5.2.0 → langfuse_haystack-7.0.0}/example/chat.py +0 -0
  13. {langfuse_haystack-5.2.0 → langfuse_haystack-7.0.0}/example/requirements.txt +0 -0
  14. {langfuse_haystack-5.2.0 → langfuse_haystack-7.0.0}/pydoc/config_docusaurus.yml +0 -0
  15. {langfuse_haystack-5.2.0 → langfuse_haystack-7.0.0}/src/haystack_integrations/components/connectors/langfuse/__init__.py +0 -0
  16. {langfuse_haystack-5.2.0 → langfuse_haystack-7.0.0}/src/haystack_integrations/components/connectors/py.typed +0 -0
  17. {langfuse_haystack-5.2.0 → langfuse_haystack-7.0.0}/src/haystack_integrations/tracing/langfuse/__init__.py +0 -0
  18. {langfuse_haystack-5.2.0 → langfuse_haystack-7.0.0}/src/haystack_integrations/tracing/py.typed +0 -0
  19. {langfuse_haystack-5.2.0 → langfuse_haystack-7.0.0}/tests/__init__.py +0 -0
  20. {langfuse_haystack-5.2.0 → langfuse_haystack-7.0.0}/tests/conftest.py +0 -0
  21. {langfuse_haystack-5.2.0 → langfuse_haystack-7.0.0}/tests/test_tracing.py +0 -0
@@ -1,5 +1,28 @@
1
1
  # Changelog
2
2
 
3
+ ## [integrations/langfuse-v6.0.0] - 2026-09-11
4
+
5
+ ### 🐛 Bug Fixes
6
+
7
+ - Fix new issues raised by ruff 0.16.0 (#3670)
8
+ - Standardize license files (#3771)
9
+
10
+ ### 🚜 Refactor
11
+
12
+ - [**breaking**] Langfuse connector - move tracer creation at warm_up (#3949)
13
+
14
+ ### ⚙️ CI
15
+
16
+ - Improve changelog generation; fix existing changelogs (#3883)
17
+
18
+
19
+ ## [integrations/langfuse-v5.2.0] - 2026-07-15
20
+
21
+ ### 🐛 Bug Fixes
22
+
23
+ - *(langfuse)* Replace noisy ToolInvoker input with focused tool call arguments (#3520)
24
+
25
+
3
26
  ## [integrations/langfuse-v5.1.0] - 2026-07-08
4
27
 
5
28
  ### 🐛 Bug Fixes
@@ -184,14 +207,19 @@
184
207
 
185
208
  - Properly cleanup Langfuse tracing context after pipeline run failures (#1999)
186
209
 
210
+ ### 🧹 Chores
211
+
212
+ - Remove black (#1985)
213
+
214
+
215
+ ## [integrations/langfuse-v2.1.0] - 2025-06-16
216
+
187
217
 
188
218
  ### 🧹 Chores
189
219
 
190
220
  - Pin langfuse<3.0.0 (#1904)
191
221
  - Align core-integrations Hatch scripts (#1898)
192
222
  - Update md files for new hatch scripts (#1911)
193
- - Remove black (#1985)
194
-
195
223
 
196
224
  ## [integrations/langfuse-v2.0.1] - 2025-06-02
197
225
 
@@ -248,10 +276,15 @@
248
276
 
249
277
  ### 🚀 Features
250
278
 
251
- - Adapt Ollama metadata to OpenAI format; support Ollama in Langfuse (#1577)
252
279
  - Unify traces of sub-pipelines within pipelines with Langfuse (#1624)
253
280
 
254
281
 
282
+ ## [integrations/langfuse-v0.10.0] - 2025-04-04
283
+
284
+ ### 🚀 Features
285
+
286
+ - Adapt Ollama metadata to OpenAI format; support Ollama in Langfuse (#1577)
287
+
255
288
 
256
289
  ## [integrations/langfuse-v0.9.0] - 2025-04-04
257
290
 
@@ -311,6 +344,12 @@
311
344
 
312
345
  ## [integrations/langfuse-v0.6.2] - 2025-01-02
313
346
 
347
+ ### 🌀 Miscellaneous
348
+
349
+ - Fix messages conversion to OpenAI format (#1272)
350
+
351
+ ## [integrations/langfuse-v0.6.1] - 2024-12-11
352
+
314
353
  ### 🚀 Features
315
354
 
316
355
  - Warn if LangfuseTracer initialized without tracing enabled (#1231)
@@ -322,7 +361,6 @@
322
361
  ### 🌀 Miscellaneous
323
362
 
324
363
  - Chore: Fix tracing_context_var lint errors (#1220)
325
- - Fix messages conversion to OpenAI format (#1272)
326
364
 
327
365
  ## [integrations/langfuse-v0.6.0] - 2024-11-18
328
366
 
@@ -355,6 +393,9 @@
355
393
 
356
394
  - Langfuse - support generation span for more LLMs (#1087)
357
395
 
396
+
397
+ ## [integrations/langfuse-v0.3.0] - 2024-09-11
398
+
358
399
  ### 🚜 Refactor
359
400
 
360
401
  - Remove usage of deprecated `ChatMessage.to_openai_format` (#1001)
@@ -390,10 +431,6 @@
390
431
 
391
432
  ## [integrations/langfuse-v0.1.0] - 2024-06-13
392
433
 
393
- ### 🚀 Features
394
-
395
- - Langfuse integration (#686)
396
-
397
434
  ### 🐛 Bug Fixes
398
435
 
399
436
  - Performance optimizations and value error when streaming in langfuse (#798)
@@ -407,4 +444,10 @@
407
444
  - Chore: change the pydoc renderer class (#718)
408
445
  - Docs: add missing api references (#728)
409
446
 
447
+ ## [integrations/langfuse-v0.0.4] - 2024-05-02
448
+
449
+ ### 🚀 Features
450
+
451
+ - Langfuse integration (#686)
452
+
410
453
  <!-- generated by git-cliff -->
@@ -58,7 +58,7 @@ APPENDIX: How to apply the Apache License to your work.
58
58
 
59
59
  To apply the Apache License to your work, attach the following boilerplate notice, with the fields enclosed by brackets "[]" replaced with your own identifying information. (Don't include the brackets!) The text should be enclosed in the appropriate comment syntax for the file format. We also recommend that a file or class name and description of purpose be included on the same "printed page" as the copyright notice for easier identification within third-party archives.
60
60
 
61
- Copyright [yyyy] [name of copyright owner]
61
+ Copyright 2023-present deepset GmbH
62
62
 
63
63
  Licensed under the Apache License, Version 2.0 (the "License");
64
64
  you may not use this file except in compliance with the License.
@@ -1,6 +1,6 @@
1
- Metadata-Version: 2.4
1
+ Metadata-Version: 2.5
2
2
  Name: langfuse-haystack
3
- Version: 5.2.0
3
+ Version: 7.0.0
4
4
  Summary: Langfuse integration for Haystack
5
5
  Project-URL: Documentation, https://github.com/deepset-ai/haystack-core-integrations/tree/main/integrations/langfuse#readme
6
6
  Project-URL: Issues, https://github.com/deepset-ai/haystack-core-integrations/issues
@@ -17,7 +17,7 @@ Classifier: Programming Language :: Python :: 3.13
17
17
  Classifier: Programming Language :: Python :: Implementation :: CPython
18
18
  Classifier: Programming Language :: Python :: Implementation :: PyPy
19
19
  Requires-Python: >=3.10
20
- Requires-Dist: haystack-ai>=2.22.0
20
+ Requires-Dist: haystack-ai>=3.0.0
21
21
  Requires-Dist: langfuse>=4.0.0
22
22
  Description-Content-Type: text/markdown
23
23
 
@@ -21,7 +21,7 @@ classifiers = [
21
21
  "Programming Language :: Python :: Implementation :: CPython",
22
22
  "Programming Language :: Python :: Implementation :: PyPy",
23
23
  ]
24
- dependencies = ["haystack-ai>=2.22.0", "langfuse>=4.0.0"]
24
+ dependencies = ["haystack-ai>=3.0.0", "langfuse>=4.0.0"]
25
25
 
26
26
  [project.urls]
27
27
  Documentation = "https://github.com/deepset-ai/haystack-core-integrations/tree/main/integrations/langfuse#readme"
@@ -130,6 +130,7 @@ ignore = [
130
130
  "PLR0912",
131
131
  "PLR0913",
132
132
  "PLR0915",
133
+ "PLR0917",
133
134
  # Asserts
134
135
  "S101",
135
136
  # Allow `Any` - used legitimately for dynamic types and SDK boundaries
@@ -157,18 +157,28 @@ class LangfuseConnector:
157
157
  self.span_handler = span_handler
158
158
  self.host = host
159
159
  self.langfuse_client_kwargs = langfuse_client_kwargs
160
+ self._httpx_client = httpx_client
161
+ self.tracer: LangfuseTracer | None = None
162
+
163
+ def warm_up(self) -> None:
164
+ """
165
+ Initialize the Langfuse client and enable tracing once.
166
+ """
167
+ if self.tracer is not None:
168
+ return
169
+
160
170
  resolved_langfuse_client_kwargs = {
161
- "secret_key": secret_key.resolve_value() if secret_key else None,
162
- "public_key": public_key.resolve_value() if public_key else None,
163
- "httpx_client": httpx_client,
164
- "host": host,
165
- **(langfuse_client_kwargs or {}),
171
+ "secret_key": self.secret_key.resolve_value() if self.secret_key else None,
172
+ "public_key": self.public_key.resolve_value() if self.public_key else None,
173
+ "httpx_client": self._httpx_client,
174
+ "host": self.host,
175
+ **(self.langfuse_client_kwargs or {}),
166
176
  }
167
177
  self.tracer = LangfuseTracer(
168
178
  tracer=Langfuse(**resolved_langfuse_client_kwargs),
169
- name=name,
170
- public=public,
171
- span_handler=span_handler,
179
+ name=self.name,
180
+ public=self.public,
181
+ span_handler=self.span_handler,
172
182
  )
173
183
  tracing.enable_tracing(self.tracer)
174
184
 
@@ -186,6 +196,8 @@ class LangfuseConnector:
186
196
  - `trace_url`: The URL to the tracing data.
187
197
  - `trace_id`: The ID of the trace.
188
198
  """
199
+ self.warm_up()
200
+ assert self.tracer is not None
189
201
  logger.debug(
190
202
  "Langfuse tracer invoked with the following context: '{invocation_context}'",
191
203
  invocation_context=invocation_context,
@@ -6,8 +6,7 @@ import contextlib
6
6
  import os
7
7
  import sys
8
8
  from abc import ABC, abstractmethod
9
- from collections import Counter
10
- from collections.abc import Iterator
9
+ from collections.abc import Iterator, Sequence
11
10
  from contextlib import AbstractContextManager
12
11
  from contextvars import ContextVar
13
12
  from dataclasses import dataclass
@@ -15,14 +14,15 @@ from datetime import datetime
15
14
  from typing import Any, Literal, cast
16
15
 
17
16
  from haystack import default_from_dict, default_to_dict, logging
18
- from haystack.dataclasses import ChatMessage
17
+ from haystack.dataclasses import ChatMessage, FileContent, ImageContent, TextContent
18
+ from haystack.tools import flatten_tools_or_toolsets
19
19
  from haystack.tracing import Span, Tracer
20
20
  from haystack.tracing import tracer as proxy_tracer
21
21
  from haystack.tracing import utils as tracing_utils
22
22
 
23
23
  import langfuse
24
+ from langfuse import LangfuseAgent, LangfuseGeneration, LangfuseTool, propagate_attributes
24
25
  from langfuse import LangfuseSpan as LangfuseClientSpan
25
- from langfuse import propagate_attributes
26
26
  from langfuse.types import TraceContext
27
27
 
28
28
  logger = logging.getLogger(__name__)
@@ -39,9 +39,16 @@ _COMPONENT_NAME_KEY = "haystack.component.name"
39
39
  _COMPONENT_TYPE_KEY = "haystack.component.type"
40
40
  _COMPONENT_OUTPUT_KEY = "haystack.component.output"
41
41
  _COMPONENT_INPUT_KEY = "haystack.component.input"
42
+ _AGENT_STEP_OPERATION = "haystack.agent.step"
43
+ _AGENT_STEP_LLM_OPERATION = "haystack.agent.step.llm"
44
+ _AGENT_STEP_TOOL_OPERATION = "haystack.agent.step.tool"
45
+ _AGENT_STEP_KEY = "haystack.agent.step"
46
+ _AGENT_STEP_LLM_INPUT_KEY = "haystack.agent.step.llm.input"
47
+ _AGENT_STEP_LLM_OUTPUT_KEY = "haystack.agent.step.llm.output"
48
+ _TOOL_NAME_KEY = "haystack.tool.name"
42
49
 
43
50
  # Type alias for observation span types
44
- ObservationSpanType = Literal["tool", "agent", "retriever", "embedding", "generation"]
51
+ ObservationSpanType = Literal["tool", "agent", "chain", "retriever", "embedding", "generation"]
45
52
 
46
53
  # External session metadata for trace correlation (Haystack system)
47
54
  # Stores trace_id, user_id, session_id, tags, version for root trace creation
@@ -88,27 +95,19 @@ class LangfuseSpan(Span):
88
95
  """
89
96
  if not proxy_tracer.is_content_tracing_enabled:
90
97
  return
98
+ # Only generation and agent observations carry chat messages, other spans like tool calls get a coerced value
99
+ is_chat = isinstance(self._span, (LangfuseGeneration, LangfuseAgent))
91
100
  if key.endswith(".input"):
92
- if "messages" in value:
93
- messages = [m.to_openai_dict_format(require_tool_call_ids=False) for m in (value.get("messages") or [])]
94
- if isinstance(gen_kwargs := value.get("generation_kwargs"), dict):
95
- self._span.update(input={"messages": messages, "generation_kwargs": gen_kwargs})
96
- else:
97
- self._span.update(input=messages)
98
- else:
99
- coerced_value = tracing_utils.coerce_tag_value(value)
100
- self._span.update(input=coerced_value)
101
+ self._span.update(
102
+ input=_format_chat_input(value=value) if is_chat else tracing_utils.coerce_tag_value(value)
103
+ )
101
104
  elif key.endswith(".output"):
102
- if "replies" in value:
103
- replies_list = value.get("replies") or []
104
- if all(isinstance(r, ChatMessage) for r in replies_list):
105
- replies = [m.to_openai_dict_format(require_tool_call_ids=False) for m in replies_list]
106
- else:
107
- replies = replies_list
108
- self._span.update(output=replies)
105
+ if is_chat:
106
+ self._span.update(output=_format_chat_output(value=value))
107
+ elif isinstance(self._span, LangfuseTool):
108
+ self._span.update(output=_format_tool_output(value=value))
109
109
  else:
110
- coerced_value = tracing_utils.coerce_tag_value(value)
111
- self._span.update(output=coerced_value)
110
+ self._span.update(output=tracing_utils.coerce_tag_value(value))
112
111
 
113
112
  self._data[key] = value
114
113
 
@@ -276,6 +275,150 @@ def _sanitize_usage_data(usage: dict[str, Any]) -> dict[str, Any]:
276
275
  return sanitized
277
276
 
278
277
 
278
+ def _to_openai_content_parts(parts: Sequence[TextContent | ImageContent | FileContent]) -> list[dict[str, Any]]:
279
+ """
280
+ Convert content parts to the `text`, `image_url` and `file` parts of OpenAI user messages.
281
+
282
+ :param parts: The content parts, e.g. the result of a tool.
283
+ :returns: The content parts in OpenAI format.
284
+ """
285
+ content: list[dict[str, Any]] = []
286
+ for part in parts:
287
+ if isinstance(part, TextContent):
288
+ content.append({"type": "text", "text": part.text})
289
+ elif isinstance(part, ImageContent):
290
+ image_url = f"data:{part.mime_type or 'image/jpeg'};base64,{part.base64_image}"
291
+ content.append({"type": "image_url", "image_url": {"url": image_url}})
292
+ elif isinstance(part, FileContent):
293
+ file_data = f"data:{part.mime_type or 'application/pdf'};base64,{part.base64_data}"
294
+ content.append({"type": "file", "file": {"file_data": file_data, "filename": part.filename}})
295
+ return content
296
+
297
+
298
+ def _to_openai_message(message: ChatMessage) -> dict[str, Any]:
299
+ """
300
+ Convert a ChatMessage to OpenAI Chat Completions format for Langfuse.
301
+
302
+ Tool results made of content parts get `text`, `image_url` and `file` parts, as used in OpenAI user messages.
303
+
304
+ :param message: The ChatMessage to convert.
305
+ :returns: The message in OpenAI format.
306
+ :raises ValueError: If the message has no OpenAI format, e.g. because it has no content.
307
+ """
308
+ result = message.tool_call_result
309
+ if result is None or isinstance(result.result, str):
310
+ return message.to_openai_dict_format(require_tool_call_ids=False)
311
+
312
+ openai_message: dict[str, Any] = {"role": "tool", "content": _to_openai_content_parts(parts=result.result)}
313
+ if result.origin.id is not None:
314
+ openai_message["tool_call_id"] = result.origin.id
315
+ return openai_message
316
+
317
+
318
+ def _format_tool_output(value: Any) -> Any:
319
+ """
320
+ Format the result of a tool call for Langfuse.
321
+
322
+ :param value: The tool result.
323
+ :returns: The content parts in OpenAI format if the result is a list of content parts, e.g. text and images.
324
+ Any other result is returned as a coerced tag value.
325
+ """
326
+ if (
327
+ isinstance(value, list)
328
+ and value
329
+ and all(isinstance(part, (TextContent, ImageContent, FileContent)) for part in value)
330
+ ):
331
+ return _to_openai_content_parts(parts=value)
332
+ return tracing_utils.coerce_tag_value(value)
333
+
334
+
335
+ def _format_chat_input(value: Any) -> Any:
336
+ """
337
+ Format the input of a generation or agent span for Langfuse.
338
+
339
+ :param value: The traced input, e.g. the inputs of a ChatGenerator.
340
+ :returns: The messages in OpenAI format. If `generation_kwargs` or `tools` are set, a dictionary with the
341
+ `messages` and the `generation_kwargs` and OpenAI tool definitions is returned instead.
342
+ Messages that have no OpenAI format are returned as a coerced tag value.
343
+ Inputs without `messages` are returned as a coerced tag value.
344
+ """
345
+ if "messages" not in value:
346
+ return tracing_utils.coerce_tag_value(value)
347
+ messages: Any
348
+ try:
349
+ messages = [_to_openai_message(message=m) for m in (value.get("messages") or [])]
350
+ except ValueError:
351
+ messages = tracing_utils.coerce_tag_value(value.get("messages"))
352
+
353
+ formatted: dict[str, Any] = {"messages": messages}
354
+ if isinstance(gen_kwargs := value.get("generation_kwargs"), dict):
355
+ formatted["generation_kwargs"] = gen_kwargs
356
+ # Langfuse shows the tools of `{"messages": ..., "tools": ...}` inputs next to the messages
357
+ if tools := value.get("tools"):
358
+ formatted["tools"] = [{"type": "function", "function": t.tool_spec} for t in flatten_tools_or_toolsets(tools)]
359
+ return formatted if len(formatted) > 1 else messages
360
+
361
+
362
+ def _format_chat_output(value: Any) -> Any:
363
+ """
364
+ Format the output of a generation or agent span for Langfuse.
365
+
366
+ :param value: The traced output, e.g. the outputs of a ChatGenerator.
367
+ :returns: The replies in OpenAI format. String replies are returned as they are.
368
+ Replies that have no OpenAI format are returned as a coerced tag value.
369
+ Outputs without `replies` are returned as a coerced tag value.
370
+ """
371
+ if "replies" not in value:
372
+ return tracing_utils.coerce_tag_value(value)
373
+ replies = value.get("replies") or []
374
+ # Generators that aren't ChatGenerators return string replies
375
+ if not all(isinstance(r, ChatMessage) for r in replies):
376
+ return replies
377
+ try:
378
+ return [_to_openai_message(message=m) for m in replies]
379
+ except ValueError:
380
+ return tracing_utils.coerce_tag_value(replies)
381
+
382
+
383
+ def _update_generation_details(
384
+ span: LangfuseSpan, chat_generator_inputs: dict[str, Any], chat_generator_output: dict[str, Any]
385
+ ) -> None:
386
+ """
387
+ Add the model details of a ChatGenerator call to the span.
388
+
389
+ The model, token usage and completion start time come from the first reply, the model parameters from the
390
+ `generation_kwargs` passed to the ChatGenerator.
391
+
392
+ :param span: The generation span.
393
+ :param chat_generator_inputs: The inputs of the ChatGenerator.
394
+ :param chat_generator_output: The outputs of the ChatGenerator.
395
+ """
396
+ update_kwargs: dict[str, Any] = {}
397
+ if replies := chat_generator_output.get("replies"):
398
+ meta = replies[0].meta
399
+ completion_start_time = meta.get("completion_start_time")
400
+ if completion_start_time:
401
+ try:
402
+ completion_start_time = datetime.fromisoformat(completion_start_time)
403
+ except ValueError:
404
+ logger.error(f"Failed to parse completion_start_time: {completion_start_time}")
405
+ completion_start_time = None
406
+ usage = meta.get("usage")
407
+ update_kwargs["usage_details"] = _sanitize_usage_data(usage=usage) if usage else None
408
+ update_kwargs["model"] = meta.get("model")
409
+ update_kwargs["completion_start_time"] = completion_start_time
410
+ if generation_kwargs := chat_generator_inputs.get("generation_kwargs"):
411
+ # Langfuse model parameters only take primitive values, so nested values like `response_format` are coerced
412
+ update_kwargs["model_parameters"] = {
413
+ key: value
414
+ if value is None or isinstance(value, tracing_utils.PRIMITIVE_TYPES)
415
+ else tracing_utils.coerce_tag_value(value)
416
+ for key, value in generation_kwargs.items()
417
+ }
418
+ if update_kwargs:
419
+ span.raw_span().update(**update_kwargs)
420
+
421
+
279
422
  class DefaultSpanHandler(SpanHandler):
280
423
  """DefaultSpanHandler provides the default Langfuse tracing behavior for Haystack."""
281
424
 
@@ -313,9 +456,19 @@ class DefaultSpanHandler(SpanHandler):
313
456
  return span
314
457
 
315
458
  span_type = None
459
+ name = context.name
316
460
 
317
- if context.component_type == "ToolInvoker":
461
+ # Agent spans carry no component tags, so they're matched by operation name
462
+ if context.operation_name == _AGENT_STEP_OPERATION:
463
+ span_type = "chain"
464
+ name = f"agent step {context.tags.get(_AGENT_STEP_KEY)}"
465
+ elif context.operation_name == _AGENT_STEP_LLM_OPERATION:
466
+ span_type = "generation"
467
+ name = "llm"
468
+ elif context.operation_name == _AGENT_STEP_TOOL_OPERATION:
318
469
  span_type = "tool"
470
+ tool_name = context.tags.get(_TOOL_NAME_KEY)
471
+ name = f"tool - {tool_name}" if tool_name else "tool"
319
472
  elif context.operation_name == "haystack.agent.run":
320
473
  span_type = "agent"
321
474
  elif context.component_type and context.component_type.endswith("Retriever"):
@@ -327,12 +480,10 @@ class DefaultSpanHandler(SpanHandler):
327
480
 
328
481
  if span_type:
329
482
  return LangfuseSpan(
330
- self.tracer.start_as_current_observation(
331
- name=context.name, as_type=cast(ObservationSpanType, span_type)
332
- )
483
+ self.tracer.start_as_current_observation(name=name, as_type=cast(ObservationSpanType, span_type))
333
484
  )
334
485
  else:
335
- return LangfuseSpan(self.tracer.start_as_current_observation(name=context.name))
486
+ return LangfuseSpan(self.tracer.start_as_current_observation(name=name))
336
487
 
337
488
  def handle(self, span: LangfuseSpan, component_type: str | None) -> None:
338
489
  """Process and enrich a span after component execution."""
@@ -342,66 +493,18 @@ class DefaultSpanHandler(SpanHandler):
342
493
  coerced_input = tracing_utils.coerce_tag_value(span.get_data().get(_PIPELINE_INPUT_KEY))
343
494
  coerced_output = tracing_utils.coerce_tag_value(span.get_data().get(_PIPELINE_OUTPUT_KEY))
344
495
  span.raw_span().update(input=coerced_input, output=coerced_output)
345
- # special case for ToolInvoker (to update the span name to be: `original_component_name - [tool_names]`)
346
- if component_type == "ToolInvoker":
347
- tool_names: list[str] = []
348
- tool_calls_input: list[dict[str, Any]] = []
349
- messages = span.get_data().get(_COMPONENT_INPUT_KEY, {}).get("messages", [])
350
- for message in messages:
351
- if isinstance(message, ChatMessage) and message.tool_calls:
352
- for call in message.tool_calls:
353
- tool_names.append(call.tool_name)
354
- tool_calls_input.append({"tool_name": call.tool_name, "arguments": call.arguments})
355
-
356
- if tool_names:
357
- # Fallback to "ToolInvoker" if we can't retrieve component name
358
- tool_invoker_name = span.get_data().get(_COMPONENT_NAME_KEY, "ToolInvoker")
359
- tool_counts = Counter(tool_names) # how many times each tool was called
360
- formatted_names = [f"{name} (x{count})" if count > 1 else name for name, count in tool_counts.items()]
361
- span.raw_span().update(name=f"{tool_invoker_name} - {sorted(formatted_names)}")
362
-
363
- if tool_calls_input and proxy_tracer.is_content_tracing_enabled:
364
- # Replace the noisy full message history with just the tool call arguments
365
- span.raw_span().update(input=tool_calls_input)
366
-
367
- output_messages = span.get_data().get(_COMPONENT_OUTPUT_KEY, {}).get("tool_messages", [])
368
- tool_results: list[dict[str, Any]] = []
369
- for message in output_messages:
370
- if isinstance(message, ChatMessage) and message.tool_call_results:
371
- for tcr in message.tool_call_results:
372
- origin = tcr.origin
373
- # Keys `name`, `arguments` and `id` let Langfuse detect these as tool
374
- # calls at ingestion and populate the Tool Call Name filter in the UI.
375
- tool_results.append(
376
- {
377
- "id": origin.id if origin else None,
378
- "name": origin.tool_name if origin else None,
379
- "arguments": origin.arguments if origin else None,
380
- "result": tcr.result,
381
- "error": tcr.error,
382
- }
383
- )
384
- if tool_results:
385
- span.raw_span().update(output=tool_results)
386
-
387
- if component_type and component_type.endswith("ChatGenerator"):
388
- replies = span.get_data().get(_COMPONENT_OUTPUT_KEY, {}).get("replies")
389
- if replies:
390
- meta = replies[0].meta
391
- completion_start_time = meta.get("completion_start_time")
392
- if completion_start_time:
393
- try:
394
- completion_start_time = datetime.fromisoformat(completion_start_time)
395
- except ValueError:
396
- logger.error(f"Failed to parse completion_start_time: {completion_start_time}")
397
- completion_start_time = None
398
- usage = meta.get("usage")
399
- sanitized_usage = _sanitize_usage_data(usage) if usage else None
400
- span.raw_span().update(
401
- usage_details=sanitized_usage,
402
- model=meta.get("model"),
403
- completion_start_time=completion_start_time,
404
- )
496
+ if _AGENT_STEP_LLM_OUTPUT_KEY in span.get_data():
497
+ _update_generation_details(
498
+ span=span,
499
+ chat_generator_inputs=span.get_data().get(_AGENT_STEP_LLM_INPUT_KEY, {}),
500
+ chat_generator_output=span.get_data()[_AGENT_STEP_LLM_OUTPUT_KEY],
501
+ )
502
+ elif component_type and component_type.endswith("ChatGenerator"):
503
+ _update_generation_details(
504
+ span=span,
505
+ chat_generator_inputs=span.get_data().get(_COMPONENT_INPUT_KEY, {}),
506
+ chat_generator_output=span.get_data().get(_COMPONENT_OUTPUT_KEY, {}),
507
+ )
405
508
  elif component_type and component_type.endswith("Generator"):
406
509
  meta = span.get_data().get(_COMPONENT_OUTPUT_KEY, {}).get("meta")
407
510
  if meta:
@@ -6,48 +6,53 @@ import os
6
6
 
7
7
  os.environ["HAYSTACK_CONTENT_TRACING_ENABLED"] = "true"
8
8
 
9
- from unittest.mock import Mock
9
+ from unittest.mock import Mock, patch
10
10
 
11
- from haystack import Pipeline
11
+ import httpx
12
+ import pytest
13
+ from haystack import Pipeline, tracing
12
14
  from haystack.components.builders import ChatPromptBuilder
13
15
  from haystack.components.generators.chat import OpenAIChatGenerator
16
+ from haystack.tracing.tracer import NullTracer
14
17
  from haystack.utils import Secret
15
18
 
16
19
  from haystack_integrations.components.connectors.langfuse import LangfuseConnector
17
20
  from haystack_integrations.tracing.langfuse import DefaultSpanHandler
18
21
 
19
22
 
23
+ @pytest.fixture
24
+ def mock_langfuse(monkeypatch):
25
+ monkeypatch.setattr(tracing.tracer, "actual_tracer", NullTracer())
26
+ with patch("haystack_integrations.components.connectors.langfuse.langfuse_connector.Langfuse") as constructor:
27
+ constructor.return_value.get_trace_url.return_value = "https://example.com/trace"
28
+ constructor.return_value.get_current_trace_id.return_value = "12345"
29
+ yield constructor
30
+
31
+
20
32
  class CustomSpanHandler(DefaultSpanHandler):
21
33
  def handle(self, span, component_type=None):
22
34
  pass
23
35
 
24
36
 
25
- class TestLangfuseConnector:
26
- def test_run(self, monkeypatch):
27
- monkeypatch.setenv("LANGFUSE_SECRET_KEY", "secret")
28
- monkeypatch.setenv("LANGFUSE_PUBLIC_KEY", "public")
29
-
37
+ class TestRun:
38
+ def test_run(self, mock_langfuse):
30
39
  langfuse_connector = LangfuseConnector(
31
40
  name="Chat example - OpenAI",
32
41
  public=True,
33
- secret_key=Secret.from_env_var("LANGFUSE_SECRET_KEY"),
34
- public_key=Secret.from_env_var("LANGFUSE_PUBLIC_KEY"),
42
+ secret_key=Secret.from_token("secret"),
43
+ public_key=Secret.from_token("public"),
35
44
  )
36
45
 
37
- mock_tracer = Mock()
38
- mock_tracer.get_trace_url.return_value = "https://example.com/trace"
39
- mock_tracer.get_trace_id.return_value = "12345"
40
- langfuse_connector.tracer = mock_tracer
41
-
42
46
  response = langfuse_connector.run(invocation_context={"some_key": "some_value"})
43
47
  assert response["name"] == "Chat example - OpenAI"
44
48
  assert response["trace_url"] == "https://example.com/trace"
45
49
  assert response["trace_id"] == "12345"
50
+ mock_langfuse.assert_called_once()
51
+ assert tracing.tracer.actual_tracer is langfuse_connector.tracer
46
52
 
47
- def test_to_dict(self, monkeypatch):
48
- monkeypatch.setenv("LANGFUSE_SECRET_KEY", "secret")
49
- monkeypatch.setenv("LANGFUSE_PUBLIC_KEY", "public")
50
53
 
54
+ class TestSerialization:
55
+ def test_to_dict(self):
51
56
  langfuse_connector = LangfuseConnector(name="Chat example - OpenAI")
52
57
  serialized = langfuse_connector.to_dict()
53
58
 
@@ -72,10 +77,7 @@ class TestLangfuseConnector:
72
77
  },
73
78
  }
74
79
 
75
- def test_to_dict_with_params(self, monkeypatch):
76
- monkeypatch.setenv("LANGFUSE_SECRET_KEY", "secret")
77
- monkeypatch.setenv("LANGFUSE_PUBLIC_KEY", "public")
78
-
80
+ def test_to_dict_with_params(self):
79
81
  langfuse_connector = LangfuseConnector(
80
82
  name="Chat example - OpenAI",
81
83
  public=True,
@@ -111,10 +113,7 @@ class TestLangfuseConnector:
111
113
  },
112
114
  }
113
115
 
114
- def test_from_dict(self, monkeypatch):
115
- monkeypatch.setenv("LANGFUSE_SECRET_KEY", "secret")
116
- monkeypatch.setenv("LANGFUSE_PUBLIC_KEY", "public")
117
-
116
+ def test_from_dict(self):
118
117
  data = {
119
118
  "type": "haystack_integrations.components.connectors.langfuse.langfuse_connector.LangfuseConnector",
120
119
  "init_parameters": {
@@ -144,10 +143,7 @@ class TestLangfuseConnector:
144
143
  assert langfuse_connector.host is None
145
144
  assert langfuse_connector.langfuse_client_kwargs is None
146
145
 
147
- def test_from_dict_without_span_handler(self, monkeypatch):
148
- monkeypatch.setenv("LANGFUSE_SECRET_KEY", "secret")
149
- monkeypatch.setenv("LANGFUSE_PUBLIC_KEY", "public")
150
-
146
+ def test_from_dict_without_span_handler(self):
151
147
  # All keys that would point to None (span_handler, host, langfuse_client_kwargs) are intentionally absent
152
148
  data = {
153
149
  "type": "haystack_integrations.components.connectors.langfuse.langfuse_connector.LangfuseConnector",
@@ -175,10 +171,7 @@ class TestLangfuseConnector:
175
171
  assert langfuse_connector.host is None
176
172
  assert langfuse_connector.langfuse_client_kwargs is None
177
173
 
178
- def test_from_dict_with_params(self, monkeypatch):
179
- monkeypatch.setenv("LANGFUSE_SECRET_KEY", "secret")
180
- monkeypatch.setenv("LANGFUSE_PUBLIC_KEY", "public")
181
-
174
+ def test_from_dict_with_params(self):
182
175
  data = {
183
176
  "type": "haystack_integrations.components.connectors.langfuse.langfuse_connector.LangfuseConnector",
184
177
  "init_parameters": {
@@ -212,10 +205,8 @@ class TestLangfuseConnector:
212
205
  assert langfuse_connector.host == "https://example.com"
213
206
  assert langfuse_connector.langfuse_client_kwargs == {"timeout": 30.0}
214
207
 
215
- def test_from_dict_with_legacy_span_handler_format(self, monkeypatch):
208
+ def test_from_dict_with_legacy_span_handler_format(self):
216
209
  # Pipelines serialized before this fix wrap span_handler as {"type": ..., "data": {...}}
217
- monkeypatch.setenv("LANGFUSE_SECRET_KEY", "secret")
218
- monkeypatch.setenv("LANGFUSE_PUBLIC_KEY", "public")
219
210
 
220
211
  data = {
221
212
  "type": "haystack_integrations.components.connectors.langfuse.langfuse_connector.LangfuseConnector",
@@ -249,8 +240,6 @@ class TestLangfuseConnector:
249
240
 
250
241
  def test_pipeline_serialization(self, monkeypatch):
251
242
  # Set test env vars
252
- monkeypatch.setenv("LANGFUSE_SECRET_KEY", "secret")
253
- monkeypatch.setenv("LANGFUSE_PUBLIC_KEY", "public")
254
243
  monkeypatch.setenv("OPENAI_API_KEY", "openai_api_key")
255
244
 
256
245
  # Create pipeline with OpenAI LLM
@@ -285,3 +274,54 @@ class TestLangfuseConnector:
285
274
 
286
275
  # Verify pipeline is the same
287
276
  assert new_pipe == pipe
277
+
278
+
279
+ class TestComponentLifecycle:
280
+ @pytest.mark.parametrize("key", ["LANGFUSE_SECRET_KEY", "LANGFUSE_PUBLIC_KEY"])
281
+ def test_key_resolved_at_warm_up_not_init(self, key, monkeypatch, mock_langfuse):
282
+ monkeypatch.setenv("LANGFUSE_SECRET_KEY", "secret")
283
+ monkeypatch.setenv("LANGFUSE_PUBLIC_KEY", "public")
284
+ monkeypatch.delenv(key)
285
+ connector = LangfuseConnector("test")
286
+ assert connector.tracer is None
287
+ mock_langfuse.assert_not_called()
288
+ assert not tracing.is_tracing_enabled()
289
+
290
+ with pytest.raises(ValueError, match=key):
291
+ connector.warm_up()
292
+ assert connector.tracer is None
293
+ mock_langfuse.assert_not_called()
294
+
295
+ monkeypatch.setenv(key, "available-now")
296
+ connector.warm_up()
297
+ assert connector.tracer is not None
298
+ mock_langfuse.assert_called_once()
299
+
300
+ def test_warm_up_passes_configuration_and_is_idempotent(self, mock_langfuse):
301
+ client = Mock(spec=httpx.Client)
302
+ handler = CustomSpanHandler()
303
+ connector = LangfuseConnector(
304
+ "configured",
305
+ public=True,
306
+ public_key=Secret.from_token("public"),
307
+ secret_key=Secret.from_token("secret"),
308
+ httpx_client=client,
309
+ span_handler=handler,
310
+ host="https://example.com",
311
+ langfuse_client_kwargs={"timeout": 30, "host": "https://override.example.com"},
312
+ )
313
+ connector.warm_up()
314
+ first_tracer = connector.tracer
315
+ connector.warm_up()
316
+ assert connector.tracer is first_tracer
317
+ mock_langfuse.assert_called_once_with(
318
+ public_key="public",
319
+ secret_key="secret",
320
+ httpx_client=client,
321
+ host="https://override.example.com",
322
+ timeout=30,
323
+ )
324
+ assert handler.tracer is mock_langfuse.return_value
325
+ assert connector.tracer is not None
326
+ assert connector.tracer._name == "configured"
327
+ assert connector.tracer._public is True
@@ -3,13 +3,18 @@
3
3
  # SPDX-License-Identifier: Apache-2.0
4
4
 
5
5
  import asyncio
6
+ import base64
6
7
  import datetime
7
8
  import logging
8
9
  import sys
9
10
  from unittest.mock import MagicMock, Mock, patch
10
11
 
11
12
  import pytest
12
- from haystack.dataclasses import ChatMessage, ToolCall
13
+ from haystack.dataclasses import ChatMessage, ChatRole, FileContent, ImageContent, TextContent, ToolCall
14
+ from haystack.tools import Tool, Toolset
15
+ from haystack.tracing import utils as tracing_utils
16
+ from langfuse import LangfuseAgent, LangfuseGeneration, LangfuseTool
17
+ from langfuse import LangfuseSpan as LangfuseClientSpan
13
18
 
14
19
  from haystack_integrations.tracing.langfuse.tracer import (
15
20
  _COMPONENT_OUTPUT_KEY,
@@ -32,8 +37,8 @@ def mock_get_client():
32
37
  class MockContextManager:
33
38
  """Mock context manager that simulates Langfuse v4 context managers"""
34
39
 
35
- def __init__(self, name="mock_span"):
36
- self._span = MockSpan(name)
40
+ def __init__(self, name="mock_span", span=None):
41
+ self._span = span or MockSpan(name)
37
42
 
38
43
  def __enter__(self):
39
44
  return self._span
@@ -139,8 +144,9 @@ class TestLangfuseSpan:
139
144
  mock_context_manager._span.update.assert_called_with(output="output_value")
140
145
 
141
146
  # set_content_tag method can update input and output of the span object with messages/replies
142
- def test_set_content_tag_updates_input_and_output_with_messages(self):
143
- mock_context_manager = MockContextManager()
147
+ @pytest.mark.parametrize("observation_class", [LangfuseGeneration, LangfuseAgent])
148
+ def test_set_content_tag_updates_input_and_output_with_messages(self, observation_class):
149
+ mock_context_manager = MockContextManager(span=Mock(spec=observation_class))
144
150
  span = LangfuseSpan(mock_context_manager)
145
151
 
146
152
  with patch("haystack_integrations.tracing.langfuse.tracer.proxy_tracer.is_content_tracing_enabled", True):
@@ -180,8 +186,40 @@ class TestLangfuseSpan:
180
186
  # check we handle properly string list replies
181
187
  assert mock_context_manager._span.update.call_args_list[0][1] == {"output": ["reply1", "reply2"]}
182
188
 
189
+ def test_set_content_tag_input_with_tools(self):
190
+ mock_context_manager = MockContextManager(span=Mock(spec=LangfuseGeneration))
191
+ span = LangfuseSpan(mock_context_manager)
192
+ weather = Tool(
193
+ name="weather",
194
+ description="Get the weather",
195
+ parameters={"type": "object", "properties": {"city": {"type": "string"}}},
196
+ function=lambda city: city,
197
+ )
198
+
199
+ with patch("haystack_integrations.tracing.langfuse.tracer.proxy_tracer.is_content_tracing_enabled", True):
200
+ span.set_content_tag(
201
+ "haystack.agent.step.llm.input",
202
+ {"messages": [ChatMessage.from_user("message")], "tools": Toolset([weather])},
203
+ )
204
+
205
+ mock_context_manager._span.update.assert_called_once_with(
206
+ input={
207
+ "messages": [{"role": "user", "content": "message"}],
208
+ "tools": [
209
+ {
210
+ "type": "function",
211
+ "function": {
212
+ "name": "weather",
213
+ "description": "Get the weather",
214
+ "parameters": {"type": "object", "properties": {"city": {"type": "string"}}},
215
+ },
216
+ }
217
+ ],
218
+ }
219
+ )
220
+
183
221
  def test_set_content_tag_messages_none_does_not_raise(self):
184
- mock_context_manager = MockContextManager()
222
+ mock_context_manager = MockContextManager(span=Mock(spec=LangfuseGeneration))
185
223
  span = LangfuseSpan(mock_context_manager)
186
224
 
187
225
  with patch("haystack_integrations.tracing.langfuse.tracer.proxy_tracer.is_content_tracing_enabled", True):
@@ -190,7 +228,7 @@ class TestLangfuseSpan:
190
228
  assert mock_context_manager._span.update.call_args_list[0][1] == {"input": []}
191
229
 
192
230
  def test_set_content_tag_replies_none_does_not_raise(self):
193
- mock_context_manager = MockContextManager()
231
+ mock_context_manager = MockContextManager(span=Mock(spec=LangfuseGeneration))
194
232
  span = LangfuseSpan(mock_context_manager)
195
233
 
196
234
  with patch("haystack_integrations.tracing.langfuse.tracer.proxy_tracer.is_content_tracing_enabled", True):
@@ -198,6 +236,91 @@ class TestLangfuseSpan:
198
236
  assert mock_context_manager._span.update.call_count == 1
199
237
  assert mock_context_manager._span.update.call_args_list[0][1] == {"output": []}
200
238
 
239
+ @pytest.mark.parametrize(
240
+ "key,value,expected",
241
+ [
242
+ ("haystack.agent.step.tool.input", {"messages": ["hi", "there"]}, '{"messages": ["hi", "there"]}'),
243
+ ("haystack.agent.step.tool.output", None, ""),
244
+ ("haystack.agent.step.tool.output", 42, 42),
245
+ ("haystack.agent.step.tool.output", "No replies found", "No replies found"),
246
+ ("haystack.agent.step.tool.output", {"replies": 5}, '{"replies": 5}'),
247
+ ("haystack.component.input", {"messages": ["hi", "there"]}, '{"messages": ["hi", "there"]}'),
248
+ ],
249
+ )
250
+ def test_set_content_tag_non_chat_span_coerces_value(self, key, value, expected):
251
+ span_spec = LangfuseTool if key.startswith("haystack.agent.step.tool") else LangfuseClientSpan
252
+ mock_context_manager = MockContextManager(span=Mock(spec=span_spec))
253
+ span = LangfuseSpan(mock_context_manager)
254
+
255
+ with patch("haystack_integrations.tracing.langfuse.tracer.proxy_tracer.is_content_tracing_enabled", True):
256
+ span.set_content_tag(key, value)
257
+
258
+ field = "input" if key.endswith(".input") else "output"
259
+ mock_context_manager._span.update.assert_called_once_with(**{field: expected})
260
+
261
+ def test_set_content_tag_tool_result_with_image_and_file(self):
262
+ mock_context_manager = MockContextManager(span=Mock(spec=LangfuseGeneration))
263
+ span = LangfuseSpan(mock_context_manager)
264
+ png = base64.b64encode(b"\x89PNG\r\n\x1a\n").decode()
265
+ pdf = base64.b64encode(b"%PDF-1.4").decode()
266
+ tool_message = ChatMessage.from_tool(
267
+ tool_result=[
268
+ TextContent("chart"),
269
+ ImageContent(base64_image=png, mime_type="image/png"),
270
+ FileContent(base64_data=pdf, mime_type="application/pdf", filename="report.pdf"),
271
+ ],
272
+ origin=ToolCall(tool_name="plot", arguments={}, id="call_1"),
273
+ )
274
+
275
+ with patch("haystack_integrations.tracing.langfuse.tracer.proxy_tracer.is_content_tracing_enabled", True):
276
+ span.set_content_tag("haystack.agent.step.llm.input", {"messages": [tool_message]})
277
+
278
+ mock_context_manager._span.update.assert_called_once_with(
279
+ input=[
280
+ {
281
+ "role": "tool",
282
+ "content": [
283
+ {"type": "text", "text": "chart"},
284
+ {"type": "image_url", "image_url": {"url": f"data:image/png;base64,{png}"}},
285
+ {
286
+ "type": "file",
287
+ "file": {"file_data": f"data:application/pdf;base64,{pdf}", "filename": "report.pdf"},
288
+ },
289
+ ],
290
+ "tool_call_id": "call_1",
291
+ }
292
+ ]
293
+ )
294
+
295
+ def test_set_content_tag_tool_output_with_image(self):
296
+ mock_context_manager = MockContextManager(span=Mock(spec=LangfuseTool))
297
+ span = LangfuseSpan(mock_context_manager)
298
+ png = base64.b64encode(b"\x89PNG\r\n\x1a\n").decode()
299
+
300
+ with patch("haystack_integrations.tracing.langfuse.tracer.proxy_tracer.is_content_tracing_enabled", True):
301
+ span.set_content_tag(
302
+ "haystack.agent.step.tool.output",
303
+ [TextContent("chart"), ImageContent(base64_image=png, mime_type="image/png")],
304
+ )
305
+
306
+ mock_context_manager._span.update.assert_called_once_with(
307
+ output=[
308
+ {"type": "text", "text": "chart"},
309
+ {"type": "image_url", "image_url": {"url": f"data:image/png;base64,{png}"}},
310
+ ]
311
+ )
312
+
313
+ def test_set_content_tag_messages_without_openai_format_are_coerced(self):
314
+ mock_context_manager = MockContextManager(span=Mock(spec=LangfuseGeneration))
315
+ span = LangfuseSpan(mock_context_manager)
316
+ # A user message without content has no OpenAI format
317
+ messages = [ChatMessage.from_user("hi"), ChatMessage(_role=ChatRole.USER, _content=[])]
318
+
319
+ with patch("haystack_integrations.tracing.langfuse.tracer.proxy_tracer.is_content_tracing_enabled", True):
320
+ span.set_content_tag("haystack.agent.step.llm.input", {"messages": messages})
321
+
322
+ mock_context_manager._span.update.assert_called_once_with(input=tracing_utils.coerce_tag_value(messages))
323
+
201
324
 
202
325
  class TestSpanContext:
203
326
  def test_post_init(self):
@@ -344,6 +467,48 @@ class TestDefaultSpanHandler:
344
467
  ),
345
468
  }
346
469
 
470
+ def test_handle_agent_step_llm(self):
471
+ mock_span = Mock()
472
+ mock_span.raw_span.return_value = mock_span
473
+ mock_span.get_data.return_value = {
474
+ "haystack.agent.step.llm.output": {
475
+ "replies": [
476
+ ChatMessage.from_assistant(
477
+ "This the LLM's response",
478
+ meta={"model": "test_model", "usage": {"prompt_tokens": 10, "completion_tokens": 5}},
479
+ )
480
+ ]
481
+ },
482
+ }
483
+
484
+ handler = DefaultSpanHandler()
485
+ handler.handle(mock_span, component_type=None)
486
+
487
+ mock_span.update.assert_called_once_with(
488
+ usage_details={"input_tokens": 10, "output_tokens": 5}, model="test_model", completion_start_time=None
489
+ )
490
+
491
+ def test_handle_agent_step_llm_model_parameters(self):
492
+ mock_span = Mock()
493
+ mock_span.raw_span.return_value = mock_span
494
+ mock_span.get_data.return_value = {
495
+ "haystack.agent.step.llm.input": {
496
+ "messages": [ChatMessage.from_user("message")],
497
+ "generation_kwargs": {"temperature": 0.2, "response_format": {"type": "json_object"}},
498
+ },
499
+ "haystack.agent.step.llm.output": {"replies": [ChatMessage.from_assistant("reply")]},
500
+ }
501
+
502
+ handler = DefaultSpanHandler()
503
+ handler.handle(mock_span, component_type=None)
504
+
505
+ mock_span.update.assert_called_once_with(
506
+ usage_details=None,
507
+ model=None,
508
+ completion_start_time=None,
509
+ model_parameters={"temperature": 0.2, "response_format": '{"type": "json_object"}'},
510
+ )
511
+
347
512
  def test_handle_bad_completion_start_time(self, caplog):
348
513
  mock_span = Mock()
349
514
  mock_span.raw_span.return_value = mock_span
@@ -463,6 +628,33 @@ class TestDefaultSpanHandler:
463
628
  name="SentenceTransformersDocumentEmbedder", as_type="embedding"
464
629
  )
465
630
 
631
+ @pytest.mark.parametrize(
632
+ "operation_name,tags,expected_name,expected_type",
633
+ [
634
+ ("haystack.agent.step", {"haystack.agent.step": 1}, "agent step 1", "chain"),
635
+ ("haystack.agent.step.llm", {}, "llm", "generation"),
636
+ ("haystack.agent.step.tool", {"haystack.tool.name": "weather_tool"}, "tool - weather_tool", "tool"),
637
+ ],
638
+ )
639
+ def test_create_span_agent_operations(self, operation_name, tags, expected_name, expected_type):
640
+ mock_client = Mock()
641
+ mock_client.start_as_current_observation = Mock(return_value=MockContextManager())
642
+
643
+ handler = DefaultSpanHandler()
644
+ handler.init_tracer(mock_client)
645
+
646
+ context = SpanContext(
647
+ name=operation_name,
648
+ operation_name=operation_name,
649
+ component_type=None,
650
+ tags=tags,
651
+ parent_span=LangfuseSpan(mock_client.start_as_current_observation()),
652
+ )
653
+ mock_client.start_as_current_observation.reset_mock()
654
+
655
+ handler.create_span(context)
656
+ mock_client.start_as_current_observation.assert_called_once_with(name=expected_name, as_type=expected_type)
657
+
466
658
  def test_create_span_non_component(self):
467
659
  """Test that non-matching components create default span type."""
468
660
  mock_client = Mock()
@@ -698,129 +890,6 @@ class TestLangfuseTracer:
698
890
  assert span.raw_span()._data["model"] == "test_model"
699
891
  assert span.raw_span()._data["completion_start_time"] == datetime.datetime(2021, 7, 27, 16, 2, 8, 12345) # noqa: DTZ001
700
892
 
701
- def test_handle_tool_invoker(self):
702
- """
703
- Test that the ToolInvoker span name is updated correctly with the tool names invoked for better UI/UX
704
- """
705
- mock_span = Mock()
706
- mock_span.raw_span.return_value = mock_span
707
-
708
- # Simulate data for the ToolInvoker component
709
- span_data = {
710
- "haystack.component.name": "tool_invoker",
711
- "haystack.component.type": "ToolInvoker",
712
- "haystack.component.input": {
713
- "messages": [
714
- # Create a chat message with tool calls
715
- ChatMessage.from_assistant(
716
- text="Calling tools",
717
- tool_calls=[
718
- ToolCall(tool_name="search_tool", arguments={"query": "test"}),
719
- ToolCall(tool_name="search_tool", arguments={"query": "another test"}),
720
- ToolCall(tool_name="weather_tool", arguments={"location": "Berlin"}),
721
- ],
722
- )
723
- ]
724
- },
725
- }
726
-
727
- mock_span.get_data.return_value = span_data
728
-
729
- handler = DefaultSpanHandler()
730
- handler.handle(mock_span, component_type="ToolInvoker")
731
-
732
- assert mock_span.update.call_count >= 1
733
- name_update_call = None
734
- for call in mock_span.update.call_args_list:
735
- if "name" in call[1]:
736
- name_update_call = call
737
- break
738
-
739
- assert name_update_call is not None, "No call to update the span name was made"
740
- updated_name = name_update_call[1]["name"]
741
-
742
- # verify the format of the updated span name to be: `original_component_name - [list_of_tool_names]`
743
- assert updated_name != "tool_invoker", "Expected 'tool_invoker` to be upddated with tool names"
744
- assert " - " in updated_name, f"Expected ' - ' in {updated_name}"
745
- assert "[" in updated_name, f"Expected '[' in {updated_name}"
746
- assert "]" in updated_name, f"Expected ']' in {updated_name}"
747
- assert "tool_invoker" in updated_name, f"Expected 'tool_invoker' in {updated_name}"
748
- assert "search_tool (x2)" in updated_name, f"Expected 'search_tool (x2)' in {updated_name}"
749
- assert "weather_tool" in updated_name, f"Expected 'weather_tool' in {updated_name}"
750
-
751
- def test_handle_tool_invoker_input_output_with_content_tracing(self):
752
- """
753
- Test that ToolInvoker spans replace the noisy full-message input with just tool call
754
- arguments, and populate output with tool results, when content tracing is enabled.
755
- """
756
- mock_span = Mock()
757
- mock_span.raw_span.return_value = mock_span
758
-
759
- tool_call = ToolCall(id="call_123", tool_name="search_tool", arguments={"query": "RAG pipelines"})
760
-
761
- span_data = {
762
- "haystack.component.name": "tool_invoker",
763
- "haystack.component.type": "ToolInvoker",
764
- "haystack.component.input": {
765
- "messages": [
766
- ChatMessage.from_user("what is RAG?"),
767
- ChatMessage.from_assistant(text="Calling search", tool_calls=[tool_call]),
768
- ]
769
- },
770
- "haystack.component.output": {
771
- "tool_messages": [ChatMessage.from_tool("RAG stands for Retrieval-Augmented Generation", tool_call)]
772
- },
773
- }
774
- mock_span.get_data.return_value = span_data
775
-
776
- handler = DefaultSpanHandler()
777
- with patch("haystack_integrations.tracing.langfuse.tracer.proxy_tracer.is_content_tracing_enabled", True):
778
- handler.handle(mock_span, component_type="ToolInvoker")
779
-
780
- update_calls = {k: v for call in mock_span.update.call_args_list for k, v in call[1].items()}
781
-
782
- # input should be just the tool call arguments, not the full message list
783
- assert update_calls["input"] == [{"tool_name": "search_tool", "arguments": {"query": "RAG pipelines"}}]
784
- # output carries id/name/arguments so Langfuse detects the tool call and populates its filter
785
- assert update_calls["output"] == [
786
- {
787
- "id": "call_123",
788
- "name": "search_tool",
789
- "arguments": {"query": "RAG pipelines"},
790
- "result": "RAG stands for Retrieval-Augmented Generation",
791
- "error": False,
792
- }
793
- ]
794
-
795
- def test_handle_tool_invoker_no_content_tracing(self):
796
- """
797
- Test that ToolInvoker input/output is NOT updated when content tracing is disabled.
798
- The span name update (tool names) should still happen.
799
- """
800
- mock_span = Mock()
801
- mock_span.raw_span.return_value = mock_span
802
-
803
- tool_call = ToolCall(tool_name="weather_tool", arguments={"location": "Tokyo"})
804
-
805
- span_data = {
806
- "haystack.component.name": "tool_invoker",
807
- "haystack.component.type": "ToolInvoker",
808
- "haystack.component.input": {"messages": [ChatMessage.from_assistant(text="", tool_calls=[tool_call])]},
809
- "haystack.component.output": {"tool_messages": [ChatMessage.from_tool("Sunny, 28°C", tool_call)]},
810
- }
811
- mock_span.get_data.return_value = span_data
812
-
813
- handler = DefaultSpanHandler()
814
- with patch("haystack_integrations.tracing.langfuse.tracer.proxy_tracer.is_content_tracing_enabled", False):
815
- handler.handle(mock_span, component_type="ToolInvoker")
816
-
817
- update_kwargs_keys = {k for call in mock_span.update.call_args_list for k in call[1]}
818
- # name should still be updated
819
- assert "name" in update_kwargs_keys
820
- # input and output must NOT be set when content tracing is off
821
- assert "input" not in update_kwargs_keys
822
- assert "output" not in update_kwargs_keys
823
-
824
893
  def test_trace_generation_invalid_start_time(self):
825
894
  with patch("haystack_integrations.tracing.langfuse.tracer.langfuse.get_client"):
826
895
  tracer = LangfuseTracer(tracer=MockLangfuseClient(), name="Haystack", public=False)