gooddata-eval 1.73.1.dev1__tar.gz → 1.73.1.dev3__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (91) hide show
  1. {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/PKG-INFO +2 -2
  2. {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/pyproject.toml +2 -2
  3. {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/agentic/alert_skill.py +27 -1
  4. {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/agentic/conversation.py +41 -4
  5. {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/agentic/guardrail.py +9 -1
  6. {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/agentic/metric_skill.py +81 -9
  7. {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/agentic/visualization.py +37 -3
  8. {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/chat/sse_client.py +14 -2
  9. {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/evaluators/general_question.py +8 -2
  10. {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/evaluators/guardrail.py +11 -2
  11. {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/evaluators/search_tool.py +4 -1
  12. {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/evaluators/visualization.py +13 -2
  13. gooddata_eval-1.73.1.dev3/src/gooddata_eval/core/models.py +270 -0
  14. {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/reporting/json_report.py +3 -0
  15. {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/runner.py +19 -1
  16. {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/tests/test_agentic_alert_skill.py +2 -0
  17. {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/tests/test_agentic_conversation.py +72 -0
  18. {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/tests/test_agentic_guardrail.py +2 -0
  19. {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/tests/test_agentic_metric_skill.py +130 -8
  20. {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/tests/test_agentic_visualization.py +2 -0
  21. {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/tests/test_runner.py +53 -0
  22. {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/tests/test_sse_client.py +192 -4
  23. gooddata_eval-1.73.1.dev1/src/gooddata_eval/core/models.py +0 -154
  24. {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/.gitignore +0 -0
  25. {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/LICENSE.txt +0 -0
  26. {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/Makefile +0 -0
  27. {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/README.md +0 -0
  28. {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/__init__.py +0 -0
  29. {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/_version.py +0 -0
  30. {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/cli/__init__.py +0 -0
  31. {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/cli/agentic_runner.py +0 -0
  32. {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/cli/main.py +0 -0
  33. {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/__init__.py +0 -0
  34. {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/agentic/__init__.py +0 -0
  35. {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/agentic/_catalog.py +0 -0
  36. {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/agentic/_langfuse.py +0 -0
  37. {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/agentic/general_question.py +0 -0
  38. {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/agentic/kda_skill.py +0 -0
  39. {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/agentic/search_tool.py +0 -0
  40. {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/chat/__init__.py +0 -0
  41. {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/config.py +0 -0
  42. {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/connection.py +0 -0
  43. {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/dataset/__init__.py +0 -0
  44. {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/dataset/langfuse_source.py +0 -0
  45. {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/dataset/local.py +0 -0
  46. {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/evaluators/__init__.py +0 -0
  47. {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/evaluators/_deep_subset.py +0 -0
  48. {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/evaluators/_llm_judge.py +0 -0
  49. {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/evaluators/_text_utils.py +0 -0
  50. {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/evaluators/alert_skill.py +0 -0
  51. {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/evaluators/base.py +0 -0
  52. {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/evaluators/metric_skill.py +0 -0
  53. {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/evaluators/summary.py +0 -0
  54. {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/langfuse/__init__.py +0 -0
  55. {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/langfuse/sink.py +0 -0
  56. {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/reporting/__init__.py +0 -0
  57. {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/reporting/console.py +0 -0
  58. {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/scoring.py +0 -0
  59. {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/summary/__init__.py +0 -0
  60. {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/summary/http_client.py +0 -0
  61. {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/workspace.py +0 -0
  62. {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/tests/__init__.py +0 -0
  63. {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/tests/conftest.py +0 -0
  64. {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/tests/fixtures/sample_dataset/metric_skill_create.json +0 -0
  65. {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/tests/fixtures/sample_dataset/visualization_revenue.json +0 -0
  66. {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/tests/fixtures/sse_visualization_stream.txt +0 -0
  67. {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/tests/test_agentic_general_question.py +0 -0
  68. {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/tests/test_agentic_kda_skill.py +0 -0
  69. {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/tests/test_agentic_langfuse_trace.py +0 -0
  70. {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/tests/test_agentic_run_context.py +0 -0
  71. {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/tests/test_agentic_runner.py +0 -0
  72. {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/tests/test_agentic_search_tool.py +0 -0
  73. {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/tests/test_alert_skill_evaluator.py +0 -0
  74. {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/tests/test_cli.py +0 -0
  75. {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/tests/test_connection.py +0 -0
  76. {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/tests/test_deep_subset.py +0 -0
  77. {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/tests/test_langfuse_sink.py +0 -0
  78. {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/tests/test_langfuse_source.py +0 -0
  79. {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/tests/test_llm_judge.py +0 -0
  80. {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/tests/test_local_loader.py +0 -0
  81. {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/tests/test_metric_skill_evaluator.py +0 -0
  82. {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/tests/test_models.py +0 -0
  83. {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/tests/test_reporting.py +0 -0
  84. {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/tests/test_scoring.py +0 -0
  85. {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/tests/test_search_tool_evaluator.py +0 -0
  86. {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/tests/test_summary_client.py +0 -0
  87. {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/tests/test_summary_evaluator.py +0 -0
  88. {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/tests/test_text_evaluators.py +0 -0
  89. {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/tests/test_visualization_evaluator.py +0 -0
  90. {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/tests/test_workspace.py +0 -0
  91. {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/tox.ini +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: gooddata-eval
3
- Version: 1.73.1.dev1
3
+ Version: 1.73.1.dev3
4
4
  Summary: Evaluate the GoodData AI agent against your own questions and models.
5
5
  Project-URL: Source, https://github.com/gooddata/gooddata-python-sdk
6
6
  Author-email: GoodData <support@gooddata.com>
@@ -17,7 +17,7 @@ Classifier: Topic :: Scientific/Engineering
17
17
  Classifier: Topic :: Software Development
18
18
  Classifier: Typing :: Typed
19
19
  Requires-Python: >=3.10
20
- Requires-Dist: gooddata-sdk~=1.73.1.dev1
20
+ Requires-Dist: gooddata-sdk~=1.73.1.dev3
21
21
  Requires-Dist: httpx<1.0,>=0.27
22
22
  Requires-Dist: orjson<4.0.0,>=3.9.15
23
23
  Requires-Dist: pydantic<3.0,>=2.6
@@ -1,7 +1,7 @@
1
1
  # (C) 2026 GoodData Corporation
2
2
  [project]
3
3
  name = "gooddata-eval"
4
- version = "1.73.1.dev1"
4
+ version = "1.73.1.dev3"
5
5
  description = "Evaluate the GoodData AI agent against your own questions and models."
6
6
  readme = "README.md"
7
7
  license = "MIT"
@@ -11,7 +11,7 @@ authors = [
11
11
  keywords = ["gooddata", "ai", "evaluation", "llm", "analytics", "cli"]
12
12
  requires-python = ">=3.10"
13
13
  dependencies = [
14
- "gooddata-sdk~=1.73.1.dev1",
14
+ "gooddata-sdk~=1.73.1.dev3",
15
15
  "httpx>=0.27,<1.0",
16
16
  "orjson>=3.9.15,<4.0.0",
17
17
  "pydantic>=2.6,<3.0",
@@ -14,7 +14,7 @@ from gooddata_sdk import GoodDataSdk
14
14
  from gooddata_eval.core.agentic._catalog import CatalogMetricAlert
15
15
  from gooddata_eval.core.chat.sse_client import ChatClient
16
16
  from gooddata_eval.core.config import ReasoningEffort
17
- from gooddata_eval.core.models import AgenticEvalOutcome, ToolCallEvent
17
+ from gooddata_eval.core.models import AgenticEvalOutcome, ReasoningStepEvent, ToolCallEvent, build_latency_breakdown
18
18
 
19
19
  try:
20
20
  from openai import OpenAI as _OpenAI
@@ -345,6 +345,8 @@ class AlertRunResult:
345
345
  actual_alert_arguments: dict
346
346
  reasoning_steps: list[str] = field(default_factory=list)
347
347
  response_id: str | None = None
348
+ tool_call_events: list[ToolCallEvent] = field(default_factory=list)
349
+ reasoning_step_events: list[ReasoningStepEvent] = field(default_factory=list)
348
350
 
349
351
 
350
352
  @dataclass
@@ -495,6 +497,11 @@ def run_agentic_alert_skill(
495
497
  tool_called = False
496
498
  reasoning_steps: list[str] = []
497
499
  response_id: str | None = None
500
+ all_tool_call_events: list[ToolCallEvent] = []
501
+ all_reasoning_step_events: list[ReasoningStepEvent] = []
502
+ turn_offset = 0.0 # each turn's call_ts/ts restarts near 0 -- shift by prior turns' wall time
503
+ tool_index_offset = 0
504
+ reasoning_index_offset = 0
498
505
  # conversation_history stores prior turns for GPT-4o context.
499
506
  # Roles follow GPT-4o's perspective: "assistant"=agent text, "user"=sim-user reply.
500
507
  conversation_history: list = []
@@ -504,6 +511,21 @@ def run_agentic_alert_skill(
504
511
  chat_result = client.send_message(conv_id, current_question)
505
512
  reasoning_steps.extend(chat_result.reasoning_steps or [])
506
513
  response_id = chat_result.response_id or response_id
514
+ for tc in chat_result.tool_call_events or []:
515
+ if tc.call_ts is not None:
516
+ tc.call_ts += turn_offset
517
+ if tc.result_ts is not None:
518
+ tc.result_ts += turn_offset
519
+ if tc.index is not None:
520
+ tc.index += tool_index_offset
521
+ for rs in chat_result.reasoning_step_events or []:
522
+ rs.ts += turn_offset
523
+ rs.index += reasoning_index_offset
524
+ all_tool_call_events.extend(chat_result.tool_call_events or [])
525
+ all_reasoning_step_events.extend(chat_result.reasoning_step_events or [])
526
+ tool_index_offset += len(chat_result.tool_call_events or [])
527
+ reasoning_index_offset += len(chat_result.reasoning_step_events or [])
528
+ turn_offset += chat_result.turn_wall_clock_sec or 0.0
507
529
  alert_id, actual_args, tool_called = _extract_alert_call(chat_result.tool_call_events or [])
508
530
  if tool_called:
509
531
  alert_id_to_delete = alert_id
@@ -541,6 +563,8 @@ def run_agentic_alert_skill(
541
563
  actual_alert_arguments=actual_args,
542
564
  reasoning_steps=reasoning_steps,
543
565
  response_id=response_id,
566
+ tool_call_events=all_tool_call_events,
567
+ reasoning_step_events=all_reasoning_step_events,
544
568
  )
545
569
  finally:
546
570
  if alert_id_to_delete:
@@ -717,6 +741,7 @@ def evaluate_agentic_alert_skill(
717
741
  "metric_correct": ev.metric_correct,
718
742
  "recipients_correct": ev.recipients_correct,
719
743
  "actual_alert_arguments": best.actual_alert_arguments,
744
+ "latency_breakdown": build_latency_breakdown(best.tool_call_events, best.reasoning_step_events),
720
745
  }
721
746
  raise exc
722
747
  best = summary.best
@@ -734,5 +759,6 @@ def evaluate_agentic_alert_skill(
734
759
  "metric_correct": ev.metric_correct,
735
760
  "recipients_correct": ev.recipients_correct,
736
761
  "actual_alert_arguments": best.actual_alert_arguments,
762
+ "latency_breakdown": build_latency_breakdown(best.tool_call_events, best.reasoning_step_events),
737
763
  },
738
764
  )
@@ -15,7 +15,13 @@ from gooddata_eval.core.agentic.alert_skill import render_alert_proposal
15
15
  from gooddata_eval.core.agentic.metric_skill import _delete_metric, _extract_created_metric_ids, _extract_metric_result
16
16
  from gooddata_eval.core.chat.sse_client import ChatClient
17
17
  from gooddata_eval.core.config import ReasoningEffort
18
- from gooddata_eval.core.models import AgenticEvalOutcome, ChatResult, ToolCallEvent
18
+ from gooddata_eval.core.models import (
19
+ AgenticEvalOutcome,
20
+ ChatResult,
21
+ ReasoningStepEvent,
22
+ ToolCallEvent,
23
+ build_latency_breakdown,
24
+ )
19
25
  from gooddata_eval.core.scoring import (
20
26
  check_filters,
21
27
  check_viz_type,
@@ -199,9 +205,12 @@ def _get_sim_user_response(agent_message: str, turn: TurnDefinition, expected_ou
199
205
  generate_simulated_response,
200
206
  )
201
207
 
202
- return generate_simulated_response(agent_message, expected_output)
203
- except Exception:
204
- pass
208
+ # A conversation turn only ever carries one expected_output (no multi-candidate
209
+ # list like agent_metric_skill's fixtures) -- wrap it as a single-item list to
210
+ # match generate_simulated_response's signature.
211
+ return generate_simulated_response(agent_message, [expected_output], turn.message)
212
+ except Exception as exc:
213
+ print(f"[SIM-USER] metric branch failed for turn {turn.turn_id}: {exc}")
205
214
 
206
215
  # Generic fallback for other skill types or when expected_output is absent
207
216
  import os # noqa: PLC0415
@@ -254,6 +263,8 @@ class ConversationResult:
254
263
  total_clarification_turns: int
255
264
  reasoning_steps: list[str] = field(default_factory=list)
256
265
  response_id: str | None = None
266
+ tool_call_events: list[ToolCallEvent] = field(default_factory=list)
267
+ reasoning_step_events: list[ReasoningStepEvent] = field(default_factory=list)
257
268
 
258
269
 
259
270
  def run_agentic_conversation(
@@ -287,6 +298,14 @@ def run_agentic_conversation(
287
298
  created_metric_ids: list[str] = []
288
299
  reasoning_steps: list[str] = []
289
300
  response_id: str | None = None
301
+ conversation_tool_call_events: list[ToolCallEvent] = []
302
+ conversation_reasoning_step_events: list[ReasoningStepEvent] = []
303
+ # Every send_message() call (across every logical turn AND every clarification
304
+ # sub-turn within it) restarts call_ts/ts near 0 -- these run across the whole
305
+ # conversation, not reset per logical turn, so every one of those calls shifts them.
306
+ turn_offset = 0.0
307
+ tool_index_offset = 0
308
+ reasoning_index_offset = 0
290
309
 
291
310
  try:
292
311
  if initial_conversation_id is not None:
@@ -322,7 +341,22 @@ def run_agentic_conversation(
322
341
  for _iter in range(max_clarification_turns + 1):
323
342
  chat_result = client.send_message(conversation_id, current_message)
324
343
  final_result = chat_result
344
+ for tc in chat_result.tool_call_events or []:
345
+ if tc.call_ts is not None:
346
+ tc.call_ts += turn_offset
347
+ if tc.result_ts is not None:
348
+ tc.result_ts += turn_offset
349
+ if tc.index is not None:
350
+ tc.index += tool_index_offset
351
+ for rs in chat_result.reasoning_step_events or []:
352
+ rs.ts += turn_offset
353
+ rs.index += reasoning_index_offset
325
354
  all_tool_calls.extend(chat_result.tool_call_events or [])
355
+ conversation_tool_call_events.extend(chat_result.tool_call_events or [])
356
+ conversation_reasoning_step_events.extend(chat_result.reasoning_step_events or [])
357
+ tool_index_offset += len(chat_result.tool_call_events or [])
358
+ reasoning_index_offset += len(chat_result.reasoning_step_events or [])
359
+ turn_offset += chat_result.turn_wall_clock_sec or 0.0
326
360
  reasoning_steps.extend(chat_result.reasoning_steps or [])
327
361
  response_id = chat_result.response_id or response_id
328
362
 
@@ -390,6 +424,8 @@ def run_agentic_conversation(
390
424
  total_clarification_turns=total_clarification_turns,
391
425
  reasoning_steps=reasoning_steps,
392
426
  response_id=response_id,
427
+ tool_call_events=conversation_tool_call_events,
428
+ reasoning_step_events=conversation_reasoning_step_events,
393
429
  )
394
430
 
395
431
 
@@ -408,6 +444,7 @@ def _conversation_detail(result: ConversationResult) -> dict:
408
444
  }
409
445
  for tr in result.turn_results
410
446
  ],
447
+ "latency_breakdown": build_latency_breakdown(result.tool_call_events, result.reasoning_step_events),
411
448
  }
412
449
 
413
450
 
@@ -8,7 +8,7 @@ from dataclasses import dataclass, field
8
8
  from gooddata_eval.core.chat.sse_client import ChatClient
9
9
  from gooddata_eval.core.config import ReasoningEffort
10
10
  from gooddata_eval.core.evaluators._llm_judge import LLMJudge
11
- from gooddata_eval.core.models import AgenticEvalOutcome
11
+ from gooddata_eval.core.models import AgenticEvalOutcome, ReasoningStepEvent, ToolCallEvent, build_latency_breakdown
12
12
 
13
13
  _DEFAULT_K = 1
14
14
 
@@ -52,6 +52,8 @@ class GuardrailResult:
52
52
  reasoning: str
53
53
  reasoning_steps: list[str] = field(default_factory=list)
54
54
  response_id: str | None = None
55
+ tool_call_events: list[ToolCallEvent] = field(default_factory=list)
56
+ reasoning_step_events: list[ReasoningStepEvent] = field(default_factory=list)
55
57
 
56
58
 
57
59
  @dataclass
@@ -100,6 +102,8 @@ def run_agentic_guardrail(
100
102
  reasoning=reasoning,
101
103
  reasoning_steps=list(chat_result.reasoning_steps or []),
102
104
  response_id=chat_result.response_id,
105
+ tool_call_events=list(chat_result.tool_call_events or []),
106
+ reasoning_step_events=list(chat_result.reasoning_step_events or []),
103
107
  )
104
108
  )
105
109
  finally:
@@ -124,6 +128,8 @@ def run_agentic_guardrail(
124
128
  reasoning=reasoning,
125
129
  reasoning_steps=list(chat_result.reasoning_steps or []),
126
130
  response_id=chat_result.response_id,
131
+ tool_call_events=list(chat_result.tool_call_events or []),
132
+ reasoning_step_events=list(chat_result.reasoning_step_events or []),
127
133
  )
128
134
  )
129
135
  finally:
@@ -245,6 +251,7 @@ def evaluate_agentic_guardrail(
245
251
  "judge_passed": best.passed,
246
252
  "judge_reasoning": best.reasoning,
247
253
  "actual_output": best.actual_output,
254
+ "latency_breakdown": build_latency_breakdown(best.tool_call_events, best.reasoning_step_events),
248
255
  }
249
256
  raise exc
250
257
  best = summary.best
@@ -256,5 +263,6 @@ def evaluate_agentic_guardrail(
256
263
  "judge_passed": best.passed,
257
264
  "judge_reasoning": best.reasoning,
258
265
  "actual_output": best.actual_output,
266
+ "latency_breakdown": build_latency_breakdown(best.tool_call_events, best.reasoning_step_events),
259
267
  },
260
268
  )
@@ -12,7 +12,7 @@ from gooddata_sdk import GoodDataSdk
12
12
 
13
13
  from gooddata_eval.core.chat.sse_client import ChatClient
14
14
  from gooddata_eval.core.config import ReasoningEffort
15
- from gooddata_eval.core.models import AgenticEvalOutcome, ToolCallEvent
15
+ from gooddata_eval.core.models import AgenticEvalOutcome, ReasoningStepEvent, ToolCallEvent, build_latency_breakdown
16
16
 
17
17
  try:
18
18
  from openai import OpenAI as _OpenAI
@@ -30,6 +30,11 @@ _INNER_SELECT_RE = re.compile(r"\(\s*SELECT\s*\{([^}]+)\}\s*\)", re.IGNORECASE)
30
30
  # Everything else in MAQL (keywords, operators, numbers, punctuation) carries no
31
31
  # case-sensitive meaning, per the MAQL reference (SELECT/BY/WHERE/FOR PREVIOUS/etc.
32
32
  # are case-insensitive; only {..} identifiers and quoted literal values are not).
33
+ # Feeds _normalize_maql, the scoring comparator (_best_maql_match) -- do not widen this
34
+ # to handle \X escapes without confirming MAQL literals actually support backslash
35
+ # escaping (unconfirmed; see PR #1760 review). A wrong guess here silently changes
36
+ # maql_correct for the whole eval dataset, not just a hint. _no_where_clause_hint()
37
+ # below has its own, separately-scoped regex for that reason.
33
38
  _PROTECTED_RE = re.compile(r"\{[^}]*\}|\"[^\"]*\"|'[^']*'")
34
39
 
35
40
 
@@ -99,9 +104,43 @@ class SimulatedResponseError(RuntimeError):
99
104
  """
100
105
 
101
106
 
102
- def generate_simulated_response(agent_message: str, expected_output: dict) -> str:
107
+ # Separate from _PROTECTED_RE on purpose: this one only feeds a same-turn LLM-prompt hint
108
+ # (see _no_where_clause_hint), never the scoring comparator, so it can afford to consume
109
+ # \X escape sequences inside quoted literals without risking maql_correct semantics.
110
+ _HINT_PROTECTED_RE = re.compile(r"\{[^}]*\}|\"(?:[^\"\\]|\\.)*\"|'(?:[^'\\]|\\.)*'")
111
+
112
+
113
+ def _no_where_clause_hint(expected_maqls: list[str]) -> str:
114
+ """Deterministic nudge for when NONE of the accepted candidate MAQLs has a WHERE clause.
115
+
116
+ Without this, whether to add a filter is left entirely to the simulating LLM's judgment
117
+ of what the original request "implies" -- the same fuzzy reasoning that caused it to
118
+ inject an unrequested filter in the first place (QA-29094). Checks every candidate, not
119
+ just the first: _best_maql_match accepts any of them, so hinting off just candidate 0
120
+ would risk steering the agent away from a filtered candidate the scorer would still have
121
+ accepted (the mirror-image of the original bug). Strips {type/id} identifiers and quoted
122
+ literals first so a "where" substring inside one of those -- e.g.
123
+ `{metric/somewhere_sales}`, or a literal value containing the word -- doesn't get
124
+ mistaken for a real WHERE clause.
125
+ """
126
+ for maql in expected_maqls:
127
+ outside_protected = _HINT_PROTECTED_RE.sub(" ", maql)
128
+ if re.search(r"\bWHERE\b", outside_protected, re.IGNORECASE):
129
+ return ""
130
+ return (
131
+ " This metric needs no filter. If the assistant asks about excluding or filtering "
132
+ "anything, say no filter is needed."
133
+ )
134
+
135
+
136
+ def generate_simulated_response(agent_message: str, expected_outputs: list[dict], original_question: str) -> str:
103
137
  """Generate a user reply to keep the metric-skill conversation going (gpt-4o-mini).
104
138
 
139
+ ``expected_outputs`` is the fixture's full candidate list (as accepted by
140
+ ``_best_maql_match``), not just the first one -- the ground-truth MAQL woven into the
141
+ prompt still comes from candidate 0, but the no-filter hint checks all of them (see
142
+ ``_no_where_clause_hint``).
143
+
105
144
  Raises:
106
145
  SimulatedResponseError: openai is not installed, OPENAI_API_KEY is unset, or the
107
146
  provider call failed.
@@ -116,15 +155,23 @@ def generate_simulated_response(agent_message: str, expected_output: dict) -> st
116
155
  raise SimulatedResponseError("OPENAI_API_KEY environment variable is not set")
117
156
 
118
157
  client = OpenAI(api_key=api_key)
119
- expected_maql = expected_output.get("maql", "")
158
+ expected_maql = expected_outputs[0].get("maql", "") if expected_outputs else ""
159
+ expected_maqls = [eo.get("maql", "") for eo in expected_outputs]
120
160
  prompt = (
121
161
  f"You are simulating a user in a conversation with a BI assistant that creates metrics. "
162
+ f"The user's original request was: '{original_question}'. "
122
163
  f"The assistant said: '{agent_message}'. "
123
164
  f"The user's ground-truth intended metric is exactly this MAQL: {expected_maql}. "
124
- f"Reply as the user. You MUST ensure every clause of that MAQL (including any WHERE/filter "
125
- f"conditions) is eventually satisfied, and quote field/label identifiers verbatim from it -- "
126
- f"never paraphrase or drop a clause, even if the assistant's question doesn't explicitly ask "
127
- f"about it. If the assistant's offered options omit a required filter, add it yourself."
165
+ f"Reply as the user. If the assistant is asking a clarifying question rather than proposing "
166
+ f"a metric, answer that question directly using the ground-truth MAQL -- quote field/label "
167
+ f"identifiers verbatim -- instead of merely agreeing. "
168
+ f"If the assistant's proposal already satisfies the ORIGINAL REQUEST above, agree and confirm "
169
+ f"-- do not introduce new requirements the original request never mentioned. "
170
+ f"Only if the assistant's proposal is missing something the original request actually implies "
171
+ f"(e.g. a filter/clause from the ground-truth MAQL that is a reasonable reading of the original "
172
+ f"request), point it out and add it yourself, quoting field/label identifiers verbatim from the "
173
+ f"ground-truth MAQL."
174
+ f"{_no_where_clause_hint(expected_maqls)}"
128
175
  )
129
176
  try:
130
177
  response = client.chat.completions.create(
@@ -150,6 +197,8 @@ class MetricRunResult:
150
197
  total_turns: float
151
198
  reasoning_steps: list[str] = field(default_factory=list)
152
199
  response_id: str | None = None
200
+ tool_call_events: list[ToolCallEvent] = field(default_factory=list)
201
+ reasoning_step_events: list[ReasoningStepEvent] = field(default_factory=list)
153
202
 
154
203
 
155
204
  @dataclass
@@ -232,13 +281,17 @@ def _execute_single_metric_run(
232
281
  ``_delete_metric``) so it cannot leak into — and be reused by — a later test
233
282
  sharing the workspace.
234
283
  """
235
- primary_expected = expected_outputs[0] if expected_outputs else {}
236
284
  metric_result: dict | None = None
237
285
  created_metric_ids: list[str] = []
238
286
  turns = 0
239
287
  current_question = question
240
288
  reasoning_steps: list[str] = []
241
289
  response_id: str | None = None
290
+ all_tool_call_events: list[ToolCallEvent] = []
291
+ all_reasoning_step_events: list[ReasoningStepEvent] = []
292
+ turn_offset = 0.0 # each turn's call_ts/ts restarts near 0 -- shift by prior turns' wall time
293
+ tool_index_offset = 0
294
+ reasoning_index_offset = 0
242
295
 
243
296
  try:
244
297
  for _iteration in range(max_iterations):
@@ -246,6 +299,21 @@ def _execute_single_metric_run(
246
299
  chat_result = client.send_message(conversation_id, current_question)
247
300
  reasoning_steps.extend(chat_result.reasoning_steps or [])
248
301
  response_id = chat_result.response_id or response_id
302
+ for tc in chat_result.tool_call_events or []:
303
+ if tc.call_ts is not None:
304
+ tc.call_ts += turn_offset
305
+ if tc.result_ts is not None:
306
+ tc.result_ts += turn_offset
307
+ if tc.index is not None:
308
+ tc.index += tool_index_offset
309
+ for rs in chat_result.reasoning_step_events or []:
310
+ rs.ts += turn_offset
311
+ rs.index += reasoning_index_offset
312
+ all_tool_call_events.extend(chat_result.tool_call_events or [])
313
+ all_reasoning_step_events.extend(chat_result.reasoning_step_events or [])
314
+ tool_index_offset += len(chat_result.tool_call_events or [])
315
+ reasoning_index_offset += len(chat_result.reasoning_step_events or [])
316
+ turn_offset += chat_result.turn_wall_clock_sec or 0.0
249
317
  for metric_id in _extract_created_metric_ids(chat_result.tool_call_events or []):
250
318
  if metric_id not in created_metric_ids:
251
319
  created_metric_ids.append(metric_id)
@@ -259,7 +327,7 @@ def _execute_single_metric_run(
259
327
  if _iteration >= max_iterations - 1:
260
328
  break
261
329
  try:
262
- current_question = generate_simulated_response(response_text, primary_expected)
330
+ current_question = generate_simulated_response(response_text, expected_outputs, question)
263
331
  except SimulatedResponseError as exc:
264
332
  print(f"[SIM-USER] Simulated reply failed for conversation {conversation_id}: {exc}")
265
333
  break
@@ -276,6 +344,8 @@ def _execute_single_metric_run(
276
344
  total_turns=float(turns),
277
345
  reasoning_steps=reasoning_steps,
278
346
  response_id=response_id,
347
+ tool_call_events=all_tool_call_events,
348
+ reasoning_step_events=all_reasoning_step_events,
279
349
  )
280
350
  finally:
281
351
  for metric_id in created_metric_ids:
@@ -457,6 +527,7 @@ def evaluate_agentic_metric_skill(
457
527
  "maql_correct": best.maql_correct,
458
528
  "expected_maql_candidates": [c.get("maql", "") for c in expected_outputs_list],
459
529
  "actual_maql": best.actual_maql,
530
+ "latency_breakdown": build_latency_breakdown(best.tool_call_events, best.reasoning_step_events),
460
531
  }
461
532
  raise exc
462
533
  best = summary.best
@@ -470,5 +541,6 @@ def evaluate_agentic_metric_skill(
470
541
  "maql_correct": best.maql_correct,
471
542
  "expected_maql_candidates": [c.get("maql", "") for c in expected_outputs_list],
472
543
  "actual_maql": best.actual_maql,
544
+ "latency_breakdown": build_latency_breakdown(best.tool_call_events, best.reasoning_step_events),
473
545
  },
474
546
  )
@@ -18,7 +18,13 @@ from gooddata_eval.core.evaluators.visualization import (
18
18
  _evaluate_against_candidates,
19
19
  evaluation_result_detail,
20
20
  )
21
- from gooddata_eval.core.models import AgenticEvalOutcome, CreatedVisualization, ToolCallEvent
21
+ from gooddata_eval.core.models import (
22
+ AgenticEvalOutcome,
23
+ CreatedVisualization,
24
+ ReasoningStepEvent,
25
+ ToolCallEvent,
26
+ build_latency_breakdown,
27
+ )
22
28
  from gooddata_eval.core.scoring import get_dimension_uri_set, get_metric_uri_set, uri_to_display_name
23
29
 
24
30
  _DEFAULT_K = 2
@@ -37,6 +43,8 @@ class RunResult:
37
43
  total_steps: float
38
44
  reasoning_steps: list[str] = field(default_factory=list)
39
45
  response_id: str | None = None
46
+ tool_call_events: list[ToolCallEvent] = field(default_factory=list)
47
+ reasoning_step_events: list[ReasoningStepEvent] = field(default_factory=list)
40
48
 
41
49
 
42
50
  @dataclass
@@ -161,8 +169,12 @@ def _execute_single_run(
161
169
  total_turns = 0.0
162
170
  total_steps = 0.0
163
171
  all_tool_call_events: list[ToolCallEvent] = []
172
+ all_reasoning_step_events: list[ReasoningStepEvent] = []
164
173
  reasoning_steps: list[str] = []
165
174
  response_id: str | None = None
175
+ turn_offset = 0.0 # each turn's call_ts/ts restarts near 0 -- shift by prior turns' wall time
176
+ reasoning_index_offset = 0 # ditto for ReasoningStepEvent.index, which also restarts per turn
177
+ tool_index_offset = 0 # ditto for ToolCallEvent.index
166
178
  simulated_response_guide = expected_outputs[0] # primary candidate guides the simulated user
167
179
 
168
180
  current_result = client.send_message(conversation_id, question)
@@ -170,9 +182,23 @@ def _execute_single_run(
170
182
  for iteration in range(max_iterations):
171
183
  total_turns += 1.0
172
184
  total_steps += float(current_result.reasoning_step_count)
185
+ for tc in current_result.tool_call_events:
186
+ if tc.call_ts is not None:
187
+ tc.call_ts += turn_offset
188
+ if tc.result_ts is not None:
189
+ tc.result_ts += turn_offset
190
+ if tc.index is not None:
191
+ tc.index += tool_index_offset
192
+ for rs in current_result.reasoning_step_events:
193
+ rs.ts += turn_offset
194
+ rs.index += reasoning_index_offset
173
195
  all_tool_call_events.extend(current_result.tool_call_events)
196
+ all_reasoning_step_events.extend(current_result.reasoning_step_events)
197
+ tool_index_offset += len(current_result.tool_call_events)
198
+ reasoning_index_offset += len(current_result.reasoning_step_events)
174
199
  reasoning_steps.extend(current_result.reasoning_steps or [])
175
200
  response_id = current_result.response_id or response_id
201
+ turn_offset += current_result.turn_wall_clock_sec or 0.0
176
202
 
177
203
  viz_produced = bool(current_result.created_visualizations and current_result.created_visualizations.objects)
178
204
  if viz_produced:
@@ -201,6 +227,8 @@ def _execute_single_run(
201
227
  total_steps=total_steps,
202
228
  reasoning_steps=reasoning_steps,
203
229
  response_id=response_id,
230
+ tool_call_events=all_tool_call_events,
231
+ reasoning_step_events=all_reasoning_step_events,
204
232
  )
205
233
 
206
234
 
@@ -434,12 +462,18 @@ def evaluate_agentic_visualization(
434
462
  exc.reasoning_steps = best.reasoning_steps
435
463
  exc.conversation_id = best.conversation_id
436
464
  exc.response_id = best.response_id
437
- exc.detail = evaluation_result_detail(ev)
465
+ exc.detail = {
466
+ **evaluation_result_detail(ev),
467
+ "latency_breakdown": build_latency_breakdown(best.tool_call_events, best.reasoning_step_events),
468
+ }
438
469
  raise exc
439
470
  best = summary.best
440
471
  return AgenticEvalOutcome(
441
472
  reasoning_steps=best.reasoning_steps,
442
473
  conversation_id=best.conversation_id,
443
474
  response_id=best.response_id,
444
- detail=evaluation_result_detail(best.eval_result),
475
+ detail={
476
+ **evaluation_result_detail(best.eval_result),
477
+ "latency_breakdown": build_latency_breakdown(best.tool_call_events, best.reasoning_step_events),
478
+ },
445
479
  )
@@ -132,6 +132,11 @@ class _SseAccumulator:
132
132
  adhoc_viz_args: list[dict[str, Any]] = field(default_factory=list)
133
133
  response_id: str | None = None
134
134
  stream_ended: bool = False
135
+ # Reference point for call_ts/result_ts below -- client-observed receipt time, not a
136
+ # server timestamp, so only meaningful as an offset within this one turn. Wrapped in a
137
+ # lambda, not passed as `time.monotonic` directly -- a bare function reference binds at
138
+ # class-body execution (import time), before tests can monkeypatch `sse_mod.time.monotonic`.
139
+ t0: float = field(default_factory=lambda: time.monotonic())
135
140
 
136
141
 
137
142
  def _handle_text(content: dict[str, Any], acc: _SseAccumulator) -> None:
@@ -160,17 +165,22 @@ def _handle_multipart(content: dict[str, Any], acc: _SseAccumulator) -> None:
160
165
  def _handle_reasoning(content: dict[str, Any], acc: _SseAccumulator) -> None:
161
166
  summary = content.get("summary", "")
162
167
  if summary:
163
- acc.reasoning_steps.append({"summary": summary})
168
+ acc.reasoning_steps.append(
169
+ {"summary": summary, "ts": round(time.monotonic() - acc.t0, 3), "index": len(acc.reasoning_steps)}
170
+ )
164
171
 
165
172
 
166
173
  def _handle_tool_call(content: dict[str, Any], acc: _SseAccumulator) -> None:
167
174
  call_id = content.get("callId", "")
168
- acc.call_id_to_event_index[call_id] = len(acc.tool_call_events)
175
+ idx = len(acc.tool_call_events)
176
+ acc.call_id_to_event_index[call_id] = idx
169
177
  acc.tool_call_events.append(
170
178
  {
171
179
  "functionName": content.get("name", ""),
172
180
  "functionArguments": json.dumps(content.get("arguments", {})),
173
181
  "result": None,
182
+ "call_ts": round(time.monotonic() - acc.t0, 3),
183
+ "index": idx,
174
184
  }
175
185
  )
176
186
  # Stash visualization definition from create_adhoc_visualization so we can
@@ -186,6 +196,7 @@ def _handle_tool_result(content: dict[str, Any], acc: _SseAccumulator) -> None:
186
196
  idx = acc.call_id_to_event_index.get(call_id)
187
197
  if idx is not None:
188
198
  acc.tool_call_events[idx]["result"] = content.get("result", "")
199
+ acc.tool_call_events[idx]["result_ts"] = round(time.monotonic() - acc.t0, 3)
189
200
 
190
201
 
191
202
  def _build_chat_result(acc: _SseAccumulator) -> ChatResult:
@@ -195,6 +206,7 @@ def _build_chat_result(acc: _SseAccumulator) -> ChatResult:
195
206
  "toolCallEvents": acc.tool_call_events,
196
207
  "reasoningStepCount": len(acc.reasoning_steps),
197
208
  "reasoningSteps": [step["summary"] for step in acc.reasoning_steps],
209
+ "reasoningStepEvents": acc.reasoning_steps,
198
210
  }
199
211
  if acc.visualizations:
200
212
  payload["createdVisualizations"] = {
@@ -4,7 +4,7 @@
4
4
  from gooddata_eval.core.evaluators._llm_judge import LLMJudge
5
5
  from gooddata_eval.core.evaluators._text_utils import extract_text
6
6
  from gooddata_eval.core.evaluators.base import ItemEvaluation
7
- from gooddata_eval.core.models import ChatResult, DatasetItem
7
+ from gooddata_eval.core.models import ChatResult, DatasetItem, build_latency_breakdown
8
8
 
9
9
  _EVALUATION_STEPS = [
10
10
  "Read the INPUT (the user's question) and the EXPECTED OUTPUT (a description of what a correct answer must contain).",
@@ -30,5 +30,11 @@ class GeneralQuestionEvaluator:
30
30
  return ItemEvaluation(
31
31
  passed=passed,
32
32
  rank_key=(int(passed),),
33
- detail={"judge_reasoning": reasoning, "actual_output": actual},
33
+ detail={
34
+ "judge_reasoning": reasoning,
35
+ "actual_output": actual,
36
+ "latency_breakdown": build_latency_breakdown(
37
+ chat_result.tool_call_events, chat_result.reasoning_step_events
38
+ ),
39
+ },
34
40
  )
@@ -4,7 +4,7 @@
4
4
  from gooddata_eval.core.evaluators._llm_judge import LLMJudge
5
5
  from gooddata_eval.core.evaluators._text_utils import extract_text
6
6
  from gooddata_eval.core.evaluators.base import ItemEvaluation
7
- from gooddata_eval.core.models import ChatResult, DatasetItem
7
+ from gooddata_eval.core.models import ChatResult, DatasetItem, build_latency_breakdown
8
8
 
9
9
  _EVALUATION_STEPS = [
10
10
  "Read the INPUT (the user's message) and the EXPECTED OUTPUT (a description of how the agent should refuse or redirect).",
@@ -29,7 +29,13 @@ class GuardrailEvaluator:
29
29
  passed=False,
30
30
  rank_key=(False,),
31
31
  # no_visualization=False → quality_score=0 (correctly bad)
32
- detail={"no_visualization": False, "judge_reasoning": "visualization produced — auto-fail"},
32
+ detail={
33
+ "no_visualization": False,
34
+ "judge_reasoning": "visualization produced — auto-fail",
35
+ "latency_breakdown": build_latency_breakdown(
36
+ chat_result.tool_call_events, chat_result.reasoning_step_events
37
+ ),
38
+ },
33
39
  )
34
40
 
35
41
  actual = extract_text(chat_result)
@@ -48,5 +54,8 @@ class GuardrailEvaluator:
48
54
  "judge_passed": passed,
49
55
  "judge_reasoning": reasoning,
50
56
  "actual_output": actual,
57
+ "latency_breakdown": build_latency_breakdown(
58
+ chat_result.tool_call_events, chat_result.reasoning_step_events
59
+ ),
51
60
  },
52
61
  )
@@ -2,7 +2,7 @@
2
2
  """Evaluator for search_tool: agent must call the catalog search with expected parameters."""
3
3
 
4
4
  from gooddata_eval.core.evaluators.base import ItemEvaluation
5
- from gooddata_eval.core.models import ChatResult, DatasetItem
5
+ from gooddata_eval.core.models import ChatResult, DatasetItem, build_latency_breakdown
6
6
 
7
7
 
8
8
  def _normalize_str_list(value: object, *, lowercase: bool = False) -> list[str]:
@@ -55,5 +55,8 @@ class SearchToolEvaluator:
55
55
  "tool_correctness": tool_correctness,
56
56
  "expected_function": expected_fn,
57
57
  "calls_found": len(matching_events),
58
+ "latency_breakdown": build_latency_breakdown(
59
+ chat_result.tool_call_events, chat_result.reasoning_step_events
60
+ ),
58
61
  },
59
62
  )