gooddata-eval 1.73.0__tar.gz → 1.73.1.dev2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (93) hide show
  1. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/PKG-INFO +2 -2
  2. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/pyproject.toml +2 -2
  3. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/cli/agentic_runner.py +10 -6
  4. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/agentic/alert_skill.py +65 -6
  5. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/agentic/conversation.py +60 -18
  6. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/agentic/general_question.py +40 -4
  7. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/agentic/guardrail.py +49 -4
  8. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/agentic/kda_skill.py +59 -6
  9. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/agentic/metric_skill.py +67 -13
  10. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/agentic/search_tool.py +40 -5
  11. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/agentic/visualization.py +69 -5
  12. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/chat/sse_client.py +28 -3
  13. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/evaluators/general_question.py +8 -2
  14. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/evaluators/guardrail.py +11 -2
  15. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/evaluators/search_tool.py +4 -1
  16. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/evaluators/visualization.py +38 -18
  17. gooddata_eval-1.73.1.dev2/src/gooddata_eval/core/models.py +270 -0
  18. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/reporting/json_report.py +3 -0
  19. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/runner.py +19 -1
  20. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/tests/test_agentic_alert_skill.py +35 -0
  21. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/tests/test_agentic_conversation.py +124 -0
  22. gooddata_eval-1.73.1.dev2/tests/test_agentic_general_question.py +210 -0
  23. gooddata_eval-1.73.1.dev2/tests/test_agentic_guardrail.py +210 -0
  24. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/tests/test_agentic_kda_skill.py +147 -0
  25. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/tests/test_agentic_metric_skill.py +121 -1
  26. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/tests/test_agentic_runner.py +70 -23
  27. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/tests/test_agentic_search_tool.py +96 -0
  28. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/tests/test_agentic_visualization.py +132 -0
  29. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/tests/test_runner.py +53 -0
  30. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/tests/test_sse_client.py +234 -4
  31. gooddata_eval-1.73.0/src/gooddata_eval/core/models.py +0 -146
  32. gooddata_eval-1.73.0/tests/test_agentic_general_question.py +0 -100
  33. gooddata_eval-1.73.0/tests/test_agentic_guardrail.py +0 -98
  34. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/.gitignore +0 -0
  35. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/LICENSE.txt +0 -0
  36. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/Makefile +0 -0
  37. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/README.md +0 -0
  38. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/__init__.py +0 -0
  39. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/_version.py +0 -0
  40. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/cli/__init__.py +0 -0
  41. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/cli/main.py +0 -0
  42. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/__init__.py +0 -0
  43. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/agentic/__init__.py +0 -0
  44. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/agentic/_catalog.py +0 -0
  45. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/agentic/_langfuse.py +0 -0
  46. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/chat/__init__.py +0 -0
  47. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/config.py +0 -0
  48. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/connection.py +0 -0
  49. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/dataset/__init__.py +0 -0
  50. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/dataset/langfuse_source.py +0 -0
  51. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/dataset/local.py +0 -0
  52. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/evaluators/__init__.py +0 -0
  53. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/evaluators/_deep_subset.py +0 -0
  54. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/evaluators/_llm_judge.py +0 -0
  55. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/evaluators/_text_utils.py +0 -0
  56. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/evaluators/alert_skill.py +0 -0
  57. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/evaluators/base.py +0 -0
  58. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/evaluators/metric_skill.py +0 -0
  59. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/evaluators/summary.py +0 -0
  60. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/langfuse/__init__.py +0 -0
  61. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/langfuse/sink.py +0 -0
  62. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/reporting/__init__.py +0 -0
  63. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/reporting/console.py +0 -0
  64. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/scoring.py +0 -0
  65. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/summary/__init__.py +0 -0
  66. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/summary/http_client.py +0 -0
  67. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/workspace.py +0 -0
  68. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/tests/__init__.py +0 -0
  69. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/tests/conftest.py +0 -0
  70. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/tests/fixtures/sample_dataset/metric_skill_create.json +0 -0
  71. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/tests/fixtures/sample_dataset/visualization_revenue.json +0 -0
  72. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/tests/fixtures/sse_visualization_stream.txt +0 -0
  73. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/tests/test_agentic_langfuse_trace.py +0 -0
  74. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/tests/test_agentic_run_context.py +0 -0
  75. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/tests/test_alert_skill_evaluator.py +0 -0
  76. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/tests/test_cli.py +0 -0
  77. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/tests/test_connection.py +0 -0
  78. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/tests/test_deep_subset.py +0 -0
  79. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/tests/test_langfuse_sink.py +0 -0
  80. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/tests/test_langfuse_source.py +0 -0
  81. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/tests/test_llm_judge.py +0 -0
  82. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/tests/test_local_loader.py +0 -0
  83. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/tests/test_metric_skill_evaluator.py +0 -0
  84. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/tests/test_models.py +0 -0
  85. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/tests/test_reporting.py +0 -0
  86. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/tests/test_scoring.py +0 -0
  87. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/tests/test_search_tool_evaluator.py +0 -0
  88. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/tests/test_summary_client.py +0 -0
  89. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/tests/test_summary_evaluator.py +0 -0
  90. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/tests/test_text_evaluators.py +0 -0
  91. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/tests/test_visualization_evaluator.py +0 -0
  92. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/tests/test_workspace.py +0 -0
  93. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/tox.ini +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: gooddata-eval
3
- Version: 1.73.0
3
+ Version: 1.73.1.dev2
4
4
  Summary: Evaluate the GoodData AI agent against your own questions and models.
5
5
  Project-URL: Source, https://github.com/gooddata/gooddata-python-sdk
6
6
  Author-email: GoodData <support@gooddata.com>
@@ -17,7 +17,7 @@ Classifier: Topic :: Scientific/Engineering
17
17
  Classifier: Topic :: Software Development
18
18
  Classifier: Typing :: Typed
19
19
  Requires-Python: >=3.10
20
- Requires-Dist: gooddata-sdk~=1.73.0
20
+ Requires-Dist: gooddata-sdk~=1.73.1.dev2
21
21
  Requires-Dist: httpx<1.0,>=0.27
22
22
  Requires-Dist: orjson<4.0.0,>=3.9.15
23
23
  Requires-Dist: pydantic<3.0,>=2.6
@@ -1,7 +1,7 @@
1
1
  # (C) 2026 GoodData Corporation
2
2
  [project]
3
3
  name = "gooddata-eval"
4
- version = "1.73.0"
4
+ version = "1.73.1.dev2"
5
5
  description = "Evaluate the GoodData AI agent against your own questions and models."
6
6
  readme = "README.md"
7
7
  license = "MIT"
@@ -11,7 +11,7 @@ authors = [
11
11
  keywords = ["gooddata", "ai", "evaluation", "llm", "analytics", "cli"]
12
12
  requires-python = ">=3.10"
13
13
  dependencies = [
14
- "gooddata-sdk~=1.73.0",
14
+ "gooddata-sdk~=1.73.1.dev2",
15
15
  "httpx>=0.27,<1.0",
16
16
  "orjson>=3.9.15,<4.0.0",
17
17
  "pydantic>=2.6,<3.0",
@@ -86,12 +86,12 @@ def _dispatch_agentic(
86
86
  model_version_override: str | None,
87
87
  reasoning_effort: ReasoningEffort | None = None,
88
88
  agent_id: str | None = None,
89
- ) -> AgenticEvalOutcome | list[str] | None:
89
+ ) -> AgenticEvalOutcome:
90
90
  """Call the appropriate evaluate_agentic_* function for the item's test_kind.
91
91
 
92
- Returns whatever that function returns -- alert_skill/metric_skill/conversation return
93
- an AgenticEvalOutcome; the rest still return None
94
- (unchanged).
92
+ Every evaluate_agentic_* function returns an AgenticEvalOutcome (reasoning_steps,
93
+ conversation_id, response_id, detail) on success and attaches the same four attributes
94
+ to its raised *AssertionError on failure -- no kind is exempt.
95
95
  """
96
96
  kind = item.test_kind
97
97
  eo = item.expected_output
@@ -174,13 +174,14 @@ def _dispatch_agentic(
174
174
  **lf_kw,
175
175
  )
176
176
  elif kind == "agentic_kda_skill":
177
- evaluate_agentic_kda_skill(
177
+ return evaluate_agentic_kda_skill(
178
178
  host=host,
179
179
  token=token,
180
180
  workspace_id=workspace_id,
181
181
  question=item.question,
182
182
  expected_output=eo if isinstance(eo, dict) else {},
183
183
  k=k,
184
+ agent_id=agent_id,
184
185
  **lf_kw,
185
186
  )
186
187
  elif kind == "agentic_conversation":
@@ -240,19 +241,22 @@ def run_agentic_items(
240
241
  reasoning_steps = outcome.reasoning_steps
241
242
  conversation_id = outcome.conversation_id
242
243
  response_id = outcome.response_id
244
+ detail = outcome.detail
243
245
  else:
244
- reasoning_steps, conversation_id, response_id = outcome, None, None
246
+ reasoning_steps, conversation_id, response_id, detail = outcome, None, None, {}
245
247
  item_report.pass_at_k = True
246
248
  item_report.runs = k
247
249
  item_report.reasoning_steps = reasoning_steps or []
248
250
  item_report.conversation_id = conversation_id
249
251
  item_report.response_id = response_id
252
+ item_report.best_detail = detail or {}
250
253
  except AssertionError as exc:
251
254
  item_report.pass_at_k = False
252
255
  item_report.runs = k
253
256
  item_report.reasoning_steps = getattr(exc, "reasoning_steps", None) or []
254
257
  item_report.conversation_id = getattr(exc, "conversation_id", None)
255
258
  item_report.response_id = getattr(exc, "response_id", None)
259
+ item_report.best_detail = getattr(exc, "detail", None) or {}
256
260
  print(f"[agentic] {item.id} FAIL: {exc}", flush=True)
257
261
  except Exception as exc:
258
262
  item_report.error = f"{type(exc).__name__}: {exc}"
@@ -14,7 +14,7 @@ from gooddata_sdk import GoodDataSdk
14
14
  from gooddata_eval.core.agentic._catalog import CatalogMetricAlert
15
15
  from gooddata_eval.core.chat.sse_client import ChatClient
16
16
  from gooddata_eval.core.config import ReasoningEffort
17
- from gooddata_eval.core.models import AgenticEvalOutcome, ToolCallEvent
17
+ from gooddata_eval.core.models import AgenticEvalOutcome, ReasoningStepEvent, ToolCallEvent, build_latency_breakdown
18
18
 
19
19
  try:
20
20
  from openai import OpenAI as _OpenAI
@@ -160,8 +160,18 @@ def _check_recipients(expected: CatalogMetricAlert, actual_args: dict, sdk: Good
160
160
  act_recip = []
161
161
  if set(expected.recipients) == set(act_recip or []):
162
162
  return True
163
- act_internal = actual_args.get("internal_recipients")
164
- if sdk is not None and isinstance(act_internal, list) and act_internal:
163
+ act_internal_raw = actual_args.get("internal_recipients")
164
+ # internal_recipients is declared `anyOf: [array of string, string, null]` in the
165
+ # create_metric_alert tool schema -- a single id as a bare string is schema-legal,
166
+ # not a malformed call, so it needs the same string/list normalization already
167
+ # applied to recipients/external_recipients above.
168
+ if isinstance(act_internal_raw, str):
169
+ act_internal = [act_internal_raw]
170
+ elif isinstance(act_internal_raw, list):
171
+ act_internal = act_internal_raw
172
+ else:
173
+ act_internal = []
174
+ if sdk is not None and act_internal:
165
175
  internal_recipient_ids = _resolve_internal_recipient_ids(sdk, expected.recipients)
166
176
  if internal_recipient_ids & set(act_internal):
167
177
  return True
@@ -335,6 +345,8 @@ class AlertRunResult:
335
345
  actual_alert_arguments: dict
336
346
  reasoning_steps: list[str] = field(default_factory=list)
337
347
  response_id: str | None = None
348
+ tool_call_events: list[ToolCallEvent] = field(default_factory=list)
349
+ reasoning_step_events: list[ReasoningStepEvent] = field(default_factory=list)
338
350
 
339
351
 
340
352
  @dataclass
@@ -485,6 +497,11 @@ def run_agentic_alert_skill(
485
497
  tool_called = False
486
498
  reasoning_steps: list[str] = []
487
499
  response_id: str | None = None
500
+ all_tool_call_events: list[ToolCallEvent] = []
501
+ all_reasoning_step_events: list[ReasoningStepEvent] = []
502
+ turn_offset = 0.0 # each turn's call_ts/ts restarts near 0 -- shift by prior turns' wall time
503
+ tool_index_offset = 0
504
+ reasoning_index_offset = 0
488
505
  # conversation_history stores prior turns for GPT-4o context.
489
506
  # Roles follow GPT-4o's perspective: "assistant"=agent text, "user"=sim-user reply.
490
507
  conversation_history: list = []
@@ -494,6 +511,21 @@ def run_agentic_alert_skill(
494
511
  chat_result = client.send_message(conv_id, current_question)
495
512
  reasoning_steps.extend(chat_result.reasoning_steps or [])
496
513
  response_id = chat_result.response_id or response_id
514
+ for tc in chat_result.tool_call_events or []:
515
+ if tc.call_ts is not None:
516
+ tc.call_ts += turn_offset
517
+ if tc.result_ts is not None:
518
+ tc.result_ts += turn_offset
519
+ if tc.index is not None:
520
+ tc.index += tool_index_offset
521
+ for rs in chat_result.reasoning_step_events or []:
522
+ rs.ts += turn_offset
523
+ rs.index += reasoning_index_offset
524
+ all_tool_call_events.extend(chat_result.tool_call_events or [])
525
+ all_reasoning_step_events.extend(chat_result.reasoning_step_events or [])
526
+ tool_index_offset += len(chat_result.tool_call_events or [])
527
+ reasoning_index_offset += len(chat_result.reasoning_step_events or [])
528
+ turn_offset += chat_result.turn_wall_clock_sec or 0.0
497
529
  alert_id, actual_args, tool_called = _extract_alert_call(chat_result.tool_call_events or [])
498
530
  if tool_called:
499
531
  alert_id_to_delete = alert_id
@@ -531,6 +563,8 @@ def run_agentic_alert_skill(
531
563
  actual_alert_arguments=actual_args,
532
564
  reasoning_steps=reasoning_steps,
533
565
  response_id=response_id,
566
+ tool_call_events=all_tool_call_events,
567
+ reasoning_step_events=all_reasoning_step_events,
534
568
  )
535
569
  finally:
536
570
  if alert_id_to_delete:
@@ -584,6 +618,7 @@ class AlertSkillAssertionError(AssertionError):
584
618
  reasoning_steps: list[str]
585
619
  conversation_id: str
586
620
  response_id: str | None
621
+ detail: dict
587
622
 
588
623
 
589
624
  def evaluate_agentic_alert_skill(
@@ -697,9 +732,33 @@ def evaluate_agentic_alert_skill(
697
732
  exc.reasoning_steps = best.reasoning_steps
698
733
  exc.conversation_id = best.conversation_id
699
734
  exc.response_id = best.response_id
735
+ exc.detail = {
736
+ "alert_created": ev.alert_created,
737
+ "operator_correct": ev.operator_correct,
738
+ "threshold_correct": ev.threshold_correct,
739
+ "trigger_correct": ev.trigger_correct,
740
+ "filters_correct": ev.filters_correct,
741
+ "metric_correct": ev.metric_correct,
742
+ "recipients_correct": ev.recipients_correct,
743
+ "actual_alert_arguments": best.actual_alert_arguments,
744
+ "latency_breakdown": build_latency_breakdown(best.tool_call_events, best.reasoning_step_events),
745
+ }
700
746
  raise exc
747
+ best = summary.best
748
+ ev = best.eval
701
749
  return AgenticEvalOutcome(
702
- reasoning_steps=summary.best.reasoning_steps,
703
- conversation_id=summary.best.conversation_id,
704
- response_id=summary.best.response_id,
750
+ reasoning_steps=best.reasoning_steps,
751
+ conversation_id=best.conversation_id,
752
+ response_id=best.response_id,
753
+ detail={
754
+ "alert_created": ev.alert_created,
755
+ "operator_correct": ev.operator_correct,
756
+ "threshold_correct": ev.threshold_correct,
757
+ "trigger_correct": ev.trigger_correct,
758
+ "filters_correct": ev.filters_correct,
759
+ "metric_correct": ev.metric_correct,
760
+ "recipients_correct": ev.recipients_correct,
761
+ "actual_alert_arguments": best.actual_alert_arguments,
762
+ "latency_breakdown": build_latency_breakdown(best.tool_call_events, best.reasoning_step_events),
763
+ },
705
764
  )
@@ -12,10 +12,16 @@ from gooddata_sdk import GoodDataSdk
12
12
  from pydantic import BaseModel
13
13
 
14
14
  from gooddata_eval.core.agentic.alert_skill import render_alert_proposal
15
- from gooddata_eval.core.agentic.metric_skill import _delete_metric, _extract_created_metric_ids
15
+ from gooddata_eval.core.agentic.metric_skill import _delete_metric, _extract_created_metric_ids, _extract_metric_result
16
16
  from gooddata_eval.core.chat.sse_client import ChatClient
17
17
  from gooddata_eval.core.config import ReasoningEffort
18
- from gooddata_eval.core.models import AgenticEvalOutcome, ChatResult, ToolCallEvent
18
+ from gooddata_eval.core.models import (
19
+ AgenticEvalOutcome,
20
+ ChatResult,
21
+ ReasoningStepEvent,
22
+ ToolCallEvent,
23
+ build_latency_breakdown,
24
+ )
19
25
  from gooddata_eval.core.scoring import (
20
26
  check_filters,
21
27
  check_viz_type,
@@ -120,7 +126,7 @@ def _check_output_present(turn: TurnDefinition, chat_result: ChatResult) -> bool
120
126
  and getattr(chat_result.created_visualizations, "objects", chat_result.created_visualizations)
121
127
  )
122
128
  if otype == "metric":
123
- return any(tc.function_name == "create_metric" for tc in (chat_result.tool_call_events or []))
129
+ return _extract_metric_result(chat_result.tool_call_events or []) is not None
124
130
  if otype == "tool_call":
125
131
  expected_tool = turn.expected_tool_name
126
132
  if not expected_tool:
@@ -129,19 +135,6 @@ def _check_output_present(turn: TurnDefinition, chat_result: ChatResult) -> bool
129
135
  return False
130
136
 
131
137
 
132
- def _extract_metric_from_turn(tool_call_events: list[ToolCallEvent]) -> dict | None:
133
- """Extract the result payload from the create_metric tool call, if present."""
134
- for tc in tool_call_events:
135
- if tc.function_name != "create_metric":
136
- continue
137
- if not tc.result:
138
- continue
139
- result_data = tc.parsed_result()
140
- if result_data is not None:
141
- return result_data.get("data", result_data)
142
- return None
143
-
144
-
145
138
  def _check_output_correct(turn: TurnDefinition, chat_result: ChatResult) -> bool | None:
146
139
  """Check output correctness against expected_output when defined.
147
140
 
@@ -186,7 +179,7 @@ def _check_output_correct(turn: TurnDefinition, chat_result: ChatResult) -> bool
186
179
  return all(results) if results else None
187
180
 
188
181
  if otype == "metric":
189
- metric_result = _extract_metric_from_turn(chat_result.tool_call_events or [])
182
+ metric_result = _extract_metric_result(chat_result.tool_call_events or [])
190
183
  if not metric_result:
191
184
  return False
192
185
  return _normalize_maql(metric_result.get("maql", "")) == _normalize_maql(expected.get("maql", ""))
@@ -267,6 +260,8 @@ class ConversationResult:
267
260
  total_clarification_turns: int
268
261
  reasoning_steps: list[str] = field(default_factory=list)
269
262
  response_id: str | None = None
263
+ tool_call_events: list[ToolCallEvent] = field(default_factory=list)
264
+ reasoning_step_events: list[ReasoningStepEvent] = field(default_factory=list)
270
265
 
271
266
 
272
267
  def run_agentic_conversation(
@@ -300,6 +295,14 @@ def run_agentic_conversation(
300
295
  created_metric_ids: list[str] = []
301
296
  reasoning_steps: list[str] = []
302
297
  response_id: str | None = None
298
+ conversation_tool_call_events: list[ToolCallEvent] = []
299
+ conversation_reasoning_step_events: list[ReasoningStepEvent] = []
300
+ # Every send_message() call (across every logical turn AND every clarification
301
+ # sub-turn within it) restarts call_ts/ts near 0 -- these run across the whole
302
+ # conversation, not reset per logical turn, so every one of those calls shifts them.
303
+ turn_offset = 0.0
304
+ tool_index_offset = 0
305
+ reasoning_index_offset = 0
303
306
 
304
307
  try:
305
308
  if initial_conversation_id is not None:
@@ -335,7 +338,22 @@ def run_agentic_conversation(
335
338
  for _iter in range(max_clarification_turns + 1):
336
339
  chat_result = client.send_message(conversation_id, current_message)
337
340
  final_result = chat_result
341
+ for tc in chat_result.tool_call_events or []:
342
+ if tc.call_ts is not None:
343
+ tc.call_ts += turn_offset
344
+ if tc.result_ts is not None:
345
+ tc.result_ts += turn_offset
346
+ if tc.index is not None:
347
+ tc.index += tool_index_offset
348
+ for rs in chat_result.reasoning_step_events or []:
349
+ rs.ts += turn_offset
350
+ rs.index += reasoning_index_offset
338
351
  all_tool_calls.extend(chat_result.tool_call_events or [])
352
+ conversation_tool_call_events.extend(chat_result.tool_call_events or [])
353
+ conversation_reasoning_step_events.extend(chat_result.reasoning_step_events or [])
354
+ tool_index_offset += len(chat_result.tool_call_events or [])
355
+ reasoning_index_offset += len(chat_result.reasoning_step_events or [])
356
+ turn_offset += chat_result.turn_wall_clock_sec or 0.0
339
357
  reasoning_steps.extend(chat_result.reasoning_steps or [])
340
358
  response_id = chat_result.response_id or response_id
341
359
 
@@ -362,7 +380,7 @@ def run_agentic_conversation(
362
380
 
363
381
  # Capture metric output for $ref resolution in subsequent turns.
364
382
  if final_result and turn.expected_output_type == "metric":
365
- metric_data = _extract_metric_from_turn(all_tool_calls)
383
+ metric_data = _extract_metric_result(all_tool_calls)
366
384
  if metric_data:
367
385
  turn_outputs[turn.turn_id] = metric_data
368
386
 
@@ -403,9 +421,30 @@ def run_agentic_conversation(
403
421
  total_clarification_turns=total_clarification_turns,
404
422
  reasoning_steps=reasoning_steps,
405
423
  response_id=response_id,
424
+ tool_call_events=conversation_tool_call_events,
425
+ reasoning_step_events=conversation_reasoning_step_events,
406
426
  )
407
427
 
408
428
 
429
+ def _conversation_detail(result: ConversationResult) -> dict:
430
+ return {
431
+ "full_skill_coverage": result.full_skill_coverage,
432
+ "total_clarification_turns": result.total_clarification_turns,
433
+ "turns": [
434
+ {
435
+ "turn_id": tr.turn_id,
436
+ "expected_skill": tr.expected_skill,
437
+ "skill_routing": tr.skill_routing,
438
+ "output_present": tr.output_present,
439
+ "output_correct": tr.output_correct,
440
+ "activated_skills": tr.activated_skills,
441
+ }
442
+ for tr in result.turn_results
443
+ ],
444
+ "latency_breakdown": build_latency_breakdown(result.tool_call_events, result.reasoning_step_events),
445
+ }
446
+
447
+
409
448
  class ConversationAssertionError(AssertionError):
410
449
  """Raised when a conversation evaluation fails."""
411
450
 
@@ -413,6 +452,7 @@ class ConversationAssertionError(AssertionError):
413
452
  reasoning_steps: list[str]
414
453
  conversation_id: str
415
454
  response_id: str | None
455
+ detail: dict
416
456
 
417
457
 
418
458
  def evaluate_agentic_conversation(
@@ -524,9 +564,11 @@ def evaluate_agentic_conversation(
524
564
  exc.reasoning_steps = result.reasoning_steps
525
565
  exc.conversation_id = result.conversation_id
526
566
  exc.response_id = result.response_id
567
+ exc.detail = _conversation_detail(result)
527
568
  raise exc
528
569
  return AgenticEvalOutcome(
529
570
  reasoning_steps=result.reasoning_steps,
530
571
  conversation_id=result.conversation_id,
531
572
  response_id=result.response_id,
573
+ detail=_conversation_detail(result),
532
574
  )
@@ -3,11 +3,12 @@
3
3
 
4
4
  from __future__ import annotations
5
5
 
6
- from dataclasses import dataclass
6
+ from dataclasses import dataclass, field
7
7
 
8
8
  from gooddata_eval.core.chat.sse_client import ChatClient
9
9
  from gooddata_eval.core.config import ReasoningEffort
10
10
  from gooddata_eval.core.evaluators._llm_judge import LLMJudge
11
+ from gooddata_eval.core.models import AgenticEvalOutcome
11
12
 
12
13
  _DEFAULT_K = 1
13
14
 
@@ -52,6 +53,8 @@ class GeneralQuestionResult:
52
53
  passed: bool
53
54
  llm_judge_score: float
54
55
  reasoning: str
56
+ reasoning_steps: list[str] = field(default_factory=list)
57
+ response_id: str | None = None
55
58
 
56
59
 
57
60
  @dataclass
@@ -98,6 +101,8 @@ def run_agentic_general_question(
98
101
  passed=passed,
99
102
  llm_judge_score=llm_judge_score,
100
103
  reasoning=reasoning,
104
+ reasoning_steps=list(chat_result.reasoning_steps or []),
105
+ response_id=chat_result.response_id,
101
106
  )
102
107
  )
103
108
  finally:
@@ -120,6 +125,8 @@ def run_agentic_general_question(
120
125
  passed=passed,
121
126
  llm_judge_score=llm_judge_score,
122
127
  reasoning=reasoning,
128
+ reasoning_steps=list(chat_result.reasoning_steps or []),
129
+ response_id=chat_result.response_id,
123
130
  )
124
131
  )
125
132
  finally:
@@ -142,6 +149,10 @@ class GeneralQuestionAssertionError(AssertionError):
142
149
  """Raised when a general-question evaluation fails."""
143
150
 
144
151
  __tracebackhide__ = True
152
+ reasoning_steps: list[str]
153
+ conversation_id: str
154
+ response_id: str | None
155
+ detail: dict
145
156
 
146
157
 
147
158
  def evaluate_agentic_general_question(
@@ -160,8 +171,13 @@ def evaluate_agentic_general_question(
160
171
  model_version_override: str | None = None,
161
172
  run_metadata_extra: dict | None = None,
162
173
  reasoning_effort: ReasoningEffort | None = None,
163
- ) -> None:
164
- """Run general-question evaluation, log to Langfuse, and raise on failure."""
174
+ ) -> AgenticEvalOutcome:
175
+ """Run general-question evaluation, log to Langfuse, and raise GeneralQuestionAssertionError on failure.
176
+
177
+ Returns the best run's outcome (reasoning_steps, conversation_id, response_id) as an
178
+ AgenticEvalOutcome on success; on failure the same three values are attached to the
179
+ raised exception as ``.reasoning_steps``/``.conversation_id``/``.response_id``.
180
+ """
165
181
  from datetime import datetime as _dt # noqa: PLC0415
166
182
  from datetime import timezone as _tz # noqa: PLC0415
167
183
 
@@ -223,6 +239,26 @@ def evaluate_agentic_general_question(
223
239
 
224
240
  if not summary.pass_at_k:
225
241
  best = summary.best
226
- raise GeneralQuestionAssertionError(
242
+ exc = GeneralQuestionAssertionError(
227
243
  f"General question assertion failed. passed={best.passed}. Reasoning: {best.reasoning}"
228
244
  )
245
+ exc.reasoning_steps = best.reasoning_steps
246
+ exc.conversation_id = best.conversation_id
247
+ exc.response_id = best.response_id
248
+ exc.detail = {
249
+ "judge_passed": best.passed,
250
+ "judge_reasoning": best.reasoning,
251
+ "actual_output": best.actual_output,
252
+ }
253
+ raise exc
254
+ best = summary.best
255
+ return AgenticEvalOutcome(
256
+ reasoning_steps=best.reasoning_steps,
257
+ conversation_id=best.conversation_id,
258
+ response_id=best.response_id,
259
+ detail={
260
+ "judge_passed": best.passed,
261
+ "judge_reasoning": best.reasoning,
262
+ "actual_output": best.actual_output,
263
+ },
264
+ )
@@ -3,11 +3,12 @@
3
3
 
4
4
  from __future__ import annotations
5
5
 
6
- from dataclasses import dataclass
6
+ from dataclasses import dataclass, field
7
7
 
8
8
  from gooddata_eval.core.chat.sse_client import ChatClient
9
9
  from gooddata_eval.core.config import ReasoningEffort
10
10
  from gooddata_eval.core.evaluators._llm_judge import LLMJudge
11
+ from gooddata_eval.core.models import AgenticEvalOutcome, ReasoningStepEvent, ToolCallEvent, build_latency_breakdown
11
12
 
12
13
  _DEFAULT_K = 1
13
14
 
@@ -49,6 +50,10 @@ class GuardrailResult:
49
50
  passed: bool
50
51
  llm_judge_score: float
51
52
  reasoning: str
53
+ reasoning_steps: list[str] = field(default_factory=list)
54
+ response_id: str | None = None
55
+ tool_call_events: list[ToolCallEvent] = field(default_factory=list)
56
+ reasoning_step_events: list[ReasoningStepEvent] = field(default_factory=list)
52
57
 
53
58
 
54
59
  @dataclass
@@ -95,6 +100,10 @@ def run_agentic_guardrail(
95
100
  passed=passed,
96
101
  llm_judge_score=llm_judge_score,
97
102
  reasoning=reasoning,
103
+ reasoning_steps=list(chat_result.reasoning_steps or []),
104
+ response_id=chat_result.response_id,
105
+ tool_call_events=list(chat_result.tool_call_events or []),
106
+ reasoning_step_events=list(chat_result.reasoning_step_events or []),
98
107
  )
99
108
  )
100
109
  finally:
@@ -117,6 +126,10 @@ def run_agentic_guardrail(
117
126
  passed=passed,
118
127
  llm_judge_score=llm_judge_score,
119
128
  reasoning=reasoning,
129
+ reasoning_steps=list(chat_result.reasoning_steps or []),
130
+ response_id=chat_result.response_id,
131
+ tool_call_events=list(chat_result.tool_call_events or []),
132
+ reasoning_step_events=list(chat_result.reasoning_step_events or []),
120
133
  )
121
134
  )
122
135
  finally:
@@ -139,6 +152,10 @@ class GuardrailAssertionError(AssertionError):
139
152
  """Raised when a guardrail evaluation fails."""
140
153
 
141
154
  __tracebackhide__ = True
155
+ reasoning_steps: list[str]
156
+ conversation_id: str
157
+ response_id: str | None
158
+ detail: dict
142
159
 
143
160
 
144
161
  def evaluate_agentic_guardrail(
@@ -157,8 +174,14 @@ def evaluate_agentic_guardrail(
157
174
  model_version_override: str | None = None,
158
175
  run_metadata_extra: dict | None = None,
159
176
  reasoning_effort: ReasoningEffort | None = None,
160
- ) -> None:
161
- """Run guardrail evaluation, log to Langfuse, and raise on failure."""
177
+ ) -> AgenticEvalOutcome:
178
+ """Run guardrail evaluation, log to Langfuse, and raise GuardrailAssertionError on failure.
179
+
180
+ Returns the best run's outcome (reasoning_steps, conversation_id, response_id) as an
181
+ AgenticEvalOutcome on success; on failure the same three values are attached to the
182
+ raised exception as ``.reasoning_steps``/``.conversation_id``/``.response_id`` (mirrors
183
+ `evaluate_agentic_metric_skill`'s idiom) so callers can retrieve them either way.
184
+ """
162
185
  from datetime import datetime as _dt # noqa: PLC0415
163
186
  from datetime import timezone as _tz # noqa: PLC0415
164
187
 
@@ -220,4 +243,26 @@ def evaluate_agentic_guardrail(
220
243
 
221
244
  if not summary.pass_at_k:
222
245
  best = summary.best
223
- raise GuardrailAssertionError(f"Guardrail assertion failed. passed={best.passed}. Reasoning: {best.reasoning}")
246
+ exc = GuardrailAssertionError(f"Guardrail assertion failed. passed={best.passed}. Reasoning: {best.reasoning}")
247
+ exc.reasoning_steps = best.reasoning_steps
248
+ exc.conversation_id = best.conversation_id
249
+ exc.response_id = best.response_id
250
+ exc.detail = {
251
+ "judge_passed": best.passed,
252
+ "judge_reasoning": best.reasoning,
253
+ "actual_output": best.actual_output,
254
+ "latency_breakdown": build_latency_breakdown(best.tool_call_events, best.reasoning_step_events),
255
+ }
256
+ raise exc
257
+ best = summary.best
258
+ return AgenticEvalOutcome(
259
+ reasoning_steps=best.reasoning_steps,
260
+ conversation_id=best.conversation_id,
261
+ response_id=best.response_id,
262
+ detail={
263
+ "judge_passed": best.passed,
264
+ "judge_reasoning": best.reasoning,
265
+ "actual_output": best.actual_output,
266
+ "latency_breakdown": build_latency_breakdown(best.tool_call_events, best.reasoning_step_events),
267
+ },
268
+ )