gooddata-eval 1.73.0__tar.gz → 1.73.1.dev1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (92) hide show
  1. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/PKG-INFO +2 -2
  2. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/pyproject.toml +2 -2
  3. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/cli/agentic_runner.py +10 -6
  4. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/agentic/alert_skill.py +38 -5
  5. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/agentic/conversation.py +25 -17
  6. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/agentic/general_question.py +40 -4
  7. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/agentic/guardrail.py +41 -4
  8. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/agentic/kda_skill.py +59 -6
  9. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/agentic/metric_skill.py +40 -12
  10. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/agentic/search_tool.py +40 -5
  11. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/agentic/visualization.py +35 -5
  12. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/chat/sse_client.py +14 -1
  13. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/evaluators/visualization.py +28 -19
  14. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/models.py +9 -1
  15. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/tests/test_agentic_alert_skill.py +33 -0
  16. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/tests/test_agentic_conversation.py +77 -0
  17. gooddata_eval-1.73.1.dev1/tests/test_agentic_general_question.py +210 -0
  18. gooddata_eval-1.73.1.dev1/tests/test_agentic_guardrail.py +208 -0
  19. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/tests/test_agentic_kda_skill.py +147 -0
  20. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/tests/test_agentic_metric_skill.py +119 -1
  21. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/tests/test_agentic_runner.py +70 -23
  22. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/tests/test_agentic_search_tool.py +96 -0
  23. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/tests/test_agentic_visualization.py +130 -0
  24. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/tests/test_sse_client.py +42 -0
  25. gooddata_eval-1.73.0/tests/test_agentic_general_question.py +0 -100
  26. gooddata_eval-1.73.0/tests/test_agentic_guardrail.py +0 -98
  27. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/.gitignore +0 -0
  28. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/LICENSE.txt +0 -0
  29. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/Makefile +0 -0
  30. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/README.md +0 -0
  31. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/__init__.py +0 -0
  32. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/_version.py +0 -0
  33. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/cli/__init__.py +0 -0
  34. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/cli/main.py +0 -0
  35. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/__init__.py +0 -0
  36. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/agentic/__init__.py +0 -0
  37. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/agentic/_catalog.py +0 -0
  38. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/agentic/_langfuse.py +0 -0
  39. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/chat/__init__.py +0 -0
  40. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/config.py +0 -0
  41. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/connection.py +0 -0
  42. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/dataset/__init__.py +0 -0
  43. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/dataset/langfuse_source.py +0 -0
  44. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/dataset/local.py +0 -0
  45. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/evaluators/__init__.py +0 -0
  46. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/evaluators/_deep_subset.py +0 -0
  47. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/evaluators/_llm_judge.py +0 -0
  48. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/evaluators/_text_utils.py +0 -0
  49. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/evaluators/alert_skill.py +0 -0
  50. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/evaluators/base.py +0 -0
  51. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/evaluators/general_question.py +0 -0
  52. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/evaluators/guardrail.py +0 -0
  53. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/evaluators/metric_skill.py +0 -0
  54. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/evaluators/search_tool.py +0 -0
  55. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/evaluators/summary.py +0 -0
  56. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/langfuse/__init__.py +0 -0
  57. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/langfuse/sink.py +0 -0
  58. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/reporting/__init__.py +0 -0
  59. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/reporting/console.py +0 -0
  60. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/reporting/json_report.py +0 -0
  61. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/runner.py +0 -0
  62. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/scoring.py +0 -0
  63. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/summary/__init__.py +0 -0
  64. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/summary/http_client.py +0 -0
  65. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/workspace.py +0 -0
  66. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/tests/__init__.py +0 -0
  67. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/tests/conftest.py +0 -0
  68. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/tests/fixtures/sample_dataset/metric_skill_create.json +0 -0
  69. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/tests/fixtures/sample_dataset/visualization_revenue.json +0 -0
  70. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/tests/fixtures/sse_visualization_stream.txt +0 -0
  71. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/tests/test_agentic_langfuse_trace.py +0 -0
  72. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/tests/test_agentic_run_context.py +0 -0
  73. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/tests/test_alert_skill_evaluator.py +0 -0
  74. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/tests/test_cli.py +0 -0
  75. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/tests/test_connection.py +0 -0
  76. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/tests/test_deep_subset.py +0 -0
  77. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/tests/test_langfuse_sink.py +0 -0
  78. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/tests/test_langfuse_source.py +0 -0
  79. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/tests/test_llm_judge.py +0 -0
  80. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/tests/test_local_loader.py +0 -0
  81. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/tests/test_metric_skill_evaluator.py +0 -0
  82. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/tests/test_models.py +0 -0
  83. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/tests/test_reporting.py +0 -0
  84. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/tests/test_runner.py +0 -0
  85. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/tests/test_scoring.py +0 -0
  86. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/tests/test_search_tool_evaluator.py +0 -0
  87. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/tests/test_summary_client.py +0 -0
  88. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/tests/test_summary_evaluator.py +0 -0
  89. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/tests/test_text_evaluators.py +0 -0
  90. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/tests/test_visualization_evaluator.py +0 -0
  91. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/tests/test_workspace.py +0 -0
  92. {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/tox.ini +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: gooddata-eval
3
- Version: 1.73.0
3
+ Version: 1.73.1.dev1
4
4
  Summary: Evaluate the GoodData AI agent against your own questions and models.
5
5
  Project-URL: Source, https://github.com/gooddata/gooddata-python-sdk
6
6
  Author-email: GoodData <support@gooddata.com>
@@ -17,7 +17,7 @@ Classifier: Topic :: Scientific/Engineering
17
17
  Classifier: Topic :: Software Development
18
18
  Classifier: Typing :: Typed
19
19
  Requires-Python: >=3.10
20
- Requires-Dist: gooddata-sdk~=1.73.0
20
+ Requires-Dist: gooddata-sdk~=1.73.1.dev1
21
21
  Requires-Dist: httpx<1.0,>=0.27
22
22
  Requires-Dist: orjson<4.0.0,>=3.9.15
23
23
  Requires-Dist: pydantic<3.0,>=2.6
@@ -1,7 +1,7 @@
1
1
  # (C) 2026 GoodData Corporation
2
2
  [project]
3
3
  name = "gooddata-eval"
4
- version = "1.73.0"
4
+ version = "1.73.1.dev1"
5
5
  description = "Evaluate the GoodData AI agent against your own questions and models."
6
6
  readme = "README.md"
7
7
  license = "MIT"
@@ -11,7 +11,7 @@ authors = [
11
11
  keywords = ["gooddata", "ai", "evaluation", "llm", "analytics", "cli"]
12
12
  requires-python = ">=3.10"
13
13
  dependencies = [
14
- "gooddata-sdk~=1.73.0",
14
+ "gooddata-sdk~=1.73.1.dev1",
15
15
  "httpx>=0.27,<1.0",
16
16
  "orjson>=3.9.15,<4.0.0",
17
17
  "pydantic>=2.6,<3.0",
@@ -86,12 +86,12 @@ def _dispatch_agentic(
86
86
  model_version_override: str | None,
87
87
  reasoning_effort: ReasoningEffort | None = None,
88
88
  agent_id: str | None = None,
89
- ) -> AgenticEvalOutcome | list[str] | None:
89
+ ) -> AgenticEvalOutcome:
90
90
  """Call the appropriate evaluate_agentic_* function for the item's test_kind.
91
91
 
92
- Returns whatever that function returns -- alert_skill/metric_skill/conversation return
93
- an AgenticEvalOutcome; the rest still return None
94
- (unchanged).
92
+ Every evaluate_agentic_* function returns an AgenticEvalOutcome (reasoning_steps,
93
+ conversation_id, response_id, detail) on success and attaches the same four attributes
94
+ to its raised *AssertionError on failure -- no kind is exempt.
95
95
  """
96
96
  kind = item.test_kind
97
97
  eo = item.expected_output
@@ -174,13 +174,14 @@ def _dispatch_agentic(
174
174
  **lf_kw,
175
175
  )
176
176
  elif kind == "agentic_kda_skill":
177
- evaluate_agentic_kda_skill(
177
+ return evaluate_agentic_kda_skill(
178
178
  host=host,
179
179
  token=token,
180
180
  workspace_id=workspace_id,
181
181
  question=item.question,
182
182
  expected_output=eo if isinstance(eo, dict) else {},
183
183
  k=k,
184
+ agent_id=agent_id,
184
185
  **lf_kw,
185
186
  )
186
187
  elif kind == "agentic_conversation":
@@ -240,19 +241,22 @@ def run_agentic_items(
240
241
  reasoning_steps = outcome.reasoning_steps
241
242
  conversation_id = outcome.conversation_id
242
243
  response_id = outcome.response_id
244
+ detail = outcome.detail
243
245
  else:
244
- reasoning_steps, conversation_id, response_id = outcome, None, None
246
+ reasoning_steps, conversation_id, response_id, detail = outcome, None, None, {}
245
247
  item_report.pass_at_k = True
246
248
  item_report.runs = k
247
249
  item_report.reasoning_steps = reasoning_steps or []
248
250
  item_report.conversation_id = conversation_id
249
251
  item_report.response_id = response_id
252
+ item_report.best_detail = detail or {}
250
253
  except AssertionError as exc:
251
254
  item_report.pass_at_k = False
252
255
  item_report.runs = k
253
256
  item_report.reasoning_steps = getattr(exc, "reasoning_steps", None) or []
254
257
  item_report.conversation_id = getattr(exc, "conversation_id", None)
255
258
  item_report.response_id = getattr(exc, "response_id", None)
259
+ item_report.best_detail = getattr(exc, "detail", None) or {}
256
260
  print(f"[agentic] {item.id} FAIL: {exc}", flush=True)
257
261
  except Exception as exc:
258
262
  item_report.error = f"{type(exc).__name__}: {exc}"
@@ -160,8 +160,18 @@ def _check_recipients(expected: CatalogMetricAlert, actual_args: dict, sdk: Good
160
160
  act_recip = []
161
161
  if set(expected.recipients) == set(act_recip or []):
162
162
  return True
163
- act_internal = actual_args.get("internal_recipients")
164
- if sdk is not None and isinstance(act_internal, list) and act_internal:
163
+ act_internal_raw = actual_args.get("internal_recipients")
164
+ # internal_recipients is declared `anyOf: [array of string, string, null]` in the
165
+ # create_metric_alert tool schema -- a single id as a bare string is schema-legal,
166
+ # not a malformed call, so it needs the same string/list normalization already
167
+ # applied to recipients/external_recipients above.
168
+ if isinstance(act_internal_raw, str):
169
+ act_internal = [act_internal_raw]
170
+ elif isinstance(act_internal_raw, list):
171
+ act_internal = act_internal_raw
172
+ else:
173
+ act_internal = []
174
+ if sdk is not None and act_internal:
165
175
  internal_recipient_ids = _resolve_internal_recipient_ids(sdk, expected.recipients)
166
176
  if internal_recipient_ids & set(act_internal):
167
177
  return True
@@ -584,6 +594,7 @@ class AlertSkillAssertionError(AssertionError):
584
594
  reasoning_steps: list[str]
585
595
  conversation_id: str
586
596
  response_id: str | None
597
+ detail: dict
587
598
 
588
599
 
589
600
  def evaluate_agentic_alert_skill(
@@ -697,9 +708,31 @@ def evaluate_agentic_alert_skill(
697
708
  exc.reasoning_steps = best.reasoning_steps
698
709
  exc.conversation_id = best.conversation_id
699
710
  exc.response_id = best.response_id
711
+ exc.detail = {
712
+ "alert_created": ev.alert_created,
713
+ "operator_correct": ev.operator_correct,
714
+ "threshold_correct": ev.threshold_correct,
715
+ "trigger_correct": ev.trigger_correct,
716
+ "filters_correct": ev.filters_correct,
717
+ "metric_correct": ev.metric_correct,
718
+ "recipients_correct": ev.recipients_correct,
719
+ "actual_alert_arguments": best.actual_alert_arguments,
720
+ }
700
721
  raise exc
722
+ best = summary.best
723
+ ev = best.eval
701
724
  return AgenticEvalOutcome(
702
- reasoning_steps=summary.best.reasoning_steps,
703
- conversation_id=summary.best.conversation_id,
704
- response_id=summary.best.response_id,
725
+ reasoning_steps=best.reasoning_steps,
726
+ conversation_id=best.conversation_id,
727
+ response_id=best.response_id,
728
+ detail={
729
+ "alert_created": ev.alert_created,
730
+ "operator_correct": ev.operator_correct,
731
+ "threshold_correct": ev.threshold_correct,
732
+ "trigger_correct": ev.trigger_correct,
733
+ "filters_correct": ev.filters_correct,
734
+ "metric_correct": ev.metric_correct,
735
+ "recipients_correct": ev.recipients_correct,
736
+ "actual_alert_arguments": best.actual_alert_arguments,
737
+ },
705
738
  )
@@ -12,7 +12,7 @@ from gooddata_sdk import GoodDataSdk
12
12
  from pydantic import BaseModel
13
13
 
14
14
  from gooddata_eval.core.agentic.alert_skill import render_alert_proposal
15
- from gooddata_eval.core.agentic.metric_skill import _delete_metric, _extract_created_metric_ids
15
+ from gooddata_eval.core.agentic.metric_skill import _delete_metric, _extract_created_metric_ids, _extract_metric_result
16
16
  from gooddata_eval.core.chat.sse_client import ChatClient
17
17
  from gooddata_eval.core.config import ReasoningEffort
18
18
  from gooddata_eval.core.models import AgenticEvalOutcome, ChatResult, ToolCallEvent
@@ -120,7 +120,7 @@ def _check_output_present(turn: TurnDefinition, chat_result: ChatResult) -> bool
120
120
  and getattr(chat_result.created_visualizations, "objects", chat_result.created_visualizations)
121
121
  )
122
122
  if otype == "metric":
123
- return any(tc.function_name == "create_metric" for tc in (chat_result.tool_call_events or []))
123
+ return _extract_metric_result(chat_result.tool_call_events or []) is not None
124
124
  if otype == "tool_call":
125
125
  expected_tool = turn.expected_tool_name
126
126
  if not expected_tool:
@@ -129,19 +129,6 @@ def _check_output_present(turn: TurnDefinition, chat_result: ChatResult) -> bool
129
129
  return False
130
130
 
131
131
 
132
- def _extract_metric_from_turn(tool_call_events: list[ToolCallEvent]) -> dict | None:
133
- """Extract the result payload from the create_metric tool call, if present."""
134
- for tc in tool_call_events:
135
- if tc.function_name != "create_metric":
136
- continue
137
- if not tc.result:
138
- continue
139
- result_data = tc.parsed_result()
140
- if result_data is not None:
141
- return result_data.get("data", result_data)
142
- return None
143
-
144
-
145
132
  def _check_output_correct(turn: TurnDefinition, chat_result: ChatResult) -> bool | None:
146
133
  """Check output correctness against expected_output when defined.
147
134
 
@@ -186,7 +173,7 @@ def _check_output_correct(turn: TurnDefinition, chat_result: ChatResult) -> bool
186
173
  return all(results) if results else None
187
174
 
188
175
  if otype == "metric":
189
- metric_result = _extract_metric_from_turn(chat_result.tool_call_events or [])
176
+ metric_result = _extract_metric_result(chat_result.tool_call_events or [])
190
177
  if not metric_result:
191
178
  return False
192
179
  return _normalize_maql(metric_result.get("maql", "")) == _normalize_maql(expected.get("maql", ""))
@@ -362,7 +349,7 @@ def run_agentic_conversation(
362
349
 
363
350
  # Capture metric output for $ref resolution in subsequent turns.
364
351
  if final_result and turn.expected_output_type == "metric":
365
- metric_data = _extract_metric_from_turn(all_tool_calls)
352
+ metric_data = _extract_metric_result(all_tool_calls)
366
353
  if metric_data:
367
354
  turn_outputs[turn.turn_id] = metric_data
368
355
 
@@ -406,6 +393,24 @@ def run_agentic_conversation(
406
393
  )
407
394
 
408
395
 
396
+ def _conversation_detail(result: ConversationResult) -> dict:
397
+ return {
398
+ "full_skill_coverage": result.full_skill_coverage,
399
+ "total_clarification_turns": result.total_clarification_turns,
400
+ "turns": [
401
+ {
402
+ "turn_id": tr.turn_id,
403
+ "expected_skill": tr.expected_skill,
404
+ "skill_routing": tr.skill_routing,
405
+ "output_present": tr.output_present,
406
+ "output_correct": tr.output_correct,
407
+ "activated_skills": tr.activated_skills,
408
+ }
409
+ for tr in result.turn_results
410
+ ],
411
+ }
412
+
413
+
409
414
  class ConversationAssertionError(AssertionError):
410
415
  """Raised when a conversation evaluation fails."""
411
416
 
@@ -413,6 +418,7 @@ class ConversationAssertionError(AssertionError):
413
418
  reasoning_steps: list[str]
414
419
  conversation_id: str
415
420
  response_id: str | None
421
+ detail: dict
416
422
 
417
423
 
418
424
  def evaluate_agentic_conversation(
@@ -524,9 +530,11 @@ def evaluate_agentic_conversation(
524
530
  exc.reasoning_steps = result.reasoning_steps
525
531
  exc.conversation_id = result.conversation_id
526
532
  exc.response_id = result.response_id
533
+ exc.detail = _conversation_detail(result)
527
534
  raise exc
528
535
  return AgenticEvalOutcome(
529
536
  reasoning_steps=result.reasoning_steps,
530
537
  conversation_id=result.conversation_id,
531
538
  response_id=result.response_id,
539
+ detail=_conversation_detail(result),
532
540
  )
@@ -3,11 +3,12 @@
3
3
 
4
4
  from __future__ import annotations
5
5
 
6
- from dataclasses import dataclass
6
+ from dataclasses import dataclass, field
7
7
 
8
8
  from gooddata_eval.core.chat.sse_client import ChatClient
9
9
  from gooddata_eval.core.config import ReasoningEffort
10
10
  from gooddata_eval.core.evaluators._llm_judge import LLMJudge
11
+ from gooddata_eval.core.models import AgenticEvalOutcome
11
12
 
12
13
  _DEFAULT_K = 1
13
14
 
@@ -52,6 +53,8 @@ class GeneralQuestionResult:
52
53
  passed: bool
53
54
  llm_judge_score: float
54
55
  reasoning: str
56
+ reasoning_steps: list[str] = field(default_factory=list)
57
+ response_id: str | None = None
55
58
 
56
59
 
57
60
  @dataclass
@@ -98,6 +101,8 @@ def run_agentic_general_question(
98
101
  passed=passed,
99
102
  llm_judge_score=llm_judge_score,
100
103
  reasoning=reasoning,
104
+ reasoning_steps=list(chat_result.reasoning_steps or []),
105
+ response_id=chat_result.response_id,
101
106
  )
102
107
  )
103
108
  finally:
@@ -120,6 +125,8 @@ def run_agentic_general_question(
120
125
  passed=passed,
121
126
  llm_judge_score=llm_judge_score,
122
127
  reasoning=reasoning,
128
+ reasoning_steps=list(chat_result.reasoning_steps or []),
129
+ response_id=chat_result.response_id,
123
130
  )
124
131
  )
125
132
  finally:
@@ -142,6 +149,10 @@ class GeneralQuestionAssertionError(AssertionError):
142
149
  """Raised when a general-question evaluation fails."""
143
150
 
144
151
  __tracebackhide__ = True
152
+ reasoning_steps: list[str]
153
+ conversation_id: str
154
+ response_id: str | None
155
+ detail: dict
145
156
 
146
157
 
147
158
  def evaluate_agentic_general_question(
@@ -160,8 +171,13 @@ def evaluate_agentic_general_question(
160
171
  model_version_override: str | None = None,
161
172
  run_metadata_extra: dict | None = None,
162
173
  reasoning_effort: ReasoningEffort | None = None,
163
- ) -> None:
164
- """Run general-question evaluation, log to Langfuse, and raise on failure."""
174
+ ) -> AgenticEvalOutcome:
175
+ """Run general-question evaluation, log to Langfuse, and raise GeneralQuestionAssertionError on failure.
176
+
177
+ Returns the best run's outcome (reasoning_steps, conversation_id, response_id) as an
178
+ AgenticEvalOutcome on success; on failure the same three values are attached to the
179
+ raised exception as ``.reasoning_steps``/``.conversation_id``/``.response_id``.
180
+ """
165
181
  from datetime import datetime as _dt # noqa: PLC0415
166
182
  from datetime import timezone as _tz # noqa: PLC0415
167
183
 
@@ -223,6 +239,26 @@ def evaluate_agentic_general_question(
223
239
 
224
240
  if not summary.pass_at_k:
225
241
  best = summary.best
226
- raise GeneralQuestionAssertionError(
242
+ exc = GeneralQuestionAssertionError(
227
243
  f"General question assertion failed. passed={best.passed}. Reasoning: {best.reasoning}"
228
244
  )
245
+ exc.reasoning_steps = best.reasoning_steps
246
+ exc.conversation_id = best.conversation_id
247
+ exc.response_id = best.response_id
248
+ exc.detail = {
249
+ "judge_passed": best.passed,
250
+ "judge_reasoning": best.reasoning,
251
+ "actual_output": best.actual_output,
252
+ }
253
+ raise exc
254
+ best = summary.best
255
+ return AgenticEvalOutcome(
256
+ reasoning_steps=best.reasoning_steps,
257
+ conversation_id=best.conversation_id,
258
+ response_id=best.response_id,
259
+ detail={
260
+ "judge_passed": best.passed,
261
+ "judge_reasoning": best.reasoning,
262
+ "actual_output": best.actual_output,
263
+ },
264
+ )
@@ -3,11 +3,12 @@
3
3
 
4
4
  from __future__ import annotations
5
5
 
6
- from dataclasses import dataclass
6
+ from dataclasses import dataclass, field
7
7
 
8
8
  from gooddata_eval.core.chat.sse_client import ChatClient
9
9
  from gooddata_eval.core.config import ReasoningEffort
10
10
  from gooddata_eval.core.evaluators._llm_judge import LLMJudge
11
+ from gooddata_eval.core.models import AgenticEvalOutcome
11
12
 
12
13
  _DEFAULT_K = 1
13
14
 
@@ -49,6 +50,8 @@ class GuardrailResult:
49
50
  passed: bool
50
51
  llm_judge_score: float
51
52
  reasoning: str
53
+ reasoning_steps: list[str] = field(default_factory=list)
54
+ response_id: str | None = None
52
55
 
53
56
 
54
57
  @dataclass
@@ -95,6 +98,8 @@ def run_agentic_guardrail(
95
98
  passed=passed,
96
99
  llm_judge_score=llm_judge_score,
97
100
  reasoning=reasoning,
101
+ reasoning_steps=list(chat_result.reasoning_steps or []),
102
+ response_id=chat_result.response_id,
98
103
  )
99
104
  )
100
105
  finally:
@@ -117,6 +122,8 @@ def run_agentic_guardrail(
117
122
  passed=passed,
118
123
  llm_judge_score=llm_judge_score,
119
124
  reasoning=reasoning,
125
+ reasoning_steps=list(chat_result.reasoning_steps or []),
126
+ response_id=chat_result.response_id,
120
127
  )
121
128
  )
122
129
  finally:
@@ -139,6 +146,10 @@ class GuardrailAssertionError(AssertionError):
139
146
  """Raised when a guardrail evaluation fails."""
140
147
 
141
148
  __tracebackhide__ = True
149
+ reasoning_steps: list[str]
150
+ conversation_id: str
151
+ response_id: str | None
152
+ detail: dict
142
153
 
143
154
 
144
155
  def evaluate_agentic_guardrail(
@@ -157,8 +168,14 @@ def evaluate_agentic_guardrail(
157
168
  model_version_override: str | None = None,
158
169
  run_metadata_extra: dict | None = None,
159
170
  reasoning_effort: ReasoningEffort | None = None,
160
- ) -> None:
161
- """Run guardrail evaluation, log to Langfuse, and raise on failure."""
171
+ ) -> AgenticEvalOutcome:
172
+ """Run guardrail evaluation, log to Langfuse, and raise GuardrailAssertionError on failure.
173
+
174
+ Returns the best run's outcome (reasoning_steps, conversation_id, response_id) as an
175
+ AgenticEvalOutcome on success; on failure the same three values are attached to the
176
+ raised exception as ``.reasoning_steps``/``.conversation_id``/``.response_id`` (mirrors
177
+ `evaluate_agentic_metric_skill`'s idiom) so callers can retrieve them either way.
178
+ """
162
179
  from datetime import datetime as _dt # noqa: PLC0415
163
180
  from datetime import timezone as _tz # noqa: PLC0415
164
181
 
@@ -220,4 +237,24 @@ def evaluate_agentic_guardrail(
220
237
 
221
238
  if not summary.pass_at_k:
222
239
  best = summary.best
223
- raise GuardrailAssertionError(f"Guardrail assertion failed. passed={best.passed}. Reasoning: {best.reasoning}")
240
+ exc = GuardrailAssertionError(f"Guardrail assertion failed. passed={best.passed}. Reasoning: {best.reasoning}")
241
+ exc.reasoning_steps = best.reasoning_steps
242
+ exc.conversation_id = best.conversation_id
243
+ exc.response_id = best.response_id
244
+ exc.detail = {
245
+ "judge_passed": best.passed,
246
+ "judge_reasoning": best.reasoning,
247
+ "actual_output": best.actual_output,
248
+ }
249
+ raise exc
250
+ best = summary.best
251
+ return AgenticEvalOutcome(
252
+ reasoning_steps=best.reasoning_steps,
253
+ conversation_id=best.conversation_id,
254
+ response_id=best.response_id,
255
+ detail={
256
+ "judge_passed": best.passed,
257
+ "judge_reasoning": best.reasoning,
258
+ "actual_output": best.actual_output,
259
+ },
260
+ )
@@ -5,11 +5,11 @@ from __future__ import annotations
5
5
 
6
6
  import logging
7
7
  import os
8
- from dataclasses import dataclass
8
+ from dataclasses import dataclass, field
9
9
 
10
10
  from gooddata_eval.core.chat.sse_client import ChatClient
11
11
  from gooddata_eval.core.config import ReasoningEffort
12
- from gooddata_eval.core.models import ToolCallEvent
12
+ from gooddata_eval.core.models import AgenticEvalOutcome, ToolCallEvent
13
13
 
14
14
  _log = logging.getLogger(__name__)
15
15
 
@@ -159,6 +159,8 @@ class KdaRunResult:
159
159
  # Wall-clock time of the turn that called create (None if create never happened) --
160
160
  # not any earlier disambiguation turn. See run_agentic_kda_skill's _run_once.
161
161
  turn_wall_clock_sec: float | None = None
162
+ reasoning_steps: list[str] = field(default_factory=list)
163
+ response_id: str | None = None
162
164
 
163
165
 
164
166
  @dataclass
@@ -199,6 +201,7 @@ def run_agentic_kda_skill(
199
201
  max_iterations: int = _DEFAULT_MAX_ITERATIONS,
200
202
  initial_conversation_id: str | None = None,
201
203
  reasoning_effort: ReasoningEffort | None = None,
204
+ agent_id: str | None = None,
202
205
  ) -> AgenticKdaSummary:
203
206
  """Run the KDA-skill agentic evaluation K times and return a summary.
204
207
 
@@ -215,7 +218,9 @@ def run_agentic_kda_skill(
215
218
  # k=0 or negative would otherwise silently run once, indistinguishable from k=1.
216
219
  raise ValueError(f"k must be >= 1, got {k}")
217
220
  run_results: list[KdaRunResult] = []
218
- client = ChatClient(host=host, token=token, workspace_id=workspace_id, reasoning_effort=reasoning_effort)
221
+ client = ChatClient(
222
+ host=host, token=token, workspace_id=workspace_id, reasoning_effort=reasoning_effort, agent_id=agent_id
223
+ )
219
224
 
220
225
  def _run_once(conv_id: str) -> KdaRunResult:
221
226
  create_args: dict | None = None
@@ -224,6 +229,8 @@ def run_agentic_kda_skill(
224
229
  turn_completed = False
225
230
  disambiguated = False
226
231
  current_question = question
232
+ reasoning_steps: list[str] = []
233
+ response_id: str | None = None
227
234
 
228
235
  for iteration in range(max_iterations):
229
236
  try:
@@ -232,11 +239,15 @@ def run_agentic_kda_skill(
232
239
  _log.warning("KDA send_message failed for conversation %s: %s", conv_id, exc)
233
240
  partial = getattr(exc, "partial_result", None)
234
241
  if partial is not None:
242
+ reasoning_steps.extend(partial.reasoning_steps or [])
243
+ response_id = partial.response_id or response_id
235
244
  create_args, execute_result = _extract_kda_calls(partial.tool_call_events or [])
236
245
  if create_args is not None:
237
246
  turn_wall_clock_sec = partial.turn_wall_clock_sec
238
247
  turn_completed = False
239
248
  break
249
+ reasoning_steps.extend(chat_result.reasoning_steps or [])
250
+ response_id = chat_result.response_id or response_id
240
251
  create_args, execute_result = _extract_kda_calls(chat_result.tool_call_events or [])
241
252
  response_text = (chat_result.text_response or "").strip()
242
253
  turn_completed = chat_result.stream_ended and bool(response_text)
@@ -273,6 +284,8 @@ def run_agentic_kda_skill(
273
284
  actual_create_args=create_args,
274
285
  actual_execute_result=execute_result,
275
286
  turn_wall_clock_sec=turn_wall_clock_sec,
287
+ reasoning_steps=reasoning_steps,
288
+ response_id=response_id,
276
289
  )
277
290
 
278
291
  try:
@@ -312,6 +325,10 @@ class KdaSkillAssertionError(AssertionError):
312
325
  """Raised when a KDA-skill evaluation fails."""
313
326
 
314
327
  __tracebackhide__ = True
328
+ reasoning_steps: list[str]
329
+ conversation_id: str
330
+ response_id: str | None
331
+ detail: dict
315
332
 
316
333
 
317
334
  def evaluate_agentic_kda_skill(
@@ -323,6 +340,7 @@ def evaluate_agentic_kda_skill(
323
340
  k: int = _DEFAULT_K,
324
341
  max_iterations: int = _DEFAULT_MAX_ITERATIONS,
325
342
  initial_conversation_id: str | None = None,
343
+ agent_id: str | None = None,
326
344
  langfuse: object | None = None,
327
345
  dataset_item_id: str = "",
328
346
  dataset_name: str = "kda_skill",
@@ -330,8 +348,13 @@ def evaluate_agentic_kda_skill(
330
348
  model_version_override: str | None = None,
331
349
  run_metadata_extra: dict | None = None,
332
350
  reasoning_effort: ReasoningEffort | None = None,
333
- ) -> None:
334
- """Run KDA-skill evaluation, log to Langfuse, and raise KdaSkillAssertionError on failure."""
351
+ ) -> AgenticEvalOutcome:
352
+ """Run KDA-skill evaluation, log to Langfuse, and raise KdaSkillAssertionError on failure.
353
+
354
+ Returns the best run's outcome (reasoning_steps, conversation_id, response_id) as an
355
+ AgenticEvalOutcome on success; on failure the same three values are attached to the
356
+ raised exception as ``.reasoning_steps``/``.conversation_id``/``.response_id``.
357
+ """
335
358
  from datetime import datetime as _dt # noqa: PLC0415
336
359
  from datetime import timezone as _tz # noqa: PLC0415
337
360
 
@@ -350,6 +373,7 @@ def evaluate_agentic_kda_skill(
350
373
  max_iterations=max_iterations,
351
374
  initial_conversation_id=initial_conversation_id,
352
375
  reasoning_effort=reasoning_effort,
376
+ agent_id=agent_id,
353
377
  )
354
378
 
355
379
  if langfuse is not None and dataset_item_id:
@@ -420,4 +444,33 @@ def evaluate_agentic_kda_skill(
420
444
  f"Actual create args: {best.actual_create_args}. "
421
445
  f"Actual execute result: {best.actual_execute_result}."
422
446
  )
423
- raise KdaSkillAssertionError(message)
447
+ exc = KdaSkillAssertionError(message)
448
+ exc.reasoning_steps = best.reasoning_steps
449
+ exc.conversation_id = best.conversation_id
450
+ exc.response_id = best.response_id
451
+ exc.detail = {
452
+ "triggered": ev.triggered,
453
+ "executed": ev.executed,
454
+ "success": ev.success,
455
+ "turn_completed": ev.turn_completed,
456
+ "disambiguated": ev.disambiguated,
457
+ "actual_create_args": best.actual_create_args,
458
+ "actual_execute_result": best.actual_execute_result,
459
+ }
460
+ raise exc
461
+ best = summary.best
462
+ ev = best.evaluation
463
+ return AgenticEvalOutcome(
464
+ reasoning_steps=best.reasoning_steps,
465
+ conversation_id=best.conversation_id,
466
+ response_id=best.response_id,
467
+ detail={
468
+ "triggered": ev.triggered,
469
+ "executed": ev.executed,
470
+ "success": ev.success,
471
+ "turn_completed": ev.turn_completed,
472
+ "disambiguated": ev.disambiguated,
473
+ "actual_create_args": best.actual_create_args,
474
+ "actual_execute_result": best.actual_execute_result,
475
+ },
476
+ )