gooddata-eval 1.72.1.dev6__tar.gz → 1.73.1.dev1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (93) hide show
  1. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/PKG-INFO +37 -2
  2. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/README.md +35 -0
  3. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/pyproject.toml +2 -2
  4. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/cli/agentic_runner.py +44 -12
  5. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/cli/main.py +14 -0
  6. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/agentic/alert_skill.py +106 -9
  7. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/agentic/conversation.py +64 -23
  8. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/agentic/general_question.py +46 -5
  9. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/agentic/guardrail.py +47 -5
  10. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/agentic/kda_skill.py +59 -6
  11. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/agentic/metric_skill.py +103 -20
  12. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/agentic/search_tool.py +46 -6
  13. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/agentic/visualization.py +56 -6
  14. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/chat/sse_client.py +19 -2
  15. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/config.py +1 -0
  16. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/evaluators/visualization.py +40 -17
  17. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/models.py +17 -0
  18. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/reporting/json_report.py +1 -0
  19. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/runner.py +2 -0
  20. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/scoring.py +13 -0
  21. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/tests/test_agentic_alert_skill.py +230 -0
  22. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/tests/test_agentic_conversation.py +210 -0
  23. gooddata_eval-1.73.1.dev1/tests/test_agentic_general_question.py +210 -0
  24. gooddata_eval-1.73.1.dev1/tests/test_agentic_guardrail.py +208 -0
  25. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/tests/test_agentic_kda_skill.py +147 -0
  26. gooddata_eval-1.73.1.dev1/tests/test_agentic_metric_skill.py +564 -0
  27. gooddata_eval-1.73.1.dev1/tests/test_agentic_runner.py +224 -0
  28. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/tests/test_agentic_search_tool.py +96 -0
  29. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/tests/test_agentic_visualization.py +130 -0
  30. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/tests/test_cli.py +110 -0
  31. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/tests/test_reporting.py +3 -0
  32. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/tests/test_runner.py +35 -0
  33. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/tests/test_scoring.py +44 -0
  34. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/tests/test_sse_client.py +76 -0
  35. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/tests/test_visualization_evaluator.py +43 -0
  36. gooddata_eval-1.72.1.dev6/tests/test_agentic_general_question.py +0 -100
  37. gooddata_eval-1.72.1.dev6/tests/test_agentic_guardrail.py +0 -98
  38. gooddata_eval-1.72.1.dev6/tests/test_agentic_metric_skill.py +0 -286
  39. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/.gitignore +0 -0
  40. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/LICENSE.txt +0 -0
  41. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/Makefile +0 -0
  42. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/__init__.py +0 -0
  43. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/_version.py +0 -0
  44. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/cli/__init__.py +0 -0
  45. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/__init__.py +0 -0
  46. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/agentic/__init__.py +0 -0
  47. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/agentic/_catalog.py +0 -0
  48. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/agentic/_langfuse.py +0 -0
  49. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/chat/__init__.py +0 -0
  50. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/connection.py +0 -0
  51. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/dataset/__init__.py +0 -0
  52. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/dataset/langfuse_source.py +0 -0
  53. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/dataset/local.py +0 -0
  54. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/evaluators/__init__.py +0 -0
  55. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/evaluators/_deep_subset.py +0 -0
  56. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/evaluators/_llm_judge.py +0 -0
  57. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/evaluators/_text_utils.py +0 -0
  58. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/evaluators/alert_skill.py +0 -0
  59. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/evaluators/base.py +0 -0
  60. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/evaluators/general_question.py +0 -0
  61. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/evaluators/guardrail.py +0 -0
  62. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/evaluators/metric_skill.py +0 -0
  63. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/evaluators/search_tool.py +0 -0
  64. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/evaluators/summary.py +0 -0
  65. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/langfuse/__init__.py +0 -0
  66. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/langfuse/sink.py +0 -0
  67. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/reporting/__init__.py +0 -0
  68. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/reporting/console.py +0 -0
  69. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/summary/__init__.py +0 -0
  70. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/summary/http_client.py +0 -0
  71. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/workspace.py +0 -0
  72. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/tests/__init__.py +0 -0
  73. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/tests/conftest.py +0 -0
  74. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/tests/fixtures/sample_dataset/metric_skill_create.json +0 -0
  75. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/tests/fixtures/sample_dataset/visualization_revenue.json +0 -0
  76. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/tests/fixtures/sse_visualization_stream.txt +0 -0
  77. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/tests/test_agentic_langfuse_trace.py +0 -0
  78. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/tests/test_agentic_run_context.py +0 -0
  79. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/tests/test_alert_skill_evaluator.py +0 -0
  80. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/tests/test_connection.py +0 -0
  81. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/tests/test_deep_subset.py +0 -0
  82. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/tests/test_langfuse_sink.py +0 -0
  83. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/tests/test_langfuse_source.py +0 -0
  84. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/tests/test_llm_judge.py +0 -0
  85. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/tests/test_local_loader.py +0 -0
  86. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/tests/test_metric_skill_evaluator.py +0 -0
  87. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/tests/test_models.py +0 -0
  88. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/tests/test_search_tool_evaluator.py +0 -0
  89. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/tests/test_summary_client.py +0 -0
  90. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/tests/test_summary_evaluator.py +0 -0
  91. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/tests/test_text_evaluators.py +0 -0
  92. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/tests/test_workspace.py +0 -0
  93. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/tox.ini +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: gooddata-eval
3
- Version: 1.72.1.dev6
3
+ Version: 1.73.1.dev1
4
4
  Summary: Evaluate the GoodData AI agent against your own questions and models.
5
5
  Project-URL: Source, https://github.com/gooddata/gooddata-python-sdk
6
6
  Author-email: GoodData <support@gooddata.com>
@@ -17,7 +17,7 @@ Classifier: Topic :: Scientific/Engineering
17
17
  Classifier: Topic :: Software Development
18
18
  Classifier: Typing :: Typed
19
19
  Requires-Python: >=3.10
20
- Requires-Dist: gooddata-sdk~=1.72.1.dev6
20
+ Requires-Dist: gooddata-sdk~=1.73.1.dev1
21
21
  Requires-Dist: httpx<1.0,>=0.27
22
22
  Requires-Dist: orjson<4.0.0,>=3.9.15
23
23
  Requires-Dist: pydantic<3.0,>=2.6
@@ -90,6 +90,40 @@ When the same model id is offered by multiple providers, use the
90
90
 
91
91
  Both provider name and provider id are accepted as the prefix.
92
92
 
93
+ ### Targeting a specific AI Hub agent
94
+
95
+ GoodData has no admin-settable "default agent": when a conversation doesn't
96
+ name one, the platform picks whichever agent was last used or last edited in
97
+ that workspace. If your org has several AI Hub agents configured (e.g. one
98
+ scoped to visualization only, another with every skill enabled), evaluating
99
+ without `--agent-id` can silently exercise the wrong one — a
100
+ `metric_skill`/`alert_skill` item run against a visualization-only agent will
101
+ never pass, no matter how well-formed the question is.
102
+
103
+ ```bash
104
+ export GD_EVAL_AGENT_ID='eval-all-skills'
105
+
106
+ gd-eval run \
107
+ --host https://your.gooddata.cloud \
108
+ --workspace ecommerce_demo \
109
+ --dataset ./my-dataset \
110
+ --model gpt-5.2 \
111
+ --runs 1 \
112
+ --json results.json
113
+ ```
114
+
115
+ Or pass it explicitly instead of via the env var:
116
+
117
+ ```bash
118
+ gd-eval run \
119
+ --host https://your.gooddata.cloud \
120
+ --workspace ecommerce_demo \
121
+ --dataset ./my-dataset \
122
+ --agent-id eval-all-skills \
123
+ --model gpt-5.2 \
124
+ --runs 1
125
+ ```
126
+
93
127
  ### All flags
94
128
 
95
129
  #### Connection
@@ -100,6 +134,7 @@ Both provider name and provider id are accepted as the prefix.
100
134
  | `--token TOKEN` | `GOODDATA_TOKEN` | API token. Pass via flag or env var. |
101
135
  | `--profile NAME` | — | Profile name in `~/.gooddata/profiles.yaml` (same file as the `gdc` CLI). |
102
136
  | `--workspace ID` | — | **Required.** Workspace id to evaluate against. |
137
+ | `--agent-id ID` | `GD_EVAL_AGENT_ID` | AI Hub agent every conversation should target. GoodData has no admin-settable default agent — without this, each conversation falls back to whichever agent the platform's last-used/last-edited heuristic resolves, which may not have every skill under test enabled. |
103
138
 
104
139
  #### Dataset source (pick one)
105
140
 
@@ -62,6 +62,40 @@ When the same model id is offered by multiple providers, use the
62
62
 
63
63
  Both provider name and provider id are accepted as the prefix.
64
64
 
65
+ ### Targeting a specific AI Hub agent
66
+
67
+ GoodData has no admin-settable "default agent": when a conversation doesn't
68
+ name one, the platform picks whichever agent was last used or last edited in
69
+ that workspace. If your org has several AI Hub agents configured (e.g. one
70
+ scoped to visualization only, another with every skill enabled), evaluating
71
+ without `--agent-id` can silently exercise the wrong one — a
72
+ `metric_skill`/`alert_skill` item run against a visualization-only agent will
73
+ never pass, no matter how well-formed the question is.
74
+
75
+ ```bash
76
+ export GD_EVAL_AGENT_ID='eval-all-skills'
77
+
78
+ gd-eval run \
79
+ --host https://your.gooddata.cloud \
80
+ --workspace ecommerce_demo \
81
+ --dataset ./my-dataset \
82
+ --model gpt-5.2 \
83
+ --runs 1 \
84
+ --json results.json
85
+ ```
86
+
87
+ Or pass it explicitly instead of via the env var:
88
+
89
+ ```bash
90
+ gd-eval run \
91
+ --host https://your.gooddata.cloud \
92
+ --workspace ecommerce_demo \
93
+ --dataset ./my-dataset \
94
+ --agent-id eval-all-skills \
95
+ --model gpt-5.2 \
96
+ --runs 1
97
+ ```
98
+
65
99
  ### All flags
66
100
 
67
101
  #### Connection
@@ -72,6 +106,7 @@ Both provider name and provider id are accepted as the prefix.
72
106
  | `--token TOKEN` | `GOODDATA_TOKEN` | API token. Pass via flag or env var. |
73
107
  | `--profile NAME` | — | Profile name in `~/.gooddata/profiles.yaml` (same file as the `gdc` CLI). |
74
108
  | `--workspace ID` | — | **Required.** Workspace id to evaluate against. |
109
+ | `--agent-id ID` | `GD_EVAL_AGENT_ID` | AI Hub agent every conversation should target. GoodData has no admin-settable default agent — without this, each conversation falls back to whichever agent the platform's last-used/last-edited heuristic resolves, which may not have every skill under test enabled. |
75
110
 
76
111
  #### Dataset source (pick one)
77
112
 
@@ -1,7 +1,7 @@
1
1
  # (C) 2026 GoodData Corporation
2
2
  [project]
3
3
  name = "gooddata-eval"
4
- version = "1.72.1.dev6"
4
+ version = "1.73.1.dev1"
5
5
  description = "Evaluate the GoodData AI agent against your own questions and models."
6
6
  readme = "README.md"
7
7
  license = "MIT"
@@ -11,7 +11,7 @@ authors = [
11
11
  keywords = ["gooddata", "ai", "evaluation", "llm", "analytics", "cli"]
12
12
  requires-python = ">=3.10"
13
13
  dependencies = [
14
- "gooddata-sdk~=1.72.1.dev6",
14
+ "gooddata-sdk~=1.73.1.dev1",
15
15
  "httpx>=0.27,<1.0",
16
16
  "orjson>=3.9.15,<4.0.0",
17
17
  "pydantic>=2.6,<3.0",
@@ -16,7 +16,7 @@ from gooddata_eval.core.agentic.metric_skill import evaluate_agentic_metric_skil
16
16
  from gooddata_eval.core.agentic.search_tool import evaluate_agentic_search_tool
17
17
  from gooddata_eval.core.agentic.visualization import evaluate_agentic_visualization
18
18
  from gooddata_eval.core.config import ReasoningEffort
19
- from gooddata_eval.core.models import CreatedVisualization, DatasetItem
19
+ from gooddata_eval.core.models import AgenticEvalOutcome, CreatedVisualization, DatasetItem
20
20
  from gooddata_eval.core.runner import EvalReport, ItemReport
21
21
 
22
22
 
@@ -85,8 +85,14 @@ def _dispatch_agentic(
85
85
  run_ts: str,
86
86
  model_version_override: str | None,
87
87
  reasoning_effort: ReasoningEffort | None = None,
88
- ) -> None:
89
- """Call the appropriate evaluate_agentic_* function for the item's test_kind."""
88
+ agent_id: str | None = None,
89
+ ) -> AgenticEvalOutcome:
90
+ """Call the appropriate evaluate_agentic_* function for the item's test_kind.
91
+
92
+ Every evaluate_agentic_* function returns an AgenticEvalOutcome (reasoning_steps,
93
+ conversation_id, response_id, detail) on success and attaches the same four attributes
94
+ to its raised *AssertionError on failure -- no kind is exempt.
95
+ """
90
96
  kind = item.test_kind
91
97
  eo = item.expected_output
92
98
  lf_kw: _LfKw = {
@@ -99,85 +105,93 @@ def _dispatch_agentic(
99
105
  }
100
106
 
101
107
  if kind in ("vis_agentic", "agentic_visualization"):
102
- evaluate_agentic_visualization(
108
+ return evaluate_agentic_visualization(
103
109
  host=host,
104
110
  token=token,
105
111
  workspace_id=workspace_id,
106
112
  question=item.question,
107
113
  expected_outputs=_parse_visualization_expected(eo),
108
114
  k=k,
115
+ agent_id=agent_id,
109
116
  **lf_kw,
110
117
  )
111
118
  elif kind == "agentic_metric_skill":
112
- evaluate_agentic_metric_skill(
119
+ return evaluate_agentic_metric_skill(
113
120
  host=host,
114
121
  token=token,
115
122
  workspace_id=workspace_id,
116
123
  question=item.question,
117
124
  expected_output=eo if isinstance(eo, (dict, list)) else {},
118
125
  k=k,
126
+ agent_id=agent_id,
119
127
  **lf_kw,
120
128
  )
121
129
  elif kind == "agentic_alert_skill":
122
- evaluate_agentic_alert_skill(
130
+ return evaluate_agentic_alert_skill(
123
131
  host=host,
124
132
  token=token,
125
133
  workspace_id=workspace_id,
126
134
  question=item.question,
127
135
  expected_output=eo if isinstance(eo, dict) else {},
128
136
  k=k,
137
+ agent_id=agent_id,
129
138
  **lf_kw,
130
139
  )
131
140
  elif kind == "agentic_search":
132
141
  eo_dict = eo if isinstance(eo, dict) else {}
133
142
  tool_call = eo_dict.get("tool_call", {})
134
143
  expected_args = tool_call.get("function_arguments", eo_dict)
135
- evaluate_agentic_search_tool(
144
+ return evaluate_agentic_search_tool(
136
145
  host=host,
137
146
  token=token,
138
147
  workspace_id=workspace_id,
139
148
  question=item.question,
140
149
  expected_tool_call=expected_args,
141
150
  k=k,
151
+ agent_id=agent_id,
142
152
  **lf_kw,
143
153
  )
144
154
  elif kind == "agentic_general_question":
145
- evaluate_agentic_general_question(
155
+ return evaluate_agentic_general_question(
146
156
  host=host,
147
157
  token=token,
148
158
  workspace_id=workspace_id,
149
159
  question=item.question,
150
160
  expected_output=eo if isinstance(eo, str) else str(eo),
151
161
  k=k,
162
+ agent_id=agent_id,
152
163
  **lf_kw,
153
164
  )
154
165
  elif kind == "agentic_guardrail":
155
- evaluate_agentic_guardrail(
166
+ return evaluate_agentic_guardrail(
156
167
  host=host,
157
168
  token=token,
158
169
  workspace_id=workspace_id,
159
170
  question=item.question,
160
171
  expected_output=eo if isinstance(eo, str) else str(eo),
161
172
  k=k,
173
+ agent_id=agent_id,
162
174
  **lf_kw,
163
175
  )
164
176
  elif kind == "agentic_kda_skill":
165
- evaluate_agentic_kda_skill(
177
+ return evaluate_agentic_kda_skill(
166
178
  host=host,
167
179
  token=token,
168
180
  workspace_id=workspace_id,
169
181
  question=item.question,
170
182
  expected_output=eo if isinstance(eo, dict) else {},
171
183
  k=k,
184
+ agent_id=agent_id,
172
185
  **lf_kw,
173
186
  )
174
187
  elif kind == "agentic_conversation":
175
188
  fixture_data = eo.get("fixture") or eo if isinstance(eo, dict) else {}
176
- evaluate_agentic_conversation(
189
+ return evaluate_agentic_conversation(
177
190
  host=host,
178
191
  token=token,
179
192
  workspace_id=workspace_id,
180
193
  fixture=ConversationFixture.model_validate(fixture_data),
194
+ agent_id=agent_id,
181
195
  **lf_kw,
182
196
  )
183
197
  else:
@@ -197,6 +211,7 @@ def run_agentic_items(
197
211
  run_ts: str,
198
212
  on_item_start: Any = None,
199
213
  on_item_done: Any = None,
214
+ agent_id: str | None = None,
200
215
  ) -> EvalReport:
201
216
  """Run agentic items through evaluate_agentic_* and return an EvalReport."""
202
217
  langfuse = make_langfuse_client() if use_langfuse else None
@@ -219,12 +234,29 @@ def run_agentic_items(
219
234
  )
220
235
  t0 = time.perf_counter()
221
236
  try:
222
- _dispatch_agentic(item, host, token, workspace_id, k, langfuse, run_ts, model_version, reasoning_effort)
237
+ outcome = _dispatch_agentic(
238
+ item, host, token, workspace_id, k, langfuse, run_ts, model_version, reasoning_effort, agent_id
239
+ )
240
+ if isinstance(outcome, AgenticEvalOutcome):
241
+ reasoning_steps = outcome.reasoning_steps
242
+ conversation_id = outcome.conversation_id
243
+ response_id = outcome.response_id
244
+ detail = outcome.detail
245
+ else:
246
+ reasoning_steps, conversation_id, response_id, detail = outcome, None, None, {}
223
247
  item_report.pass_at_k = True
224
248
  item_report.runs = k
249
+ item_report.reasoning_steps = reasoning_steps or []
250
+ item_report.conversation_id = conversation_id
251
+ item_report.response_id = response_id
252
+ item_report.best_detail = detail or {}
225
253
  except AssertionError as exc:
226
254
  item_report.pass_at_k = False
227
255
  item_report.runs = k
256
+ item_report.reasoning_steps = getattr(exc, "reasoning_steps", None) or []
257
+ item_report.conversation_id = getattr(exc, "conversation_id", None)
258
+ item_report.response_id = getattr(exc, "response_id", None)
259
+ item_report.best_detail = getattr(exc, "detail", None) or {}
228
260
  print(f"[agentic] {item.id} FAIL: {exc}", flush=True)
229
261
  except Exception as exc:
230
262
  item_report.error = f"{type(exc).__name__}: {exc}"
@@ -2,6 +2,7 @@
2
2
  """`gd-eval` command-line entry point."""
3
3
 
4
4
  import argparse
5
+ import os
5
6
  import sys
6
7
  import threading
7
8
  from datetime import datetime, timezone
@@ -117,6 +118,16 @@ def _build_parser() -> argparse.ArgumentParser:
117
118
  action="store_true",
118
119
  help="Log scores and traces to Langfuse (requires --langfuse-dataset and LANGFUSE_* env vars).",
119
120
  )
121
+ run.add_argument(
122
+ "--agent-id",
123
+ dest="agent_id",
124
+ help=(
125
+ "AI Hub agent id every conversation should target (or set GD_EVAL_AGENT_ID). "
126
+ "GoodData has no admin-settable default agent -- without this, each conversation "
127
+ "falls back to whichever agent the platform's last-used/last-edited heuristic "
128
+ "resolves, which may not have every skill under test enabled."
129
+ ),
130
+ )
120
131
  models_cmd = sub.add_parser("models", help="List LLM providers and models configured in the org.")
121
132
  models_cmd.add_argument("--host", help="GoodData host URL.")
122
133
  models_cmd.add_argument("--token", help="API token (or set GOODDATA_TOKEN).")
@@ -347,6 +358,7 @@ def _run(config: RunConfig) -> int:
347
358
  run_ts=run_ts,
348
359
  on_item_start=on_item_start,
349
360
  on_item_done=on_item_done,
361
+ agent_id=config.agent_id,
350
362
  )
351
363
 
352
364
  # --- non-agentic items (single-turn, use Evaluator) ---
@@ -357,6 +369,7 @@ def _run(config: RunConfig) -> int:
357
369
  workspace_id=config.workspace_id,
358
370
  preserve_failed=config.preserve_failed,
359
371
  reasoning_effort=config.reasoning_effort,
372
+ agent_id=config.agent_id,
360
373
  ),
361
374
  SummaryClient(host=config.host, token=config.token, workspace_id=config.workspace_id),
362
375
  )
@@ -449,6 +462,7 @@ def main(argv: list[str] | None = None) -> int:
449
462
  kind=args.kind,
450
463
  preserve_failed=args.preserve_failed,
451
464
  reasoning_effort=args.reasoning_effort,
465
+ agent_id=args.agent_id or os.environ.get("GD_EVAL_AGENT_ID"),
452
466
  )
453
467
  return _run(config)
454
468
  except (
@@ -6,7 +6,7 @@ from __future__ import annotations
6
6
  import json
7
7
  import os
8
8
  import re
9
- from dataclasses import dataclass
9
+ from dataclasses import dataclass, field
10
10
  from typing import Any
11
11
 
12
12
  from gooddata_sdk import GoodDataSdk
@@ -14,7 +14,7 @@ from gooddata_sdk import GoodDataSdk
14
14
  from gooddata_eval.core.agentic._catalog import CatalogMetricAlert
15
15
  from gooddata_eval.core.chat.sse_client import ChatClient
16
16
  from gooddata_eval.core.config import ReasoningEffort
17
- from gooddata_eval.core.models import ToolCallEvent
17
+ from gooddata_eval.core.models import AgenticEvalOutcome, ToolCallEvent
18
18
 
19
19
  try:
20
20
  from openai import OpenAI as _OpenAI
@@ -119,7 +119,31 @@ def _check_metric(expected: CatalogMetricAlert, actual_args: dict) -> bool:
119
119
  return expected.metric_id == act_metric
120
120
 
121
121
 
122
- def _check_recipients(expected: CatalogMetricAlert, actual_args: dict) -> bool:
122
+ def _resolve_internal_recipient_ids(sdk: GoodDataSdk, emails: list[str]) -> set[str]:
123
+ """Best-effort map of expected recipient emails to internal GoodData user ids.
124
+
125
+ Some notification channels are workspace-restricted to internal users --
126
+ `create_metric_alert` then addresses the alert by internal user id
127
+ (`internal_recipients`), never by email, so an expected email has to be
128
+ resolved before it can be compared against that field. Failures (no
129
+ matching user, no permission, network error) are swallowed: the caller
130
+ treats an empty result the same as "this delivery path doesn't match",
131
+ which is correct -- it doesn't mean the alert itself failed.
132
+ """
133
+ if not emails:
134
+ return set()
135
+ try:
136
+ # RSQL quoted-string escaping: backslash first, then the enclosing quote char,
137
+ # or an email like o'hara@example.com breaks the filter into invalid RSQL.
138
+ escaped = [email.replace("\\", "\\\\").replace("'", "\\'") for email in emails]
139
+ quoted = ",".join(f"'{email}'" for email in escaped)
140
+ resp = sdk._client.entities_api.get_all_entities_users(filter=f"email=in=({quoted})")
141
+ return {u.id for u in (resp.data or [])}
142
+ except Exception:
143
+ return set()
144
+
145
+
146
+ def _check_recipients(expected: CatalogMetricAlert, actual_args: dict, sdk: GoodDataSdk | None = None) -> bool:
123
147
  if not expected.recipients:
124
148
  return True
125
149
  act_recip_raw = actual_args.get("recipients", actual_args.get("external_recipients"))
@@ -134,7 +158,24 @@ def _check_recipients(expected: CatalogMetricAlert, actual_args: dict) -> bool:
134
158
  act_recip = act_recip_raw
135
159
  else:
136
160
  act_recip = []
137
- return set(expected.recipients) == set(act_recip or [])
161
+ if set(expected.recipients) == set(act_recip or []):
162
+ return True
163
+ act_internal_raw = actual_args.get("internal_recipients")
164
+ # internal_recipients is declared `anyOf: [array of string, string, null]` in the
165
+ # create_metric_alert tool schema -- a single id as a bare string is schema-legal,
166
+ # not a malformed call, so it needs the same string/list normalization already
167
+ # applied to recipients/external_recipients above.
168
+ if isinstance(act_internal_raw, str):
169
+ act_internal = [act_internal_raw]
170
+ elif isinstance(act_internal_raw, list):
171
+ act_internal = act_internal_raw
172
+ else:
173
+ act_internal = []
174
+ if sdk is not None and act_internal:
175
+ internal_recipient_ids = _resolve_internal_recipient_ids(sdk, expected.recipients)
176
+ if internal_recipient_ids & set(act_internal):
177
+ return True
178
+ return False
138
179
 
139
180
 
140
181
  def generate_simulated_alert_response(
@@ -302,6 +343,8 @@ class AlertRunResult:
302
343
  alert_id: str | None
303
344
  eval: AlertEvaluation
304
345
  actual_alert_arguments: dict
346
+ reasoning_steps: list[str] = field(default_factory=list)
347
+ response_id: str | None = None
305
348
 
306
349
 
307
350
  @dataclass
@@ -434,11 +477,14 @@ def run_agentic_alert_skill(
434
477
  max_iterations: int = _DEFAULT_MAX_ITERATIONS,
435
478
  initial_conversation_id: str | None = None,
436
479
  reasoning_effort: ReasoningEffort | None = None,
480
+ agent_id: str | None = None,
437
481
  ) -> AgenticAlertSummary:
438
482
  """Run the alert-skill agentic evaluation K times and return a summary."""
439
483
  expected = _normalize_expected_output(expected_output)
440
484
  run_results: list[AlertRunResult] = []
441
- client = ChatClient(host=host, token=token, workspace_id=workspace_id, reasoning_effort=reasoning_effort)
485
+ client = ChatClient(
486
+ host=host, token=token, workspace_id=workspace_id, reasoning_effort=reasoning_effort, agent_id=agent_id
487
+ )
442
488
  sdk = GoodDataSdk.create(host, token)
443
489
 
444
490
  def _run_once(conv_id: str) -> AlertRunResult:
@@ -447,6 +493,8 @@ def run_agentic_alert_skill(
447
493
  alert_id: str | None = None
448
494
  actual_args: dict = {}
449
495
  tool_called = False
496
+ reasoning_steps: list[str] = []
497
+ response_id: str | None = None
450
498
  # conversation_history stores prior turns for GPT-4o context.
451
499
  # Roles follow GPT-4o's perspective: "assistant"=agent text, "user"=sim-user reply.
452
500
  conversation_history: list = []
@@ -454,6 +502,8 @@ def run_agentic_alert_skill(
454
502
 
455
503
  for _iteration in range(max_iterations):
456
504
  chat_result = client.send_message(conv_id, current_question)
505
+ reasoning_steps.extend(chat_result.reasoning_steps or [])
506
+ response_id = chat_result.response_id or response_id
457
507
  alert_id, actual_args, tool_called = _extract_alert_call(chat_result.tool_call_events or [])
458
508
  if tool_called:
459
509
  alert_id_to_delete = alert_id
@@ -482,13 +532,15 @@ def run_agentic_alert_skill(
482
532
  trigger_correct=tool_called and _check_trigger(expected, actual_args),
483
533
  filters_correct=tool_called and _check_filters(expected, actual_args),
484
534
  metric_correct=tool_called and _check_metric(expected, actual_args),
485
- recipients_correct=tool_called and _check_recipients(expected, actual_args),
535
+ recipients_correct=tool_called and _check_recipients(expected, actual_args, sdk=sdk),
486
536
  )
487
537
  return AlertRunResult(
488
538
  conversation_id=conv_id,
489
539
  alert_id=alert_id,
490
540
  eval=ev,
491
541
  actual_alert_arguments=actual_args,
542
+ reasoning_steps=reasoning_steps,
543
+ response_id=response_id,
492
544
  )
493
545
  finally:
494
546
  if alert_id_to_delete:
@@ -539,6 +591,10 @@ class AlertSkillAssertionError(AssertionError):
539
591
  """Raised when an alert-skill evaluation fails."""
540
592
 
541
593
  __tracebackhide__ = True
594
+ reasoning_steps: list[str]
595
+ conversation_id: str
596
+ response_id: str | None
597
+ detail: dict
542
598
 
543
599
 
544
600
  def evaluate_agentic_alert_skill(
@@ -550,6 +606,7 @@ def evaluate_agentic_alert_skill(
550
606
  k: int = _DEFAULT_K,
551
607
  max_iterations: int = _DEFAULT_MAX_ITERATIONS,
552
608
  initial_conversation_id: str | None = None,
609
+ agent_id: str | None = None,
553
610
  langfuse: object | None = None,
554
611
  dataset_item_id: str = "",
555
612
  dataset_name: str = "alert_skill",
@@ -557,8 +614,16 @@ def evaluate_agentic_alert_skill(
557
614
  model_version_override: str | None = None,
558
615
  run_metadata_extra: dict | None = None,
559
616
  reasoning_effort: ReasoningEffort | None = None,
560
- ) -> None:
561
- """Run alert-skill evaluation, log to Langfuse, and raise AlertSkillAssertionError on failure."""
617
+ ) -> AgenticEvalOutcome:
618
+ """Run alert-skill evaluation, log to Langfuse, and raise AlertSkillAssertionError on failure.
619
+
620
+ Returns the best run's outcome (reasoning_steps, conversation_id, response_id) as an
621
+ AgenticEvalOutcome on success; on failure the same three values are attached to the
622
+ raised exception as
623
+ ``.reasoning_steps``/``.conversation_id``/``.response_id`` (mirrors the
624
+ `conversation_id`-on-exception idiom in `ChatClient.ask()`) so callers can retrieve them
625
+ either way.
626
+ """
562
627
  from datetime import datetime as _dt # noqa: PLC0415
563
628
  from datetime import timezone as _tz # noqa: PLC0415
564
629
 
@@ -577,6 +642,7 @@ def evaluate_agentic_alert_skill(
577
642
  max_iterations=max_iterations,
578
643
  initial_conversation_id=initial_conversation_id,
579
644
  reasoning_effort=reasoning_effort,
645
+ agent_id=agent_id,
580
646
  )
581
647
 
582
648
  if langfuse is not None and dataset_item_id:
@@ -631,7 +697,7 @@ def evaluate_agentic_alert_skill(
631
697
  if not summary.pass_at_k:
632
698
  best = summary.best
633
699
  ev = best.eval
634
- raise AlertSkillAssertionError(
700
+ exc = AlertSkillAssertionError(
635
701
  f"Alert skill assertion failed. strict_pass={ev.strict_pass}. "
636
702
  f"alert_created={ev.alert_created}, operator_correct={ev.operator_correct}, "
637
703
  f"threshold_correct={ev.threshold_correct}, trigger_correct={ev.trigger_correct}, "
@@ -639,3 +705,34 @@ def evaluate_agentic_alert_skill(
639
705
  f"recipients_correct={ev.recipients_correct}. "
640
706
  f"Actual args: {best.actual_alert_arguments}"
641
707
  )
708
+ exc.reasoning_steps = best.reasoning_steps
709
+ exc.conversation_id = best.conversation_id
710
+ exc.response_id = best.response_id
711
+ exc.detail = {
712
+ "alert_created": ev.alert_created,
713
+ "operator_correct": ev.operator_correct,
714
+ "threshold_correct": ev.threshold_correct,
715
+ "trigger_correct": ev.trigger_correct,
716
+ "filters_correct": ev.filters_correct,
717
+ "metric_correct": ev.metric_correct,
718
+ "recipients_correct": ev.recipients_correct,
719
+ "actual_alert_arguments": best.actual_alert_arguments,
720
+ }
721
+ raise exc
722
+ best = summary.best
723
+ ev = best.eval
724
+ return AgenticEvalOutcome(
725
+ reasoning_steps=best.reasoning_steps,
726
+ conversation_id=best.conversation_id,
727
+ response_id=best.response_id,
728
+ detail={
729
+ "alert_created": ev.alert_created,
730
+ "operator_correct": ev.operator_correct,
731
+ "threshold_correct": ev.threshold_correct,
732
+ "trigger_correct": ev.trigger_correct,
733
+ "filters_correct": ev.filters_correct,
734
+ "metric_correct": ev.metric_correct,
735
+ "recipients_correct": ev.recipients_correct,
736
+ "actual_alert_arguments": best.actual_alert_arguments,
737
+ },
738
+ )