gooddata-eval 1.72.1.dev6__tar.gz → 1.73.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (90) hide show
  1. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.0}/PKG-INFO +37 -2
  2. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.0}/README.md +35 -0
  3. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.0}/pyproject.toml +2 -2
  4. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.0}/src/gooddata_eval/cli/agentic_runner.py +39 -11
  5. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.0}/src/gooddata_eval/cli/main.py +14 -0
  6. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.0}/src/gooddata_eval/core/agentic/alert_skill.py +73 -9
  7. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.0}/src/gooddata_eval/core/agentic/conversation.py +39 -6
  8. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.0}/src/gooddata_eval/core/agentic/general_question.py +6 -1
  9. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.0}/src/gooddata_eval/core/agentic/guardrail.py +6 -1
  10. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.0}/src/gooddata_eval/core/agentic/metric_skill.py +66 -11
  11. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.0}/src/gooddata_eval/core/agentic/search_tool.py +6 -1
  12. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.0}/src/gooddata_eval/core/agentic/visualization.py +21 -1
  13. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.0}/src/gooddata_eval/core/chat/sse_client.py +5 -1
  14. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.0}/src/gooddata_eval/core/config.py +1 -0
  15. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.0}/src/gooddata_eval/core/evaluators/visualization.py +14 -0
  16. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.0}/src/gooddata_eval/core/models.py +9 -0
  17. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.0}/src/gooddata_eval/core/reporting/json_report.py +1 -0
  18. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.0}/src/gooddata_eval/core/runner.py +2 -0
  19. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.0}/src/gooddata_eval/core/scoring.py +13 -0
  20. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.0}/tests/test_agentic_alert_skill.py +197 -0
  21. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.0}/tests/test_agentic_conversation.py +133 -0
  22. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.0}/tests/test_agentic_metric_skill.py +161 -1
  23. gooddata_eval-1.73.0/tests/test_agentic_runner.py +177 -0
  24. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.0}/tests/test_cli.py +110 -0
  25. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.0}/tests/test_reporting.py +3 -0
  26. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.0}/tests/test_runner.py +35 -0
  27. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.0}/tests/test_scoring.py +44 -0
  28. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.0}/tests/test_sse_client.py +34 -0
  29. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.0}/tests/test_visualization_evaluator.py +43 -0
  30. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.0}/.gitignore +0 -0
  31. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.0}/LICENSE.txt +0 -0
  32. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.0}/Makefile +0 -0
  33. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.0}/src/gooddata_eval/__init__.py +0 -0
  34. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.0}/src/gooddata_eval/_version.py +0 -0
  35. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.0}/src/gooddata_eval/cli/__init__.py +0 -0
  36. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.0}/src/gooddata_eval/core/__init__.py +0 -0
  37. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.0}/src/gooddata_eval/core/agentic/__init__.py +0 -0
  38. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.0}/src/gooddata_eval/core/agentic/_catalog.py +0 -0
  39. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.0}/src/gooddata_eval/core/agentic/_langfuse.py +0 -0
  40. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.0}/src/gooddata_eval/core/agentic/kda_skill.py +0 -0
  41. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.0}/src/gooddata_eval/core/chat/__init__.py +0 -0
  42. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.0}/src/gooddata_eval/core/connection.py +0 -0
  43. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.0}/src/gooddata_eval/core/dataset/__init__.py +0 -0
  44. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.0}/src/gooddata_eval/core/dataset/langfuse_source.py +0 -0
  45. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.0}/src/gooddata_eval/core/dataset/local.py +0 -0
  46. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.0}/src/gooddata_eval/core/evaluators/__init__.py +0 -0
  47. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.0}/src/gooddata_eval/core/evaluators/_deep_subset.py +0 -0
  48. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.0}/src/gooddata_eval/core/evaluators/_llm_judge.py +0 -0
  49. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.0}/src/gooddata_eval/core/evaluators/_text_utils.py +0 -0
  50. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.0}/src/gooddata_eval/core/evaluators/alert_skill.py +0 -0
  51. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.0}/src/gooddata_eval/core/evaluators/base.py +0 -0
  52. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.0}/src/gooddata_eval/core/evaluators/general_question.py +0 -0
  53. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.0}/src/gooddata_eval/core/evaluators/guardrail.py +0 -0
  54. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.0}/src/gooddata_eval/core/evaluators/metric_skill.py +0 -0
  55. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.0}/src/gooddata_eval/core/evaluators/search_tool.py +0 -0
  56. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.0}/src/gooddata_eval/core/evaluators/summary.py +0 -0
  57. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.0}/src/gooddata_eval/core/langfuse/__init__.py +0 -0
  58. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.0}/src/gooddata_eval/core/langfuse/sink.py +0 -0
  59. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.0}/src/gooddata_eval/core/reporting/__init__.py +0 -0
  60. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.0}/src/gooddata_eval/core/reporting/console.py +0 -0
  61. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.0}/src/gooddata_eval/core/summary/__init__.py +0 -0
  62. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.0}/src/gooddata_eval/core/summary/http_client.py +0 -0
  63. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.0}/src/gooddata_eval/core/workspace.py +0 -0
  64. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.0}/tests/__init__.py +0 -0
  65. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.0}/tests/conftest.py +0 -0
  66. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.0}/tests/fixtures/sample_dataset/metric_skill_create.json +0 -0
  67. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.0}/tests/fixtures/sample_dataset/visualization_revenue.json +0 -0
  68. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.0}/tests/fixtures/sse_visualization_stream.txt +0 -0
  69. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.0}/tests/test_agentic_general_question.py +0 -0
  70. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.0}/tests/test_agentic_guardrail.py +0 -0
  71. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.0}/tests/test_agentic_kda_skill.py +0 -0
  72. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.0}/tests/test_agentic_langfuse_trace.py +0 -0
  73. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.0}/tests/test_agentic_run_context.py +0 -0
  74. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.0}/tests/test_agentic_search_tool.py +0 -0
  75. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.0}/tests/test_agentic_visualization.py +0 -0
  76. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.0}/tests/test_alert_skill_evaluator.py +0 -0
  77. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.0}/tests/test_connection.py +0 -0
  78. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.0}/tests/test_deep_subset.py +0 -0
  79. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.0}/tests/test_langfuse_sink.py +0 -0
  80. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.0}/tests/test_langfuse_source.py +0 -0
  81. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.0}/tests/test_llm_judge.py +0 -0
  82. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.0}/tests/test_local_loader.py +0 -0
  83. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.0}/tests/test_metric_skill_evaluator.py +0 -0
  84. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.0}/tests/test_models.py +0 -0
  85. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.0}/tests/test_search_tool_evaluator.py +0 -0
  86. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.0}/tests/test_summary_client.py +0 -0
  87. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.0}/tests/test_summary_evaluator.py +0 -0
  88. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.0}/tests/test_text_evaluators.py +0 -0
  89. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.0}/tests/test_workspace.py +0 -0
  90. {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.0}/tox.ini +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: gooddata-eval
3
- Version: 1.72.1.dev6
3
+ Version: 1.73.0
4
4
  Summary: Evaluate the GoodData AI agent against your own questions and models.
5
5
  Project-URL: Source, https://github.com/gooddata/gooddata-python-sdk
6
6
  Author-email: GoodData <support@gooddata.com>
@@ -17,7 +17,7 @@ Classifier: Topic :: Scientific/Engineering
17
17
  Classifier: Topic :: Software Development
18
18
  Classifier: Typing :: Typed
19
19
  Requires-Python: >=3.10
20
- Requires-Dist: gooddata-sdk~=1.72.1.dev6
20
+ Requires-Dist: gooddata-sdk~=1.73.0
21
21
  Requires-Dist: httpx<1.0,>=0.27
22
22
  Requires-Dist: orjson<4.0.0,>=3.9.15
23
23
  Requires-Dist: pydantic<3.0,>=2.6
@@ -90,6 +90,40 @@ When the same model id is offered by multiple providers, use the
90
90
 
91
91
  Both provider name and provider id are accepted as the prefix.
92
92
 
93
+ ### Targeting a specific AI Hub agent
94
+
95
+ GoodData has no admin-settable "default agent": when a conversation doesn't
96
+ name one, the platform picks whichever agent was last used or last edited in
97
+ that workspace. If your org has several AI Hub agents configured (e.g. one
98
+ scoped to visualization only, another with every skill enabled), evaluating
99
+ without `--agent-id` can silently exercise the wrong one — a
100
+ `metric_skill`/`alert_skill` item run against a visualization-only agent will
101
+ never pass, no matter how well-formed the question is.
102
+
103
+ ```bash
104
+ export GD_EVAL_AGENT_ID='eval-all-skills'
105
+
106
+ gd-eval run \
107
+ --host https://your.gooddata.cloud \
108
+ --workspace ecommerce_demo \
109
+ --dataset ./my-dataset \
110
+ --model gpt-5.2 \
111
+ --runs 1 \
112
+ --json results.json
113
+ ```
114
+
115
+ Or pass it explicitly instead of via the env var:
116
+
117
+ ```bash
118
+ gd-eval run \
119
+ --host https://your.gooddata.cloud \
120
+ --workspace ecommerce_demo \
121
+ --dataset ./my-dataset \
122
+ --agent-id eval-all-skills \
123
+ --model gpt-5.2 \
124
+ --runs 1
125
+ ```
126
+
93
127
  ### All flags
94
128
 
95
129
  #### Connection
@@ -100,6 +134,7 @@ Both provider name and provider id are accepted as the prefix.
100
134
  | `--token TOKEN` | `GOODDATA_TOKEN` | API token. Pass via flag or env var. |
101
135
  | `--profile NAME` | — | Profile name in `~/.gooddata/profiles.yaml` (same file as the `gdc` CLI). |
102
136
  | `--workspace ID` | — | **Required.** Workspace id to evaluate against. |
137
+ | `--agent-id ID` | `GD_EVAL_AGENT_ID` | AI Hub agent every conversation should target. GoodData has no admin-settable default agent — without this, each conversation falls back to whichever agent the platform's last-used/last-edited heuristic resolves, which may not have every skill under test enabled. |
103
138
 
104
139
  #### Dataset source (pick one)
105
140
 
@@ -62,6 +62,40 @@ When the same model id is offered by multiple providers, use the
62
62
 
63
63
  Both provider name and provider id are accepted as the prefix.
64
64
 
65
+ ### Targeting a specific AI Hub agent
66
+
67
+ GoodData has no admin-settable "default agent": when a conversation doesn't
68
+ name one, the platform picks whichever agent was last used or last edited in
69
+ that workspace. If your org has several AI Hub agents configured (e.g. one
70
+ scoped to visualization only, another with every skill enabled), evaluating
71
+ without `--agent-id` can silently exercise the wrong one — a
72
+ `metric_skill`/`alert_skill` item run against a visualization-only agent will
73
+ never pass, no matter how well-formed the question is.
74
+
75
+ ```bash
76
+ export GD_EVAL_AGENT_ID='eval-all-skills'
77
+
78
+ gd-eval run \
79
+ --host https://your.gooddata.cloud \
80
+ --workspace ecommerce_demo \
81
+ --dataset ./my-dataset \
82
+ --model gpt-5.2 \
83
+ --runs 1 \
84
+ --json results.json
85
+ ```
86
+
87
+ Or pass it explicitly instead of via the env var:
88
+
89
+ ```bash
90
+ gd-eval run \
91
+ --host https://your.gooddata.cloud \
92
+ --workspace ecommerce_demo \
93
+ --dataset ./my-dataset \
94
+ --agent-id eval-all-skills \
95
+ --model gpt-5.2 \
96
+ --runs 1
97
+ ```
98
+
65
99
  ### All flags
66
100
 
67
101
  #### Connection
@@ -72,6 +106,7 @@ Both provider name and provider id are accepted as the prefix.
72
106
  | `--token TOKEN` | `GOODDATA_TOKEN` | API token. Pass via flag or env var. |
73
107
  | `--profile NAME` | — | Profile name in `~/.gooddata/profiles.yaml` (same file as the `gdc` CLI). |
74
108
  | `--workspace ID` | — | **Required.** Workspace id to evaluate against. |
109
+ | `--agent-id ID` | `GD_EVAL_AGENT_ID` | AI Hub agent every conversation should target. GoodData has no admin-settable default agent — without this, each conversation falls back to whichever agent the platform's last-used/last-edited heuristic resolves, which may not have every skill under test enabled. |
75
110
 
76
111
  #### Dataset source (pick one)
77
112
 
@@ -1,7 +1,7 @@
1
1
  # (C) 2026 GoodData Corporation
2
2
  [project]
3
3
  name = "gooddata-eval"
4
- version = "1.72.1.dev6"
4
+ version = "1.73.0"
5
5
  description = "Evaluate the GoodData AI agent against your own questions and models."
6
6
  readme = "README.md"
7
7
  license = "MIT"
@@ -11,7 +11,7 @@ authors = [
11
11
  keywords = ["gooddata", "ai", "evaluation", "llm", "analytics", "cli"]
12
12
  requires-python = ">=3.10"
13
13
  dependencies = [
14
- "gooddata-sdk~=1.72.1.dev6",
14
+ "gooddata-sdk~=1.73.0",
15
15
  "httpx>=0.27,<1.0",
16
16
  "orjson>=3.9.15,<4.0.0",
17
17
  "pydantic>=2.6,<3.0",
@@ -16,7 +16,7 @@ from gooddata_eval.core.agentic.metric_skill import evaluate_agentic_metric_skil
16
16
  from gooddata_eval.core.agentic.search_tool import evaluate_agentic_search_tool
17
17
  from gooddata_eval.core.agentic.visualization import evaluate_agentic_visualization
18
18
  from gooddata_eval.core.config import ReasoningEffort
19
- from gooddata_eval.core.models import CreatedVisualization, DatasetItem
19
+ from gooddata_eval.core.models import AgenticEvalOutcome, CreatedVisualization, DatasetItem
20
20
  from gooddata_eval.core.runner import EvalReport, ItemReport
21
21
 
22
22
 
@@ -85,8 +85,14 @@ def _dispatch_agentic(
85
85
  run_ts: str,
86
86
  model_version_override: str | None,
87
87
  reasoning_effort: ReasoningEffort | None = None,
88
- ) -> None:
89
- """Call the appropriate evaluate_agentic_* function for the item's test_kind."""
88
+ agent_id: str | None = None,
89
+ ) -> AgenticEvalOutcome | list[str] | None:
90
+ """Call the appropriate evaluate_agentic_* function for the item's test_kind.
91
+
92
+ Returns whatever that function returns -- alert_skill/metric_skill/conversation return
93
+ an AgenticEvalOutcome; the rest still return None
94
+ (unchanged).
95
+ """
90
96
  kind = item.test_kind
91
97
  eo = item.expected_output
92
98
  lf_kw: _LfKw = {
@@ -99,66 +105,72 @@ def _dispatch_agentic(
99
105
  }
100
106
 
101
107
  if kind in ("vis_agentic", "agentic_visualization"):
102
- evaluate_agentic_visualization(
108
+ return evaluate_agentic_visualization(
103
109
  host=host,
104
110
  token=token,
105
111
  workspace_id=workspace_id,
106
112
  question=item.question,
107
113
  expected_outputs=_parse_visualization_expected(eo),
108
114
  k=k,
115
+ agent_id=agent_id,
109
116
  **lf_kw,
110
117
  )
111
118
  elif kind == "agentic_metric_skill":
112
- evaluate_agentic_metric_skill(
119
+ return evaluate_agentic_metric_skill(
113
120
  host=host,
114
121
  token=token,
115
122
  workspace_id=workspace_id,
116
123
  question=item.question,
117
124
  expected_output=eo if isinstance(eo, (dict, list)) else {},
118
125
  k=k,
126
+ agent_id=agent_id,
119
127
  **lf_kw,
120
128
  )
121
129
  elif kind == "agentic_alert_skill":
122
- evaluate_agentic_alert_skill(
130
+ return evaluate_agentic_alert_skill(
123
131
  host=host,
124
132
  token=token,
125
133
  workspace_id=workspace_id,
126
134
  question=item.question,
127
135
  expected_output=eo if isinstance(eo, dict) else {},
128
136
  k=k,
137
+ agent_id=agent_id,
129
138
  **lf_kw,
130
139
  )
131
140
  elif kind == "agentic_search":
132
141
  eo_dict = eo if isinstance(eo, dict) else {}
133
142
  tool_call = eo_dict.get("tool_call", {})
134
143
  expected_args = tool_call.get("function_arguments", eo_dict)
135
- evaluate_agentic_search_tool(
144
+ return evaluate_agentic_search_tool(
136
145
  host=host,
137
146
  token=token,
138
147
  workspace_id=workspace_id,
139
148
  question=item.question,
140
149
  expected_tool_call=expected_args,
141
150
  k=k,
151
+ agent_id=agent_id,
142
152
  **lf_kw,
143
153
  )
144
154
  elif kind == "agentic_general_question":
145
- evaluate_agentic_general_question(
155
+ return evaluate_agentic_general_question(
146
156
  host=host,
147
157
  token=token,
148
158
  workspace_id=workspace_id,
149
159
  question=item.question,
150
160
  expected_output=eo if isinstance(eo, str) else str(eo),
151
161
  k=k,
162
+ agent_id=agent_id,
152
163
  **lf_kw,
153
164
  )
154
165
  elif kind == "agentic_guardrail":
155
- evaluate_agentic_guardrail(
166
+ return evaluate_agentic_guardrail(
156
167
  host=host,
157
168
  token=token,
158
169
  workspace_id=workspace_id,
159
170
  question=item.question,
160
171
  expected_output=eo if isinstance(eo, str) else str(eo),
161
172
  k=k,
173
+ agent_id=agent_id,
162
174
  **lf_kw,
163
175
  )
164
176
  elif kind == "agentic_kda_skill":
@@ -173,11 +185,12 @@ def _dispatch_agentic(
173
185
  )
174
186
  elif kind == "agentic_conversation":
175
187
  fixture_data = eo.get("fixture") or eo if isinstance(eo, dict) else {}
176
- evaluate_agentic_conversation(
188
+ return evaluate_agentic_conversation(
177
189
  host=host,
178
190
  token=token,
179
191
  workspace_id=workspace_id,
180
192
  fixture=ConversationFixture.model_validate(fixture_data),
193
+ agent_id=agent_id,
181
194
  **lf_kw,
182
195
  )
183
196
  else:
@@ -197,6 +210,7 @@ def run_agentic_items(
197
210
  run_ts: str,
198
211
  on_item_start: Any = None,
199
212
  on_item_done: Any = None,
213
+ agent_id: str | None = None,
200
214
  ) -> EvalReport:
201
215
  """Run agentic items through evaluate_agentic_* and return an EvalReport."""
202
216
  langfuse = make_langfuse_client() if use_langfuse else None
@@ -219,12 +233,26 @@ def run_agentic_items(
219
233
  )
220
234
  t0 = time.perf_counter()
221
235
  try:
222
- _dispatch_agentic(item, host, token, workspace_id, k, langfuse, run_ts, model_version, reasoning_effort)
236
+ outcome = _dispatch_agentic(
237
+ item, host, token, workspace_id, k, langfuse, run_ts, model_version, reasoning_effort, agent_id
238
+ )
239
+ if isinstance(outcome, AgenticEvalOutcome):
240
+ reasoning_steps = outcome.reasoning_steps
241
+ conversation_id = outcome.conversation_id
242
+ response_id = outcome.response_id
243
+ else:
244
+ reasoning_steps, conversation_id, response_id = outcome, None, None
223
245
  item_report.pass_at_k = True
224
246
  item_report.runs = k
247
+ item_report.reasoning_steps = reasoning_steps or []
248
+ item_report.conversation_id = conversation_id
249
+ item_report.response_id = response_id
225
250
  except AssertionError as exc:
226
251
  item_report.pass_at_k = False
227
252
  item_report.runs = k
253
+ item_report.reasoning_steps = getattr(exc, "reasoning_steps", None) or []
254
+ item_report.conversation_id = getattr(exc, "conversation_id", None)
255
+ item_report.response_id = getattr(exc, "response_id", None)
228
256
  print(f"[agentic] {item.id} FAIL: {exc}", flush=True)
229
257
  except Exception as exc:
230
258
  item_report.error = f"{type(exc).__name__}: {exc}"
@@ -2,6 +2,7 @@
2
2
  """`gd-eval` command-line entry point."""
3
3
 
4
4
  import argparse
5
+ import os
5
6
  import sys
6
7
  import threading
7
8
  from datetime import datetime, timezone
@@ -117,6 +118,16 @@ def _build_parser() -> argparse.ArgumentParser:
117
118
  action="store_true",
118
119
  help="Log scores and traces to Langfuse (requires --langfuse-dataset and LANGFUSE_* env vars).",
119
120
  )
121
+ run.add_argument(
122
+ "--agent-id",
123
+ dest="agent_id",
124
+ help=(
125
+ "AI Hub agent id every conversation should target (or set GD_EVAL_AGENT_ID). "
126
+ "GoodData has no admin-settable default agent -- without this, each conversation "
127
+ "falls back to whichever agent the platform's last-used/last-edited heuristic "
128
+ "resolves, which may not have every skill under test enabled."
129
+ ),
130
+ )
120
131
  models_cmd = sub.add_parser("models", help="List LLM providers and models configured in the org.")
121
132
  models_cmd.add_argument("--host", help="GoodData host URL.")
122
133
  models_cmd.add_argument("--token", help="API token (or set GOODDATA_TOKEN).")
@@ -347,6 +358,7 @@ def _run(config: RunConfig) -> int:
347
358
  run_ts=run_ts,
348
359
  on_item_start=on_item_start,
349
360
  on_item_done=on_item_done,
361
+ agent_id=config.agent_id,
350
362
  )
351
363
 
352
364
  # --- non-agentic items (single-turn, use Evaluator) ---
@@ -357,6 +369,7 @@ def _run(config: RunConfig) -> int:
357
369
  workspace_id=config.workspace_id,
358
370
  preserve_failed=config.preserve_failed,
359
371
  reasoning_effort=config.reasoning_effort,
372
+ agent_id=config.agent_id,
360
373
  ),
361
374
  SummaryClient(host=config.host, token=config.token, workspace_id=config.workspace_id),
362
375
  )
@@ -449,6 +462,7 @@ def main(argv: list[str] | None = None) -> int:
449
462
  kind=args.kind,
450
463
  preserve_failed=args.preserve_failed,
451
464
  reasoning_effort=args.reasoning_effort,
465
+ agent_id=args.agent_id or os.environ.get("GD_EVAL_AGENT_ID"),
452
466
  )
453
467
  return _run(config)
454
468
  except (
@@ -6,7 +6,7 @@ from __future__ import annotations
6
6
  import json
7
7
  import os
8
8
  import re
9
- from dataclasses import dataclass
9
+ from dataclasses import dataclass, field
10
10
  from typing import Any
11
11
 
12
12
  from gooddata_sdk import GoodDataSdk
@@ -14,7 +14,7 @@ from gooddata_sdk import GoodDataSdk
14
14
  from gooddata_eval.core.agentic._catalog import CatalogMetricAlert
15
15
  from gooddata_eval.core.chat.sse_client import ChatClient
16
16
  from gooddata_eval.core.config import ReasoningEffort
17
- from gooddata_eval.core.models import ToolCallEvent
17
+ from gooddata_eval.core.models import AgenticEvalOutcome, ToolCallEvent
18
18
 
19
19
  try:
20
20
  from openai import OpenAI as _OpenAI
@@ -119,7 +119,31 @@ def _check_metric(expected: CatalogMetricAlert, actual_args: dict) -> bool:
119
119
  return expected.metric_id == act_metric
120
120
 
121
121
 
122
- def _check_recipients(expected: CatalogMetricAlert, actual_args: dict) -> bool:
122
+ def _resolve_internal_recipient_ids(sdk: GoodDataSdk, emails: list[str]) -> set[str]:
123
+ """Best-effort map of expected recipient emails to internal GoodData user ids.
124
+
125
+ Some notification channels are workspace-restricted to internal users --
126
+ `create_metric_alert` then addresses the alert by internal user id
127
+ (`internal_recipients`), never by email, so an expected email has to be
128
+ resolved before it can be compared against that field. Failures (no
129
+ matching user, no permission, network error) are swallowed: the caller
130
+ treats an empty result the same as "this delivery path doesn't match",
131
+ which is correct -- it doesn't mean the alert itself failed.
132
+ """
133
+ if not emails:
134
+ return set()
135
+ try:
136
+ # RSQL quoted-string escaping: backslash first, then the enclosing quote char,
137
+ # or an email like o'hara@example.com breaks the filter into invalid RSQL.
138
+ escaped = [email.replace("\\", "\\\\").replace("'", "\\'") for email in emails]
139
+ quoted = ",".join(f"'{email}'" for email in escaped)
140
+ resp = sdk._client.entities_api.get_all_entities_users(filter=f"email=in=({quoted})")
141
+ return {u.id for u in (resp.data or [])}
142
+ except Exception:
143
+ return set()
144
+
145
+
146
+ def _check_recipients(expected: CatalogMetricAlert, actual_args: dict, sdk: GoodDataSdk | None = None) -> bool:
123
147
  if not expected.recipients:
124
148
  return True
125
149
  act_recip_raw = actual_args.get("recipients", actual_args.get("external_recipients"))
@@ -134,7 +158,14 @@ def _check_recipients(expected: CatalogMetricAlert, actual_args: dict) -> bool:
134
158
  act_recip = act_recip_raw
135
159
  else:
136
160
  act_recip = []
137
- return set(expected.recipients) == set(act_recip or [])
161
+ if set(expected.recipients) == set(act_recip or []):
162
+ return True
163
+ act_internal = actual_args.get("internal_recipients")
164
+ if sdk is not None and isinstance(act_internal, list) and act_internal:
165
+ internal_recipient_ids = _resolve_internal_recipient_ids(sdk, expected.recipients)
166
+ if internal_recipient_ids & set(act_internal):
167
+ return True
168
+ return False
138
169
 
139
170
 
140
171
  def generate_simulated_alert_response(
@@ -302,6 +333,8 @@ class AlertRunResult:
302
333
  alert_id: str | None
303
334
  eval: AlertEvaluation
304
335
  actual_alert_arguments: dict
336
+ reasoning_steps: list[str] = field(default_factory=list)
337
+ response_id: str | None = None
305
338
 
306
339
 
307
340
  @dataclass
@@ -434,11 +467,14 @@ def run_agentic_alert_skill(
434
467
  max_iterations: int = _DEFAULT_MAX_ITERATIONS,
435
468
  initial_conversation_id: str | None = None,
436
469
  reasoning_effort: ReasoningEffort | None = None,
470
+ agent_id: str | None = None,
437
471
  ) -> AgenticAlertSummary:
438
472
  """Run the alert-skill agentic evaluation K times and return a summary."""
439
473
  expected = _normalize_expected_output(expected_output)
440
474
  run_results: list[AlertRunResult] = []
441
- client = ChatClient(host=host, token=token, workspace_id=workspace_id, reasoning_effort=reasoning_effort)
475
+ client = ChatClient(
476
+ host=host, token=token, workspace_id=workspace_id, reasoning_effort=reasoning_effort, agent_id=agent_id
477
+ )
442
478
  sdk = GoodDataSdk.create(host, token)
443
479
 
444
480
  def _run_once(conv_id: str) -> AlertRunResult:
@@ -447,6 +483,8 @@ def run_agentic_alert_skill(
447
483
  alert_id: str | None = None
448
484
  actual_args: dict = {}
449
485
  tool_called = False
486
+ reasoning_steps: list[str] = []
487
+ response_id: str | None = None
450
488
  # conversation_history stores prior turns for GPT-4o context.
451
489
  # Roles follow GPT-4o's perspective: "assistant"=agent text, "user"=sim-user reply.
452
490
  conversation_history: list = []
@@ -454,6 +492,8 @@ def run_agentic_alert_skill(
454
492
 
455
493
  for _iteration in range(max_iterations):
456
494
  chat_result = client.send_message(conv_id, current_question)
495
+ reasoning_steps.extend(chat_result.reasoning_steps or [])
496
+ response_id = chat_result.response_id or response_id
457
497
  alert_id, actual_args, tool_called = _extract_alert_call(chat_result.tool_call_events or [])
458
498
  if tool_called:
459
499
  alert_id_to_delete = alert_id
@@ -482,13 +522,15 @@ def run_agentic_alert_skill(
482
522
  trigger_correct=tool_called and _check_trigger(expected, actual_args),
483
523
  filters_correct=tool_called and _check_filters(expected, actual_args),
484
524
  metric_correct=tool_called and _check_metric(expected, actual_args),
485
- recipients_correct=tool_called and _check_recipients(expected, actual_args),
525
+ recipients_correct=tool_called and _check_recipients(expected, actual_args, sdk=sdk),
486
526
  )
487
527
  return AlertRunResult(
488
528
  conversation_id=conv_id,
489
529
  alert_id=alert_id,
490
530
  eval=ev,
491
531
  actual_alert_arguments=actual_args,
532
+ reasoning_steps=reasoning_steps,
533
+ response_id=response_id,
492
534
  )
493
535
  finally:
494
536
  if alert_id_to_delete:
@@ -539,6 +581,9 @@ class AlertSkillAssertionError(AssertionError):
539
581
  """Raised when an alert-skill evaluation fails."""
540
582
 
541
583
  __tracebackhide__ = True
584
+ reasoning_steps: list[str]
585
+ conversation_id: str
586
+ response_id: str | None
542
587
 
543
588
 
544
589
  def evaluate_agentic_alert_skill(
@@ -550,6 +595,7 @@ def evaluate_agentic_alert_skill(
550
595
  k: int = _DEFAULT_K,
551
596
  max_iterations: int = _DEFAULT_MAX_ITERATIONS,
552
597
  initial_conversation_id: str | None = None,
598
+ agent_id: str | None = None,
553
599
  langfuse: object | None = None,
554
600
  dataset_item_id: str = "",
555
601
  dataset_name: str = "alert_skill",
@@ -557,8 +603,16 @@ def evaluate_agentic_alert_skill(
557
603
  model_version_override: str | None = None,
558
604
  run_metadata_extra: dict | None = None,
559
605
  reasoning_effort: ReasoningEffort | None = None,
560
- ) -> None:
561
- """Run alert-skill evaluation, log to Langfuse, and raise AlertSkillAssertionError on failure."""
606
+ ) -> AgenticEvalOutcome:
607
+ """Run alert-skill evaluation, log to Langfuse, and raise AlertSkillAssertionError on failure.
608
+
609
+ Returns the best run's outcome (reasoning_steps, conversation_id, response_id) as an
610
+ AgenticEvalOutcome on success; on failure the same three values are attached to the
611
+ raised exception as
612
+ ``.reasoning_steps``/``.conversation_id``/``.response_id`` (mirrors the
613
+ `conversation_id`-on-exception idiom in `ChatClient.ask()`) so callers can retrieve them
614
+ either way.
615
+ """
562
616
  from datetime import datetime as _dt # noqa: PLC0415
563
617
  from datetime import timezone as _tz # noqa: PLC0415
564
618
 
@@ -577,6 +631,7 @@ def evaluate_agentic_alert_skill(
577
631
  max_iterations=max_iterations,
578
632
  initial_conversation_id=initial_conversation_id,
579
633
  reasoning_effort=reasoning_effort,
634
+ agent_id=agent_id,
580
635
  )
581
636
 
582
637
  if langfuse is not None and dataset_item_id:
@@ -631,7 +686,7 @@ def evaluate_agentic_alert_skill(
631
686
  if not summary.pass_at_k:
632
687
  best = summary.best
633
688
  ev = best.eval
634
- raise AlertSkillAssertionError(
689
+ exc = AlertSkillAssertionError(
635
690
  f"Alert skill assertion failed. strict_pass={ev.strict_pass}. "
636
691
  f"alert_created={ev.alert_created}, operator_correct={ev.operator_correct}, "
637
692
  f"threshold_correct={ev.threshold_correct}, trigger_correct={ev.trigger_correct}, "
@@ -639,3 +694,12 @@ def evaluate_agentic_alert_skill(
639
694
  f"recipients_correct={ev.recipients_correct}. "
640
695
  f"Actual args: {best.actual_alert_arguments}"
641
696
  )
697
+ exc.reasoning_steps = best.reasoning_steps
698
+ exc.conversation_id = best.conversation_id
699
+ exc.response_id = best.response_id
700
+ raise exc
701
+ return AgenticEvalOutcome(
702
+ reasoning_steps=summary.best.reasoning_steps,
703
+ conversation_id=summary.best.conversation_id,
704
+ response_id=summary.best.response_id,
705
+ )
@@ -5,7 +5,7 @@ from __future__ import annotations
5
5
 
6
6
  import json
7
7
  import re
8
- from dataclasses import dataclass
8
+ from dataclasses import dataclass, field
9
9
  from typing import Literal
10
10
 
11
11
  from gooddata_sdk import GoodDataSdk
@@ -15,7 +15,7 @@ from gooddata_eval.core.agentic.alert_skill import render_alert_proposal
15
15
  from gooddata_eval.core.agentic.metric_skill import _delete_metric, _extract_created_metric_ids
16
16
  from gooddata_eval.core.chat.sse_client import ChatClient
17
17
  from gooddata_eval.core.config import ReasoningEffort
18
- from gooddata_eval.core.models import ChatResult, ToolCallEvent
18
+ from gooddata_eval.core.models import AgenticEvalOutcome, ChatResult, ToolCallEvent
19
19
  from gooddata_eval.core.scoring import (
20
20
  check_filters,
21
21
  check_viz_type,
@@ -265,6 +265,8 @@ class ConversationResult:
265
265
  full_skill_coverage: bool
266
266
  conversation_success: bool
267
267
  total_clarification_turns: int
268
+ reasoning_steps: list[str] = field(default_factory=list)
269
+ response_id: str | None = None
268
270
 
269
271
 
270
272
  def run_agentic_conversation(
@@ -275,6 +277,7 @@ def run_agentic_conversation(
275
277
  max_clarification_turns: int = _DEFAULT_MAX_CLARIFICATION_TURNS,
276
278
  initial_conversation_id: str | None = None,
277
279
  reasoning_effort: ReasoningEffort | None = None,
280
+ agent_id: str | None = None,
278
281
  ) -> ConversationResult:
279
282
  """Run a multi-turn, multi-skill conversation evaluation (no K-runs).
280
283
 
@@ -282,7 +285,9 @@ def run_agentic_conversation(
282
285
  trigger up to *max_clarification_turns* additional rounds of simulated-user
283
286
  replies before the agent produces the expected output.
284
287
  """
285
- client = ChatClient(host=host, token=token, workspace_id=workspace_id, reasoning_effort=reasoning_effort)
288
+ client = ChatClient(
289
+ host=host, token=token, workspace_id=workspace_id, reasoning_effort=reasoning_effort, agent_id=agent_id
290
+ )
286
291
  sdk = GoodDataSdk.create(host, token)
287
292
  turn_results: list[TurnResult] = []
288
293
  turn_outputs: dict[str, dict] = {}
@@ -293,6 +298,8 @@ def run_agentic_conversation(
293
298
  # not persist in the (shared) workspace and get reused by a later test. Deferred to
294
299
  # the end — a later turn may $ref a metric an earlier turn created.
295
300
  created_metric_ids: list[str] = []
301
+ reasoning_steps: list[str] = []
302
+ response_id: str | None = None
296
303
 
297
304
  try:
298
305
  if initial_conversation_id is not None:
@@ -329,6 +336,8 @@ def run_agentic_conversation(
329
336
  chat_result = client.send_message(conversation_id, current_message)
330
337
  final_result = chat_result
331
338
  all_tool_calls.extend(chat_result.tool_call_events or [])
339
+ reasoning_steps.extend(chat_result.reasoning_steps or [])
340
+ response_id = chat_result.response_id or response_id
332
341
 
333
342
  if _check_output_present(resolved_turn, chat_result):
334
343
  break
@@ -392,6 +401,8 @@ def run_agentic_conversation(
392
401
  full_skill_coverage=full_skill_coverage,
393
402
  conversation_success=conversation_success,
394
403
  total_clarification_turns=total_clarification_turns,
404
+ reasoning_steps=reasoning_steps,
405
+ response_id=response_id,
395
406
  )
396
407
 
397
408
 
@@ -399,6 +410,9 @@ class ConversationAssertionError(AssertionError):
399
410
  """Raised when a conversation evaluation fails."""
400
411
 
401
412
  __tracebackhide__ = True
413
+ reasoning_steps: list[str]
414
+ conversation_id: str
415
+ response_id: str | None
402
416
 
403
417
 
404
418
  def evaluate_agentic_conversation(
@@ -408,6 +422,7 @@ def evaluate_agentic_conversation(
408
422
  fixture: ConversationFixture,
409
423
  max_clarification_turns: int = _DEFAULT_MAX_CLARIFICATION_TURNS,
410
424
  initial_conversation_id: str | None = None,
425
+ agent_id: str | None = None,
411
426
  langfuse: object | None = None,
412
427
  dataset_item_id: str = "",
413
428
  dataset_name: str = "conversation",
@@ -415,8 +430,16 @@ def evaluate_agentic_conversation(
415
430
  model_version_override: str | None = None,
416
431
  run_metadata_extra: dict | None = None,
417
432
  reasoning_effort: ReasoningEffort | None = None,
418
- ) -> None:
419
- """Run conversation evaluation, log to Langfuse, and raise on failure."""
433
+ ) -> AgenticEvalOutcome:
434
+ """Run conversation evaluation, log to Langfuse, and raise on failure.
435
+
436
+ Returns the conversation's outcome (reasoning_steps, conversation_id, response_id) as
437
+ an AgenticEvalOutcome on success; on failure the same three values are attached to the
438
+ raised exception as
439
+ ``.reasoning_steps``/``.conversation_id``/``.response_id`` (mirrors the
440
+ `conversation_id`-on-exception idiom in `ChatClient.ask()`) so callers can retrieve them
441
+ either way.
442
+ """
420
443
  from datetime import datetime as _dt # noqa: PLC0415
421
444
  from datetime import timezone as _tz # noqa: PLC0415
422
445
 
@@ -433,6 +456,7 @@ def evaluate_agentic_conversation(
433
456
  max_clarification_turns=max_clarification_turns,
434
457
  initial_conversation_id=initial_conversation_id,
435
458
  reasoning_effort=reasoning_effort,
459
+ agent_id=agent_id,
436
460
  )
437
461
 
438
462
  if langfuse is not None and dataset_item_id:
@@ -492,8 +516,17 @@ def evaluate_agentic_conversation(
492
516
 
493
517
  if not result.conversation_success:
494
518
  failed_turns = [tr for tr in result.turn_results if not tr.skill_success]
495
- raise ConversationAssertionError(
519
+ exc = ConversationAssertionError(
496
520
  f"Conversation assertion failed. "
497
521
  f"full_skill_coverage={result.full_skill_coverage}. "
498
522
  f"Failed turns: {[t.turn_id for t in failed_turns]}"
499
523
  )
524
+ exc.reasoning_steps = result.reasoning_steps
525
+ exc.conversation_id = result.conversation_id
526
+ exc.response_id = result.response_id
527
+ raise exc
528
+ return AgenticEvalOutcome(
529
+ reasoning_steps=result.reasoning_steps,
530
+ conversation_id=result.conversation_id,
531
+ response_id=result.response_id,
532
+ )