gooddata-eval 1.71.0__tar.gz → 1.71.1.dev1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (86) hide show
  1. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/PKG-INFO +2 -2
  2. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/pyproject.toml +2 -2
  3. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/core/agentic/alert_skill.py +22 -5
  4. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/core/agentic/conversation.py +5 -1
  5. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/core/chat/sse_client.py +7 -0
  6. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/core/models.py +4 -0
  7. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/tests/test_agentic_alert_skill.py +92 -0
  8. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/tests/test_agentic_conversation.py +62 -1
  9. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/tests/test_sse_client.py +38 -0
  10. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/.gitignore +0 -0
  11. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/LICENSE.txt +0 -0
  12. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/Makefile +0 -0
  13. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/README.md +0 -0
  14. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/__init__.py +0 -0
  15. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/_version.py +0 -0
  16. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/cli/__init__.py +0 -0
  17. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/cli/agentic_runner.py +0 -0
  18. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/cli/main.py +0 -0
  19. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/core/__init__.py +0 -0
  20. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/core/agentic/__init__.py +0 -0
  21. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/core/agentic/_catalog.py +0 -0
  22. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/core/agentic/_langfuse.py +0 -0
  23. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/core/agentic/general_question.py +0 -0
  24. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/core/agentic/guardrail.py +0 -0
  25. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/core/agentic/metric_skill.py +0 -0
  26. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/core/agentic/search_tool.py +0 -0
  27. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/core/agentic/visualization.py +0 -0
  28. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/core/chat/__init__.py +0 -0
  29. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/core/config.py +0 -0
  30. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/core/connection.py +0 -0
  31. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/core/dataset/__init__.py +0 -0
  32. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/core/dataset/langfuse_source.py +0 -0
  33. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/core/dataset/local.py +0 -0
  34. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/core/evaluators/__init__.py +0 -0
  35. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/core/evaluators/_deep_subset.py +0 -0
  36. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/core/evaluators/_llm_judge.py +0 -0
  37. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/core/evaluators/_text_utils.py +0 -0
  38. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/core/evaluators/alert_skill.py +0 -0
  39. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/core/evaluators/base.py +0 -0
  40. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/core/evaluators/general_question.py +0 -0
  41. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/core/evaluators/guardrail.py +0 -0
  42. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/core/evaluators/metric_skill.py +0 -0
  43. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/core/evaluators/search_tool.py +0 -0
  44. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/core/evaluators/summary.py +0 -0
  45. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/core/evaluators/visualization.py +0 -0
  46. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/core/langfuse/__init__.py +0 -0
  47. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/core/langfuse/sink.py +0 -0
  48. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/core/reporting/__init__.py +0 -0
  49. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/core/reporting/console.py +0 -0
  50. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/core/reporting/json_report.py +0 -0
  51. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/core/runner.py +0 -0
  52. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/core/scoring.py +0 -0
  53. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/core/summary/__init__.py +0 -0
  54. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/core/summary/http_client.py +0 -0
  55. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/core/workspace.py +0 -0
  56. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/tests/__init__.py +0 -0
  57. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/tests/conftest.py +0 -0
  58. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/tests/fixtures/sample_dataset/metric_skill_create.json +0 -0
  59. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/tests/fixtures/sample_dataset/visualization_revenue.json +0 -0
  60. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/tests/fixtures/sse_visualization_stream.txt +0 -0
  61. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/tests/test_agentic_general_question.py +0 -0
  62. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/tests/test_agentic_guardrail.py +0 -0
  63. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/tests/test_agentic_metric_skill.py +0 -0
  64. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/tests/test_agentic_run_context.py +0 -0
  65. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/tests/test_agentic_search_tool.py +0 -0
  66. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/tests/test_agentic_visualization.py +0 -0
  67. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/tests/test_alert_skill_evaluator.py +0 -0
  68. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/tests/test_cli.py +0 -0
  69. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/tests/test_connection.py +0 -0
  70. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/tests/test_deep_subset.py +0 -0
  71. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/tests/test_langfuse_sink.py +0 -0
  72. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/tests/test_langfuse_source.py +0 -0
  73. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/tests/test_llm_judge.py +0 -0
  74. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/tests/test_local_loader.py +0 -0
  75. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/tests/test_metric_skill_evaluator.py +0 -0
  76. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/tests/test_models.py +0 -0
  77. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/tests/test_reporting.py +0 -0
  78. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/tests/test_runner.py +0 -0
  79. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/tests/test_scoring.py +0 -0
  80. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/tests/test_search_tool_evaluator.py +0 -0
  81. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/tests/test_summary_client.py +0 -0
  82. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/tests/test_summary_evaluator.py +0 -0
  83. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/tests/test_text_evaluators.py +0 -0
  84. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/tests/test_visualization_evaluator.py +0 -0
  85. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/tests/test_workspace.py +0 -0
  86. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/tox.ini +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: gooddata-eval
3
- Version: 1.71.0
3
+ Version: 1.71.1.dev1
4
4
  Summary: Evaluate the GoodData AI agent against your own questions and models.
5
5
  Project-URL: Source, https://github.com/gooddata/gooddata-python-sdk
6
6
  Author-email: GoodData <support@gooddata.com>
@@ -17,7 +17,7 @@ Classifier: Topic :: Scientific/Engineering
17
17
  Classifier: Topic :: Software Development
18
18
  Classifier: Typing :: Typed
19
19
  Requires-Python: >=3.10
20
- Requires-Dist: gooddata-sdk~=1.71.0
20
+ Requires-Dist: gooddata-sdk~=1.71.1.dev1
21
21
  Requires-Dist: httpx<1.0,>=0.27
22
22
  Requires-Dist: orjson<4.0.0,>=3.9.15
23
23
  Requires-Dist: pydantic<3.0,>=2.6
@@ -1,7 +1,7 @@
1
1
  # (C) 2026 GoodData Corporation
2
2
  [project]
3
3
  name = "gooddata-eval"
4
- version = "1.71.0"
4
+ version = "1.71.1.dev1"
5
5
  description = "Evaluate the GoodData AI agent against your own questions and models."
6
6
  readme = "README.md"
7
7
  license = "MIT"
@@ -11,7 +11,7 @@ authors = [
11
11
  keywords = ["gooddata", "ai", "evaluation", "llm", "analytics", "cli"]
12
12
  requires-python = ">=3.10"
13
13
  dependencies = [
14
- "gooddata-sdk~=1.71.0",
14
+ "gooddata-sdk~=1.71.1.dev1",
15
15
  "httpx>=0.27,<1.0",
16
16
  "orjson>=3.9.15,<4.0.0",
17
17
  "pydantic>=2.6,<3.0",
@@ -311,11 +311,26 @@ def _extract_alert_call(tool_call_events: list[ToolCallEvent]) -> tuple[str | No
311
311
  return None, {}, False
312
312
 
313
313
 
314
- def _is_asking_clarification(text: str) -> bool:
315
- if not text:
316
- return False
317
- t = text.lower()
318
- return "?" in t or "could you" in t or "please" in t or "clarif" in t
314
+ def render_alert_proposal(proposal: dict) -> str:
315
+ """Render an alert-proposal part as the text the simulated user reacts to.
316
+
317
+ The alert skill's confirmation step deliberately emits no text part (GDAI-2032) — the
318
+ prompt and the CTA live only in the proposal payload, which the frontend renders as a
319
+ widget. Dumping the payload (rather than prose) keeps recipients, condition, trigger and
320
+ dashboard visible so the simulated user can still verify them against its goal, and does
321
+ not need updating whenever ``AlertProposal`` grows a field.
322
+ """
323
+ cta = proposal.get("cta") or "Should I create this alert?"
324
+ summary = {k: v for k, v in proposal.items() if k != "cta"}
325
+ alert = dict(summary.get("alert") or {})
326
+ # The AFM execution block is opaque wire dicts — noise that would crowd out the fields
327
+ # the simulated user actually has to check.
328
+ alert.pop("execution", None)
329
+ if "alert" in summary:
330
+ # Key off presence, not truthiness: an alert whose only key was `execution` must
331
+ # still be replaced, otherwise the original (execution-bearing) dict survives.
332
+ summary["alert"] = alert
333
+ return f"{cta}\n\nAlert proposal:\n{json.dumps(summary, indent=2, sort_keys=True)}"
319
334
 
320
335
 
321
336
  def run_agentic_alert_skill(
@@ -352,6 +367,8 @@ def run_agentic_alert_skill(
352
367
  alert_id_to_delete = alert_id
353
368
  break
354
369
  response_text = (chat_result.text_response or "").strip()
370
+ if not response_text and chat_result.alert_proposals:
371
+ response_text = render_alert_proposal(chat_result.alert_proposals[-1])
355
372
  # Stop if agent gave a completely empty response (stuck)
356
373
  if not response_text and not chat_result.tool_call_events:
357
374
  break
@@ -11,6 +11,7 @@ from typing import Literal
11
11
  from gooddata_sdk import GoodDataSdk
12
12
  from pydantic import BaseModel
13
13
 
14
+ from gooddata_eval.core.agentic.alert_skill import render_alert_proposal
14
15
  from gooddata_eval.core.agentic.metric_skill import _delete_metric, _extract_created_metric_ids
15
16
  from gooddata_eval.core.chat.sse_client import ChatClient
16
17
  from gooddata_eval.core.models import ChatResult, ToolCallEvent
@@ -322,7 +323,10 @@ def run_agentic_conversation(
322
323
  break
323
324
 
324
325
  response_text = (chat_result.text_response or "").strip()
325
- if _is_asking_clarification(response_text) and clarification_turns < max_clarification_turns:
326
+ if not response_text and chat_result.alert_proposals:
327
+ response_text = render_alert_proposal(chat_result.alert_proposals[-1])
328
+ asking = _is_asking_clarification(response_text) or bool(chat_result.alert_proposals)
329
+ if asking and clarification_turns < max_clarification_turns:
326
330
  clarification_turns += 1
327
331
  total_clarification_turns += 1
328
332
  current_message = _get_sim_user_response(response_text, resolved_turn, resolved_expected)
@@ -102,6 +102,7 @@ class _SseAccumulator:
102
102
  text_parts: list[str] = field(default_factory=list)
103
103
  viz_reasoning_parts: list[str] = field(default_factory=list)
104
104
  visualizations: list[dict[str, Any]] = field(default_factory=list)
105
+ alert_proposals: list[dict[str, Any]] = field(default_factory=list)
105
106
  tool_call_events: list[dict[str, Any]] = field(default_factory=list)
106
107
  call_id_to_event_index: dict[str, int] = field(default_factory=dict)
107
108
  reasoning_steps: list[dict[str, Any]] = field(default_factory=list)
@@ -125,6 +126,11 @@ def _handle_multipart(content: dict[str, Any], acc: _SseAccumulator) -> None:
125
126
  acc.viz_reasoning_parts.append(t)
126
127
  elif ptype == "visualization" and part.get("visualization"):
127
128
  acc.visualizations.append(part["visualization"])
129
+ elif ptype == "alertProposal":
130
+ # Record the part even when the server could not resolve the proposal payload
131
+ # (``alertProposal: null``) — its mere presence is the confirmation signal, and
132
+ # the reader falls back to a default CTA.
133
+ acc.alert_proposals.append(part.get("alertProposal") or {})
128
134
 
129
135
 
130
136
  def _handle_reasoning(content: dict[str, Any], acc: _SseAccumulator) -> None:
@@ -161,6 +167,7 @@ def _handle_tool_result(content: dict[str, Any], acc: _SseAccumulator) -> None:
161
167
  def _build_chat_result(acc: _SseAccumulator) -> ChatResult:
162
168
  payload: dict[str, Any] = {
163
169
  "textResponse": "\n".join(acc.text_parts) or None,
170
+ "alertProposals": acc.alert_proposals,
164
171
  "toolCallEvents": acc.tool_call_events,
165
172
  "reasoningStepCount": len(acc.reasoning_steps),
166
173
  }
@@ -92,6 +92,10 @@ class ChatResult(BaseModel):
92
92
 
93
93
  text_response: str | None = Field(default=None, alias="textResponse")
94
94
  created_visualizations: CreatedVisualizations | None = Field(default=None, alias="createdVisualizations")
95
+ # Alert-proposal parts of the agent's multipart response. The alert skill's confirmation
96
+ # step emits ONLY this part (no text part), so its `cta` is the only "the agent is asking
97
+ # a question" signal the simulated-user loops can key off.
98
+ alert_proposals: list[dict] = Field(default_factory=list, alias="alertProposals")
95
99
  tool_call_events: list[ToolCallEvent] = Field(default_factory=list, alias="toolCallEvents")
96
100
  reasoning_step_count: int = Field(default=0, alias="reasoningStepCount")
97
101
  conversation_id: str | None = Field(default=None, alias="conversationId")
@@ -8,10 +8,23 @@ from gooddata_eval.core.agentic.alert_skill import (
8
8
  _deep_subset,
9
9
  _normalize_expected_output,
10
10
  _to_number,
11
+ render_alert_proposal,
11
12
  run_agentic_alert_skill,
12
13
  )
13
14
  from gooddata_eval.core.models import ChatResult
14
15
 
16
+ _PROPOSAL = {
17
+ "title": "# of Orders Alert - Greater Than 500",
18
+ "cta": "Should I create this alert?",
19
+ "recipients": [{"email": "admin@gooddata.com"}],
20
+ "dashboard": {"id": "dash-1", "title": "Orders overview"},
21
+ "alert": {
22
+ "trigger": "ALWAYS",
23
+ "condition": {"comparison": {"operator": "GREATER_THAN", "right": {"value": 500}}},
24
+ "execution": {"measures": [{"opaque": "afm"}]},
25
+ },
26
+ }
27
+
15
28
 
16
29
  def test_to_number_int():
17
30
  assert _to_number("42") == 42
@@ -156,3 +169,82 @@ def test_run_agentic_alert_skill_creates_fresh_conversations_for_remaining_runs(
156
169
  )
157
170
  assert mock_client.create_conversation.call_count == 2
158
171
  assert mock_client.delete_conversation.call_count == 2
172
+
173
+
174
+ def test_render_alert_proposal_keeps_verifiable_fields_and_drops_afm():
175
+ rendered = render_alert_proposal(_PROPOSAL)
176
+ # The CTA leads so the simulated user reads it as a question.
177
+ assert rendered.startswith("Should I create this alert?")
178
+ # Rule 3 of the sim-user prompt requires verifying recipients against its goal.
179
+ assert "admin@gooddata.com" in rendered
180
+ assert "GREATER_THAN" in rendered
181
+ assert "Orders overview" in rendered
182
+ # Opaque AFM wire dicts must not crowd out the fields above.
183
+ assert "execution" not in rendered
184
+
185
+
186
+ def test_render_alert_proposal_drops_afm_when_execution_is_the_only_alert_field():
187
+ # Truthiness-gated replacement used to leave the original execution-bearing dict in place.
188
+ rendered = render_alert_proposal({"alert": {"execution": {"measures": [{"opaque": "afm"}]}}})
189
+ assert "execution" not in rendered
190
+ assert "opaque" not in rendered
191
+
192
+
193
+ def test_render_alert_proposal_falls_back_to_default_cta():
194
+ assert render_alert_proposal({}).startswith("Should I create this alert?")
195
+
196
+
197
+ def test_run_agentic_alert_skill_answers_proposal_only_confirmation_turn():
198
+ """GDAI-2032 regression: confirmation turn has no text part, only an alertProposal.
199
+
200
+ Without the fallback the simulated user is handed an empty agent message, so the agent
201
+ never receives an explicit "yes" and create_metric_alert is never called.
202
+ """
203
+ proposal_turn = ChatResult.model_validate(
204
+ {
205
+ "text_response": None,
206
+ "alertProposals": [_PROPOSAL],
207
+ "tool_call_events": [
208
+ {"functionName": "prepare_metric_alert_proposal", "functionArguments": "{}", "result": None}
209
+ ],
210
+ }
211
+ )
212
+ created_turn = ChatResult.model_validate(
213
+ {
214
+ "text_response": "Alert created.",
215
+ "tool_call_events": [
216
+ {
217
+ "functionName": "create_metric_alert",
218
+ "functionArguments": '{"operator": "GREATER_THAN", "threshold": 500}',
219
+ "result": '{"id": "alert-1"}',
220
+ }
221
+ ],
222
+ }
223
+ )
224
+ mock_client = MagicMock()
225
+ mock_client.send_message.side_effect = [proposal_turn, created_turn]
226
+
227
+ with (
228
+ patch("gooddata_eval.core.agentic.alert_skill.ChatClient", return_value=mock_client),
229
+ patch(
230
+ "gooddata_eval.core.agentic.alert_skill.generate_simulated_alert_response",
231
+ return_value="Yes, please proceed to create the alert.",
232
+ ) as mock_sim,
233
+ patch("gooddata_eval.core.agentic.alert_skill._delete_alert"),
234
+ ):
235
+ summary = run_agentic_alert_skill(
236
+ host="http://host",
237
+ token="tok",
238
+ workspace_id="ws1",
239
+ question="Notify me whenever the number of orders goes above 500",
240
+ expected_output={"operator": "GREATER_THAN", "threshold": 500},
241
+ k=1,
242
+ max_iterations=6,
243
+ initial_conversation_id="conv-1",
244
+ )
245
+
246
+ agent_message = mock_sim.call_args.args[0]
247
+ assert "Should I create this alert?" in agent_message
248
+ assert "admin@gooddata.com" in agent_message
249
+ assert summary.best.eval.alert_created is True
250
+ assert summary.best.alert_id == "alert-1"
@@ -10,7 +10,7 @@ from gooddata_eval.core.agentic.conversation import (
10
10
  _resolve_refs,
11
11
  run_agentic_conversation,
12
12
  )
13
- from gooddata_eval.core.models import ToolCallEvent
13
+ from gooddata_eval.core.models import ChatResult, ToolCallEvent
14
14
 
15
15
 
16
16
  def _skills_tc(*skills):
@@ -291,3 +291,64 @@ def test_run_agentic_conversation_deletes_metrics_even_when_a_later_turn_raises(
291
291
  )
292
292
 
293
293
  mock_sdk._client.entities_api.delete_entity_metrics.assert_called_once_with("ws1", "m1")
294
+
295
+
296
+ def _alert_turn_fixture():
297
+ return ConversationFixture(
298
+ id="conv-alert",
299
+ expected_skills=["alert"],
300
+ turns=[
301
+ TurnDefinition(
302
+ turn_id="create_alert",
303
+ message="Now alert me when the metric drops below 100.",
304
+ expected_skill="alert",
305
+ expected_output_type="tool_call",
306
+ expected_tool_name="create_metric_alert",
307
+ )
308
+ ],
309
+ )
310
+
311
+
312
+ def test_run_agentic_conversation_treats_alert_proposal_as_a_clarification():
313
+ """GDAI-2032 regression: a proposal-only turn has no text, so the old text-only check
314
+ stopped the turn instead of replying, and create_metric_alert never happened."""
315
+ proposal_turn = ChatResult.model_validate(
316
+ {
317
+ "text_response": None,
318
+ "alertProposals": [{"cta": "Should I create this alert?", "recipients": [{"email": "a@b.com"}]}],
319
+ "toolCallEvents": [
320
+ {"functionName": "set_skills", "functionArguments": '{"skills": ["alert"]}', "result": None},
321
+ {"functionName": "prepare_metric_alert_proposal", "functionArguments": "{}", "result": None},
322
+ ],
323
+ }
324
+ )
325
+ created_turn = ChatResult.model_validate(
326
+ {
327
+ "text_response": "Alert created.",
328
+ "toolCallEvents": [
329
+ {"functionName": "create_metric_alert", "functionArguments": "{}", "result": '{"id": "alert-1"}'}
330
+ ],
331
+ }
332
+ )
333
+ mock_client = MagicMock()
334
+ mock_client.create_conversation.return_value = "conv-1"
335
+ mock_client.send_message.side_effect = [proposal_turn, created_turn]
336
+
337
+ with (
338
+ patch("gooddata_eval.core.agentic.conversation.ChatClient", return_value=mock_client),
339
+ patch("gooddata_eval.core.agentic.conversation.GoodDataSdk"),
340
+ patch(
341
+ "gooddata_eval.core.agentic.conversation._get_sim_user_response",
342
+ return_value="Yes, please create it.",
343
+ ) as mock_sim,
344
+ ):
345
+ result = run_agentic_conversation(
346
+ host="http://host/api/v1/actions/workspaces/ws1/ai",
347
+ token="tok",
348
+ workspace_id="ws1",
349
+ fixture=_alert_turn_fixture(),
350
+ )
351
+
352
+ assert "Should I create this alert?" in mock_sim.call_args.args[0]
353
+ assert result.turn_results[0].clarification_turns_used == 1
354
+ assert result.turn_results[0].skill_success is True
@@ -82,6 +82,44 @@ def test_parse_sse_lines_prefers_multipart_viz_over_adhoc_fallback():
82
82
  assert result.created_visualizations.objects[0].id == "real"
83
83
 
84
84
 
85
+ def test_parse_sse_lines_collects_alert_proposal_without_text_part():
86
+ """The alert skill's confirmation turn emits ONLY an alertProposal part (GDAI-2032).
87
+
88
+ Pins the wire contract the simulated-user loops depend on: part ``type`` is
89
+ ``alertProposal`` and the payload lives under the ``alertProposal`` key.
90
+ """
91
+ proposal = {
92
+ "title": "# of Orders Alert - Greater Than 500",
93
+ "cta": "Should I create this alert?",
94
+ "recipients": [{"email": "admin@gooddata.com"}],
95
+ "alert": {"trigger": "ALWAYS", "execution": {"measures": [{"opaque": "afm"}]}},
96
+ }
97
+ lines = [
98
+ 'data: {"item": {"role": "assistant", "content": {"type": "toolCall", "callId": "c1", '
99
+ '"name": "prepare_metric_alert_proposal", "arguments": {}}}}',
100
+ f'data: {{"item": {{"role": "assistant", "content": {{"type": "multipart", '
101
+ f'"parts": [{{"type": "alertProposal", "alertProposal": {json.dumps(proposal)}}}]}}}}}}',
102
+ ]
103
+ result = parse_sse_lines(lines)
104
+ assert result.text_response is None
105
+ assert result.alert_proposals == [proposal]
106
+
107
+
108
+ def test_parse_sse_lines_keeps_alert_proposal_part_when_payload_is_null():
109
+ """Presence of the part is the confirmation signal even if the server did not resolve it."""
110
+ lines = [
111
+ 'data: {"item": {"role": "assistant", "content": {"type": "multipart", '
112
+ '"parts": [{"type": "alertProposal", "alertProposal": null}]}}}',
113
+ ]
114
+ result = parse_sse_lines(lines)
115
+ assert result.alert_proposals == [{}]
116
+
117
+
118
+ def test_parse_sse_lines_has_no_alert_proposals_by_default():
119
+ lines = ['data: {"item": {"role": "assistant", "content": {"type": "text", "text": "Done"}}}']
120
+ assert parse_sse_lines(lines).alert_proposals == []
121
+
122
+
85
123
  @pytest.mark.parametrize("code", [429, 502, 503, 504])
86
124
  def test_parse_sse_lines_transient_status_codes(code):
87
125
  with pytest.raises(TransientChatError) as ei: