gooddata-eval 1.72.1.dev5__tar.gz → 1.72.1.dev6__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (89) hide show
  1. {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/PKG-INFO +2 -2
  2. {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/pyproject.toml +2 -2
  3. {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/src/gooddata_eval/core/agentic/conversation.py +26 -17
  4. {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/src/gooddata_eval/core/agentic/metric_skill.py +33 -19
  5. {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/tests/test_agentic_conversation.py +115 -0
  6. {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/tests/test_agentic_metric_skill.py +61 -1
  7. {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/.gitignore +0 -0
  8. {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/LICENSE.txt +0 -0
  9. {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/Makefile +0 -0
  10. {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/README.md +0 -0
  11. {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/src/gooddata_eval/__init__.py +0 -0
  12. {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/src/gooddata_eval/_version.py +0 -0
  13. {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/src/gooddata_eval/cli/__init__.py +0 -0
  14. {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/src/gooddata_eval/cli/agentic_runner.py +0 -0
  15. {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/src/gooddata_eval/cli/main.py +0 -0
  16. {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/src/gooddata_eval/core/__init__.py +0 -0
  17. {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/src/gooddata_eval/core/agentic/__init__.py +0 -0
  18. {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/src/gooddata_eval/core/agentic/_catalog.py +0 -0
  19. {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/src/gooddata_eval/core/agentic/_langfuse.py +0 -0
  20. {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/src/gooddata_eval/core/agentic/alert_skill.py +0 -0
  21. {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/src/gooddata_eval/core/agentic/general_question.py +0 -0
  22. {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/src/gooddata_eval/core/agentic/guardrail.py +0 -0
  23. {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/src/gooddata_eval/core/agentic/kda_skill.py +0 -0
  24. {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/src/gooddata_eval/core/agentic/search_tool.py +0 -0
  25. {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/src/gooddata_eval/core/agentic/visualization.py +0 -0
  26. {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/src/gooddata_eval/core/chat/__init__.py +0 -0
  27. {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/src/gooddata_eval/core/chat/sse_client.py +0 -0
  28. {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/src/gooddata_eval/core/config.py +0 -0
  29. {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/src/gooddata_eval/core/connection.py +0 -0
  30. {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/src/gooddata_eval/core/dataset/__init__.py +0 -0
  31. {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/src/gooddata_eval/core/dataset/langfuse_source.py +0 -0
  32. {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/src/gooddata_eval/core/dataset/local.py +0 -0
  33. {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/src/gooddata_eval/core/evaluators/__init__.py +0 -0
  34. {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/src/gooddata_eval/core/evaluators/_deep_subset.py +0 -0
  35. {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/src/gooddata_eval/core/evaluators/_llm_judge.py +0 -0
  36. {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/src/gooddata_eval/core/evaluators/_text_utils.py +0 -0
  37. {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/src/gooddata_eval/core/evaluators/alert_skill.py +0 -0
  38. {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/src/gooddata_eval/core/evaluators/base.py +0 -0
  39. {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/src/gooddata_eval/core/evaluators/general_question.py +0 -0
  40. {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/src/gooddata_eval/core/evaluators/guardrail.py +0 -0
  41. {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/src/gooddata_eval/core/evaluators/metric_skill.py +0 -0
  42. {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/src/gooddata_eval/core/evaluators/search_tool.py +0 -0
  43. {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/src/gooddata_eval/core/evaluators/summary.py +0 -0
  44. {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/src/gooddata_eval/core/evaluators/visualization.py +0 -0
  45. {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/src/gooddata_eval/core/langfuse/__init__.py +0 -0
  46. {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/src/gooddata_eval/core/langfuse/sink.py +0 -0
  47. {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/src/gooddata_eval/core/models.py +0 -0
  48. {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/src/gooddata_eval/core/reporting/__init__.py +0 -0
  49. {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/src/gooddata_eval/core/reporting/console.py +0 -0
  50. {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/src/gooddata_eval/core/reporting/json_report.py +0 -0
  51. {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/src/gooddata_eval/core/runner.py +0 -0
  52. {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/src/gooddata_eval/core/scoring.py +0 -0
  53. {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/src/gooddata_eval/core/summary/__init__.py +0 -0
  54. {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/src/gooddata_eval/core/summary/http_client.py +0 -0
  55. {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/src/gooddata_eval/core/workspace.py +0 -0
  56. {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/tests/__init__.py +0 -0
  57. {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/tests/conftest.py +0 -0
  58. {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/tests/fixtures/sample_dataset/metric_skill_create.json +0 -0
  59. {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/tests/fixtures/sample_dataset/visualization_revenue.json +0 -0
  60. {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/tests/fixtures/sse_visualization_stream.txt +0 -0
  61. {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/tests/test_agentic_alert_skill.py +0 -0
  62. {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/tests/test_agentic_general_question.py +0 -0
  63. {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/tests/test_agentic_guardrail.py +0 -0
  64. {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/tests/test_agentic_kda_skill.py +0 -0
  65. {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/tests/test_agentic_langfuse_trace.py +0 -0
  66. {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/tests/test_agentic_run_context.py +0 -0
  67. {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/tests/test_agentic_search_tool.py +0 -0
  68. {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/tests/test_agentic_visualization.py +0 -0
  69. {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/tests/test_alert_skill_evaluator.py +0 -0
  70. {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/tests/test_cli.py +0 -0
  71. {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/tests/test_connection.py +0 -0
  72. {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/tests/test_deep_subset.py +0 -0
  73. {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/tests/test_langfuse_sink.py +0 -0
  74. {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/tests/test_langfuse_source.py +0 -0
  75. {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/tests/test_llm_judge.py +0 -0
  76. {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/tests/test_local_loader.py +0 -0
  77. {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/tests/test_metric_skill_evaluator.py +0 -0
  78. {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/tests/test_models.py +0 -0
  79. {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/tests/test_reporting.py +0 -0
  80. {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/tests/test_runner.py +0 -0
  81. {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/tests/test_scoring.py +0 -0
  82. {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/tests/test_search_tool_evaluator.py +0 -0
  83. {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/tests/test_sse_client.py +0 -0
  84. {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/tests/test_summary_client.py +0 -0
  85. {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/tests/test_summary_evaluator.py +0 -0
  86. {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/tests/test_text_evaluators.py +0 -0
  87. {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/tests/test_visualization_evaluator.py +0 -0
  88. {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/tests/test_workspace.py +0 -0
  89. {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/tox.ini +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: gooddata-eval
3
- Version: 1.72.1.dev5
3
+ Version: 1.72.1.dev6
4
4
  Summary: Evaluate the GoodData AI agent against your own questions and models.
5
5
  Project-URL: Source, https://github.com/gooddata/gooddata-python-sdk
6
6
  Author-email: GoodData <support@gooddata.com>
@@ -17,7 +17,7 @@ Classifier: Topic :: Scientific/Engineering
17
17
  Classifier: Topic :: Software Development
18
18
  Classifier: Typing :: Typed
19
19
  Requires-Python: >=3.10
20
- Requires-Dist: gooddata-sdk~=1.72.1.dev5
20
+ Requires-Dist: gooddata-sdk~=1.72.1.dev6
21
21
  Requires-Dist: httpx<1.0,>=0.27
22
22
  Requires-Dist: orjson<4.0.0,>=3.9.15
23
23
  Requires-Dist: pydantic<3.0,>=2.6
@@ -1,7 +1,7 @@
1
1
  # (C) 2026 GoodData Corporation
2
2
  [project]
3
3
  name = "gooddata-eval"
4
- version = "1.72.1.dev5"
4
+ version = "1.72.1.dev6"
5
5
  description = "Evaluate the GoodData AI agent against your own questions and models."
6
6
  readme = "README.md"
7
7
  license = "MIT"
@@ -11,7 +11,7 @@ authors = [
11
11
  keywords = ["gooddata", "ai", "evaluation", "llm", "analytics", "cli"]
12
12
  requires-python = ">=3.10"
13
13
  dependencies = [
14
- "gooddata-sdk~=1.72.1.dev5",
14
+ "gooddata-sdk~=1.72.1.dev6",
15
15
  "httpx>=0.27,<1.0",
16
16
  "orjson>=3.9.15,<4.0.0",
17
17
  "pydantic>=2.6,<3.0",
@@ -25,6 +25,8 @@ from gooddata_eval.core.scoring import (
25
25
 
26
26
  _REF_PATTERN = re.compile(r"\$ref:([\w_]+)\.([\w_]+)")
27
27
 
28
+ _DEFAULT_MAX_CLARIFICATION_TURNS = 7
29
+
28
30
 
29
31
  class TurnDefinition(BaseModel):
30
32
  """Definition of a single turn in a multi-turn conversation evaluation."""
@@ -192,13 +194,6 @@ def _check_output_correct(turn: TurnDefinition, chat_result: ChatResult) -> bool
192
194
  return None
193
195
 
194
196
 
195
- def _is_asking_clarification(text: str) -> bool:
196
- if not text:
197
- return False
198
- t = text.lower()
199
- return "?" in t or "could you" in t or "please" in t or "clarif" in t
200
-
201
-
202
197
  def _get_sim_user_response(agent_message: str, turn: TurnDefinition, expected_output: dict | None) -> str:
203
198
  """Generate a simulated user reply to an agent clarification question."""
204
199
  otype = turn.expected_output_type
@@ -277,7 +272,7 @@ def run_agentic_conversation(
277
272
  token: str,
278
273
  workspace_id: str,
279
274
  fixture: ConversationFixture,
280
- max_clarification_turns: int = 20,
275
+ max_clarification_turns: int = _DEFAULT_MAX_CLARIFICATION_TURNS,
281
276
  initial_conversation_id: str | None = None,
282
277
  reasoning_effort: ReasoningEffort | None = None,
283
278
  ) -> ConversationResult:
@@ -307,8 +302,22 @@ def run_agentic_conversation(
307
302
  owns_conversation = True
308
303
 
309
304
  for turn in fixture.turns:
310
- # Resolve $ref placeholders using outputs captured from prior turns.
311
- resolved_expected = _resolve_refs(turn.expected_output, turn_outputs)
305
+ try:
306
+ resolved_expected = _resolve_refs(turn.expected_output, turn_outputs)
307
+ except ValueError as exc:
308
+ print(f"[SKIP] turn '{turn.turn_id}': {exc}")
309
+ turn_results.append(
310
+ TurnResult(
311
+ turn_id=turn.turn_id,
312
+ expected_skill=turn.expected_skill,
313
+ skill_routing=False,
314
+ output_present=False,
315
+ no_error=False,
316
+ activated_skills=[],
317
+ output_correct=False,
318
+ )
319
+ )
320
+ continue
312
321
  resolved_turn = turn.model_copy(update={"expected_output": resolved_expected})
313
322
 
314
323
  clarification_turns = 0
@@ -327,13 +336,13 @@ def run_agentic_conversation(
327
336
  response_text = (chat_result.text_response or "").strip()
328
337
  if not response_text and chat_result.alert_proposals:
329
338
  response_text = render_alert_proposal(chat_result.alert_proposals[-1])
330
- asking = _is_asking_clarification(response_text) or bool(chat_result.alert_proposals)
331
- if asking and clarification_turns < max_clarification_turns:
332
- clarification_turns += 1
333
- total_clarification_turns += 1
334
- current_message = _get_sim_user_response(response_text, resolved_turn, resolved_expected)
335
- else:
339
+ if not response_text and not chat_result.tool_call_events:
340
+ break
341
+ if clarification_turns >= max_clarification_turns:
336
342
  break
343
+ clarification_turns += 1
344
+ total_clarification_turns += 1
345
+ current_message = _get_sim_user_response(response_text, resolved_turn, resolved_expected)
337
346
 
338
347
  activated = _activated_skills(all_tool_calls)
339
348
  skill_routing = turn.expected_skill in activated if activated else False
@@ -397,7 +406,7 @@ def evaluate_agentic_conversation(
397
406
  token: str,
398
407
  workspace_id: str,
399
408
  fixture: ConversationFixture,
400
- max_clarification_turns: int = 20,
409
+ max_clarification_turns: int = _DEFAULT_MAX_CLARIFICATION_TURNS,
401
410
  initial_conversation_id: str | None = None,
402
411
  langfuse: object | None = None,
403
412
  dataset_item_id: str = "",
@@ -72,16 +72,29 @@ def _best_maql_match(actual_maql: str, expected_outputs: list[dict]) -> tuple[bo
72
72
  return False, expected_outputs[0].get("maql", "") if expected_outputs else ""
73
73
 
74
74
 
75
+ class SimulatedResponseError(RuntimeError):
76
+ """The simulated user could not reply: openai missing, no API key, or the provider failed.
77
+
78
+ Carries every expected setup/provider failure so callers can end the run without
79
+ swallowing programming errors raised from the same call.
80
+ """
81
+
82
+
75
83
  def generate_simulated_response(agent_message: str, expected_output: dict) -> str:
76
- """Generate a user reply to keep the metric-skill conversation going (gpt-4o-mini)."""
84
+ """Generate a user reply to keep the metric-skill conversation going (gpt-4o-mini).
85
+
86
+ Raises:
87
+ SimulatedResponseError: openai is not installed, OPENAI_API_KEY is unset, or the
88
+ provider call failed.
89
+ """
77
90
  try:
78
- from openai import OpenAI # noqa: PLC0415
91
+ from openai import OpenAI, OpenAIError # noqa: PLC0415
79
92
  except ImportError as exc:
80
- raise RuntimeError("openai package is required for generate_simulated_response") from exc
93
+ raise SimulatedResponseError("openai package is required for generate_simulated_response") from exc
81
94
 
82
95
  api_key = os.environ.get("OPENAI_API_KEY")
83
96
  if not api_key:
84
- raise OSError("OPENAI_API_KEY environment variable is not set")
97
+ raise SimulatedResponseError("OPENAI_API_KEY environment variable is not set")
85
98
 
86
99
  client = OpenAI(api_key=api_key)
87
100
  expected_maql = expected_output.get("maql", "")
@@ -91,12 +104,15 @@ def generate_simulated_response(agent_message: str, expected_output: dict) -> st
91
104
  f"The user originally asked to create a metric with MAQL: {expected_maql}. "
92
105
  f"Reply briefly as the user, providing any clarification the assistant needs."
93
106
  )
94
- response = client.chat.completions.create(
95
- model="gpt-4o-mini",
96
- messages=[{"role": "user", "content": prompt}],
97
- max_tokens=150,
98
- temperature=0,
99
- )
107
+ try:
108
+ response = client.chat.completions.create(
109
+ model="gpt-4o-mini",
110
+ messages=[{"role": "user", "content": prompt}],
111
+ max_tokens=150,
112
+ temperature=0,
113
+ )
114
+ except OpenAIError as exc:
115
+ raise SimulatedResponseError(f"simulated user reply failed: {exc}") from exc
100
116
  return response.choices[0].message.content or "Please proceed."
101
117
 
102
118
 
@@ -166,13 +182,6 @@ def _delete_metric(sdk: GoodDataSdk, workspace_id: str, metric_id: str) -> None:
166
182
  print(f"[CLEANUP] Failed to delete metric {metric_id}: {exc}")
167
183
 
168
184
 
169
- def _is_asking_clarification(text: str) -> bool:
170
- if not text:
171
- return False
172
- t = text.lower()
173
- return "?" in t or "could you" in t or "please provide" in t or "clarif" in t
174
-
175
-
176
185
  def _execute_single_metric_run(
177
186
  client: ChatClient,
178
187
  sdk: GoodDataSdk,
@@ -204,9 +213,14 @@ def _execute_single_metric_run(
204
213
  metric_id_to_delete = candidate.get("metric_id")
205
214
  break
206
215
  response_text = (chat_result.text_response or "").strip()
207
- if _is_asking_clarification(response_text):
216
+ if not response_text and not chat_result.tool_call_events:
217
+ break
218
+ if _iteration >= max_iterations - 1:
219
+ break
220
+ try:
208
221
  current_question = generate_simulated_response(response_text, primary_expected)
209
- else:
222
+ except SimulatedResponseError as exc:
223
+ print(f"[SIM-USER] Simulated reply failed for conversation {conversation_id}: {exc}")
210
224
  break
211
225
 
212
226
  actual_maql = (metric_result or {}).get("maql", "")
@@ -352,3 +352,118 @@ def test_run_agentic_conversation_treats_alert_proposal_as_a_clarification():
352
352
  assert "Should I create this alert?" in mock_sim.call_args.args[0]
353
353
  assert result.turn_results[0].clarification_turns_used == 1
354
354
  assert result.turn_results[0].skill_success is True
355
+
356
+
357
+ def _viz_turn_result(text=None, viz=None, tool_calls=()):
358
+ r = MagicMock()
359
+ r.text_response = text
360
+ r.created_visualizations = viz
361
+ r.tool_call_events = list(tool_calls)
362
+ r.alert_proposals = []
363
+ return r
364
+
365
+
366
+ def test_run_agentic_conversation_replies_to_a_statement_without_a_question_mark():
367
+ """QA-28982 regression: gpt-5.2 answered "I need to confirm ... Next I'll:" -- no question
368
+ mark, so the old substring heuristic ended the turn and no metric was ever created."""
369
+ mock_client = MagicMock()
370
+ mock_client.create_conversation.return_value = "conv-1"
371
+ stalling_turn = _viz_turn_result(
372
+ text="I can create that, but first I need to confirm which Net Sales calculation to use. Next I'll: ...",
373
+ tool_calls=[_skills_tc("metric")],
374
+ )
375
+ mock_client.send_message.side_effect = [
376
+ stalling_turn,
377
+ _metric_turn_result([_skills_tc("metric"), _create_metric_tc("m1")]),
378
+ ]
379
+
380
+ with (
381
+ patch("gooddata_eval.core.agentic.conversation.ChatClient", return_value=mock_client),
382
+ patch("gooddata_eval.core.agentic.conversation.GoodDataSdk"),
383
+ patch(
384
+ "gooddata_eval.core.agentic.conversation._get_sim_user_response",
385
+ return_value="Go ahead with Net Sales.",
386
+ ) as mock_sim,
387
+ ):
388
+ result = run_agentic_conversation(
389
+ host="http://host/api/v1/actions/workspaces/ws1/ai",
390
+ token="tok",
391
+ workspace_id="ws1",
392
+ fixture=_two_metric_turn_fixture().model_copy(update={"turns": _two_metric_turn_fixture().turns[:1]}),
393
+ )
394
+
395
+ mock_sim.assert_called_once()
396
+ assert result.turn_results[0].clarification_turns_used == 1
397
+ assert result.turn_results[0].skill_success is True
398
+
399
+
400
+ def test_run_agentic_conversation_stops_when_the_agent_says_nothing():
401
+ """An agent that returns neither text nor tool calls is stuck -- no point replying to it."""
402
+ mock_client = MagicMock()
403
+ mock_client.create_conversation.return_value = "conv-1"
404
+ mock_client.send_message.return_value = _viz_turn_result(text=None)
405
+
406
+ with (
407
+ patch("gooddata_eval.core.agentic.conversation.ChatClient", return_value=mock_client),
408
+ patch("gooddata_eval.core.agentic.conversation.GoodDataSdk"),
409
+ patch("gooddata_eval.core.agentic.conversation._get_sim_user_response") as mock_sim,
410
+ ):
411
+ result = run_agentic_conversation(
412
+ host="http://host/api/v1/actions/workspaces/ws1/ai",
413
+ token="tok",
414
+ workspace_id="ws1",
415
+ fixture=_two_metric_turn_fixture().model_copy(update={"turns": _two_metric_turn_fixture().turns[:1]}),
416
+ )
417
+
418
+ mock_sim.assert_not_called()
419
+ assert mock_client.send_message.call_count == 1
420
+ assert result.turn_results[0].skill_success is False
421
+
422
+
423
+ def test_run_agentic_conversation_records_a_failed_turn_when_a_ref_cannot_be_resolved():
424
+ """QA-28982 regression: turn 1 producing no metric used to raise ValueError out of the whole
425
+ run, hiding which turn broke and skipping every later turn."""
426
+ mock_client = MagicMock()
427
+ mock_client.create_conversation.return_value = "conv-1"
428
+ mock_client.send_message.side_effect = [
429
+ _viz_turn_result(text="Which Net Sales metric?", tool_calls=[_skills_tc("metric")]),
430
+ _viz_turn_result(text="Working on it.", tool_calls=[_skills_tc("metric")]),
431
+ _metric_turn_result([_skills_tc("metric"), _create_metric_tc("m2")]),
432
+ ]
433
+ fixture = ConversationFixture(
434
+ id="test-ref",
435
+ expected_skills=["metric"],
436
+ turns=[
437
+ TurnDefinition(
438
+ turn_id="t1", message="Create shared", expected_skill="metric", expected_output_type="metric"
439
+ ),
440
+ TurnDefinition(
441
+ turn_id="t2",
442
+ message="Chart it",
443
+ expected_skill="visualization",
444
+ expected_output={"metrics": ["metric/$ref:t1.metric_id"]},
445
+ ),
446
+ TurnDefinition(
447
+ turn_id="t3", message="Create another", expected_skill="metric", expected_output_type="metric"
448
+ ),
449
+ ],
450
+ )
451
+
452
+ with (
453
+ patch("gooddata_eval.core.agentic.conversation.ChatClient", return_value=mock_client),
454
+ patch("gooddata_eval.core.agentic.conversation.GoodDataSdk"),
455
+ patch("gooddata_eval.core.agentic.conversation._get_sim_user_response", return_value="Go ahead."),
456
+ ):
457
+ result = run_agentic_conversation(
458
+ host="http://host/api/v1/actions/workspaces/ws1/ai",
459
+ token="tok",
460
+ workspace_id="ws1",
461
+ fixture=fixture,
462
+ max_clarification_turns=1,
463
+ )
464
+
465
+ assert [t.turn_id for t in result.turn_results] == ["t1", "t2", "t3"]
466
+ assert result.turn_results[0].skill_success is False
467
+ assert result.turn_results[1].no_error is False
468
+ assert result.turn_results[2].skill_success is True
469
+ assert result.conversation_success is False
@@ -1,13 +1,17 @@
1
1
  # (C) 2026 GoodData Corporation. All rights reserved.
2
2
  # SPDX-License-Identifier: LicenseRef-GoodData-Enterprise
3
+ import os
4
+ import sys
3
5
  from unittest.mock import MagicMock, patch
4
6
 
5
7
  import pytest
6
8
  from gooddata_eval.core.agentic.metric_skill import (
7
9
  AgenticMetricSummary,
8
10
  MetricRunResult,
11
+ SimulatedResponseError,
9
12
  _delete_metric,
10
13
  _normalize_maql,
14
+ generate_simulated_response,
11
15
  run_agentic_metric_skill,
12
16
  )
13
17
  from gooddata_eval.core.models import ChatResult
@@ -83,7 +87,13 @@ def test_run_agentic_metric_skill_closes_client_on_no_result():
83
87
  "reasoningStepCount": 1,
84
88
  }
85
89
  )
86
- with patch("gooddata_eval.core.agentic.metric_skill.ChatClient", return_value=mock_client):
90
+ with (
91
+ patch("gooddata_eval.core.agentic.metric_skill.ChatClient", return_value=mock_client),
92
+ patch(
93
+ "gooddata_eval.core.agentic.metric_skill.generate_simulated_response",
94
+ return_value="Go ahead and create it.",
95
+ ) as mock_sim,
96
+ ):
87
97
  summary = run_agentic_metric_skill(
88
98
  host="http://host/api/v1/actions/workspaces/ws1/ai",
89
99
  token="tok",
@@ -96,6 +106,7 @@ def test_run_agentic_metric_skill_closes_client_on_no_result():
96
106
  mock_client.close.assert_called_once()
97
107
  assert summary.pass_at_k is False
98
108
  assert summary.best.metric_created is False
109
+ mock_sim.assert_called_once_with("I will work on that.", {"maql": "SELECT {metric/foo}"})
99
110
 
100
111
 
101
112
  def test_run_agentic_metric_skill_uses_initial_conversation_for_run_0():
@@ -224,3 +235,52 @@ def test_run_agentic_metric_skill_deletes_metric_even_when_teardown_fails():
224
235
  )
225
236
 
226
237
  mock_sdk._client.entities_api.delete_entity_metrics.assert_called_once_with("ws1", "foo_metric")
238
+
239
+
240
+ def test_generate_simulated_response_without_an_api_key():
241
+ with (
242
+ patch.dict(sys.modules, {"openai": MagicMock()}),
243
+ patch.dict(os.environ, {}, clear=True),
244
+ pytest.raises(SimulatedResponseError, match="OPENAI_API_KEY"),
245
+ ):
246
+ generate_simulated_response("Which brand field?", {"maql": "SELECT {metric/foo}"})
247
+
248
+
249
+ def test_generate_simulated_response_without_the_openai_package():
250
+ with (
251
+ patch.dict(sys.modules, {"openai": None}),
252
+ pytest.raises(SimulatedResponseError, match="openai package is required"),
253
+ ):
254
+ generate_simulated_response("Which brand field?", {"maql": "SELECT {metric/foo}"})
255
+
256
+
257
+ def test_run_agentic_metric_skill_fails_the_run_when_the_simulated_reply_cannot_be_generated():
258
+ exc = SimulatedResponseError("OPENAI_API_KEY environment variable is not set")
259
+ mock_client = MagicMock()
260
+ mock_client.create_conversation.return_value = "conv-1"
261
+ mock_client.send_message.return_value = ChatResult.model_validate(
262
+ {
263
+ "textResponse": "Which brand field should I count?",
264
+ "toolCallEvents": [],
265
+ "reasoningStepCount": 1,
266
+ }
267
+ )
268
+ with (
269
+ patch("gooddata_eval.core.agentic.metric_skill.ChatClient", return_value=mock_client),
270
+ patch("gooddata_eval.core.agentic.metric_skill.generate_simulated_response", side_effect=exc) as mock_sim,
271
+ ):
272
+ summary = run_agentic_metric_skill(
273
+ host="http://host/api/v1/actions/workspaces/ws1/ai",
274
+ token="tok",
275
+ workspace_id="ws1",
276
+ question="Create metric foo",
277
+ expected_output={"maql": "SELECT {metric/foo}"},
278
+ k=1,
279
+ max_iterations=3,
280
+ )
281
+
282
+ assert summary.pass_at_k is False
283
+ assert summary.best.metric_created is False
284
+ assert summary.best.total_turns == 1.0
285
+ mock_client.close.assert_called_once()
286
+ mock_sim.assert_called_once_with("Which brand field should I count?", {"maql": "SELECT {metric/foo}"})