gooddata-eval 1.72.1.dev3__tar.gz → 1.72.1.dev4__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (89) hide show
  1. {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/PKG-INFO +2 -2
  2. {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/pyproject.toml +2 -2
  3. {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/agentic/kda_skill.py +89 -49
  4. {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/tests/test_agentic_kda_skill.py +167 -36
  5. {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/.gitignore +0 -0
  6. {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/LICENSE.txt +0 -0
  7. {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/Makefile +0 -0
  8. {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/README.md +0 -0
  9. {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/__init__.py +0 -0
  10. {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/_version.py +0 -0
  11. {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/cli/__init__.py +0 -0
  12. {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/cli/agentic_runner.py +0 -0
  13. {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/cli/main.py +0 -0
  14. {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/__init__.py +0 -0
  15. {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/agentic/__init__.py +0 -0
  16. {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/agentic/_catalog.py +0 -0
  17. {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/agentic/_langfuse.py +0 -0
  18. {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/agentic/alert_skill.py +0 -0
  19. {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/agentic/conversation.py +0 -0
  20. {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/agentic/general_question.py +0 -0
  21. {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/agentic/guardrail.py +0 -0
  22. {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/agentic/metric_skill.py +0 -0
  23. {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/agentic/search_tool.py +0 -0
  24. {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/agentic/visualization.py +0 -0
  25. {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/chat/__init__.py +0 -0
  26. {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/chat/sse_client.py +0 -0
  27. {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/config.py +0 -0
  28. {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/connection.py +0 -0
  29. {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/dataset/__init__.py +0 -0
  30. {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/dataset/langfuse_source.py +0 -0
  31. {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/dataset/local.py +0 -0
  32. {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/evaluators/__init__.py +0 -0
  33. {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/evaluators/_deep_subset.py +0 -0
  34. {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/evaluators/_llm_judge.py +0 -0
  35. {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/evaluators/_text_utils.py +0 -0
  36. {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/evaluators/alert_skill.py +0 -0
  37. {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/evaluators/base.py +0 -0
  38. {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/evaluators/general_question.py +0 -0
  39. {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/evaluators/guardrail.py +0 -0
  40. {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/evaluators/metric_skill.py +0 -0
  41. {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/evaluators/search_tool.py +0 -0
  42. {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/evaluators/summary.py +0 -0
  43. {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/evaluators/visualization.py +0 -0
  44. {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/langfuse/__init__.py +0 -0
  45. {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/langfuse/sink.py +0 -0
  46. {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/models.py +0 -0
  47. {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/reporting/__init__.py +0 -0
  48. {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/reporting/console.py +0 -0
  49. {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/reporting/json_report.py +0 -0
  50. {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/runner.py +0 -0
  51. {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/scoring.py +0 -0
  52. {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/summary/__init__.py +0 -0
  53. {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/summary/http_client.py +0 -0
  54. {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/workspace.py +0 -0
  55. {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/tests/__init__.py +0 -0
  56. {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/tests/conftest.py +0 -0
  57. {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/tests/fixtures/sample_dataset/metric_skill_create.json +0 -0
  58. {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/tests/fixtures/sample_dataset/visualization_revenue.json +0 -0
  59. {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/tests/fixtures/sse_visualization_stream.txt +0 -0
  60. {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/tests/test_agentic_alert_skill.py +0 -0
  61. {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/tests/test_agentic_conversation.py +0 -0
  62. {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/tests/test_agentic_general_question.py +0 -0
  63. {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/tests/test_agentic_guardrail.py +0 -0
  64. {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/tests/test_agentic_langfuse_trace.py +0 -0
  65. {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/tests/test_agentic_metric_skill.py +0 -0
  66. {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/tests/test_agentic_run_context.py +0 -0
  67. {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/tests/test_agentic_search_tool.py +0 -0
  68. {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/tests/test_agentic_visualization.py +0 -0
  69. {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/tests/test_alert_skill_evaluator.py +0 -0
  70. {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/tests/test_cli.py +0 -0
  71. {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/tests/test_connection.py +0 -0
  72. {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/tests/test_deep_subset.py +0 -0
  73. {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/tests/test_langfuse_sink.py +0 -0
  74. {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/tests/test_langfuse_source.py +0 -0
  75. {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/tests/test_llm_judge.py +0 -0
  76. {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/tests/test_local_loader.py +0 -0
  77. {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/tests/test_metric_skill_evaluator.py +0 -0
  78. {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/tests/test_models.py +0 -0
  79. {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/tests/test_reporting.py +0 -0
  80. {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/tests/test_runner.py +0 -0
  81. {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/tests/test_scoring.py +0 -0
  82. {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/tests/test_search_tool_evaluator.py +0 -0
  83. {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/tests/test_sse_client.py +0 -0
  84. {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/tests/test_summary_client.py +0 -0
  85. {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/tests/test_summary_evaluator.py +0 -0
  86. {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/tests/test_text_evaluators.py +0 -0
  87. {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/tests/test_visualization_evaluator.py +0 -0
  88. {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/tests/test_workspace.py +0 -0
  89. {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/tox.ini +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: gooddata-eval
3
- Version: 1.72.1.dev3
3
+ Version: 1.72.1.dev4
4
4
  Summary: Evaluate the GoodData AI agent against your own questions and models.
5
5
  Project-URL: Source, https://github.com/gooddata/gooddata-python-sdk
6
6
  Author-email: GoodData <support@gooddata.com>
@@ -17,7 +17,7 @@ Classifier: Topic :: Scientific/Engineering
17
17
  Classifier: Topic :: Software Development
18
18
  Classifier: Typing :: Typed
19
19
  Requires-Python: >=3.10
20
- Requires-Dist: gooddata-sdk~=1.72.1.dev3
20
+ Requires-Dist: gooddata-sdk~=1.72.1.dev4
21
21
  Requires-Dist: httpx<1.0,>=0.27
22
22
  Requires-Dist: orjson<4.0.0,>=3.9.15
23
23
  Requires-Dist: pydantic<3.0,>=2.6
@@ -1,7 +1,7 @@
1
1
  # (C) 2026 GoodData Corporation
2
2
  [project]
3
3
  name = "gooddata-eval"
4
- version = "1.72.1.dev3"
4
+ version = "1.72.1.dev4"
5
5
  description = "Evaluate the GoodData AI agent against your own questions and models."
6
6
  readme = "README.md"
7
7
  license = "MIT"
@@ -11,7 +11,7 @@ authors = [
11
11
  keywords = ["gooddata", "ai", "evaluation", "llm", "analytics", "cli"]
12
12
  requires-python = ">=3.10"
13
13
  dependencies = [
14
- "gooddata-sdk~=1.72.1.dev3",
14
+ "gooddata-sdk~=1.72.1.dev4",
15
15
  "httpx>=0.27,<1.0",
16
16
  "orjson>=3.9.15,<4.0.0",
17
17
  "pydantic>=2.6,<3.0",
@@ -5,7 +5,6 @@ from __future__ import annotations
5
5
 
6
6
  import logging
7
7
  import os
8
- import re
9
8
  from dataclasses import dataclass
10
9
 
11
10
  from gooddata_eval.core.chat.sse_client import ChatClient
@@ -15,36 +14,79 @@ from gooddata_eval.core.models import ToolCallEvent
15
14
  _log = logging.getLogger(__name__)
16
15
 
17
16
  _DEFAULT_K = 1
18
- # Disambiguation safety net only (create+execute always run together in the same
19
- # turn) -- 3 covers metric and period each needing their own clarifying question.
20
- _DEFAULT_MAX_ITERATIONS = 3
17
+ # Disambiguation safety net only (create+execute always run together in the same turn) --
18
+ # 3 real questions' worth (metric, period, +1 slack) since a simulated reply is now sent on
19
+ # every non-final turn (see run_agentic_kda_skill), not just ones classified as a question.
20
+ _DEFAULT_MAX_ITERATIONS = 4
21
21
 
22
22
 
23
- def _is_asking_kda_clarification(text: str) -> bool:
24
- """True if ``text`` reads as the agent asking for input, not a final answer.
25
-
26
- KDA-specific, not shared with metric_skill.py/conversation.py -- each skill's
27
- disambiguation heuristic has already drifted independently. Requires the text to
28
- end on "?" (a "?" anywhere also matches a final answer that merely quotes one).
23
+ def _build_period_hint(expected_output: dict) -> str | None:
24
+ """Build a period hint from whichever of expected_output's Date Attribute/Analyzed
25
+ Period/Reference Period are present -- a question about only one of them (e.g. "which
26
+ date dimension?") must still get an answerable hint, not None just because the other
27
+ two are absent.
28
+ """
29
+ date_attr = expected_output.get("Date Attribute")
30
+ analyzed = expected_output.get("Analyzed Period")
31
+ reference_period = expected_output.get("Reference Period")
32
+ if not (date_attr or analyzed or reference_period):
33
+ return None
34
+ parts = []
35
+ if date_attr:
36
+ parts.append(date_attr)
37
+ if analyzed and reference_period:
38
+ parts.append(f"comparing {analyzed} to {reference_period}")
39
+ elif analyzed:
40
+ parts.append(f"period {analyzed}")
41
+ elif reference_period:
42
+ parts.append(f"compared to {reference_period}")
43
+ return ", ".join(parts)
44
+
45
+
46
+ def _build_clarification_prompt(
47
+ agent_message: str, measure_candidates: dict | list[dict] | None, period_hint: str | None
48
+ ) -> str:
49
+ """Build the simulated-user prompt, referencing only whatever candidates/period-hint
50
+ are actually usable -- an empty/None candidate must drop the "acceptable metric/fact"
51
+ clause entirely rather than assert a literal "None" as if it were a real option.
29
52
  """
30
- if not text:
31
- return False
32
- t = text.strip().lower()
33
- if t.endswith("?"):
34
- return True
35
- # "To clarify, ..." means "in other words" (a final answer), not a request for one --
36
- # strip it first so "clarif" below only matches genuine clarification requests.
37
- t = re.sub(r"^(just )?to clarify,?\s*", "", t)
38
- return "could you" in t or "please provide" in t or "clarif" in t
39
-
40
-
41
- def generate_simulated_kda_response(agent_message: str, measure_candidates: dict | list[dict] | None) -> str:
53
+ candidates = [
54
+ c for c in (measure_candidates if isinstance(measure_candidates, list) else [measure_candidates]) if c
55
+ ]
56
+ reference = ""
57
+ if candidates:
58
+ candidate_desc = "; or ".join(
59
+ f"{c.get('type')} '{c.get('id')}'" + (f" (aggregation {c['aggregation']})" if c.get("aggregation") else "")
60
+ for c in candidates
61
+ )
62
+ reference = f"an acceptable metric/fact is {candidate_desc}"
63
+ if period_hint:
64
+ reference = (
65
+ f"{reference}; the intended time period is {period_hint}"
66
+ if reference
67
+ else f"the intended time period is {period_hint}"
68
+ )
69
+ return (
70
+ f"You are simulating a user in a conversation with a BI assistant that runs key driver "
71
+ f"analysis. The assistant asked: '{agent_message}'. "
72
+ + (f"For reference, {reference}. " if reference else "")
73
+ + "Reply briefly as the user, answering whichever of those the assistant actually asked about."
74
+ )
75
+
76
+
77
+ def generate_simulated_kda_response(
78
+ agent_message: str,
79
+ measure_candidates: dict | list[dict] | None,
80
+ period_hint: str | None = None,
81
+ ) -> str:
42
82
  """Generate a user reply to keep the KDA-skill conversation going (gpt-4o-mini).
43
83
 
44
- Used only when the agent asks a clarifying question instead of triggering KDA
45
- directly. Picks *any* candidate from ``measure_candidates`` -- scope only needs KDA
46
- to trigger, not the resulting measure to be exactly right. Always OpenAI regardless
47
- of the combo's own provider -- this is test-harness plumbing, not the system under test.
84
+ Called on any turn that didn't trigger KDA, whatever the agent's response actually
85
+ said -- most often a clarifying question about the measure, the period, or both, so
86
+ both are given as reference and the reply answers whichever was actually asked.
87
+ Scope only needs KDA to trigger, not the resulting measure/period to be exactly
88
+ right. Always OpenAI regardless of the combo's own provider -- this is
89
+ test-harness plumbing, not the system under test.
48
90
  """
49
91
  try:
50
92
  from openai import OpenAI # noqa: PLC0415
@@ -56,17 +98,7 @@ def generate_simulated_kda_response(agent_message: str, measure_candidates: dict
56
98
  raise OSError("OPENAI_API_KEY environment variable is not set")
57
99
 
58
100
  client = OpenAI(api_key=api_key)
59
- candidates = measure_candidates if isinstance(measure_candidates, list) else [measure_candidates or {}]
60
- candidate_desc = "; or ".join(
61
- f"{c.get('type')} '{c.get('id')}'" + (f" (aggregation {c['aggregation']})" if c.get("aggregation") else "")
62
- for c in candidates
63
- )
64
- prompt = (
65
- f"You are simulating a user in a conversation with a BI assistant that runs key driver "
66
- f"analysis. The assistant said: '{agent_message}'. "
67
- f"The user is happy to proceed with any of the following: {candidate_desc}. "
68
- f"Reply briefly as the user, picking whichever of those the assistant offered."
69
- )
101
+ prompt = _build_clarification_prompt(agent_message, measure_candidates, period_hint)
70
102
  response = client.chat.completions.create(
71
103
  model="gpt-4o-mini",
72
104
  messages=[{"role": "user", "content": prompt}],
@@ -171,9 +203,12 @@ def run_agentic_kda_skill(
171
203
 
172
204
  Each run is normally one message, one turn -- create and execute are always called
173
205
  together in the same turn (the skill's own system prompt: "NO confirmation needed").
174
- The only thing that can extend a run up to ``max_iterations`` turns is the agent
175
- asking a clarifying question instead of triggering KDA directly; a simulated user
176
- reply nudges it forward.
206
+ A run only extends past turn 1, up to ``max_iterations``, when the agent's response
207
+ has no create call and isn't empty; a simulated user reply is then always sent, with
208
+ no attempt to classify whether the text was actually asking for input (matching
209
+ visualization.py/alert_skill.py's own break conditions) -- missing a genuine
210
+ clarifying question hard-fails the run, while sending one after an unrecognized final
211
+ answer only costs one harmless extra turn, so the asymmetry favors never guessing.
177
212
  """
178
213
  if k < 1:
179
214
  # k=0 or negative would otherwise silently run once, indistinguishable from k=1.
@@ -212,17 +247,22 @@ def run_agentic_kda_skill(
212
247
  # execute tool isn't available at all when data-sharing is off for the org).
213
248
  turn_wall_clock_sec = chat_result.turn_wall_clock_sec
214
249
  break
250
+ if not response_text:
251
+ break
215
252
  if iteration >= max_iterations - 1:
216
253
  break
217
- if _is_asking_kda_clarification(response_text):
218
- measure_candidates = expected_output.get("Measure") if isinstance(expected_output, dict) else None
219
- try:
220
- current_question = generate_simulated_kda_response(response_text, measure_candidates)
221
- disambiguated = True
222
- except Exception as exc: # noqa: BLE001 -- safety net, not the assertion; end only this run
223
- _log.warning("Simulated KDA user reply failed for conversation %s: %s", conv_id, exc)
224
- break
225
- else:
254
+ # No text classification -- matches visualization.py/alert_skill.py: break only on
255
+ # the goal signal (create_args set) or an empty response, otherwise always send a
256
+ # simulated reply. A false positive (agent had already given a final answer) costs
257
+ # one harmless extra turn; a false negative (missing a genuine clarifying question)
258
+ # would hard-fail the run, so the asymmetry favors never trying to tell them apart.
259
+ measure_candidates = expected_output.get("Measure") if isinstance(expected_output, dict) else None
260
+ period_hint = _build_period_hint(expected_output) if isinstance(expected_output, dict) else None
261
+ try:
262
+ current_question = generate_simulated_kda_response(response_text, measure_candidates, period_hint)
263
+ disambiguated = True
264
+ except Exception as exc: # noqa: BLE001 -- safety net, not the assertion; end only this run
265
+ _log.warning("Simulated KDA user reply failed for conversation %s: %s", conv_id, exc)
226
266
  break
227
267
 
228
268
  ev = _evaluate_run(create_args, execute_result, turn_completed, disambiguated)
@@ -8,9 +8,10 @@ import pytest
8
8
  from gooddata_eval.core.agentic.kda_skill import (
9
9
  KdaEvaluation,
10
10
  KdaSkillAssertionError,
11
+ _build_clarification_prompt,
12
+ _build_period_hint,
11
13
  _evaluate_run,
12
14
  _extract_kda_calls,
13
- _is_asking_kda_clarification,
14
15
  evaluate_agentic_kda_skill,
15
16
  run_agentic_kda_skill,
16
17
  )
@@ -67,51 +68,59 @@ def _no_kda_chat_result(
67
68
 
68
69
 
69
70
  # --------------------------------------------------------------------------- #
70
- # _is_asking_kda_clarification
71
+ # _build_clarification_prompt
71
72
  # --------------------------------------------------------------------------- #
72
- @pytest.mark.parametrize(
73
- "text",
74
- ["Could you clarify which metric?", "Please provide the date range.", "Did you mean revenue?"],
75
- )
76
- def test_is_asking_kda_clarification_true(text):
77
- assert _is_asking_kda_clarification(text) is True
73
+ def test_build_clarification_prompt_omits_reference_clause_when_no_candidates_or_period():
74
+ # Regression (chi My's review): with no usable candidates, the old code still asserted
75
+ # "an acceptable metric/fact is None 'None'" as if it were a real option -- likely to
76
+ # make the simulated user invent a metric literally named "None". No candidates and no
77
+ # period hint must drop the whole "For reference, ..." clause instead.
78
+ prompt = _build_clarification_prompt("Which date range?", None, None)
79
+ assert "None" not in prompt
80
+ assert "For reference" not in prompt
78
81
 
79
82
 
80
- def test_is_asking_kda_clarification_false_on_plain_statement():
81
- assert _is_asking_kda_clarification("Here is the key driver analysis result.") is False
83
+ def test_build_clarification_prompt_includes_only_period_hint_when_no_candidates():
84
+ prompt = _build_clarification_prompt("Which period?", None, "2026-2 vs 2026-1")
85
+ assert "None" not in prompt
86
+ assert "the intended time period is 2026-2 vs 2026-1" in prompt
82
87
 
83
88
 
84
- def test_is_asking_kda_clarification_false_on_empty():
85
- assert _is_asking_kda_clarification("") is False
89
+ def test_build_clarification_prompt_includes_candidates_and_period_hint():
90
+ prompt = _build_clarification_prompt(
91
+ "Which metric and period?", {"type": "metric", "id": "revenue"}, "2026-2 vs 2026-1"
92
+ )
93
+ assert "an acceptable metric/fact is metric 'revenue'" in prompt
94
+ assert "the intended time period is 2026-2 vs 2026-1" in prompt
86
95
 
87
96
 
88
- def test_is_asking_kda_clarification_false_when_question_mark_is_not_the_final_answer():
89
- # Regression guard for the original bug: a final answer that merely quotes or
90
- # rhetorically references a question must not be mistaken for a clarifying question.
91
- text = 'The user asked "what changed?" so here is the key driver breakdown they requested.'
92
- assert _is_asking_kda_clarification(text) is False
97
+ # --------------------------------------------------------------------------- #
98
+ # _build_period_hint
99
+ # --------------------------------------------------------------------------- #
100
+ def test_build_period_hint_none_when_no_period_fields_present():
101
+ assert _build_period_hint({"Measure": {"type": "metric", "id": "revenue"}}) is None
93
102
 
94
103
 
95
- @pytest.mark.parametrize(
96
- "text",
97
- [
98
- "To clarify, revenue rose 12% quarter over quarter.",
99
- "Just to clarify, the increase was driven by the South region.",
100
- ],
101
- )
102
- def test_is_asking_kda_clarification_false_on_to_clarify_discourse_marker(text):
103
- # Regression guard: "to clarify, ..." is a discourse marker ("in other words") that
104
- # introduces a restated FINAL answer, not a request for one -- the bare "clarif" in t
105
- # substring check would otherwise mistake this for a clarifying question and burn a
106
- # simulated-reply turn on an answer that was already complete.
107
- assert _is_asking_kda_clarification(text) is False
104
+ def test_build_period_hint_all_three_fields():
105
+ hint = _build_period_hint(
106
+ {"Date Attribute": "transaction_date.quarter", "Analyzed Period": "2026-2", "Reference Period": "2026-1"}
107
+ )
108
+ assert hint == "transaction_date.quarter, comparing 2026-2 to 2026-1"
108
109
 
109
110
 
110
- def test_is_asking_kda_clarification_true_for_genuine_clarify_request_despite_marker_strip():
111
- # The discourse-marker strip must not eat a genuine request that happens to start the
112
- # same way it's phrased in practice. No trailing "?" here specifically so this exercises
113
- # the "could you" substring check post-strip, not the separate endswith("?") check.
114
- assert _is_asking_kda_clarification("To clarify, could you tell me which region you mean") is True
111
+ def test_build_period_hint_date_attribute_only():
112
+ # A dataset item carrying only Date Attribute (agent asks "which date dimension should
113
+ # I use?") must still get an answerable hint -- this used to require all 3 fields and
114
+ # reproduced the same gap the metric-clarification fix closed, just narrower.
115
+ assert _build_period_hint({"Date Attribute": "transaction_date.quarter"}) == "transaction_date.quarter"
116
+
117
+
118
+ def test_build_period_hint_analyzed_period_only():
119
+ assert _build_period_hint({"Analyzed Period": "2026-2"}) == "period 2026-2"
120
+
121
+
122
+ def test_build_period_hint_reference_period_only():
123
+ assert _build_period_hint({"Reference Period": "2026-1"}) == "compared to 2026-1"
115
124
 
116
125
 
117
126
  # --------------------------------------------------------------------------- #
@@ -427,6 +436,128 @@ def test_run_agentic_kda_skill_marks_disambiguated_after_a_simulated_reply():
427
436
  assert summary.best.evaluation.triggered is True
428
437
 
429
438
 
439
+ def test_run_agentic_kda_skill_disambiguates_on_question_followed_by_option_list():
440
+ # Regression (QA-28800): the real captured response ends with a bullet list of
441
+ # candidate metrics. Before this module dropped text classification in favor of
442
+ # always retrying on a non-empty, non-triggering response (matching
443
+ # visualization.py/alert_skill.py), a heuristic that only matched "?" endings gave
444
+ # up after turn 1 (triggered=False) instead of ever nudging the simulated user to
445
+ # pick one.
446
+ mock_client = MagicMock()
447
+ mock_client.create_conversation.return_value = "conv-1"
448
+ mock_client.send_message.side_effect = [
449
+ _no_kda_chat_result(
450
+ 'I found two different "Total Net Revenue" metrics in your data model. '
451
+ "Which one should I analyze for the 2024 vs 2023 drop?\n\n"
452
+ "- {metric/metric_l1_sql_net_sales_summary_net_revenue}\n"
453
+ "- {metric/metric_l1_total_net_revenue}"
454
+ ),
455
+ _kda_chat_result(success=True),
456
+ ]
457
+
458
+ with (
459
+ patch("gooddata_eval.core.agentic.kda_skill.ChatClient", return_value=mock_client),
460
+ patch(
461
+ "gooddata_eval.core.agentic.kda_skill.generate_simulated_kda_response",
462
+ return_value="Use metric_l1_sql_net_sales_summary_net_revenue.",
463
+ ) as mock_simulate,
464
+ ):
465
+ summary = run_agentic_kda_skill(
466
+ host="http://host/api/v1/actions/workspaces/ws1/ai",
467
+ token="tok",
468
+ workspace_id="ws1",
469
+ question="Why did Total Net Revenue of Net Sales Summary drop in 2024 compared to 2023?",
470
+ expected_output=_EXPECTED,
471
+ k=1,
472
+ max_iterations=2,
473
+ )
474
+
475
+ mock_simulate.assert_called_once()
476
+ assert summary.best.evaluation.disambiguated is True
477
+ assert summary.best.evaluation.triggered is True
478
+ assert mock_client.send_message.call_count == 2
479
+
480
+
481
+ def test_run_agentic_kda_skill_retries_on_bold_markdown_option_list_with_no_space():
482
+ # Regression (chi My's review): a prior classifier-based fix required a space right
483
+ # after the list marker, so "**Option 1**: revenue" (bold markdown, no space between
484
+ # the two asterisks) would have been misread as a final answer. Dropping content
485
+ # classification entirely (see run_agentic_kda_skill's docstring) makes this -- and any
486
+ # other future response shape -- a non-issue: a non-triggering, non-empty response
487
+ # always gets a simulated reply now, regardless of how it's formatted.
488
+ mock_client = MagicMock()
489
+ mock_client.create_conversation.return_value = "conv-1"
490
+ mock_client.send_message.side_effect = [
491
+ _no_kda_chat_result("Which one should I analyze?\n**Option 1**: revenue\n**Option 2**: gross profit"),
492
+ _kda_chat_result(success=True),
493
+ ]
494
+
495
+ with (
496
+ patch("gooddata_eval.core.agentic.kda_skill.ChatClient", return_value=mock_client),
497
+ patch(
498
+ "gooddata_eval.core.agentic.kda_skill.generate_simulated_kda_response",
499
+ return_value="Use revenue.",
500
+ ) as mock_simulate,
501
+ ):
502
+ summary = run_agentic_kda_skill(
503
+ host="http://host/api/v1/actions/workspaces/ws1/ai",
504
+ token="tok",
505
+ workspace_id="ws1",
506
+ question="Why did revenue drop?",
507
+ expected_output=_EXPECTED,
508
+ k=1,
509
+ max_iterations=2,
510
+ )
511
+
512
+ mock_simulate.assert_called_once()
513
+ assert summary.best.evaluation.disambiguated is True
514
+ assert summary.best.evaluation.triggered is True
515
+
516
+
517
+ def test_run_agentic_kda_skill_disambiguates_on_period_clarification():
518
+ # generate_simulated_kda_response used to only know about measure candidates -- if the
519
+ # agent asked about the PERIOD instead, it had nothing period-specific to answer with.
520
+ # Verify the period hint built from expected_output's Date Attribute/Analyzed
521
+ # Period/Reference Period reaches the simulated-reply call.
522
+ expected_output = {
523
+ "Measure": {"type": "metric", "id": "revenue"},
524
+ "Date Attribute": "transaction_date.quarter",
525
+ "Analyzed Period": "2026-2",
526
+ "Reference Period": "2026-1",
527
+ }
528
+ mock_client = MagicMock()
529
+ mock_client.create_conversation.return_value = "conv-1"
530
+ mock_client.send_message.side_effect = [
531
+ _no_kda_chat_result("Which period would you like to compare?"),
532
+ _kda_chat_result(success=True),
533
+ ]
534
+
535
+ with (
536
+ patch("gooddata_eval.core.agentic.kda_skill.ChatClient", return_value=mock_client),
537
+ patch(
538
+ "gooddata_eval.core.agentic.kda_skill.generate_simulated_kda_response",
539
+ return_value="Compare 2026-2 to 2026-1.",
540
+ ) as mock_simulate,
541
+ ):
542
+ summary = run_agentic_kda_skill(
543
+ host="http://host/api/v1/actions/workspaces/ws1/ai",
544
+ token="tok",
545
+ workspace_id="ws1",
546
+ question="Why did revenue drop?",
547
+ expected_output=expected_output,
548
+ k=1,
549
+ max_iterations=2,
550
+ )
551
+
552
+ mock_simulate.assert_called_once_with(
553
+ "Which period would you like to compare?",
554
+ {"type": "metric", "id": "revenue"},
555
+ "transaction_date.quarter, comparing 2026-2 to 2026-1",
556
+ )
557
+ assert summary.best.evaluation.disambiguated is True
558
+ assert summary.best.evaluation.triggered is True
559
+
560
+
430
561
  def test_run_agentic_kda_skill_disambiguates_when_expected_output_is_not_a_dict():
431
562
  # DatasetItem.expected_output on the gdc-nas side allows str/list, not just dict.
432
563
  # expected_output.get("Measure") would raise AttributeError on those shapes, silently
@@ -457,7 +588,7 @@ def test_run_agentic_kda_skill_disambiguates_when_expected_output_is_not_a_dict(
457
588
  max_iterations=2,
458
589
  )
459
590
 
460
- mock_generate.assert_called_once_with("Could you clarify which measure?", None)
591
+ mock_generate.assert_called_once_with("Could you clarify which measure?", None, None)
461
592
  assert summary.best.evaluation.disambiguated is True
462
593
  assert summary.best.evaluation.triggered is True
463
594