gooddata-eval 1.73.1.dev2__tar.gz → 1.73.1.dev3__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (90) hide show
  1. {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/PKG-INFO +2 -2
  2. {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/pyproject.toml +2 -2
  3. {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/agentic/conversation.py +6 -3
  4. {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/agentic/metric_skill.py +54 -8
  5. {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/tests/test_agentic_conversation.py +25 -0
  6. {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/tests/test_agentic_metric_skill.py +128 -8
  7. {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/.gitignore +0 -0
  8. {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/LICENSE.txt +0 -0
  9. {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/Makefile +0 -0
  10. {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/README.md +0 -0
  11. {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/__init__.py +0 -0
  12. {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/_version.py +0 -0
  13. {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/cli/__init__.py +0 -0
  14. {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/cli/agentic_runner.py +0 -0
  15. {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/cli/main.py +0 -0
  16. {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/__init__.py +0 -0
  17. {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/agentic/__init__.py +0 -0
  18. {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/agentic/_catalog.py +0 -0
  19. {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/agentic/_langfuse.py +0 -0
  20. {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/agentic/alert_skill.py +0 -0
  21. {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/agentic/general_question.py +0 -0
  22. {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/agentic/guardrail.py +0 -0
  23. {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/agentic/kda_skill.py +0 -0
  24. {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/agentic/search_tool.py +0 -0
  25. {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/agentic/visualization.py +0 -0
  26. {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/chat/__init__.py +0 -0
  27. {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/chat/sse_client.py +0 -0
  28. {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/config.py +0 -0
  29. {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/connection.py +0 -0
  30. {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/dataset/__init__.py +0 -0
  31. {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/dataset/langfuse_source.py +0 -0
  32. {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/dataset/local.py +0 -0
  33. {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/evaluators/__init__.py +0 -0
  34. {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/evaluators/_deep_subset.py +0 -0
  35. {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/evaluators/_llm_judge.py +0 -0
  36. {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/evaluators/_text_utils.py +0 -0
  37. {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/evaluators/alert_skill.py +0 -0
  38. {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/evaluators/base.py +0 -0
  39. {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/evaluators/general_question.py +0 -0
  40. {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/evaluators/guardrail.py +0 -0
  41. {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/evaluators/metric_skill.py +0 -0
  42. {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/evaluators/search_tool.py +0 -0
  43. {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/evaluators/summary.py +0 -0
  44. {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/evaluators/visualization.py +0 -0
  45. {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/langfuse/__init__.py +0 -0
  46. {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/langfuse/sink.py +0 -0
  47. {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/models.py +0 -0
  48. {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/reporting/__init__.py +0 -0
  49. {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/reporting/console.py +0 -0
  50. {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/reporting/json_report.py +0 -0
  51. {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/runner.py +0 -0
  52. {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/scoring.py +0 -0
  53. {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/summary/__init__.py +0 -0
  54. {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/summary/http_client.py +0 -0
  55. {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/workspace.py +0 -0
  56. {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/tests/__init__.py +0 -0
  57. {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/tests/conftest.py +0 -0
  58. {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/tests/fixtures/sample_dataset/metric_skill_create.json +0 -0
  59. {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/tests/fixtures/sample_dataset/visualization_revenue.json +0 -0
  60. {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/tests/fixtures/sse_visualization_stream.txt +0 -0
  61. {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/tests/test_agentic_alert_skill.py +0 -0
  62. {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/tests/test_agentic_general_question.py +0 -0
  63. {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/tests/test_agentic_guardrail.py +0 -0
  64. {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/tests/test_agentic_kda_skill.py +0 -0
  65. {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/tests/test_agentic_langfuse_trace.py +0 -0
  66. {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/tests/test_agentic_run_context.py +0 -0
  67. {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/tests/test_agentic_runner.py +0 -0
  68. {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/tests/test_agentic_search_tool.py +0 -0
  69. {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/tests/test_agentic_visualization.py +0 -0
  70. {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/tests/test_alert_skill_evaluator.py +0 -0
  71. {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/tests/test_cli.py +0 -0
  72. {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/tests/test_connection.py +0 -0
  73. {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/tests/test_deep_subset.py +0 -0
  74. {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/tests/test_langfuse_sink.py +0 -0
  75. {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/tests/test_langfuse_source.py +0 -0
  76. {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/tests/test_llm_judge.py +0 -0
  77. {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/tests/test_local_loader.py +0 -0
  78. {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/tests/test_metric_skill_evaluator.py +0 -0
  79. {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/tests/test_models.py +0 -0
  80. {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/tests/test_reporting.py +0 -0
  81. {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/tests/test_runner.py +0 -0
  82. {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/tests/test_scoring.py +0 -0
  83. {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/tests/test_search_tool_evaluator.py +0 -0
  84. {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/tests/test_sse_client.py +0 -0
  85. {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/tests/test_summary_client.py +0 -0
  86. {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/tests/test_summary_evaluator.py +0 -0
  87. {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/tests/test_text_evaluators.py +0 -0
  88. {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/tests/test_visualization_evaluator.py +0 -0
  89. {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/tests/test_workspace.py +0 -0
  90. {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/tox.ini +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: gooddata-eval
3
- Version: 1.73.1.dev2
3
+ Version: 1.73.1.dev3
4
4
  Summary: Evaluate the GoodData AI agent against your own questions and models.
5
5
  Project-URL: Source, https://github.com/gooddata/gooddata-python-sdk
6
6
  Author-email: GoodData <support@gooddata.com>
@@ -17,7 +17,7 @@ Classifier: Topic :: Scientific/Engineering
17
17
  Classifier: Topic :: Software Development
18
18
  Classifier: Typing :: Typed
19
19
  Requires-Python: >=3.10
20
- Requires-Dist: gooddata-sdk~=1.73.1.dev2
20
+ Requires-Dist: gooddata-sdk~=1.73.1.dev3
21
21
  Requires-Dist: httpx<1.0,>=0.27
22
22
  Requires-Dist: orjson<4.0.0,>=3.9.15
23
23
  Requires-Dist: pydantic<3.0,>=2.6
@@ -1,7 +1,7 @@
1
1
  # (C) 2026 GoodData Corporation
2
2
  [project]
3
3
  name = "gooddata-eval"
4
- version = "1.73.1.dev2"
4
+ version = "1.73.1.dev3"
5
5
  description = "Evaluate the GoodData AI agent against your own questions and models."
6
6
  readme = "README.md"
7
7
  license = "MIT"
@@ -11,7 +11,7 @@ authors = [
11
11
  keywords = ["gooddata", "ai", "evaluation", "llm", "analytics", "cli"]
12
12
  requires-python = ">=3.10"
13
13
  dependencies = [
14
- "gooddata-sdk~=1.73.1.dev2",
14
+ "gooddata-sdk~=1.73.1.dev3",
15
15
  "httpx>=0.27,<1.0",
16
16
  "orjson>=3.9.15,<4.0.0",
17
17
  "pydantic>=2.6,<3.0",
@@ -205,9 +205,12 @@ def _get_sim_user_response(agent_message: str, turn: TurnDefinition, expected_ou
205
205
  generate_simulated_response,
206
206
  )
207
207
 
208
- return generate_simulated_response(agent_message, expected_output)
209
- except Exception:
210
- pass
208
+ # A conversation turn only ever carries one expected_output (no multi-candidate
209
+ # list like agent_metric_skill's fixtures) -- wrap it as a single-item list to
210
+ # match generate_simulated_response's signature.
211
+ return generate_simulated_response(agent_message, [expected_output], turn.message)
212
+ except Exception as exc:
213
+ print(f"[SIM-USER] metric branch failed for turn {turn.turn_id}: {exc}")
211
214
 
212
215
  # Generic fallback for other skill types or when expected_output is absent
213
216
  import os # noqa: PLC0415
@@ -30,6 +30,11 @@ _INNER_SELECT_RE = re.compile(r"\(\s*SELECT\s*\{([^}]+)\}\s*\)", re.IGNORECASE)
30
30
  # Everything else in MAQL (keywords, operators, numbers, punctuation) carries no
31
31
  # case-sensitive meaning, per the MAQL reference (SELECT/BY/WHERE/FOR PREVIOUS/etc.
32
32
  # are case-insensitive; only {..} identifiers and quoted literal values are not).
33
+ # Feeds _normalize_maql, the scoring comparator (_best_maql_match) -- do not widen this
34
+ # to handle \X escapes without confirming MAQL literals actually support backslash
35
+ # escaping (unconfirmed; see PR #1760 review). A wrong guess here silently changes
36
+ # maql_correct for the whole eval dataset, not just a hint. _no_where_clause_hint()
37
+ # below has its own, separately-scoped regex for that reason.
33
38
  _PROTECTED_RE = re.compile(r"\{[^}]*\}|\"[^\"]*\"|'[^']*'")
34
39
 
35
40
 
@@ -99,9 +104,43 @@ class SimulatedResponseError(RuntimeError):
99
104
  """
100
105
 
101
106
 
102
- def generate_simulated_response(agent_message: str, expected_output: dict) -> str:
107
+ # Separate from _PROTECTED_RE on purpose: this one only feeds a same-turn LLM-prompt hint
108
+ # (see _no_where_clause_hint), never the scoring comparator, so it can afford to consume
109
+ # \X escape sequences inside quoted literals without risking maql_correct semantics.
110
+ _HINT_PROTECTED_RE = re.compile(r"\{[^}]*\}|\"(?:[^\"\\]|\\.)*\"|'(?:[^'\\]|\\.)*'")
111
+
112
+
113
+ def _no_where_clause_hint(expected_maqls: list[str]) -> str:
114
+ """Deterministic nudge for when NONE of the accepted candidate MAQLs has a WHERE clause.
115
+
116
+ Without this, whether to add a filter is left entirely to the simulating LLM's judgment
117
+ of what the original request "implies" -- the same fuzzy reasoning that caused it to
118
+ inject an unrequested filter in the first place (QA-29094). Checks every candidate, not
119
+ just the first: _best_maql_match accepts any of them, so hinting off just candidate 0
120
+ would risk steering the agent away from a filtered candidate the scorer would still have
121
+ accepted (the mirror-image of the original bug). Strips {type/id} identifiers and quoted
122
+ literals first so a "where" substring inside one of those -- e.g.
123
+ `{metric/somewhere_sales}`, or a literal value containing the word -- doesn't get
124
+ mistaken for a real WHERE clause.
125
+ """
126
+ for maql in expected_maqls:
127
+ outside_protected = _HINT_PROTECTED_RE.sub(" ", maql)
128
+ if re.search(r"\bWHERE\b", outside_protected, re.IGNORECASE):
129
+ return ""
130
+ return (
131
+ " This metric needs no filter. If the assistant asks about excluding or filtering "
132
+ "anything, say no filter is needed."
133
+ )
134
+
135
+
136
+ def generate_simulated_response(agent_message: str, expected_outputs: list[dict], original_question: str) -> str:
103
137
  """Generate a user reply to keep the metric-skill conversation going (gpt-4o-mini).
104
138
 
139
+ ``expected_outputs`` is the fixture's full candidate list (as accepted by
140
+ ``_best_maql_match``), not just the first one -- the ground-truth MAQL woven into the
141
+ prompt still comes from candidate 0, but the no-filter hint checks all of them (see
142
+ ``_no_where_clause_hint``).
143
+
105
144
  Raises:
106
145
  SimulatedResponseError: openai is not installed, OPENAI_API_KEY is unset, or the
107
146
  provider call failed.
@@ -116,15 +155,23 @@ def generate_simulated_response(agent_message: str, expected_output: dict) -> st
116
155
  raise SimulatedResponseError("OPENAI_API_KEY environment variable is not set")
117
156
 
118
157
  client = OpenAI(api_key=api_key)
119
- expected_maql = expected_output.get("maql", "")
158
+ expected_maql = expected_outputs[0].get("maql", "") if expected_outputs else ""
159
+ expected_maqls = [eo.get("maql", "") for eo in expected_outputs]
120
160
  prompt = (
121
161
  f"You are simulating a user in a conversation with a BI assistant that creates metrics. "
162
+ f"The user's original request was: '{original_question}'. "
122
163
  f"The assistant said: '{agent_message}'. "
123
164
  f"The user's ground-truth intended metric is exactly this MAQL: {expected_maql}. "
124
- f"Reply as the user. You MUST ensure every clause of that MAQL (including any WHERE/filter "
125
- f"conditions) is eventually satisfied, and quote field/label identifiers verbatim from it -- "
126
- f"never paraphrase or drop a clause, even if the assistant's question doesn't explicitly ask "
127
- f"about it. If the assistant's offered options omit a required filter, add it yourself."
165
+ f"Reply as the user. If the assistant is asking a clarifying question rather than proposing "
166
+ f"a metric, answer that question directly using the ground-truth MAQL -- quote field/label "
167
+ f"identifiers verbatim -- instead of merely agreeing. "
168
+ f"If the assistant's proposal already satisfies the ORIGINAL REQUEST above, agree and confirm "
169
+ f"-- do not introduce new requirements the original request never mentioned. "
170
+ f"Only if the assistant's proposal is missing something the original request actually implies "
171
+ f"(e.g. a filter/clause from the ground-truth MAQL that is a reasonable reading of the original "
172
+ f"request), point it out and add it yourself, quoting field/label identifiers verbatim from the "
173
+ f"ground-truth MAQL."
174
+ f"{_no_where_clause_hint(expected_maqls)}"
128
175
  )
129
176
  try:
130
177
  response = client.chat.completions.create(
@@ -234,7 +281,6 @@ def _execute_single_metric_run(
234
281
  ``_delete_metric``) so it cannot leak into — and be reused by — a later test
235
282
  sharing the workspace.
236
283
  """
237
- primary_expected = expected_outputs[0] if expected_outputs else {}
238
284
  metric_result: dict | None = None
239
285
  created_metric_ids: list[str] = []
240
286
  turns = 0
@@ -281,7 +327,7 @@ def _execute_single_metric_run(
281
327
  if _iteration >= max_iterations - 1:
282
328
  break
283
329
  try:
284
- current_question = generate_simulated_response(response_text, primary_expected)
330
+ current_question = generate_simulated_response(response_text, expected_outputs, question)
285
331
  except SimulatedResponseError as exc:
286
332
  print(f"[SIM-USER] Simulated reply failed for conversation {conversation_id}: {exc}")
287
333
  break
@@ -8,6 +8,7 @@ from gooddata_eval.core.agentic.conversation import (
8
8
  ConversationFixture,
9
9
  TurnDefinition,
10
10
  TurnResult,
11
+ _get_sim_user_response,
11
12
  _resolve_refs,
12
13
  evaluate_agentic_conversation,
13
14
  run_agentic_conversation,
@@ -107,6 +108,30 @@ def test_resolve_refs_substitutes():
107
108
  assert result == {"maql": "SELECT {metric/foo}"}
108
109
 
109
110
 
111
+ def test_get_sim_user_response_metric_branch_forwards_the_turn_message():
112
+ """QA-29094 follow-up: every test in this file patches out `_get_sim_user_response`
113
+ itself, so its metric branch (which forwards to
114
+ ``metric_skill.generate_simulated_response``) had 0% coverage -- a future signature
115
+ change there would raise inside the bare ``except Exception`` and silently fall through
116
+ to the generic fallback prompt instead of failing loudly."""
117
+ turn = TurnDefinition(
118
+ turn_id="t1",
119
+ message="I need a metric for total ordered units",
120
+ expected_skill="metric",
121
+ expected_output_type="metric",
122
+ )
123
+ expected_output = {"maql": "SELECT SUM({fact/order_unit_quantity})"}
124
+
125
+ with patch(
126
+ "gooddata_eval.core.agentic.metric_skill.generate_simulated_response",
127
+ return_value="Yes, that works.",
128
+ ) as mock_sim:
129
+ reply = _get_sim_user_response("Should I create this metric?", turn, expected_output)
130
+
131
+ assert reply == "Yes, that works."
132
+ mock_sim.assert_called_once_with("Should I create this metric?", [expected_output], turn.message)
133
+
134
+
110
135
  def test_run_agentic_conversation_single_turn():
111
136
  mock_client = MagicMock()
112
137
  mock_client.create_conversation.return_value = "conv-1"
@@ -13,6 +13,7 @@ from gooddata_eval.core.agentic.metric_skill import (
13
13
  SimulatedResponseError,
14
14
  _delete_metric,
15
15
  _extract_metric_result,
16
+ _no_where_clause_hint,
16
17
  _normalize_maql,
17
18
  evaluate_agentic_metric_skill,
18
19
  generate_simulated_response,
@@ -91,6 +92,54 @@ def test_normalize_maql_removes_select_wrapper():
91
92
  assert _normalize_maql("(SELECT {metric/abc})") == "{metric/abc}"
92
93
 
93
94
 
95
+ def test_no_where_clause_hint_is_empty_when_a_candidate_has_a_where_clause():
96
+ assert _no_where_clause_hint(['SELECT {metric/foo} WHERE {label/status} = "active"']) == ""
97
+
98
+
99
+ def test_no_where_clause_hint_is_present_when_no_candidate_has_a_where_clause():
100
+ """QA-29094 follow-up: whether to add a filter must not be left to the simulating LLM's
101
+ judgment of what the original request "implies" -- that fuzzy reasoning is exactly what
102
+ caused it to inject an unrequested filter in the first place."""
103
+ hint = _no_where_clause_hint(["SELECT SUM({fact/order_unit_quantity})"])
104
+ assert hint != ""
105
+ assert "no filter is needed" in hint
106
+
107
+
108
+ def test_no_where_clause_hint_stays_silent_if_any_candidate_has_a_where_clause():
109
+ """PR #1760 review (Henry): _no_where_clause_hint used to see only expected_outputs[0].
110
+ A fixture like agent_metric_skill_4.json lists an unfiltered candidate first and a
111
+ filtered one second -- both accepted by _best_maql_match. Hinting "no filter needed"
112
+ off candidate 0 alone would steer the agent away from the filtered candidate even
113
+ though the scorer would still take it -- the mirror image of the original QA-29094 bug.
114
+ """
115
+ candidates = [
116
+ "SELECT SUM({fact/order_unit_quantity})",
117
+ 'SELECT SUM({fact/order_unit_quantity}) WHERE {label/order_status} = "Processed"',
118
+ ]
119
+ assert _no_where_clause_hint(candidates) == ""
120
+
121
+
122
+ def test_no_where_clause_hint_ignores_where_inside_an_identifier():
123
+ """CodeRabbit finding on PR #1760: a naive substring check treats the "where" inside
124
+ an identifier like {metric/somewhere_sales} as a real WHERE clause and wrongly stays
125
+ silent -- it must be stripped as a protected span before matching."""
126
+ assert _no_where_clause_hint(["SELECT {metric/somewhere_sales}"]) != ""
127
+
128
+
129
+ def test_no_where_clause_hint_ignores_where_inside_a_quoted_literal():
130
+ assert _no_where_clause_hint(['SELECT {metric/x} = "somewhere nearby"']) != ""
131
+
132
+
133
+ def test_no_where_clause_hint_ignores_where_inside_a_literal_with_an_escaped_quote():
134
+ """CodeRabbit finding on PR #1760: an escaped quote inside a quoted literal ended the
135
+ protected-span match early, leaking the rest of the literal's text -- including a
136
+ standalone WHERE -- as unprotected. Uses _HINT_PROTECTED_RE (escape-aware), kept
137
+ separate from the shared _PROTECTED_RE that feeds the maql_correct comparator (PR
138
+ #1760 review, Henry) -- see test_normalize_maql_does_not_consume_escape_sequences."""
139
+ maql = 'SELECT {metric/x} = "Jane\\"s store WHERE something"'
140
+ assert _no_where_clause_hint([maql]) != ""
141
+
142
+
94
143
  def test_generate_simulated_response_prompt_preserves_maql_fidelity(monkeypatch):
95
144
  """Regression test for a live-reproduced bug: the old prompt ("reply briefly",
96
145
  no instruction to cover clauses the assistant didn't ask about) let the
@@ -111,19 +160,75 @@ def test_generate_simulated_response_prompt_preserves_maql_fidelity(monkeypatch)
111
160
  monkeypatch.setitem(sys.modules, "openai", fake_openai_module)
112
161
 
113
162
  expected_output = {"maql": 'SELECT {metric/spend_amount_-_cutcgco} WHERE {label/ecommerce_indicator_code} = "1"'}
114
- generate_simulated_response("Which base metric should I use?", expected_output)
163
+ generate_simulated_response(
164
+ "Which base metric should I use?", [expected_output], "I need a metric for spend amount"
165
+ )
115
166
 
116
167
  call_kwargs = mock_client.chat.completions.create.call_args.kwargs
117
168
  sent_prompt = call_kwargs["messages"][0]["content"]
118
169
 
119
170
  assert expected_output["maql"] in sent_prompt
120
171
  assert "verbatim" in sent_prompt
121
- assert "every clause" in sent_prompt
122
- assert "WHERE" in sent_prompt or "filter" in sent_prompt.lower()
123
- assert "reply briefly" not in sent_prompt.lower()
172
+ assert "filter" in sent_prompt.lower()
173
+ # Guards against a truncated reply mid-MAQL -- the LLM was cutting fidelity short under
174
+ # the old, lower budget before this was raised (see the docstring above).
124
175
  assert call_kwargs["max_tokens"] >= 300
125
176
 
126
177
 
178
+ def test_generate_simulated_response_prompt_agrees_when_the_original_request_is_already_satisfied(monkeypatch):
179
+ """Regression test for QA-29094: the old prompt told the simulated user to force every
180
+ clause of the ground-truth MAQL regardless of what the original request actually asked
181
+ for, so it would inject filters/constraints the user never mentioned even when the
182
+ assistant's proposal already matched the request. The prompt must now carry the
183
+ original request and instruct the simulated user to agree when it's already satisfied.
184
+ """
185
+ monkeypatch.setenv("OPENAI_API_KEY", "test-key")
186
+ mock_client = MagicMock()
187
+ mock_response = MagicMock()
188
+ mock_response.choices = [MagicMock(message=MagicMock(content="ok"))]
189
+ mock_client.chat.completions.create.return_value = mock_response
190
+ fake_openai_module = types.SimpleNamespace(OpenAI=MagicMock(return_value=mock_client), OpenAIError=Exception)
191
+ monkeypatch.setitem(sys.modules, "openai", fake_openai_module)
192
+
193
+ original_question = "I need a metric for total ordered units called Total Order Quantity"
194
+ expected_output = {"maql": "SELECT SUM({fact/order_unit_quantity}) WHERE {fact/order_status} != 'cancelled'"}
195
+ generate_simulated_response("Should I create this metric?", [expected_output], original_question)
196
+
197
+ sent_prompt = mock_client.chat.completions.create.call_args.kwargs["messages"][0]["content"]
198
+
199
+ # Structural checks on the interpolated data -- robust to prompt-wording edits.
200
+ assert original_question in sent_prompt
201
+ assert expected_output["maql"] in sent_prompt
202
+ assert "reply briefly" not in sent_prompt.lower()
203
+ # A ground-truth MAQL with a WHERE clause must not trigger the no-filter-needed hint.
204
+ assert "no filter is needed" not in sent_prompt
205
+
206
+
207
+ def test_generate_simulated_response_prompt_handles_a_clarifying_question(monkeypatch):
208
+ """QA-29094 follow-up: the two-branch prompt ("already satisfies" / "missing something")
209
+ both assume the assistant made a proposal -- but the dominant real case is the assistant
210
+ asking a clarifying question first (no proposal exists yet to judge as satisfying or not).
211
+ Without an explicit instruction, the simulating LLM could classify "nothing proposed yet"
212
+ as trivially "satisfied" and reply "yes, that works", leaving the agent no closer to a
213
+ usable metric and burning iterations."""
214
+ monkeypatch.setenv("OPENAI_API_KEY", "test-key")
215
+ mock_client = MagicMock()
216
+ mock_response = MagicMock()
217
+ mock_response.choices = [MagicMock(message=MagicMock(content="ok"))]
218
+ mock_client.chat.completions.create.return_value = mock_response
219
+ fake_openai_module = types.SimpleNamespace(OpenAI=MagicMock(return_value=mock_client), OpenAIError=Exception)
220
+ monkeypatch.setitem(sys.modules, "openai", fake_openai_module)
221
+
222
+ expected_output = {"maql": "SELECT SUM({fact/order_unit_quantity})"}
223
+ generate_simulated_response("Which base metric should I use?", [expected_output], "I need total ordered units")
224
+
225
+ sent_prompt = mock_client.chat.completions.create.call_args.kwargs["messages"][0]["content"]
226
+
227
+ assert "clarifying question" in sent_prompt
228
+ # No WHERE clause in the ground truth -- the no-filter hint must fire here too.
229
+ assert "no filter is needed" in sent_prompt
230
+
231
+
127
232
  def test_normalize_maql_is_case_insensitive_for_keywords():
128
233
  """Regression test for a live-reproduced bug: 'FOR PREVIOUS(...)' vs
129
234
  'FOR Previous(...)' scored as a mismatch even though MAQL keywords are
@@ -147,6 +252,19 @@ def test_normalize_maql_preserves_quoted_literal_case():
147
252
  assert _normalize_maql('WHERE {label/status} = "Active"') != _normalize_maql('WHERE {label/status} = "active"')
148
253
 
149
254
 
255
+ def test_normalize_maql_does_not_consume_escape_sequences():
256
+ """PR #1760 review (Henry): _PROTECTED_RE feeds this comparator (via
257
+ _casefold_outside_protected), so it must NOT treat \\X as an escape sequence unless
258
+ MAQL literals are confirmed to support backslash escaping (unconfirmed). A `\\"`
259
+ inside a literal must still end that literal at the next real quote -- not swallow
260
+ everything up to the following quoted value, which would leave a real keyword like
261
+ AND uncasefolded and a later literal's case wrongly casefolded."""
262
+ maql = 'SELECT {metric/x} WHERE {label/path} = "C:\\" AND {label/y} = "Active"'
263
+ normalized = _normalize_maql(maql)
264
+ assert "and {label/y}" in normalized # AND is a keyword outside the literal -- casefolded
265
+ assert '"Active"' in normalized # the second literal's case is untouched -- not "active"
266
+
267
+
150
268
  def test_metric_run_result_fields():
151
269
  r = MetricRunResult(
152
270
  conversation_id="c1",
@@ -228,7 +346,7 @@ def test_run_agentic_metric_skill_closes_client_on_no_result():
228
346
  mock_client.close.assert_called_once()
229
347
  assert summary.pass_at_k is False
230
348
  assert summary.best.metric_created is False
231
- mock_sim.assert_called_once_with("I will work on that.", {"maql": "SELECT {metric/foo}"})
349
+ mock_sim.assert_called_once_with("I will work on that.", [{"maql": "SELECT {metric/foo}"}], "Create metric foo")
232
350
 
233
351
 
234
352
  def test_run_agentic_metric_skill_uses_initial_conversation_for_run_0():
@@ -408,7 +526,7 @@ def test_generate_simulated_response_without_an_api_key():
408
526
  patch.dict(os.environ, {}, clear=True),
409
527
  pytest.raises(SimulatedResponseError, match="OPENAI_API_KEY"),
410
528
  ):
411
- generate_simulated_response("Which brand field?", {"maql": "SELECT {metric/foo}"})
529
+ generate_simulated_response("Which brand field?", [{"maql": "SELECT {metric/foo}"}], "I need a metric for foo")
412
530
 
413
531
 
414
532
  def test_generate_simulated_response_without_the_openai_package():
@@ -416,7 +534,7 @@ def test_generate_simulated_response_without_the_openai_package():
416
534
  patch.dict(sys.modules, {"openai": None}),
417
535
  pytest.raises(SimulatedResponseError, match="openai package is required"),
418
536
  ):
419
- generate_simulated_response("Which brand field?", {"maql": "SELECT {metric/foo}"})
537
+ generate_simulated_response("Which brand field?", [{"maql": "SELECT {metric/foo}"}], "I need a metric for foo")
420
538
 
421
539
 
422
540
  def test_run_agentic_metric_skill_fails_the_run_when_the_simulated_reply_cannot_be_generated():
@@ -448,7 +566,9 @@ def test_run_agentic_metric_skill_fails_the_run_when_the_simulated_reply_cannot_
448
566
  assert summary.best.metric_created is False
449
567
  assert summary.best.total_turns == 1.0
450
568
  mock_client.close.assert_called_once()
451
- mock_sim.assert_called_once_with("Which brand field should I count?", {"maql": "SELECT {metric/foo}"})
569
+ mock_sim.assert_called_once_with(
570
+ "Which brand field should I count?", [{"maql": "SELECT {metric/foo}"}], "Create metric foo"
571
+ )
452
572
 
453
573
 
454
574
  def test_run_agentic_metric_skill_accumulates_reasoning_steps_across_iterations():