gooddata-eval 1.73.1.dev2__tar.gz → 1.73.1.dev3__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/PKG-INFO +2 -2
- {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/pyproject.toml +2 -2
- {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/agentic/conversation.py +6 -3
- {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/agentic/metric_skill.py +54 -8
- {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/tests/test_agentic_conversation.py +25 -0
- {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/tests/test_agentic_metric_skill.py +128 -8
- {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/.gitignore +0 -0
- {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/LICENSE.txt +0 -0
- {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/Makefile +0 -0
- {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/README.md +0 -0
- {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/__init__.py +0 -0
- {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/_version.py +0 -0
- {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/cli/__init__.py +0 -0
- {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/cli/agentic_runner.py +0 -0
- {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/cli/main.py +0 -0
- {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/__init__.py +0 -0
- {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/agentic/__init__.py +0 -0
- {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/agentic/_catalog.py +0 -0
- {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/agentic/_langfuse.py +0 -0
- {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/agentic/alert_skill.py +0 -0
- {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/agentic/general_question.py +0 -0
- {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/agentic/guardrail.py +0 -0
- {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/agentic/kda_skill.py +0 -0
- {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/agentic/search_tool.py +0 -0
- {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/agentic/visualization.py +0 -0
- {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/chat/__init__.py +0 -0
- {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/chat/sse_client.py +0 -0
- {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/config.py +0 -0
- {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/connection.py +0 -0
- {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/dataset/__init__.py +0 -0
- {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/dataset/langfuse_source.py +0 -0
- {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/dataset/local.py +0 -0
- {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/evaluators/__init__.py +0 -0
- {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/evaluators/_deep_subset.py +0 -0
- {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/evaluators/_llm_judge.py +0 -0
- {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/evaluators/_text_utils.py +0 -0
- {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/evaluators/alert_skill.py +0 -0
- {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/evaluators/base.py +0 -0
- {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/evaluators/general_question.py +0 -0
- {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/evaluators/guardrail.py +0 -0
- {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/evaluators/metric_skill.py +0 -0
- {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/evaluators/search_tool.py +0 -0
- {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/evaluators/summary.py +0 -0
- {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/evaluators/visualization.py +0 -0
- {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/langfuse/__init__.py +0 -0
- {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/langfuse/sink.py +0 -0
- {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/models.py +0 -0
- {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/reporting/__init__.py +0 -0
- {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/reporting/console.py +0 -0
- {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/reporting/json_report.py +0 -0
- {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/runner.py +0 -0
- {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/scoring.py +0 -0
- {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/summary/__init__.py +0 -0
- {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/summary/http_client.py +0 -0
- {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/workspace.py +0 -0
- {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/tests/__init__.py +0 -0
- {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/tests/conftest.py +0 -0
- {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/tests/fixtures/sample_dataset/metric_skill_create.json +0 -0
- {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/tests/fixtures/sample_dataset/visualization_revenue.json +0 -0
- {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/tests/fixtures/sse_visualization_stream.txt +0 -0
- {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/tests/test_agentic_alert_skill.py +0 -0
- {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/tests/test_agentic_general_question.py +0 -0
- {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/tests/test_agentic_guardrail.py +0 -0
- {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/tests/test_agentic_kda_skill.py +0 -0
- {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/tests/test_agentic_langfuse_trace.py +0 -0
- {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/tests/test_agentic_run_context.py +0 -0
- {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/tests/test_agentic_runner.py +0 -0
- {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/tests/test_agentic_search_tool.py +0 -0
- {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/tests/test_agentic_visualization.py +0 -0
- {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/tests/test_alert_skill_evaluator.py +0 -0
- {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/tests/test_cli.py +0 -0
- {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/tests/test_connection.py +0 -0
- {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/tests/test_deep_subset.py +0 -0
- {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/tests/test_langfuse_sink.py +0 -0
- {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/tests/test_langfuse_source.py +0 -0
- {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/tests/test_llm_judge.py +0 -0
- {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/tests/test_local_loader.py +0 -0
- {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/tests/test_metric_skill_evaluator.py +0 -0
- {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/tests/test_models.py +0 -0
- {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/tests/test_reporting.py +0 -0
- {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/tests/test_runner.py +0 -0
- {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/tests/test_scoring.py +0 -0
- {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/tests/test_search_tool_evaluator.py +0 -0
- {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/tests/test_sse_client.py +0 -0
- {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/tests/test_summary_client.py +0 -0
- {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/tests/test_summary_evaluator.py +0 -0
- {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/tests/test_text_evaluators.py +0 -0
- {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/tests/test_visualization_evaluator.py +0 -0
- {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/tests/test_workspace.py +0 -0
- {gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/tox.ini +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: gooddata-eval
|
|
3
|
-
Version: 1.73.1.
|
|
3
|
+
Version: 1.73.1.dev3
|
|
4
4
|
Summary: Evaluate the GoodData AI agent against your own questions and models.
|
|
5
5
|
Project-URL: Source, https://github.com/gooddata/gooddata-python-sdk
|
|
6
6
|
Author-email: GoodData <support@gooddata.com>
|
|
@@ -17,7 +17,7 @@ Classifier: Topic :: Scientific/Engineering
|
|
|
17
17
|
Classifier: Topic :: Software Development
|
|
18
18
|
Classifier: Typing :: Typed
|
|
19
19
|
Requires-Python: >=3.10
|
|
20
|
-
Requires-Dist: gooddata-sdk~=1.73.1.
|
|
20
|
+
Requires-Dist: gooddata-sdk~=1.73.1.dev3
|
|
21
21
|
Requires-Dist: httpx<1.0,>=0.27
|
|
22
22
|
Requires-Dist: orjson<4.0.0,>=3.9.15
|
|
23
23
|
Requires-Dist: pydantic<3.0,>=2.6
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
# (C) 2026 GoodData Corporation
|
|
2
2
|
[project]
|
|
3
3
|
name = "gooddata-eval"
|
|
4
|
-
version = "1.73.1.
|
|
4
|
+
version = "1.73.1.dev3"
|
|
5
5
|
description = "Evaluate the GoodData AI agent against your own questions and models."
|
|
6
6
|
readme = "README.md"
|
|
7
7
|
license = "MIT"
|
|
@@ -11,7 +11,7 @@ authors = [
|
|
|
11
11
|
keywords = ["gooddata", "ai", "evaluation", "llm", "analytics", "cli"]
|
|
12
12
|
requires-python = ">=3.10"
|
|
13
13
|
dependencies = [
|
|
14
|
-
"gooddata-sdk~=1.73.1.
|
|
14
|
+
"gooddata-sdk~=1.73.1.dev3",
|
|
15
15
|
"httpx>=0.27,<1.0",
|
|
16
16
|
"orjson>=3.9.15,<4.0.0",
|
|
17
17
|
"pydantic>=2.6,<3.0",
|
|
@@ -205,9 +205,12 @@ def _get_sim_user_response(agent_message: str, turn: TurnDefinition, expected_ou
|
|
|
205
205
|
generate_simulated_response,
|
|
206
206
|
)
|
|
207
207
|
|
|
208
|
-
|
|
209
|
-
|
|
210
|
-
|
|
208
|
+
# A conversation turn only ever carries one expected_output (no multi-candidate
|
|
209
|
+
# list like agent_metric_skill's fixtures) -- wrap it as a single-item list to
|
|
210
|
+
# match generate_simulated_response's signature.
|
|
211
|
+
return generate_simulated_response(agent_message, [expected_output], turn.message)
|
|
212
|
+
except Exception as exc:
|
|
213
|
+
print(f"[SIM-USER] metric branch failed for turn {turn.turn_id}: {exc}")
|
|
211
214
|
|
|
212
215
|
# Generic fallback for other skill types or when expected_output is absent
|
|
213
216
|
import os # noqa: PLC0415
|
|
@@ -30,6 +30,11 @@ _INNER_SELECT_RE = re.compile(r"\(\s*SELECT\s*\{([^}]+)\}\s*\)", re.IGNORECASE)
|
|
|
30
30
|
# Everything else in MAQL (keywords, operators, numbers, punctuation) carries no
|
|
31
31
|
# case-sensitive meaning, per the MAQL reference (SELECT/BY/WHERE/FOR PREVIOUS/etc.
|
|
32
32
|
# are case-insensitive; only {..} identifiers and quoted literal values are not).
|
|
33
|
+
# Feeds _normalize_maql, the scoring comparator (_best_maql_match) -- do not widen this
|
|
34
|
+
# to handle \X escapes without confirming MAQL literals actually support backslash
|
|
35
|
+
# escaping (unconfirmed; see PR #1760 review). A wrong guess here silently changes
|
|
36
|
+
# maql_correct for the whole eval dataset, not just a hint. _no_where_clause_hint()
|
|
37
|
+
# below has its own, separately-scoped regex for that reason.
|
|
33
38
|
_PROTECTED_RE = re.compile(r"\{[^}]*\}|\"[^\"]*\"|'[^']*'")
|
|
34
39
|
|
|
35
40
|
|
|
@@ -99,9 +104,43 @@ class SimulatedResponseError(RuntimeError):
|
|
|
99
104
|
"""
|
|
100
105
|
|
|
101
106
|
|
|
102
|
-
|
|
107
|
+
# Separate from _PROTECTED_RE on purpose: this one only feeds a same-turn LLM-prompt hint
|
|
108
|
+
# (see _no_where_clause_hint), never the scoring comparator, so it can afford to consume
|
|
109
|
+
# \X escape sequences inside quoted literals without risking maql_correct semantics.
|
|
110
|
+
_HINT_PROTECTED_RE = re.compile(r"\{[^}]*\}|\"(?:[^\"\\]|\\.)*\"|'(?:[^'\\]|\\.)*'")
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
def _no_where_clause_hint(expected_maqls: list[str]) -> str:
|
|
114
|
+
"""Deterministic nudge for when NONE of the accepted candidate MAQLs has a WHERE clause.
|
|
115
|
+
|
|
116
|
+
Without this, whether to add a filter is left entirely to the simulating LLM's judgment
|
|
117
|
+
of what the original request "implies" -- the same fuzzy reasoning that caused it to
|
|
118
|
+
inject an unrequested filter in the first place (QA-29094). Checks every candidate, not
|
|
119
|
+
just the first: _best_maql_match accepts any of them, so hinting off just candidate 0
|
|
120
|
+
would risk steering the agent away from a filtered candidate the scorer would still have
|
|
121
|
+
accepted (the mirror-image of the original bug). Strips {type/id} identifiers and quoted
|
|
122
|
+
literals first so a "where" substring inside one of those -- e.g.
|
|
123
|
+
`{metric/somewhere_sales}`, or a literal value containing the word -- doesn't get
|
|
124
|
+
mistaken for a real WHERE clause.
|
|
125
|
+
"""
|
|
126
|
+
for maql in expected_maqls:
|
|
127
|
+
outside_protected = _HINT_PROTECTED_RE.sub(" ", maql)
|
|
128
|
+
if re.search(r"\bWHERE\b", outside_protected, re.IGNORECASE):
|
|
129
|
+
return ""
|
|
130
|
+
return (
|
|
131
|
+
" This metric needs no filter. If the assistant asks about excluding or filtering "
|
|
132
|
+
"anything, say no filter is needed."
|
|
133
|
+
)
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
def generate_simulated_response(agent_message: str, expected_outputs: list[dict], original_question: str) -> str:
|
|
103
137
|
"""Generate a user reply to keep the metric-skill conversation going (gpt-4o-mini).
|
|
104
138
|
|
|
139
|
+
``expected_outputs`` is the fixture's full candidate list (as accepted by
|
|
140
|
+
``_best_maql_match``), not just the first one -- the ground-truth MAQL woven into the
|
|
141
|
+
prompt still comes from candidate 0, but the no-filter hint checks all of them (see
|
|
142
|
+
``_no_where_clause_hint``).
|
|
143
|
+
|
|
105
144
|
Raises:
|
|
106
145
|
SimulatedResponseError: openai is not installed, OPENAI_API_KEY is unset, or the
|
|
107
146
|
provider call failed.
|
|
@@ -116,15 +155,23 @@ def generate_simulated_response(agent_message: str, expected_output: dict) -> st
|
|
|
116
155
|
raise SimulatedResponseError("OPENAI_API_KEY environment variable is not set")
|
|
117
156
|
|
|
118
157
|
client = OpenAI(api_key=api_key)
|
|
119
|
-
expected_maql =
|
|
158
|
+
expected_maql = expected_outputs[0].get("maql", "") if expected_outputs else ""
|
|
159
|
+
expected_maqls = [eo.get("maql", "") for eo in expected_outputs]
|
|
120
160
|
prompt = (
|
|
121
161
|
f"You are simulating a user in a conversation with a BI assistant that creates metrics. "
|
|
162
|
+
f"The user's original request was: '{original_question}'. "
|
|
122
163
|
f"The assistant said: '{agent_message}'. "
|
|
123
164
|
f"The user's ground-truth intended metric is exactly this MAQL: {expected_maql}. "
|
|
124
|
-
f"Reply as the user.
|
|
125
|
-
f"
|
|
126
|
-
f"
|
|
127
|
-
f"
|
|
165
|
+
f"Reply as the user. If the assistant is asking a clarifying question rather than proposing "
|
|
166
|
+
f"a metric, answer that question directly using the ground-truth MAQL -- quote field/label "
|
|
167
|
+
f"identifiers verbatim -- instead of merely agreeing. "
|
|
168
|
+
f"If the assistant's proposal already satisfies the ORIGINAL REQUEST above, agree and confirm "
|
|
169
|
+
f"-- do not introduce new requirements the original request never mentioned. "
|
|
170
|
+
f"Only if the assistant's proposal is missing something the original request actually implies "
|
|
171
|
+
f"(e.g. a filter/clause from the ground-truth MAQL that is a reasonable reading of the original "
|
|
172
|
+
f"request), point it out and add it yourself, quoting field/label identifiers verbatim from the "
|
|
173
|
+
f"ground-truth MAQL."
|
|
174
|
+
f"{_no_where_clause_hint(expected_maqls)}"
|
|
128
175
|
)
|
|
129
176
|
try:
|
|
130
177
|
response = client.chat.completions.create(
|
|
@@ -234,7 +281,6 @@ def _execute_single_metric_run(
|
|
|
234
281
|
``_delete_metric``) so it cannot leak into — and be reused by — a later test
|
|
235
282
|
sharing the workspace.
|
|
236
283
|
"""
|
|
237
|
-
primary_expected = expected_outputs[0] if expected_outputs else {}
|
|
238
284
|
metric_result: dict | None = None
|
|
239
285
|
created_metric_ids: list[str] = []
|
|
240
286
|
turns = 0
|
|
@@ -281,7 +327,7 @@ def _execute_single_metric_run(
|
|
|
281
327
|
if _iteration >= max_iterations - 1:
|
|
282
328
|
break
|
|
283
329
|
try:
|
|
284
|
-
current_question = generate_simulated_response(response_text,
|
|
330
|
+
current_question = generate_simulated_response(response_text, expected_outputs, question)
|
|
285
331
|
except SimulatedResponseError as exc:
|
|
286
332
|
print(f"[SIM-USER] Simulated reply failed for conversation {conversation_id}: {exc}")
|
|
287
333
|
break
|
|
@@ -8,6 +8,7 @@ from gooddata_eval.core.agentic.conversation import (
|
|
|
8
8
|
ConversationFixture,
|
|
9
9
|
TurnDefinition,
|
|
10
10
|
TurnResult,
|
|
11
|
+
_get_sim_user_response,
|
|
11
12
|
_resolve_refs,
|
|
12
13
|
evaluate_agentic_conversation,
|
|
13
14
|
run_agentic_conversation,
|
|
@@ -107,6 +108,30 @@ def test_resolve_refs_substitutes():
|
|
|
107
108
|
assert result == {"maql": "SELECT {metric/foo}"}
|
|
108
109
|
|
|
109
110
|
|
|
111
|
+
def test_get_sim_user_response_metric_branch_forwards_the_turn_message():
|
|
112
|
+
"""QA-29094 follow-up: every test in this file patches out `_get_sim_user_response`
|
|
113
|
+
itself, so its metric branch (which forwards to
|
|
114
|
+
``metric_skill.generate_simulated_response``) had 0% coverage -- a future signature
|
|
115
|
+
change there would raise inside the bare ``except Exception`` and silently fall through
|
|
116
|
+
to the generic fallback prompt instead of failing loudly."""
|
|
117
|
+
turn = TurnDefinition(
|
|
118
|
+
turn_id="t1",
|
|
119
|
+
message="I need a metric for total ordered units",
|
|
120
|
+
expected_skill="metric",
|
|
121
|
+
expected_output_type="metric",
|
|
122
|
+
)
|
|
123
|
+
expected_output = {"maql": "SELECT SUM({fact/order_unit_quantity})"}
|
|
124
|
+
|
|
125
|
+
with patch(
|
|
126
|
+
"gooddata_eval.core.agentic.metric_skill.generate_simulated_response",
|
|
127
|
+
return_value="Yes, that works.",
|
|
128
|
+
) as mock_sim:
|
|
129
|
+
reply = _get_sim_user_response("Should I create this metric?", turn, expected_output)
|
|
130
|
+
|
|
131
|
+
assert reply == "Yes, that works."
|
|
132
|
+
mock_sim.assert_called_once_with("Should I create this metric?", [expected_output], turn.message)
|
|
133
|
+
|
|
134
|
+
|
|
110
135
|
def test_run_agentic_conversation_single_turn():
|
|
111
136
|
mock_client = MagicMock()
|
|
112
137
|
mock_client.create_conversation.return_value = "conv-1"
|
|
@@ -13,6 +13,7 @@ from gooddata_eval.core.agentic.metric_skill import (
|
|
|
13
13
|
SimulatedResponseError,
|
|
14
14
|
_delete_metric,
|
|
15
15
|
_extract_metric_result,
|
|
16
|
+
_no_where_clause_hint,
|
|
16
17
|
_normalize_maql,
|
|
17
18
|
evaluate_agentic_metric_skill,
|
|
18
19
|
generate_simulated_response,
|
|
@@ -91,6 +92,54 @@ def test_normalize_maql_removes_select_wrapper():
|
|
|
91
92
|
assert _normalize_maql("(SELECT {metric/abc})") == "{metric/abc}"
|
|
92
93
|
|
|
93
94
|
|
|
95
|
+
def test_no_where_clause_hint_is_empty_when_a_candidate_has_a_where_clause():
|
|
96
|
+
assert _no_where_clause_hint(['SELECT {metric/foo} WHERE {label/status} = "active"']) == ""
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
def test_no_where_clause_hint_is_present_when_no_candidate_has_a_where_clause():
|
|
100
|
+
"""QA-29094 follow-up: whether to add a filter must not be left to the simulating LLM's
|
|
101
|
+
judgment of what the original request "implies" -- that fuzzy reasoning is exactly what
|
|
102
|
+
caused it to inject an unrequested filter in the first place."""
|
|
103
|
+
hint = _no_where_clause_hint(["SELECT SUM({fact/order_unit_quantity})"])
|
|
104
|
+
assert hint != ""
|
|
105
|
+
assert "no filter is needed" in hint
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
def test_no_where_clause_hint_stays_silent_if_any_candidate_has_a_where_clause():
|
|
109
|
+
"""PR #1760 review (Henry): _no_where_clause_hint used to see only expected_outputs[0].
|
|
110
|
+
A fixture like agent_metric_skill_4.json lists an unfiltered candidate first and a
|
|
111
|
+
filtered one second -- both accepted by _best_maql_match. Hinting "no filter needed"
|
|
112
|
+
off candidate 0 alone would steer the agent away from the filtered candidate even
|
|
113
|
+
though the scorer would still take it -- the mirror image of the original QA-29094 bug.
|
|
114
|
+
"""
|
|
115
|
+
candidates = [
|
|
116
|
+
"SELECT SUM({fact/order_unit_quantity})",
|
|
117
|
+
'SELECT SUM({fact/order_unit_quantity}) WHERE {label/order_status} = "Processed"',
|
|
118
|
+
]
|
|
119
|
+
assert _no_where_clause_hint(candidates) == ""
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
def test_no_where_clause_hint_ignores_where_inside_an_identifier():
|
|
123
|
+
"""CodeRabbit finding on PR #1760: a naive substring check treats the "where" inside
|
|
124
|
+
an identifier like {metric/somewhere_sales} as a real WHERE clause and wrongly stays
|
|
125
|
+
silent -- it must be stripped as a protected span before matching."""
|
|
126
|
+
assert _no_where_clause_hint(["SELECT {metric/somewhere_sales}"]) != ""
|
|
127
|
+
|
|
128
|
+
|
|
129
|
+
def test_no_where_clause_hint_ignores_where_inside_a_quoted_literal():
|
|
130
|
+
assert _no_where_clause_hint(['SELECT {metric/x} = "somewhere nearby"']) != ""
|
|
131
|
+
|
|
132
|
+
|
|
133
|
+
def test_no_where_clause_hint_ignores_where_inside_a_literal_with_an_escaped_quote():
|
|
134
|
+
"""CodeRabbit finding on PR #1760: an escaped quote inside a quoted literal ended the
|
|
135
|
+
protected-span match early, leaking the rest of the literal's text -- including a
|
|
136
|
+
standalone WHERE -- as unprotected. Uses _HINT_PROTECTED_RE (escape-aware), kept
|
|
137
|
+
separate from the shared _PROTECTED_RE that feeds the maql_correct comparator (PR
|
|
138
|
+
#1760 review, Henry) -- see test_normalize_maql_does_not_consume_escape_sequences."""
|
|
139
|
+
maql = 'SELECT {metric/x} = "Jane\\"s store WHERE something"'
|
|
140
|
+
assert _no_where_clause_hint([maql]) != ""
|
|
141
|
+
|
|
142
|
+
|
|
94
143
|
def test_generate_simulated_response_prompt_preserves_maql_fidelity(monkeypatch):
|
|
95
144
|
"""Regression test for a live-reproduced bug: the old prompt ("reply briefly",
|
|
96
145
|
no instruction to cover clauses the assistant didn't ask about) let the
|
|
@@ -111,19 +160,75 @@ def test_generate_simulated_response_prompt_preserves_maql_fidelity(monkeypatch)
|
|
|
111
160
|
monkeypatch.setitem(sys.modules, "openai", fake_openai_module)
|
|
112
161
|
|
|
113
162
|
expected_output = {"maql": 'SELECT {metric/spend_amount_-_cutcgco} WHERE {label/ecommerce_indicator_code} = "1"'}
|
|
114
|
-
generate_simulated_response(
|
|
163
|
+
generate_simulated_response(
|
|
164
|
+
"Which base metric should I use?", [expected_output], "I need a metric for spend amount"
|
|
165
|
+
)
|
|
115
166
|
|
|
116
167
|
call_kwargs = mock_client.chat.completions.create.call_args.kwargs
|
|
117
168
|
sent_prompt = call_kwargs["messages"][0]["content"]
|
|
118
169
|
|
|
119
170
|
assert expected_output["maql"] in sent_prompt
|
|
120
171
|
assert "verbatim" in sent_prompt
|
|
121
|
-
assert "
|
|
122
|
-
|
|
123
|
-
|
|
172
|
+
assert "filter" in sent_prompt.lower()
|
|
173
|
+
# Guards against a truncated reply mid-MAQL -- the LLM was cutting fidelity short under
|
|
174
|
+
# the old, lower budget before this was raised (see the docstring above).
|
|
124
175
|
assert call_kwargs["max_tokens"] >= 300
|
|
125
176
|
|
|
126
177
|
|
|
178
|
+
def test_generate_simulated_response_prompt_agrees_when_the_original_request_is_already_satisfied(monkeypatch):
|
|
179
|
+
"""Regression test for QA-29094: the old prompt told the simulated user to force every
|
|
180
|
+
clause of the ground-truth MAQL regardless of what the original request actually asked
|
|
181
|
+
for, so it would inject filters/constraints the user never mentioned even when the
|
|
182
|
+
assistant's proposal already matched the request. The prompt must now carry the
|
|
183
|
+
original request and instruct the simulated user to agree when it's already satisfied.
|
|
184
|
+
"""
|
|
185
|
+
monkeypatch.setenv("OPENAI_API_KEY", "test-key")
|
|
186
|
+
mock_client = MagicMock()
|
|
187
|
+
mock_response = MagicMock()
|
|
188
|
+
mock_response.choices = [MagicMock(message=MagicMock(content="ok"))]
|
|
189
|
+
mock_client.chat.completions.create.return_value = mock_response
|
|
190
|
+
fake_openai_module = types.SimpleNamespace(OpenAI=MagicMock(return_value=mock_client), OpenAIError=Exception)
|
|
191
|
+
monkeypatch.setitem(sys.modules, "openai", fake_openai_module)
|
|
192
|
+
|
|
193
|
+
original_question = "I need a metric for total ordered units called Total Order Quantity"
|
|
194
|
+
expected_output = {"maql": "SELECT SUM({fact/order_unit_quantity}) WHERE {fact/order_status} != 'cancelled'"}
|
|
195
|
+
generate_simulated_response("Should I create this metric?", [expected_output], original_question)
|
|
196
|
+
|
|
197
|
+
sent_prompt = mock_client.chat.completions.create.call_args.kwargs["messages"][0]["content"]
|
|
198
|
+
|
|
199
|
+
# Structural checks on the interpolated data -- robust to prompt-wording edits.
|
|
200
|
+
assert original_question in sent_prompt
|
|
201
|
+
assert expected_output["maql"] in sent_prompt
|
|
202
|
+
assert "reply briefly" not in sent_prompt.lower()
|
|
203
|
+
# A ground-truth MAQL with a WHERE clause must not trigger the no-filter-needed hint.
|
|
204
|
+
assert "no filter is needed" not in sent_prompt
|
|
205
|
+
|
|
206
|
+
|
|
207
|
+
def test_generate_simulated_response_prompt_handles_a_clarifying_question(monkeypatch):
|
|
208
|
+
"""QA-29094 follow-up: the two-branch prompt ("already satisfies" / "missing something")
|
|
209
|
+
both assume the assistant made a proposal -- but the dominant real case is the assistant
|
|
210
|
+
asking a clarifying question first (no proposal exists yet to judge as satisfying or not).
|
|
211
|
+
Without an explicit instruction, the simulating LLM could classify "nothing proposed yet"
|
|
212
|
+
as trivially "satisfied" and reply "yes, that works", leaving the agent no closer to a
|
|
213
|
+
usable metric and burning iterations."""
|
|
214
|
+
monkeypatch.setenv("OPENAI_API_KEY", "test-key")
|
|
215
|
+
mock_client = MagicMock()
|
|
216
|
+
mock_response = MagicMock()
|
|
217
|
+
mock_response.choices = [MagicMock(message=MagicMock(content="ok"))]
|
|
218
|
+
mock_client.chat.completions.create.return_value = mock_response
|
|
219
|
+
fake_openai_module = types.SimpleNamespace(OpenAI=MagicMock(return_value=mock_client), OpenAIError=Exception)
|
|
220
|
+
monkeypatch.setitem(sys.modules, "openai", fake_openai_module)
|
|
221
|
+
|
|
222
|
+
expected_output = {"maql": "SELECT SUM({fact/order_unit_quantity})"}
|
|
223
|
+
generate_simulated_response("Which base metric should I use?", [expected_output], "I need total ordered units")
|
|
224
|
+
|
|
225
|
+
sent_prompt = mock_client.chat.completions.create.call_args.kwargs["messages"][0]["content"]
|
|
226
|
+
|
|
227
|
+
assert "clarifying question" in sent_prompt
|
|
228
|
+
# No WHERE clause in the ground truth -- the no-filter hint must fire here too.
|
|
229
|
+
assert "no filter is needed" in sent_prompt
|
|
230
|
+
|
|
231
|
+
|
|
127
232
|
def test_normalize_maql_is_case_insensitive_for_keywords():
|
|
128
233
|
"""Regression test for a live-reproduced bug: 'FOR PREVIOUS(...)' vs
|
|
129
234
|
'FOR Previous(...)' scored as a mismatch even though MAQL keywords are
|
|
@@ -147,6 +252,19 @@ def test_normalize_maql_preserves_quoted_literal_case():
|
|
|
147
252
|
assert _normalize_maql('WHERE {label/status} = "Active"') != _normalize_maql('WHERE {label/status} = "active"')
|
|
148
253
|
|
|
149
254
|
|
|
255
|
+
def test_normalize_maql_does_not_consume_escape_sequences():
|
|
256
|
+
"""PR #1760 review (Henry): _PROTECTED_RE feeds this comparator (via
|
|
257
|
+
_casefold_outside_protected), so it must NOT treat \\X as an escape sequence unless
|
|
258
|
+
MAQL literals are confirmed to support backslash escaping (unconfirmed). A `\\"`
|
|
259
|
+
inside a literal must still end that literal at the next real quote -- not swallow
|
|
260
|
+
everything up to the following quoted value, which would leave a real keyword like
|
|
261
|
+
AND uncasefolded and a later literal's case wrongly casefolded."""
|
|
262
|
+
maql = 'SELECT {metric/x} WHERE {label/path} = "C:\\" AND {label/y} = "Active"'
|
|
263
|
+
normalized = _normalize_maql(maql)
|
|
264
|
+
assert "and {label/y}" in normalized # AND is a keyword outside the literal -- casefolded
|
|
265
|
+
assert '"Active"' in normalized # the second literal's case is untouched -- not "active"
|
|
266
|
+
|
|
267
|
+
|
|
150
268
|
def test_metric_run_result_fields():
|
|
151
269
|
r = MetricRunResult(
|
|
152
270
|
conversation_id="c1",
|
|
@@ -228,7 +346,7 @@ def test_run_agentic_metric_skill_closes_client_on_no_result():
|
|
|
228
346
|
mock_client.close.assert_called_once()
|
|
229
347
|
assert summary.pass_at_k is False
|
|
230
348
|
assert summary.best.metric_created is False
|
|
231
|
-
mock_sim.assert_called_once_with("I will work on that.", {"maql": "SELECT {metric/foo}"})
|
|
349
|
+
mock_sim.assert_called_once_with("I will work on that.", [{"maql": "SELECT {metric/foo}"}], "Create metric foo")
|
|
232
350
|
|
|
233
351
|
|
|
234
352
|
def test_run_agentic_metric_skill_uses_initial_conversation_for_run_0():
|
|
@@ -408,7 +526,7 @@ def test_generate_simulated_response_without_an_api_key():
|
|
|
408
526
|
patch.dict(os.environ, {}, clear=True),
|
|
409
527
|
pytest.raises(SimulatedResponseError, match="OPENAI_API_KEY"),
|
|
410
528
|
):
|
|
411
|
-
generate_simulated_response("Which brand field?", {"maql": "SELECT {metric/foo}"})
|
|
529
|
+
generate_simulated_response("Which brand field?", [{"maql": "SELECT {metric/foo}"}], "I need a metric for foo")
|
|
412
530
|
|
|
413
531
|
|
|
414
532
|
def test_generate_simulated_response_without_the_openai_package():
|
|
@@ -416,7 +534,7 @@ def test_generate_simulated_response_without_the_openai_package():
|
|
|
416
534
|
patch.dict(sys.modules, {"openai": None}),
|
|
417
535
|
pytest.raises(SimulatedResponseError, match="openai package is required"),
|
|
418
536
|
):
|
|
419
|
-
generate_simulated_response("Which brand field?", {"maql": "SELECT {metric/foo}"})
|
|
537
|
+
generate_simulated_response("Which brand field?", [{"maql": "SELECT {metric/foo}"}], "I need a metric for foo")
|
|
420
538
|
|
|
421
539
|
|
|
422
540
|
def test_run_agentic_metric_skill_fails_the_run_when_the_simulated_reply_cannot_be_generated():
|
|
@@ -448,7 +566,9 @@ def test_run_agentic_metric_skill_fails_the_run_when_the_simulated_reply_cannot_
|
|
|
448
566
|
assert summary.best.metric_created is False
|
|
449
567
|
assert summary.best.total_turns == 1.0
|
|
450
568
|
mock_client.close.assert_called_once()
|
|
451
|
-
mock_sim.assert_called_once_with(
|
|
569
|
+
mock_sim.assert_called_once_with(
|
|
570
|
+
"Which brand field should I count?", [{"maql": "SELECT {metric/foo}"}], "Create metric foo"
|
|
571
|
+
)
|
|
452
572
|
|
|
453
573
|
|
|
454
574
|
def test_run_agentic_metric_skill_accumulates_reasoning_steps_across_iterations():
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/cli/agentic_runner.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
{gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/agentic/__init__.py
RENAMED
|
File without changes
|
{gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/agentic/_catalog.py
RENAMED
|
File without changes
|
{gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/agentic/_langfuse.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
{gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/agentic/guardrail.py
RENAMED
|
File without changes
|
{gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/agentic/kda_skill.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
{gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/chat/__init__.py
RENAMED
|
File without changes
|
{gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/chat/sse_client.py
RENAMED
|
File without changes
|
|
File without changes
|
{gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/connection.py
RENAMED
|
File without changes
|
{gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/dataset/__init__.py
RENAMED
|
File without changes
|
|
File without changes
|
{gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/dataset/local.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/evaluators/base.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/evaluators/summary.py
RENAMED
|
File without changes
|
|
File without changes
|
{gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/langfuse/__init__.py
RENAMED
|
File without changes
|
{gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/langfuse/sink.py
RENAMED
|
File without changes
|
|
File without changes
|
{gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/reporting/__init__.py
RENAMED
|
File without changes
|
{gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/reporting/console.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/summary/__init__.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/tests/fixtures/sse_visualization_stream.txt
RENAMED
|
File without changes
|
|
File without changes
|
{gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/tests/test_agentic_general_question.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
{gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/tests/test_agentic_langfuse_trace.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/tests/test_metric_skill_evaluator.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{gooddata_eval-1.73.1.dev2 → gooddata_eval-1.73.1.dev3}/tests/test_visualization_evaluator.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|