gooddata-eval 1.72.0__tar.gz → 1.72.1.dev1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/PKG-INFO +2 -2
- {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/pyproject.toml +2 -2
- {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/src/gooddata_eval/core/agentic/alert_skill.py +112 -20
- gooddata_eval-1.72.1.dev1/tests/test_agentic_alert_skill.py +491 -0
- gooddata_eval-1.72.0/tests/test_agentic_alert_skill.py +0 -250
- {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/.gitignore +0 -0
- {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/LICENSE.txt +0 -0
- {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/Makefile +0 -0
- {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/README.md +0 -0
- {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/src/gooddata_eval/__init__.py +0 -0
- {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/src/gooddata_eval/_version.py +0 -0
- {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/src/gooddata_eval/cli/__init__.py +0 -0
- {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/src/gooddata_eval/cli/agentic_runner.py +0 -0
- {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/src/gooddata_eval/cli/main.py +0 -0
- {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/src/gooddata_eval/core/__init__.py +0 -0
- {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/src/gooddata_eval/core/agentic/__init__.py +0 -0
- {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/src/gooddata_eval/core/agentic/_catalog.py +0 -0
- {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/src/gooddata_eval/core/agentic/_langfuse.py +0 -0
- {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/src/gooddata_eval/core/agentic/conversation.py +0 -0
- {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/src/gooddata_eval/core/agentic/general_question.py +0 -0
- {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/src/gooddata_eval/core/agentic/guardrail.py +0 -0
- {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/src/gooddata_eval/core/agentic/metric_skill.py +0 -0
- {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/src/gooddata_eval/core/agentic/search_tool.py +0 -0
- {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/src/gooddata_eval/core/agentic/visualization.py +0 -0
- {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/src/gooddata_eval/core/chat/__init__.py +0 -0
- {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/src/gooddata_eval/core/chat/sse_client.py +0 -0
- {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/src/gooddata_eval/core/config.py +0 -0
- {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/src/gooddata_eval/core/connection.py +0 -0
- {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/src/gooddata_eval/core/dataset/__init__.py +0 -0
- {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/src/gooddata_eval/core/dataset/langfuse_source.py +0 -0
- {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/src/gooddata_eval/core/dataset/local.py +0 -0
- {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/src/gooddata_eval/core/evaluators/__init__.py +0 -0
- {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/src/gooddata_eval/core/evaluators/_deep_subset.py +0 -0
- {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/src/gooddata_eval/core/evaluators/_llm_judge.py +0 -0
- {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/src/gooddata_eval/core/evaluators/_text_utils.py +0 -0
- {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/src/gooddata_eval/core/evaluators/alert_skill.py +0 -0
- {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/src/gooddata_eval/core/evaluators/base.py +0 -0
- {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/src/gooddata_eval/core/evaluators/general_question.py +0 -0
- {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/src/gooddata_eval/core/evaluators/guardrail.py +0 -0
- {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/src/gooddata_eval/core/evaluators/metric_skill.py +0 -0
- {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/src/gooddata_eval/core/evaluators/search_tool.py +0 -0
- {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/src/gooddata_eval/core/evaluators/summary.py +0 -0
- {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/src/gooddata_eval/core/evaluators/visualization.py +0 -0
- {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/src/gooddata_eval/core/langfuse/__init__.py +0 -0
- {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/src/gooddata_eval/core/langfuse/sink.py +0 -0
- {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/src/gooddata_eval/core/models.py +0 -0
- {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/src/gooddata_eval/core/reporting/__init__.py +0 -0
- {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/src/gooddata_eval/core/reporting/console.py +0 -0
- {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/src/gooddata_eval/core/reporting/json_report.py +0 -0
- {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/src/gooddata_eval/core/runner.py +0 -0
- {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/src/gooddata_eval/core/scoring.py +0 -0
- {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/src/gooddata_eval/core/summary/__init__.py +0 -0
- {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/src/gooddata_eval/core/summary/http_client.py +0 -0
- {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/src/gooddata_eval/core/workspace.py +0 -0
- {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/tests/__init__.py +0 -0
- {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/tests/conftest.py +0 -0
- {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/tests/fixtures/sample_dataset/metric_skill_create.json +0 -0
- {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/tests/fixtures/sample_dataset/visualization_revenue.json +0 -0
- {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/tests/fixtures/sse_visualization_stream.txt +0 -0
- {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/tests/test_agentic_conversation.py +0 -0
- {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/tests/test_agentic_general_question.py +0 -0
- {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/tests/test_agentic_guardrail.py +0 -0
- {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/tests/test_agentic_metric_skill.py +0 -0
- {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/tests/test_agentic_run_context.py +0 -0
- {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/tests/test_agentic_search_tool.py +0 -0
- {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/tests/test_agentic_visualization.py +0 -0
- {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/tests/test_alert_skill_evaluator.py +0 -0
- {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/tests/test_cli.py +0 -0
- {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/tests/test_connection.py +0 -0
- {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/tests/test_deep_subset.py +0 -0
- {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/tests/test_langfuse_sink.py +0 -0
- {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/tests/test_langfuse_source.py +0 -0
- {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/tests/test_llm_judge.py +0 -0
- {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/tests/test_local_loader.py +0 -0
- {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/tests/test_metric_skill_evaluator.py +0 -0
- {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/tests/test_models.py +0 -0
- {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/tests/test_reporting.py +0 -0
- {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/tests/test_runner.py +0 -0
- {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/tests/test_scoring.py +0 -0
- {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/tests/test_search_tool_evaluator.py +0 -0
- {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/tests/test_sse_client.py +0 -0
- {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/tests/test_summary_client.py +0 -0
- {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/tests/test_summary_evaluator.py +0 -0
- {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/tests/test_text_evaluators.py +0 -0
- {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/tests/test_visualization_evaluator.py +0 -0
- {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/tests/test_workspace.py +0 -0
- {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/tox.ini +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: gooddata-eval
|
|
3
|
-
Version: 1.72.
|
|
3
|
+
Version: 1.72.1.dev1
|
|
4
4
|
Summary: Evaluate the GoodData AI agent against your own questions and models.
|
|
5
5
|
Project-URL: Source, https://github.com/gooddata/gooddata-python-sdk
|
|
6
6
|
Author-email: GoodData <support@gooddata.com>
|
|
@@ -17,7 +17,7 @@ Classifier: Topic :: Scientific/Engineering
|
|
|
17
17
|
Classifier: Topic :: Software Development
|
|
18
18
|
Classifier: Typing :: Typed
|
|
19
19
|
Requires-Python: >=3.10
|
|
20
|
-
Requires-Dist: gooddata-sdk~=1.72.
|
|
20
|
+
Requires-Dist: gooddata-sdk~=1.72.1.dev1
|
|
21
21
|
Requires-Dist: httpx<1.0,>=0.27
|
|
22
22
|
Requires-Dist: orjson<4.0.0,>=3.9.15
|
|
23
23
|
Requires-Dist: pydantic<3.0,>=2.6
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
# (C) 2026 GoodData Corporation
|
|
2
2
|
[project]
|
|
3
3
|
name = "gooddata-eval"
|
|
4
|
-
version = "1.72.
|
|
4
|
+
version = "1.72.1.dev1"
|
|
5
5
|
description = "Evaluate the GoodData AI agent against your own questions and models."
|
|
6
6
|
readme = "README.md"
|
|
7
7
|
license = "MIT"
|
|
@@ -11,7 +11,7 @@ authors = [
|
|
|
11
11
|
keywords = ["gooddata", "ai", "evaluation", "llm", "analytics", "cli"]
|
|
12
12
|
requires-python = ">=3.10"
|
|
13
13
|
dependencies = [
|
|
14
|
-
"gooddata-sdk~=1.72.
|
|
14
|
+
"gooddata-sdk~=1.72.1.dev1",
|
|
15
15
|
"httpx>=0.27,<1.0",
|
|
16
16
|
"orjson>=3.9.15,<4.0.0",
|
|
17
17
|
"pydantic>=2.6,<3.0",
|
{gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/src/gooddata_eval/core/agentic/alert_skill.py
RENAMED
|
@@ -27,6 +27,13 @@ _DEFAULT_MAX_ITERATIONS = 6
|
|
|
27
27
|
_TRIGGER_DISPLAY_TO_API = {"Every time": "ALWAYS", "One time": "ONCE"}
|
|
28
28
|
_ALWAYS_TRIGGER_VALUES = {"Every time", "ALWAYS", "not specified"}
|
|
29
29
|
|
|
30
|
+
_TRIGGER_INSTRUCTIONS = {
|
|
31
|
+
"ALWAYS": (
|
|
32
|
+
"alert me EVERY TIME the condition is met — not once per day, week or month, and not only the first time"
|
|
33
|
+
),
|
|
34
|
+
"ONCE": "alert me ONLY THE FIRST TIME the condition is met, then stop",
|
|
35
|
+
}
|
|
36
|
+
|
|
30
37
|
|
|
31
38
|
def _to_number(value: object) -> float | int | None:
|
|
32
39
|
"""Convert string/number to int or float, None on failure."""
|
|
@@ -94,9 +101,11 @@ def _check_trigger(expected: CatalogMetricAlert, actual_args: dict) -> bool:
|
|
|
94
101
|
|
|
95
102
|
def _check_filters(expected: CatalogMetricAlert, actual_args: dict) -> bool:
|
|
96
103
|
exp_filters = expected.filters
|
|
97
|
-
act_filters = actual_args.get("filters", actual_args.get("attribute_filters"))
|
|
98
|
-
if
|
|
104
|
+
act_filters = actual_args.get("filters", actual_args.get("attribute_filters")) or []
|
|
105
|
+
if exp_filters is None:
|
|
99
106
|
return True
|
|
107
|
+
if not exp_filters:
|
|
108
|
+
return not act_filters
|
|
100
109
|
if not act_filters:
|
|
101
110
|
return False
|
|
102
111
|
return _deep_subset(exp_filters, act_filters)
|
|
@@ -132,8 +141,15 @@ def generate_simulated_alert_response(
|
|
|
132
141
|
agent_message: str,
|
|
133
142
|
expected: CatalogMetricAlert,
|
|
134
143
|
conversation_history: list,
|
|
144
|
+
question: str = "",
|
|
135
145
|
) -> str:
|
|
136
|
-
"""Stateful sim-user reply for alert-skill conversation (gpt-4o).
|
|
146
|
+
"""Stateful sim-user reply for alert-skill conversation (gpt-4o).
|
|
147
|
+
|
|
148
|
+
``question`` is the fixture's original request. The sim-user is first called with an empty
|
|
149
|
+
history — the opening question went straight to the agent, never to the sim-user — so
|
|
150
|
+
without it rule 5's "the filters your original request implies" refers to text the model
|
|
151
|
+
cannot see. Optional (defaults to "") to keep the signature backwards compatible.
|
|
152
|
+
"""
|
|
137
153
|
if _OpenAI is None:
|
|
138
154
|
raise RuntimeError(
|
|
139
155
|
"openai package is required for generate_simulated_alert_response. "
|
|
@@ -147,30 +163,83 @@ def generate_simulated_alert_response(
|
|
|
147
163
|
|
|
148
164
|
metric = expected.metric_id or "not specified"
|
|
149
165
|
operator = expected.operator
|
|
150
|
-
|
|
166
|
+
# BETWEEN / NOT_BETWEEN carry their value in threshold_from/threshold_to, so `threshold` is
|
|
167
|
+
# None for them. Rule 3 asks the sim-user to verify the threshold, and reporting "not
|
|
168
|
+
# specified" made it demand the agent delete both bounds of a BETWEEN condition — an
|
|
169
|
+
# impossible request that burned every iteration without the alert ever being created.
|
|
170
|
+
threshold: str | float | int
|
|
171
|
+
if expected.operator in ("BETWEEN", "NOT_BETWEEN") and (
|
|
172
|
+
expected.threshold_from is not None or expected.threshold_to is not None
|
|
173
|
+
):
|
|
174
|
+
threshold = f"between {expected.threshold_from} and {expected.threshold_to}"
|
|
175
|
+
elif expected.threshold is not None:
|
|
176
|
+
threshold = expected.threshold
|
|
177
|
+
else:
|
|
178
|
+
threshold = "not specified"
|
|
151
179
|
recipients = ", ".join(expected.recipients) if expected.recipients else "not specified"
|
|
152
180
|
trigger = expected.trigger
|
|
153
181
|
filters = expected.filters
|
|
154
182
|
|
|
155
|
-
|
|
156
|
-
|
|
157
|
-
|
|
158
|
-
|
|
159
|
-
|
|
183
|
+
# "not specified" is the normalizer's stand-in for an absent trigger, which the product
|
|
184
|
+
# persists as its ALWAYS default and `_check_trigger` asserts as ALWAYS. Both the cadence to
|
|
185
|
+
# ask for (rule 6) and the goal text (rule 1) use the resolved value: reporting the raw
|
|
186
|
+
# placeholder made rule 3 treat the trigger as unconstrained, so the sim-user would confirm a
|
|
187
|
+
# ONCE/ONCE_PER_INTERVAL proposal that the assertion then failed.
|
|
188
|
+
trigger_key = "ALWAYS" if trigger in _ALWAYS_TRIGGER_VALUES else trigger
|
|
189
|
+
trigger_request = _TRIGGER_INSTRUCTIONS.get(trigger_key, f"set the trigger to {trigger}")
|
|
190
|
+
|
|
191
|
+
# Three branches, matching the three states of `expected.filters`. `[]` and `None` must not
|
|
192
|
+
# share one: telling the sim-user "you want NO filters" on an unstated expectation makes it
|
|
193
|
+
# refuse filters the request genuinely implies (e.g. "orders from the United States"), which
|
|
194
|
+
# quietly turns that fixture into a weaker test rather than a failing one.
|
|
195
|
+
if filters:
|
|
196
|
+
filters_rule = (
|
|
197
|
+
f"5. Your alert needs exactly these filters and NOTHING else: {filters}. "
|
|
198
|
+
"If the agent offers, proposes or asks about any further date/time window, "
|
|
199
|
+
"evaluation period or granularity, refuse it and repeat that these are the only "
|
|
200
|
+
"filters you want.\n"
|
|
201
|
+
)
|
|
202
|
+
elif filters == []:
|
|
203
|
+
filters_rule = (
|
|
204
|
+
"5. Your alert must have NO filters and NO date/time window — it evaluates over all time. "
|
|
205
|
+
"If the agent asks which time period each check should cover, or offers a choice such as "
|
|
206
|
+
"'last Day / Week / Month', do NOT pick one: reply that you want no date filter at all, "
|
|
207
|
+
"all time. Never invent a period, a granularity or an 'evaluate each run on a X basis' "
|
|
208
|
+
"instruction the goal did not ask for.\n"
|
|
209
|
+
)
|
|
210
|
+
else:
|
|
211
|
+
filters_rule = (
|
|
212
|
+
"5. Ask only for the filters your original request implies — do not invent an evaluation "
|
|
213
|
+
"period, granularity or date window that was not requested. If the agent offers a choice "
|
|
214
|
+
"such as 'last Day / Week / Month' that your request never mentioned, say you do not want "
|
|
215
|
+
"a date window.\n"
|
|
216
|
+
)
|
|
217
|
+
|
|
218
|
+
original_request = f'Your original request to the agent was: "{question}"\n' if question else ""
|
|
219
|
+
|
|
160
220
|
system_prompt = (
|
|
161
221
|
"You are a user requesting creation of an alert for a metric from an AI agent. "
|
|
162
222
|
"Respond naturally but always steer toward the exact values you were given.\n"
|
|
163
|
-
|
|
223
|
+
+ original_request
|
|
224
|
+
+ "Rules you MUST follow:\n"
|
|
164
225
|
f"1. Your goal: metric={metric}, operator={operator}, threshold={threshold}, "
|
|
165
|
-
f"recipients={recipients}, trigger={
|
|
226
|
+
f"recipients={recipients}, trigger={trigger_key}" + (f", filters={filters}" if filters else "") + ".\n"
|
|
166
227
|
"2. Never revert or change a decision that was already confirmed in a previous turn.\n"
|
|
167
|
-
"3. If the agent shows a final summary
|
|
168
|
-
"
|
|
169
|
-
"
|
|
228
|
+
"3. If the agent shows a final summary, an alert proposal or asks for confirmation, check "
|
|
229
|
+
" ALL of these against your goal: recipients, trigger (how often you are alerted), "
|
|
230
|
+
" filters / time window, threshold and operator. If ANY of them differs — for example the "
|
|
231
|
+
" summary says 'once per day/week/month' but your goal is every time, or it lists a date "
|
|
232
|
+
" filter you never asked for — do NOT confirm: name the wrong field, state the correct "
|
|
233
|
+
" value and ask the agent to fix it. Say 'Yes, please proceed to create the alert.' ONLY "
|
|
234
|
+
" when every one of those fields matches your goal.\n"
|
|
235
|
+
" A field your goal reports as 'not specified' is one you have NO expectation about: "
|
|
236
|
+
" accept whatever the agent chose for it and never ask for it to be removed.\n"
|
|
170
237
|
"4. Proactively include your email recipient in your first reply. "
|
|
171
238
|
" Do not wait for the agent to ask — state it alongside the metric and condition answers.\n"
|
|
172
|
-
+
|
|
173
|
-
+ "
|
|
239
|
+
+ filters_rule
|
|
240
|
+
+ f"6. Proactively state how often you want to be alerted in your first reply: {trigger_request}. "
|
|
241
|
+
" Repeat it if the agent proposes a different cadence.\n"
|
|
242
|
+
"Reply concisely and directly."
|
|
174
243
|
)
|
|
175
244
|
|
|
176
245
|
messages: list = [{"role": "system", "content": system_prompt}]
|
|
@@ -257,6 +326,29 @@ def _case_insensitive_get(d: dict, *keys: str) -> Any:
|
|
|
257
326
|
return None
|
|
258
327
|
|
|
259
328
|
|
|
329
|
+
_NO_FILTER_MARKERS = ("none", "all time")
|
|
330
|
+
|
|
331
|
+
|
|
332
|
+
def _normalize_expected_filters(expected: dict) -> list | str | None:
|
|
333
|
+
"""
|
|
334
|
+
* ``Filters`` list -> that list (exact expectation)
|
|
335
|
+
* "None (All time)" in either -> ``[]`` (stated: no filters; extras fail)
|
|
336
|
+
* anything else / absent -> ``None`` (unstated; filters not asserted)
|
|
337
|
+
"""
|
|
338
|
+
filters = _case_insensitive_get(expected, "filters")
|
|
339
|
+
if isinstance(filters, list):
|
|
340
|
+
return filters
|
|
341
|
+
time_window = _case_insensitive_get(expected, "time window/filters", "time_window")
|
|
342
|
+
for candidate in (filters, time_window):
|
|
343
|
+
if isinstance(candidate, str) and any(kw in candidate.lower() for kw in _NO_FILTER_MARKERS):
|
|
344
|
+
return []
|
|
345
|
+
# Prose that is not a no-filter marker ("Product Category = X") describes a filter without
|
|
346
|
+
# encoding it, so it cannot be compared: returning it made `_check_filters` fall through to
|
|
347
|
+
# `_deep_subset(str, list)`, which can never match. `None` is what the contract above
|
|
348
|
+
# promises — the sim-user derives such filters from the original request instead.
|
|
349
|
+
return None
|
|
350
|
+
|
|
351
|
+
|
|
260
352
|
def _normalize_expected_output(expected: dict) -> CatalogMetricAlert:
|
|
261
353
|
"""Parse expected_output dict into CatalogMetricAlert, accepting display-format or internal-format keys."""
|
|
262
354
|
operator = _case_insensitive_get(expected, "operator") or "GREATER_THAN"
|
|
@@ -280,9 +372,7 @@ def _normalize_expected_output(expected: dict) -> CatalogMetricAlert:
|
|
|
280
372
|
else:
|
|
281
373
|
recipients = list(raw_recip)
|
|
282
374
|
|
|
283
|
-
filters =
|
|
284
|
-
if isinstance(filters, str) and any(kw in filters for kw in ("None", "All time")):
|
|
285
|
-
filters = None
|
|
375
|
+
filters = _normalize_expected_filters(expected)
|
|
286
376
|
|
|
287
377
|
return CatalogMetricAlert(
|
|
288
378
|
operator=operator,
|
|
@@ -377,7 +467,9 @@ def run_agentic_alert_skill(
|
|
|
377
467
|
# Stop before generating a follow-up for the last iteration
|
|
378
468
|
if _iteration >= max_iterations - 1:
|
|
379
469
|
break
|
|
380
|
-
follow_up = generate_simulated_alert_response(
|
|
470
|
+
follow_up = generate_simulated_alert_response(
|
|
471
|
+
response_text, expected, conversation_history, question=question
|
|
472
|
+
)
|
|
381
473
|
# Record this exchange so the next call has full history
|
|
382
474
|
conversation_history.append({"role": "assistant", "content": response_text})
|
|
383
475
|
conversation_history.append({"role": "user", "content": follow_up})
|