gooddata-eval 1.72.0__tar.gz → 1.72.1.dev1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (87) hide show
  1. {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/PKG-INFO +2 -2
  2. {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/pyproject.toml +2 -2
  3. {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/src/gooddata_eval/core/agentic/alert_skill.py +112 -20
  4. gooddata_eval-1.72.1.dev1/tests/test_agentic_alert_skill.py +491 -0
  5. gooddata_eval-1.72.0/tests/test_agentic_alert_skill.py +0 -250
  6. {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/.gitignore +0 -0
  7. {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/LICENSE.txt +0 -0
  8. {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/Makefile +0 -0
  9. {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/README.md +0 -0
  10. {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/src/gooddata_eval/__init__.py +0 -0
  11. {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/src/gooddata_eval/_version.py +0 -0
  12. {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/src/gooddata_eval/cli/__init__.py +0 -0
  13. {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/src/gooddata_eval/cli/agentic_runner.py +0 -0
  14. {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/src/gooddata_eval/cli/main.py +0 -0
  15. {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/src/gooddata_eval/core/__init__.py +0 -0
  16. {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/src/gooddata_eval/core/agentic/__init__.py +0 -0
  17. {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/src/gooddata_eval/core/agentic/_catalog.py +0 -0
  18. {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/src/gooddata_eval/core/agentic/_langfuse.py +0 -0
  19. {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/src/gooddata_eval/core/agentic/conversation.py +0 -0
  20. {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/src/gooddata_eval/core/agentic/general_question.py +0 -0
  21. {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/src/gooddata_eval/core/agentic/guardrail.py +0 -0
  22. {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/src/gooddata_eval/core/agentic/metric_skill.py +0 -0
  23. {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/src/gooddata_eval/core/agentic/search_tool.py +0 -0
  24. {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/src/gooddata_eval/core/agentic/visualization.py +0 -0
  25. {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/src/gooddata_eval/core/chat/__init__.py +0 -0
  26. {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/src/gooddata_eval/core/chat/sse_client.py +0 -0
  27. {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/src/gooddata_eval/core/config.py +0 -0
  28. {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/src/gooddata_eval/core/connection.py +0 -0
  29. {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/src/gooddata_eval/core/dataset/__init__.py +0 -0
  30. {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/src/gooddata_eval/core/dataset/langfuse_source.py +0 -0
  31. {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/src/gooddata_eval/core/dataset/local.py +0 -0
  32. {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/src/gooddata_eval/core/evaluators/__init__.py +0 -0
  33. {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/src/gooddata_eval/core/evaluators/_deep_subset.py +0 -0
  34. {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/src/gooddata_eval/core/evaluators/_llm_judge.py +0 -0
  35. {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/src/gooddata_eval/core/evaluators/_text_utils.py +0 -0
  36. {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/src/gooddata_eval/core/evaluators/alert_skill.py +0 -0
  37. {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/src/gooddata_eval/core/evaluators/base.py +0 -0
  38. {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/src/gooddata_eval/core/evaluators/general_question.py +0 -0
  39. {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/src/gooddata_eval/core/evaluators/guardrail.py +0 -0
  40. {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/src/gooddata_eval/core/evaluators/metric_skill.py +0 -0
  41. {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/src/gooddata_eval/core/evaluators/search_tool.py +0 -0
  42. {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/src/gooddata_eval/core/evaluators/summary.py +0 -0
  43. {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/src/gooddata_eval/core/evaluators/visualization.py +0 -0
  44. {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/src/gooddata_eval/core/langfuse/__init__.py +0 -0
  45. {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/src/gooddata_eval/core/langfuse/sink.py +0 -0
  46. {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/src/gooddata_eval/core/models.py +0 -0
  47. {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/src/gooddata_eval/core/reporting/__init__.py +0 -0
  48. {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/src/gooddata_eval/core/reporting/console.py +0 -0
  49. {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/src/gooddata_eval/core/reporting/json_report.py +0 -0
  50. {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/src/gooddata_eval/core/runner.py +0 -0
  51. {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/src/gooddata_eval/core/scoring.py +0 -0
  52. {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/src/gooddata_eval/core/summary/__init__.py +0 -0
  53. {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/src/gooddata_eval/core/summary/http_client.py +0 -0
  54. {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/src/gooddata_eval/core/workspace.py +0 -0
  55. {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/tests/__init__.py +0 -0
  56. {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/tests/conftest.py +0 -0
  57. {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/tests/fixtures/sample_dataset/metric_skill_create.json +0 -0
  58. {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/tests/fixtures/sample_dataset/visualization_revenue.json +0 -0
  59. {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/tests/fixtures/sse_visualization_stream.txt +0 -0
  60. {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/tests/test_agentic_conversation.py +0 -0
  61. {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/tests/test_agentic_general_question.py +0 -0
  62. {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/tests/test_agentic_guardrail.py +0 -0
  63. {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/tests/test_agentic_metric_skill.py +0 -0
  64. {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/tests/test_agentic_run_context.py +0 -0
  65. {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/tests/test_agentic_search_tool.py +0 -0
  66. {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/tests/test_agentic_visualization.py +0 -0
  67. {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/tests/test_alert_skill_evaluator.py +0 -0
  68. {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/tests/test_cli.py +0 -0
  69. {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/tests/test_connection.py +0 -0
  70. {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/tests/test_deep_subset.py +0 -0
  71. {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/tests/test_langfuse_sink.py +0 -0
  72. {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/tests/test_langfuse_source.py +0 -0
  73. {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/tests/test_llm_judge.py +0 -0
  74. {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/tests/test_local_loader.py +0 -0
  75. {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/tests/test_metric_skill_evaluator.py +0 -0
  76. {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/tests/test_models.py +0 -0
  77. {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/tests/test_reporting.py +0 -0
  78. {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/tests/test_runner.py +0 -0
  79. {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/tests/test_scoring.py +0 -0
  80. {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/tests/test_search_tool_evaluator.py +0 -0
  81. {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/tests/test_sse_client.py +0 -0
  82. {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/tests/test_summary_client.py +0 -0
  83. {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/tests/test_summary_evaluator.py +0 -0
  84. {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/tests/test_text_evaluators.py +0 -0
  85. {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/tests/test_visualization_evaluator.py +0 -0
  86. {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/tests/test_workspace.py +0 -0
  87. {gooddata_eval-1.72.0 → gooddata_eval-1.72.1.dev1}/tox.ini +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: gooddata-eval
3
- Version: 1.72.0
3
+ Version: 1.72.1.dev1
4
4
  Summary: Evaluate the GoodData AI agent against your own questions and models.
5
5
  Project-URL: Source, https://github.com/gooddata/gooddata-python-sdk
6
6
  Author-email: GoodData <support@gooddata.com>
@@ -17,7 +17,7 @@ Classifier: Topic :: Scientific/Engineering
17
17
  Classifier: Topic :: Software Development
18
18
  Classifier: Typing :: Typed
19
19
  Requires-Python: >=3.10
20
- Requires-Dist: gooddata-sdk~=1.72.0
20
+ Requires-Dist: gooddata-sdk~=1.72.1.dev1
21
21
  Requires-Dist: httpx<1.0,>=0.27
22
22
  Requires-Dist: orjson<4.0.0,>=3.9.15
23
23
  Requires-Dist: pydantic<3.0,>=2.6
@@ -1,7 +1,7 @@
1
1
  # (C) 2026 GoodData Corporation
2
2
  [project]
3
3
  name = "gooddata-eval"
4
- version = "1.72.0"
4
+ version = "1.72.1.dev1"
5
5
  description = "Evaluate the GoodData AI agent against your own questions and models."
6
6
  readme = "README.md"
7
7
  license = "MIT"
@@ -11,7 +11,7 @@ authors = [
11
11
  keywords = ["gooddata", "ai", "evaluation", "llm", "analytics", "cli"]
12
12
  requires-python = ">=3.10"
13
13
  dependencies = [
14
- "gooddata-sdk~=1.72.0",
14
+ "gooddata-sdk~=1.72.1.dev1",
15
15
  "httpx>=0.27,<1.0",
16
16
  "orjson>=3.9.15,<4.0.0",
17
17
  "pydantic>=2.6,<3.0",
@@ -27,6 +27,13 @@ _DEFAULT_MAX_ITERATIONS = 6
27
27
  _TRIGGER_DISPLAY_TO_API = {"Every time": "ALWAYS", "One time": "ONCE"}
28
28
  _ALWAYS_TRIGGER_VALUES = {"Every time", "ALWAYS", "not specified"}
29
29
 
30
+ _TRIGGER_INSTRUCTIONS = {
31
+ "ALWAYS": (
32
+ "alert me EVERY TIME the condition is met — not once per day, week or month, and not only the first time"
33
+ ),
34
+ "ONCE": "alert me ONLY THE FIRST TIME the condition is met, then stop",
35
+ }
36
+
30
37
 
31
38
  def _to_number(value: object) -> float | int | None:
32
39
  """Convert string/number to int or float, None on failure."""
@@ -94,9 +101,11 @@ def _check_trigger(expected: CatalogMetricAlert, actual_args: dict) -> bool:
94
101
 
95
102
  def _check_filters(expected: CatalogMetricAlert, actual_args: dict) -> bool:
96
103
  exp_filters = expected.filters
97
- act_filters = actual_args.get("filters", actual_args.get("attribute_filters"))
98
- if not exp_filters:
104
+ act_filters = actual_args.get("filters", actual_args.get("attribute_filters")) or []
105
+ if exp_filters is None:
99
106
  return True
107
+ if not exp_filters:
108
+ return not act_filters
100
109
  if not act_filters:
101
110
  return False
102
111
  return _deep_subset(exp_filters, act_filters)
@@ -132,8 +141,15 @@ def generate_simulated_alert_response(
132
141
  agent_message: str,
133
142
  expected: CatalogMetricAlert,
134
143
  conversation_history: list,
144
+ question: str = "",
135
145
  ) -> str:
136
- """Stateful sim-user reply for alert-skill conversation (gpt-4o)."""
146
+ """Stateful sim-user reply for alert-skill conversation (gpt-4o).
147
+
148
+ ``question`` is the fixture's original request. The sim-user is first called with an empty
149
+ history — the opening question went straight to the agent, never to the sim-user — so
150
+ without it rule 5's "the filters your original request implies" refers to text the model
151
+ cannot see. Optional (defaults to "") to keep the signature backwards compatible.
152
+ """
137
153
  if _OpenAI is None:
138
154
  raise RuntimeError(
139
155
  "openai package is required for generate_simulated_alert_response. "
@@ -147,30 +163,83 @@ def generate_simulated_alert_response(
147
163
 
148
164
  metric = expected.metric_id or "not specified"
149
165
  operator = expected.operator
150
- threshold = expected.threshold if expected.threshold is not None else "not specified"
166
+ # BETWEEN / NOT_BETWEEN carry their value in threshold_from/threshold_to, so `threshold` is
167
+ # None for them. Rule 3 asks the sim-user to verify the threshold, and reporting "not
168
+ # specified" made it demand the agent delete both bounds of a BETWEEN condition — an
169
+ # impossible request that burned every iteration without the alert ever being created.
170
+ threshold: str | float | int
171
+ if expected.operator in ("BETWEEN", "NOT_BETWEEN") and (
172
+ expected.threshold_from is not None or expected.threshold_to is not None
173
+ ):
174
+ threshold = f"between {expected.threshold_from} and {expected.threshold_to}"
175
+ elif expected.threshold is not None:
176
+ threshold = expected.threshold
177
+ else:
178
+ threshold = "not specified"
151
179
  recipients = ", ".join(expected.recipients) if expected.recipients else "not specified"
152
180
  trigger = expected.trigger
153
181
  filters = expected.filters
154
182
 
155
- trigger_line = (
156
- f"5. Proactively tell the agent the trigger is '{trigger}' in your first reply.\n"
157
- if trigger not in _ALWAYS_TRIGGER_VALUES
158
- else ""
159
- )
183
+ # "not specified" is the normalizer's stand-in for an absent trigger, which the product
184
+ # persists as its ALWAYS default and `_check_trigger` asserts as ALWAYS. Both the cadence to
185
+ # ask for (rule 6) and the goal text (rule 1) use the resolved value: reporting the raw
186
+ # placeholder made rule 3 treat the trigger as unconstrained, so the sim-user would confirm a
187
+ # ONCE/ONCE_PER_INTERVAL proposal that the assertion then failed.
188
+ trigger_key = "ALWAYS" if trigger in _ALWAYS_TRIGGER_VALUES else trigger
189
+ trigger_request = _TRIGGER_INSTRUCTIONS.get(trigger_key, f"set the trigger to {trigger}")
190
+
191
+ # Three branches, matching the three states of `expected.filters`. `[]` and `None` must not
192
+ # share one: telling the sim-user "you want NO filters" on an unstated expectation makes it
193
+ # refuse filters the request genuinely implies (e.g. "orders from the United States"), which
194
+ # quietly turns that fixture into a weaker test rather than a failing one.
195
+ if filters:
196
+ filters_rule = (
197
+ f"5. Your alert needs exactly these filters and NOTHING else: {filters}. "
198
+ "If the agent offers, proposes or asks about any further date/time window, "
199
+ "evaluation period or granularity, refuse it and repeat that these are the only "
200
+ "filters you want.\n"
201
+ )
202
+ elif filters == []:
203
+ filters_rule = (
204
+ "5. Your alert must have NO filters and NO date/time window — it evaluates over all time. "
205
+ "If the agent asks which time period each check should cover, or offers a choice such as "
206
+ "'last Day / Week / Month', do NOT pick one: reply that you want no date filter at all, "
207
+ "all time. Never invent a period, a granularity or an 'evaluate each run on a X basis' "
208
+ "instruction the goal did not ask for.\n"
209
+ )
210
+ else:
211
+ filters_rule = (
212
+ "5. Ask only for the filters your original request implies — do not invent an evaluation "
213
+ "period, granularity or date window that was not requested. If the agent offers a choice "
214
+ "such as 'last Day / Week / Month' that your request never mentioned, say you do not want "
215
+ "a date window.\n"
216
+ )
217
+
218
+ original_request = f'Your original request to the agent was: "{question}"\n' if question else ""
219
+
160
220
  system_prompt = (
161
221
  "You are a user requesting creation of an alert for a metric from an AI agent. "
162
222
  "Respond naturally but always steer toward the exact values you were given.\n"
163
- "Rules you MUST follow:\n"
223
+ + original_request
224
+ + "Rules you MUST follow:\n"
164
225
  f"1. Your goal: metric={metric}, operator={operator}, threshold={threshold}, "
165
- f"recipients={recipients}, trigger={trigger}" + (f", filters={filters}" if filters else "") + ".\n"
226
+ f"recipients={recipients}, trigger={trigger_key}" + (f", filters={filters}" if filters else "") + ".\n"
166
227
  "2. Never revert or change a decision that was already confirmed in a previous turn.\n"
167
- "3. If the agent shows a final summary and asks for confirmation, verify that the "
168
- " recipients match your goal. If they differ, correct them. "
169
- " Once recipients are correct, say 'Yes, please proceed to create the alert.'\n"
228
+ "3. If the agent shows a final summary, an alert proposal or asks for confirmation, check "
229
+ " ALL of these against your goal: recipients, trigger (how often you are alerted), "
230
+ " filters / time window, threshold and operator. If ANY of them differs — for example the "
231
+ " summary says 'once per day/week/month' but your goal is every time, or it lists a date "
232
+ " filter you never asked for — do NOT confirm: name the wrong field, state the correct "
233
+ " value and ask the agent to fix it. Say 'Yes, please proceed to create the alert.' ONLY "
234
+ " when every one of those fields matches your goal.\n"
235
+ " A field your goal reports as 'not specified' is one you have NO expectation about: "
236
+ " accept whatever the agent chose for it and never ask for it to be removed.\n"
170
237
  "4. Proactively include your email recipient in your first reply. "
171
238
  " Do not wait for the agent to ask — state it alongside the metric and condition answers.\n"
172
- + trigger_line
173
- + "Reply concisely and directly."
239
+ + filters_rule
240
+ + f"6. Proactively state how often you want to be alerted in your first reply: {trigger_request}. "
241
+ " Repeat it if the agent proposes a different cadence.\n"
242
+ "Reply concisely and directly."
174
243
  )
175
244
 
176
245
  messages: list = [{"role": "system", "content": system_prompt}]
@@ -257,6 +326,29 @@ def _case_insensitive_get(d: dict, *keys: str) -> Any:
257
326
  return None
258
327
 
259
328
 
329
+ _NO_FILTER_MARKERS = ("none", "all time")
330
+
331
+
332
+ def _normalize_expected_filters(expected: dict) -> list | str | None:
333
+ """
334
+ * ``Filters`` list -> that list (exact expectation)
335
+ * "None (All time)" in either -> ``[]`` (stated: no filters; extras fail)
336
+ * anything else / absent -> ``None`` (unstated; filters not asserted)
337
+ """
338
+ filters = _case_insensitive_get(expected, "filters")
339
+ if isinstance(filters, list):
340
+ return filters
341
+ time_window = _case_insensitive_get(expected, "time window/filters", "time_window")
342
+ for candidate in (filters, time_window):
343
+ if isinstance(candidate, str) and any(kw in candidate.lower() for kw in _NO_FILTER_MARKERS):
344
+ return []
345
+ # Prose that is not a no-filter marker ("Product Category = X") describes a filter without
346
+ # encoding it, so it cannot be compared: returning it made `_check_filters` fall through to
347
+ # `_deep_subset(str, list)`, which can never match. `None` is what the contract above
348
+ # promises — the sim-user derives such filters from the original request instead.
349
+ return None
350
+
351
+
260
352
  def _normalize_expected_output(expected: dict) -> CatalogMetricAlert:
261
353
  """Parse expected_output dict into CatalogMetricAlert, accepting display-format or internal-format keys."""
262
354
  operator = _case_insensitive_get(expected, "operator") or "GREATER_THAN"
@@ -280,9 +372,7 @@ def _normalize_expected_output(expected: dict) -> CatalogMetricAlert:
280
372
  else:
281
373
  recipients = list(raw_recip)
282
374
 
283
- filters = _case_insensitive_get(expected, "filters")
284
- if isinstance(filters, str) and any(kw in filters for kw in ("None", "All time")):
285
- filters = None
375
+ filters = _normalize_expected_filters(expected)
286
376
 
287
377
  return CatalogMetricAlert(
288
378
  operator=operator,
@@ -377,7 +467,9 @@ def run_agentic_alert_skill(
377
467
  # Stop before generating a follow-up for the last iteration
378
468
  if _iteration >= max_iterations - 1:
379
469
  break
380
- follow_up = generate_simulated_alert_response(response_text, expected, conversation_history)
470
+ follow_up = generate_simulated_alert_response(
471
+ response_text, expected, conversation_history, question=question
472
+ )
381
473
  # Record this exchange so the next call has full history
382
474
  conversation_history.append({"role": "assistant", "content": response_text})
383
475
  conversation_history.append({"role": "user", "content": follow_up})