gooddata-eval 1.74.1.dev1__tar.gz → 1.74.1.dev2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (101) hide show
  1. gooddata_eval-1.74.1.dev2/AGENTS.md +114 -0
  2. gooddata_eval-1.74.1.dev2/CLAUDE.md +1 -0
  3. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/PKG-INFO +2 -2
  4. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/pyproject.toml +4 -4
  5. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/agentic/_catalog.py +41 -0
  6. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/agentic/alert_skill.py +175 -23
  7. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/agentic/conversation.py +119 -39
  8. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/agentic/general_question.py +14 -2
  9. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/agentic/guardrail.py +2 -1
  10. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/agentic/kda_skill.py +34 -2
  11. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/agentic/metric_skill.py +16 -78
  12. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/agentic/search_tool.py +14 -1
  13. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/agentic/visualization.py +11 -15
  14. gooddata_eval-1.74.1.dev2/src/gooddata_eval/core/chat/render.py +47 -0
  15. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/chat/sse_client.py +34 -2
  16. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/evaluators/__init__.py +13 -5
  17. gooddata_eval-1.74.1.dev2/src/gooddata_eval/core/evaluators/_maql.py +103 -0
  18. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/evaluators/_text_utils.py +3 -4
  19. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/evaluators/alert_skill.py +11 -2
  20. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/evaluators/general_question.py +4 -0
  21. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/evaluators/metric_skill.py +16 -4
  22. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/models.py +42 -0
  23. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/tests/test_agentic_alert_skill.py +285 -0
  24. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/tests/test_agentic_conversation.py +277 -1
  25. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/tests/test_agentic_general_question.py +41 -0
  26. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/tests/test_agentic_kda_skill.py +79 -0
  27. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/tests/test_agentic_metric_skill.py +3 -47
  28. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/tests/test_agentic_search_tool.py +42 -0
  29. gooddata_eval-1.74.1.dev2/tests/test_chat_render.py +118 -0
  30. gooddata_eval-1.74.1.dev2/tests/test_maql_normalize.py +106 -0
  31. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/tests/test_runner.py +1 -0
  32. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/tests/test_text_evaluators.py +28 -0
  33. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/.gitignore +0 -0
  34. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/LICENSE.txt +0 -0
  35. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/Makefile +0 -0
  36. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/README.md +0 -0
  37. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/__init__.py +0 -0
  38. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/_version.py +0 -0
  39. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/cli/__init__.py +0 -0
  40. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/cli/agentic_runner.py +0 -0
  41. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/cli/main.py +0 -0
  42. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/__init__.py +0 -0
  43. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/_output.py +0 -0
  44. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/agentic/__init__.py +0 -0
  45. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/agentic/_langfuse.py +0 -0
  46. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/agentic/_trace_linker.py +0 -0
  47. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/chat/__init__.py +0 -0
  48. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/config.py +0 -0
  49. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/connection.py +0 -0
  50. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/dataset/__init__.py +0 -0
  51. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/dataset/langfuse_source.py +0 -0
  52. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/dataset/local.py +0 -0
  53. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/evaluators/_deep_subset.py +0 -0
  54. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/evaluators/_llm_judge.py +0 -0
  55. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/evaluators/base.py +0 -0
  56. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/evaluators/guardrail.py +0 -0
  57. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/evaluators/search_tool.py +0 -0
  58. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/evaluators/summary.py +0 -0
  59. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/evaluators/visualization.py +0 -0
  60. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/langfuse/__init__.py +0 -0
  61. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/langfuse/sink.py +0 -0
  62. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/reporting/__init__.py +0 -0
  63. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/reporting/console.py +0 -0
  64. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/reporting/json_report.py +0 -0
  65. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/runner.py +0 -0
  66. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/scoring.py +0 -0
  67. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/summary/__init__.py +0 -0
  68. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/summary/http_client.py +0 -0
  69. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/timing.py +0 -0
  70. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/workspace.py +0 -0
  71. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/tests/__init__.py +0 -0
  72. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/tests/conftest.py +0 -0
  73. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/tests/fixtures/sample_dataset/metric_skill_create.json +0 -0
  74. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/tests/fixtures/sample_dataset/visualization_revenue.json +0 -0
  75. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/tests/fixtures/sse_visualization_stream.txt +0 -0
  76. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/tests/test_agentic_guardrail.py +0 -0
  77. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/tests/test_agentic_langfuse_trace.py +0 -0
  78. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/tests/test_agentic_run_context.py +0 -0
  79. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/tests/test_agentic_runner.py +0 -0
  80. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/tests/test_agentic_visualization.py +0 -0
  81. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/tests/test_alert_skill_evaluator.py +0 -0
  82. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/tests/test_cli.py +0 -0
  83. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/tests/test_connection.py +0 -0
  84. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/tests/test_deep_subset.py +0 -0
  85. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/tests/test_langfuse_sink.py +0 -0
  86. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/tests/test_langfuse_source.py +0 -0
  87. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/tests/test_llm_judge.py +0 -0
  88. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/tests/test_local_loader.py +0 -0
  89. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/tests/test_metric_skill_evaluator.py +0 -0
  90. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/tests/test_models.py +0 -0
  91. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/tests/test_reporting.py +0 -0
  92. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/tests/test_scoring.py +0 -0
  93. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/tests/test_search_tool_evaluator.py +0 -0
  94. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/tests/test_sse_client.py +0 -0
  95. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/tests/test_summary_client.py +0 -0
  96. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/tests/test_summary_evaluator.py +0 -0
  97. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/tests/test_timing.py +0 -0
  98. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/tests/test_trace_linker.py +0 -0
  99. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/tests/test_visualization_evaluator.py +0 -0
  100. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/tests/test_workspace.py +0 -0
  101. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/tox.ini +0 -0
@@ -0,0 +1,114 @@
1
+ # gooddata-eval
2
+
3
+ `gd-eval` — a CLI and library that drives the GoodData AI agent (a separate service, in
4
+ `gdc-nas`) through a dataset of natural-language questions and scores what comes back,
5
+ including side-by-side comparison across models. Each dataset item is a JSON envelope
6
+ loaded from a local folder or pulled from a Langfuse dataset. Results are aggregated into
7
+ pass@K / pass^K reports and optionally pushed to Langfuse as scored traces tied to a
8
+ dataset run. The newest and most actively developed package in the repo.
9
+
10
+ ## Owns
11
+
12
+ - The `gd-eval` CLI (`gd-eval run`, `gd-eval models`)
13
+ - Dataset loading and the evaluation run loop
14
+ - Per-capability evaluators and their scoring
15
+ - Result reporting, and pushing runs, scores and trace links to Langfuse
16
+
17
+ ## Does NOT Own
18
+
19
+ - The agent under evaluation — that lives in `gdc-nas` (gen-ai)
20
+ - Platform access → `gooddata-sdk`
21
+
22
+ ## Architecture
23
+
24
+ | Path | Role |
25
+ |---|---|
26
+ | `cli/` | argument parsing, and `agentic_runner` — the agentic dispatch and concurrency phases |
27
+ | `core/agentic/` | multi-turn agentic evaluation per capability, **plus** all Langfuse trace polling and linking (`_langfuse.py`, `_trace_linker.py`) |
28
+ | `core/chat/` | SSE client for the agent's streaming chat endpoint |
29
+ | `core/summary/` | HTTP client for the dedicated dashboard-summary endpoint — a single-shot chat backend, not reporting |
30
+ | `core/dataset/` | dataset format and loading |
31
+ | `core/evaluators/` | single-shot evaluators and their registry |
32
+ | `core/langfuse/` | `sink.py` only — pushes single-turn scores and dataset-run items |
33
+ | `core/reporting/` | console and JSON output rendering |
34
+ | `core/scoring.py`, `core/runner.py` | scoring and orchestration |
35
+ | `core/models.py` | `DatasetItem`, `ChatResult`, `ItemReport` and friends |
36
+
37
+ **Depends on**: `gooddata-sdk`, `httpx`, `pydantic`, `orjson`, `rich`. The LLM-judge
38
+ evaluator is an optional extra (`llm-judge`, pulling `openai>=1.45,<2.0`); every `openai`
39
+ import site is guarded or deferred so the base install stays usable without it — keep it
40
+ that way.
41
+
42
+ ### Two evaluation paths that share almost nothing
43
+
44
+ - **Single-shot** kinds send one chat turn and are scored by an `Evaluator` (a Protocol:
45
+ a `test_kind` attribute plus `evaluate(item, chat_result) -> ItemEvaluation`) looked up
46
+ from a registry in `core/evaluators/__init__.py`.
47
+ - **Agentic** kinds (`agentic_*`, `vis_agentic`) drive a full multi-turn conversation over
48
+ the SSE endpoint and are dispatched by an explicit `if`/`elif` chain in
49
+ `cli/agentic_runner.py`.
50
+
51
+ Many capabilities exist in **both** forms — visualization, metric skill, alert skill,
52
+ search, general question and guardrail each have a single-turn and a multi-turn
53
+ implementation, sometimes under different `test_kind` strings (`search_tool` vs
54
+ `agentic_search`). These are parallel implementations, not layers.
55
+
56
+ ### Dataset items
57
+
58
+ `DatasetItem` is the envelope: `id`, `dataset_name`, `test_kind`, `question`, and
59
+ `expected_output: Any`. `expected_output` is deliberately untyped — each evaluator parses
60
+ its own shape. `test_kind` on the item is what labels the result, not the evaluator class,
61
+ which is why `knowledge_question` can reuse `GeneralQuestionEvaluator` verbatim.
62
+ `dashboard_summary` items additionally need `summary_input`.
63
+
64
+ ## Gotchas
65
+
66
+ **Adding an evaluator is a registry change, not a naming convention.** Single-shot kinds go
67
+ into `_EAGER_EVALUATORS`, or `_LAZY_EVALUATOR_MODULES` plus `_LAZY_EVALUATOR_CLASSES`, in
68
+ `core/evaluators/__init__.py`. Agentic kinds need the string added to `AGENTIC_TEST_KINDS`
69
+ and a new branch in `_dispatch_agentic`. Test file naming follows the capability, but
70
+ naming a test file correctly registers nothing.
71
+
72
+ **Parallel-safety is a reviewed allowlist, and getting it wrong corrupts results.**
73
+ `WORKSPACE_MUTATING_TEST_KINDS` is computed as `AGENTIC_TEST_KINDS - PARALLEL_SAFE_TEST_KINDS`,
74
+ so a newly added kind defaults to workspace-mutating and runs serially in its own phase.
75
+ That default is correct: agent tool calls create real server-side objects (metrics, alerts).
76
+ Adding a kind to `PARALLEL_SAFE_TEST_KINDS` is a deliberate assertion that it is read-only,
77
+ which nothing in the package can prove for you.
78
+
79
+ **The SSE client's retry predicate is load-bearing.** `core/chat/sse_client.py` retries
80
+ 429/502/503/504 and `httpx.RemoteProtocolError` (a mid-stream disconnect) with exponential
81
+ backoff, and treats a `METADATA_SYNC_IN_PROGRESS` payload as transient. The
82
+ `RemoteProtocolError` case was added after it was confirmed live to contaminate a small
83
+ percentage of visualization runs with a hard fail and no retry. Narrowing that predicate
84
+ reintroduces the problem.
85
+
86
+ **Langfuse trace linking is deliberately off the item critical path.** Polling for trace
87
+ ingestion has no pass/fail signal and inflates or misattributes per-item latency, so
88
+ `BackgroundTraceLinker` defers it and is drained before the report renders
89
+ (`run_trace_link_inline` is the synchronous alternative). Do not "fix" a slow item by
90
+ making trace scoring synchronous again.
91
+
92
+ **Scoring weights do not sum to 1.** `quality_score` is the fraction of boolean-valued keys
93
+ in `best_detail` that are true, falling back to `pass_at_k` when there are none (text
94
+ evaluators). `value_score` is `0.6 * quality + 0.2 * speed` — the 0.8 total is what the
95
+ code does; treat it as intentional unless you have checked with the owner.
96
+
97
+ ### Fixture shapes
98
+
99
+ Group-by / attribute expectations in the alert-skill fixtures are written in AAC shape,
100
+ while the tool arguments the agent emits are AFM-shaped. Never deep-compare those two
101
+ directly — convert, or compare field by field. This applies specifically to the
102
+ attribute/group-by fields: `Filters` in the same fixtures is AFM-shaped on both sides and
103
+ is correctly deep-compared as-is. The attribute comparison itself lands with the
104
+ alert group-by work currently on `jt/gdai-2175-eval-alert-attributes`, so on `master` this
105
+ is guidance for the incoming code rather than a description of what is already there.
106
+
107
+ ## Testing
108
+
109
+ Plain pytest under `tests/`, no cassettes — the agent is stubbed with
110
+ `unittest.mock`. Tests are named per capability (`test_agentic_*.py`), which is the
111
+ convention to follow when adding one.
112
+
113
+ `ty` is configured here with `allowed-unresolved-imports` for `openai.**` and
114
+ `gooddata_api_client.**`; do not widen that list to paper over a real typing problem.
@@ -0,0 +1 @@
1
+ @AGENTS.md
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: gooddata-eval
3
- Version: 1.74.1.dev1
3
+ Version: 1.74.1.dev2
4
4
  Summary: Evaluate the GoodData AI agent against your own questions and models.
5
5
  Project-URL: Source, https://github.com/gooddata/gooddata-python-sdk
6
6
  Author-email: GoodData <support@gooddata.com>
@@ -17,7 +17,7 @@ Classifier: Topic :: Scientific/Engineering
17
17
  Classifier: Topic :: Software Development
18
18
  Classifier: Typing :: Typed
19
19
  Requires-Python: >=3.10
20
- Requires-Dist: gooddata-sdk~=1.74.1.dev1
20
+ Requires-Dist: gooddata-sdk~=1.74.1.dev2
21
21
  Requires-Dist: httpx<1.0,>=0.27
22
22
  Requires-Dist: orjson<4.0.0,>=3.9.15
23
23
  Requires-Dist: pydantic<3.0,>=2.6
@@ -1,7 +1,7 @@
1
1
  # (C) 2026 GoodData Corporation
2
2
  [project]
3
3
  name = "gooddata-eval"
4
- version = "1.74.1.dev1"
4
+ version = "1.74.1.dev2"
5
5
  description = "Evaluate the GoodData AI agent against your own questions and models."
6
6
  readme = "README.md"
7
7
  license = "MIT"
@@ -11,7 +11,7 @@ authors = [
11
11
  keywords = ["gooddata", "ai", "evaluation", "llm", "analytics", "cli"]
12
12
  requires-python = ">=3.10"
13
13
  dependencies = [
14
- "gooddata-sdk~=1.74.1.dev1",
14
+ "gooddata-sdk~=1.74.1.dev2",
15
15
  "httpx>=0.27,<1.0",
16
16
  "orjson>=3.9.15,<4.0.0",
17
17
  "pydantic>=2.6,<3.0",
@@ -43,8 +43,8 @@ dev = [
43
43
  "pytest>=8.3.5",
44
44
  ]
45
45
  test = [
46
- "pytest~=8.3.4",
47
- "pytest-cov~=6.0.0",
46
+ "pytest~=9.1.1",
47
+ "pytest-cov~=7.1.0",
48
48
  "pytest-json-report==1.5.0",
49
49
  "pytest-mock>=3.14.0",
50
50
  ]
@@ -2,6 +2,41 @@
2
2
  from __future__ import annotations
3
3
 
4
4
  from dataclasses import dataclass, field
5
+ from enum import Enum
6
+
7
+
8
+ class AnomalyDetectionGranularity(str, Enum):
9
+ """Detection intervals an anomaly alert accepts.
10
+
11
+ Mirrors gen-ai's enum of the same name; `StrEnum` is unavailable on the 3.10 floor, so
12
+ the `str` mixin carries the comparison against the raw tool argument.
13
+ """
14
+
15
+ HOUR = "HOUR"
16
+ DAY = "DAY"
17
+ WEEK = "WEEK"
18
+ MONTH = "MONTH"
19
+ QUARTER = "QUARTER"
20
+ YEAR = "YEAR"
21
+
22
+ @classmethod
23
+ def parse(cls, value: object) -> AnomalyDetectionGranularity | None:
24
+ """Coerce a fixture value, or None when the fixture states none.
25
+
26
+ Raises ValueError on an unknown interval: fixtures are hand-written, and a typo has
27
+ to fail before the run spends an API call rather than score the item against an
28
+ interval the product cannot produce.
29
+ """
30
+ if value is None:
31
+ return None
32
+ candidate = str(value).strip().upper()
33
+ if not candidate:
34
+ return None
35
+ try:
36
+ return cls(candidate)
37
+ except ValueError:
38
+ expected = ", ".join(member.value for member in cls)
39
+ raise ValueError(f"Invalid granularity {value!r}; expected one of {expected}.") from None
5
40
 
6
41
 
7
42
  @dataclass
@@ -28,6 +63,10 @@ class CatalogMetricAlert:
28
63
  """List of recipient email addresses."""
29
64
  filters: list | str | None = None
30
65
  """Attribute filters applied to the alert condition."""
66
+ attributes: list | None = None
67
+ """Expected group-by attributes; ``None`` means the fixture states no expectation."""
68
+ granularity: AnomalyDetectionGranularity | None = None
69
+ """Detection interval for an ANOMALY alert (DAY/WEEK/MONTH/...). Not a date filter."""
31
70
 
32
71
  @classmethod
33
72
  def from_dict(cls, d: dict) -> CatalogMetricAlert:
@@ -46,4 +85,6 @@ class CatalogMetricAlert:
46
85
  metric_id=d.get("metric_id"),
47
86
  recipients=recipients,
48
87
  filters=d.get("filters"),
88
+ attributes=d.get("attributes"),
89
+ granularity=AnomalyDetectionGranularity.parse(d.get("granularity")),
49
90
  )
@@ -11,7 +11,7 @@ from typing import Any
11
11
 
12
12
  from gooddata_sdk import GoodDataSdk
13
13
 
14
- from gooddata_eval.core.agentic._catalog import CatalogMetricAlert
14
+ from gooddata_eval.core.agentic._catalog import AnomalyDetectionGranularity, CatalogMetricAlert
15
15
  from gooddata_eval.core.agentic._trace_linker import (
16
16
  RunIdentity,
17
17
  RunTraceContext,
@@ -21,6 +21,7 @@ from gooddata_eval.core.agentic._trace_linker import (
21
21
  submit_trace_scoring,
22
22
  utc_now,
23
23
  )
24
+ from gooddata_eval.core.chat.render import render_answer_text
24
25
  from gooddata_eval.core.chat.sse_client import ChatClient
25
26
  from gooddata_eval.core.config import ReasoningEffort
26
27
  from gooddata_eval.core.models import (
@@ -29,6 +30,7 @@ from gooddata_eval.core.models import (
29
30
  ReasoningStepEvent,
30
31
  ToolCallEvent,
31
32
  build_latency_breakdown,
33
+ shift_and_index_events,
32
34
  )
33
35
 
34
36
  try:
@@ -126,6 +128,94 @@ def _check_filters(expected: CatalogMetricAlert, actual_args: dict) -> bool:
126
128
  return _deep_subset(exp_filters, act_filters)
127
129
 
128
130
 
131
+ def _attribute_label_ids(items: list, *, side: str) -> list[str]:
132
+ """Canonicalise group-by entries to bare label ids, whatever spelling they arrive in.
133
+
134
+ The two sides of the comparison speak different vocabularies for the same grouping.
135
+ Fixtures author the AAC tool-input form, ``{"using": "label/x"}``; ``create_metric_alert``
136
+ receives the resolved AFM form, ``{"localIdentifier": "a0", "label": {"identifier":
137
+ {"id": "x", "type": "label"}}}``, forwarded verbatim from ``prepare_metric_alert_proposal``.
138
+ Identity is therefore the only thing they can be compared on.
139
+
140
+ A shape not listed here, or a URI prefix other than ``label/``, raises: ``label/x`` and
141
+ ``attribute/x`` are different objects, and an unknown spelling must fail loudly rather
142
+ than quietly compare unequal.
143
+ """
144
+ if not isinstance(items, list):
145
+ raise ValueError(f"Unrecognised {side} group-by attributes, expected a list: {items!r}")
146
+ ids: list[str] = []
147
+ for item in items:
148
+ raw: object = None
149
+ if isinstance(item, str):
150
+ raw = item
151
+ elif isinstance(item, dict):
152
+ label = item.get("label")
153
+ identifier = item.get("identifier")
154
+ if isinstance(item.get("using"), str):
155
+ raw = item["using"]
156
+ elif isinstance(label, dict) and isinstance(label.get("identifier"), dict):
157
+ raw = label["identifier"].get("id")
158
+ elif isinstance(identifier, dict):
159
+ raw = identifier.get("id")
160
+ if not isinstance(raw, str) or not raw:
161
+ raise ValueError(f"Unrecognised {side} group-by attribute entry: {item!r}")
162
+ prefix, slash, rest = raw.partition("/")
163
+ if not slash:
164
+ ids.append(raw)
165
+ elif prefix == "label" and rest:
166
+ ids.append(rest)
167
+ else:
168
+ raise ValueError(f"Unrecognised {side} group-by attribute reference: {raw!r}")
169
+ return ids
170
+
171
+
172
+ def _check_attributes(expected: CatalogMetricAlert, actual_args: dict) -> bool:
173
+ """Compare group-by identity only.
174
+
175
+ Per-entry properties — ``showAllValues``, the converter-assigned ``localIdentifier`` —
176
+ are deliberately not asserted, and the comparison is a multiset so entry order does not
177
+ matter.
178
+ """
179
+ exp_attributes = expected.attributes
180
+ if exp_attributes is None:
181
+ return True
182
+ act_attributes = actual_args.get("attributes")
183
+ if act_attributes is None:
184
+ # Arguments are raw `json.loads` output, where an unset nullable argument arrives as
185
+ # null rather than absent. Both spellings of "no grouping" have to land on [], which
186
+ # is why this is not `actual_args.get("attributes", [])`.
187
+ act_attributes = []
188
+ elif not isinstance(act_attributes, list):
189
+ # An argument that is not a list of groupings is the agent answering wrongly, so it
190
+ # scores False. Raising instead would make the runner record an ERROR, and errored
191
+ # items are excluded from the failure count — a malformed answer must not rank above
192
+ # a merely wrong one. An unreadable *entry* still raises, in `_attribute_label_ids`:
193
+ # entries are typed at the tool boundary, so the plausible cause there is the wire
194
+ # format moving, which has to be unmissable.
195
+ return False
196
+ if not exp_attributes:
197
+ return not act_attributes
198
+ exp_ids = sorted(_attribute_label_ids(exp_attributes, side="expected"))
199
+ act_ids = sorted(_attribute_label_ids(act_attributes, side="actual"))
200
+ return exp_ids == act_ids
201
+
202
+
203
+ def _check_granularity(expected: CatalogMetricAlert, actual_args: dict) -> bool:
204
+ """Compare the ANOMALY detection interval when the fixture states one.
205
+
206
+ ``None`` means unasserted, mirroring ``attributes``: only the ANOMALY items carry a
207
+ ``Granularity``, and every other item must stay unaffected. The expectation is already
208
+ canonical by the time it lands here; the tool argument is a raw string, so only that
209
+ side needs folding.
210
+ """
211
+ if expected.granularity is None:
212
+ return True
213
+ actual = actual_args.get("granularity")
214
+ if not actual:
215
+ return False
216
+ return str(actual).strip().upper() == expected.granularity.value
217
+
218
+
129
219
  def _check_metric(expected: CatalogMetricAlert, actual_args: dict) -> bool:
130
220
  if not expected.metric_id:
131
221
  return True
@@ -257,20 +347,38 @@ def generate_simulated_alert_response(
257
347
  )
258
348
  elif filters == []:
259
349
  filters_rule = (
260
- "5. Your alert must have NO filters and NO date/time window — it evaluates over all time. "
261
- "If the agent asks which time period each check should cover, or offers a choice such as "
262
- "'last Day / Week / Month', do NOT pick one: reply that you want no date filter at all, "
263
- "all time. Never invent a period, a granularity or an 'evaluate each run on a X basis' "
264
- "instruction the goal did not ask for.\n"
350
+ "5. Your alert must have NO filters and NO date/time window on the metric — it evaluates "
351
+ "over all time. If the agent asks which time period each check should cover, or offers a "
352
+ "choice such as 'last Day / Week / Month', do NOT pick one: reply that you want no date "
353
+ "filter at all, all time.\n"
265
354
  )
266
355
  else:
267
356
  filters_rule = (
268
357
  "5. Ask only for the filters your original request implies — do not invent an evaluation "
269
- "period, granularity or date window that was not requested. If the agent offers a choice "
358
+ "period or date window that was not requested. If the agent offers a choice "
270
359
  "such as 'last Day / Week / Month' that your request never mentioned, say you do not want "
271
360
  "a date window.\n"
272
361
  )
273
362
 
363
+ if operator == "ANOMALY":
364
+ # The fallback keeps the conversation alive when the fixture names no interval -- an
365
+ # anomaly alert cannot be created without one. It is deliberately NOT mirrored into
366
+ # `expected.granularity`: `_check_granularity` asserts only what the fixture stated,
367
+ # and scoring an item against an interval it never asked for is the defect this rule
368
+ # exists to undo.
369
+ granularity = (expected.granularity or AnomalyDetectionGranularity.DAY).value
370
+ anomaly_rule = (
371
+ "7. This is an ANOMALY alert. Anomaly detection REQUIRES a time granularity, and that "
372
+ f"granularity is NOT a date filter. State it in your first reply and repeat it whenever "
373
+ f"asked: use {granularity} granularity. Rule 5 constrains filters on the metric only — it "
374
+ "never applies to this detection interval, so never refuse to give one.\n"
375
+ )
376
+ else:
377
+ anomaly_rule = (
378
+ "7. Do not invent an evaluation period, a granularity or an 'evaluate each run on a X "
379
+ "basis' instruction your goal never asked for.\n"
380
+ )
381
+
274
382
  original_request = f'Your original request to the agent was: "{question}"\n' if question else ""
275
383
 
276
384
  system_prompt = (
@@ -294,8 +402,7 @@ def generate_simulated_alert_response(
294
402
  " Do not wait for the agent to ask — state it alongside the metric and condition answers.\n"
295
403
  + filters_rule
296
404
  + f"6. Proactively state how often you want to be alerted in your first reply: {trigger_request}. "
297
- " Repeat it if the agent proposes a different cadence.\n"
298
- "Reply concisely and directly."
405
+ " Repeat it if the agent proposes a different cadence.\n" + anomaly_rule + "Reply concisely and directly."
299
406
  )
300
407
 
301
408
  messages: list = [{"role": "system", "content": system_prompt}]
@@ -334,6 +441,8 @@ class AlertEvaluation:
334
441
  filters_correct: bool
335
442
  metric_correct: bool
336
443
  recipients_correct: bool
444
+ attributes_correct: bool = True
445
+ granularity_correct: bool = True
337
446
 
338
447
  @property
339
448
  def strict_pass(self) -> bool:
@@ -346,6 +455,8 @@ class AlertEvaluation:
346
455
  self.filters_correct,
347
456
  self.metric_correct,
348
457
  self.recipients_correct,
458
+ self.attributes_correct,
459
+ self.granularity_correct,
349
460
  ]
350
461
  )
351
462
 
@@ -409,6 +520,35 @@ def _normalize_expected_filters(expected: dict) -> list | str | None:
409
520
  return None
410
521
 
411
522
 
523
+ _NO_GROUPING_MARKERS = ("none", "no grouping")
524
+
525
+
526
+ def _normalize_expected_attributes(expected: dict) -> list | None:
527
+ """
528
+ * ``Attributes`` list -> that list (exact expectation)
529
+ * "None" / "no grouping" -> ``[]`` (stated: no group-by; extras fail)
530
+ * absent, or other prose -> ``None`` (unstated; grouping not asserted)
531
+
532
+ A date narrows an alert as a group-by as well as a filter, and a group-by makes it fire
533
+ per period value instead of on the latest one — so ``[]`` has to be expressible separately
534
+ from "absent", exactly as it is for ``filters``.
535
+
536
+ The simulated user is told nothing about groupings, so a non-empty expectation requires the
537
+ item's own question to request that grouping; ``[]`` needs no such support, because the
538
+ simulated user does not invent a grouping and the check verifies it did not.
539
+ """
540
+ attributes = _case_insensitive_get(expected, "attributes")
541
+ if isinstance(attributes, list):
542
+ # Validated here so a malformed fixture fails before the run spends an API call.
543
+ _attribute_label_ids(attributes, side="expected")
544
+ return attributes
545
+ if attributes is None:
546
+ return None
547
+ if isinstance(attributes, str):
548
+ return [] if any(kw in attributes.lower() for kw in _NO_GROUPING_MARKERS) else None
549
+ raise ValueError(f"Attributes expectation must be a list or a display string, got {type(attributes).__name__}")
550
+
551
+
412
552
  def _normalize_expected_output(expected: dict) -> CatalogMetricAlert:
413
553
  """Parse expected_output dict into CatalogMetricAlert, accepting display-format or internal-format keys."""
414
554
  operator = _case_insensitive_get(expected, "operator") or "GREATER_THAN"
@@ -433,6 +573,11 @@ def _normalize_expected_output(expected: dict) -> CatalogMetricAlert:
433
573
  recipients = list(raw_recip)
434
574
 
435
575
  filters = _normalize_expected_filters(expected)
576
+ attributes = _normalize_expected_attributes(expected)
577
+
578
+ granularity = AnomalyDetectionGranularity.parse(
579
+ _case_insensitive_get(expected, "granularity", "detection granularity")
580
+ )
436
581
 
437
582
  return CatalogMetricAlert(
438
583
  operator=operator,
@@ -443,6 +588,8 @@ def _normalize_expected_output(expected: dict) -> CatalogMetricAlert:
443
588
  metric_id=metric_id,
444
589
  recipients=recipients,
445
590
  filters=filters,
591
+ attributes=attributes,
592
+ granularity=granularity,
446
593
  )
447
594
 
448
595
 
@@ -526,21 +673,14 @@ def run_agentic_alert_skill(
526
673
  chat_result = client.send_message(conv_id, current_question)
527
674
  reasoning_steps.extend(chat_result.reasoning_steps or [])
528
675
  response_id = chat_result.response_id or response_id
529
- for tc in chat_result.tool_call_events or []:
530
- if tc.call_ts is not None:
531
- tc.call_ts += turn_offset
532
- if tc.result_ts is not None:
533
- tc.result_ts += turn_offset
534
- if tc.index is not None:
535
- tc.index += tool_index_offset
536
- for rs in chat_result.reasoning_step_events or []:
537
- rs.ts += turn_offset
538
- rs.index += reasoning_index_offset
676
+ turn_offset, tool_index_offset, reasoning_index_offset = shift_and_index_events(
677
+ chat_result,
678
+ turn_offset=turn_offset,
679
+ tool_index_offset=tool_index_offset,
680
+ reasoning_index_offset=reasoning_index_offset,
681
+ )
539
682
  all_tool_call_events.extend(chat_result.tool_call_events or [])
540
683
  all_reasoning_step_events.extend(chat_result.reasoning_step_events or [])
541
- tool_index_offset += len(chat_result.tool_call_events or [])
542
- reasoning_index_offset += len(chat_result.reasoning_step_events or [])
543
- turn_offset += chat_result.turn_wall_clock_sec or 0.0
544
684
  alert_id, actual_args, tool_called = _extract_alert_call(chat_result.tool_call_events or [])
545
685
  if tool_called:
546
686
  alert_id_to_delete = alert_id
@@ -548,6 +688,8 @@ def run_agentic_alert_skill(
548
688
  response_text = (chat_result.text_response or "").strip()
549
689
  if not response_text and chat_result.alert_proposals:
550
690
  response_text = render_alert_proposal(chat_result.alert_proposals[-1])
691
+ if not response_text:
692
+ response_text = render_answer_text(chat_result)
551
693
  # Stop if agent gave a completely empty response (stuck)
552
694
  if not response_text and not chat_result.tool_call_events:
553
695
  break
@@ -570,6 +712,8 @@ def run_agentic_alert_skill(
570
712
  filters_correct=tool_called and _check_filters(expected, actual_args),
571
713
  metric_correct=tool_called and _check_metric(expected, actual_args),
572
714
  recipients_correct=tool_called and _check_recipients(expected, actual_args, sdk=sdk),
715
+ attributes_correct=tool_called and _check_attributes(expected, actual_args),
716
+ granularity_correct=tool_called and _check_granularity(expected, actual_args),
573
717
  )
574
718
  return AlertRunResult(
575
719
  conversation_id=conv_id,
@@ -615,6 +759,8 @@ def run_agentic_alert_skill(
615
759
  r.eval.filters_correct,
616
760
  r.eval.metric_correct,
617
761
  r.eval.recipients_correct,
762
+ r.eval.attributes_correct,
763
+ r.eval.granularity_correct,
618
764
  ]
619
765
  ),
620
766
  )
@@ -689,6 +835,8 @@ def evaluate_agentic_alert_skill(
689
835
  "filters_correct": ev.filters_correct,
690
836
  "metric_correct": ev.metric_correct,
691
837
  "recipients_correct": ev.recipients_correct,
838
+ "attributes_correct": ev.attributes_correct,
839
+ "granularity_correct": ev.granularity_correct,
692
840
  }
693
841
  with ctx.observe(pt, run_idx) as tid:
694
842
  for score_name, value in strict_checks.items():
@@ -735,6 +883,8 @@ def evaluate_agentic_alert_skill(
735
883
  "filters_correct": ev.filters_correct,
736
884
  "metric_correct": ev.metric_correct,
737
885
  "recipients_correct": ev.recipients_correct,
886
+ "attributes_correct": ev.attributes_correct,
887
+ "granularity_correct": ev.granularity_correct,
738
888
  "actual_alert_arguments": best.actual_alert_arguments,
739
889
  "latency_breakdown": build_latency_breakdown(best.tool_call_events, best.reasoning_step_events),
740
890
  }
@@ -745,7 +895,9 @@ def evaluate_agentic_alert_skill(
745
895
  f"alert_created={ev.alert_created}, operator_correct={ev.operator_correct}, "
746
896
  f"threshold_correct={ev.threshold_correct}, trigger_correct={ev.trigger_correct}, "
747
897
  f"filters_correct={ev.filters_correct}, metric_correct={ev.metric_correct}, "
748
- f"recipients_correct={ev.recipients_correct}. "
898
+ f"recipients_correct={ev.recipients_correct}, "
899
+ f"attributes_correct={ev.attributes_correct}, "
900
+ f"granularity_correct={ev.granularity_correct}. "
749
901
  f"Actual args: {best.actual_alert_arguments}"
750
902
  )
751
903
  exc.reasoning_steps = best.reasoning_steps