gooddata-eval 1.74.1.dev1__tar.gz → 1.74.1.dev2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- gooddata_eval-1.74.1.dev2/AGENTS.md +114 -0
- gooddata_eval-1.74.1.dev2/CLAUDE.md +1 -0
- {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/PKG-INFO +2 -2
- {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/pyproject.toml +4 -4
- {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/agentic/_catalog.py +41 -0
- {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/agentic/alert_skill.py +175 -23
- {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/agentic/conversation.py +119 -39
- {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/agentic/general_question.py +14 -2
- {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/agentic/guardrail.py +2 -1
- {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/agentic/kda_skill.py +34 -2
- {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/agentic/metric_skill.py +16 -78
- {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/agentic/search_tool.py +14 -1
- {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/agentic/visualization.py +11 -15
- gooddata_eval-1.74.1.dev2/src/gooddata_eval/core/chat/render.py +47 -0
- {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/chat/sse_client.py +34 -2
- {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/evaluators/__init__.py +13 -5
- gooddata_eval-1.74.1.dev2/src/gooddata_eval/core/evaluators/_maql.py +103 -0
- {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/evaluators/_text_utils.py +3 -4
- {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/evaluators/alert_skill.py +11 -2
- {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/evaluators/general_question.py +4 -0
- {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/evaluators/metric_skill.py +16 -4
- {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/models.py +42 -0
- {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/tests/test_agentic_alert_skill.py +285 -0
- {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/tests/test_agentic_conversation.py +277 -1
- {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/tests/test_agentic_general_question.py +41 -0
- {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/tests/test_agentic_kda_skill.py +79 -0
- {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/tests/test_agentic_metric_skill.py +3 -47
- {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/tests/test_agentic_search_tool.py +42 -0
- gooddata_eval-1.74.1.dev2/tests/test_chat_render.py +118 -0
- gooddata_eval-1.74.1.dev2/tests/test_maql_normalize.py +106 -0
- {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/tests/test_runner.py +1 -0
- {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/tests/test_text_evaluators.py +28 -0
- {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/.gitignore +0 -0
- {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/LICENSE.txt +0 -0
- {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/Makefile +0 -0
- {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/README.md +0 -0
- {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/__init__.py +0 -0
- {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/_version.py +0 -0
- {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/cli/__init__.py +0 -0
- {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/cli/agentic_runner.py +0 -0
- {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/cli/main.py +0 -0
- {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/__init__.py +0 -0
- {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/_output.py +0 -0
- {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/agentic/__init__.py +0 -0
- {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/agentic/_langfuse.py +0 -0
- {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/agentic/_trace_linker.py +0 -0
- {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/chat/__init__.py +0 -0
- {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/config.py +0 -0
- {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/connection.py +0 -0
- {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/dataset/__init__.py +0 -0
- {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/dataset/langfuse_source.py +0 -0
- {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/dataset/local.py +0 -0
- {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/evaluators/_deep_subset.py +0 -0
- {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/evaluators/_llm_judge.py +0 -0
- {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/evaluators/base.py +0 -0
- {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/evaluators/guardrail.py +0 -0
- {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/evaluators/search_tool.py +0 -0
- {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/evaluators/summary.py +0 -0
- {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/evaluators/visualization.py +0 -0
- {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/langfuse/__init__.py +0 -0
- {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/langfuse/sink.py +0 -0
- {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/reporting/__init__.py +0 -0
- {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/reporting/console.py +0 -0
- {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/reporting/json_report.py +0 -0
- {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/runner.py +0 -0
- {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/scoring.py +0 -0
- {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/summary/__init__.py +0 -0
- {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/summary/http_client.py +0 -0
- {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/timing.py +0 -0
- {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/workspace.py +0 -0
- {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/tests/__init__.py +0 -0
- {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/tests/conftest.py +0 -0
- {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/tests/fixtures/sample_dataset/metric_skill_create.json +0 -0
- {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/tests/fixtures/sample_dataset/visualization_revenue.json +0 -0
- {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/tests/fixtures/sse_visualization_stream.txt +0 -0
- {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/tests/test_agentic_guardrail.py +0 -0
- {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/tests/test_agentic_langfuse_trace.py +0 -0
- {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/tests/test_agentic_run_context.py +0 -0
- {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/tests/test_agentic_runner.py +0 -0
- {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/tests/test_agentic_visualization.py +0 -0
- {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/tests/test_alert_skill_evaluator.py +0 -0
- {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/tests/test_cli.py +0 -0
- {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/tests/test_connection.py +0 -0
- {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/tests/test_deep_subset.py +0 -0
- {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/tests/test_langfuse_sink.py +0 -0
- {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/tests/test_langfuse_source.py +0 -0
- {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/tests/test_llm_judge.py +0 -0
- {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/tests/test_local_loader.py +0 -0
- {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/tests/test_metric_skill_evaluator.py +0 -0
- {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/tests/test_models.py +0 -0
- {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/tests/test_reporting.py +0 -0
- {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/tests/test_scoring.py +0 -0
- {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/tests/test_search_tool_evaluator.py +0 -0
- {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/tests/test_sse_client.py +0 -0
- {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/tests/test_summary_client.py +0 -0
- {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/tests/test_summary_evaluator.py +0 -0
- {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/tests/test_timing.py +0 -0
- {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/tests/test_trace_linker.py +0 -0
- {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/tests/test_visualization_evaluator.py +0 -0
- {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/tests/test_workspace.py +0 -0
- {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/tox.ini +0 -0
|
@@ -0,0 +1,114 @@
|
|
|
1
|
+
# gooddata-eval
|
|
2
|
+
|
|
3
|
+
`gd-eval` — a CLI and library that drives the GoodData AI agent (a separate service, in
|
|
4
|
+
`gdc-nas`) through a dataset of natural-language questions and scores what comes back,
|
|
5
|
+
including side-by-side comparison across models. Each dataset item is a JSON envelope
|
|
6
|
+
loaded from a local folder or pulled from a Langfuse dataset. Results are aggregated into
|
|
7
|
+
pass@K / pass^K reports and optionally pushed to Langfuse as scored traces tied to a
|
|
8
|
+
dataset run. The newest and most actively developed package in the repo.
|
|
9
|
+
|
|
10
|
+
## Owns
|
|
11
|
+
|
|
12
|
+
- The `gd-eval` CLI (`gd-eval run`, `gd-eval models`)
|
|
13
|
+
- Dataset loading and the evaluation run loop
|
|
14
|
+
- Per-capability evaluators and their scoring
|
|
15
|
+
- Result reporting, and pushing runs, scores and trace links to Langfuse
|
|
16
|
+
|
|
17
|
+
## Does NOT Own
|
|
18
|
+
|
|
19
|
+
- The agent under evaluation — that lives in `gdc-nas` (gen-ai)
|
|
20
|
+
- Platform access → `gooddata-sdk`
|
|
21
|
+
|
|
22
|
+
## Architecture
|
|
23
|
+
|
|
24
|
+
| Path | Role |
|
|
25
|
+
|---|---|
|
|
26
|
+
| `cli/` | argument parsing, and `agentic_runner` — the agentic dispatch and concurrency phases |
|
|
27
|
+
| `core/agentic/` | multi-turn agentic evaluation per capability, **plus** all Langfuse trace polling and linking (`_langfuse.py`, `_trace_linker.py`) |
|
|
28
|
+
| `core/chat/` | SSE client for the agent's streaming chat endpoint |
|
|
29
|
+
| `core/summary/` | HTTP client for the dedicated dashboard-summary endpoint — a single-shot chat backend, not reporting |
|
|
30
|
+
| `core/dataset/` | dataset format and loading |
|
|
31
|
+
| `core/evaluators/` | single-shot evaluators and their registry |
|
|
32
|
+
| `core/langfuse/` | `sink.py` only — pushes single-turn scores and dataset-run items |
|
|
33
|
+
| `core/reporting/` | console and JSON output rendering |
|
|
34
|
+
| `core/scoring.py`, `core/runner.py` | scoring and orchestration |
|
|
35
|
+
| `core/models.py` | `DatasetItem`, `ChatResult`, `ItemReport` and friends |
|
|
36
|
+
|
|
37
|
+
**Depends on**: `gooddata-sdk`, `httpx`, `pydantic`, `orjson`, `rich`. The LLM-judge
|
|
38
|
+
evaluator is an optional extra (`llm-judge`, pulling `openai>=1.45,<2.0`); every `openai`
|
|
39
|
+
import site is guarded or deferred so the base install stays usable without it — keep it
|
|
40
|
+
that way.
|
|
41
|
+
|
|
42
|
+
### Two evaluation paths that share almost nothing
|
|
43
|
+
|
|
44
|
+
- **Single-shot** kinds send one chat turn and are scored by an `Evaluator` (a Protocol:
|
|
45
|
+
a `test_kind` attribute plus `evaluate(item, chat_result) -> ItemEvaluation`) looked up
|
|
46
|
+
from a registry in `core/evaluators/__init__.py`.
|
|
47
|
+
- **Agentic** kinds (`agentic_*`, `vis_agentic`) drive a full multi-turn conversation over
|
|
48
|
+
the SSE endpoint and are dispatched by an explicit `if`/`elif` chain in
|
|
49
|
+
`cli/agentic_runner.py`.
|
|
50
|
+
|
|
51
|
+
Many capabilities exist in **both** forms — visualization, metric skill, alert skill,
|
|
52
|
+
search, general question and guardrail each have a single-turn and a multi-turn
|
|
53
|
+
implementation, sometimes under different `test_kind` strings (`search_tool` vs
|
|
54
|
+
`agentic_search`). These are parallel implementations, not layers.
|
|
55
|
+
|
|
56
|
+
### Dataset items
|
|
57
|
+
|
|
58
|
+
`DatasetItem` is the envelope: `id`, `dataset_name`, `test_kind`, `question`, and
|
|
59
|
+
`expected_output: Any`. `expected_output` is deliberately untyped — each evaluator parses
|
|
60
|
+
its own shape. `test_kind` on the item is what labels the result, not the evaluator class,
|
|
61
|
+
which is why `knowledge_question` can reuse `GeneralQuestionEvaluator` verbatim.
|
|
62
|
+
`dashboard_summary` items additionally need `summary_input`.
|
|
63
|
+
|
|
64
|
+
## Gotchas
|
|
65
|
+
|
|
66
|
+
**Adding an evaluator is a registry change, not a naming convention.** Single-shot kinds go
|
|
67
|
+
into `_EAGER_EVALUATORS`, or `_LAZY_EVALUATOR_MODULES` plus `_LAZY_EVALUATOR_CLASSES`, in
|
|
68
|
+
`core/evaluators/__init__.py`. Agentic kinds need the string added to `AGENTIC_TEST_KINDS`
|
|
69
|
+
and a new branch in `_dispatch_agentic`. Test file naming follows the capability, but
|
|
70
|
+
naming a test file correctly registers nothing.
|
|
71
|
+
|
|
72
|
+
**Parallel-safety is a reviewed allowlist, and getting it wrong corrupts results.**
|
|
73
|
+
`WORKSPACE_MUTATING_TEST_KINDS` is computed as `AGENTIC_TEST_KINDS - PARALLEL_SAFE_TEST_KINDS`,
|
|
74
|
+
so a newly added kind defaults to workspace-mutating and runs serially in its own phase.
|
|
75
|
+
That default is correct: agent tool calls create real server-side objects (metrics, alerts).
|
|
76
|
+
Adding a kind to `PARALLEL_SAFE_TEST_KINDS` is a deliberate assertion that it is read-only,
|
|
77
|
+
which nothing in the package can prove for you.
|
|
78
|
+
|
|
79
|
+
**The SSE client's retry predicate is load-bearing.** `core/chat/sse_client.py` retries
|
|
80
|
+
429/502/503/504 and `httpx.RemoteProtocolError` (a mid-stream disconnect) with exponential
|
|
81
|
+
backoff, and treats a `METADATA_SYNC_IN_PROGRESS` payload as transient. The
|
|
82
|
+
`RemoteProtocolError` case was added after it was confirmed live to contaminate a small
|
|
83
|
+
percentage of visualization runs with a hard fail and no retry. Narrowing that predicate
|
|
84
|
+
reintroduces the problem.
|
|
85
|
+
|
|
86
|
+
**Langfuse trace linking is deliberately off the item critical path.** Polling for trace
|
|
87
|
+
ingestion has no pass/fail signal and inflates or misattributes per-item latency, so
|
|
88
|
+
`BackgroundTraceLinker` defers it and is drained before the report renders
|
|
89
|
+
(`run_trace_link_inline` is the synchronous alternative). Do not "fix" a slow item by
|
|
90
|
+
making trace scoring synchronous again.
|
|
91
|
+
|
|
92
|
+
**Scoring weights do not sum to 1.** `quality_score` is the fraction of boolean-valued keys
|
|
93
|
+
in `best_detail` that are true, falling back to `pass_at_k` when there are none (text
|
|
94
|
+
evaluators). `value_score` is `0.6 * quality + 0.2 * speed` — the 0.8 total is what the
|
|
95
|
+
code does; treat it as intentional unless you have checked with the owner.
|
|
96
|
+
|
|
97
|
+
### Fixture shapes
|
|
98
|
+
|
|
99
|
+
Group-by / attribute expectations in the alert-skill fixtures are written in AAC shape,
|
|
100
|
+
while the tool arguments the agent emits are AFM-shaped. Never deep-compare those two
|
|
101
|
+
directly — convert, or compare field by field. This applies specifically to the
|
|
102
|
+
attribute/group-by fields: `Filters` in the same fixtures is AFM-shaped on both sides and
|
|
103
|
+
is correctly deep-compared as-is. The attribute comparison itself lands with the
|
|
104
|
+
alert group-by work currently on `jt/gdai-2175-eval-alert-attributes`, so on `master` this
|
|
105
|
+
is guidance for the incoming code rather than a description of what is already there.
|
|
106
|
+
|
|
107
|
+
## Testing
|
|
108
|
+
|
|
109
|
+
Plain pytest under `tests/`, no cassettes — the agent is stubbed with
|
|
110
|
+
`unittest.mock`. Tests are named per capability (`test_agentic_*.py`), which is the
|
|
111
|
+
convention to follow when adding one.
|
|
112
|
+
|
|
113
|
+
`ty` is configured here with `allowed-unresolved-imports` for `openai.**` and
|
|
114
|
+
`gooddata_api_client.**`; do not widen that list to paper over a real typing problem.
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
@AGENTS.md
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: gooddata-eval
|
|
3
|
-
Version: 1.74.1.
|
|
3
|
+
Version: 1.74.1.dev2
|
|
4
4
|
Summary: Evaluate the GoodData AI agent against your own questions and models.
|
|
5
5
|
Project-URL: Source, https://github.com/gooddata/gooddata-python-sdk
|
|
6
6
|
Author-email: GoodData <support@gooddata.com>
|
|
@@ -17,7 +17,7 @@ Classifier: Topic :: Scientific/Engineering
|
|
|
17
17
|
Classifier: Topic :: Software Development
|
|
18
18
|
Classifier: Typing :: Typed
|
|
19
19
|
Requires-Python: >=3.10
|
|
20
|
-
Requires-Dist: gooddata-sdk~=1.74.1.
|
|
20
|
+
Requires-Dist: gooddata-sdk~=1.74.1.dev2
|
|
21
21
|
Requires-Dist: httpx<1.0,>=0.27
|
|
22
22
|
Requires-Dist: orjson<4.0.0,>=3.9.15
|
|
23
23
|
Requires-Dist: pydantic<3.0,>=2.6
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
# (C) 2026 GoodData Corporation
|
|
2
2
|
[project]
|
|
3
3
|
name = "gooddata-eval"
|
|
4
|
-
version = "1.74.1.
|
|
4
|
+
version = "1.74.1.dev2"
|
|
5
5
|
description = "Evaluate the GoodData AI agent against your own questions and models."
|
|
6
6
|
readme = "README.md"
|
|
7
7
|
license = "MIT"
|
|
@@ -11,7 +11,7 @@ authors = [
|
|
|
11
11
|
keywords = ["gooddata", "ai", "evaluation", "llm", "analytics", "cli"]
|
|
12
12
|
requires-python = ">=3.10"
|
|
13
13
|
dependencies = [
|
|
14
|
-
"gooddata-sdk~=1.74.1.
|
|
14
|
+
"gooddata-sdk~=1.74.1.dev2",
|
|
15
15
|
"httpx>=0.27,<1.0",
|
|
16
16
|
"orjson>=3.9.15,<4.0.0",
|
|
17
17
|
"pydantic>=2.6,<3.0",
|
|
@@ -43,8 +43,8 @@ dev = [
|
|
|
43
43
|
"pytest>=8.3.5",
|
|
44
44
|
]
|
|
45
45
|
test = [
|
|
46
|
-
"pytest~=
|
|
47
|
-
"pytest-cov~=
|
|
46
|
+
"pytest~=9.1.1",
|
|
47
|
+
"pytest-cov~=7.1.0",
|
|
48
48
|
"pytest-json-report==1.5.0",
|
|
49
49
|
"pytest-mock>=3.14.0",
|
|
50
50
|
]
|
{gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/agentic/_catalog.py
RENAMED
|
@@ -2,6 +2,41 @@
|
|
|
2
2
|
from __future__ import annotations
|
|
3
3
|
|
|
4
4
|
from dataclasses import dataclass, field
|
|
5
|
+
from enum import Enum
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
class AnomalyDetectionGranularity(str, Enum):
|
|
9
|
+
"""Detection intervals an anomaly alert accepts.
|
|
10
|
+
|
|
11
|
+
Mirrors gen-ai's enum of the same name; `StrEnum` is unavailable on the 3.10 floor, so
|
|
12
|
+
the `str` mixin carries the comparison against the raw tool argument.
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
HOUR = "HOUR"
|
|
16
|
+
DAY = "DAY"
|
|
17
|
+
WEEK = "WEEK"
|
|
18
|
+
MONTH = "MONTH"
|
|
19
|
+
QUARTER = "QUARTER"
|
|
20
|
+
YEAR = "YEAR"
|
|
21
|
+
|
|
22
|
+
@classmethod
|
|
23
|
+
def parse(cls, value: object) -> AnomalyDetectionGranularity | None:
|
|
24
|
+
"""Coerce a fixture value, or None when the fixture states none.
|
|
25
|
+
|
|
26
|
+
Raises ValueError on an unknown interval: fixtures are hand-written, and a typo has
|
|
27
|
+
to fail before the run spends an API call rather than score the item against an
|
|
28
|
+
interval the product cannot produce.
|
|
29
|
+
"""
|
|
30
|
+
if value is None:
|
|
31
|
+
return None
|
|
32
|
+
candidate = str(value).strip().upper()
|
|
33
|
+
if not candidate:
|
|
34
|
+
return None
|
|
35
|
+
try:
|
|
36
|
+
return cls(candidate)
|
|
37
|
+
except ValueError:
|
|
38
|
+
expected = ", ".join(member.value for member in cls)
|
|
39
|
+
raise ValueError(f"Invalid granularity {value!r}; expected one of {expected}.") from None
|
|
5
40
|
|
|
6
41
|
|
|
7
42
|
@dataclass
|
|
@@ -28,6 +63,10 @@ class CatalogMetricAlert:
|
|
|
28
63
|
"""List of recipient email addresses."""
|
|
29
64
|
filters: list | str | None = None
|
|
30
65
|
"""Attribute filters applied to the alert condition."""
|
|
66
|
+
attributes: list | None = None
|
|
67
|
+
"""Expected group-by attributes; ``None`` means the fixture states no expectation."""
|
|
68
|
+
granularity: AnomalyDetectionGranularity | None = None
|
|
69
|
+
"""Detection interval for an ANOMALY alert (DAY/WEEK/MONTH/...). Not a date filter."""
|
|
31
70
|
|
|
32
71
|
@classmethod
|
|
33
72
|
def from_dict(cls, d: dict) -> CatalogMetricAlert:
|
|
@@ -46,4 +85,6 @@ class CatalogMetricAlert:
|
|
|
46
85
|
metric_id=d.get("metric_id"),
|
|
47
86
|
recipients=recipients,
|
|
48
87
|
filters=d.get("filters"),
|
|
88
|
+
attributes=d.get("attributes"),
|
|
89
|
+
granularity=AnomalyDetectionGranularity.parse(d.get("granularity")),
|
|
49
90
|
)
|
|
@@ -11,7 +11,7 @@ from typing import Any
|
|
|
11
11
|
|
|
12
12
|
from gooddata_sdk import GoodDataSdk
|
|
13
13
|
|
|
14
|
-
from gooddata_eval.core.agentic._catalog import CatalogMetricAlert
|
|
14
|
+
from gooddata_eval.core.agentic._catalog import AnomalyDetectionGranularity, CatalogMetricAlert
|
|
15
15
|
from gooddata_eval.core.agentic._trace_linker import (
|
|
16
16
|
RunIdentity,
|
|
17
17
|
RunTraceContext,
|
|
@@ -21,6 +21,7 @@ from gooddata_eval.core.agentic._trace_linker import (
|
|
|
21
21
|
submit_trace_scoring,
|
|
22
22
|
utc_now,
|
|
23
23
|
)
|
|
24
|
+
from gooddata_eval.core.chat.render import render_answer_text
|
|
24
25
|
from gooddata_eval.core.chat.sse_client import ChatClient
|
|
25
26
|
from gooddata_eval.core.config import ReasoningEffort
|
|
26
27
|
from gooddata_eval.core.models import (
|
|
@@ -29,6 +30,7 @@ from gooddata_eval.core.models import (
|
|
|
29
30
|
ReasoningStepEvent,
|
|
30
31
|
ToolCallEvent,
|
|
31
32
|
build_latency_breakdown,
|
|
33
|
+
shift_and_index_events,
|
|
32
34
|
)
|
|
33
35
|
|
|
34
36
|
try:
|
|
@@ -126,6 +128,94 @@ def _check_filters(expected: CatalogMetricAlert, actual_args: dict) -> bool:
|
|
|
126
128
|
return _deep_subset(exp_filters, act_filters)
|
|
127
129
|
|
|
128
130
|
|
|
131
|
+
def _attribute_label_ids(items: list, *, side: str) -> list[str]:
|
|
132
|
+
"""Canonicalise group-by entries to bare label ids, whatever spelling they arrive in.
|
|
133
|
+
|
|
134
|
+
The two sides of the comparison speak different vocabularies for the same grouping.
|
|
135
|
+
Fixtures author the AAC tool-input form, ``{"using": "label/x"}``; ``create_metric_alert``
|
|
136
|
+
receives the resolved AFM form, ``{"localIdentifier": "a0", "label": {"identifier":
|
|
137
|
+
{"id": "x", "type": "label"}}}``, forwarded verbatim from ``prepare_metric_alert_proposal``.
|
|
138
|
+
Identity is therefore the only thing they can be compared on.
|
|
139
|
+
|
|
140
|
+
A shape not listed here, or a URI prefix other than ``label/``, raises: ``label/x`` and
|
|
141
|
+
``attribute/x`` are different objects, and an unknown spelling must fail loudly rather
|
|
142
|
+
than quietly compare unequal.
|
|
143
|
+
"""
|
|
144
|
+
if not isinstance(items, list):
|
|
145
|
+
raise ValueError(f"Unrecognised {side} group-by attributes, expected a list: {items!r}")
|
|
146
|
+
ids: list[str] = []
|
|
147
|
+
for item in items:
|
|
148
|
+
raw: object = None
|
|
149
|
+
if isinstance(item, str):
|
|
150
|
+
raw = item
|
|
151
|
+
elif isinstance(item, dict):
|
|
152
|
+
label = item.get("label")
|
|
153
|
+
identifier = item.get("identifier")
|
|
154
|
+
if isinstance(item.get("using"), str):
|
|
155
|
+
raw = item["using"]
|
|
156
|
+
elif isinstance(label, dict) and isinstance(label.get("identifier"), dict):
|
|
157
|
+
raw = label["identifier"].get("id")
|
|
158
|
+
elif isinstance(identifier, dict):
|
|
159
|
+
raw = identifier.get("id")
|
|
160
|
+
if not isinstance(raw, str) or not raw:
|
|
161
|
+
raise ValueError(f"Unrecognised {side} group-by attribute entry: {item!r}")
|
|
162
|
+
prefix, slash, rest = raw.partition("/")
|
|
163
|
+
if not slash:
|
|
164
|
+
ids.append(raw)
|
|
165
|
+
elif prefix == "label" and rest:
|
|
166
|
+
ids.append(rest)
|
|
167
|
+
else:
|
|
168
|
+
raise ValueError(f"Unrecognised {side} group-by attribute reference: {raw!r}")
|
|
169
|
+
return ids
|
|
170
|
+
|
|
171
|
+
|
|
172
|
+
def _check_attributes(expected: CatalogMetricAlert, actual_args: dict) -> bool:
|
|
173
|
+
"""Compare group-by identity only.
|
|
174
|
+
|
|
175
|
+
Per-entry properties — ``showAllValues``, the converter-assigned ``localIdentifier`` —
|
|
176
|
+
are deliberately not asserted, and the comparison is a multiset so entry order does not
|
|
177
|
+
matter.
|
|
178
|
+
"""
|
|
179
|
+
exp_attributes = expected.attributes
|
|
180
|
+
if exp_attributes is None:
|
|
181
|
+
return True
|
|
182
|
+
act_attributes = actual_args.get("attributes")
|
|
183
|
+
if act_attributes is None:
|
|
184
|
+
# Arguments are raw `json.loads` output, where an unset nullable argument arrives as
|
|
185
|
+
# null rather than absent. Both spellings of "no grouping" have to land on [], which
|
|
186
|
+
# is why this is not `actual_args.get("attributes", [])`.
|
|
187
|
+
act_attributes = []
|
|
188
|
+
elif not isinstance(act_attributes, list):
|
|
189
|
+
# An argument that is not a list of groupings is the agent answering wrongly, so it
|
|
190
|
+
# scores False. Raising instead would make the runner record an ERROR, and errored
|
|
191
|
+
# items are excluded from the failure count — a malformed answer must not rank above
|
|
192
|
+
# a merely wrong one. An unreadable *entry* still raises, in `_attribute_label_ids`:
|
|
193
|
+
# entries are typed at the tool boundary, so the plausible cause there is the wire
|
|
194
|
+
# format moving, which has to be unmissable.
|
|
195
|
+
return False
|
|
196
|
+
if not exp_attributes:
|
|
197
|
+
return not act_attributes
|
|
198
|
+
exp_ids = sorted(_attribute_label_ids(exp_attributes, side="expected"))
|
|
199
|
+
act_ids = sorted(_attribute_label_ids(act_attributes, side="actual"))
|
|
200
|
+
return exp_ids == act_ids
|
|
201
|
+
|
|
202
|
+
|
|
203
|
+
def _check_granularity(expected: CatalogMetricAlert, actual_args: dict) -> bool:
|
|
204
|
+
"""Compare the ANOMALY detection interval when the fixture states one.
|
|
205
|
+
|
|
206
|
+
``None`` means unasserted, mirroring ``attributes``: only the ANOMALY items carry a
|
|
207
|
+
``Granularity``, and every other item must stay unaffected. The expectation is already
|
|
208
|
+
canonical by the time it lands here; the tool argument is a raw string, so only that
|
|
209
|
+
side needs folding.
|
|
210
|
+
"""
|
|
211
|
+
if expected.granularity is None:
|
|
212
|
+
return True
|
|
213
|
+
actual = actual_args.get("granularity")
|
|
214
|
+
if not actual:
|
|
215
|
+
return False
|
|
216
|
+
return str(actual).strip().upper() == expected.granularity.value
|
|
217
|
+
|
|
218
|
+
|
|
129
219
|
def _check_metric(expected: CatalogMetricAlert, actual_args: dict) -> bool:
|
|
130
220
|
if not expected.metric_id:
|
|
131
221
|
return True
|
|
@@ -257,20 +347,38 @@ def generate_simulated_alert_response(
|
|
|
257
347
|
)
|
|
258
348
|
elif filters == []:
|
|
259
349
|
filters_rule = (
|
|
260
|
-
"5. Your alert must have NO filters and NO date/time window — it evaluates
|
|
261
|
-
"If the agent asks which time period each check should cover, or offers a
|
|
262
|
-
"'last Day / Week / Month', do NOT pick one: reply that you want no date
|
|
263
|
-
"
|
|
264
|
-
"instruction the goal did not ask for.\n"
|
|
350
|
+
"5. Your alert must have NO filters and NO date/time window on the metric — it evaluates "
|
|
351
|
+
"over all time. If the agent asks which time period each check should cover, or offers a "
|
|
352
|
+
"choice such as 'last Day / Week / Month', do NOT pick one: reply that you want no date "
|
|
353
|
+
"filter at all, all time.\n"
|
|
265
354
|
)
|
|
266
355
|
else:
|
|
267
356
|
filters_rule = (
|
|
268
357
|
"5. Ask only for the filters your original request implies — do not invent an evaluation "
|
|
269
|
-
"period
|
|
358
|
+
"period or date window that was not requested. If the agent offers a choice "
|
|
270
359
|
"such as 'last Day / Week / Month' that your request never mentioned, say you do not want "
|
|
271
360
|
"a date window.\n"
|
|
272
361
|
)
|
|
273
362
|
|
|
363
|
+
if operator == "ANOMALY":
|
|
364
|
+
# The fallback keeps the conversation alive when the fixture names no interval -- an
|
|
365
|
+
# anomaly alert cannot be created without one. It is deliberately NOT mirrored into
|
|
366
|
+
# `expected.granularity`: `_check_granularity` asserts only what the fixture stated,
|
|
367
|
+
# and scoring an item against an interval it never asked for is the defect this rule
|
|
368
|
+
# exists to undo.
|
|
369
|
+
granularity = (expected.granularity or AnomalyDetectionGranularity.DAY).value
|
|
370
|
+
anomaly_rule = (
|
|
371
|
+
"7. This is an ANOMALY alert. Anomaly detection REQUIRES a time granularity, and that "
|
|
372
|
+
f"granularity is NOT a date filter. State it in your first reply and repeat it whenever "
|
|
373
|
+
f"asked: use {granularity} granularity. Rule 5 constrains filters on the metric only — it "
|
|
374
|
+
"never applies to this detection interval, so never refuse to give one.\n"
|
|
375
|
+
)
|
|
376
|
+
else:
|
|
377
|
+
anomaly_rule = (
|
|
378
|
+
"7. Do not invent an evaluation period, a granularity or an 'evaluate each run on a X "
|
|
379
|
+
"basis' instruction your goal never asked for.\n"
|
|
380
|
+
)
|
|
381
|
+
|
|
274
382
|
original_request = f'Your original request to the agent was: "{question}"\n' if question else ""
|
|
275
383
|
|
|
276
384
|
system_prompt = (
|
|
@@ -294,8 +402,7 @@ def generate_simulated_alert_response(
|
|
|
294
402
|
" Do not wait for the agent to ask — state it alongside the metric and condition answers.\n"
|
|
295
403
|
+ filters_rule
|
|
296
404
|
+ f"6. Proactively state how often you want to be alerted in your first reply: {trigger_request}. "
|
|
297
|
-
" Repeat it if the agent proposes a different cadence.\n"
|
|
298
|
-
"Reply concisely and directly."
|
|
405
|
+
" Repeat it if the agent proposes a different cadence.\n" + anomaly_rule + "Reply concisely and directly."
|
|
299
406
|
)
|
|
300
407
|
|
|
301
408
|
messages: list = [{"role": "system", "content": system_prompt}]
|
|
@@ -334,6 +441,8 @@ class AlertEvaluation:
|
|
|
334
441
|
filters_correct: bool
|
|
335
442
|
metric_correct: bool
|
|
336
443
|
recipients_correct: bool
|
|
444
|
+
attributes_correct: bool = True
|
|
445
|
+
granularity_correct: bool = True
|
|
337
446
|
|
|
338
447
|
@property
|
|
339
448
|
def strict_pass(self) -> bool:
|
|
@@ -346,6 +455,8 @@ class AlertEvaluation:
|
|
|
346
455
|
self.filters_correct,
|
|
347
456
|
self.metric_correct,
|
|
348
457
|
self.recipients_correct,
|
|
458
|
+
self.attributes_correct,
|
|
459
|
+
self.granularity_correct,
|
|
349
460
|
]
|
|
350
461
|
)
|
|
351
462
|
|
|
@@ -409,6 +520,35 @@ def _normalize_expected_filters(expected: dict) -> list | str | None:
|
|
|
409
520
|
return None
|
|
410
521
|
|
|
411
522
|
|
|
523
|
+
_NO_GROUPING_MARKERS = ("none", "no grouping")
|
|
524
|
+
|
|
525
|
+
|
|
526
|
+
def _normalize_expected_attributes(expected: dict) -> list | None:
|
|
527
|
+
"""
|
|
528
|
+
* ``Attributes`` list -> that list (exact expectation)
|
|
529
|
+
* "None" / "no grouping" -> ``[]`` (stated: no group-by; extras fail)
|
|
530
|
+
* absent, or other prose -> ``None`` (unstated; grouping not asserted)
|
|
531
|
+
|
|
532
|
+
A date narrows an alert as a group-by as well as a filter, and a group-by makes it fire
|
|
533
|
+
per period value instead of on the latest one — so ``[]`` has to be expressible separately
|
|
534
|
+
from "absent", exactly as it is for ``filters``.
|
|
535
|
+
|
|
536
|
+
The simulated user is told nothing about groupings, so a non-empty expectation requires the
|
|
537
|
+
item's own question to request that grouping; ``[]`` needs no such support, because the
|
|
538
|
+
simulated user does not invent a grouping and the check verifies it did not.
|
|
539
|
+
"""
|
|
540
|
+
attributes = _case_insensitive_get(expected, "attributes")
|
|
541
|
+
if isinstance(attributes, list):
|
|
542
|
+
# Validated here so a malformed fixture fails before the run spends an API call.
|
|
543
|
+
_attribute_label_ids(attributes, side="expected")
|
|
544
|
+
return attributes
|
|
545
|
+
if attributes is None:
|
|
546
|
+
return None
|
|
547
|
+
if isinstance(attributes, str):
|
|
548
|
+
return [] if any(kw in attributes.lower() for kw in _NO_GROUPING_MARKERS) else None
|
|
549
|
+
raise ValueError(f"Attributes expectation must be a list or a display string, got {type(attributes).__name__}")
|
|
550
|
+
|
|
551
|
+
|
|
412
552
|
def _normalize_expected_output(expected: dict) -> CatalogMetricAlert:
|
|
413
553
|
"""Parse expected_output dict into CatalogMetricAlert, accepting display-format or internal-format keys."""
|
|
414
554
|
operator = _case_insensitive_get(expected, "operator") or "GREATER_THAN"
|
|
@@ -433,6 +573,11 @@ def _normalize_expected_output(expected: dict) -> CatalogMetricAlert:
|
|
|
433
573
|
recipients = list(raw_recip)
|
|
434
574
|
|
|
435
575
|
filters = _normalize_expected_filters(expected)
|
|
576
|
+
attributes = _normalize_expected_attributes(expected)
|
|
577
|
+
|
|
578
|
+
granularity = AnomalyDetectionGranularity.parse(
|
|
579
|
+
_case_insensitive_get(expected, "granularity", "detection granularity")
|
|
580
|
+
)
|
|
436
581
|
|
|
437
582
|
return CatalogMetricAlert(
|
|
438
583
|
operator=operator,
|
|
@@ -443,6 +588,8 @@ def _normalize_expected_output(expected: dict) -> CatalogMetricAlert:
|
|
|
443
588
|
metric_id=metric_id,
|
|
444
589
|
recipients=recipients,
|
|
445
590
|
filters=filters,
|
|
591
|
+
attributes=attributes,
|
|
592
|
+
granularity=granularity,
|
|
446
593
|
)
|
|
447
594
|
|
|
448
595
|
|
|
@@ -526,21 +673,14 @@ def run_agentic_alert_skill(
|
|
|
526
673
|
chat_result = client.send_message(conv_id, current_question)
|
|
527
674
|
reasoning_steps.extend(chat_result.reasoning_steps or [])
|
|
528
675
|
response_id = chat_result.response_id or response_id
|
|
529
|
-
|
|
530
|
-
|
|
531
|
-
|
|
532
|
-
|
|
533
|
-
|
|
534
|
-
|
|
535
|
-
tc.index += tool_index_offset
|
|
536
|
-
for rs in chat_result.reasoning_step_events or []:
|
|
537
|
-
rs.ts += turn_offset
|
|
538
|
-
rs.index += reasoning_index_offset
|
|
676
|
+
turn_offset, tool_index_offset, reasoning_index_offset = shift_and_index_events(
|
|
677
|
+
chat_result,
|
|
678
|
+
turn_offset=turn_offset,
|
|
679
|
+
tool_index_offset=tool_index_offset,
|
|
680
|
+
reasoning_index_offset=reasoning_index_offset,
|
|
681
|
+
)
|
|
539
682
|
all_tool_call_events.extend(chat_result.tool_call_events or [])
|
|
540
683
|
all_reasoning_step_events.extend(chat_result.reasoning_step_events or [])
|
|
541
|
-
tool_index_offset += len(chat_result.tool_call_events or [])
|
|
542
|
-
reasoning_index_offset += len(chat_result.reasoning_step_events or [])
|
|
543
|
-
turn_offset += chat_result.turn_wall_clock_sec or 0.0
|
|
544
684
|
alert_id, actual_args, tool_called = _extract_alert_call(chat_result.tool_call_events or [])
|
|
545
685
|
if tool_called:
|
|
546
686
|
alert_id_to_delete = alert_id
|
|
@@ -548,6 +688,8 @@ def run_agentic_alert_skill(
|
|
|
548
688
|
response_text = (chat_result.text_response or "").strip()
|
|
549
689
|
if not response_text and chat_result.alert_proposals:
|
|
550
690
|
response_text = render_alert_proposal(chat_result.alert_proposals[-1])
|
|
691
|
+
if not response_text:
|
|
692
|
+
response_text = render_answer_text(chat_result)
|
|
551
693
|
# Stop if agent gave a completely empty response (stuck)
|
|
552
694
|
if not response_text and not chat_result.tool_call_events:
|
|
553
695
|
break
|
|
@@ -570,6 +712,8 @@ def run_agentic_alert_skill(
|
|
|
570
712
|
filters_correct=tool_called and _check_filters(expected, actual_args),
|
|
571
713
|
metric_correct=tool_called and _check_metric(expected, actual_args),
|
|
572
714
|
recipients_correct=tool_called and _check_recipients(expected, actual_args, sdk=sdk),
|
|
715
|
+
attributes_correct=tool_called and _check_attributes(expected, actual_args),
|
|
716
|
+
granularity_correct=tool_called and _check_granularity(expected, actual_args),
|
|
573
717
|
)
|
|
574
718
|
return AlertRunResult(
|
|
575
719
|
conversation_id=conv_id,
|
|
@@ -615,6 +759,8 @@ def run_agentic_alert_skill(
|
|
|
615
759
|
r.eval.filters_correct,
|
|
616
760
|
r.eval.metric_correct,
|
|
617
761
|
r.eval.recipients_correct,
|
|
762
|
+
r.eval.attributes_correct,
|
|
763
|
+
r.eval.granularity_correct,
|
|
618
764
|
]
|
|
619
765
|
),
|
|
620
766
|
)
|
|
@@ -689,6 +835,8 @@ def evaluate_agentic_alert_skill(
|
|
|
689
835
|
"filters_correct": ev.filters_correct,
|
|
690
836
|
"metric_correct": ev.metric_correct,
|
|
691
837
|
"recipients_correct": ev.recipients_correct,
|
|
838
|
+
"attributes_correct": ev.attributes_correct,
|
|
839
|
+
"granularity_correct": ev.granularity_correct,
|
|
692
840
|
}
|
|
693
841
|
with ctx.observe(pt, run_idx) as tid:
|
|
694
842
|
for score_name, value in strict_checks.items():
|
|
@@ -735,6 +883,8 @@ def evaluate_agentic_alert_skill(
|
|
|
735
883
|
"filters_correct": ev.filters_correct,
|
|
736
884
|
"metric_correct": ev.metric_correct,
|
|
737
885
|
"recipients_correct": ev.recipients_correct,
|
|
886
|
+
"attributes_correct": ev.attributes_correct,
|
|
887
|
+
"granularity_correct": ev.granularity_correct,
|
|
738
888
|
"actual_alert_arguments": best.actual_alert_arguments,
|
|
739
889
|
"latency_breakdown": build_latency_breakdown(best.tool_call_events, best.reasoning_step_events),
|
|
740
890
|
}
|
|
@@ -745,7 +895,9 @@ def evaluate_agentic_alert_skill(
|
|
|
745
895
|
f"alert_created={ev.alert_created}, operator_correct={ev.operator_correct}, "
|
|
746
896
|
f"threshold_correct={ev.threshold_correct}, trigger_correct={ev.trigger_correct}, "
|
|
747
897
|
f"filters_correct={ev.filters_correct}, metric_correct={ev.metric_correct}, "
|
|
748
|
-
f"recipients_correct={ev.recipients_correct}
|
|
898
|
+
f"recipients_correct={ev.recipients_correct}, "
|
|
899
|
+
f"attributes_correct={ev.attributes_correct}, "
|
|
900
|
+
f"granularity_correct={ev.granularity_correct}. "
|
|
749
901
|
f"Actual args: {best.actual_alert_arguments}"
|
|
750
902
|
)
|
|
751
903
|
exc.reasoning_steps = best.reasoning_steps
|