gooddata-eval 1.72.1.dev6__tar.gz → 1.73.1.dev1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/PKG-INFO +37 -2
- {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/README.md +35 -0
- {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/pyproject.toml +2 -2
- {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/cli/agentic_runner.py +44 -12
- {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/cli/main.py +14 -0
- {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/agentic/alert_skill.py +106 -9
- {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/agentic/conversation.py +64 -23
- {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/agentic/general_question.py +46 -5
- {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/agentic/guardrail.py +47 -5
- {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/agentic/kda_skill.py +59 -6
- {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/agentic/metric_skill.py +103 -20
- {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/agentic/search_tool.py +46 -6
- {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/agentic/visualization.py +56 -6
- {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/chat/sse_client.py +19 -2
- {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/config.py +1 -0
- {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/evaluators/visualization.py +40 -17
- {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/models.py +17 -0
- {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/reporting/json_report.py +1 -0
- {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/runner.py +2 -0
- {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/scoring.py +13 -0
- {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/tests/test_agentic_alert_skill.py +230 -0
- {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/tests/test_agentic_conversation.py +210 -0
- gooddata_eval-1.73.1.dev1/tests/test_agentic_general_question.py +210 -0
- gooddata_eval-1.73.1.dev1/tests/test_agentic_guardrail.py +208 -0
- {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/tests/test_agentic_kda_skill.py +147 -0
- gooddata_eval-1.73.1.dev1/tests/test_agentic_metric_skill.py +564 -0
- gooddata_eval-1.73.1.dev1/tests/test_agentic_runner.py +224 -0
- {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/tests/test_agentic_search_tool.py +96 -0
- {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/tests/test_agentic_visualization.py +130 -0
- {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/tests/test_cli.py +110 -0
- {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/tests/test_reporting.py +3 -0
- {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/tests/test_runner.py +35 -0
- {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/tests/test_scoring.py +44 -0
- {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/tests/test_sse_client.py +76 -0
- {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/tests/test_visualization_evaluator.py +43 -0
- gooddata_eval-1.72.1.dev6/tests/test_agentic_general_question.py +0 -100
- gooddata_eval-1.72.1.dev6/tests/test_agentic_guardrail.py +0 -98
- gooddata_eval-1.72.1.dev6/tests/test_agentic_metric_skill.py +0 -286
- {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/.gitignore +0 -0
- {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/LICENSE.txt +0 -0
- {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/Makefile +0 -0
- {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/__init__.py +0 -0
- {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/_version.py +0 -0
- {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/cli/__init__.py +0 -0
- {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/__init__.py +0 -0
- {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/agentic/__init__.py +0 -0
- {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/agentic/_catalog.py +0 -0
- {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/agentic/_langfuse.py +0 -0
- {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/chat/__init__.py +0 -0
- {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/connection.py +0 -0
- {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/dataset/__init__.py +0 -0
- {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/dataset/langfuse_source.py +0 -0
- {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/dataset/local.py +0 -0
- {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/evaluators/__init__.py +0 -0
- {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/evaluators/_deep_subset.py +0 -0
- {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/evaluators/_llm_judge.py +0 -0
- {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/evaluators/_text_utils.py +0 -0
- {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/evaluators/alert_skill.py +0 -0
- {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/evaluators/base.py +0 -0
- {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/evaluators/general_question.py +0 -0
- {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/evaluators/guardrail.py +0 -0
- {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/evaluators/metric_skill.py +0 -0
- {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/evaluators/search_tool.py +0 -0
- {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/evaluators/summary.py +0 -0
- {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/langfuse/__init__.py +0 -0
- {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/langfuse/sink.py +0 -0
- {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/reporting/__init__.py +0 -0
- {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/reporting/console.py +0 -0
- {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/summary/__init__.py +0 -0
- {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/summary/http_client.py +0 -0
- {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/workspace.py +0 -0
- {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/tests/__init__.py +0 -0
- {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/tests/conftest.py +0 -0
- {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/tests/fixtures/sample_dataset/metric_skill_create.json +0 -0
- {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/tests/fixtures/sample_dataset/visualization_revenue.json +0 -0
- {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/tests/fixtures/sse_visualization_stream.txt +0 -0
- {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/tests/test_agentic_langfuse_trace.py +0 -0
- {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/tests/test_agentic_run_context.py +0 -0
- {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/tests/test_alert_skill_evaluator.py +0 -0
- {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/tests/test_connection.py +0 -0
- {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/tests/test_deep_subset.py +0 -0
- {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/tests/test_langfuse_sink.py +0 -0
- {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/tests/test_langfuse_source.py +0 -0
- {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/tests/test_llm_judge.py +0 -0
- {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/tests/test_local_loader.py +0 -0
- {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/tests/test_metric_skill_evaluator.py +0 -0
- {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/tests/test_models.py +0 -0
- {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/tests/test_search_tool_evaluator.py +0 -0
- {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/tests/test_summary_client.py +0 -0
- {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/tests/test_summary_evaluator.py +0 -0
- {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/tests/test_text_evaluators.py +0 -0
- {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/tests/test_workspace.py +0 -0
- {gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/tox.ini +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: gooddata-eval
|
|
3
|
-
Version: 1.
|
|
3
|
+
Version: 1.73.1.dev1
|
|
4
4
|
Summary: Evaluate the GoodData AI agent against your own questions and models.
|
|
5
5
|
Project-URL: Source, https://github.com/gooddata/gooddata-python-sdk
|
|
6
6
|
Author-email: GoodData <support@gooddata.com>
|
|
@@ -17,7 +17,7 @@ Classifier: Topic :: Scientific/Engineering
|
|
|
17
17
|
Classifier: Topic :: Software Development
|
|
18
18
|
Classifier: Typing :: Typed
|
|
19
19
|
Requires-Python: >=3.10
|
|
20
|
-
Requires-Dist: gooddata-sdk~=1.
|
|
20
|
+
Requires-Dist: gooddata-sdk~=1.73.1.dev1
|
|
21
21
|
Requires-Dist: httpx<1.0,>=0.27
|
|
22
22
|
Requires-Dist: orjson<4.0.0,>=3.9.15
|
|
23
23
|
Requires-Dist: pydantic<3.0,>=2.6
|
|
@@ -90,6 +90,40 @@ When the same model id is offered by multiple providers, use the
|
|
|
90
90
|
|
|
91
91
|
Both provider name and provider id are accepted as the prefix.
|
|
92
92
|
|
|
93
|
+
### Targeting a specific AI Hub agent
|
|
94
|
+
|
|
95
|
+
GoodData has no admin-settable "default agent": when a conversation doesn't
|
|
96
|
+
name one, the platform picks whichever agent was last used or last edited in
|
|
97
|
+
that workspace. If your org has several AI Hub agents configured (e.g. one
|
|
98
|
+
scoped to visualization only, another with every skill enabled), evaluating
|
|
99
|
+
without `--agent-id` can silently exercise the wrong one — a
|
|
100
|
+
`metric_skill`/`alert_skill` item run against a visualization-only agent will
|
|
101
|
+
never pass, no matter how well-formed the question is.
|
|
102
|
+
|
|
103
|
+
```bash
|
|
104
|
+
export GD_EVAL_AGENT_ID='eval-all-skills'
|
|
105
|
+
|
|
106
|
+
gd-eval run \
|
|
107
|
+
--host https://your.gooddata.cloud \
|
|
108
|
+
--workspace ecommerce_demo \
|
|
109
|
+
--dataset ./my-dataset \
|
|
110
|
+
--model gpt-5.2 \
|
|
111
|
+
--runs 1 \
|
|
112
|
+
--json results.json
|
|
113
|
+
```
|
|
114
|
+
|
|
115
|
+
Or pass it explicitly instead of via the env var:
|
|
116
|
+
|
|
117
|
+
```bash
|
|
118
|
+
gd-eval run \
|
|
119
|
+
--host https://your.gooddata.cloud \
|
|
120
|
+
--workspace ecommerce_demo \
|
|
121
|
+
--dataset ./my-dataset \
|
|
122
|
+
--agent-id eval-all-skills \
|
|
123
|
+
--model gpt-5.2 \
|
|
124
|
+
--runs 1
|
|
125
|
+
```
|
|
126
|
+
|
|
93
127
|
### All flags
|
|
94
128
|
|
|
95
129
|
#### Connection
|
|
@@ -100,6 +134,7 @@ Both provider name and provider id are accepted as the prefix.
|
|
|
100
134
|
| `--token TOKEN` | `GOODDATA_TOKEN` | API token. Pass via flag or env var. |
|
|
101
135
|
| `--profile NAME` | — | Profile name in `~/.gooddata/profiles.yaml` (same file as the `gdc` CLI). |
|
|
102
136
|
| `--workspace ID` | — | **Required.** Workspace id to evaluate against. |
|
|
137
|
+
| `--agent-id ID` | `GD_EVAL_AGENT_ID` | AI Hub agent every conversation should target. GoodData has no admin-settable default agent — without this, each conversation falls back to whichever agent the platform's last-used/last-edited heuristic resolves, which may not have every skill under test enabled. |
|
|
103
138
|
|
|
104
139
|
#### Dataset source (pick one)
|
|
105
140
|
|
|
@@ -62,6 +62,40 @@ When the same model id is offered by multiple providers, use the
|
|
|
62
62
|
|
|
63
63
|
Both provider name and provider id are accepted as the prefix.
|
|
64
64
|
|
|
65
|
+
### Targeting a specific AI Hub agent
|
|
66
|
+
|
|
67
|
+
GoodData has no admin-settable "default agent": when a conversation doesn't
|
|
68
|
+
name one, the platform picks whichever agent was last used or last edited in
|
|
69
|
+
that workspace. If your org has several AI Hub agents configured (e.g. one
|
|
70
|
+
scoped to visualization only, another with every skill enabled), evaluating
|
|
71
|
+
without `--agent-id` can silently exercise the wrong one — a
|
|
72
|
+
`metric_skill`/`alert_skill` item run against a visualization-only agent will
|
|
73
|
+
never pass, no matter how well-formed the question is.
|
|
74
|
+
|
|
75
|
+
```bash
|
|
76
|
+
export GD_EVAL_AGENT_ID='eval-all-skills'
|
|
77
|
+
|
|
78
|
+
gd-eval run \
|
|
79
|
+
--host https://your.gooddata.cloud \
|
|
80
|
+
--workspace ecommerce_demo \
|
|
81
|
+
--dataset ./my-dataset \
|
|
82
|
+
--model gpt-5.2 \
|
|
83
|
+
--runs 1 \
|
|
84
|
+
--json results.json
|
|
85
|
+
```
|
|
86
|
+
|
|
87
|
+
Or pass it explicitly instead of via the env var:
|
|
88
|
+
|
|
89
|
+
```bash
|
|
90
|
+
gd-eval run \
|
|
91
|
+
--host https://your.gooddata.cloud \
|
|
92
|
+
--workspace ecommerce_demo \
|
|
93
|
+
--dataset ./my-dataset \
|
|
94
|
+
--agent-id eval-all-skills \
|
|
95
|
+
--model gpt-5.2 \
|
|
96
|
+
--runs 1
|
|
97
|
+
```
|
|
98
|
+
|
|
65
99
|
### All flags
|
|
66
100
|
|
|
67
101
|
#### Connection
|
|
@@ -72,6 +106,7 @@ Both provider name and provider id are accepted as the prefix.
|
|
|
72
106
|
| `--token TOKEN` | `GOODDATA_TOKEN` | API token. Pass via flag or env var. |
|
|
73
107
|
| `--profile NAME` | — | Profile name in `~/.gooddata/profiles.yaml` (same file as the `gdc` CLI). |
|
|
74
108
|
| `--workspace ID` | — | **Required.** Workspace id to evaluate against. |
|
|
109
|
+
| `--agent-id ID` | `GD_EVAL_AGENT_ID` | AI Hub agent every conversation should target. GoodData has no admin-settable default agent — without this, each conversation falls back to whichever agent the platform's last-used/last-edited heuristic resolves, which may not have every skill under test enabled. |
|
|
75
110
|
|
|
76
111
|
#### Dataset source (pick one)
|
|
77
112
|
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
# (C) 2026 GoodData Corporation
|
|
2
2
|
[project]
|
|
3
3
|
name = "gooddata-eval"
|
|
4
|
-
version = "1.
|
|
4
|
+
version = "1.73.1.dev1"
|
|
5
5
|
description = "Evaluate the GoodData AI agent against your own questions and models."
|
|
6
6
|
readme = "README.md"
|
|
7
7
|
license = "MIT"
|
|
@@ -11,7 +11,7 @@ authors = [
|
|
|
11
11
|
keywords = ["gooddata", "ai", "evaluation", "llm", "analytics", "cli"]
|
|
12
12
|
requires-python = ">=3.10"
|
|
13
13
|
dependencies = [
|
|
14
|
-
"gooddata-sdk~=1.
|
|
14
|
+
"gooddata-sdk~=1.73.1.dev1",
|
|
15
15
|
"httpx>=0.27,<1.0",
|
|
16
16
|
"orjson>=3.9.15,<4.0.0",
|
|
17
17
|
"pydantic>=2.6,<3.0",
|
{gooddata_eval-1.72.1.dev6 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/cli/agentic_runner.py
RENAMED
|
@@ -16,7 +16,7 @@ from gooddata_eval.core.agentic.metric_skill import evaluate_agentic_metric_skil
|
|
|
16
16
|
from gooddata_eval.core.agentic.search_tool import evaluate_agentic_search_tool
|
|
17
17
|
from gooddata_eval.core.agentic.visualization import evaluate_agentic_visualization
|
|
18
18
|
from gooddata_eval.core.config import ReasoningEffort
|
|
19
|
-
from gooddata_eval.core.models import CreatedVisualization, DatasetItem
|
|
19
|
+
from gooddata_eval.core.models import AgenticEvalOutcome, CreatedVisualization, DatasetItem
|
|
20
20
|
from gooddata_eval.core.runner import EvalReport, ItemReport
|
|
21
21
|
|
|
22
22
|
|
|
@@ -85,8 +85,14 @@ def _dispatch_agentic(
|
|
|
85
85
|
run_ts: str,
|
|
86
86
|
model_version_override: str | None,
|
|
87
87
|
reasoning_effort: ReasoningEffort | None = None,
|
|
88
|
-
|
|
89
|
-
|
|
88
|
+
agent_id: str | None = None,
|
|
89
|
+
) -> AgenticEvalOutcome:
|
|
90
|
+
"""Call the appropriate evaluate_agentic_* function for the item's test_kind.
|
|
91
|
+
|
|
92
|
+
Every evaluate_agentic_* function returns an AgenticEvalOutcome (reasoning_steps,
|
|
93
|
+
conversation_id, response_id, detail) on success and attaches the same four attributes
|
|
94
|
+
to its raised *AssertionError on failure -- no kind is exempt.
|
|
95
|
+
"""
|
|
90
96
|
kind = item.test_kind
|
|
91
97
|
eo = item.expected_output
|
|
92
98
|
lf_kw: _LfKw = {
|
|
@@ -99,85 +105,93 @@ def _dispatch_agentic(
|
|
|
99
105
|
}
|
|
100
106
|
|
|
101
107
|
if kind in ("vis_agentic", "agentic_visualization"):
|
|
102
|
-
evaluate_agentic_visualization(
|
|
108
|
+
return evaluate_agentic_visualization(
|
|
103
109
|
host=host,
|
|
104
110
|
token=token,
|
|
105
111
|
workspace_id=workspace_id,
|
|
106
112
|
question=item.question,
|
|
107
113
|
expected_outputs=_parse_visualization_expected(eo),
|
|
108
114
|
k=k,
|
|
115
|
+
agent_id=agent_id,
|
|
109
116
|
**lf_kw,
|
|
110
117
|
)
|
|
111
118
|
elif kind == "agentic_metric_skill":
|
|
112
|
-
evaluate_agentic_metric_skill(
|
|
119
|
+
return evaluate_agentic_metric_skill(
|
|
113
120
|
host=host,
|
|
114
121
|
token=token,
|
|
115
122
|
workspace_id=workspace_id,
|
|
116
123
|
question=item.question,
|
|
117
124
|
expected_output=eo if isinstance(eo, (dict, list)) else {},
|
|
118
125
|
k=k,
|
|
126
|
+
agent_id=agent_id,
|
|
119
127
|
**lf_kw,
|
|
120
128
|
)
|
|
121
129
|
elif kind == "agentic_alert_skill":
|
|
122
|
-
evaluate_agentic_alert_skill(
|
|
130
|
+
return evaluate_agentic_alert_skill(
|
|
123
131
|
host=host,
|
|
124
132
|
token=token,
|
|
125
133
|
workspace_id=workspace_id,
|
|
126
134
|
question=item.question,
|
|
127
135
|
expected_output=eo if isinstance(eo, dict) else {},
|
|
128
136
|
k=k,
|
|
137
|
+
agent_id=agent_id,
|
|
129
138
|
**lf_kw,
|
|
130
139
|
)
|
|
131
140
|
elif kind == "agentic_search":
|
|
132
141
|
eo_dict = eo if isinstance(eo, dict) else {}
|
|
133
142
|
tool_call = eo_dict.get("tool_call", {})
|
|
134
143
|
expected_args = tool_call.get("function_arguments", eo_dict)
|
|
135
|
-
evaluate_agentic_search_tool(
|
|
144
|
+
return evaluate_agentic_search_tool(
|
|
136
145
|
host=host,
|
|
137
146
|
token=token,
|
|
138
147
|
workspace_id=workspace_id,
|
|
139
148
|
question=item.question,
|
|
140
149
|
expected_tool_call=expected_args,
|
|
141
150
|
k=k,
|
|
151
|
+
agent_id=agent_id,
|
|
142
152
|
**lf_kw,
|
|
143
153
|
)
|
|
144
154
|
elif kind == "agentic_general_question":
|
|
145
|
-
evaluate_agentic_general_question(
|
|
155
|
+
return evaluate_agentic_general_question(
|
|
146
156
|
host=host,
|
|
147
157
|
token=token,
|
|
148
158
|
workspace_id=workspace_id,
|
|
149
159
|
question=item.question,
|
|
150
160
|
expected_output=eo if isinstance(eo, str) else str(eo),
|
|
151
161
|
k=k,
|
|
162
|
+
agent_id=agent_id,
|
|
152
163
|
**lf_kw,
|
|
153
164
|
)
|
|
154
165
|
elif kind == "agentic_guardrail":
|
|
155
|
-
evaluate_agentic_guardrail(
|
|
166
|
+
return evaluate_agentic_guardrail(
|
|
156
167
|
host=host,
|
|
157
168
|
token=token,
|
|
158
169
|
workspace_id=workspace_id,
|
|
159
170
|
question=item.question,
|
|
160
171
|
expected_output=eo if isinstance(eo, str) else str(eo),
|
|
161
172
|
k=k,
|
|
173
|
+
agent_id=agent_id,
|
|
162
174
|
**lf_kw,
|
|
163
175
|
)
|
|
164
176
|
elif kind == "agentic_kda_skill":
|
|
165
|
-
evaluate_agentic_kda_skill(
|
|
177
|
+
return evaluate_agentic_kda_skill(
|
|
166
178
|
host=host,
|
|
167
179
|
token=token,
|
|
168
180
|
workspace_id=workspace_id,
|
|
169
181
|
question=item.question,
|
|
170
182
|
expected_output=eo if isinstance(eo, dict) else {},
|
|
171
183
|
k=k,
|
|
184
|
+
agent_id=agent_id,
|
|
172
185
|
**lf_kw,
|
|
173
186
|
)
|
|
174
187
|
elif kind == "agentic_conversation":
|
|
175
188
|
fixture_data = eo.get("fixture") or eo if isinstance(eo, dict) else {}
|
|
176
|
-
evaluate_agentic_conversation(
|
|
189
|
+
return evaluate_agentic_conversation(
|
|
177
190
|
host=host,
|
|
178
191
|
token=token,
|
|
179
192
|
workspace_id=workspace_id,
|
|
180
193
|
fixture=ConversationFixture.model_validate(fixture_data),
|
|
194
|
+
agent_id=agent_id,
|
|
181
195
|
**lf_kw,
|
|
182
196
|
)
|
|
183
197
|
else:
|
|
@@ -197,6 +211,7 @@ def run_agentic_items(
|
|
|
197
211
|
run_ts: str,
|
|
198
212
|
on_item_start: Any = None,
|
|
199
213
|
on_item_done: Any = None,
|
|
214
|
+
agent_id: str | None = None,
|
|
200
215
|
) -> EvalReport:
|
|
201
216
|
"""Run agentic items through evaluate_agentic_* and return an EvalReport."""
|
|
202
217
|
langfuse = make_langfuse_client() if use_langfuse else None
|
|
@@ -219,12 +234,29 @@ def run_agentic_items(
|
|
|
219
234
|
)
|
|
220
235
|
t0 = time.perf_counter()
|
|
221
236
|
try:
|
|
222
|
-
_dispatch_agentic(
|
|
237
|
+
outcome = _dispatch_agentic(
|
|
238
|
+
item, host, token, workspace_id, k, langfuse, run_ts, model_version, reasoning_effort, agent_id
|
|
239
|
+
)
|
|
240
|
+
if isinstance(outcome, AgenticEvalOutcome):
|
|
241
|
+
reasoning_steps = outcome.reasoning_steps
|
|
242
|
+
conversation_id = outcome.conversation_id
|
|
243
|
+
response_id = outcome.response_id
|
|
244
|
+
detail = outcome.detail
|
|
245
|
+
else:
|
|
246
|
+
reasoning_steps, conversation_id, response_id, detail = outcome, None, None, {}
|
|
223
247
|
item_report.pass_at_k = True
|
|
224
248
|
item_report.runs = k
|
|
249
|
+
item_report.reasoning_steps = reasoning_steps or []
|
|
250
|
+
item_report.conversation_id = conversation_id
|
|
251
|
+
item_report.response_id = response_id
|
|
252
|
+
item_report.best_detail = detail or {}
|
|
225
253
|
except AssertionError as exc:
|
|
226
254
|
item_report.pass_at_k = False
|
|
227
255
|
item_report.runs = k
|
|
256
|
+
item_report.reasoning_steps = getattr(exc, "reasoning_steps", None) or []
|
|
257
|
+
item_report.conversation_id = getattr(exc, "conversation_id", None)
|
|
258
|
+
item_report.response_id = getattr(exc, "response_id", None)
|
|
259
|
+
item_report.best_detail = getattr(exc, "detail", None) or {}
|
|
228
260
|
print(f"[agentic] {item.id} FAIL: {exc}", flush=True)
|
|
229
261
|
except Exception as exc:
|
|
230
262
|
item_report.error = f"{type(exc).__name__}: {exc}"
|
|
@@ -2,6 +2,7 @@
|
|
|
2
2
|
"""`gd-eval` command-line entry point."""
|
|
3
3
|
|
|
4
4
|
import argparse
|
|
5
|
+
import os
|
|
5
6
|
import sys
|
|
6
7
|
import threading
|
|
7
8
|
from datetime import datetime, timezone
|
|
@@ -117,6 +118,16 @@ def _build_parser() -> argparse.ArgumentParser:
|
|
|
117
118
|
action="store_true",
|
|
118
119
|
help="Log scores and traces to Langfuse (requires --langfuse-dataset and LANGFUSE_* env vars).",
|
|
119
120
|
)
|
|
121
|
+
run.add_argument(
|
|
122
|
+
"--agent-id",
|
|
123
|
+
dest="agent_id",
|
|
124
|
+
help=(
|
|
125
|
+
"AI Hub agent id every conversation should target (or set GD_EVAL_AGENT_ID). "
|
|
126
|
+
"GoodData has no admin-settable default agent -- without this, each conversation "
|
|
127
|
+
"falls back to whichever agent the platform's last-used/last-edited heuristic "
|
|
128
|
+
"resolves, which may not have every skill under test enabled."
|
|
129
|
+
),
|
|
130
|
+
)
|
|
120
131
|
models_cmd = sub.add_parser("models", help="List LLM providers and models configured in the org.")
|
|
121
132
|
models_cmd.add_argument("--host", help="GoodData host URL.")
|
|
122
133
|
models_cmd.add_argument("--token", help="API token (or set GOODDATA_TOKEN).")
|
|
@@ -347,6 +358,7 @@ def _run(config: RunConfig) -> int:
|
|
|
347
358
|
run_ts=run_ts,
|
|
348
359
|
on_item_start=on_item_start,
|
|
349
360
|
on_item_done=on_item_done,
|
|
361
|
+
agent_id=config.agent_id,
|
|
350
362
|
)
|
|
351
363
|
|
|
352
364
|
# --- non-agentic items (single-turn, use Evaluator) ---
|
|
@@ -357,6 +369,7 @@ def _run(config: RunConfig) -> int:
|
|
|
357
369
|
workspace_id=config.workspace_id,
|
|
358
370
|
preserve_failed=config.preserve_failed,
|
|
359
371
|
reasoning_effort=config.reasoning_effort,
|
|
372
|
+
agent_id=config.agent_id,
|
|
360
373
|
),
|
|
361
374
|
SummaryClient(host=config.host, token=config.token, workspace_id=config.workspace_id),
|
|
362
375
|
)
|
|
@@ -449,6 +462,7 @@ def main(argv: list[str] | None = None) -> int:
|
|
|
449
462
|
kind=args.kind,
|
|
450
463
|
preserve_failed=args.preserve_failed,
|
|
451
464
|
reasoning_effort=args.reasoning_effort,
|
|
465
|
+
agent_id=args.agent_id or os.environ.get("GD_EVAL_AGENT_ID"),
|
|
452
466
|
)
|
|
453
467
|
return _run(config)
|
|
454
468
|
except (
|
|
@@ -6,7 +6,7 @@ from __future__ import annotations
|
|
|
6
6
|
import json
|
|
7
7
|
import os
|
|
8
8
|
import re
|
|
9
|
-
from dataclasses import dataclass
|
|
9
|
+
from dataclasses import dataclass, field
|
|
10
10
|
from typing import Any
|
|
11
11
|
|
|
12
12
|
from gooddata_sdk import GoodDataSdk
|
|
@@ -14,7 +14,7 @@ from gooddata_sdk import GoodDataSdk
|
|
|
14
14
|
from gooddata_eval.core.agentic._catalog import CatalogMetricAlert
|
|
15
15
|
from gooddata_eval.core.chat.sse_client import ChatClient
|
|
16
16
|
from gooddata_eval.core.config import ReasoningEffort
|
|
17
|
-
from gooddata_eval.core.models import ToolCallEvent
|
|
17
|
+
from gooddata_eval.core.models import AgenticEvalOutcome, ToolCallEvent
|
|
18
18
|
|
|
19
19
|
try:
|
|
20
20
|
from openai import OpenAI as _OpenAI
|
|
@@ -119,7 +119,31 @@ def _check_metric(expected: CatalogMetricAlert, actual_args: dict) -> bool:
|
|
|
119
119
|
return expected.metric_id == act_metric
|
|
120
120
|
|
|
121
121
|
|
|
122
|
-
def
|
|
122
|
+
def _resolve_internal_recipient_ids(sdk: GoodDataSdk, emails: list[str]) -> set[str]:
|
|
123
|
+
"""Best-effort map of expected recipient emails to internal GoodData user ids.
|
|
124
|
+
|
|
125
|
+
Some notification channels are workspace-restricted to internal users --
|
|
126
|
+
`create_metric_alert` then addresses the alert by internal user id
|
|
127
|
+
(`internal_recipients`), never by email, so an expected email has to be
|
|
128
|
+
resolved before it can be compared against that field. Failures (no
|
|
129
|
+
matching user, no permission, network error) are swallowed: the caller
|
|
130
|
+
treats an empty result the same as "this delivery path doesn't match",
|
|
131
|
+
which is correct -- it doesn't mean the alert itself failed.
|
|
132
|
+
"""
|
|
133
|
+
if not emails:
|
|
134
|
+
return set()
|
|
135
|
+
try:
|
|
136
|
+
# RSQL quoted-string escaping: backslash first, then the enclosing quote char,
|
|
137
|
+
# or an email like o'hara@example.com breaks the filter into invalid RSQL.
|
|
138
|
+
escaped = [email.replace("\\", "\\\\").replace("'", "\\'") for email in emails]
|
|
139
|
+
quoted = ",".join(f"'{email}'" for email in escaped)
|
|
140
|
+
resp = sdk._client.entities_api.get_all_entities_users(filter=f"email=in=({quoted})")
|
|
141
|
+
return {u.id for u in (resp.data or [])}
|
|
142
|
+
except Exception:
|
|
143
|
+
return set()
|
|
144
|
+
|
|
145
|
+
|
|
146
|
+
def _check_recipients(expected: CatalogMetricAlert, actual_args: dict, sdk: GoodDataSdk | None = None) -> bool:
|
|
123
147
|
if not expected.recipients:
|
|
124
148
|
return True
|
|
125
149
|
act_recip_raw = actual_args.get("recipients", actual_args.get("external_recipients"))
|
|
@@ -134,7 +158,24 @@ def _check_recipients(expected: CatalogMetricAlert, actual_args: dict) -> bool:
|
|
|
134
158
|
act_recip = act_recip_raw
|
|
135
159
|
else:
|
|
136
160
|
act_recip = []
|
|
137
|
-
|
|
161
|
+
if set(expected.recipients) == set(act_recip or []):
|
|
162
|
+
return True
|
|
163
|
+
act_internal_raw = actual_args.get("internal_recipients")
|
|
164
|
+
# internal_recipients is declared `anyOf: [array of string, string, null]` in the
|
|
165
|
+
# create_metric_alert tool schema -- a single id as a bare string is schema-legal,
|
|
166
|
+
# not a malformed call, so it needs the same string/list normalization already
|
|
167
|
+
# applied to recipients/external_recipients above.
|
|
168
|
+
if isinstance(act_internal_raw, str):
|
|
169
|
+
act_internal = [act_internal_raw]
|
|
170
|
+
elif isinstance(act_internal_raw, list):
|
|
171
|
+
act_internal = act_internal_raw
|
|
172
|
+
else:
|
|
173
|
+
act_internal = []
|
|
174
|
+
if sdk is not None and act_internal:
|
|
175
|
+
internal_recipient_ids = _resolve_internal_recipient_ids(sdk, expected.recipients)
|
|
176
|
+
if internal_recipient_ids & set(act_internal):
|
|
177
|
+
return True
|
|
178
|
+
return False
|
|
138
179
|
|
|
139
180
|
|
|
140
181
|
def generate_simulated_alert_response(
|
|
@@ -302,6 +343,8 @@ class AlertRunResult:
|
|
|
302
343
|
alert_id: str | None
|
|
303
344
|
eval: AlertEvaluation
|
|
304
345
|
actual_alert_arguments: dict
|
|
346
|
+
reasoning_steps: list[str] = field(default_factory=list)
|
|
347
|
+
response_id: str | None = None
|
|
305
348
|
|
|
306
349
|
|
|
307
350
|
@dataclass
|
|
@@ -434,11 +477,14 @@ def run_agentic_alert_skill(
|
|
|
434
477
|
max_iterations: int = _DEFAULT_MAX_ITERATIONS,
|
|
435
478
|
initial_conversation_id: str | None = None,
|
|
436
479
|
reasoning_effort: ReasoningEffort | None = None,
|
|
480
|
+
agent_id: str | None = None,
|
|
437
481
|
) -> AgenticAlertSummary:
|
|
438
482
|
"""Run the alert-skill agentic evaluation K times and return a summary."""
|
|
439
483
|
expected = _normalize_expected_output(expected_output)
|
|
440
484
|
run_results: list[AlertRunResult] = []
|
|
441
|
-
client = ChatClient(
|
|
485
|
+
client = ChatClient(
|
|
486
|
+
host=host, token=token, workspace_id=workspace_id, reasoning_effort=reasoning_effort, agent_id=agent_id
|
|
487
|
+
)
|
|
442
488
|
sdk = GoodDataSdk.create(host, token)
|
|
443
489
|
|
|
444
490
|
def _run_once(conv_id: str) -> AlertRunResult:
|
|
@@ -447,6 +493,8 @@ def run_agentic_alert_skill(
|
|
|
447
493
|
alert_id: str | None = None
|
|
448
494
|
actual_args: dict = {}
|
|
449
495
|
tool_called = False
|
|
496
|
+
reasoning_steps: list[str] = []
|
|
497
|
+
response_id: str | None = None
|
|
450
498
|
# conversation_history stores prior turns for GPT-4o context.
|
|
451
499
|
# Roles follow GPT-4o's perspective: "assistant"=agent text, "user"=sim-user reply.
|
|
452
500
|
conversation_history: list = []
|
|
@@ -454,6 +502,8 @@ def run_agentic_alert_skill(
|
|
|
454
502
|
|
|
455
503
|
for _iteration in range(max_iterations):
|
|
456
504
|
chat_result = client.send_message(conv_id, current_question)
|
|
505
|
+
reasoning_steps.extend(chat_result.reasoning_steps or [])
|
|
506
|
+
response_id = chat_result.response_id or response_id
|
|
457
507
|
alert_id, actual_args, tool_called = _extract_alert_call(chat_result.tool_call_events or [])
|
|
458
508
|
if tool_called:
|
|
459
509
|
alert_id_to_delete = alert_id
|
|
@@ -482,13 +532,15 @@ def run_agentic_alert_skill(
|
|
|
482
532
|
trigger_correct=tool_called and _check_trigger(expected, actual_args),
|
|
483
533
|
filters_correct=tool_called and _check_filters(expected, actual_args),
|
|
484
534
|
metric_correct=tool_called and _check_metric(expected, actual_args),
|
|
485
|
-
recipients_correct=tool_called and _check_recipients(expected, actual_args),
|
|
535
|
+
recipients_correct=tool_called and _check_recipients(expected, actual_args, sdk=sdk),
|
|
486
536
|
)
|
|
487
537
|
return AlertRunResult(
|
|
488
538
|
conversation_id=conv_id,
|
|
489
539
|
alert_id=alert_id,
|
|
490
540
|
eval=ev,
|
|
491
541
|
actual_alert_arguments=actual_args,
|
|
542
|
+
reasoning_steps=reasoning_steps,
|
|
543
|
+
response_id=response_id,
|
|
492
544
|
)
|
|
493
545
|
finally:
|
|
494
546
|
if alert_id_to_delete:
|
|
@@ -539,6 +591,10 @@ class AlertSkillAssertionError(AssertionError):
|
|
|
539
591
|
"""Raised when an alert-skill evaluation fails."""
|
|
540
592
|
|
|
541
593
|
__tracebackhide__ = True
|
|
594
|
+
reasoning_steps: list[str]
|
|
595
|
+
conversation_id: str
|
|
596
|
+
response_id: str | None
|
|
597
|
+
detail: dict
|
|
542
598
|
|
|
543
599
|
|
|
544
600
|
def evaluate_agentic_alert_skill(
|
|
@@ -550,6 +606,7 @@ def evaluate_agentic_alert_skill(
|
|
|
550
606
|
k: int = _DEFAULT_K,
|
|
551
607
|
max_iterations: int = _DEFAULT_MAX_ITERATIONS,
|
|
552
608
|
initial_conversation_id: str | None = None,
|
|
609
|
+
agent_id: str | None = None,
|
|
553
610
|
langfuse: object | None = None,
|
|
554
611
|
dataset_item_id: str = "",
|
|
555
612
|
dataset_name: str = "alert_skill",
|
|
@@ -557,8 +614,16 @@ def evaluate_agentic_alert_skill(
|
|
|
557
614
|
model_version_override: str | None = None,
|
|
558
615
|
run_metadata_extra: dict | None = None,
|
|
559
616
|
reasoning_effort: ReasoningEffort | None = None,
|
|
560
|
-
) ->
|
|
561
|
-
"""Run alert-skill evaluation, log to Langfuse, and raise AlertSkillAssertionError on failure.
|
|
617
|
+
) -> AgenticEvalOutcome:
|
|
618
|
+
"""Run alert-skill evaluation, log to Langfuse, and raise AlertSkillAssertionError on failure.
|
|
619
|
+
|
|
620
|
+
Returns the best run's outcome (reasoning_steps, conversation_id, response_id) as an
|
|
621
|
+
AgenticEvalOutcome on success; on failure the same three values are attached to the
|
|
622
|
+
raised exception as
|
|
623
|
+
``.reasoning_steps``/``.conversation_id``/``.response_id`` (mirrors the
|
|
624
|
+
`conversation_id`-on-exception idiom in `ChatClient.ask()`) so callers can retrieve them
|
|
625
|
+
either way.
|
|
626
|
+
"""
|
|
562
627
|
from datetime import datetime as _dt # noqa: PLC0415
|
|
563
628
|
from datetime import timezone as _tz # noqa: PLC0415
|
|
564
629
|
|
|
@@ -577,6 +642,7 @@ def evaluate_agentic_alert_skill(
|
|
|
577
642
|
max_iterations=max_iterations,
|
|
578
643
|
initial_conversation_id=initial_conversation_id,
|
|
579
644
|
reasoning_effort=reasoning_effort,
|
|
645
|
+
agent_id=agent_id,
|
|
580
646
|
)
|
|
581
647
|
|
|
582
648
|
if langfuse is not None and dataset_item_id:
|
|
@@ -631,7 +697,7 @@ def evaluate_agentic_alert_skill(
|
|
|
631
697
|
if not summary.pass_at_k:
|
|
632
698
|
best = summary.best
|
|
633
699
|
ev = best.eval
|
|
634
|
-
|
|
700
|
+
exc = AlertSkillAssertionError(
|
|
635
701
|
f"Alert skill assertion failed. strict_pass={ev.strict_pass}. "
|
|
636
702
|
f"alert_created={ev.alert_created}, operator_correct={ev.operator_correct}, "
|
|
637
703
|
f"threshold_correct={ev.threshold_correct}, trigger_correct={ev.trigger_correct}, "
|
|
@@ -639,3 +705,34 @@ def evaluate_agentic_alert_skill(
|
|
|
639
705
|
f"recipients_correct={ev.recipients_correct}. "
|
|
640
706
|
f"Actual args: {best.actual_alert_arguments}"
|
|
641
707
|
)
|
|
708
|
+
exc.reasoning_steps = best.reasoning_steps
|
|
709
|
+
exc.conversation_id = best.conversation_id
|
|
710
|
+
exc.response_id = best.response_id
|
|
711
|
+
exc.detail = {
|
|
712
|
+
"alert_created": ev.alert_created,
|
|
713
|
+
"operator_correct": ev.operator_correct,
|
|
714
|
+
"threshold_correct": ev.threshold_correct,
|
|
715
|
+
"trigger_correct": ev.trigger_correct,
|
|
716
|
+
"filters_correct": ev.filters_correct,
|
|
717
|
+
"metric_correct": ev.metric_correct,
|
|
718
|
+
"recipients_correct": ev.recipients_correct,
|
|
719
|
+
"actual_alert_arguments": best.actual_alert_arguments,
|
|
720
|
+
}
|
|
721
|
+
raise exc
|
|
722
|
+
best = summary.best
|
|
723
|
+
ev = best.eval
|
|
724
|
+
return AgenticEvalOutcome(
|
|
725
|
+
reasoning_steps=best.reasoning_steps,
|
|
726
|
+
conversation_id=best.conversation_id,
|
|
727
|
+
response_id=best.response_id,
|
|
728
|
+
detail={
|
|
729
|
+
"alert_created": ev.alert_created,
|
|
730
|
+
"operator_correct": ev.operator_correct,
|
|
731
|
+
"threshold_correct": ev.threshold_correct,
|
|
732
|
+
"trigger_correct": ev.trigger_correct,
|
|
733
|
+
"filters_correct": ev.filters_correct,
|
|
734
|
+
"metric_correct": ev.metric_correct,
|
|
735
|
+
"recipients_correct": ev.recipients_correct,
|
|
736
|
+
"actual_alert_arguments": best.actual_alert_arguments,
|
|
737
|
+
},
|
|
738
|
+
)
|