gooddata-eval 1.72.1.dev5__tar.gz → 1.73.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.73.0}/PKG-INFO +37 -2
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.73.0}/README.md +35 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.73.0}/pyproject.toml +2 -2
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.73.0}/src/gooddata_eval/cli/agentic_runner.py +39 -11
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.73.0}/src/gooddata_eval/cli/main.py +14 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.73.0}/src/gooddata_eval/core/agentic/alert_skill.py +73 -9
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.73.0}/src/gooddata_eval/core/agentic/conversation.py +65 -23
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.73.0}/src/gooddata_eval/core/agentic/general_question.py +6 -1
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.73.0}/src/gooddata_eval/core/agentic/guardrail.py +6 -1
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.73.0}/src/gooddata_eval/core/agentic/metric_skill.py +98 -29
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.73.0}/src/gooddata_eval/core/agentic/search_tool.py +6 -1
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.73.0}/src/gooddata_eval/core/agentic/visualization.py +21 -1
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.73.0}/src/gooddata_eval/core/chat/sse_client.py +5 -1
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.73.0}/src/gooddata_eval/core/config.py +1 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.73.0}/src/gooddata_eval/core/evaluators/visualization.py +14 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.73.0}/src/gooddata_eval/core/models.py +9 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.73.0}/src/gooddata_eval/core/reporting/json_report.py +1 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.73.0}/src/gooddata_eval/core/runner.py +2 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.73.0}/src/gooddata_eval/core/scoring.py +13 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.73.0}/tests/test_agentic_alert_skill.py +197 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.73.0}/tests/test_agentic_conversation.py +248 -0
- gooddata_eval-1.73.0/tests/test_agentic_metric_skill.py +446 -0
- gooddata_eval-1.73.0/tests/test_agentic_runner.py +177 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.73.0}/tests/test_cli.py +110 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.73.0}/tests/test_reporting.py +3 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.73.0}/tests/test_runner.py +35 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.73.0}/tests/test_scoring.py +44 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.73.0}/tests/test_sse_client.py +34 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.73.0}/tests/test_visualization_evaluator.py +43 -0
- gooddata_eval-1.72.1.dev5/tests/test_agentic_metric_skill.py +0 -226
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.73.0}/.gitignore +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.73.0}/LICENSE.txt +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.73.0}/Makefile +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.73.0}/src/gooddata_eval/__init__.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.73.0}/src/gooddata_eval/_version.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.73.0}/src/gooddata_eval/cli/__init__.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.73.0}/src/gooddata_eval/core/__init__.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.73.0}/src/gooddata_eval/core/agentic/__init__.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.73.0}/src/gooddata_eval/core/agentic/_catalog.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.73.0}/src/gooddata_eval/core/agentic/_langfuse.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.73.0}/src/gooddata_eval/core/agentic/kda_skill.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.73.0}/src/gooddata_eval/core/chat/__init__.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.73.0}/src/gooddata_eval/core/connection.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.73.0}/src/gooddata_eval/core/dataset/__init__.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.73.0}/src/gooddata_eval/core/dataset/langfuse_source.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.73.0}/src/gooddata_eval/core/dataset/local.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.73.0}/src/gooddata_eval/core/evaluators/__init__.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.73.0}/src/gooddata_eval/core/evaluators/_deep_subset.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.73.0}/src/gooddata_eval/core/evaluators/_llm_judge.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.73.0}/src/gooddata_eval/core/evaluators/_text_utils.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.73.0}/src/gooddata_eval/core/evaluators/alert_skill.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.73.0}/src/gooddata_eval/core/evaluators/base.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.73.0}/src/gooddata_eval/core/evaluators/general_question.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.73.0}/src/gooddata_eval/core/evaluators/guardrail.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.73.0}/src/gooddata_eval/core/evaluators/metric_skill.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.73.0}/src/gooddata_eval/core/evaluators/search_tool.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.73.0}/src/gooddata_eval/core/evaluators/summary.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.73.0}/src/gooddata_eval/core/langfuse/__init__.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.73.0}/src/gooddata_eval/core/langfuse/sink.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.73.0}/src/gooddata_eval/core/reporting/__init__.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.73.0}/src/gooddata_eval/core/reporting/console.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.73.0}/src/gooddata_eval/core/summary/__init__.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.73.0}/src/gooddata_eval/core/summary/http_client.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.73.0}/src/gooddata_eval/core/workspace.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.73.0}/tests/__init__.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.73.0}/tests/conftest.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.73.0}/tests/fixtures/sample_dataset/metric_skill_create.json +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.73.0}/tests/fixtures/sample_dataset/visualization_revenue.json +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.73.0}/tests/fixtures/sse_visualization_stream.txt +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.73.0}/tests/test_agentic_general_question.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.73.0}/tests/test_agentic_guardrail.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.73.0}/tests/test_agentic_kda_skill.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.73.0}/tests/test_agentic_langfuse_trace.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.73.0}/tests/test_agentic_run_context.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.73.0}/tests/test_agentic_search_tool.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.73.0}/tests/test_agentic_visualization.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.73.0}/tests/test_alert_skill_evaluator.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.73.0}/tests/test_connection.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.73.0}/tests/test_deep_subset.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.73.0}/tests/test_langfuse_sink.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.73.0}/tests/test_langfuse_source.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.73.0}/tests/test_llm_judge.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.73.0}/tests/test_local_loader.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.73.0}/tests/test_metric_skill_evaluator.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.73.0}/tests/test_models.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.73.0}/tests/test_search_tool_evaluator.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.73.0}/tests/test_summary_client.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.73.0}/tests/test_summary_evaluator.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.73.0}/tests/test_text_evaluators.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.73.0}/tests/test_workspace.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.73.0}/tox.ini +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: gooddata-eval
|
|
3
|
-
Version: 1.
|
|
3
|
+
Version: 1.73.0
|
|
4
4
|
Summary: Evaluate the GoodData AI agent against your own questions and models.
|
|
5
5
|
Project-URL: Source, https://github.com/gooddata/gooddata-python-sdk
|
|
6
6
|
Author-email: GoodData <support@gooddata.com>
|
|
@@ -17,7 +17,7 @@ Classifier: Topic :: Scientific/Engineering
|
|
|
17
17
|
Classifier: Topic :: Software Development
|
|
18
18
|
Classifier: Typing :: Typed
|
|
19
19
|
Requires-Python: >=3.10
|
|
20
|
-
Requires-Dist: gooddata-sdk~=1.
|
|
20
|
+
Requires-Dist: gooddata-sdk~=1.73.0
|
|
21
21
|
Requires-Dist: httpx<1.0,>=0.27
|
|
22
22
|
Requires-Dist: orjson<4.0.0,>=3.9.15
|
|
23
23
|
Requires-Dist: pydantic<3.0,>=2.6
|
|
@@ -90,6 +90,40 @@ When the same model id is offered by multiple providers, use the
|
|
|
90
90
|
|
|
91
91
|
Both provider name and provider id are accepted as the prefix.
|
|
92
92
|
|
|
93
|
+
### Targeting a specific AI Hub agent
|
|
94
|
+
|
|
95
|
+
GoodData has no admin-settable "default agent": when a conversation doesn't
|
|
96
|
+
name one, the platform picks whichever agent was last used or last edited in
|
|
97
|
+
that workspace. If your org has several AI Hub agents configured (e.g. one
|
|
98
|
+
scoped to visualization only, another with every skill enabled), evaluating
|
|
99
|
+
without `--agent-id` can silently exercise the wrong one — a
|
|
100
|
+
`metric_skill`/`alert_skill` item run against a visualization-only agent will
|
|
101
|
+
never pass, no matter how well-formed the question is.
|
|
102
|
+
|
|
103
|
+
```bash
|
|
104
|
+
export GD_EVAL_AGENT_ID='eval-all-skills'
|
|
105
|
+
|
|
106
|
+
gd-eval run \
|
|
107
|
+
--host https://your.gooddata.cloud \
|
|
108
|
+
--workspace ecommerce_demo \
|
|
109
|
+
--dataset ./my-dataset \
|
|
110
|
+
--model gpt-5.2 \
|
|
111
|
+
--runs 1 \
|
|
112
|
+
--json results.json
|
|
113
|
+
```
|
|
114
|
+
|
|
115
|
+
Or pass it explicitly instead of via the env var:
|
|
116
|
+
|
|
117
|
+
```bash
|
|
118
|
+
gd-eval run \
|
|
119
|
+
--host https://your.gooddata.cloud \
|
|
120
|
+
--workspace ecommerce_demo \
|
|
121
|
+
--dataset ./my-dataset \
|
|
122
|
+
--agent-id eval-all-skills \
|
|
123
|
+
--model gpt-5.2 \
|
|
124
|
+
--runs 1
|
|
125
|
+
```
|
|
126
|
+
|
|
93
127
|
### All flags
|
|
94
128
|
|
|
95
129
|
#### Connection
|
|
@@ -100,6 +134,7 @@ Both provider name and provider id are accepted as the prefix.
|
|
|
100
134
|
| `--token TOKEN` | `GOODDATA_TOKEN` | API token. Pass via flag or env var. |
|
|
101
135
|
| `--profile NAME` | — | Profile name in `~/.gooddata/profiles.yaml` (same file as the `gdc` CLI). |
|
|
102
136
|
| `--workspace ID` | — | **Required.** Workspace id to evaluate against. |
|
|
137
|
+
| `--agent-id ID` | `GD_EVAL_AGENT_ID` | AI Hub agent every conversation should target. GoodData has no admin-settable default agent — without this, each conversation falls back to whichever agent the platform's last-used/last-edited heuristic resolves, which may not have every skill under test enabled. |
|
|
103
138
|
|
|
104
139
|
#### Dataset source (pick one)
|
|
105
140
|
|
|
@@ -62,6 +62,40 @@ When the same model id is offered by multiple providers, use the
|
|
|
62
62
|
|
|
63
63
|
Both provider name and provider id are accepted as the prefix.
|
|
64
64
|
|
|
65
|
+
### Targeting a specific AI Hub agent
|
|
66
|
+
|
|
67
|
+
GoodData has no admin-settable "default agent": when a conversation doesn't
|
|
68
|
+
name one, the platform picks whichever agent was last used or last edited in
|
|
69
|
+
that workspace. If your org has several AI Hub agents configured (e.g. one
|
|
70
|
+
scoped to visualization only, another with every skill enabled), evaluating
|
|
71
|
+
without `--agent-id` can silently exercise the wrong one — a
|
|
72
|
+
`metric_skill`/`alert_skill` item run against a visualization-only agent will
|
|
73
|
+
never pass, no matter how well-formed the question is.
|
|
74
|
+
|
|
75
|
+
```bash
|
|
76
|
+
export GD_EVAL_AGENT_ID='eval-all-skills'
|
|
77
|
+
|
|
78
|
+
gd-eval run \
|
|
79
|
+
--host https://your.gooddata.cloud \
|
|
80
|
+
--workspace ecommerce_demo \
|
|
81
|
+
--dataset ./my-dataset \
|
|
82
|
+
--model gpt-5.2 \
|
|
83
|
+
--runs 1 \
|
|
84
|
+
--json results.json
|
|
85
|
+
```
|
|
86
|
+
|
|
87
|
+
Or pass it explicitly instead of via the env var:
|
|
88
|
+
|
|
89
|
+
```bash
|
|
90
|
+
gd-eval run \
|
|
91
|
+
--host https://your.gooddata.cloud \
|
|
92
|
+
--workspace ecommerce_demo \
|
|
93
|
+
--dataset ./my-dataset \
|
|
94
|
+
--agent-id eval-all-skills \
|
|
95
|
+
--model gpt-5.2 \
|
|
96
|
+
--runs 1
|
|
97
|
+
```
|
|
98
|
+
|
|
65
99
|
### All flags
|
|
66
100
|
|
|
67
101
|
#### Connection
|
|
@@ -72,6 +106,7 @@ Both provider name and provider id are accepted as the prefix.
|
|
|
72
106
|
| `--token TOKEN` | `GOODDATA_TOKEN` | API token. Pass via flag or env var. |
|
|
73
107
|
| `--profile NAME` | — | Profile name in `~/.gooddata/profiles.yaml` (same file as the `gdc` CLI). |
|
|
74
108
|
| `--workspace ID` | — | **Required.** Workspace id to evaluate against. |
|
|
109
|
+
| `--agent-id ID` | `GD_EVAL_AGENT_ID` | AI Hub agent every conversation should target. GoodData has no admin-settable default agent — without this, each conversation falls back to whichever agent the platform's last-used/last-edited heuristic resolves, which may not have every skill under test enabled. |
|
|
75
110
|
|
|
76
111
|
#### Dataset source (pick one)
|
|
77
112
|
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
# (C) 2026 GoodData Corporation
|
|
2
2
|
[project]
|
|
3
3
|
name = "gooddata-eval"
|
|
4
|
-
version = "1.
|
|
4
|
+
version = "1.73.0"
|
|
5
5
|
description = "Evaluate the GoodData AI agent against your own questions and models."
|
|
6
6
|
readme = "README.md"
|
|
7
7
|
license = "MIT"
|
|
@@ -11,7 +11,7 @@ authors = [
|
|
|
11
11
|
keywords = ["gooddata", "ai", "evaluation", "llm", "analytics", "cli"]
|
|
12
12
|
requires-python = ">=3.10"
|
|
13
13
|
dependencies = [
|
|
14
|
-
"gooddata-sdk~=1.
|
|
14
|
+
"gooddata-sdk~=1.73.0",
|
|
15
15
|
"httpx>=0.27,<1.0",
|
|
16
16
|
"orjson>=3.9.15,<4.0.0",
|
|
17
17
|
"pydantic>=2.6,<3.0",
|
|
@@ -16,7 +16,7 @@ from gooddata_eval.core.agentic.metric_skill import evaluate_agentic_metric_skil
|
|
|
16
16
|
from gooddata_eval.core.agentic.search_tool import evaluate_agentic_search_tool
|
|
17
17
|
from gooddata_eval.core.agentic.visualization import evaluate_agentic_visualization
|
|
18
18
|
from gooddata_eval.core.config import ReasoningEffort
|
|
19
|
-
from gooddata_eval.core.models import CreatedVisualization, DatasetItem
|
|
19
|
+
from gooddata_eval.core.models import AgenticEvalOutcome, CreatedVisualization, DatasetItem
|
|
20
20
|
from gooddata_eval.core.runner import EvalReport, ItemReport
|
|
21
21
|
|
|
22
22
|
|
|
@@ -85,8 +85,14 @@ def _dispatch_agentic(
|
|
|
85
85
|
run_ts: str,
|
|
86
86
|
model_version_override: str | None,
|
|
87
87
|
reasoning_effort: ReasoningEffort | None = None,
|
|
88
|
-
|
|
89
|
-
|
|
88
|
+
agent_id: str | None = None,
|
|
89
|
+
) -> AgenticEvalOutcome | list[str] | None:
|
|
90
|
+
"""Call the appropriate evaluate_agentic_* function for the item's test_kind.
|
|
91
|
+
|
|
92
|
+
Returns whatever that function returns -- alert_skill/metric_skill/conversation return
|
|
93
|
+
an AgenticEvalOutcome; the rest still return None
|
|
94
|
+
(unchanged).
|
|
95
|
+
"""
|
|
90
96
|
kind = item.test_kind
|
|
91
97
|
eo = item.expected_output
|
|
92
98
|
lf_kw: _LfKw = {
|
|
@@ -99,66 +105,72 @@ def _dispatch_agentic(
|
|
|
99
105
|
}
|
|
100
106
|
|
|
101
107
|
if kind in ("vis_agentic", "agentic_visualization"):
|
|
102
|
-
evaluate_agentic_visualization(
|
|
108
|
+
return evaluate_agentic_visualization(
|
|
103
109
|
host=host,
|
|
104
110
|
token=token,
|
|
105
111
|
workspace_id=workspace_id,
|
|
106
112
|
question=item.question,
|
|
107
113
|
expected_outputs=_parse_visualization_expected(eo),
|
|
108
114
|
k=k,
|
|
115
|
+
agent_id=agent_id,
|
|
109
116
|
**lf_kw,
|
|
110
117
|
)
|
|
111
118
|
elif kind == "agentic_metric_skill":
|
|
112
|
-
evaluate_agentic_metric_skill(
|
|
119
|
+
return evaluate_agentic_metric_skill(
|
|
113
120
|
host=host,
|
|
114
121
|
token=token,
|
|
115
122
|
workspace_id=workspace_id,
|
|
116
123
|
question=item.question,
|
|
117
124
|
expected_output=eo if isinstance(eo, (dict, list)) else {},
|
|
118
125
|
k=k,
|
|
126
|
+
agent_id=agent_id,
|
|
119
127
|
**lf_kw,
|
|
120
128
|
)
|
|
121
129
|
elif kind == "agentic_alert_skill":
|
|
122
|
-
evaluate_agentic_alert_skill(
|
|
130
|
+
return evaluate_agentic_alert_skill(
|
|
123
131
|
host=host,
|
|
124
132
|
token=token,
|
|
125
133
|
workspace_id=workspace_id,
|
|
126
134
|
question=item.question,
|
|
127
135
|
expected_output=eo if isinstance(eo, dict) else {},
|
|
128
136
|
k=k,
|
|
137
|
+
agent_id=agent_id,
|
|
129
138
|
**lf_kw,
|
|
130
139
|
)
|
|
131
140
|
elif kind == "agentic_search":
|
|
132
141
|
eo_dict = eo if isinstance(eo, dict) else {}
|
|
133
142
|
tool_call = eo_dict.get("tool_call", {})
|
|
134
143
|
expected_args = tool_call.get("function_arguments", eo_dict)
|
|
135
|
-
evaluate_agentic_search_tool(
|
|
144
|
+
return evaluate_agentic_search_tool(
|
|
136
145
|
host=host,
|
|
137
146
|
token=token,
|
|
138
147
|
workspace_id=workspace_id,
|
|
139
148
|
question=item.question,
|
|
140
149
|
expected_tool_call=expected_args,
|
|
141
150
|
k=k,
|
|
151
|
+
agent_id=agent_id,
|
|
142
152
|
**lf_kw,
|
|
143
153
|
)
|
|
144
154
|
elif kind == "agentic_general_question":
|
|
145
|
-
evaluate_agentic_general_question(
|
|
155
|
+
return evaluate_agentic_general_question(
|
|
146
156
|
host=host,
|
|
147
157
|
token=token,
|
|
148
158
|
workspace_id=workspace_id,
|
|
149
159
|
question=item.question,
|
|
150
160
|
expected_output=eo if isinstance(eo, str) else str(eo),
|
|
151
161
|
k=k,
|
|
162
|
+
agent_id=agent_id,
|
|
152
163
|
**lf_kw,
|
|
153
164
|
)
|
|
154
165
|
elif kind == "agentic_guardrail":
|
|
155
|
-
evaluate_agentic_guardrail(
|
|
166
|
+
return evaluate_agentic_guardrail(
|
|
156
167
|
host=host,
|
|
157
168
|
token=token,
|
|
158
169
|
workspace_id=workspace_id,
|
|
159
170
|
question=item.question,
|
|
160
171
|
expected_output=eo if isinstance(eo, str) else str(eo),
|
|
161
172
|
k=k,
|
|
173
|
+
agent_id=agent_id,
|
|
162
174
|
**lf_kw,
|
|
163
175
|
)
|
|
164
176
|
elif kind == "agentic_kda_skill":
|
|
@@ -173,11 +185,12 @@ def _dispatch_agentic(
|
|
|
173
185
|
)
|
|
174
186
|
elif kind == "agentic_conversation":
|
|
175
187
|
fixture_data = eo.get("fixture") or eo if isinstance(eo, dict) else {}
|
|
176
|
-
evaluate_agentic_conversation(
|
|
188
|
+
return evaluate_agentic_conversation(
|
|
177
189
|
host=host,
|
|
178
190
|
token=token,
|
|
179
191
|
workspace_id=workspace_id,
|
|
180
192
|
fixture=ConversationFixture.model_validate(fixture_data),
|
|
193
|
+
agent_id=agent_id,
|
|
181
194
|
**lf_kw,
|
|
182
195
|
)
|
|
183
196
|
else:
|
|
@@ -197,6 +210,7 @@ def run_agentic_items(
|
|
|
197
210
|
run_ts: str,
|
|
198
211
|
on_item_start: Any = None,
|
|
199
212
|
on_item_done: Any = None,
|
|
213
|
+
agent_id: str | None = None,
|
|
200
214
|
) -> EvalReport:
|
|
201
215
|
"""Run agentic items through evaluate_agentic_* and return an EvalReport."""
|
|
202
216
|
langfuse = make_langfuse_client() if use_langfuse else None
|
|
@@ -219,12 +233,26 @@ def run_agentic_items(
|
|
|
219
233
|
)
|
|
220
234
|
t0 = time.perf_counter()
|
|
221
235
|
try:
|
|
222
|
-
_dispatch_agentic(
|
|
236
|
+
outcome = _dispatch_agentic(
|
|
237
|
+
item, host, token, workspace_id, k, langfuse, run_ts, model_version, reasoning_effort, agent_id
|
|
238
|
+
)
|
|
239
|
+
if isinstance(outcome, AgenticEvalOutcome):
|
|
240
|
+
reasoning_steps = outcome.reasoning_steps
|
|
241
|
+
conversation_id = outcome.conversation_id
|
|
242
|
+
response_id = outcome.response_id
|
|
243
|
+
else:
|
|
244
|
+
reasoning_steps, conversation_id, response_id = outcome, None, None
|
|
223
245
|
item_report.pass_at_k = True
|
|
224
246
|
item_report.runs = k
|
|
247
|
+
item_report.reasoning_steps = reasoning_steps or []
|
|
248
|
+
item_report.conversation_id = conversation_id
|
|
249
|
+
item_report.response_id = response_id
|
|
225
250
|
except AssertionError as exc:
|
|
226
251
|
item_report.pass_at_k = False
|
|
227
252
|
item_report.runs = k
|
|
253
|
+
item_report.reasoning_steps = getattr(exc, "reasoning_steps", None) or []
|
|
254
|
+
item_report.conversation_id = getattr(exc, "conversation_id", None)
|
|
255
|
+
item_report.response_id = getattr(exc, "response_id", None)
|
|
228
256
|
print(f"[agentic] {item.id} FAIL: {exc}", flush=True)
|
|
229
257
|
except Exception as exc:
|
|
230
258
|
item_report.error = f"{type(exc).__name__}: {exc}"
|
|
@@ -2,6 +2,7 @@
|
|
|
2
2
|
"""`gd-eval` command-line entry point."""
|
|
3
3
|
|
|
4
4
|
import argparse
|
|
5
|
+
import os
|
|
5
6
|
import sys
|
|
6
7
|
import threading
|
|
7
8
|
from datetime import datetime, timezone
|
|
@@ -117,6 +118,16 @@ def _build_parser() -> argparse.ArgumentParser:
|
|
|
117
118
|
action="store_true",
|
|
118
119
|
help="Log scores and traces to Langfuse (requires --langfuse-dataset and LANGFUSE_* env vars).",
|
|
119
120
|
)
|
|
121
|
+
run.add_argument(
|
|
122
|
+
"--agent-id",
|
|
123
|
+
dest="agent_id",
|
|
124
|
+
help=(
|
|
125
|
+
"AI Hub agent id every conversation should target (or set GD_EVAL_AGENT_ID). "
|
|
126
|
+
"GoodData has no admin-settable default agent -- without this, each conversation "
|
|
127
|
+
"falls back to whichever agent the platform's last-used/last-edited heuristic "
|
|
128
|
+
"resolves, which may not have every skill under test enabled."
|
|
129
|
+
),
|
|
130
|
+
)
|
|
120
131
|
models_cmd = sub.add_parser("models", help="List LLM providers and models configured in the org.")
|
|
121
132
|
models_cmd.add_argument("--host", help="GoodData host URL.")
|
|
122
133
|
models_cmd.add_argument("--token", help="API token (or set GOODDATA_TOKEN).")
|
|
@@ -347,6 +358,7 @@ def _run(config: RunConfig) -> int:
|
|
|
347
358
|
run_ts=run_ts,
|
|
348
359
|
on_item_start=on_item_start,
|
|
349
360
|
on_item_done=on_item_done,
|
|
361
|
+
agent_id=config.agent_id,
|
|
350
362
|
)
|
|
351
363
|
|
|
352
364
|
# --- non-agentic items (single-turn, use Evaluator) ---
|
|
@@ -357,6 +369,7 @@ def _run(config: RunConfig) -> int:
|
|
|
357
369
|
workspace_id=config.workspace_id,
|
|
358
370
|
preserve_failed=config.preserve_failed,
|
|
359
371
|
reasoning_effort=config.reasoning_effort,
|
|
372
|
+
agent_id=config.agent_id,
|
|
360
373
|
),
|
|
361
374
|
SummaryClient(host=config.host, token=config.token, workspace_id=config.workspace_id),
|
|
362
375
|
)
|
|
@@ -449,6 +462,7 @@ def main(argv: list[str] | None = None) -> int:
|
|
|
449
462
|
kind=args.kind,
|
|
450
463
|
preserve_failed=args.preserve_failed,
|
|
451
464
|
reasoning_effort=args.reasoning_effort,
|
|
465
|
+
agent_id=args.agent_id or os.environ.get("GD_EVAL_AGENT_ID"),
|
|
452
466
|
)
|
|
453
467
|
return _run(config)
|
|
454
468
|
except (
|
{gooddata_eval-1.72.1.dev5 → gooddata_eval-1.73.0}/src/gooddata_eval/core/agentic/alert_skill.py
RENAMED
|
@@ -6,7 +6,7 @@ from __future__ import annotations
|
|
|
6
6
|
import json
|
|
7
7
|
import os
|
|
8
8
|
import re
|
|
9
|
-
from dataclasses import dataclass
|
|
9
|
+
from dataclasses import dataclass, field
|
|
10
10
|
from typing import Any
|
|
11
11
|
|
|
12
12
|
from gooddata_sdk import GoodDataSdk
|
|
@@ -14,7 +14,7 @@ from gooddata_sdk import GoodDataSdk
|
|
|
14
14
|
from gooddata_eval.core.agentic._catalog import CatalogMetricAlert
|
|
15
15
|
from gooddata_eval.core.chat.sse_client import ChatClient
|
|
16
16
|
from gooddata_eval.core.config import ReasoningEffort
|
|
17
|
-
from gooddata_eval.core.models import ToolCallEvent
|
|
17
|
+
from gooddata_eval.core.models import AgenticEvalOutcome, ToolCallEvent
|
|
18
18
|
|
|
19
19
|
try:
|
|
20
20
|
from openai import OpenAI as _OpenAI
|
|
@@ -119,7 +119,31 @@ def _check_metric(expected: CatalogMetricAlert, actual_args: dict) -> bool:
|
|
|
119
119
|
return expected.metric_id == act_metric
|
|
120
120
|
|
|
121
121
|
|
|
122
|
-
def
|
|
122
|
+
def _resolve_internal_recipient_ids(sdk: GoodDataSdk, emails: list[str]) -> set[str]:
|
|
123
|
+
"""Best-effort map of expected recipient emails to internal GoodData user ids.
|
|
124
|
+
|
|
125
|
+
Some notification channels are workspace-restricted to internal users --
|
|
126
|
+
`create_metric_alert` then addresses the alert by internal user id
|
|
127
|
+
(`internal_recipients`), never by email, so an expected email has to be
|
|
128
|
+
resolved before it can be compared against that field. Failures (no
|
|
129
|
+
matching user, no permission, network error) are swallowed: the caller
|
|
130
|
+
treats an empty result the same as "this delivery path doesn't match",
|
|
131
|
+
which is correct -- it doesn't mean the alert itself failed.
|
|
132
|
+
"""
|
|
133
|
+
if not emails:
|
|
134
|
+
return set()
|
|
135
|
+
try:
|
|
136
|
+
# RSQL quoted-string escaping: backslash first, then the enclosing quote char,
|
|
137
|
+
# or an email like o'hara@example.com breaks the filter into invalid RSQL.
|
|
138
|
+
escaped = [email.replace("\\", "\\\\").replace("'", "\\'") for email in emails]
|
|
139
|
+
quoted = ",".join(f"'{email}'" for email in escaped)
|
|
140
|
+
resp = sdk._client.entities_api.get_all_entities_users(filter=f"email=in=({quoted})")
|
|
141
|
+
return {u.id for u in (resp.data or [])}
|
|
142
|
+
except Exception:
|
|
143
|
+
return set()
|
|
144
|
+
|
|
145
|
+
|
|
146
|
+
def _check_recipients(expected: CatalogMetricAlert, actual_args: dict, sdk: GoodDataSdk | None = None) -> bool:
|
|
123
147
|
if not expected.recipients:
|
|
124
148
|
return True
|
|
125
149
|
act_recip_raw = actual_args.get("recipients", actual_args.get("external_recipients"))
|
|
@@ -134,7 +158,14 @@ def _check_recipients(expected: CatalogMetricAlert, actual_args: dict) -> bool:
|
|
|
134
158
|
act_recip = act_recip_raw
|
|
135
159
|
else:
|
|
136
160
|
act_recip = []
|
|
137
|
-
|
|
161
|
+
if set(expected.recipients) == set(act_recip or []):
|
|
162
|
+
return True
|
|
163
|
+
act_internal = actual_args.get("internal_recipients")
|
|
164
|
+
if sdk is not None and isinstance(act_internal, list) and act_internal:
|
|
165
|
+
internal_recipient_ids = _resolve_internal_recipient_ids(sdk, expected.recipients)
|
|
166
|
+
if internal_recipient_ids & set(act_internal):
|
|
167
|
+
return True
|
|
168
|
+
return False
|
|
138
169
|
|
|
139
170
|
|
|
140
171
|
def generate_simulated_alert_response(
|
|
@@ -302,6 +333,8 @@ class AlertRunResult:
|
|
|
302
333
|
alert_id: str | None
|
|
303
334
|
eval: AlertEvaluation
|
|
304
335
|
actual_alert_arguments: dict
|
|
336
|
+
reasoning_steps: list[str] = field(default_factory=list)
|
|
337
|
+
response_id: str | None = None
|
|
305
338
|
|
|
306
339
|
|
|
307
340
|
@dataclass
|
|
@@ -434,11 +467,14 @@ def run_agentic_alert_skill(
|
|
|
434
467
|
max_iterations: int = _DEFAULT_MAX_ITERATIONS,
|
|
435
468
|
initial_conversation_id: str | None = None,
|
|
436
469
|
reasoning_effort: ReasoningEffort | None = None,
|
|
470
|
+
agent_id: str | None = None,
|
|
437
471
|
) -> AgenticAlertSummary:
|
|
438
472
|
"""Run the alert-skill agentic evaluation K times and return a summary."""
|
|
439
473
|
expected = _normalize_expected_output(expected_output)
|
|
440
474
|
run_results: list[AlertRunResult] = []
|
|
441
|
-
client = ChatClient(
|
|
475
|
+
client = ChatClient(
|
|
476
|
+
host=host, token=token, workspace_id=workspace_id, reasoning_effort=reasoning_effort, agent_id=agent_id
|
|
477
|
+
)
|
|
442
478
|
sdk = GoodDataSdk.create(host, token)
|
|
443
479
|
|
|
444
480
|
def _run_once(conv_id: str) -> AlertRunResult:
|
|
@@ -447,6 +483,8 @@ def run_agentic_alert_skill(
|
|
|
447
483
|
alert_id: str | None = None
|
|
448
484
|
actual_args: dict = {}
|
|
449
485
|
tool_called = False
|
|
486
|
+
reasoning_steps: list[str] = []
|
|
487
|
+
response_id: str | None = None
|
|
450
488
|
# conversation_history stores prior turns for GPT-4o context.
|
|
451
489
|
# Roles follow GPT-4o's perspective: "assistant"=agent text, "user"=sim-user reply.
|
|
452
490
|
conversation_history: list = []
|
|
@@ -454,6 +492,8 @@ def run_agentic_alert_skill(
|
|
|
454
492
|
|
|
455
493
|
for _iteration in range(max_iterations):
|
|
456
494
|
chat_result = client.send_message(conv_id, current_question)
|
|
495
|
+
reasoning_steps.extend(chat_result.reasoning_steps or [])
|
|
496
|
+
response_id = chat_result.response_id or response_id
|
|
457
497
|
alert_id, actual_args, tool_called = _extract_alert_call(chat_result.tool_call_events or [])
|
|
458
498
|
if tool_called:
|
|
459
499
|
alert_id_to_delete = alert_id
|
|
@@ -482,13 +522,15 @@ def run_agentic_alert_skill(
|
|
|
482
522
|
trigger_correct=tool_called and _check_trigger(expected, actual_args),
|
|
483
523
|
filters_correct=tool_called and _check_filters(expected, actual_args),
|
|
484
524
|
metric_correct=tool_called and _check_metric(expected, actual_args),
|
|
485
|
-
recipients_correct=tool_called and _check_recipients(expected, actual_args),
|
|
525
|
+
recipients_correct=tool_called and _check_recipients(expected, actual_args, sdk=sdk),
|
|
486
526
|
)
|
|
487
527
|
return AlertRunResult(
|
|
488
528
|
conversation_id=conv_id,
|
|
489
529
|
alert_id=alert_id,
|
|
490
530
|
eval=ev,
|
|
491
531
|
actual_alert_arguments=actual_args,
|
|
532
|
+
reasoning_steps=reasoning_steps,
|
|
533
|
+
response_id=response_id,
|
|
492
534
|
)
|
|
493
535
|
finally:
|
|
494
536
|
if alert_id_to_delete:
|
|
@@ -539,6 +581,9 @@ class AlertSkillAssertionError(AssertionError):
|
|
|
539
581
|
"""Raised when an alert-skill evaluation fails."""
|
|
540
582
|
|
|
541
583
|
__tracebackhide__ = True
|
|
584
|
+
reasoning_steps: list[str]
|
|
585
|
+
conversation_id: str
|
|
586
|
+
response_id: str | None
|
|
542
587
|
|
|
543
588
|
|
|
544
589
|
def evaluate_agentic_alert_skill(
|
|
@@ -550,6 +595,7 @@ def evaluate_agentic_alert_skill(
|
|
|
550
595
|
k: int = _DEFAULT_K,
|
|
551
596
|
max_iterations: int = _DEFAULT_MAX_ITERATIONS,
|
|
552
597
|
initial_conversation_id: str | None = None,
|
|
598
|
+
agent_id: str | None = None,
|
|
553
599
|
langfuse: object | None = None,
|
|
554
600
|
dataset_item_id: str = "",
|
|
555
601
|
dataset_name: str = "alert_skill",
|
|
@@ -557,8 +603,16 @@ def evaluate_agentic_alert_skill(
|
|
|
557
603
|
model_version_override: str | None = None,
|
|
558
604
|
run_metadata_extra: dict | None = None,
|
|
559
605
|
reasoning_effort: ReasoningEffort | None = None,
|
|
560
|
-
) ->
|
|
561
|
-
"""Run alert-skill evaluation, log to Langfuse, and raise AlertSkillAssertionError on failure.
|
|
606
|
+
) -> AgenticEvalOutcome:
|
|
607
|
+
"""Run alert-skill evaluation, log to Langfuse, and raise AlertSkillAssertionError on failure.
|
|
608
|
+
|
|
609
|
+
Returns the best run's outcome (reasoning_steps, conversation_id, response_id) as an
|
|
610
|
+
AgenticEvalOutcome on success; on failure the same three values are attached to the
|
|
611
|
+
raised exception as
|
|
612
|
+
``.reasoning_steps``/``.conversation_id``/``.response_id`` (mirrors the
|
|
613
|
+
`conversation_id`-on-exception idiom in `ChatClient.ask()`) so callers can retrieve them
|
|
614
|
+
either way.
|
|
615
|
+
"""
|
|
562
616
|
from datetime import datetime as _dt # noqa: PLC0415
|
|
563
617
|
from datetime import timezone as _tz # noqa: PLC0415
|
|
564
618
|
|
|
@@ -577,6 +631,7 @@ def evaluate_agentic_alert_skill(
|
|
|
577
631
|
max_iterations=max_iterations,
|
|
578
632
|
initial_conversation_id=initial_conversation_id,
|
|
579
633
|
reasoning_effort=reasoning_effort,
|
|
634
|
+
agent_id=agent_id,
|
|
580
635
|
)
|
|
581
636
|
|
|
582
637
|
if langfuse is not None and dataset_item_id:
|
|
@@ -631,7 +686,7 @@ def evaluate_agentic_alert_skill(
|
|
|
631
686
|
if not summary.pass_at_k:
|
|
632
687
|
best = summary.best
|
|
633
688
|
ev = best.eval
|
|
634
|
-
|
|
689
|
+
exc = AlertSkillAssertionError(
|
|
635
690
|
f"Alert skill assertion failed. strict_pass={ev.strict_pass}. "
|
|
636
691
|
f"alert_created={ev.alert_created}, operator_correct={ev.operator_correct}, "
|
|
637
692
|
f"threshold_correct={ev.threshold_correct}, trigger_correct={ev.trigger_correct}, "
|
|
@@ -639,3 +694,12 @@ def evaluate_agentic_alert_skill(
|
|
|
639
694
|
f"recipients_correct={ev.recipients_correct}. "
|
|
640
695
|
f"Actual args: {best.actual_alert_arguments}"
|
|
641
696
|
)
|
|
697
|
+
exc.reasoning_steps = best.reasoning_steps
|
|
698
|
+
exc.conversation_id = best.conversation_id
|
|
699
|
+
exc.response_id = best.response_id
|
|
700
|
+
raise exc
|
|
701
|
+
return AgenticEvalOutcome(
|
|
702
|
+
reasoning_steps=summary.best.reasoning_steps,
|
|
703
|
+
conversation_id=summary.best.conversation_id,
|
|
704
|
+
response_id=summary.best.response_id,
|
|
705
|
+
)
|