gooddata-eval 1.71.1.dev2__tar.gz → 1.72.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/PKG-INFO +2 -2
- {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/pyproject.toml +2 -2
- {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/src/gooddata_eval/core/scoring.py +70 -20
- gooddata_eval-1.72.0/tests/test_scoring.py +163 -0
- {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/tests/test_visualization_evaluator.py +28 -0
- gooddata_eval-1.71.1.dev2/tests/test_scoring.py +0 -66
- {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/.gitignore +0 -0
- {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/LICENSE.txt +0 -0
- {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/Makefile +0 -0
- {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/README.md +0 -0
- {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/src/gooddata_eval/__init__.py +0 -0
- {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/src/gooddata_eval/_version.py +0 -0
- {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/src/gooddata_eval/cli/__init__.py +0 -0
- {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/src/gooddata_eval/cli/agentic_runner.py +0 -0
- {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/src/gooddata_eval/cli/main.py +0 -0
- {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/src/gooddata_eval/core/__init__.py +0 -0
- {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/src/gooddata_eval/core/agentic/__init__.py +0 -0
- {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/src/gooddata_eval/core/agentic/_catalog.py +0 -0
- {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/src/gooddata_eval/core/agentic/_langfuse.py +0 -0
- {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/src/gooddata_eval/core/agentic/alert_skill.py +0 -0
- {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/src/gooddata_eval/core/agentic/conversation.py +0 -0
- {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/src/gooddata_eval/core/agentic/general_question.py +0 -0
- {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/src/gooddata_eval/core/agentic/guardrail.py +0 -0
- {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/src/gooddata_eval/core/agentic/metric_skill.py +0 -0
- {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/src/gooddata_eval/core/agentic/search_tool.py +0 -0
- {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/src/gooddata_eval/core/agentic/visualization.py +0 -0
- {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/src/gooddata_eval/core/chat/__init__.py +0 -0
- {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/src/gooddata_eval/core/chat/sse_client.py +0 -0
- {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/src/gooddata_eval/core/config.py +0 -0
- {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/src/gooddata_eval/core/connection.py +0 -0
- {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/src/gooddata_eval/core/dataset/__init__.py +0 -0
- {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/src/gooddata_eval/core/dataset/langfuse_source.py +0 -0
- {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/src/gooddata_eval/core/dataset/local.py +0 -0
- {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/src/gooddata_eval/core/evaluators/__init__.py +0 -0
- {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/src/gooddata_eval/core/evaluators/_deep_subset.py +0 -0
- {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/src/gooddata_eval/core/evaluators/_llm_judge.py +0 -0
- {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/src/gooddata_eval/core/evaluators/_text_utils.py +0 -0
- {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/src/gooddata_eval/core/evaluators/alert_skill.py +0 -0
- {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/src/gooddata_eval/core/evaluators/base.py +0 -0
- {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/src/gooddata_eval/core/evaluators/general_question.py +0 -0
- {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/src/gooddata_eval/core/evaluators/guardrail.py +0 -0
- {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/src/gooddata_eval/core/evaluators/metric_skill.py +0 -0
- {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/src/gooddata_eval/core/evaluators/search_tool.py +0 -0
- {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/src/gooddata_eval/core/evaluators/summary.py +0 -0
- {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/src/gooddata_eval/core/evaluators/visualization.py +0 -0
- {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/src/gooddata_eval/core/langfuse/__init__.py +0 -0
- {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/src/gooddata_eval/core/langfuse/sink.py +0 -0
- {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/src/gooddata_eval/core/models.py +0 -0
- {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/src/gooddata_eval/core/reporting/__init__.py +0 -0
- {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/src/gooddata_eval/core/reporting/console.py +0 -0
- {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/src/gooddata_eval/core/reporting/json_report.py +0 -0
- {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/src/gooddata_eval/core/runner.py +0 -0
- {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/src/gooddata_eval/core/summary/__init__.py +0 -0
- {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/src/gooddata_eval/core/summary/http_client.py +0 -0
- {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/src/gooddata_eval/core/workspace.py +0 -0
- {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/tests/__init__.py +0 -0
- {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/tests/conftest.py +0 -0
- {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/tests/fixtures/sample_dataset/metric_skill_create.json +0 -0
- {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/tests/fixtures/sample_dataset/visualization_revenue.json +0 -0
- {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/tests/fixtures/sse_visualization_stream.txt +0 -0
- {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/tests/test_agentic_alert_skill.py +0 -0
- {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/tests/test_agentic_conversation.py +0 -0
- {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/tests/test_agentic_general_question.py +0 -0
- {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/tests/test_agentic_guardrail.py +0 -0
- {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/tests/test_agentic_metric_skill.py +0 -0
- {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/tests/test_agentic_run_context.py +0 -0
- {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/tests/test_agentic_search_tool.py +0 -0
- {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/tests/test_agentic_visualization.py +0 -0
- {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/tests/test_alert_skill_evaluator.py +0 -0
- {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/tests/test_cli.py +0 -0
- {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/tests/test_connection.py +0 -0
- {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/tests/test_deep_subset.py +0 -0
- {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/tests/test_langfuse_sink.py +0 -0
- {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/tests/test_langfuse_source.py +0 -0
- {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/tests/test_llm_judge.py +0 -0
- {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/tests/test_local_loader.py +0 -0
- {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/tests/test_metric_skill_evaluator.py +0 -0
- {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/tests/test_models.py +0 -0
- {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/tests/test_reporting.py +0 -0
- {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/tests/test_runner.py +0 -0
- {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/tests/test_search_tool_evaluator.py +0 -0
- {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/tests/test_sse_client.py +0 -0
- {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/tests/test_summary_client.py +0 -0
- {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/tests/test_summary_evaluator.py +0 -0
- {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/tests/test_text_evaluators.py +0 -0
- {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/tests/test_workspace.py +0 -0
- {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/tox.ini +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: gooddata-eval
|
|
3
|
-
Version: 1.
|
|
3
|
+
Version: 1.72.0
|
|
4
4
|
Summary: Evaluate the GoodData AI agent against your own questions and models.
|
|
5
5
|
Project-URL: Source, https://github.com/gooddata/gooddata-python-sdk
|
|
6
6
|
Author-email: GoodData <support@gooddata.com>
|
|
@@ -17,7 +17,7 @@ Classifier: Topic :: Scientific/Engineering
|
|
|
17
17
|
Classifier: Topic :: Software Development
|
|
18
18
|
Classifier: Typing :: Typed
|
|
19
19
|
Requires-Python: >=3.10
|
|
20
|
-
Requires-Dist: gooddata-sdk~=1.
|
|
20
|
+
Requires-Dist: gooddata-sdk~=1.72.0
|
|
21
21
|
Requires-Dist: httpx<1.0,>=0.27
|
|
22
22
|
Requires-Dist: orjson<4.0.0,>=3.9.15
|
|
23
23
|
Requires-Dist: pydantic<3.0,>=2.6
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
# (C) 2026 GoodData Corporation
|
|
2
2
|
[project]
|
|
3
3
|
name = "gooddata-eval"
|
|
4
|
-
version = "1.
|
|
4
|
+
version = "1.72.0"
|
|
5
5
|
description = "Evaluate the GoodData AI agent against your own questions and models."
|
|
6
6
|
readme = "README.md"
|
|
7
7
|
license = "MIT"
|
|
@@ -11,7 +11,7 @@ authors = [
|
|
|
11
11
|
keywords = ["gooddata", "ai", "evaluation", "llm", "analytics", "cli"]
|
|
12
12
|
requires-python = ">=3.10"
|
|
13
13
|
dependencies = [
|
|
14
|
-
"gooddata-sdk~=1.
|
|
14
|
+
"gooddata-sdk~=1.72.0",
|
|
15
15
|
"httpx>=0.27,<1.0",
|
|
16
16
|
"orjson>=3.9.15,<4.0.0",
|
|
17
17
|
"pydantic>=2.6,<3.0",
|
|
@@ -63,29 +63,46 @@ def uri_to_display_name(uri: str) -> str:
|
|
|
63
63
|
|
|
64
64
|
|
|
65
65
|
def validate_cross_references(viz: CreatedVisualization) -> tuple[bool, list[str]]:
|
|
66
|
-
"""Validate ranking-filter `using`/`attribute` resolve to correct URI prefixes.
|
|
66
|
+
"""Validate ranking-filter `using`/`attribute` resolve to correct URI prefixes.
|
|
67
|
+
|
|
68
|
+
Always returns `(ok, errors)` — a malformed filter produces an error entry, never an
|
|
69
|
+
exception. Anything unusable (None, empty, non-string) used to reach `.startswith()`
|
|
70
|
+
or `dict.get()` and blow up with AttributeError/TypeError mid-evaluation.
|
|
71
|
+
|
|
72
|
+
`using` is required by the AAC schema, `attribute` is optional (see
|
|
73
|
+
`_normalize_ranking_filter`), so an absent/None/empty `attribute` is accepted silently.
|
|
74
|
+
"""
|
|
67
75
|
errors: list[str] = []
|
|
68
76
|
fields = viz.query.fields
|
|
69
77
|
for filter_key, filter_dict in viz.query.filter_by.items():
|
|
70
78
|
if filter_dict.get("type") != "ranking_filter":
|
|
71
79
|
continue
|
|
72
|
-
using_val = filter_dict.get("using"
|
|
73
|
-
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
|
|
80
|
-
)
|
|
81
|
-
if "attribute" in filter_dict:
|
|
82
|
-
attr_val = filter_dict["attribute"]
|
|
83
|
-
attr_uri = _resolve_alias_to_uri(attr_val, fields)
|
|
84
|
-
if not attr_uri.startswith(("label/", "attribute/")):
|
|
80
|
+
using_val = filter_dict.get("using")
|
|
81
|
+
if not isinstance(using_val, str) or not using_val:
|
|
82
|
+
errors.append(f"ranking filter '{filter_key}': using={using_val!r} — a metric/ or fact/ URI is required")
|
|
83
|
+
else:
|
|
84
|
+
using_uri = _resolve_alias_to_uri(using_val, fields)
|
|
85
|
+
field_def = fields.get(using_val)
|
|
86
|
+
is_adhoc_agg = isinstance(field_def, AacQueryField) and bool(field_def.aggregation)
|
|
87
|
+
if not using_uri.startswith(("metric/", "fact/")) and not is_adhoc_agg:
|
|
85
88
|
errors.append(
|
|
86
|
-
f"ranking filter '{filter_key}':
|
|
87
|
-
f"resolves to '{
|
|
89
|
+
f"ranking filter '{filter_key}': using='{using_val}' "
|
|
90
|
+
f"resolves to '{using_uri}' — expected a metric/ or fact/ URI"
|
|
88
91
|
)
|
|
92
|
+
attr_val = filter_dict.get("attribute")
|
|
93
|
+
if attr_val is None or attr_val == "":
|
|
94
|
+
continue
|
|
95
|
+
if not isinstance(attr_val, str):
|
|
96
|
+
errors.append(
|
|
97
|
+
f"ranking filter '{filter_key}': attribute={attr_val!r} — expected a label/ or attribute/ URI"
|
|
98
|
+
)
|
|
99
|
+
continue
|
|
100
|
+
attr_uri = _resolve_alias_to_uri(attr_val, fields)
|
|
101
|
+
if not attr_uri.startswith(("label/", "attribute/")):
|
|
102
|
+
errors.append(
|
|
103
|
+
f"ranking filter '{filter_key}': attribute='{attr_val}' "
|
|
104
|
+
f"resolves to '{attr_uri}' — expected a label/ or attribute/ URI"
|
|
105
|
+
)
|
|
89
106
|
return len(errors) == 0, errors
|
|
90
107
|
|
|
91
108
|
|
|
@@ -99,11 +116,43 @@ def _normalize_date_filter(filter_dict: dict, _fields: dict) -> dict:
|
|
|
99
116
|
}
|
|
100
117
|
|
|
101
118
|
|
|
102
|
-
def
|
|
119
|
+
def _sole_dimension_uri(viz: CreatedVisualization) -> str | None:
|
|
120
|
+
"""URI of the visualization's only dimension, or None when it has zero or several."""
|
|
121
|
+
dim_uris = get_dimension_uri_set(viz)
|
|
122
|
+
return next(iter(dim_uris)) if len(dim_uris) == 1 else None
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
def _normalize_ranking_filter(
|
|
126
|
+
filter_dict: dict,
|
|
127
|
+
fields: dict[str, AacQueryField | str],
|
|
128
|
+
sole_dim_uri: str | None = None,
|
|
129
|
+
) -> dict:
|
|
130
|
+
"""Canonicalize a ranking filter so equivalent filters compare equal.
|
|
131
|
+
|
|
132
|
+
`attribute` is optional in the AAC schema (gen-ai models it as `NotRequired[str]` /
|
|
133
|
+
`str | None`), and when it is omitted AFM ranks over every dimension of the result. For a
|
|
134
|
+
single-dimension visualization that is exactly "rank by that one dimension", so an omitted
|
|
135
|
+
attribute is filled in with `sole_dim_uri` instead of comparing as an empty string — the
|
|
136
|
+
agent and the dataset may legitimately express the same filter either way.
|
|
137
|
+
|
|
138
|
+
The substitution is deliberately gated on there being exactly ONE dimension: with two or
|
|
139
|
+
more, omitting `attribute` ranks over the dimension *tuple*, which is a different filter,
|
|
140
|
+
so those stay strict. Callers pass the sole dimension of the visualization the filter
|
|
141
|
+
belongs to, which makes the comparison symmetric — it does not matter which side omitted it.
|
|
142
|
+
|
|
143
|
+
Missing, None and "" are all treated as "not specified"; so is a non-string, which
|
|
144
|
+
`validate_cross_references` reports separately rather than crashing the comparison.
|
|
145
|
+
"""
|
|
146
|
+
attr_val = filter_dict.get("attribute")
|
|
147
|
+
if not isinstance(attr_val, str) or not attr_val:
|
|
148
|
+
dim_uri = sole_dim_uri or ""
|
|
149
|
+
else:
|
|
150
|
+
dim_uri = _resolve_alias_to_uri(attr_val, fields)
|
|
151
|
+
using_val = filter_dict.get("using")
|
|
103
152
|
entry: dict = {
|
|
104
153
|
"type": "ranking_filter",
|
|
105
|
-
"metric_uri": _resolve_alias_to_uri(
|
|
106
|
-
"dim_uri":
|
|
154
|
+
"metric_uri": _resolve_alias_to_uri(using_val, fields) if isinstance(using_val, str) else "",
|
|
155
|
+
"dim_uri": dim_uri,
|
|
107
156
|
}
|
|
108
157
|
if "top" in filter_dict:
|
|
109
158
|
entry["top"] = filter_dict["top"]
|
|
@@ -127,12 +176,13 @@ def _split_and_normalize_filters(viz: CreatedVisualization) -> tuple[set[str], s
|
|
|
127
176
|
ranking_set: set[str] = set()
|
|
128
177
|
attr_set: set[str] = set()
|
|
129
178
|
fields = viz.query.fields
|
|
179
|
+
sole_dim_uri = _sole_dimension_uri(viz)
|
|
130
180
|
for filter_dict in viz.query.filter_by.values():
|
|
131
181
|
ft = filter_dict.get("type")
|
|
132
182
|
if ft == "date_filter":
|
|
133
183
|
date_set.add(json.dumps(_normalize_date_filter(filter_dict, fields), sort_keys=True))
|
|
134
184
|
elif ft == "ranking_filter":
|
|
135
|
-
ranking_set.add(json.dumps(_normalize_ranking_filter(filter_dict, fields), sort_keys=True))
|
|
185
|
+
ranking_set.add(json.dumps(_normalize_ranking_filter(filter_dict, fields, sole_dim_uri), sort_keys=True))
|
|
136
186
|
elif ft == "attribute_filter":
|
|
137
187
|
attr_set.add(json.dumps(_normalize_attribute_filter(filter_dict, fields), sort_keys=True))
|
|
138
188
|
return date_set, ranking_set, attr_set
|
|
@@ -0,0 +1,163 @@
|
|
|
1
|
+
# (C) 2026 GoodData Corporation
|
|
2
|
+
from gooddata_eval.core.models import CreatedVisualization
|
|
3
|
+
from gooddata_eval.core.scoring import (
|
|
4
|
+
check_filters,
|
|
5
|
+
check_viz_type,
|
|
6
|
+
get_dimension_uri_set,
|
|
7
|
+
get_metric_uri_set,
|
|
8
|
+
uri_to_display_name,
|
|
9
|
+
validate_cross_references,
|
|
10
|
+
)
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
def _viz(**kw) -> CreatedVisualization:
|
|
14
|
+
base = {"id": "v", "type": "", "query": {"fields": {}, "filter_by": {}}}
|
|
15
|
+
base.update(kw)
|
|
16
|
+
return CreatedVisualization.model_validate(base)
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def test_metric_and_dimension_uri_sets_resolve_aliases():
|
|
20
|
+
viz = _viz(
|
|
21
|
+
query={
|
|
22
|
+
"fields": {"m_rev": {"using": "metric/revenue"}, "d_q": {"using": "label/date.quarter"}},
|
|
23
|
+
"filter_by": {},
|
|
24
|
+
},
|
|
25
|
+
metrics=["m_rev"],
|
|
26
|
+
view_by=["d_q"],
|
|
27
|
+
)
|
|
28
|
+
assert get_metric_uri_set(viz) == {"metric/revenue"}
|
|
29
|
+
assert get_dimension_uri_set(viz) == {"label/date.quarter"}
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def test_uri_to_display_name():
|
|
33
|
+
assert uri_to_display_name("metric/net_sales") == "net sales"
|
|
34
|
+
assert uri_to_display_name("label/date.month") == "date - month"
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def test_validate_cross_references_flags_bad_ranking_using():
|
|
38
|
+
viz = _viz(
|
|
39
|
+
query={
|
|
40
|
+
"fields": {"d_q": {"using": "label/date.quarter"}},
|
|
41
|
+
"filter_by": {"f_rank": {"type": "ranking_filter", "top": 5, "using": "d_q"}},
|
|
42
|
+
}
|
|
43
|
+
)
|
|
44
|
+
ok, errors = validate_cross_references(viz)
|
|
45
|
+
assert ok is False
|
|
46
|
+
assert errors and "ranking filter" in errors[0]
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def test_check_viz_type_empty_expected_is_wildcard():
|
|
50
|
+
expected = _viz(type="")
|
|
51
|
+
actual = _viz(type="column_chart")
|
|
52
|
+
assert check_viz_type(expected, actual) is True
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def test_check_viz_type_strict_match_normalizes():
|
|
56
|
+
expected = _viz(type="column_chart")
|
|
57
|
+
actual = _viz(type="COLUMN")
|
|
58
|
+
assert check_viz_type(expected, actual) is True
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
def test_check_filters_exact_attribute_match():
|
|
62
|
+
f = {"f_a": {"type": "attribute_filter", "using": "label/region", "state": {"include": ["EMEA"]}}}
|
|
63
|
+
expected = _viz(query={"fields": {}, "filter_by": f})
|
|
64
|
+
actual = _viz(query={"fields": {}, "filter_by": f})
|
|
65
|
+
scores = check_filters(expected, actual)
|
|
66
|
+
assert scores.all_ok is True
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
# --- ranking-filter `attribute` is optional on single-dimension visualizations (QA-28615) ---
|
|
70
|
+
#
|
|
71
|
+
# `attribute` is NotRequired in the AAC schema and AFM ranks over the whole result when it is
|
|
72
|
+
# absent, so on a one-dimension chart "omitted" and "the sole dimension" mean the same filter.
|
|
73
|
+
# The comparator used to demand an exact match and failed those as filters_correct=False.
|
|
74
|
+
|
|
75
|
+
_M = {"m_sales": {"using": "metric/net_sales"}}
|
|
76
|
+
# same URI behind two different aliases — normalization must be alias-independent
|
|
77
|
+
_ONE_DIM_A = {**_M, "d_product_id": {"using": "label/product_id"}}
|
|
78
|
+
_ONE_DIM_B = {**_M, "d_product": {"using": "label/product_id"}}
|
|
79
|
+
_TWO_DIM = {**_M, "d_brand": {"using": "label/product_brand"}, "d_city": {"using": "label/customer_city"}}
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def _rank_viz(fields, dims, **filter_overrides):
|
|
83
|
+
rank = {"type": "ranking_filter", "using": "m_sales", "top": 1, **filter_overrides}
|
|
84
|
+
return _viz(
|
|
85
|
+
type="bar_chart",
|
|
86
|
+
query={"fields": fields, "filter_by": {"f_rank": rank}},
|
|
87
|
+
metrics=["m_sales"],
|
|
88
|
+
view_by=dims,
|
|
89
|
+
)
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
def test_ranking_attribute_optional_on_single_dimension_viz():
|
|
93
|
+
"""Expected names the attribute, actual omits it — one dimension, so they are equivalent."""
|
|
94
|
+
expected = _rank_viz(_ONE_DIM_A, ["d_product_id"], attribute="d_product_id")
|
|
95
|
+
actual = _rank_viz(_ONE_DIM_B, ["d_product"])
|
|
96
|
+
scores = check_filters(expected, actual)
|
|
97
|
+
assert scores.ranking_ok is True
|
|
98
|
+
assert scores.all_ok is True
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
def test_ranking_attribute_optional_is_symmetric():
|
|
102
|
+
"""Reverse direction: the dataset omits the attribute and the agent supplies it."""
|
|
103
|
+
expected = _rank_viz(_ONE_DIM_A, ["d_product_id"])
|
|
104
|
+
actual = _rank_viz(_ONE_DIM_B, ["d_product"], attribute="d_product")
|
|
105
|
+
assert check_filters(expected, actual).ranking_ok is True
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
def test_ranking_attribute_none_and_empty_are_the_same_as_omitted():
|
|
109
|
+
expected = _rank_viz(_ONE_DIM_A, ["d_product_id"], attribute="d_product_id")
|
|
110
|
+
for omitted in ({"attribute": None}, {"attribute": ""}):
|
|
111
|
+
actual = _rank_viz(_ONE_DIM_B, ["d_product"], **omitted)
|
|
112
|
+
assert check_filters(expected, actual).ranking_ok is True, omitted
|
|
113
|
+
|
|
114
|
+
|
|
115
|
+
def test_ranking_attribute_still_required_on_multi_dimension_viz():
|
|
116
|
+
"""Two dimensions: omitting the attribute ranks over the tuple, so it stays strict."""
|
|
117
|
+
expected = _rank_viz(_TWO_DIM, ["d_brand", "d_city"], attribute="d_brand")
|
|
118
|
+
actual = _rank_viz(_TWO_DIM, ["d_brand", "d_city"])
|
|
119
|
+
assert check_filters(expected, actual).ranking_ok is False
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
def test_ranking_attribute_omitted_does_not_mask_a_wrong_top_n():
|
|
123
|
+
expected = _rank_viz(_ONE_DIM_A, ["d_product_id"], attribute="d_product_id", top=1)
|
|
124
|
+
actual = _rank_viz(_ONE_DIM_B, ["d_product"], top=5)
|
|
125
|
+
assert check_filters(expected, actual).ranking_ok is False
|
|
126
|
+
|
|
127
|
+
|
|
128
|
+
def test_ranking_attribute_omitted_does_not_mask_a_wrong_dimension():
|
|
129
|
+
expected = _rank_viz(_ONE_DIM_A, ["d_product_id"], attribute="d_product_id")
|
|
130
|
+
actual = _rank_viz(_TWO_DIM, ["d_brand"]) # single dim, but a different one
|
|
131
|
+
assert check_filters(expected, actual).ranking_ok is False
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
def test_validate_cross_references_never_raises_on_empty_or_none_uris():
|
|
135
|
+
"""Each of these used to raise AttributeError/TypeError instead of returning a score.
|
|
136
|
+
|
|
137
|
+
Every case carries its expected verdict: `attribute` is optional so None/"" are valid,
|
|
138
|
+
while a non-string attribute or a missing/None `using` must be reported as an error.
|
|
139
|
+
Asserting the verdict is what stops a malformed filter from silently passing as valid.
|
|
140
|
+
"""
|
|
141
|
+
cases = [
|
|
142
|
+
({"type": "ranking_filter", "using": "m_sales", "top": 5, "attribute": None}, True),
|
|
143
|
+
({"type": "ranking_filter", "using": "m_sales", "top": 5, "attribute": ""}, True),
|
|
144
|
+
({"type": "ranking_filter", "using": "m_sales", "top": 5, "attribute": []}, False),
|
|
145
|
+
({"type": "ranking_filter", "using": None, "top": 5}, False),
|
|
146
|
+
({"type": "ranking_filter", "top": 5}, False),
|
|
147
|
+
]
|
|
148
|
+
for rank, expected_ok in cases:
|
|
149
|
+
viz = _viz(query={"fields": _M, "filter_by": {"f_rank": rank}})
|
|
150
|
+
ok, errors = validate_cross_references(viz)
|
|
151
|
+
assert isinstance(ok, bool) and isinstance(errors, list), rank
|
|
152
|
+
assert ok is expected_ok, rank
|
|
153
|
+
assert bool(errors) is not expected_ok, rank
|
|
154
|
+
|
|
155
|
+
|
|
156
|
+
def test_validate_cross_references_accepts_omitted_attribute_but_flags_missing_using():
|
|
157
|
+
omitted = _viz(query={"fields": _M, "filter_by": {"f": {"type": "ranking_filter", "using": "m_sales", "top": 5}}})
|
|
158
|
+
assert validate_cross_references(omitted) == (True, [])
|
|
159
|
+
|
|
160
|
+
no_using = _viz(query={"fields": _M, "filter_by": {"f": {"type": "ranking_filter", "top": 5}}})
|
|
161
|
+
ok, errors = validate_cross_references(no_using)
|
|
162
|
+
assert ok is False
|
|
163
|
+
assert "is required" in errors[0]
|
|
@@ -99,3 +99,31 @@ def test_evaluator_skill_not_activated_when_wrong_skill_name():
|
|
|
99
99
|
)
|
|
100
100
|
result = ev.evaluate(_item(_expected()), chat)
|
|
101
101
|
assert result.detail["skill_activated"] is False
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
def _ranked(attribute: str | None, dim_alias: str = "d_q"):
|
|
105
|
+
"""Single-dimension chart with a top-1 ranking filter, optionally naming the attribute."""
|
|
106
|
+
rank = {"type": "ranking_filter", "using": "m_rev", "top": 1}
|
|
107
|
+
if attribute is not None:
|
|
108
|
+
rank["attribute"] = attribute
|
|
109
|
+
return {
|
|
110
|
+
"id": "x",
|
|
111
|
+
"type": "column_chart",
|
|
112
|
+
"query": {
|
|
113
|
+
"fields": {"m_rev": {"using": "metric/revenue"}, dim_alias: {"using": "label/date.quarter"}},
|
|
114
|
+
"filter_by": {"f_rank": rank},
|
|
115
|
+
},
|
|
116
|
+
"metrics": ["m_rev"],
|
|
117
|
+
"view_by": [dim_alias],
|
|
118
|
+
}
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
def test_evaluator_passes_when_agent_omits_ranking_attribute_on_single_dim_viz():
|
|
122
|
+
"""QA-28615: the omitted attribute resolves to the sole dimension, so the case must pass."""
|
|
123
|
+
ev = get_evaluator("visualization")
|
|
124
|
+
expected = _ranked("d_q")
|
|
125
|
+
actual = _ranked(None, dim_alias="d_quarter") # different alias, attribute omitted
|
|
126
|
+
result = ev.evaluate(_item(expected), _chat_result_with(actual))
|
|
127
|
+
assert result.detail["filter_ranking_score"] is True
|
|
128
|
+
assert result.detail["filters_correct"] is True
|
|
129
|
+
assert result.passed is True
|
|
@@ -1,66 +0,0 @@
|
|
|
1
|
-
# (C) 2026 GoodData Corporation
|
|
2
|
-
from gooddata_eval.core.models import CreatedVisualization
|
|
3
|
-
from gooddata_eval.core.scoring import (
|
|
4
|
-
check_filters,
|
|
5
|
-
check_viz_type,
|
|
6
|
-
get_dimension_uri_set,
|
|
7
|
-
get_metric_uri_set,
|
|
8
|
-
uri_to_display_name,
|
|
9
|
-
validate_cross_references,
|
|
10
|
-
)
|
|
11
|
-
|
|
12
|
-
|
|
13
|
-
def _viz(**kw) -> CreatedVisualization:
|
|
14
|
-
base = {"id": "v", "type": "", "query": {"fields": {}, "filter_by": {}}}
|
|
15
|
-
base.update(kw)
|
|
16
|
-
return CreatedVisualization.model_validate(base)
|
|
17
|
-
|
|
18
|
-
|
|
19
|
-
def test_metric_and_dimension_uri_sets_resolve_aliases():
|
|
20
|
-
viz = _viz(
|
|
21
|
-
query={
|
|
22
|
-
"fields": {"m_rev": {"using": "metric/revenue"}, "d_q": {"using": "label/date.quarter"}},
|
|
23
|
-
"filter_by": {},
|
|
24
|
-
},
|
|
25
|
-
metrics=["m_rev"],
|
|
26
|
-
view_by=["d_q"],
|
|
27
|
-
)
|
|
28
|
-
assert get_metric_uri_set(viz) == {"metric/revenue"}
|
|
29
|
-
assert get_dimension_uri_set(viz) == {"label/date.quarter"}
|
|
30
|
-
|
|
31
|
-
|
|
32
|
-
def test_uri_to_display_name():
|
|
33
|
-
assert uri_to_display_name("metric/net_sales") == "net sales"
|
|
34
|
-
assert uri_to_display_name("label/date.month") == "date - month"
|
|
35
|
-
|
|
36
|
-
|
|
37
|
-
def test_validate_cross_references_flags_bad_ranking_using():
|
|
38
|
-
viz = _viz(
|
|
39
|
-
query={
|
|
40
|
-
"fields": {"d_q": {"using": "label/date.quarter"}},
|
|
41
|
-
"filter_by": {"f_rank": {"type": "ranking_filter", "top": 5, "using": "d_q"}},
|
|
42
|
-
}
|
|
43
|
-
)
|
|
44
|
-
ok, errors = validate_cross_references(viz)
|
|
45
|
-
assert ok is False
|
|
46
|
-
assert errors and "ranking filter" in errors[0]
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
def test_check_viz_type_empty_expected_is_wildcard():
|
|
50
|
-
expected = _viz(type="")
|
|
51
|
-
actual = _viz(type="column_chart")
|
|
52
|
-
assert check_viz_type(expected, actual) is True
|
|
53
|
-
|
|
54
|
-
|
|
55
|
-
def test_check_viz_type_strict_match_normalizes():
|
|
56
|
-
expected = _viz(type="column_chart")
|
|
57
|
-
actual = _viz(type="COLUMN")
|
|
58
|
-
assert check_viz_type(expected, actual) is True
|
|
59
|
-
|
|
60
|
-
|
|
61
|
-
def test_check_filters_exact_attribute_match():
|
|
62
|
-
f = {"f_a": {"type": "attribute_filter", "using": "label/region", "state": {"include": ["EMEA"]}}}
|
|
63
|
-
expected = _viz(query={"fields": {}, "filter_by": f})
|
|
64
|
-
actual = _viz(query={"fields": {}, "filter_by": f})
|
|
65
|
-
scores = check_filters(expected, actual)
|
|
66
|
-
assert scores.all_ok is True
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/src/gooddata_eval/core/agentic/__init__.py
RENAMED
|
File without changes
|
{gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/src/gooddata_eval/core/agentic/_catalog.py
RENAMED
|
File without changes
|
{gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/src/gooddata_eval/core/agentic/_langfuse.py
RENAMED
|
File without changes
|
{gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/src/gooddata_eval/core/agentic/alert_skill.py
RENAMED
|
File without changes
|
{gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/src/gooddata_eval/core/agentic/conversation.py
RENAMED
|
File without changes
|
|
File without changes
|
{gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/src/gooddata_eval/core/agentic/guardrail.py
RENAMED
|
File without changes
|
{gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/src/gooddata_eval/core/agentic/metric_skill.py
RENAMED
|
File without changes
|
{gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/src/gooddata_eval/core/agentic/search_tool.py
RENAMED
|
File without changes
|
{gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/src/gooddata_eval/core/agentic/visualization.py
RENAMED
|
File without changes
|
|
File without changes
|
{gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/src/gooddata_eval/core/chat/sse_client.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
{gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/src/gooddata_eval/core/dataset/__init__.py
RENAMED
|
File without changes
|
{gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/src/gooddata_eval/core/dataset/langfuse_source.py
RENAMED
|
File without changes
|
|
File without changes
|
{gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/src/gooddata_eval/core/evaluators/__init__.py
RENAMED
|
File without changes
|
{gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/src/gooddata_eval/core/evaluators/_deep_subset.py
RENAMED
|
File without changes
|
{gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/src/gooddata_eval/core/evaluators/_llm_judge.py
RENAMED
|
File without changes
|
{gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/src/gooddata_eval/core/evaluators/_text_utils.py
RENAMED
|
File without changes
|
{gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/src/gooddata_eval/core/evaluators/alert_skill.py
RENAMED
|
File without changes
|
{gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/src/gooddata_eval/core/evaluators/base.py
RENAMED
|
File without changes
|
|
File without changes
|
{gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/src/gooddata_eval/core/evaluators/guardrail.py
RENAMED
|
File without changes
|
{gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/src/gooddata_eval/core/evaluators/metric_skill.py
RENAMED
|
File without changes
|
{gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/src/gooddata_eval/core/evaluators/search_tool.py
RENAMED
|
File without changes
|
{gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/src/gooddata_eval/core/evaluators/summary.py
RENAMED
|
File without changes
|
|
File without changes
|
{gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/src/gooddata_eval/core/langfuse/__init__.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
{gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/src/gooddata_eval/core/reporting/__init__.py
RENAMED
|
File without changes
|
{gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/src/gooddata_eval/core/reporting/console.py
RENAMED
|
File without changes
|
{gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/src/gooddata_eval/core/reporting/json_report.py
RENAMED
|
File without changes
|
|
File without changes
|
{gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/src/gooddata_eval/core/summary/__init__.py
RENAMED
|
File without changes
|
{gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/src/gooddata_eval/core/summary/http_client.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{gooddata_eval-1.71.1.dev2 → gooddata_eval-1.72.0}/tests/fixtures/sse_visualization_stream.txt
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|