gooddata-eval 1.71.1.dev2__tar.gz → 1.71.1.dev3__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (87) hide show
  1. {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.71.1.dev3}/PKG-INFO +2 -2
  2. {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.71.1.dev3}/pyproject.toml +2 -2
  3. {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.71.1.dev3}/src/gooddata_eval/core/scoring.py +70 -20
  4. gooddata_eval-1.71.1.dev3/tests/test_scoring.py +163 -0
  5. {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.71.1.dev3}/tests/test_visualization_evaluator.py +28 -0
  6. gooddata_eval-1.71.1.dev2/tests/test_scoring.py +0 -66
  7. {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.71.1.dev3}/.gitignore +0 -0
  8. {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.71.1.dev3}/LICENSE.txt +0 -0
  9. {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.71.1.dev3}/Makefile +0 -0
  10. {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.71.1.dev3}/README.md +0 -0
  11. {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.71.1.dev3}/src/gooddata_eval/__init__.py +0 -0
  12. {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.71.1.dev3}/src/gooddata_eval/_version.py +0 -0
  13. {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.71.1.dev3}/src/gooddata_eval/cli/__init__.py +0 -0
  14. {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.71.1.dev3}/src/gooddata_eval/cli/agentic_runner.py +0 -0
  15. {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.71.1.dev3}/src/gooddata_eval/cli/main.py +0 -0
  16. {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.71.1.dev3}/src/gooddata_eval/core/__init__.py +0 -0
  17. {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.71.1.dev3}/src/gooddata_eval/core/agentic/__init__.py +0 -0
  18. {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.71.1.dev3}/src/gooddata_eval/core/agentic/_catalog.py +0 -0
  19. {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.71.1.dev3}/src/gooddata_eval/core/agentic/_langfuse.py +0 -0
  20. {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.71.1.dev3}/src/gooddata_eval/core/agentic/alert_skill.py +0 -0
  21. {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.71.1.dev3}/src/gooddata_eval/core/agentic/conversation.py +0 -0
  22. {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.71.1.dev3}/src/gooddata_eval/core/agentic/general_question.py +0 -0
  23. {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.71.1.dev3}/src/gooddata_eval/core/agentic/guardrail.py +0 -0
  24. {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.71.1.dev3}/src/gooddata_eval/core/agentic/metric_skill.py +0 -0
  25. {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.71.1.dev3}/src/gooddata_eval/core/agentic/search_tool.py +0 -0
  26. {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.71.1.dev3}/src/gooddata_eval/core/agentic/visualization.py +0 -0
  27. {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.71.1.dev3}/src/gooddata_eval/core/chat/__init__.py +0 -0
  28. {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.71.1.dev3}/src/gooddata_eval/core/chat/sse_client.py +0 -0
  29. {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.71.1.dev3}/src/gooddata_eval/core/config.py +0 -0
  30. {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.71.1.dev3}/src/gooddata_eval/core/connection.py +0 -0
  31. {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.71.1.dev3}/src/gooddata_eval/core/dataset/__init__.py +0 -0
  32. {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.71.1.dev3}/src/gooddata_eval/core/dataset/langfuse_source.py +0 -0
  33. {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.71.1.dev3}/src/gooddata_eval/core/dataset/local.py +0 -0
  34. {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.71.1.dev3}/src/gooddata_eval/core/evaluators/__init__.py +0 -0
  35. {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.71.1.dev3}/src/gooddata_eval/core/evaluators/_deep_subset.py +0 -0
  36. {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.71.1.dev3}/src/gooddata_eval/core/evaluators/_llm_judge.py +0 -0
  37. {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.71.1.dev3}/src/gooddata_eval/core/evaluators/_text_utils.py +0 -0
  38. {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.71.1.dev3}/src/gooddata_eval/core/evaluators/alert_skill.py +0 -0
  39. {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.71.1.dev3}/src/gooddata_eval/core/evaluators/base.py +0 -0
  40. {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.71.1.dev3}/src/gooddata_eval/core/evaluators/general_question.py +0 -0
  41. {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.71.1.dev3}/src/gooddata_eval/core/evaluators/guardrail.py +0 -0
  42. {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.71.1.dev3}/src/gooddata_eval/core/evaluators/metric_skill.py +0 -0
  43. {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.71.1.dev3}/src/gooddata_eval/core/evaluators/search_tool.py +0 -0
  44. {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.71.1.dev3}/src/gooddata_eval/core/evaluators/summary.py +0 -0
  45. {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.71.1.dev3}/src/gooddata_eval/core/evaluators/visualization.py +0 -0
  46. {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.71.1.dev3}/src/gooddata_eval/core/langfuse/__init__.py +0 -0
  47. {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.71.1.dev3}/src/gooddata_eval/core/langfuse/sink.py +0 -0
  48. {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.71.1.dev3}/src/gooddata_eval/core/models.py +0 -0
  49. {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.71.1.dev3}/src/gooddata_eval/core/reporting/__init__.py +0 -0
  50. {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.71.1.dev3}/src/gooddata_eval/core/reporting/console.py +0 -0
  51. {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.71.1.dev3}/src/gooddata_eval/core/reporting/json_report.py +0 -0
  52. {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.71.1.dev3}/src/gooddata_eval/core/runner.py +0 -0
  53. {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.71.1.dev3}/src/gooddata_eval/core/summary/__init__.py +0 -0
  54. {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.71.1.dev3}/src/gooddata_eval/core/summary/http_client.py +0 -0
  55. {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.71.1.dev3}/src/gooddata_eval/core/workspace.py +0 -0
  56. {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.71.1.dev3}/tests/__init__.py +0 -0
  57. {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.71.1.dev3}/tests/conftest.py +0 -0
  58. {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.71.1.dev3}/tests/fixtures/sample_dataset/metric_skill_create.json +0 -0
  59. {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.71.1.dev3}/tests/fixtures/sample_dataset/visualization_revenue.json +0 -0
  60. {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.71.1.dev3}/tests/fixtures/sse_visualization_stream.txt +0 -0
  61. {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.71.1.dev3}/tests/test_agentic_alert_skill.py +0 -0
  62. {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.71.1.dev3}/tests/test_agentic_conversation.py +0 -0
  63. {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.71.1.dev3}/tests/test_agentic_general_question.py +0 -0
  64. {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.71.1.dev3}/tests/test_agentic_guardrail.py +0 -0
  65. {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.71.1.dev3}/tests/test_agentic_metric_skill.py +0 -0
  66. {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.71.1.dev3}/tests/test_agentic_run_context.py +0 -0
  67. {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.71.1.dev3}/tests/test_agentic_search_tool.py +0 -0
  68. {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.71.1.dev3}/tests/test_agentic_visualization.py +0 -0
  69. {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.71.1.dev3}/tests/test_alert_skill_evaluator.py +0 -0
  70. {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.71.1.dev3}/tests/test_cli.py +0 -0
  71. {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.71.1.dev3}/tests/test_connection.py +0 -0
  72. {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.71.1.dev3}/tests/test_deep_subset.py +0 -0
  73. {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.71.1.dev3}/tests/test_langfuse_sink.py +0 -0
  74. {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.71.1.dev3}/tests/test_langfuse_source.py +0 -0
  75. {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.71.1.dev3}/tests/test_llm_judge.py +0 -0
  76. {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.71.1.dev3}/tests/test_local_loader.py +0 -0
  77. {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.71.1.dev3}/tests/test_metric_skill_evaluator.py +0 -0
  78. {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.71.1.dev3}/tests/test_models.py +0 -0
  79. {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.71.1.dev3}/tests/test_reporting.py +0 -0
  80. {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.71.1.dev3}/tests/test_runner.py +0 -0
  81. {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.71.1.dev3}/tests/test_search_tool_evaluator.py +0 -0
  82. {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.71.1.dev3}/tests/test_sse_client.py +0 -0
  83. {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.71.1.dev3}/tests/test_summary_client.py +0 -0
  84. {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.71.1.dev3}/tests/test_summary_evaluator.py +0 -0
  85. {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.71.1.dev3}/tests/test_text_evaluators.py +0 -0
  86. {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.71.1.dev3}/tests/test_workspace.py +0 -0
  87. {gooddata_eval-1.71.1.dev2 → gooddata_eval-1.71.1.dev3}/tox.ini +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: gooddata-eval
3
- Version: 1.71.1.dev2
3
+ Version: 1.71.1.dev3
4
4
  Summary: Evaluate the GoodData AI agent against your own questions and models.
5
5
  Project-URL: Source, https://github.com/gooddata/gooddata-python-sdk
6
6
  Author-email: GoodData <support@gooddata.com>
@@ -17,7 +17,7 @@ Classifier: Topic :: Scientific/Engineering
17
17
  Classifier: Topic :: Software Development
18
18
  Classifier: Typing :: Typed
19
19
  Requires-Python: >=3.10
20
- Requires-Dist: gooddata-sdk~=1.71.1.dev2
20
+ Requires-Dist: gooddata-sdk~=1.71.1.dev3
21
21
  Requires-Dist: httpx<1.0,>=0.27
22
22
  Requires-Dist: orjson<4.0.0,>=3.9.15
23
23
  Requires-Dist: pydantic<3.0,>=2.6
@@ -1,7 +1,7 @@
1
1
  # (C) 2026 GoodData Corporation
2
2
  [project]
3
3
  name = "gooddata-eval"
4
- version = "1.71.1.dev2"
4
+ version = "1.71.1.dev3"
5
5
  description = "Evaluate the GoodData AI agent against your own questions and models."
6
6
  readme = "README.md"
7
7
  license = "MIT"
@@ -11,7 +11,7 @@ authors = [
11
11
  keywords = ["gooddata", "ai", "evaluation", "llm", "analytics", "cli"]
12
12
  requires-python = ">=3.10"
13
13
  dependencies = [
14
- "gooddata-sdk~=1.71.1.dev2",
14
+ "gooddata-sdk~=1.71.1.dev3",
15
15
  "httpx>=0.27,<1.0",
16
16
  "orjson>=3.9.15,<4.0.0",
17
17
  "pydantic>=2.6,<3.0",
@@ -63,29 +63,46 @@ def uri_to_display_name(uri: str) -> str:
63
63
 
64
64
 
65
65
  def validate_cross_references(viz: CreatedVisualization) -> tuple[bool, list[str]]:
66
- """Validate ranking-filter `using`/`attribute` resolve to correct URI prefixes."""
66
+ """Validate ranking-filter `using`/`attribute` resolve to correct URI prefixes.
67
+
68
+ Always returns `(ok, errors)` — a malformed filter produces an error entry, never an
69
+ exception. Anything unusable (None, empty, non-string) used to reach `.startswith()`
70
+ or `dict.get()` and blow up with AttributeError/TypeError mid-evaluation.
71
+
72
+ `using` is required by the AAC schema, `attribute` is optional (see
73
+ `_normalize_ranking_filter`), so an absent/None/empty `attribute` is accepted silently.
74
+ """
67
75
  errors: list[str] = []
68
76
  fields = viz.query.fields
69
77
  for filter_key, filter_dict in viz.query.filter_by.items():
70
78
  if filter_dict.get("type") != "ranking_filter":
71
79
  continue
72
- using_val = filter_dict.get("using", "")
73
- using_uri = _resolve_alias_to_uri(using_val, fields)
74
- field_def = fields.get(using_val)
75
- is_adhoc_agg = isinstance(field_def, AacQueryField) and bool(field_def.aggregation)
76
- if not using_uri.startswith(("metric/", "fact/")) and not is_adhoc_agg:
77
- errors.append(
78
- f"ranking filter '{filter_key}': using='{using_val}' "
79
- f"resolves to '{using_uri}' — expected a metric/ or fact/ URI"
80
- )
81
- if "attribute" in filter_dict:
82
- attr_val = filter_dict["attribute"]
83
- attr_uri = _resolve_alias_to_uri(attr_val, fields)
84
- if not attr_uri.startswith(("label/", "attribute/")):
80
+ using_val = filter_dict.get("using")
81
+ if not isinstance(using_val, str) or not using_val:
82
+ errors.append(f"ranking filter '{filter_key}': using={using_val!r} — a metric/ or fact/ URI is required")
83
+ else:
84
+ using_uri = _resolve_alias_to_uri(using_val, fields)
85
+ field_def = fields.get(using_val)
86
+ is_adhoc_agg = isinstance(field_def, AacQueryField) and bool(field_def.aggregation)
87
+ if not using_uri.startswith(("metric/", "fact/")) and not is_adhoc_agg:
85
88
  errors.append(
86
- f"ranking filter '{filter_key}': attribute='{attr_val}' "
87
- f"resolves to '{attr_uri}' — expected a label/ or attribute/ URI"
89
+ f"ranking filter '{filter_key}': using='{using_val}' "
90
+ f"resolves to '{using_uri}' — expected a metric/ or fact/ URI"
88
91
  )
92
+ attr_val = filter_dict.get("attribute")
93
+ if attr_val is None or attr_val == "":
94
+ continue
95
+ if not isinstance(attr_val, str):
96
+ errors.append(
97
+ f"ranking filter '{filter_key}': attribute={attr_val!r} — expected a label/ or attribute/ URI"
98
+ )
99
+ continue
100
+ attr_uri = _resolve_alias_to_uri(attr_val, fields)
101
+ if not attr_uri.startswith(("label/", "attribute/")):
102
+ errors.append(
103
+ f"ranking filter '{filter_key}': attribute='{attr_val}' "
104
+ f"resolves to '{attr_uri}' — expected a label/ or attribute/ URI"
105
+ )
89
106
  return len(errors) == 0, errors
90
107
 
91
108
 
@@ -99,11 +116,43 @@ def _normalize_date_filter(filter_dict: dict, _fields: dict) -> dict:
99
116
  }
100
117
 
101
118
 
102
- def _normalize_ranking_filter(filter_dict: dict, fields: dict[str, AacQueryField | str]) -> dict:
119
+ def _sole_dimension_uri(viz: CreatedVisualization) -> str | None:
120
+ """URI of the visualization's only dimension, or None when it has zero or several."""
121
+ dim_uris = get_dimension_uri_set(viz)
122
+ return next(iter(dim_uris)) if len(dim_uris) == 1 else None
123
+
124
+
125
+ def _normalize_ranking_filter(
126
+ filter_dict: dict,
127
+ fields: dict[str, AacQueryField | str],
128
+ sole_dim_uri: str | None = None,
129
+ ) -> dict:
130
+ """Canonicalize a ranking filter so equivalent filters compare equal.
131
+
132
+ `attribute` is optional in the AAC schema (gen-ai models it as `NotRequired[str]` /
133
+ `str | None`), and when it is omitted AFM ranks over every dimension of the result. For a
134
+ single-dimension visualization that is exactly "rank by that one dimension", so an omitted
135
+ attribute is filled in with `sole_dim_uri` instead of comparing as an empty string — the
136
+ agent and the dataset may legitimately express the same filter either way.
137
+
138
+ The substitution is deliberately gated on there being exactly ONE dimension: with two or
139
+ more, omitting `attribute` ranks over the dimension *tuple*, which is a different filter,
140
+ so those stay strict. Callers pass the sole dimension of the visualization the filter
141
+ belongs to, which makes the comparison symmetric — it does not matter which side omitted it.
142
+
143
+ Missing, None and "" are all treated as "not specified"; so is a non-string, which
144
+ `validate_cross_references` reports separately rather than crashing the comparison.
145
+ """
146
+ attr_val = filter_dict.get("attribute")
147
+ if not isinstance(attr_val, str) or not attr_val:
148
+ dim_uri = sole_dim_uri or ""
149
+ else:
150
+ dim_uri = _resolve_alias_to_uri(attr_val, fields)
151
+ using_val = filter_dict.get("using")
103
152
  entry: dict = {
104
153
  "type": "ranking_filter",
105
- "metric_uri": _resolve_alias_to_uri(filter_dict.get("using", ""), fields),
106
- "dim_uri": _resolve_alias_to_uri(filter_dict.get("attribute", ""), fields),
154
+ "metric_uri": _resolve_alias_to_uri(using_val, fields) if isinstance(using_val, str) else "",
155
+ "dim_uri": dim_uri,
107
156
  }
108
157
  if "top" in filter_dict:
109
158
  entry["top"] = filter_dict["top"]
@@ -127,12 +176,13 @@ def _split_and_normalize_filters(viz: CreatedVisualization) -> tuple[set[str], s
127
176
  ranking_set: set[str] = set()
128
177
  attr_set: set[str] = set()
129
178
  fields = viz.query.fields
179
+ sole_dim_uri = _sole_dimension_uri(viz)
130
180
  for filter_dict in viz.query.filter_by.values():
131
181
  ft = filter_dict.get("type")
132
182
  if ft == "date_filter":
133
183
  date_set.add(json.dumps(_normalize_date_filter(filter_dict, fields), sort_keys=True))
134
184
  elif ft == "ranking_filter":
135
- ranking_set.add(json.dumps(_normalize_ranking_filter(filter_dict, fields), sort_keys=True))
185
+ ranking_set.add(json.dumps(_normalize_ranking_filter(filter_dict, fields, sole_dim_uri), sort_keys=True))
136
186
  elif ft == "attribute_filter":
137
187
  attr_set.add(json.dumps(_normalize_attribute_filter(filter_dict, fields), sort_keys=True))
138
188
  return date_set, ranking_set, attr_set
@@ -0,0 +1,163 @@
1
+ # (C) 2026 GoodData Corporation
2
+ from gooddata_eval.core.models import CreatedVisualization
3
+ from gooddata_eval.core.scoring import (
4
+ check_filters,
5
+ check_viz_type,
6
+ get_dimension_uri_set,
7
+ get_metric_uri_set,
8
+ uri_to_display_name,
9
+ validate_cross_references,
10
+ )
11
+
12
+
13
+ def _viz(**kw) -> CreatedVisualization:
14
+ base = {"id": "v", "type": "", "query": {"fields": {}, "filter_by": {}}}
15
+ base.update(kw)
16
+ return CreatedVisualization.model_validate(base)
17
+
18
+
19
+ def test_metric_and_dimension_uri_sets_resolve_aliases():
20
+ viz = _viz(
21
+ query={
22
+ "fields": {"m_rev": {"using": "metric/revenue"}, "d_q": {"using": "label/date.quarter"}},
23
+ "filter_by": {},
24
+ },
25
+ metrics=["m_rev"],
26
+ view_by=["d_q"],
27
+ )
28
+ assert get_metric_uri_set(viz) == {"metric/revenue"}
29
+ assert get_dimension_uri_set(viz) == {"label/date.quarter"}
30
+
31
+
32
+ def test_uri_to_display_name():
33
+ assert uri_to_display_name("metric/net_sales") == "net sales"
34
+ assert uri_to_display_name("label/date.month") == "date - month"
35
+
36
+
37
+ def test_validate_cross_references_flags_bad_ranking_using():
38
+ viz = _viz(
39
+ query={
40
+ "fields": {"d_q": {"using": "label/date.quarter"}},
41
+ "filter_by": {"f_rank": {"type": "ranking_filter", "top": 5, "using": "d_q"}},
42
+ }
43
+ )
44
+ ok, errors = validate_cross_references(viz)
45
+ assert ok is False
46
+ assert errors and "ranking filter" in errors[0]
47
+
48
+
49
+ def test_check_viz_type_empty_expected_is_wildcard():
50
+ expected = _viz(type="")
51
+ actual = _viz(type="column_chart")
52
+ assert check_viz_type(expected, actual) is True
53
+
54
+
55
+ def test_check_viz_type_strict_match_normalizes():
56
+ expected = _viz(type="column_chart")
57
+ actual = _viz(type="COLUMN")
58
+ assert check_viz_type(expected, actual) is True
59
+
60
+
61
+ def test_check_filters_exact_attribute_match():
62
+ f = {"f_a": {"type": "attribute_filter", "using": "label/region", "state": {"include": ["EMEA"]}}}
63
+ expected = _viz(query={"fields": {}, "filter_by": f})
64
+ actual = _viz(query={"fields": {}, "filter_by": f})
65
+ scores = check_filters(expected, actual)
66
+ assert scores.all_ok is True
67
+
68
+
69
+ # --- ranking-filter `attribute` is optional on single-dimension visualizations (QA-28615) ---
70
+ #
71
+ # `attribute` is NotRequired in the AAC schema and AFM ranks over the whole result when it is
72
+ # absent, so on a one-dimension chart "omitted" and "the sole dimension" mean the same filter.
73
+ # The comparator used to demand an exact match and failed those as filters_correct=False.
74
+
75
+ _M = {"m_sales": {"using": "metric/net_sales"}}
76
+ # same URI behind two different aliases — normalization must be alias-independent
77
+ _ONE_DIM_A = {**_M, "d_product_id": {"using": "label/product_id"}}
78
+ _ONE_DIM_B = {**_M, "d_product": {"using": "label/product_id"}}
79
+ _TWO_DIM = {**_M, "d_brand": {"using": "label/product_brand"}, "d_city": {"using": "label/customer_city"}}
80
+
81
+
82
+ def _rank_viz(fields, dims, **filter_overrides):
83
+ rank = {"type": "ranking_filter", "using": "m_sales", "top": 1, **filter_overrides}
84
+ return _viz(
85
+ type="bar_chart",
86
+ query={"fields": fields, "filter_by": {"f_rank": rank}},
87
+ metrics=["m_sales"],
88
+ view_by=dims,
89
+ )
90
+
91
+
92
+ def test_ranking_attribute_optional_on_single_dimension_viz():
93
+ """Expected names the attribute, actual omits it — one dimension, so they are equivalent."""
94
+ expected = _rank_viz(_ONE_DIM_A, ["d_product_id"], attribute="d_product_id")
95
+ actual = _rank_viz(_ONE_DIM_B, ["d_product"])
96
+ scores = check_filters(expected, actual)
97
+ assert scores.ranking_ok is True
98
+ assert scores.all_ok is True
99
+
100
+
101
+ def test_ranking_attribute_optional_is_symmetric():
102
+ """Reverse direction: the dataset omits the attribute and the agent supplies it."""
103
+ expected = _rank_viz(_ONE_DIM_A, ["d_product_id"])
104
+ actual = _rank_viz(_ONE_DIM_B, ["d_product"], attribute="d_product")
105
+ assert check_filters(expected, actual).ranking_ok is True
106
+
107
+
108
+ def test_ranking_attribute_none_and_empty_are_the_same_as_omitted():
109
+ expected = _rank_viz(_ONE_DIM_A, ["d_product_id"], attribute="d_product_id")
110
+ for omitted in ({"attribute": None}, {"attribute": ""}):
111
+ actual = _rank_viz(_ONE_DIM_B, ["d_product"], **omitted)
112
+ assert check_filters(expected, actual).ranking_ok is True, omitted
113
+
114
+
115
+ def test_ranking_attribute_still_required_on_multi_dimension_viz():
116
+ """Two dimensions: omitting the attribute ranks over the tuple, so it stays strict."""
117
+ expected = _rank_viz(_TWO_DIM, ["d_brand", "d_city"], attribute="d_brand")
118
+ actual = _rank_viz(_TWO_DIM, ["d_brand", "d_city"])
119
+ assert check_filters(expected, actual).ranking_ok is False
120
+
121
+
122
+ def test_ranking_attribute_omitted_does_not_mask_a_wrong_top_n():
123
+ expected = _rank_viz(_ONE_DIM_A, ["d_product_id"], attribute="d_product_id", top=1)
124
+ actual = _rank_viz(_ONE_DIM_B, ["d_product"], top=5)
125
+ assert check_filters(expected, actual).ranking_ok is False
126
+
127
+
128
+ def test_ranking_attribute_omitted_does_not_mask_a_wrong_dimension():
129
+ expected = _rank_viz(_ONE_DIM_A, ["d_product_id"], attribute="d_product_id")
130
+ actual = _rank_viz(_TWO_DIM, ["d_brand"]) # single dim, but a different one
131
+ assert check_filters(expected, actual).ranking_ok is False
132
+
133
+
134
+ def test_validate_cross_references_never_raises_on_empty_or_none_uris():
135
+ """Each of these used to raise AttributeError/TypeError instead of returning a score.
136
+
137
+ Every case carries its expected verdict: `attribute` is optional so None/"" are valid,
138
+ while a non-string attribute or a missing/None `using` must be reported as an error.
139
+ Asserting the verdict is what stops a malformed filter from silently passing as valid.
140
+ """
141
+ cases = [
142
+ ({"type": "ranking_filter", "using": "m_sales", "top": 5, "attribute": None}, True),
143
+ ({"type": "ranking_filter", "using": "m_sales", "top": 5, "attribute": ""}, True),
144
+ ({"type": "ranking_filter", "using": "m_sales", "top": 5, "attribute": []}, False),
145
+ ({"type": "ranking_filter", "using": None, "top": 5}, False),
146
+ ({"type": "ranking_filter", "top": 5}, False),
147
+ ]
148
+ for rank, expected_ok in cases:
149
+ viz = _viz(query={"fields": _M, "filter_by": {"f_rank": rank}})
150
+ ok, errors = validate_cross_references(viz)
151
+ assert isinstance(ok, bool) and isinstance(errors, list), rank
152
+ assert ok is expected_ok, rank
153
+ assert bool(errors) is not expected_ok, rank
154
+
155
+
156
+ def test_validate_cross_references_accepts_omitted_attribute_but_flags_missing_using():
157
+ omitted = _viz(query={"fields": _M, "filter_by": {"f": {"type": "ranking_filter", "using": "m_sales", "top": 5}}})
158
+ assert validate_cross_references(omitted) == (True, [])
159
+
160
+ no_using = _viz(query={"fields": _M, "filter_by": {"f": {"type": "ranking_filter", "top": 5}}})
161
+ ok, errors = validate_cross_references(no_using)
162
+ assert ok is False
163
+ assert "is required" in errors[0]
@@ -99,3 +99,31 @@ def test_evaluator_skill_not_activated_when_wrong_skill_name():
99
99
  )
100
100
  result = ev.evaluate(_item(_expected()), chat)
101
101
  assert result.detail["skill_activated"] is False
102
+
103
+
104
+ def _ranked(attribute: str | None, dim_alias: str = "d_q"):
105
+ """Single-dimension chart with a top-1 ranking filter, optionally naming the attribute."""
106
+ rank = {"type": "ranking_filter", "using": "m_rev", "top": 1}
107
+ if attribute is not None:
108
+ rank["attribute"] = attribute
109
+ return {
110
+ "id": "x",
111
+ "type": "column_chart",
112
+ "query": {
113
+ "fields": {"m_rev": {"using": "metric/revenue"}, dim_alias: {"using": "label/date.quarter"}},
114
+ "filter_by": {"f_rank": rank},
115
+ },
116
+ "metrics": ["m_rev"],
117
+ "view_by": [dim_alias],
118
+ }
119
+
120
+
121
+ def test_evaluator_passes_when_agent_omits_ranking_attribute_on_single_dim_viz():
122
+ """QA-28615: the omitted attribute resolves to the sole dimension, so the case must pass."""
123
+ ev = get_evaluator("visualization")
124
+ expected = _ranked("d_q")
125
+ actual = _ranked(None, dim_alias="d_quarter") # different alias, attribute omitted
126
+ result = ev.evaluate(_item(expected), _chat_result_with(actual))
127
+ assert result.detail["filter_ranking_score"] is True
128
+ assert result.detail["filters_correct"] is True
129
+ assert result.passed is True
@@ -1,66 +0,0 @@
1
- # (C) 2026 GoodData Corporation
2
- from gooddata_eval.core.models import CreatedVisualization
3
- from gooddata_eval.core.scoring import (
4
- check_filters,
5
- check_viz_type,
6
- get_dimension_uri_set,
7
- get_metric_uri_set,
8
- uri_to_display_name,
9
- validate_cross_references,
10
- )
11
-
12
-
13
- def _viz(**kw) -> CreatedVisualization:
14
- base = {"id": "v", "type": "", "query": {"fields": {}, "filter_by": {}}}
15
- base.update(kw)
16
- return CreatedVisualization.model_validate(base)
17
-
18
-
19
- def test_metric_and_dimension_uri_sets_resolve_aliases():
20
- viz = _viz(
21
- query={
22
- "fields": {"m_rev": {"using": "metric/revenue"}, "d_q": {"using": "label/date.quarter"}},
23
- "filter_by": {},
24
- },
25
- metrics=["m_rev"],
26
- view_by=["d_q"],
27
- )
28
- assert get_metric_uri_set(viz) == {"metric/revenue"}
29
- assert get_dimension_uri_set(viz) == {"label/date.quarter"}
30
-
31
-
32
- def test_uri_to_display_name():
33
- assert uri_to_display_name("metric/net_sales") == "net sales"
34
- assert uri_to_display_name("label/date.month") == "date - month"
35
-
36
-
37
- def test_validate_cross_references_flags_bad_ranking_using():
38
- viz = _viz(
39
- query={
40
- "fields": {"d_q": {"using": "label/date.quarter"}},
41
- "filter_by": {"f_rank": {"type": "ranking_filter", "top": 5, "using": "d_q"}},
42
- }
43
- )
44
- ok, errors = validate_cross_references(viz)
45
- assert ok is False
46
- assert errors and "ranking filter" in errors[0]
47
-
48
-
49
- def test_check_viz_type_empty_expected_is_wildcard():
50
- expected = _viz(type="")
51
- actual = _viz(type="column_chart")
52
- assert check_viz_type(expected, actual) is True
53
-
54
-
55
- def test_check_viz_type_strict_match_normalizes():
56
- expected = _viz(type="column_chart")
57
- actual = _viz(type="COLUMN")
58
- assert check_viz_type(expected, actual) is True
59
-
60
-
61
- def test_check_filters_exact_attribute_match():
62
- f = {"f_a": {"type": "attribute_filter", "using": "label/region", "state": {"include": ["EMEA"]}}}
63
- expected = _viz(query={"fields": {}, "filter_by": f})
64
- actual = _viz(query={"fields": {}, "filter_by": f})
65
- scores = check_filters(expected, actual)
66
- assert scores.all_ok is True