gooddata-eval 1.70.1.dev1__tar.gz → 1.71.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (86) hide show
  1. {gooddata_eval-1.70.1.dev1 → gooddata_eval-1.71.0}/PKG-INFO +2 -2
  2. {gooddata_eval-1.70.1.dev1 → gooddata_eval-1.71.0}/pyproject.toml +2 -2
  3. {gooddata_eval-1.70.1.dev1 → gooddata_eval-1.71.0}/src/gooddata_eval/core/evaluators/alert_skill.py +20 -1
  4. {gooddata_eval-1.70.1.dev1 → gooddata_eval-1.71.0}/src/gooddata_eval/core/evaluators/metric_skill.py +6 -1
  5. {gooddata_eval-1.70.1.dev1 → gooddata_eval-1.71.0}/tests/test_alert_skill_evaluator.py +43 -2
  6. {gooddata_eval-1.70.1.dev1 → gooddata_eval-1.71.0}/tests/test_metric_skill_evaluator.py +2 -0
  7. {gooddata_eval-1.70.1.dev1 → gooddata_eval-1.71.0}/.gitignore +0 -0
  8. {gooddata_eval-1.70.1.dev1 → gooddata_eval-1.71.0}/LICENSE.txt +0 -0
  9. {gooddata_eval-1.70.1.dev1 → gooddata_eval-1.71.0}/Makefile +0 -0
  10. {gooddata_eval-1.70.1.dev1 → gooddata_eval-1.71.0}/README.md +0 -0
  11. {gooddata_eval-1.70.1.dev1 → gooddata_eval-1.71.0}/src/gooddata_eval/__init__.py +0 -0
  12. {gooddata_eval-1.70.1.dev1 → gooddata_eval-1.71.0}/src/gooddata_eval/_version.py +0 -0
  13. {gooddata_eval-1.70.1.dev1 → gooddata_eval-1.71.0}/src/gooddata_eval/cli/__init__.py +0 -0
  14. {gooddata_eval-1.70.1.dev1 → gooddata_eval-1.71.0}/src/gooddata_eval/cli/agentic_runner.py +0 -0
  15. {gooddata_eval-1.70.1.dev1 → gooddata_eval-1.71.0}/src/gooddata_eval/cli/main.py +0 -0
  16. {gooddata_eval-1.70.1.dev1 → gooddata_eval-1.71.0}/src/gooddata_eval/core/__init__.py +0 -0
  17. {gooddata_eval-1.70.1.dev1 → gooddata_eval-1.71.0}/src/gooddata_eval/core/agentic/__init__.py +0 -0
  18. {gooddata_eval-1.70.1.dev1 → gooddata_eval-1.71.0}/src/gooddata_eval/core/agentic/_catalog.py +0 -0
  19. {gooddata_eval-1.70.1.dev1 → gooddata_eval-1.71.0}/src/gooddata_eval/core/agentic/_langfuse.py +0 -0
  20. {gooddata_eval-1.70.1.dev1 → gooddata_eval-1.71.0}/src/gooddata_eval/core/agentic/alert_skill.py +0 -0
  21. {gooddata_eval-1.70.1.dev1 → gooddata_eval-1.71.0}/src/gooddata_eval/core/agentic/conversation.py +0 -0
  22. {gooddata_eval-1.70.1.dev1 → gooddata_eval-1.71.0}/src/gooddata_eval/core/agentic/general_question.py +0 -0
  23. {gooddata_eval-1.70.1.dev1 → gooddata_eval-1.71.0}/src/gooddata_eval/core/agentic/guardrail.py +0 -0
  24. {gooddata_eval-1.70.1.dev1 → gooddata_eval-1.71.0}/src/gooddata_eval/core/agentic/metric_skill.py +0 -0
  25. {gooddata_eval-1.70.1.dev1 → gooddata_eval-1.71.0}/src/gooddata_eval/core/agentic/search_tool.py +0 -0
  26. {gooddata_eval-1.70.1.dev1 → gooddata_eval-1.71.0}/src/gooddata_eval/core/agentic/visualization.py +0 -0
  27. {gooddata_eval-1.70.1.dev1 → gooddata_eval-1.71.0}/src/gooddata_eval/core/chat/__init__.py +0 -0
  28. {gooddata_eval-1.70.1.dev1 → gooddata_eval-1.71.0}/src/gooddata_eval/core/chat/sse_client.py +0 -0
  29. {gooddata_eval-1.70.1.dev1 → gooddata_eval-1.71.0}/src/gooddata_eval/core/config.py +0 -0
  30. {gooddata_eval-1.70.1.dev1 → gooddata_eval-1.71.0}/src/gooddata_eval/core/connection.py +0 -0
  31. {gooddata_eval-1.70.1.dev1 → gooddata_eval-1.71.0}/src/gooddata_eval/core/dataset/__init__.py +0 -0
  32. {gooddata_eval-1.70.1.dev1 → gooddata_eval-1.71.0}/src/gooddata_eval/core/dataset/langfuse_source.py +0 -0
  33. {gooddata_eval-1.70.1.dev1 → gooddata_eval-1.71.0}/src/gooddata_eval/core/dataset/local.py +0 -0
  34. {gooddata_eval-1.70.1.dev1 → gooddata_eval-1.71.0}/src/gooddata_eval/core/evaluators/__init__.py +0 -0
  35. {gooddata_eval-1.70.1.dev1 → gooddata_eval-1.71.0}/src/gooddata_eval/core/evaluators/_deep_subset.py +0 -0
  36. {gooddata_eval-1.70.1.dev1 → gooddata_eval-1.71.0}/src/gooddata_eval/core/evaluators/_llm_judge.py +0 -0
  37. {gooddata_eval-1.70.1.dev1 → gooddata_eval-1.71.0}/src/gooddata_eval/core/evaluators/_text_utils.py +0 -0
  38. {gooddata_eval-1.70.1.dev1 → gooddata_eval-1.71.0}/src/gooddata_eval/core/evaluators/base.py +0 -0
  39. {gooddata_eval-1.70.1.dev1 → gooddata_eval-1.71.0}/src/gooddata_eval/core/evaluators/general_question.py +0 -0
  40. {gooddata_eval-1.70.1.dev1 → gooddata_eval-1.71.0}/src/gooddata_eval/core/evaluators/guardrail.py +0 -0
  41. {gooddata_eval-1.70.1.dev1 → gooddata_eval-1.71.0}/src/gooddata_eval/core/evaluators/search_tool.py +0 -0
  42. {gooddata_eval-1.70.1.dev1 → gooddata_eval-1.71.0}/src/gooddata_eval/core/evaluators/summary.py +0 -0
  43. {gooddata_eval-1.70.1.dev1 → gooddata_eval-1.71.0}/src/gooddata_eval/core/evaluators/visualization.py +0 -0
  44. {gooddata_eval-1.70.1.dev1 → gooddata_eval-1.71.0}/src/gooddata_eval/core/langfuse/__init__.py +0 -0
  45. {gooddata_eval-1.70.1.dev1 → gooddata_eval-1.71.0}/src/gooddata_eval/core/langfuse/sink.py +0 -0
  46. {gooddata_eval-1.70.1.dev1 → gooddata_eval-1.71.0}/src/gooddata_eval/core/models.py +0 -0
  47. {gooddata_eval-1.70.1.dev1 → gooddata_eval-1.71.0}/src/gooddata_eval/core/reporting/__init__.py +0 -0
  48. {gooddata_eval-1.70.1.dev1 → gooddata_eval-1.71.0}/src/gooddata_eval/core/reporting/console.py +0 -0
  49. {gooddata_eval-1.70.1.dev1 → gooddata_eval-1.71.0}/src/gooddata_eval/core/reporting/json_report.py +0 -0
  50. {gooddata_eval-1.70.1.dev1 → gooddata_eval-1.71.0}/src/gooddata_eval/core/runner.py +0 -0
  51. {gooddata_eval-1.70.1.dev1 → gooddata_eval-1.71.0}/src/gooddata_eval/core/scoring.py +0 -0
  52. {gooddata_eval-1.70.1.dev1 → gooddata_eval-1.71.0}/src/gooddata_eval/core/summary/__init__.py +0 -0
  53. {gooddata_eval-1.70.1.dev1 → gooddata_eval-1.71.0}/src/gooddata_eval/core/summary/http_client.py +0 -0
  54. {gooddata_eval-1.70.1.dev1 → gooddata_eval-1.71.0}/src/gooddata_eval/core/workspace.py +0 -0
  55. {gooddata_eval-1.70.1.dev1 → gooddata_eval-1.71.0}/tests/__init__.py +0 -0
  56. {gooddata_eval-1.70.1.dev1 → gooddata_eval-1.71.0}/tests/conftest.py +0 -0
  57. {gooddata_eval-1.70.1.dev1 → gooddata_eval-1.71.0}/tests/fixtures/sample_dataset/metric_skill_create.json +0 -0
  58. {gooddata_eval-1.70.1.dev1 → gooddata_eval-1.71.0}/tests/fixtures/sample_dataset/visualization_revenue.json +0 -0
  59. {gooddata_eval-1.70.1.dev1 → gooddata_eval-1.71.0}/tests/fixtures/sse_visualization_stream.txt +0 -0
  60. {gooddata_eval-1.70.1.dev1 → gooddata_eval-1.71.0}/tests/test_agentic_alert_skill.py +0 -0
  61. {gooddata_eval-1.70.1.dev1 → gooddata_eval-1.71.0}/tests/test_agentic_conversation.py +0 -0
  62. {gooddata_eval-1.70.1.dev1 → gooddata_eval-1.71.0}/tests/test_agentic_general_question.py +0 -0
  63. {gooddata_eval-1.70.1.dev1 → gooddata_eval-1.71.0}/tests/test_agentic_guardrail.py +0 -0
  64. {gooddata_eval-1.70.1.dev1 → gooddata_eval-1.71.0}/tests/test_agentic_metric_skill.py +0 -0
  65. {gooddata_eval-1.70.1.dev1 → gooddata_eval-1.71.0}/tests/test_agentic_run_context.py +0 -0
  66. {gooddata_eval-1.70.1.dev1 → gooddata_eval-1.71.0}/tests/test_agentic_search_tool.py +0 -0
  67. {gooddata_eval-1.70.1.dev1 → gooddata_eval-1.71.0}/tests/test_agentic_visualization.py +0 -0
  68. {gooddata_eval-1.70.1.dev1 → gooddata_eval-1.71.0}/tests/test_cli.py +0 -0
  69. {gooddata_eval-1.70.1.dev1 → gooddata_eval-1.71.0}/tests/test_connection.py +0 -0
  70. {gooddata_eval-1.70.1.dev1 → gooddata_eval-1.71.0}/tests/test_deep_subset.py +0 -0
  71. {gooddata_eval-1.70.1.dev1 → gooddata_eval-1.71.0}/tests/test_langfuse_sink.py +0 -0
  72. {gooddata_eval-1.70.1.dev1 → gooddata_eval-1.71.0}/tests/test_langfuse_source.py +0 -0
  73. {gooddata_eval-1.70.1.dev1 → gooddata_eval-1.71.0}/tests/test_llm_judge.py +0 -0
  74. {gooddata_eval-1.70.1.dev1 → gooddata_eval-1.71.0}/tests/test_local_loader.py +0 -0
  75. {gooddata_eval-1.70.1.dev1 → gooddata_eval-1.71.0}/tests/test_models.py +0 -0
  76. {gooddata_eval-1.70.1.dev1 → gooddata_eval-1.71.0}/tests/test_reporting.py +0 -0
  77. {gooddata_eval-1.70.1.dev1 → gooddata_eval-1.71.0}/tests/test_runner.py +0 -0
  78. {gooddata_eval-1.70.1.dev1 → gooddata_eval-1.71.0}/tests/test_scoring.py +0 -0
  79. {gooddata_eval-1.70.1.dev1 → gooddata_eval-1.71.0}/tests/test_search_tool_evaluator.py +0 -0
  80. {gooddata_eval-1.70.1.dev1 → gooddata_eval-1.71.0}/tests/test_sse_client.py +0 -0
  81. {gooddata_eval-1.70.1.dev1 → gooddata_eval-1.71.0}/tests/test_summary_client.py +0 -0
  82. {gooddata_eval-1.70.1.dev1 → gooddata_eval-1.71.0}/tests/test_summary_evaluator.py +0 -0
  83. {gooddata_eval-1.70.1.dev1 → gooddata_eval-1.71.0}/tests/test_text_evaluators.py +0 -0
  84. {gooddata_eval-1.70.1.dev1 → gooddata_eval-1.71.0}/tests/test_visualization_evaluator.py +0 -0
  85. {gooddata_eval-1.70.1.dev1 → gooddata_eval-1.71.0}/tests/test_workspace.py +0 -0
  86. {gooddata_eval-1.70.1.dev1 → gooddata_eval-1.71.0}/tox.ini +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: gooddata-eval
3
- Version: 1.70.1.dev1
3
+ Version: 1.71.0
4
4
  Summary: Evaluate the GoodData AI agent against your own questions and models.
5
5
  Project-URL: Source, https://github.com/gooddata/gooddata-python-sdk
6
6
  Author-email: GoodData <support@gooddata.com>
@@ -17,7 +17,7 @@ Classifier: Topic :: Scientific/Engineering
17
17
  Classifier: Topic :: Software Development
18
18
  Classifier: Typing :: Typed
19
19
  Requires-Python: >=3.10
20
- Requires-Dist: gooddata-sdk~=1.70.1.dev1
20
+ Requires-Dist: gooddata-sdk~=1.71.0
21
21
  Requires-Dist: httpx<1.0,>=0.27
22
22
  Requires-Dist: orjson<4.0.0,>=3.9.15
23
23
  Requires-Dist: pydantic<3.0,>=2.6
@@ -1,7 +1,7 @@
1
1
  # (C) 2026 GoodData Corporation
2
2
  [project]
3
3
  name = "gooddata-eval"
4
- version = "1.70.1.dev1"
4
+ version = "1.71.0"
5
5
  description = "Evaluate the GoodData AI agent against your own questions and models."
6
6
  readme = "README.md"
7
7
  license = "MIT"
@@ -11,7 +11,7 @@ authors = [
11
11
  keywords = ["gooddata", "ai", "evaluation", "llm", "analytics", "cli"]
12
12
  requires-python = ">=3.10"
13
13
  dependencies = [
14
- "gooddata-sdk~=1.70.1.dev1",
14
+ "gooddata-sdk~=1.71.0",
15
15
  "httpx>=0.27,<1.0",
16
16
  "orjson>=3.9.15,<4.0.0",
17
17
  "pydantic>=2.6,<3.0",
@@ -25,6 +25,20 @@ def _extract_metric_id(metric_str: str) -> str | None:
25
25
  return match.group(1) if match else None
26
26
 
27
27
 
28
+ def _extract_automation_id(tool_event) -> str | None:
29
+ """Real server-assigned id of the automation this call created.
30
+
31
+ Mirrors ``core/agentic/alert_skill.py``'s ``_extract_alert_call`` — the id
32
+ lives in the tool *result* (what the server assigned), not the tool call
33
+ *arguments* (what the agent asked for).
34
+ """
35
+ result_data = tool_event.parsed_result()
36
+ if not isinstance(result_data, dict):
37
+ return None
38
+ nested_data = result_data.get("data")
39
+ return result_data.get("id") or (nested_data.get("id") if isinstance(nested_data, dict) else None)
40
+
41
+
28
42
  def _check_threshold(expected: dict, actual_args: dict) -> bool:
29
43
  operator = expected.get("Operator", "")
30
44
  if operator == "ANOMALY":
@@ -58,10 +72,11 @@ class AlertSkillEvaluator:
58
72
  return ItemEvaluation(
59
73
  passed=False,
60
74
  rank_key=(False,) * 7,
61
- detail={"alert_created": False},
75
+ detail={"alert_created": False, "automation_id": None},
62
76
  )
63
77
 
64
78
  args = tool_event.parsed_arguments()
79
+ automation_id = _extract_automation_id(tool_event)
65
80
 
66
81
  operator_correct = True
67
82
  threshold_correct = True
@@ -126,5 +141,9 @@ class AlertSkillEvaluator:
126
141
  "filters_correct": filters_correct,
127
142
  "metric_correct": metric_correct,
128
143
  "recipients_correct": recipients_correct,
144
+ # Real server-assigned id of the automation this run created —
145
+ # lets a caller (e.g. a cleanup step) delete the exact object
146
+ # created instead of diffing the workspace catalog before/after.
147
+ "automation_id": automation_id,
129
148
  },
130
149
  )
@@ -28,7 +28,7 @@ class MetricSkillEvaluator:
28
28
  return ItemEvaluation(
29
29
  passed=False,
30
30
  rank_key=(False, False, False),
31
- detail={"metric_created": False, "maql_correct": False, "format_correct": False},
31
+ detail={"metric_created": False, "maql_correct": False, "format_correct": False, "metric_id": None},
32
32
  )
33
33
 
34
34
  result = tool_event.parsed_result()
@@ -54,5 +54,10 @@ class MetricSkillEvaluator:
54
54
  "actual_maql": actual_maql,
55
55
  "expected_format": expected_format,
56
56
  "actual_format": actual_format,
57
+ # The real id of the metric this run created, straight from the
58
+ # create_metric tool result — lets a caller (e.g. a cleanup step)
59
+ # delete the exact object created instead of diffing the workspace
60
+ # catalog before/after and guessing by name.
61
+ "metric_id": payload.get("metric_id"),
57
62
  },
58
63
  )
@@ -15,11 +15,15 @@ def _item(expected_output: dict) -> DatasetItem:
15
15
  )
16
16
 
17
17
 
18
- def _chat_with_alert(args: dict) -> ChatResult:
18
+ def _chat_with_alert(args: dict, result: dict | None = None) -> ChatResult:
19
19
  return ChatResult.model_validate(
20
20
  {
21
21
  "toolCallEvents": [
22
- {"functionName": "create_metric_alert", "functionArguments": json.dumps(args), "result": "{}"}
22
+ {
23
+ "functionName": "create_metric_alert",
24
+ "functionArguments": json.dumps(args),
25
+ "result": json.dumps(result if result is not None else {}),
26
+ }
23
27
  ]
24
28
  }
25
29
  )
@@ -81,3 +85,40 @@ def test_alert_evaluator_fails_when_no_tool_call():
81
85
  )
82
86
  assert result.passed is False
83
87
  assert result.detail["alert_created"] is False
88
+ assert result.detail["automation_id"] is None
89
+
90
+
91
+ def test_alert_evaluator_extracts_automation_id_from_top_level_result():
92
+ expected = {"Operator": "LESS_THAN", "Threshold": "20000"}
93
+ actual_args = {"operator": "LESS_THAN", "threshold": 20000}
94
+ result = get_evaluator("alert_skill").evaluate(
95
+ _item(expected), _chat_with_alert(actual_args, result={"id": "automation-abc-123"})
96
+ )
97
+ assert result.detail["automation_id"] == "automation-abc-123"
98
+
99
+
100
+ def test_alert_evaluator_extracts_automation_id_from_nested_data_result():
101
+ expected = {"Operator": "LESS_THAN", "Threshold": "20000"}
102
+ actual_args = {"operator": "LESS_THAN", "threshold": 20000}
103
+ result = get_evaluator("alert_skill").evaluate(
104
+ _item(expected), _chat_with_alert(actual_args, result={"data": {"id": "automation-xyz-789"}})
105
+ )
106
+ assert result.detail["automation_id"] == "automation-xyz-789"
107
+
108
+
109
+ def test_alert_evaluator_automation_id_none_when_result_has_no_id():
110
+ expected = {"Operator": "LESS_THAN", "Threshold": "20000"}
111
+ actual_args = {"operator": "LESS_THAN", "threshold": 20000}
112
+ result = get_evaluator("alert_skill").evaluate(_item(expected), _chat_with_alert(actual_args, result={}))
113
+ assert result.detail["automation_id"] is None
114
+
115
+
116
+ def test_alert_evaluator_automation_id_none_when_nested_data_is_not_a_mapping():
117
+ # Malformed/unexpected payload shape: "data" present but not a dict -> must
118
+ # not crash, just report automation_id=None (CodeRabbit review on PR #1694).
119
+ expected = {"Operator": "LESS_THAN", "Threshold": "20000"}
120
+ actual_args = {"operator": "LESS_THAN", "threshold": 20000}
121
+ result = get_evaluator("alert_skill").evaluate(
122
+ _item(expected), _chat_with_alert(actual_args, result={"data": "not-a-mapping"})
123
+ )
124
+ assert result.detail["automation_id"] is None
@@ -32,6 +32,7 @@ def test_metric_evaluator_passes_on_exact_match():
32
32
  assert result.passed is True
33
33
  assert result.detail["maql_correct"] is True
34
34
  assert result.detail["format_correct"] is True
35
+ assert result.detail["metric_id"] == "avg_order_value"
35
36
 
36
37
 
37
38
  def test_metric_evaluator_fails_wrong_maql():
@@ -48,3 +49,4 @@ def test_metric_evaluator_fails_when_no_tool_call():
48
49
  result = ev.evaluate(_item(), empty)
49
50
  assert result.passed is False
50
51
  assert result.detail["metric_created"] is False
52
+ assert result.detail["metric_id"] is None