gooddata-eval 1.71.0__tar.gz → 1.71.1.dev1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/PKG-INFO +2 -2
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/pyproject.toml +2 -2
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/core/agentic/alert_skill.py +22 -5
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/core/agentic/conversation.py +5 -1
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/core/chat/sse_client.py +7 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/core/models.py +4 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/tests/test_agentic_alert_skill.py +92 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/tests/test_agentic_conversation.py +62 -1
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/tests/test_sse_client.py +38 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/.gitignore +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/LICENSE.txt +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/Makefile +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/README.md +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/__init__.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/_version.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/cli/__init__.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/cli/agentic_runner.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/cli/main.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/core/__init__.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/core/agentic/__init__.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/core/agentic/_catalog.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/core/agentic/_langfuse.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/core/agentic/general_question.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/core/agentic/guardrail.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/core/agentic/metric_skill.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/core/agentic/search_tool.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/core/agentic/visualization.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/core/chat/__init__.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/core/config.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/core/connection.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/core/dataset/__init__.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/core/dataset/langfuse_source.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/core/dataset/local.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/core/evaluators/__init__.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/core/evaluators/_deep_subset.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/core/evaluators/_llm_judge.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/core/evaluators/_text_utils.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/core/evaluators/alert_skill.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/core/evaluators/base.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/core/evaluators/general_question.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/core/evaluators/guardrail.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/core/evaluators/metric_skill.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/core/evaluators/search_tool.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/core/evaluators/summary.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/core/evaluators/visualization.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/core/langfuse/__init__.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/core/langfuse/sink.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/core/reporting/__init__.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/core/reporting/console.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/core/reporting/json_report.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/core/runner.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/core/scoring.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/core/summary/__init__.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/core/summary/http_client.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/core/workspace.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/tests/__init__.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/tests/conftest.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/tests/fixtures/sample_dataset/metric_skill_create.json +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/tests/fixtures/sample_dataset/visualization_revenue.json +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/tests/fixtures/sse_visualization_stream.txt +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/tests/test_agentic_general_question.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/tests/test_agentic_guardrail.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/tests/test_agentic_metric_skill.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/tests/test_agentic_run_context.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/tests/test_agentic_search_tool.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/tests/test_agentic_visualization.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/tests/test_alert_skill_evaluator.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/tests/test_cli.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/tests/test_connection.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/tests/test_deep_subset.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/tests/test_langfuse_sink.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/tests/test_langfuse_source.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/tests/test_llm_judge.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/tests/test_local_loader.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/tests/test_metric_skill_evaluator.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/tests/test_models.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/tests/test_reporting.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/tests/test_runner.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/tests/test_scoring.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/tests/test_search_tool_evaluator.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/tests/test_summary_client.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/tests/test_summary_evaluator.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/tests/test_text_evaluators.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/tests/test_visualization_evaluator.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/tests/test_workspace.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/tox.ini +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: gooddata-eval
|
|
3
|
-
Version: 1.71.
|
|
3
|
+
Version: 1.71.1.dev1
|
|
4
4
|
Summary: Evaluate the GoodData AI agent against your own questions and models.
|
|
5
5
|
Project-URL: Source, https://github.com/gooddata/gooddata-python-sdk
|
|
6
6
|
Author-email: GoodData <support@gooddata.com>
|
|
@@ -17,7 +17,7 @@ Classifier: Topic :: Scientific/Engineering
|
|
|
17
17
|
Classifier: Topic :: Software Development
|
|
18
18
|
Classifier: Typing :: Typed
|
|
19
19
|
Requires-Python: >=3.10
|
|
20
|
-
Requires-Dist: gooddata-sdk~=1.71.
|
|
20
|
+
Requires-Dist: gooddata-sdk~=1.71.1.dev1
|
|
21
21
|
Requires-Dist: httpx<1.0,>=0.27
|
|
22
22
|
Requires-Dist: orjson<4.0.0,>=3.9.15
|
|
23
23
|
Requires-Dist: pydantic<3.0,>=2.6
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
# (C) 2026 GoodData Corporation
|
|
2
2
|
[project]
|
|
3
3
|
name = "gooddata-eval"
|
|
4
|
-
version = "1.71.
|
|
4
|
+
version = "1.71.1.dev1"
|
|
5
5
|
description = "Evaluate the GoodData AI agent against your own questions and models."
|
|
6
6
|
readme = "README.md"
|
|
7
7
|
license = "MIT"
|
|
@@ -11,7 +11,7 @@ authors = [
|
|
|
11
11
|
keywords = ["gooddata", "ai", "evaluation", "llm", "analytics", "cli"]
|
|
12
12
|
requires-python = ">=3.10"
|
|
13
13
|
dependencies = [
|
|
14
|
-
"gooddata-sdk~=1.71.
|
|
14
|
+
"gooddata-sdk~=1.71.1.dev1",
|
|
15
15
|
"httpx>=0.27,<1.0",
|
|
16
16
|
"orjson>=3.9.15,<4.0.0",
|
|
17
17
|
"pydantic>=2.6,<3.0",
|
{gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/core/agentic/alert_skill.py
RENAMED
|
@@ -311,11 +311,26 @@ def _extract_alert_call(tool_call_events: list[ToolCallEvent]) -> tuple[str | No
|
|
|
311
311
|
return None, {}, False
|
|
312
312
|
|
|
313
313
|
|
|
314
|
-
def
|
|
315
|
-
|
|
316
|
-
|
|
317
|
-
|
|
318
|
-
|
|
314
|
+
def render_alert_proposal(proposal: dict) -> str:
|
|
315
|
+
"""Render an alert-proposal part as the text the simulated user reacts to.
|
|
316
|
+
|
|
317
|
+
The alert skill's confirmation step deliberately emits no text part (GDAI-2032) — the
|
|
318
|
+
prompt and the CTA live only in the proposal payload, which the frontend renders as a
|
|
319
|
+
widget. Dumping the payload (rather than prose) keeps recipients, condition, trigger and
|
|
320
|
+
dashboard visible so the simulated user can still verify them against its goal, and does
|
|
321
|
+
not need updating whenever ``AlertProposal`` grows a field.
|
|
322
|
+
"""
|
|
323
|
+
cta = proposal.get("cta") or "Should I create this alert?"
|
|
324
|
+
summary = {k: v for k, v in proposal.items() if k != "cta"}
|
|
325
|
+
alert = dict(summary.get("alert") or {})
|
|
326
|
+
# The AFM execution block is opaque wire dicts — noise that would crowd out the fields
|
|
327
|
+
# the simulated user actually has to check.
|
|
328
|
+
alert.pop("execution", None)
|
|
329
|
+
if "alert" in summary:
|
|
330
|
+
# Key off presence, not truthiness: an alert whose only key was `execution` must
|
|
331
|
+
# still be replaced, otherwise the original (execution-bearing) dict survives.
|
|
332
|
+
summary["alert"] = alert
|
|
333
|
+
return f"{cta}\n\nAlert proposal:\n{json.dumps(summary, indent=2, sort_keys=True)}"
|
|
319
334
|
|
|
320
335
|
|
|
321
336
|
def run_agentic_alert_skill(
|
|
@@ -352,6 +367,8 @@ def run_agentic_alert_skill(
|
|
|
352
367
|
alert_id_to_delete = alert_id
|
|
353
368
|
break
|
|
354
369
|
response_text = (chat_result.text_response or "").strip()
|
|
370
|
+
if not response_text and chat_result.alert_proposals:
|
|
371
|
+
response_text = render_alert_proposal(chat_result.alert_proposals[-1])
|
|
355
372
|
# Stop if agent gave a completely empty response (stuck)
|
|
356
373
|
if not response_text and not chat_result.tool_call_events:
|
|
357
374
|
break
|
{gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/core/agentic/conversation.py
RENAMED
|
@@ -11,6 +11,7 @@ from typing import Literal
|
|
|
11
11
|
from gooddata_sdk import GoodDataSdk
|
|
12
12
|
from pydantic import BaseModel
|
|
13
13
|
|
|
14
|
+
from gooddata_eval.core.agentic.alert_skill import render_alert_proposal
|
|
14
15
|
from gooddata_eval.core.agentic.metric_skill import _delete_metric, _extract_created_metric_ids
|
|
15
16
|
from gooddata_eval.core.chat.sse_client import ChatClient
|
|
16
17
|
from gooddata_eval.core.models import ChatResult, ToolCallEvent
|
|
@@ -322,7 +323,10 @@ def run_agentic_conversation(
|
|
|
322
323
|
break
|
|
323
324
|
|
|
324
325
|
response_text = (chat_result.text_response or "").strip()
|
|
325
|
-
if
|
|
326
|
+
if not response_text and chat_result.alert_proposals:
|
|
327
|
+
response_text = render_alert_proposal(chat_result.alert_proposals[-1])
|
|
328
|
+
asking = _is_asking_clarification(response_text) or bool(chat_result.alert_proposals)
|
|
329
|
+
if asking and clarification_turns < max_clarification_turns:
|
|
326
330
|
clarification_turns += 1
|
|
327
331
|
total_clarification_turns += 1
|
|
328
332
|
current_message = _get_sim_user_response(response_text, resolved_turn, resolved_expected)
|
{gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/core/chat/sse_client.py
RENAMED
|
@@ -102,6 +102,7 @@ class _SseAccumulator:
|
|
|
102
102
|
text_parts: list[str] = field(default_factory=list)
|
|
103
103
|
viz_reasoning_parts: list[str] = field(default_factory=list)
|
|
104
104
|
visualizations: list[dict[str, Any]] = field(default_factory=list)
|
|
105
|
+
alert_proposals: list[dict[str, Any]] = field(default_factory=list)
|
|
105
106
|
tool_call_events: list[dict[str, Any]] = field(default_factory=list)
|
|
106
107
|
call_id_to_event_index: dict[str, int] = field(default_factory=dict)
|
|
107
108
|
reasoning_steps: list[dict[str, Any]] = field(default_factory=list)
|
|
@@ -125,6 +126,11 @@ def _handle_multipart(content: dict[str, Any], acc: _SseAccumulator) -> None:
|
|
|
125
126
|
acc.viz_reasoning_parts.append(t)
|
|
126
127
|
elif ptype == "visualization" and part.get("visualization"):
|
|
127
128
|
acc.visualizations.append(part["visualization"])
|
|
129
|
+
elif ptype == "alertProposal":
|
|
130
|
+
# Record the part even when the server could not resolve the proposal payload
|
|
131
|
+
# (``alertProposal: null``) — its mere presence is the confirmation signal, and
|
|
132
|
+
# the reader falls back to a default CTA.
|
|
133
|
+
acc.alert_proposals.append(part.get("alertProposal") or {})
|
|
128
134
|
|
|
129
135
|
|
|
130
136
|
def _handle_reasoning(content: dict[str, Any], acc: _SseAccumulator) -> None:
|
|
@@ -161,6 +167,7 @@ def _handle_tool_result(content: dict[str, Any], acc: _SseAccumulator) -> None:
|
|
|
161
167
|
def _build_chat_result(acc: _SseAccumulator) -> ChatResult:
|
|
162
168
|
payload: dict[str, Any] = {
|
|
163
169
|
"textResponse": "\n".join(acc.text_parts) or None,
|
|
170
|
+
"alertProposals": acc.alert_proposals,
|
|
164
171
|
"toolCallEvents": acc.tool_call_events,
|
|
165
172
|
"reasoningStepCount": len(acc.reasoning_steps),
|
|
166
173
|
}
|
|
@@ -92,6 +92,10 @@ class ChatResult(BaseModel):
|
|
|
92
92
|
|
|
93
93
|
text_response: str | None = Field(default=None, alias="textResponse")
|
|
94
94
|
created_visualizations: CreatedVisualizations | None = Field(default=None, alias="createdVisualizations")
|
|
95
|
+
# Alert-proposal parts of the agent's multipart response. The alert skill's confirmation
|
|
96
|
+
# step emits ONLY this part (no text part), so its `cta` is the only "the agent is asking
|
|
97
|
+
# a question" signal the simulated-user loops can key off.
|
|
98
|
+
alert_proposals: list[dict] = Field(default_factory=list, alias="alertProposals")
|
|
95
99
|
tool_call_events: list[ToolCallEvent] = Field(default_factory=list, alias="toolCallEvents")
|
|
96
100
|
reasoning_step_count: int = Field(default=0, alias="reasoningStepCount")
|
|
97
101
|
conversation_id: str | None = Field(default=None, alias="conversationId")
|
|
@@ -8,10 +8,23 @@ from gooddata_eval.core.agentic.alert_skill import (
|
|
|
8
8
|
_deep_subset,
|
|
9
9
|
_normalize_expected_output,
|
|
10
10
|
_to_number,
|
|
11
|
+
render_alert_proposal,
|
|
11
12
|
run_agentic_alert_skill,
|
|
12
13
|
)
|
|
13
14
|
from gooddata_eval.core.models import ChatResult
|
|
14
15
|
|
|
16
|
+
_PROPOSAL = {
|
|
17
|
+
"title": "# of Orders Alert - Greater Than 500",
|
|
18
|
+
"cta": "Should I create this alert?",
|
|
19
|
+
"recipients": [{"email": "admin@gooddata.com"}],
|
|
20
|
+
"dashboard": {"id": "dash-1", "title": "Orders overview"},
|
|
21
|
+
"alert": {
|
|
22
|
+
"trigger": "ALWAYS",
|
|
23
|
+
"condition": {"comparison": {"operator": "GREATER_THAN", "right": {"value": 500}}},
|
|
24
|
+
"execution": {"measures": [{"opaque": "afm"}]},
|
|
25
|
+
},
|
|
26
|
+
}
|
|
27
|
+
|
|
15
28
|
|
|
16
29
|
def test_to_number_int():
|
|
17
30
|
assert _to_number("42") == 42
|
|
@@ -156,3 +169,82 @@ def test_run_agentic_alert_skill_creates_fresh_conversations_for_remaining_runs(
|
|
|
156
169
|
)
|
|
157
170
|
assert mock_client.create_conversation.call_count == 2
|
|
158
171
|
assert mock_client.delete_conversation.call_count == 2
|
|
172
|
+
|
|
173
|
+
|
|
174
|
+
def test_render_alert_proposal_keeps_verifiable_fields_and_drops_afm():
|
|
175
|
+
rendered = render_alert_proposal(_PROPOSAL)
|
|
176
|
+
# The CTA leads so the simulated user reads it as a question.
|
|
177
|
+
assert rendered.startswith("Should I create this alert?")
|
|
178
|
+
# Rule 3 of the sim-user prompt requires verifying recipients against its goal.
|
|
179
|
+
assert "admin@gooddata.com" in rendered
|
|
180
|
+
assert "GREATER_THAN" in rendered
|
|
181
|
+
assert "Orders overview" in rendered
|
|
182
|
+
# Opaque AFM wire dicts must not crowd out the fields above.
|
|
183
|
+
assert "execution" not in rendered
|
|
184
|
+
|
|
185
|
+
|
|
186
|
+
def test_render_alert_proposal_drops_afm_when_execution_is_the_only_alert_field():
|
|
187
|
+
# Truthiness-gated replacement used to leave the original execution-bearing dict in place.
|
|
188
|
+
rendered = render_alert_proposal({"alert": {"execution": {"measures": [{"opaque": "afm"}]}}})
|
|
189
|
+
assert "execution" not in rendered
|
|
190
|
+
assert "opaque" not in rendered
|
|
191
|
+
|
|
192
|
+
|
|
193
|
+
def test_render_alert_proposal_falls_back_to_default_cta():
|
|
194
|
+
assert render_alert_proposal({}).startswith("Should I create this alert?")
|
|
195
|
+
|
|
196
|
+
|
|
197
|
+
def test_run_agentic_alert_skill_answers_proposal_only_confirmation_turn():
|
|
198
|
+
"""GDAI-2032 regression: confirmation turn has no text part, only an alertProposal.
|
|
199
|
+
|
|
200
|
+
Without the fallback the simulated user is handed an empty agent message, so the agent
|
|
201
|
+
never receives an explicit "yes" and create_metric_alert is never called.
|
|
202
|
+
"""
|
|
203
|
+
proposal_turn = ChatResult.model_validate(
|
|
204
|
+
{
|
|
205
|
+
"text_response": None,
|
|
206
|
+
"alertProposals": [_PROPOSAL],
|
|
207
|
+
"tool_call_events": [
|
|
208
|
+
{"functionName": "prepare_metric_alert_proposal", "functionArguments": "{}", "result": None}
|
|
209
|
+
],
|
|
210
|
+
}
|
|
211
|
+
)
|
|
212
|
+
created_turn = ChatResult.model_validate(
|
|
213
|
+
{
|
|
214
|
+
"text_response": "Alert created.",
|
|
215
|
+
"tool_call_events": [
|
|
216
|
+
{
|
|
217
|
+
"functionName": "create_metric_alert",
|
|
218
|
+
"functionArguments": '{"operator": "GREATER_THAN", "threshold": 500}',
|
|
219
|
+
"result": '{"id": "alert-1"}',
|
|
220
|
+
}
|
|
221
|
+
],
|
|
222
|
+
}
|
|
223
|
+
)
|
|
224
|
+
mock_client = MagicMock()
|
|
225
|
+
mock_client.send_message.side_effect = [proposal_turn, created_turn]
|
|
226
|
+
|
|
227
|
+
with (
|
|
228
|
+
patch("gooddata_eval.core.agentic.alert_skill.ChatClient", return_value=mock_client),
|
|
229
|
+
patch(
|
|
230
|
+
"gooddata_eval.core.agentic.alert_skill.generate_simulated_alert_response",
|
|
231
|
+
return_value="Yes, please proceed to create the alert.",
|
|
232
|
+
) as mock_sim,
|
|
233
|
+
patch("gooddata_eval.core.agentic.alert_skill._delete_alert"),
|
|
234
|
+
):
|
|
235
|
+
summary = run_agentic_alert_skill(
|
|
236
|
+
host="http://host",
|
|
237
|
+
token="tok",
|
|
238
|
+
workspace_id="ws1",
|
|
239
|
+
question="Notify me whenever the number of orders goes above 500",
|
|
240
|
+
expected_output={"operator": "GREATER_THAN", "threshold": 500},
|
|
241
|
+
k=1,
|
|
242
|
+
max_iterations=6,
|
|
243
|
+
initial_conversation_id="conv-1",
|
|
244
|
+
)
|
|
245
|
+
|
|
246
|
+
agent_message = mock_sim.call_args.args[0]
|
|
247
|
+
assert "Should I create this alert?" in agent_message
|
|
248
|
+
assert "admin@gooddata.com" in agent_message
|
|
249
|
+
assert summary.best.eval.alert_created is True
|
|
250
|
+
assert summary.best.alert_id == "alert-1"
|
|
@@ -10,7 +10,7 @@ from gooddata_eval.core.agentic.conversation import (
|
|
|
10
10
|
_resolve_refs,
|
|
11
11
|
run_agentic_conversation,
|
|
12
12
|
)
|
|
13
|
-
from gooddata_eval.core.models import ToolCallEvent
|
|
13
|
+
from gooddata_eval.core.models import ChatResult, ToolCallEvent
|
|
14
14
|
|
|
15
15
|
|
|
16
16
|
def _skills_tc(*skills):
|
|
@@ -291,3 +291,64 @@ def test_run_agentic_conversation_deletes_metrics_even_when_a_later_turn_raises(
|
|
|
291
291
|
)
|
|
292
292
|
|
|
293
293
|
mock_sdk._client.entities_api.delete_entity_metrics.assert_called_once_with("ws1", "m1")
|
|
294
|
+
|
|
295
|
+
|
|
296
|
+
def _alert_turn_fixture():
|
|
297
|
+
return ConversationFixture(
|
|
298
|
+
id="conv-alert",
|
|
299
|
+
expected_skills=["alert"],
|
|
300
|
+
turns=[
|
|
301
|
+
TurnDefinition(
|
|
302
|
+
turn_id="create_alert",
|
|
303
|
+
message="Now alert me when the metric drops below 100.",
|
|
304
|
+
expected_skill="alert",
|
|
305
|
+
expected_output_type="tool_call",
|
|
306
|
+
expected_tool_name="create_metric_alert",
|
|
307
|
+
)
|
|
308
|
+
],
|
|
309
|
+
)
|
|
310
|
+
|
|
311
|
+
|
|
312
|
+
def test_run_agentic_conversation_treats_alert_proposal_as_a_clarification():
|
|
313
|
+
"""GDAI-2032 regression: a proposal-only turn has no text, so the old text-only check
|
|
314
|
+
stopped the turn instead of replying, and create_metric_alert never happened."""
|
|
315
|
+
proposal_turn = ChatResult.model_validate(
|
|
316
|
+
{
|
|
317
|
+
"text_response": None,
|
|
318
|
+
"alertProposals": [{"cta": "Should I create this alert?", "recipients": [{"email": "a@b.com"}]}],
|
|
319
|
+
"toolCallEvents": [
|
|
320
|
+
{"functionName": "set_skills", "functionArguments": '{"skills": ["alert"]}', "result": None},
|
|
321
|
+
{"functionName": "prepare_metric_alert_proposal", "functionArguments": "{}", "result": None},
|
|
322
|
+
],
|
|
323
|
+
}
|
|
324
|
+
)
|
|
325
|
+
created_turn = ChatResult.model_validate(
|
|
326
|
+
{
|
|
327
|
+
"text_response": "Alert created.",
|
|
328
|
+
"toolCallEvents": [
|
|
329
|
+
{"functionName": "create_metric_alert", "functionArguments": "{}", "result": '{"id": "alert-1"}'}
|
|
330
|
+
],
|
|
331
|
+
}
|
|
332
|
+
)
|
|
333
|
+
mock_client = MagicMock()
|
|
334
|
+
mock_client.create_conversation.return_value = "conv-1"
|
|
335
|
+
mock_client.send_message.side_effect = [proposal_turn, created_turn]
|
|
336
|
+
|
|
337
|
+
with (
|
|
338
|
+
patch("gooddata_eval.core.agentic.conversation.ChatClient", return_value=mock_client),
|
|
339
|
+
patch("gooddata_eval.core.agentic.conversation.GoodDataSdk"),
|
|
340
|
+
patch(
|
|
341
|
+
"gooddata_eval.core.agentic.conversation._get_sim_user_response",
|
|
342
|
+
return_value="Yes, please create it.",
|
|
343
|
+
) as mock_sim,
|
|
344
|
+
):
|
|
345
|
+
result = run_agentic_conversation(
|
|
346
|
+
host="http://host/api/v1/actions/workspaces/ws1/ai",
|
|
347
|
+
token="tok",
|
|
348
|
+
workspace_id="ws1",
|
|
349
|
+
fixture=_alert_turn_fixture(),
|
|
350
|
+
)
|
|
351
|
+
|
|
352
|
+
assert "Should I create this alert?" in mock_sim.call_args.args[0]
|
|
353
|
+
assert result.turn_results[0].clarification_turns_used == 1
|
|
354
|
+
assert result.turn_results[0].skill_success is True
|
|
@@ -82,6 +82,44 @@ def test_parse_sse_lines_prefers_multipart_viz_over_adhoc_fallback():
|
|
|
82
82
|
assert result.created_visualizations.objects[0].id == "real"
|
|
83
83
|
|
|
84
84
|
|
|
85
|
+
def test_parse_sse_lines_collects_alert_proposal_without_text_part():
|
|
86
|
+
"""The alert skill's confirmation turn emits ONLY an alertProposal part (GDAI-2032).
|
|
87
|
+
|
|
88
|
+
Pins the wire contract the simulated-user loops depend on: part ``type`` is
|
|
89
|
+
``alertProposal`` and the payload lives under the ``alertProposal`` key.
|
|
90
|
+
"""
|
|
91
|
+
proposal = {
|
|
92
|
+
"title": "# of Orders Alert - Greater Than 500",
|
|
93
|
+
"cta": "Should I create this alert?",
|
|
94
|
+
"recipients": [{"email": "admin@gooddata.com"}],
|
|
95
|
+
"alert": {"trigger": "ALWAYS", "execution": {"measures": [{"opaque": "afm"}]}},
|
|
96
|
+
}
|
|
97
|
+
lines = [
|
|
98
|
+
'data: {"item": {"role": "assistant", "content": {"type": "toolCall", "callId": "c1", '
|
|
99
|
+
'"name": "prepare_metric_alert_proposal", "arguments": {}}}}',
|
|
100
|
+
f'data: {{"item": {{"role": "assistant", "content": {{"type": "multipart", '
|
|
101
|
+
f'"parts": [{{"type": "alertProposal", "alertProposal": {json.dumps(proposal)}}}]}}}}}}',
|
|
102
|
+
]
|
|
103
|
+
result = parse_sse_lines(lines)
|
|
104
|
+
assert result.text_response is None
|
|
105
|
+
assert result.alert_proposals == [proposal]
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
def test_parse_sse_lines_keeps_alert_proposal_part_when_payload_is_null():
|
|
109
|
+
"""Presence of the part is the confirmation signal even if the server did not resolve it."""
|
|
110
|
+
lines = [
|
|
111
|
+
'data: {"item": {"role": "assistant", "content": {"type": "multipart", '
|
|
112
|
+
'"parts": [{"type": "alertProposal", "alertProposal": null}]}}}',
|
|
113
|
+
]
|
|
114
|
+
result = parse_sse_lines(lines)
|
|
115
|
+
assert result.alert_proposals == [{}]
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
def test_parse_sse_lines_has_no_alert_proposals_by_default():
|
|
119
|
+
lines = ['data: {"item": {"role": "assistant", "content": {"type": "text", "text": "Done"}}}']
|
|
120
|
+
assert parse_sse_lines(lines).alert_proposals == []
|
|
121
|
+
|
|
122
|
+
|
|
85
123
|
@pytest.mark.parametrize("code", [429, 502, 503, 504])
|
|
86
124
|
def test_parse_sse_lines_transient_status_codes(code):
|
|
87
125
|
with pytest.raises(TransientChatError) as ei:
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/core/agentic/__init__.py
RENAMED
|
File without changes
|
{gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/core/agentic/_catalog.py
RENAMED
|
File without changes
|
{gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/core/agentic/_langfuse.py
RENAMED
|
File without changes
|
|
File without changes
|
{gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/core/agentic/guardrail.py
RENAMED
|
File without changes
|
{gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/core/agentic/metric_skill.py
RENAMED
|
File without changes
|
{gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/core/agentic/search_tool.py
RENAMED
|
File without changes
|
{gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/core/agentic/visualization.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/core/dataset/__init__.py
RENAMED
|
File without changes
|
{gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/core/dataset/langfuse_source.py
RENAMED
|
File without changes
|
|
File without changes
|
{gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/core/evaluators/__init__.py
RENAMED
|
File without changes
|
{gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/core/evaluators/_deep_subset.py
RENAMED
|
File without changes
|
{gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/core/evaluators/_llm_judge.py
RENAMED
|
File without changes
|
{gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/core/evaluators/_text_utils.py
RENAMED
|
File without changes
|
{gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/core/evaluators/alert_skill.py
RENAMED
|
File without changes
|
{gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/core/evaluators/base.py
RENAMED
|
File without changes
|
|
File without changes
|
{gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/core/evaluators/guardrail.py
RENAMED
|
File without changes
|
{gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/core/evaluators/metric_skill.py
RENAMED
|
File without changes
|
{gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/core/evaluators/search_tool.py
RENAMED
|
File without changes
|
{gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/core/evaluators/summary.py
RENAMED
|
File without changes
|
|
File without changes
|
{gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/core/langfuse/__init__.py
RENAMED
|
File without changes
|
|
File without changes
|
{gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/core/reporting/__init__.py
RENAMED
|
File without changes
|
{gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/core/reporting/console.py
RENAMED
|
File without changes
|
{gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/core/reporting/json_report.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
{gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/core/summary/__init__.py
RENAMED
|
File without changes
|
{gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/src/gooddata_eval/core/summary/http_client.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev1}/tests/fixtures/sse_visualization_stream.txt
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|