gooddata-eval 1.72.1.dev5__tar.gz → 1.72.1.dev6__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/PKG-INFO +2 -2
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/pyproject.toml +2 -2
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/src/gooddata_eval/core/agentic/conversation.py +26 -17
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/src/gooddata_eval/core/agentic/metric_skill.py +33 -19
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/tests/test_agentic_conversation.py +115 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/tests/test_agentic_metric_skill.py +61 -1
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/.gitignore +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/LICENSE.txt +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/Makefile +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/README.md +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/src/gooddata_eval/__init__.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/src/gooddata_eval/_version.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/src/gooddata_eval/cli/__init__.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/src/gooddata_eval/cli/agentic_runner.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/src/gooddata_eval/cli/main.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/src/gooddata_eval/core/__init__.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/src/gooddata_eval/core/agentic/__init__.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/src/gooddata_eval/core/agentic/_catalog.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/src/gooddata_eval/core/agentic/_langfuse.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/src/gooddata_eval/core/agentic/alert_skill.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/src/gooddata_eval/core/agentic/general_question.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/src/gooddata_eval/core/agentic/guardrail.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/src/gooddata_eval/core/agentic/kda_skill.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/src/gooddata_eval/core/agentic/search_tool.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/src/gooddata_eval/core/agentic/visualization.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/src/gooddata_eval/core/chat/__init__.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/src/gooddata_eval/core/chat/sse_client.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/src/gooddata_eval/core/config.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/src/gooddata_eval/core/connection.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/src/gooddata_eval/core/dataset/__init__.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/src/gooddata_eval/core/dataset/langfuse_source.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/src/gooddata_eval/core/dataset/local.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/src/gooddata_eval/core/evaluators/__init__.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/src/gooddata_eval/core/evaluators/_deep_subset.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/src/gooddata_eval/core/evaluators/_llm_judge.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/src/gooddata_eval/core/evaluators/_text_utils.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/src/gooddata_eval/core/evaluators/alert_skill.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/src/gooddata_eval/core/evaluators/base.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/src/gooddata_eval/core/evaluators/general_question.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/src/gooddata_eval/core/evaluators/guardrail.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/src/gooddata_eval/core/evaluators/metric_skill.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/src/gooddata_eval/core/evaluators/search_tool.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/src/gooddata_eval/core/evaluators/summary.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/src/gooddata_eval/core/evaluators/visualization.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/src/gooddata_eval/core/langfuse/__init__.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/src/gooddata_eval/core/langfuse/sink.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/src/gooddata_eval/core/models.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/src/gooddata_eval/core/reporting/__init__.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/src/gooddata_eval/core/reporting/console.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/src/gooddata_eval/core/reporting/json_report.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/src/gooddata_eval/core/runner.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/src/gooddata_eval/core/scoring.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/src/gooddata_eval/core/summary/__init__.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/src/gooddata_eval/core/summary/http_client.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/src/gooddata_eval/core/workspace.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/tests/__init__.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/tests/conftest.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/tests/fixtures/sample_dataset/metric_skill_create.json +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/tests/fixtures/sample_dataset/visualization_revenue.json +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/tests/fixtures/sse_visualization_stream.txt +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/tests/test_agentic_alert_skill.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/tests/test_agentic_general_question.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/tests/test_agentic_guardrail.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/tests/test_agentic_kda_skill.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/tests/test_agentic_langfuse_trace.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/tests/test_agentic_run_context.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/tests/test_agentic_search_tool.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/tests/test_agentic_visualization.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/tests/test_alert_skill_evaluator.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/tests/test_cli.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/tests/test_connection.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/tests/test_deep_subset.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/tests/test_langfuse_sink.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/tests/test_langfuse_source.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/tests/test_llm_judge.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/tests/test_local_loader.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/tests/test_metric_skill_evaluator.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/tests/test_models.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/tests/test_reporting.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/tests/test_runner.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/tests/test_scoring.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/tests/test_search_tool_evaluator.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/tests/test_sse_client.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/tests/test_summary_client.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/tests/test_summary_evaluator.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/tests/test_text_evaluators.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/tests/test_visualization_evaluator.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/tests/test_workspace.py +0 -0
- {gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/tox.ini +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: gooddata-eval
|
|
3
|
-
Version: 1.72.1.
|
|
3
|
+
Version: 1.72.1.dev6
|
|
4
4
|
Summary: Evaluate the GoodData AI agent against your own questions and models.
|
|
5
5
|
Project-URL: Source, https://github.com/gooddata/gooddata-python-sdk
|
|
6
6
|
Author-email: GoodData <support@gooddata.com>
|
|
@@ -17,7 +17,7 @@ Classifier: Topic :: Scientific/Engineering
|
|
|
17
17
|
Classifier: Topic :: Software Development
|
|
18
18
|
Classifier: Typing :: Typed
|
|
19
19
|
Requires-Python: >=3.10
|
|
20
|
-
Requires-Dist: gooddata-sdk~=1.72.1.
|
|
20
|
+
Requires-Dist: gooddata-sdk~=1.72.1.dev6
|
|
21
21
|
Requires-Dist: httpx<1.0,>=0.27
|
|
22
22
|
Requires-Dist: orjson<4.0.0,>=3.9.15
|
|
23
23
|
Requires-Dist: pydantic<3.0,>=2.6
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
# (C) 2026 GoodData Corporation
|
|
2
2
|
[project]
|
|
3
3
|
name = "gooddata-eval"
|
|
4
|
-
version = "1.72.1.
|
|
4
|
+
version = "1.72.1.dev6"
|
|
5
5
|
description = "Evaluate the GoodData AI agent against your own questions and models."
|
|
6
6
|
readme = "README.md"
|
|
7
7
|
license = "MIT"
|
|
@@ -11,7 +11,7 @@ authors = [
|
|
|
11
11
|
keywords = ["gooddata", "ai", "evaluation", "llm", "analytics", "cli"]
|
|
12
12
|
requires-python = ">=3.10"
|
|
13
13
|
dependencies = [
|
|
14
|
-
"gooddata-sdk~=1.72.1.
|
|
14
|
+
"gooddata-sdk~=1.72.1.dev6",
|
|
15
15
|
"httpx>=0.27,<1.0",
|
|
16
16
|
"orjson>=3.9.15,<4.0.0",
|
|
17
17
|
"pydantic>=2.6,<3.0",
|
|
@@ -25,6 +25,8 @@ from gooddata_eval.core.scoring import (
|
|
|
25
25
|
|
|
26
26
|
_REF_PATTERN = re.compile(r"\$ref:([\w_]+)\.([\w_]+)")
|
|
27
27
|
|
|
28
|
+
_DEFAULT_MAX_CLARIFICATION_TURNS = 7
|
|
29
|
+
|
|
28
30
|
|
|
29
31
|
class TurnDefinition(BaseModel):
|
|
30
32
|
"""Definition of a single turn in a multi-turn conversation evaluation."""
|
|
@@ -192,13 +194,6 @@ def _check_output_correct(turn: TurnDefinition, chat_result: ChatResult) -> bool
|
|
|
192
194
|
return None
|
|
193
195
|
|
|
194
196
|
|
|
195
|
-
def _is_asking_clarification(text: str) -> bool:
|
|
196
|
-
if not text:
|
|
197
|
-
return False
|
|
198
|
-
t = text.lower()
|
|
199
|
-
return "?" in t or "could you" in t or "please" in t or "clarif" in t
|
|
200
|
-
|
|
201
|
-
|
|
202
197
|
def _get_sim_user_response(agent_message: str, turn: TurnDefinition, expected_output: dict | None) -> str:
|
|
203
198
|
"""Generate a simulated user reply to an agent clarification question."""
|
|
204
199
|
otype = turn.expected_output_type
|
|
@@ -277,7 +272,7 @@ def run_agentic_conversation(
|
|
|
277
272
|
token: str,
|
|
278
273
|
workspace_id: str,
|
|
279
274
|
fixture: ConversationFixture,
|
|
280
|
-
max_clarification_turns: int =
|
|
275
|
+
max_clarification_turns: int = _DEFAULT_MAX_CLARIFICATION_TURNS,
|
|
281
276
|
initial_conversation_id: str | None = None,
|
|
282
277
|
reasoning_effort: ReasoningEffort | None = None,
|
|
283
278
|
) -> ConversationResult:
|
|
@@ -307,8 +302,22 @@ def run_agentic_conversation(
|
|
|
307
302
|
owns_conversation = True
|
|
308
303
|
|
|
309
304
|
for turn in fixture.turns:
|
|
310
|
-
|
|
311
|
-
|
|
305
|
+
try:
|
|
306
|
+
resolved_expected = _resolve_refs(turn.expected_output, turn_outputs)
|
|
307
|
+
except ValueError as exc:
|
|
308
|
+
print(f"[SKIP] turn '{turn.turn_id}': {exc}")
|
|
309
|
+
turn_results.append(
|
|
310
|
+
TurnResult(
|
|
311
|
+
turn_id=turn.turn_id,
|
|
312
|
+
expected_skill=turn.expected_skill,
|
|
313
|
+
skill_routing=False,
|
|
314
|
+
output_present=False,
|
|
315
|
+
no_error=False,
|
|
316
|
+
activated_skills=[],
|
|
317
|
+
output_correct=False,
|
|
318
|
+
)
|
|
319
|
+
)
|
|
320
|
+
continue
|
|
312
321
|
resolved_turn = turn.model_copy(update={"expected_output": resolved_expected})
|
|
313
322
|
|
|
314
323
|
clarification_turns = 0
|
|
@@ -327,13 +336,13 @@ def run_agentic_conversation(
|
|
|
327
336
|
response_text = (chat_result.text_response or "").strip()
|
|
328
337
|
if not response_text and chat_result.alert_proposals:
|
|
329
338
|
response_text = render_alert_proposal(chat_result.alert_proposals[-1])
|
|
330
|
-
|
|
331
|
-
|
|
332
|
-
|
|
333
|
-
total_clarification_turns += 1
|
|
334
|
-
current_message = _get_sim_user_response(response_text, resolved_turn, resolved_expected)
|
|
335
|
-
else:
|
|
339
|
+
if not response_text and not chat_result.tool_call_events:
|
|
340
|
+
break
|
|
341
|
+
if clarification_turns >= max_clarification_turns:
|
|
336
342
|
break
|
|
343
|
+
clarification_turns += 1
|
|
344
|
+
total_clarification_turns += 1
|
|
345
|
+
current_message = _get_sim_user_response(response_text, resolved_turn, resolved_expected)
|
|
337
346
|
|
|
338
347
|
activated = _activated_skills(all_tool_calls)
|
|
339
348
|
skill_routing = turn.expected_skill in activated if activated else False
|
|
@@ -397,7 +406,7 @@ def evaluate_agentic_conversation(
|
|
|
397
406
|
token: str,
|
|
398
407
|
workspace_id: str,
|
|
399
408
|
fixture: ConversationFixture,
|
|
400
|
-
max_clarification_turns: int =
|
|
409
|
+
max_clarification_turns: int = _DEFAULT_MAX_CLARIFICATION_TURNS,
|
|
401
410
|
initial_conversation_id: str | None = None,
|
|
402
411
|
langfuse: object | None = None,
|
|
403
412
|
dataset_item_id: str = "",
|
|
@@ -72,16 +72,29 @@ def _best_maql_match(actual_maql: str, expected_outputs: list[dict]) -> tuple[bo
|
|
|
72
72
|
return False, expected_outputs[0].get("maql", "") if expected_outputs else ""
|
|
73
73
|
|
|
74
74
|
|
|
75
|
+
class SimulatedResponseError(RuntimeError):
|
|
76
|
+
"""The simulated user could not reply: openai missing, no API key, or the provider failed.
|
|
77
|
+
|
|
78
|
+
Carries every expected setup/provider failure so callers can end the run without
|
|
79
|
+
swallowing programming errors raised from the same call.
|
|
80
|
+
"""
|
|
81
|
+
|
|
82
|
+
|
|
75
83
|
def generate_simulated_response(agent_message: str, expected_output: dict) -> str:
|
|
76
|
-
"""Generate a user reply to keep the metric-skill conversation going (gpt-4o-mini).
|
|
84
|
+
"""Generate a user reply to keep the metric-skill conversation going (gpt-4o-mini).
|
|
85
|
+
|
|
86
|
+
Raises:
|
|
87
|
+
SimulatedResponseError: openai is not installed, OPENAI_API_KEY is unset, or the
|
|
88
|
+
provider call failed.
|
|
89
|
+
"""
|
|
77
90
|
try:
|
|
78
|
-
from openai import OpenAI # noqa: PLC0415
|
|
91
|
+
from openai import OpenAI, OpenAIError # noqa: PLC0415
|
|
79
92
|
except ImportError as exc:
|
|
80
|
-
raise
|
|
93
|
+
raise SimulatedResponseError("openai package is required for generate_simulated_response") from exc
|
|
81
94
|
|
|
82
95
|
api_key = os.environ.get("OPENAI_API_KEY")
|
|
83
96
|
if not api_key:
|
|
84
|
-
raise
|
|
97
|
+
raise SimulatedResponseError("OPENAI_API_KEY environment variable is not set")
|
|
85
98
|
|
|
86
99
|
client = OpenAI(api_key=api_key)
|
|
87
100
|
expected_maql = expected_output.get("maql", "")
|
|
@@ -91,12 +104,15 @@ def generate_simulated_response(agent_message: str, expected_output: dict) -> st
|
|
|
91
104
|
f"The user originally asked to create a metric with MAQL: {expected_maql}. "
|
|
92
105
|
f"Reply briefly as the user, providing any clarification the assistant needs."
|
|
93
106
|
)
|
|
94
|
-
|
|
95
|
-
|
|
96
|
-
|
|
97
|
-
|
|
98
|
-
|
|
99
|
-
|
|
107
|
+
try:
|
|
108
|
+
response = client.chat.completions.create(
|
|
109
|
+
model="gpt-4o-mini",
|
|
110
|
+
messages=[{"role": "user", "content": prompt}],
|
|
111
|
+
max_tokens=150,
|
|
112
|
+
temperature=0,
|
|
113
|
+
)
|
|
114
|
+
except OpenAIError as exc:
|
|
115
|
+
raise SimulatedResponseError(f"simulated user reply failed: {exc}") from exc
|
|
100
116
|
return response.choices[0].message.content or "Please proceed."
|
|
101
117
|
|
|
102
118
|
|
|
@@ -166,13 +182,6 @@ def _delete_metric(sdk: GoodDataSdk, workspace_id: str, metric_id: str) -> None:
|
|
|
166
182
|
print(f"[CLEANUP] Failed to delete metric {metric_id}: {exc}")
|
|
167
183
|
|
|
168
184
|
|
|
169
|
-
def _is_asking_clarification(text: str) -> bool:
|
|
170
|
-
if not text:
|
|
171
|
-
return False
|
|
172
|
-
t = text.lower()
|
|
173
|
-
return "?" in t or "could you" in t or "please provide" in t or "clarif" in t
|
|
174
|
-
|
|
175
|
-
|
|
176
185
|
def _execute_single_metric_run(
|
|
177
186
|
client: ChatClient,
|
|
178
187
|
sdk: GoodDataSdk,
|
|
@@ -204,9 +213,14 @@ def _execute_single_metric_run(
|
|
|
204
213
|
metric_id_to_delete = candidate.get("metric_id")
|
|
205
214
|
break
|
|
206
215
|
response_text = (chat_result.text_response or "").strip()
|
|
207
|
-
if
|
|
216
|
+
if not response_text and not chat_result.tool_call_events:
|
|
217
|
+
break
|
|
218
|
+
if _iteration >= max_iterations - 1:
|
|
219
|
+
break
|
|
220
|
+
try:
|
|
208
221
|
current_question = generate_simulated_response(response_text, primary_expected)
|
|
209
|
-
|
|
222
|
+
except SimulatedResponseError as exc:
|
|
223
|
+
print(f"[SIM-USER] Simulated reply failed for conversation {conversation_id}: {exc}")
|
|
210
224
|
break
|
|
211
225
|
|
|
212
226
|
actual_maql = (metric_result or {}).get("maql", "")
|
|
@@ -352,3 +352,118 @@ def test_run_agentic_conversation_treats_alert_proposal_as_a_clarification():
|
|
|
352
352
|
assert "Should I create this alert?" in mock_sim.call_args.args[0]
|
|
353
353
|
assert result.turn_results[0].clarification_turns_used == 1
|
|
354
354
|
assert result.turn_results[0].skill_success is True
|
|
355
|
+
|
|
356
|
+
|
|
357
|
+
def _viz_turn_result(text=None, viz=None, tool_calls=()):
|
|
358
|
+
r = MagicMock()
|
|
359
|
+
r.text_response = text
|
|
360
|
+
r.created_visualizations = viz
|
|
361
|
+
r.tool_call_events = list(tool_calls)
|
|
362
|
+
r.alert_proposals = []
|
|
363
|
+
return r
|
|
364
|
+
|
|
365
|
+
|
|
366
|
+
def test_run_agentic_conversation_replies_to_a_statement_without_a_question_mark():
|
|
367
|
+
"""QA-28982 regression: gpt-5.2 answered "I need to confirm ... Next I'll:" -- no question
|
|
368
|
+
mark, so the old substring heuristic ended the turn and no metric was ever created."""
|
|
369
|
+
mock_client = MagicMock()
|
|
370
|
+
mock_client.create_conversation.return_value = "conv-1"
|
|
371
|
+
stalling_turn = _viz_turn_result(
|
|
372
|
+
text="I can create that, but first I need to confirm which Net Sales calculation to use. Next I'll: ...",
|
|
373
|
+
tool_calls=[_skills_tc("metric")],
|
|
374
|
+
)
|
|
375
|
+
mock_client.send_message.side_effect = [
|
|
376
|
+
stalling_turn,
|
|
377
|
+
_metric_turn_result([_skills_tc("metric"), _create_metric_tc("m1")]),
|
|
378
|
+
]
|
|
379
|
+
|
|
380
|
+
with (
|
|
381
|
+
patch("gooddata_eval.core.agentic.conversation.ChatClient", return_value=mock_client),
|
|
382
|
+
patch("gooddata_eval.core.agentic.conversation.GoodDataSdk"),
|
|
383
|
+
patch(
|
|
384
|
+
"gooddata_eval.core.agentic.conversation._get_sim_user_response",
|
|
385
|
+
return_value="Go ahead with Net Sales.",
|
|
386
|
+
) as mock_sim,
|
|
387
|
+
):
|
|
388
|
+
result = run_agentic_conversation(
|
|
389
|
+
host="http://host/api/v1/actions/workspaces/ws1/ai",
|
|
390
|
+
token="tok",
|
|
391
|
+
workspace_id="ws1",
|
|
392
|
+
fixture=_two_metric_turn_fixture().model_copy(update={"turns": _two_metric_turn_fixture().turns[:1]}),
|
|
393
|
+
)
|
|
394
|
+
|
|
395
|
+
mock_sim.assert_called_once()
|
|
396
|
+
assert result.turn_results[0].clarification_turns_used == 1
|
|
397
|
+
assert result.turn_results[0].skill_success is True
|
|
398
|
+
|
|
399
|
+
|
|
400
|
+
def test_run_agentic_conversation_stops_when_the_agent_says_nothing():
|
|
401
|
+
"""An agent that returns neither text nor tool calls is stuck -- no point replying to it."""
|
|
402
|
+
mock_client = MagicMock()
|
|
403
|
+
mock_client.create_conversation.return_value = "conv-1"
|
|
404
|
+
mock_client.send_message.return_value = _viz_turn_result(text=None)
|
|
405
|
+
|
|
406
|
+
with (
|
|
407
|
+
patch("gooddata_eval.core.agentic.conversation.ChatClient", return_value=mock_client),
|
|
408
|
+
patch("gooddata_eval.core.agentic.conversation.GoodDataSdk"),
|
|
409
|
+
patch("gooddata_eval.core.agentic.conversation._get_sim_user_response") as mock_sim,
|
|
410
|
+
):
|
|
411
|
+
result = run_agentic_conversation(
|
|
412
|
+
host="http://host/api/v1/actions/workspaces/ws1/ai",
|
|
413
|
+
token="tok",
|
|
414
|
+
workspace_id="ws1",
|
|
415
|
+
fixture=_two_metric_turn_fixture().model_copy(update={"turns": _two_metric_turn_fixture().turns[:1]}),
|
|
416
|
+
)
|
|
417
|
+
|
|
418
|
+
mock_sim.assert_not_called()
|
|
419
|
+
assert mock_client.send_message.call_count == 1
|
|
420
|
+
assert result.turn_results[0].skill_success is False
|
|
421
|
+
|
|
422
|
+
|
|
423
|
+
def test_run_agentic_conversation_records_a_failed_turn_when_a_ref_cannot_be_resolved():
|
|
424
|
+
"""QA-28982 regression: turn 1 producing no metric used to raise ValueError out of the whole
|
|
425
|
+
run, hiding which turn broke and skipping every later turn."""
|
|
426
|
+
mock_client = MagicMock()
|
|
427
|
+
mock_client.create_conversation.return_value = "conv-1"
|
|
428
|
+
mock_client.send_message.side_effect = [
|
|
429
|
+
_viz_turn_result(text="Which Net Sales metric?", tool_calls=[_skills_tc("metric")]),
|
|
430
|
+
_viz_turn_result(text="Working on it.", tool_calls=[_skills_tc("metric")]),
|
|
431
|
+
_metric_turn_result([_skills_tc("metric"), _create_metric_tc("m2")]),
|
|
432
|
+
]
|
|
433
|
+
fixture = ConversationFixture(
|
|
434
|
+
id="test-ref",
|
|
435
|
+
expected_skills=["metric"],
|
|
436
|
+
turns=[
|
|
437
|
+
TurnDefinition(
|
|
438
|
+
turn_id="t1", message="Create shared", expected_skill="metric", expected_output_type="metric"
|
|
439
|
+
),
|
|
440
|
+
TurnDefinition(
|
|
441
|
+
turn_id="t2",
|
|
442
|
+
message="Chart it",
|
|
443
|
+
expected_skill="visualization",
|
|
444
|
+
expected_output={"metrics": ["metric/$ref:t1.metric_id"]},
|
|
445
|
+
),
|
|
446
|
+
TurnDefinition(
|
|
447
|
+
turn_id="t3", message="Create another", expected_skill="metric", expected_output_type="metric"
|
|
448
|
+
),
|
|
449
|
+
],
|
|
450
|
+
)
|
|
451
|
+
|
|
452
|
+
with (
|
|
453
|
+
patch("gooddata_eval.core.agentic.conversation.ChatClient", return_value=mock_client),
|
|
454
|
+
patch("gooddata_eval.core.agentic.conversation.GoodDataSdk"),
|
|
455
|
+
patch("gooddata_eval.core.agentic.conversation._get_sim_user_response", return_value="Go ahead."),
|
|
456
|
+
):
|
|
457
|
+
result = run_agentic_conversation(
|
|
458
|
+
host="http://host/api/v1/actions/workspaces/ws1/ai",
|
|
459
|
+
token="tok",
|
|
460
|
+
workspace_id="ws1",
|
|
461
|
+
fixture=fixture,
|
|
462
|
+
max_clarification_turns=1,
|
|
463
|
+
)
|
|
464
|
+
|
|
465
|
+
assert [t.turn_id for t in result.turn_results] == ["t1", "t2", "t3"]
|
|
466
|
+
assert result.turn_results[0].skill_success is False
|
|
467
|
+
assert result.turn_results[1].no_error is False
|
|
468
|
+
assert result.turn_results[2].skill_success is True
|
|
469
|
+
assert result.conversation_success is False
|
|
@@ -1,13 +1,17 @@
|
|
|
1
1
|
# (C) 2026 GoodData Corporation. All rights reserved.
|
|
2
2
|
# SPDX-License-Identifier: LicenseRef-GoodData-Enterprise
|
|
3
|
+
import os
|
|
4
|
+
import sys
|
|
3
5
|
from unittest.mock import MagicMock, patch
|
|
4
6
|
|
|
5
7
|
import pytest
|
|
6
8
|
from gooddata_eval.core.agentic.metric_skill import (
|
|
7
9
|
AgenticMetricSummary,
|
|
8
10
|
MetricRunResult,
|
|
11
|
+
SimulatedResponseError,
|
|
9
12
|
_delete_metric,
|
|
10
13
|
_normalize_maql,
|
|
14
|
+
generate_simulated_response,
|
|
11
15
|
run_agentic_metric_skill,
|
|
12
16
|
)
|
|
13
17
|
from gooddata_eval.core.models import ChatResult
|
|
@@ -83,7 +87,13 @@ def test_run_agentic_metric_skill_closes_client_on_no_result():
|
|
|
83
87
|
"reasoningStepCount": 1,
|
|
84
88
|
}
|
|
85
89
|
)
|
|
86
|
-
with
|
|
90
|
+
with (
|
|
91
|
+
patch("gooddata_eval.core.agentic.metric_skill.ChatClient", return_value=mock_client),
|
|
92
|
+
patch(
|
|
93
|
+
"gooddata_eval.core.agentic.metric_skill.generate_simulated_response",
|
|
94
|
+
return_value="Go ahead and create it.",
|
|
95
|
+
) as mock_sim,
|
|
96
|
+
):
|
|
87
97
|
summary = run_agentic_metric_skill(
|
|
88
98
|
host="http://host/api/v1/actions/workspaces/ws1/ai",
|
|
89
99
|
token="tok",
|
|
@@ -96,6 +106,7 @@ def test_run_agentic_metric_skill_closes_client_on_no_result():
|
|
|
96
106
|
mock_client.close.assert_called_once()
|
|
97
107
|
assert summary.pass_at_k is False
|
|
98
108
|
assert summary.best.metric_created is False
|
|
109
|
+
mock_sim.assert_called_once_with("I will work on that.", {"maql": "SELECT {metric/foo}"})
|
|
99
110
|
|
|
100
111
|
|
|
101
112
|
def test_run_agentic_metric_skill_uses_initial_conversation_for_run_0():
|
|
@@ -224,3 +235,52 @@ def test_run_agentic_metric_skill_deletes_metric_even_when_teardown_fails():
|
|
|
224
235
|
)
|
|
225
236
|
|
|
226
237
|
mock_sdk._client.entities_api.delete_entity_metrics.assert_called_once_with("ws1", "foo_metric")
|
|
238
|
+
|
|
239
|
+
|
|
240
|
+
def test_generate_simulated_response_without_an_api_key():
|
|
241
|
+
with (
|
|
242
|
+
patch.dict(sys.modules, {"openai": MagicMock()}),
|
|
243
|
+
patch.dict(os.environ, {}, clear=True),
|
|
244
|
+
pytest.raises(SimulatedResponseError, match="OPENAI_API_KEY"),
|
|
245
|
+
):
|
|
246
|
+
generate_simulated_response("Which brand field?", {"maql": "SELECT {metric/foo}"})
|
|
247
|
+
|
|
248
|
+
|
|
249
|
+
def test_generate_simulated_response_without_the_openai_package():
|
|
250
|
+
with (
|
|
251
|
+
patch.dict(sys.modules, {"openai": None}),
|
|
252
|
+
pytest.raises(SimulatedResponseError, match="openai package is required"),
|
|
253
|
+
):
|
|
254
|
+
generate_simulated_response("Which brand field?", {"maql": "SELECT {metric/foo}"})
|
|
255
|
+
|
|
256
|
+
|
|
257
|
+
def test_run_agentic_metric_skill_fails_the_run_when_the_simulated_reply_cannot_be_generated():
|
|
258
|
+
exc = SimulatedResponseError("OPENAI_API_KEY environment variable is not set")
|
|
259
|
+
mock_client = MagicMock()
|
|
260
|
+
mock_client.create_conversation.return_value = "conv-1"
|
|
261
|
+
mock_client.send_message.return_value = ChatResult.model_validate(
|
|
262
|
+
{
|
|
263
|
+
"textResponse": "Which brand field should I count?",
|
|
264
|
+
"toolCallEvents": [],
|
|
265
|
+
"reasoningStepCount": 1,
|
|
266
|
+
}
|
|
267
|
+
)
|
|
268
|
+
with (
|
|
269
|
+
patch("gooddata_eval.core.agentic.metric_skill.ChatClient", return_value=mock_client),
|
|
270
|
+
patch("gooddata_eval.core.agentic.metric_skill.generate_simulated_response", side_effect=exc) as mock_sim,
|
|
271
|
+
):
|
|
272
|
+
summary = run_agentic_metric_skill(
|
|
273
|
+
host="http://host/api/v1/actions/workspaces/ws1/ai",
|
|
274
|
+
token="tok",
|
|
275
|
+
workspace_id="ws1",
|
|
276
|
+
question="Create metric foo",
|
|
277
|
+
expected_output={"maql": "SELECT {metric/foo}"},
|
|
278
|
+
k=1,
|
|
279
|
+
max_iterations=3,
|
|
280
|
+
)
|
|
281
|
+
|
|
282
|
+
assert summary.pass_at_k is False
|
|
283
|
+
assert summary.best.metric_created is False
|
|
284
|
+
assert summary.best.total_turns == 1.0
|
|
285
|
+
mock_client.close.assert_called_once()
|
|
286
|
+
mock_sim.assert_called_once_with("Which brand field should I count?", {"maql": "SELECT {metric/foo}"})
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/src/gooddata_eval/cli/agentic_runner.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
{gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/src/gooddata_eval/core/agentic/__init__.py
RENAMED
|
File without changes
|
{gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/src/gooddata_eval/core/agentic/_catalog.py
RENAMED
|
File without changes
|
{gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/src/gooddata_eval/core/agentic/_langfuse.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
{gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/src/gooddata_eval/core/agentic/guardrail.py
RENAMED
|
File without changes
|
{gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/src/gooddata_eval/core/agentic/kda_skill.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
{gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/src/gooddata_eval/core/chat/__init__.py
RENAMED
|
File without changes
|
{gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/src/gooddata_eval/core/chat/sse_client.py
RENAMED
|
File without changes
|
|
File without changes
|
{gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/src/gooddata_eval/core/connection.py
RENAMED
|
File without changes
|
{gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/src/gooddata_eval/core/dataset/__init__.py
RENAMED
|
File without changes
|
|
File without changes
|
{gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/src/gooddata_eval/core/dataset/local.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/src/gooddata_eval/core/evaluators/base.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/src/gooddata_eval/core/evaluators/summary.py
RENAMED
|
File without changes
|
|
File without changes
|
{gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/src/gooddata_eval/core/langfuse/__init__.py
RENAMED
|
File without changes
|
{gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/src/gooddata_eval/core/langfuse/sink.py
RENAMED
|
File without changes
|
|
File without changes
|
{gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/src/gooddata_eval/core/reporting/__init__.py
RENAMED
|
File without changes
|
{gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/src/gooddata_eval/core/reporting/console.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/src/gooddata_eval/core/summary/__init__.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/tests/fixtures/sse_visualization_stream.txt
RENAMED
|
File without changes
|
|
File without changes
|
{gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/tests/test_agentic_general_question.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
{gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/tests/test_agentic_langfuse_trace.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/tests/test_metric_skill_evaluator.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{gooddata_eval-1.72.1.dev5 → gooddata_eval-1.72.1.dev6}/tests/test_visualization_evaluator.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|