gooddata-eval 1.72.1.dev3__tar.gz → 1.72.1.dev4__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/PKG-INFO +2 -2
- {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/pyproject.toml +2 -2
- {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/agentic/kda_skill.py +89 -49
- {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/tests/test_agentic_kda_skill.py +167 -36
- {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/.gitignore +0 -0
- {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/LICENSE.txt +0 -0
- {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/Makefile +0 -0
- {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/README.md +0 -0
- {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/__init__.py +0 -0
- {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/_version.py +0 -0
- {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/cli/__init__.py +0 -0
- {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/cli/agentic_runner.py +0 -0
- {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/cli/main.py +0 -0
- {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/__init__.py +0 -0
- {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/agentic/__init__.py +0 -0
- {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/agentic/_catalog.py +0 -0
- {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/agentic/_langfuse.py +0 -0
- {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/agentic/alert_skill.py +0 -0
- {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/agentic/conversation.py +0 -0
- {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/agentic/general_question.py +0 -0
- {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/agentic/guardrail.py +0 -0
- {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/agentic/metric_skill.py +0 -0
- {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/agentic/search_tool.py +0 -0
- {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/agentic/visualization.py +0 -0
- {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/chat/__init__.py +0 -0
- {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/chat/sse_client.py +0 -0
- {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/config.py +0 -0
- {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/connection.py +0 -0
- {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/dataset/__init__.py +0 -0
- {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/dataset/langfuse_source.py +0 -0
- {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/dataset/local.py +0 -0
- {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/evaluators/__init__.py +0 -0
- {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/evaluators/_deep_subset.py +0 -0
- {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/evaluators/_llm_judge.py +0 -0
- {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/evaluators/_text_utils.py +0 -0
- {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/evaluators/alert_skill.py +0 -0
- {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/evaluators/base.py +0 -0
- {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/evaluators/general_question.py +0 -0
- {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/evaluators/guardrail.py +0 -0
- {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/evaluators/metric_skill.py +0 -0
- {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/evaluators/search_tool.py +0 -0
- {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/evaluators/summary.py +0 -0
- {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/evaluators/visualization.py +0 -0
- {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/langfuse/__init__.py +0 -0
- {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/langfuse/sink.py +0 -0
- {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/models.py +0 -0
- {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/reporting/__init__.py +0 -0
- {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/reporting/console.py +0 -0
- {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/reporting/json_report.py +0 -0
- {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/runner.py +0 -0
- {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/scoring.py +0 -0
- {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/summary/__init__.py +0 -0
- {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/summary/http_client.py +0 -0
- {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/workspace.py +0 -0
- {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/tests/__init__.py +0 -0
- {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/tests/conftest.py +0 -0
- {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/tests/fixtures/sample_dataset/metric_skill_create.json +0 -0
- {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/tests/fixtures/sample_dataset/visualization_revenue.json +0 -0
- {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/tests/fixtures/sse_visualization_stream.txt +0 -0
- {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/tests/test_agentic_alert_skill.py +0 -0
- {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/tests/test_agentic_conversation.py +0 -0
- {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/tests/test_agentic_general_question.py +0 -0
- {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/tests/test_agentic_guardrail.py +0 -0
- {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/tests/test_agentic_langfuse_trace.py +0 -0
- {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/tests/test_agentic_metric_skill.py +0 -0
- {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/tests/test_agentic_run_context.py +0 -0
- {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/tests/test_agentic_search_tool.py +0 -0
- {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/tests/test_agentic_visualization.py +0 -0
- {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/tests/test_alert_skill_evaluator.py +0 -0
- {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/tests/test_cli.py +0 -0
- {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/tests/test_connection.py +0 -0
- {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/tests/test_deep_subset.py +0 -0
- {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/tests/test_langfuse_sink.py +0 -0
- {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/tests/test_langfuse_source.py +0 -0
- {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/tests/test_llm_judge.py +0 -0
- {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/tests/test_local_loader.py +0 -0
- {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/tests/test_metric_skill_evaluator.py +0 -0
- {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/tests/test_models.py +0 -0
- {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/tests/test_reporting.py +0 -0
- {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/tests/test_runner.py +0 -0
- {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/tests/test_scoring.py +0 -0
- {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/tests/test_search_tool_evaluator.py +0 -0
- {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/tests/test_sse_client.py +0 -0
- {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/tests/test_summary_client.py +0 -0
- {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/tests/test_summary_evaluator.py +0 -0
- {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/tests/test_text_evaluators.py +0 -0
- {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/tests/test_visualization_evaluator.py +0 -0
- {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/tests/test_workspace.py +0 -0
- {gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/tox.ini +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: gooddata-eval
|
|
3
|
-
Version: 1.72.1.
|
|
3
|
+
Version: 1.72.1.dev4
|
|
4
4
|
Summary: Evaluate the GoodData AI agent against your own questions and models.
|
|
5
5
|
Project-URL: Source, https://github.com/gooddata/gooddata-python-sdk
|
|
6
6
|
Author-email: GoodData <support@gooddata.com>
|
|
@@ -17,7 +17,7 @@ Classifier: Topic :: Scientific/Engineering
|
|
|
17
17
|
Classifier: Topic :: Software Development
|
|
18
18
|
Classifier: Typing :: Typed
|
|
19
19
|
Requires-Python: >=3.10
|
|
20
|
-
Requires-Dist: gooddata-sdk~=1.72.1.
|
|
20
|
+
Requires-Dist: gooddata-sdk~=1.72.1.dev4
|
|
21
21
|
Requires-Dist: httpx<1.0,>=0.27
|
|
22
22
|
Requires-Dist: orjson<4.0.0,>=3.9.15
|
|
23
23
|
Requires-Dist: pydantic<3.0,>=2.6
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
# (C) 2026 GoodData Corporation
|
|
2
2
|
[project]
|
|
3
3
|
name = "gooddata-eval"
|
|
4
|
-
version = "1.72.1.
|
|
4
|
+
version = "1.72.1.dev4"
|
|
5
5
|
description = "Evaluate the GoodData AI agent against your own questions and models."
|
|
6
6
|
readme = "README.md"
|
|
7
7
|
license = "MIT"
|
|
@@ -11,7 +11,7 @@ authors = [
|
|
|
11
11
|
keywords = ["gooddata", "ai", "evaluation", "llm", "analytics", "cli"]
|
|
12
12
|
requires-python = ">=3.10"
|
|
13
13
|
dependencies = [
|
|
14
|
-
"gooddata-sdk~=1.72.1.
|
|
14
|
+
"gooddata-sdk~=1.72.1.dev4",
|
|
15
15
|
"httpx>=0.27,<1.0",
|
|
16
16
|
"orjson>=3.9.15,<4.0.0",
|
|
17
17
|
"pydantic>=2.6,<3.0",
|
{gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/agentic/kda_skill.py
RENAMED
|
@@ -5,7 +5,6 @@ from __future__ import annotations
|
|
|
5
5
|
|
|
6
6
|
import logging
|
|
7
7
|
import os
|
|
8
|
-
import re
|
|
9
8
|
from dataclasses import dataclass
|
|
10
9
|
|
|
11
10
|
from gooddata_eval.core.chat.sse_client import ChatClient
|
|
@@ -15,36 +14,79 @@ from gooddata_eval.core.models import ToolCallEvent
|
|
|
15
14
|
_log = logging.getLogger(__name__)
|
|
16
15
|
|
|
17
16
|
_DEFAULT_K = 1
|
|
18
|
-
# Disambiguation safety net only (create+execute always run together in the same
|
|
19
|
-
#
|
|
20
|
-
|
|
17
|
+
# Disambiguation safety net only (create+execute always run together in the same turn) --
|
|
18
|
+
# 3 real questions' worth (metric, period, +1 slack) since a simulated reply is now sent on
|
|
19
|
+
# every non-final turn (see run_agentic_kda_skill), not just ones classified as a question.
|
|
20
|
+
_DEFAULT_MAX_ITERATIONS = 4
|
|
21
21
|
|
|
22
22
|
|
|
23
|
-
def
|
|
24
|
-
"""
|
|
25
|
-
|
|
26
|
-
|
|
27
|
-
|
|
28
|
-
|
|
23
|
+
def _build_period_hint(expected_output: dict) -> str | None:
|
|
24
|
+
"""Build a period hint from whichever of expected_output's Date Attribute/Analyzed
|
|
25
|
+
Period/Reference Period are present -- a question about only one of them (e.g. "which
|
|
26
|
+
date dimension?") must still get an answerable hint, not None just because the other
|
|
27
|
+
two are absent.
|
|
28
|
+
"""
|
|
29
|
+
date_attr = expected_output.get("Date Attribute")
|
|
30
|
+
analyzed = expected_output.get("Analyzed Period")
|
|
31
|
+
reference_period = expected_output.get("Reference Period")
|
|
32
|
+
if not (date_attr or analyzed or reference_period):
|
|
33
|
+
return None
|
|
34
|
+
parts = []
|
|
35
|
+
if date_attr:
|
|
36
|
+
parts.append(date_attr)
|
|
37
|
+
if analyzed and reference_period:
|
|
38
|
+
parts.append(f"comparing {analyzed} to {reference_period}")
|
|
39
|
+
elif analyzed:
|
|
40
|
+
parts.append(f"period {analyzed}")
|
|
41
|
+
elif reference_period:
|
|
42
|
+
parts.append(f"compared to {reference_period}")
|
|
43
|
+
return ", ".join(parts)
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def _build_clarification_prompt(
|
|
47
|
+
agent_message: str, measure_candidates: dict | list[dict] | None, period_hint: str | None
|
|
48
|
+
) -> str:
|
|
49
|
+
"""Build the simulated-user prompt, referencing only whatever candidates/period-hint
|
|
50
|
+
are actually usable -- an empty/None candidate must drop the "acceptable metric/fact"
|
|
51
|
+
clause entirely rather than assert a literal "None" as if it were a real option.
|
|
29
52
|
"""
|
|
30
|
-
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
|
|
34
|
-
|
|
35
|
-
|
|
36
|
-
|
|
37
|
-
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
|
|
53
|
+
candidates = [
|
|
54
|
+
c for c in (measure_candidates if isinstance(measure_candidates, list) else [measure_candidates]) if c
|
|
55
|
+
]
|
|
56
|
+
reference = ""
|
|
57
|
+
if candidates:
|
|
58
|
+
candidate_desc = "; or ".join(
|
|
59
|
+
f"{c.get('type')} '{c.get('id')}'" + (f" (aggregation {c['aggregation']})" if c.get("aggregation") else "")
|
|
60
|
+
for c in candidates
|
|
61
|
+
)
|
|
62
|
+
reference = f"an acceptable metric/fact is {candidate_desc}"
|
|
63
|
+
if period_hint:
|
|
64
|
+
reference = (
|
|
65
|
+
f"{reference}; the intended time period is {period_hint}"
|
|
66
|
+
if reference
|
|
67
|
+
else f"the intended time period is {period_hint}"
|
|
68
|
+
)
|
|
69
|
+
return (
|
|
70
|
+
f"You are simulating a user in a conversation with a BI assistant that runs key driver "
|
|
71
|
+
f"analysis. The assistant asked: '{agent_message}'. "
|
|
72
|
+
+ (f"For reference, {reference}. " if reference else "")
|
|
73
|
+
+ "Reply briefly as the user, answering whichever of those the assistant actually asked about."
|
|
74
|
+
)
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def generate_simulated_kda_response(
|
|
78
|
+
agent_message: str,
|
|
79
|
+
measure_candidates: dict | list[dict] | None,
|
|
80
|
+
period_hint: str | None = None,
|
|
81
|
+
) -> str:
|
|
42
82
|
"""Generate a user reply to keep the KDA-skill conversation going (gpt-4o-mini).
|
|
43
83
|
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
|
|
84
|
+
Called on any turn that didn't trigger KDA, whatever the agent's response actually
|
|
85
|
+
said -- most often a clarifying question about the measure, the period, or both, so
|
|
86
|
+
both are given as reference and the reply answers whichever was actually asked.
|
|
87
|
+
Scope only needs KDA to trigger, not the resulting measure/period to be exactly
|
|
88
|
+
right. Always OpenAI regardless of the combo's own provider -- this is
|
|
89
|
+
test-harness plumbing, not the system under test.
|
|
48
90
|
"""
|
|
49
91
|
try:
|
|
50
92
|
from openai import OpenAI # noqa: PLC0415
|
|
@@ -56,17 +98,7 @@ def generate_simulated_kda_response(agent_message: str, measure_candidates: dict
|
|
|
56
98
|
raise OSError("OPENAI_API_KEY environment variable is not set")
|
|
57
99
|
|
|
58
100
|
client = OpenAI(api_key=api_key)
|
|
59
|
-
|
|
60
|
-
candidate_desc = "; or ".join(
|
|
61
|
-
f"{c.get('type')} '{c.get('id')}'" + (f" (aggregation {c['aggregation']})" if c.get("aggregation") else "")
|
|
62
|
-
for c in candidates
|
|
63
|
-
)
|
|
64
|
-
prompt = (
|
|
65
|
-
f"You are simulating a user in a conversation with a BI assistant that runs key driver "
|
|
66
|
-
f"analysis. The assistant said: '{agent_message}'. "
|
|
67
|
-
f"The user is happy to proceed with any of the following: {candidate_desc}. "
|
|
68
|
-
f"Reply briefly as the user, picking whichever of those the assistant offered."
|
|
69
|
-
)
|
|
101
|
+
prompt = _build_clarification_prompt(agent_message, measure_candidates, period_hint)
|
|
70
102
|
response = client.chat.completions.create(
|
|
71
103
|
model="gpt-4o-mini",
|
|
72
104
|
messages=[{"role": "user", "content": prompt}],
|
|
@@ -171,9 +203,12 @@ def run_agentic_kda_skill(
|
|
|
171
203
|
|
|
172
204
|
Each run is normally one message, one turn -- create and execute are always called
|
|
173
205
|
together in the same turn (the skill's own system prompt: "NO confirmation needed").
|
|
174
|
-
|
|
175
|
-
|
|
176
|
-
|
|
206
|
+
A run only extends past turn 1, up to ``max_iterations``, when the agent's response
|
|
207
|
+
has no create call and isn't empty; a simulated user reply is then always sent, with
|
|
208
|
+
no attempt to classify whether the text was actually asking for input (matching
|
|
209
|
+
visualization.py/alert_skill.py's own break conditions) -- missing a genuine
|
|
210
|
+
clarifying question hard-fails the run, while sending one after an unrecognized final
|
|
211
|
+
answer only costs one harmless extra turn, so the asymmetry favors never guessing.
|
|
177
212
|
"""
|
|
178
213
|
if k < 1:
|
|
179
214
|
# k=0 or negative would otherwise silently run once, indistinguishable from k=1.
|
|
@@ -212,17 +247,22 @@ def run_agentic_kda_skill(
|
|
|
212
247
|
# execute tool isn't available at all when data-sharing is off for the org).
|
|
213
248
|
turn_wall_clock_sec = chat_result.turn_wall_clock_sec
|
|
214
249
|
break
|
|
250
|
+
if not response_text:
|
|
251
|
+
break
|
|
215
252
|
if iteration >= max_iterations - 1:
|
|
216
253
|
break
|
|
217
|
-
|
|
218
|
-
|
|
219
|
-
|
|
220
|
-
|
|
221
|
-
|
|
222
|
-
|
|
223
|
-
|
|
224
|
-
|
|
225
|
-
|
|
254
|
+
# No text classification -- matches visualization.py/alert_skill.py: break only on
|
|
255
|
+
# the goal signal (create_args set) or an empty response, otherwise always send a
|
|
256
|
+
# simulated reply. A false positive (agent had already given a final answer) costs
|
|
257
|
+
# one harmless extra turn; a false negative (missing a genuine clarifying question)
|
|
258
|
+
# would hard-fail the run, so the asymmetry favors never trying to tell them apart.
|
|
259
|
+
measure_candidates = expected_output.get("Measure") if isinstance(expected_output, dict) else None
|
|
260
|
+
period_hint = _build_period_hint(expected_output) if isinstance(expected_output, dict) else None
|
|
261
|
+
try:
|
|
262
|
+
current_question = generate_simulated_kda_response(response_text, measure_candidates, period_hint)
|
|
263
|
+
disambiguated = True
|
|
264
|
+
except Exception as exc: # noqa: BLE001 -- safety net, not the assertion; end only this run
|
|
265
|
+
_log.warning("Simulated KDA user reply failed for conversation %s: %s", conv_id, exc)
|
|
226
266
|
break
|
|
227
267
|
|
|
228
268
|
ev = _evaluate_run(create_args, execute_result, turn_completed, disambiguated)
|
|
@@ -8,9 +8,10 @@ import pytest
|
|
|
8
8
|
from gooddata_eval.core.agentic.kda_skill import (
|
|
9
9
|
KdaEvaluation,
|
|
10
10
|
KdaSkillAssertionError,
|
|
11
|
+
_build_clarification_prompt,
|
|
12
|
+
_build_period_hint,
|
|
11
13
|
_evaluate_run,
|
|
12
14
|
_extract_kda_calls,
|
|
13
|
-
_is_asking_kda_clarification,
|
|
14
15
|
evaluate_agentic_kda_skill,
|
|
15
16
|
run_agentic_kda_skill,
|
|
16
17
|
)
|
|
@@ -67,51 +68,59 @@ def _no_kda_chat_result(
|
|
|
67
68
|
|
|
68
69
|
|
|
69
70
|
# --------------------------------------------------------------------------- #
|
|
70
|
-
#
|
|
71
|
+
# _build_clarification_prompt
|
|
71
72
|
# --------------------------------------------------------------------------- #
|
|
72
|
-
|
|
73
|
-
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
|
|
77
|
-
|
|
73
|
+
def test_build_clarification_prompt_omits_reference_clause_when_no_candidates_or_period():
|
|
74
|
+
# Regression (chi My's review): with no usable candidates, the old code still asserted
|
|
75
|
+
# "an acceptable metric/fact is None 'None'" as if it were a real option -- likely to
|
|
76
|
+
# make the simulated user invent a metric literally named "None". No candidates and no
|
|
77
|
+
# period hint must drop the whole "For reference, ..." clause instead.
|
|
78
|
+
prompt = _build_clarification_prompt("Which date range?", None, None)
|
|
79
|
+
assert "None" not in prompt
|
|
80
|
+
assert "For reference" not in prompt
|
|
78
81
|
|
|
79
82
|
|
|
80
|
-
def
|
|
81
|
-
|
|
83
|
+
def test_build_clarification_prompt_includes_only_period_hint_when_no_candidates():
|
|
84
|
+
prompt = _build_clarification_prompt("Which period?", None, "2026-2 vs 2026-1")
|
|
85
|
+
assert "None" not in prompt
|
|
86
|
+
assert "the intended time period is 2026-2 vs 2026-1" in prompt
|
|
82
87
|
|
|
83
88
|
|
|
84
|
-
def
|
|
85
|
-
|
|
89
|
+
def test_build_clarification_prompt_includes_candidates_and_period_hint():
|
|
90
|
+
prompt = _build_clarification_prompt(
|
|
91
|
+
"Which metric and period?", {"type": "metric", "id": "revenue"}, "2026-2 vs 2026-1"
|
|
92
|
+
)
|
|
93
|
+
assert "an acceptable metric/fact is metric 'revenue'" in prompt
|
|
94
|
+
assert "the intended time period is 2026-2 vs 2026-1" in prompt
|
|
86
95
|
|
|
87
96
|
|
|
88
|
-
|
|
89
|
-
|
|
90
|
-
|
|
91
|
-
|
|
92
|
-
assert
|
|
97
|
+
# --------------------------------------------------------------------------- #
|
|
98
|
+
# _build_period_hint
|
|
99
|
+
# --------------------------------------------------------------------------- #
|
|
100
|
+
def test_build_period_hint_none_when_no_period_fields_present():
|
|
101
|
+
assert _build_period_hint({"Measure": {"type": "metric", "id": "revenue"}}) is None
|
|
93
102
|
|
|
94
103
|
|
|
95
|
-
|
|
96
|
-
|
|
97
|
-
|
|
98
|
-
|
|
99
|
-
|
|
100
|
-
],
|
|
101
|
-
)
|
|
102
|
-
def test_is_asking_kda_clarification_false_on_to_clarify_discourse_marker(text):
|
|
103
|
-
# Regression guard: "to clarify, ..." is a discourse marker ("in other words") that
|
|
104
|
-
# introduces a restated FINAL answer, not a request for one -- the bare "clarif" in t
|
|
105
|
-
# substring check would otherwise mistake this for a clarifying question and burn a
|
|
106
|
-
# simulated-reply turn on an answer that was already complete.
|
|
107
|
-
assert _is_asking_kda_clarification(text) is False
|
|
104
|
+
def test_build_period_hint_all_three_fields():
|
|
105
|
+
hint = _build_period_hint(
|
|
106
|
+
{"Date Attribute": "transaction_date.quarter", "Analyzed Period": "2026-2", "Reference Period": "2026-1"}
|
|
107
|
+
)
|
|
108
|
+
assert hint == "transaction_date.quarter, comparing 2026-2 to 2026-1"
|
|
108
109
|
|
|
109
110
|
|
|
110
|
-
def
|
|
111
|
-
#
|
|
112
|
-
#
|
|
113
|
-
# the
|
|
114
|
-
assert
|
|
111
|
+
def test_build_period_hint_date_attribute_only():
|
|
112
|
+
# A dataset item carrying only Date Attribute (agent asks "which date dimension should
|
|
113
|
+
# I use?") must still get an answerable hint -- this used to require all 3 fields and
|
|
114
|
+
# reproduced the same gap the metric-clarification fix closed, just narrower.
|
|
115
|
+
assert _build_period_hint({"Date Attribute": "transaction_date.quarter"}) == "transaction_date.quarter"
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
def test_build_period_hint_analyzed_period_only():
|
|
119
|
+
assert _build_period_hint({"Analyzed Period": "2026-2"}) == "period 2026-2"
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
def test_build_period_hint_reference_period_only():
|
|
123
|
+
assert _build_period_hint({"Reference Period": "2026-1"}) == "compared to 2026-1"
|
|
115
124
|
|
|
116
125
|
|
|
117
126
|
# --------------------------------------------------------------------------- #
|
|
@@ -427,6 +436,128 @@ def test_run_agentic_kda_skill_marks_disambiguated_after_a_simulated_reply():
|
|
|
427
436
|
assert summary.best.evaluation.triggered is True
|
|
428
437
|
|
|
429
438
|
|
|
439
|
+
def test_run_agentic_kda_skill_disambiguates_on_question_followed_by_option_list():
|
|
440
|
+
# Regression (QA-28800): the real captured response ends with a bullet list of
|
|
441
|
+
# candidate metrics. Before this module dropped text classification in favor of
|
|
442
|
+
# always retrying on a non-empty, non-triggering response (matching
|
|
443
|
+
# visualization.py/alert_skill.py), a heuristic that only matched "?" endings gave
|
|
444
|
+
# up after turn 1 (triggered=False) instead of ever nudging the simulated user to
|
|
445
|
+
# pick one.
|
|
446
|
+
mock_client = MagicMock()
|
|
447
|
+
mock_client.create_conversation.return_value = "conv-1"
|
|
448
|
+
mock_client.send_message.side_effect = [
|
|
449
|
+
_no_kda_chat_result(
|
|
450
|
+
'I found two different "Total Net Revenue" metrics in your data model. '
|
|
451
|
+
"Which one should I analyze for the 2024 vs 2023 drop?\n\n"
|
|
452
|
+
"- {metric/metric_l1_sql_net_sales_summary_net_revenue}\n"
|
|
453
|
+
"- {metric/metric_l1_total_net_revenue}"
|
|
454
|
+
),
|
|
455
|
+
_kda_chat_result(success=True),
|
|
456
|
+
]
|
|
457
|
+
|
|
458
|
+
with (
|
|
459
|
+
patch("gooddata_eval.core.agentic.kda_skill.ChatClient", return_value=mock_client),
|
|
460
|
+
patch(
|
|
461
|
+
"gooddata_eval.core.agentic.kda_skill.generate_simulated_kda_response",
|
|
462
|
+
return_value="Use metric_l1_sql_net_sales_summary_net_revenue.",
|
|
463
|
+
) as mock_simulate,
|
|
464
|
+
):
|
|
465
|
+
summary = run_agentic_kda_skill(
|
|
466
|
+
host="http://host/api/v1/actions/workspaces/ws1/ai",
|
|
467
|
+
token="tok",
|
|
468
|
+
workspace_id="ws1",
|
|
469
|
+
question="Why did Total Net Revenue of Net Sales Summary drop in 2024 compared to 2023?",
|
|
470
|
+
expected_output=_EXPECTED,
|
|
471
|
+
k=1,
|
|
472
|
+
max_iterations=2,
|
|
473
|
+
)
|
|
474
|
+
|
|
475
|
+
mock_simulate.assert_called_once()
|
|
476
|
+
assert summary.best.evaluation.disambiguated is True
|
|
477
|
+
assert summary.best.evaluation.triggered is True
|
|
478
|
+
assert mock_client.send_message.call_count == 2
|
|
479
|
+
|
|
480
|
+
|
|
481
|
+
def test_run_agentic_kda_skill_retries_on_bold_markdown_option_list_with_no_space():
|
|
482
|
+
# Regression (chi My's review): a prior classifier-based fix required a space right
|
|
483
|
+
# after the list marker, so "**Option 1**: revenue" (bold markdown, no space between
|
|
484
|
+
# the two asterisks) would have been misread as a final answer. Dropping content
|
|
485
|
+
# classification entirely (see run_agentic_kda_skill's docstring) makes this -- and any
|
|
486
|
+
# other future response shape -- a non-issue: a non-triggering, non-empty response
|
|
487
|
+
# always gets a simulated reply now, regardless of how it's formatted.
|
|
488
|
+
mock_client = MagicMock()
|
|
489
|
+
mock_client.create_conversation.return_value = "conv-1"
|
|
490
|
+
mock_client.send_message.side_effect = [
|
|
491
|
+
_no_kda_chat_result("Which one should I analyze?\n**Option 1**: revenue\n**Option 2**: gross profit"),
|
|
492
|
+
_kda_chat_result(success=True),
|
|
493
|
+
]
|
|
494
|
+
|
|
495
|
+
with (
|
|
496
|
+
patch("gooddata_eval.core.agentic.kda_skill.ChatClient", return_value=mock_client),
|
|
497
|
+
patch(
|
|
498
|
+
"gooddata_eval.core.agentic.kda_skill.generate_simulated_kda_response",
|
|
499
|
+
return_value="Use revenue.",
|
|
500
|
+
) as mock_simulate,
|
|
501
|
+
):
|
|
502
|
+
summary = run_agentic_kda_skill(
|
|
503
|
+
host="http://host/api/v1/actions/workspaces/ws1/ai",
|
|
504
|
+
token="tok",
|
|
505
|
+
workspace_id="ws1",
|
|
506
|
+
question="Why did revenue drop?",
|
|
507
|
+
expected_output=_EXPECTED,
|
|
508
|
+
k=1,
|
|
509
|
+
max_iterations=2,
|
|
510
|
+
)
|
|
511
|
+
|
|
512
|
+
mock_simulate.assert_called_once()
|
|
513
|
+
assert summary.best.evaluation.disambiguated is True
|
|
514
|
+
assert summary.best.evaluation.triggered is True
|
|
515
|
+
|
|
516
|
+
|
|
517
|
+
def test_run_agentic_kda_skill_disambiguates_on_period_clarification():
|
|
518
|
+
# generate_simulated_kda_response used to only know about measure candidates -- if the
|
|
519
|
+
# agent asked about the PERIOD instead, it had nothing period-specific to answer with.
|
|
520
|
+
# Verify the period hint built from expected_output's Date Attribute/Analyzed
|
|
521
|
+
# Period/Reference Period reaches the simulated-reply call.
|
|
522
|
+
expected_output = {
|
|
523
|
+
"Measure": {"type": "metric", "id": "revenue"},
|
|
524
|
+
"Date Attribute": "transaction_date.quarter",
|
|
525
|
+
"Analyzed Period": "2026-2",
|
|
526
|
+
"Reference Period": "2026-1",
|
|
527
|
+
}
|
|
528
|
+
mock_client = MagicMock()
|
|
529
|
+
mock_client.create_conversation.return_value = "conv-1"
|
|
530
|
+
mock_client.send_message.side_effect = [
|
|
531
|
+
_no_kda_chat_result("Which period would you like to compare?"),
|
|
532
|
+
_kda_chat_result(success=True),
|
|
533
|
+
]
|
|
534
|
+
|
|
535
|
+
with (
|
|
536
|
+
patch("gooddata_eval.core.agentic.kda_skill.ChatClient", return_value=mock_client),
|
|
537
|
+
patch(
|
|
538
|
+
"gooddata_eval.core.agentic.kda_skill.generate_simulated_kda_response",
|
|
539
|
+
return_value="Compare 2026-2 to 2026-1.",
|
|
540
|
+
) as mock_simulate,
|
|
541
|
+
):
|
|
542
|
+
summary = run_agentic_kda_skill(
|
|
543
|
+
host="http://host/api/v1/actions/workspaces/ws1/ai",
|
|
544
|
+
token="tok",
|
|
545
|
+
workspace_id="ws1",
|
|
546
|
+
question="Why did revenue drop?",
|
|
547
|
+
expected_output=expected_output,
|
|
548
|
+
k=1,
|
|
549
|
+
max_iterations=2,
|
|
550
|
+
)
|
|
551
|
+
|
|
552
|
+
mock_simulate.assert_called_once_with(
|
|
553
|
+
"Which period would you like to compare?",
|
|
554
|
+
{"type": "metric", "id": "revenue"},
|
|
555
|
+
"transaction_date.quarter, comparing 2026-2 to 2026-1",
|
|
556
|
+
)
|
|
557
|
+
assert summary.best.evaluation.disambiguated is True
|
|
558
|
+
assert summary.best.evaluation.triggered is True
|
|
559
|
+
|
|
560
|
+
|
|
430
561
|
def test_run_agentic_kda_skill_disambiguates_when_expected_output_is_not_a_dict():
|
|
431
562
|
# DatasetItem.expected_output on the gdc-nas side allows str/list, not just dict.
|
|
432
563
|
# expected_output.get("Measure") would raise AttributeError on those shapes, silently
|
|
@@ -457,7 +588,7 @@ def test_run_agentic_kda_skill_disambiguates_when_expected_output_is_not_a_dict(
|
|
|
457
588
|
max_iterations=2,
|
|
458
589
|
)
|
|
459
590
|
|
|
460
|
-
mock_generate.assert_called_once_with("Could you clarify which measure?", None)
|
|
591
|
+
mock_generate.assert_called_once_with("Could you clarify which measure?", None, None)
|
|
461
592
|
assert summary.best.evaluation.disambiguated is True
|
|
462
593
|
assert summary.best.evaluation.triggered is True
|
|
463
594
|
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/cli/agentic_runner.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
{gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/agentic/__init__.py
RENAMED
|
File without changes
|
{gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/agentic/_catalog.py
RENAMED
|
File without changes
|
{gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/agentic/_langfuse.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/agentic/guardrail.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/chat/__init__.py
RENAMED
|
File without changes
|
{gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/chat/sse_client.py
RENAMED
|
File without changes
|
|
File without changes
|
{gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/connection.py
RENAMED
|
File without changes
|
{gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/dataset/__init__.py
RENAMED
|
File without changes
|
|
File without changes
|
{gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/dataset/local.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/evaluators/base.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/evaluators/summary.py
RENAMED
|
File without changes
|
|
File without changes
|
{gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/langfuse/__init__.py
RENAMED
|
File without changes
|
{gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/langfuse/sink.py
RENAMED
|
File without changes
|
|
File without changes
|
{gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/reporting/__init__.py
RENAMED
|
File without changes
|
{gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/reporting/console.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/summary/__init__.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/tests/fixtures/sse_visualization_stream.txt
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
{gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/tests/test_agentic_general_question.py
RENAMED
|
File without changes
|
|
File without changes
|
{gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/tests/test_agentic_langfuse_trace.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/tests/test_metric_skill_evaluator.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{gooddata_eval-1.72.1.dev3 → gooddata_eval-1.72.1.dev4}/tests/test_visualization_evaluator.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|