gooddata-eval 1.72.1.dev3__py3-none-any.whl → 1.72.1.dev4__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- gooddata_eval/core/agentic/kda_skill.py +89 -49
- {gooddata_eval-1.72.1.dev3.dist-info → gooddata_eval-1.72.1.dev4.dist-info}/METADATA +2 -2
- {gooddata_eval-1.72.1.dev3.dist-info → gooddata_eval-1.72.1.dev4.dist-info}/RECORD +6 -6
- {gooddata_eval-1.72.1.dev3.dist-info → gooddata_eval-1.72.1.dev4.dist-info}/WHEEL +0 -0
- {gooddata_eval-1.72.1.dev3.dist-info → gooddata_eval-1.72.1.dev4.dist-info}/entry_points.txt +0 -0
- {gooddata_eval-1.72.1.dev3.dist-info → gooddata_eval-1.72.1.dev4.dist-info}/licenses/LICENSE.txt +0 -0
|
@@ -5,7 +5,6 @@ from __future__ import annotations
|
|
|
5
5
|
|
|
6
6
|
import logging
|
|
7
7
|
import os
|
|
8
|
-
import re
|
|
9
8
|
from dataclasses import dataclass
|
|
10
9
|
|
|
11
10
|
from gooddata_eval.core.chat.sse_client import ChatClient
|
|
@@ -15,36 +14,79 @@ from gooddata_eval.core.models import ToolCallEvent
|
|
|
15
14
|
_log = logging.getLogger(__name__)
|
|
16
15
|
|
|
17
16
|
_DEFAULT_K = 1
|
|
18
|
-
# Disambiguation safety net only (create+execute always run together in the same
|
|
19
|
-
#
|
|
20
|
-
|
|
17
|
+
# Disambiguation safety net only (create+execute always run together in the same turn) --
|
|
18
|
+
# 3 real questions' worth (metric, period, +1 slack) since a simulated reply is now sent on
|
|
19
|
+
# every non-final turn (see run_agentic_kda_skill), not just ones classified as a question.
|
|
20
|
+
_DEFAULT_MAX_ITERATIONS = 4
|
|
21
21
|
|
|
22
22
|
|
|
23
|
-
def
|
|
24
|
-
"""
|
|
25
|
-
|
|
26
|
-
|
|
27
|
-
|
|
28
|
-
|
|
23
|
+
def _build_period_hint(expected_output: dict) -> str | None:
|
|
24
|
+
"""Build a period hint from whichever of expected_output's Date Attribute/Analyzed
|
|
25
|
+
Period/Reference Period are present -- a question about only one of them (e.g. "which
|
|
26
|
+
date dimension?") must still get an answerable hint, not None just because the other
|
|
27
|
+
two are absent.
|
|
28
|
+
"""
|
|
29
|
+
date_attr = expected_output.get("Date Attribute")
|
|
30
|
+
analyzed = expected_output.get("Analyzed Period")
|
|
31
|
+
reference_period = expected_output.get("Reference Period")
|
|
32
|
+
if not (date_attr or analyzed or reference_period):
|
|
33
|
+
return None
|
|
34
|
+
parts = []
|
|
35
|
+
if date_attr:
|
|
36
|
+
parts.append(date_attr)
|
|
37
|
+
if analyzed and reference_period:
|
|
38
|
+
parts.append(f"comparing {analyzed} to {reference_period}")
|
|
39
|
+
elif analyzed:
|
|
40
|
+
parts.append(f"period {analyzed}")
|
|
41
|
+
elif reference_period:
|
|
42
|
+
parts.append(f"compared to {reference_period}")
|
|
43
|
+
return ", ".join(parts)
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def _build_clarification_prompt(
|
|
47
|
+
agent_message: str, measure_candidates: dict | list[dict] | None, period_hint: str | None
|
|
48
|
+
) -> str:
|
|
49
|
+
"""Build the simulated-user prompt, referencing only whatever candidates/period-hint
|
|
50
|
+
are actually usable -- an empty/None candidate must drop the "acceptable metric/fact"
|
|
51
|
+
clause entirely rather than assert a literal "None" as if it were a real option.
|
|
29
52
|
"""
|
|
30
|
-
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
|
|
34
|
-
|
|
35
|
-
|
|
36
|
-
|
|
37
|
-
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
|
|
53
|
+
candidates = [
|
|
54
|
+
c for c in (measure_candidates if isinstance(measure_candidates, list) else [measure_candidates]) if c
|
|
55
|
+
]
|
|
56
|
+
reference = ""
|
|
57
|
+
if candidates:
|
|
58
|
+
candidate_desc = "; or ".join(
|
|
59
|
+
f"{c.get('type')} '{c.get('id')}'" + (f" (aggregation {c['aggregation']})" if c.get("aggregation") else "")
|
|
60
|
+
for c in candidates
|
|
61
|
+
)
|
|
62
|
+
reference = f"an acceptable metric/fact is {candidate_desc}"
|
|
63
|
+
if period_hint:
|
|
64
|
+
reference = (
|
|
65
|
+
f"{reference}; the intended time period is {period_hint}"
|
|
66
|
+
if reference
|
|
67
|
+
else f"the intended time period is {period_hint}"
|
|
68
|
+
)
|
|
69
|
+
return (
|
|
70
|
+
f"You are simulating a user in a conversation with a BI assistant that runs key driver "
|
|
71
|
+
f"analysis. The assistant asked: '{agent_message}'. "
|
|
72
|
+
+ (f"For reference, {reference}. " if reference else "")
|
|
73
|
+
+ "Reply briefly as the user, answering whichever of those the assistant actually asked about."
|
|
74
|
+
)
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def generate_simulated_kda_response(
|
|
78
|
+
agent_message: str,
|
|
79
|
+
measure_candidates: dict | list[dict] | None,
|
|
80
|
+
period_hint: str | None = None,
|
|
81
|
+
) -> str:
|
|
42
82
|
"""Generate a user reply to keep the KDA-skill conversation going (gpt-4o-mini).
|
|
43
83
|
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
|
|
84
|
+
Called on any turn that didn't trigger KDA, whatever the agent's response actually
|
|
85
|
+
said -- most often a clarifying question about the measure, the period, or both, so
|
|
86
|
+
both are given as reference and the reply answers whichever was actually asked.
|
|
87
|
+
Scope only needs KDA to trigger, not the resulting measure/period to be exactly
|
|
88
|
+
right. Always OpenAI regardless of the combo's own provider -- this is
|
|
89
|
+
test-harness plumbing, not the system under test.
|
|
48
90
|
"""
|
|
49
91
|
try:
|
|
50
92
|
from openai import OpenAI # noqa: PLC0415
|
|
@@ -56,17 +98,7 @@ def generate_simulated_kda_response(agent_message: str, measure_candidates: dict
|
|
|
56
98
|
raise OSError("OPENAI_API_KEY environment variable is not set")
|
|
57
99
|
|
|
58
100
|
client = OpenAI(api_key=api_key)
|
|
59
|
-
|
|
60
|
-
candidate_desc = "; or ".join(
|
|
61
|
-
f"{c.get('type')} '{c.get('id')}'" + (f" (aggregation {c['aggregation']})" if c.get("aggregation") else "")
|
|
62
|
-
for c in candidates
|
|
63
|
-
)
|
|
64
|
-
prompt = (
|
|
65
|
-
f"You are simulating a user in a conversation with a BI assistant that runs key driver "
|
|
66
|
-
f"analysis. The assistant said: '{agent_message}'. "
|
|
67
|
-
f"The user is happy to proceed with any of the following: {candidate_desc}. "
|
|
68
|
-
f"Reply briefly as the user, picking whichever of those the assistant offered."
|
|
69
|
-
)
|
|
101
|
+
prompt = _build_clarification_prompt(agent_message, measure_candidates, period_hint)
|
|
70
102
|
response = client.chat.completions.create(
|
|
71
103
|
model="gpt-4o-mini",
|
|
72
104
|
messages=[{"role": "user", "content": prompt}],
|
|
@@ -171,9 +203,12 @@ def run_agentic_kda_skill(
|
|
|
171
203
|
|
|
172
204
|
Each run is normally one message, one turn -- create and execute are always called
|
|
173
205
|
together in the same turn (the skill's own system prompt: "NO confirmation needed").
|
|
174
|
-
|
|
175
|
-
|
|
176
|
-
|
|
206
|
+
A run only extends past turn 1, up to ``max_iterations``, when the agent's response
|
|
207
|
+
has no create call and isn't empty; a simulated user reply is then always sent, with
|
|
208
|
+
no attempt to classify whether the text was actually asking for input (matching
|
|
209
|
+
visualization.py/alert_skill.py's own break conditions) -- missing a genuine
|
|
210
|
+
clarifying question hard-fails the run, while sending one after an unrecognized final
|
|
211
|
+
answer only costs one harmless extra turn, so the asymmetry favors never guessing.
|
|
177
212
|
"""
|
|
178
213
|
if k < 1:
|
|
179
214
|
# k=0 or negative would otherwise silently run once, indistinguishable from k=1.
|
|
@@ -212,17 +247,22 @@ def run_agentic_kda_skill(
|
|
|
212
247
|
# execute tool isn't available at all when data-sharing is off for the org).
|
|
213
248
|
turn_wall_clock_sec = chat_result.turn_wall_clock_sec
|
|
214
249
|
break
|
|
250
|
+
if not response_text:
|
|
251
|
+
break
|
|
215
252
|
if iteration >= max_iterations - 1:
|
|
216
253
|
break
|
|
217
|
-
|
|
218
|
-
|
|
219
|
-
|
|
220
|
-
|
|
221
|
-
|
|
222
|
-
|
|
223
|
-
|
|
224
|
-
|
|
225
|
-
|
|
254
|
+
# No text classification -- matches visualization.py/alert_skill.py: break only on
|
|
255
|
+
# the goal signal (create_args set) or an empty response, otherwise always send a
|
|
256
|
+
# simulated reply. A false positive (agent had already given a final answer) costs
|
|
257
|
+
# one harmless extra turn; a false negative (missing a genuine clarifying question)
|
|
258
|
+
# would hard-fail the run, so the asymmetry favors never trying to tell them apart.
|
|
259
|
+
measure_candidates = expected_output.get("Measure") if isinstance(expected_output, dict) else None
|
|
260
|
+
period_hint = _build_period_hint(expected_output) if isinstance(expected_output, dict) else None
|
|
261
|
+
try:
|
|
262
|
+
current_question = generate_simulated_kda_response(response_text, measure_candidates, period_hint)
|
|
263
|
+
disambiguated = True
|
|
264
|
+
except Exception as exc: # noqa: BLE001 -- safety net, not the assertion; end only this run
|
|
265
|
+
_log.warning("Simulated KDA user reply failed for conversation %s: %s", conv_id, exc)
|
|
226
266
|
break
|
|
227
267
|
|
|
228
268
|
ev = _evaluate_run(create_args, execute_result, turn_completed, disambiguated)
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: gooddata-eval
|
|
3
|
-
Version: 1.72.1.
|
|
3
|
+
Version: 1.72.1.dev4
|
|
4
4
|
Summary: Evaluate the GoodData AI agent against your own questions and models.
|
|
5
5
|
Project-URL: Source, https://github.com/gooddata/gooddata-python-sdk
|
|
6
6
|
Author-email: GoodData <support@gooddata.com>
|
|
@@ -17,7 +17,7 @@ Classifier: Topic :: Scientific/Engineering
|
|
|
17
17
|
Classifier: Topic :: Software Development
|
|
18
18
|
Classifier: Typing :: Typed
|
|
19
19
|
Requires-Python: >=3.10
|
|
20
|
-
Requires-Dist: gooddata-sdk~=1.72.1.
|
|
20
|
+
Requires-Dist: gooddata-sdk~=1.72.1.dev4
|
|
21
21
|
Requires-Dist: httpx<1.0,>=0.27
|
|
22
22
|
Requires-Dist: orjson<4.0.0,>=3.9.15
|
|
23
23
|
Requires-Dist: pydantic<3.0,>=2.6
|
|
@@ -17,7 +17,7 @@ gooddata_eval/core/agentic/alert_skill.py,sha256=Be7rf1sfToQfO_pzg1BwXJcC2MwlY01
|
|
|
17
17
|
gooddata_eval/core/agentic/conversation.py,sha256=Yglf68PZ_Yb69z9ujTfX7HQcUrOzwlB85vnpul9SLMk,19073
|
|
18
18
|
gooddata_eval/core/agentic/general_question.py,sha256=Tw2M-FJJWDo0O4leD5AoYW8UBpVQWSZrZJ0fNbHTHW8,8364
|
|
19
19
|
gooddata_eval/core/agentic/guardrail.py,sha256=qE9vADYilDnmqjcwEE-HGF3XsOcGjdvsKFfLSX9IFYQ,8251
|
|
20
|
-
gooddata_eval/core/agentic/kda_skill.py,sha256=
|
|
20
|
+
gooddata_eval/core/agentic/kda_skill.py,sha256=836jOEzvXjv_F-xurm3sUp0QaFPOqu3jZqy1kMSNLxI,17951
|
|
21
21
|
gooddata_eval/core/agentic/metric_skill.py,sha256=zZs8cRcFdDLIV_rY6wQKd9o-0VbQYK6EOL8KbmZpk4A,14117
|
|
22
22
|
gooddata_eval/core/agentic/search_tool.py,sha256=L6yEs0hJymC_beYmily9ivw3M6tjLgDnF9_6kr7g11s,8020
|
|
23
23
|
gooddata_eval/core/agentic/visualization.py,sha256=rZPN6CF_XpMsmPHSwzUP4iisLBL_99g2UIjIww_GbbQ,17129
|
|
@@ -45,8 +45,8 @@ gooddata_eval/core/reporting/console.py,sha256=kCHt4sCSaSRh0ZWqaSEJOl6_Jgb6uiouS
|
|
|
45
45
|
gooddata_eval/core/reporting/json_report.py,sha256=sKPG9-uxzDsf6EJyBXoaHVIfN1g-E2rZ8fA0wtyBYb4,2943
|
|
46
46
|
gooddata_eval/core/summary/__init__.py,sha256=U54lANKm34yjqm7dZL9KkGouPbwAaQWC9wiAmb5B5g4,32
|
|
47
47
|
gooddata_eval/core/summary/http_client.py,sha256=dPArJ6zoCu1W9xmYXFA1WMDNsSOkhn3gaCpYgCUp3gk,2174
|
|
48
|
-
gooddata_eval-1.72.1.
|
|
49
|
-
gooddata_eval-1.72.1.
|
|
50
|
-
gooddata_eval-1.72.1.
|
|
51
|
-
gooddata_eval-1.72.1.
|
|
52
|
-
gooddata_eval-1.72.1.
|
|
48
|
+
gooddata_eval-1.72.1.dev4.dist-info/METADATA,sha256=H_8L61m6K2yL9IR9kQPFcQKr001D6-j4Hd1Fz2kSkuc,9995
|
|
49
|
+
gooddata_eval-1.72.1.dev4.dist-info/WHEEL,sha256=zOwg4jB6zX2kU910N-cMawjivD6tO8NEWvE12je1bVk,87
|
|
50
|
+
gooddata_eval-1.72.1.dev4.dist-info/entry_points.txt,sha256=28nFp5Viknx4haPYgzK9QHlwPFfC6rPF6RmjC2ht8EI,56
|
|
51
|
+
gooddata_eval-1.72.1.dev4.dist-info/licenses/LICENSE.txt,sha256=LVfVlC9maU3K9lgMwdZnClQQsOIbOekdcdVKr9qzJtk,256836
|
|
52
|
+
gooddata_eval-1.72.1.dev4.dist-info/RECORD,,
|
|
File without changes
|
{gooddata_eval-1.72.1.dev3.dist-info → gooddata_eval-1.72.1.dev4.dist-info}/entry_points.txt
RENAMED
|
File without changes
|
{gooddata_eval-1.72.1.dev3.dist-info → gooddata_eval-1.72.1.dev4.dist-info}/licenses/LICENSE.txt
RENAMED
|
File without changes
|