gooddata-eval 1.72.1.dev4__py3-none-any.whl → 1.72.1.dev6__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- gooddata_eval/core/agentic/alert_skill.py +1 -1
- gooddata_eval/core/agentic/conversation.py +27 -18
- gooddata_eval/core/agentic/kda_skill.py +1 -0
- gooddata_eval/core/agentic/metric_skill.py +33 -18
- gooddata_eval/core/agentic/visualization.py +1 -1
- {gooddata_eval-1.72.1.dev4.dist-info → gooddata_eval-1.72.1.dev6.dist-info}/METADATA +2 -2
- {gooddata_eval-1.72.1.dev4.dist-info → gooddata_eval-1.72.1.dev6.dist-info}/RECORD +10 -10
- {gooddata_eval-1.72.1.dev4.dist-info → gooddata_eval-1.72.1.dev6.dist-info}/WHEEL +0 -0
- {gooddata_eval-1.72.1.dev4.dist-info → gooddata_eval-1.72.1.dev6.dist-info}/entry_points.txt +0 -0
- {gooddata_eval-1.72.1.dev4.dist-info → gooddata_eval-1.72.1.dev6.dist-info}/licenses/LICENSE.txt +0 -0
|
@@ -25,6 +25,8 @@ from gooddata_eval.core.scoring import (
|
|
|
25
25
|
|
|
26
26
|
_REF_PATTERN = re.compile(r"\$ref:([\w_]+)\.([\w_]+)")
|
|
27
27
|
|
|
28
|
+
_DEFAULT_MAX_CLARIFICATION_TURNS = 7
|
|
29
|
+
|
|
28
30
|
|
|
29
31
|
class TurnDefinition(BaseModel):
|
|
30
32
|
"""Definition of a single turn in a multi-turn conversation evaluation."""
|
|
@@ -192,13 +194,6 @@ def _check_output_correct(turn: TurnDefinition, chat_result: ChatResult) -> bool
|
|
|
192
194
|
return None
|
|
193
195
|
|
|
194
196
|
|
|
195
|
-
def _is_asking_clarification(text: str) -> bool:
|
|
196
|
-
if not text:
|
|
197
|
-
return False
|
|
198
|
-
t = text.lower()
|
|
199
|
-
return "?" in t or "could you" in t or "please" in t or "clarif" in t
|
|
200
|
-
|
|
201
|
-
|
|
202
197
|
def _get_sim_user_response(agent_message: str, turn: TurnDefinition, expected_output: dict | None) -> str:
|
|
203
198
|
"""Generate a simulated user reply to an agent clarification question."""
|
|
204
199
|
otype = turn.expected_output_type
|
|
@@ -252,7 +247,7 @@ def _get_sim_user_response(agent_message: str, turn: TurnDefinition, expected_ou
|
|
|
252
247
|
),
|
|
253
248
|
},
|
|
254
249
|
],
|
|
255
|
-
temperature=0
|
|
250
|
+
temperature=0,
|
|
256
251
|
)
|
|
257
252
|
content = response.choices[0].message.content
|
|
258
253
|
return content.strip() if content else "Please proceed with sensible defaults."
|
|
@@ -277,7 +272,7 @@ def run_agentic_conversation(
|
|
|
277
272
|
token: str,
|
|
278
273
|
workspace_id: str,
|
|
279
274
|
fixture: ConversationFixture,
|
|
280
|
-
max_clarification_turns: int =
|
|
275
|
+
max_clarification_turns: int = _DEFAULT_MAX_CLARIFICATION_TURNS,
|
|
281
276
|
initial_conversation_id: str | None = None,
|
|
282
277
|
reasoning_effort: ReasoningEffort | None = None,
|
|
283
278
|
) -> ConversationResult:
|
|
@@ -307,8 +302,22 @@ def run_agentic_conversation(
|
|
|
307
302
|
owns_conversation = True
|
|
308
303
|
|
|
309
304
|
for turn in fixture.turns:
|
|
310
|
-
|
|
311
|
-
|
|
305
|
+
try:
|
|
306
|
+
resolved_expected = _resolve_refs(turn.expected_output, turn_outputs)
|
|
307
|
+
except ValueError as exc:
|
|
308
|
+
print(f"[SKIP] turn '{turn.turn_id}': {exc}")
|
|
309
|
+
turn_results.append(
|
|
310
|
+
TurnResult(
|
|
311
|
+
turn_id=turn.turn_id,
|
|
312
|
+
expected_skill=turn.expected_skill,
|
|
313
|
+
skill_routing=False,
|
|
314
|
+
output_present=False,
|
|
315
|
+
no_error=False,
|
|
316
|
+
activated_skills=[],
|
|
317
|
+
output_correct=False,
|
|
318
|
+
)
|
|
319
|
+
)
|
|
320
|
+
continue
|
|
312
321
|
resolved_turn = turn.model_copy(update={"expected_output": resolved_expected})
|
|
313
322
|
|
|
314
323
|
clarification_turns = 0
|
|
@@ -327,13 +336,13 @@ def run_agentic_conversation(
|
|
|
327
336
|
response_text = (chat_result.text_response or "").strip()
|
|
328
337
|
if not response_text and chat_result.alert_proposals:
|
|
329
338
|
response_text = render_alert_proposal(chat_result.alert_proposals[-1])
|
|
330
|
-
|
|
331
|
-
|
|
332
|
-
|
|
333
|
-
total_clarification_turns += 1
|
|
334
|
-
current_message = _get_sim_user_response(response_text, resolved_turn, resolved_expected)
|
|
335
|
-
else:
|
|
339
|
+
if not response_text and not chat_result.tool_call_events:
|
|
340
|
+
break
|
|
341
|
+
if clarification_turns >= max_clarification_turns:
|
|
336
342
|
break
|
|
343
|
+
clarification_turns += 1
|
|
344
|
+
total_clarification_turns += 1
|
|
345
|
+
current_message = _get_sim_user_response(response_text, resolved_turn, resolved_expected)
|
|
337
346
|
|
|
338
347
|
activated = _activated_skills(all_tool_calls)
|
|
339
348
|
skill_routing = turn.expected_skill in activated if activated else False
|
|
@@ -397,7 +406,7 @@ def evaluate_agentic_conversation(
|
|
|
397
406
|
token: str,
|
|
398
407
|
workspace_id: str,
|
|
399
408
|
fixture: ConversationFixture,
|
|
400
|
-
max_clarification_turns: int =
|
|
409
|
+
max_clarification_turns: int = _DEFAULT_MAX_CLARIFICATION_TURNS,
|
|
401
410
|
initial_conversation_id: str | None = None,
|
|
402
411
|
langfuse: object | None = None,
|
|
403
412
|
dataset_item_id: str = "",
|
|
@@ -103,6 +103,7 @@ def generate_simulated_kda_response(
|
|
|
103
103
|
model="gpt-4o-mini",
|
|
104
104
|
messages=[{"role": "user", "content": prompt}],
|
|
105
105
|
max_tokens=150,
|
|
106
|
+
temperature=0,
|
|
106
107
|
timeout=30,
|
|
107
108
|
)
|
|
108
109
|
return response.choices[0].message.content or "Please proceed with either option."
|
|
@@ -72,16 +72,29 @@ def _best_maql_match(actual_maql: str, expected_outputs: list[dict]) -> tuple[bo
|
|
|
72
72
|
return False, expected_outputs[0].get("maql", "") if expected_outputs else ""
|
|
73
73
|
|
|
74
74
|
|
|
75
|
+
class SimulatedResponseError(RuntimeError):
|
|
76
|
+
"""The simulated user could not reply: openai missing, no API key, or the provider failed.
|
|
77
|
+
|
|
78
|
+
Carries every expected setup/provider failure so callers can end the run without
|
|
79
|
+
swallowing programming errors raised from the same call.
|
|
80
|
+
"""
|
|
81
|
+
|
|
82
|
+
|
|
75
83
|
def generate_simulated_response(agent_message: str, expected_output: dict) -> str:
|
|
76
|
-
"""Generate a user reply to keep the metric-skill conversation going (gpt-4o-mini).
|
|
84
|
+
"""Generate a user reply to keep the metric-skill conversation going (gpt-4o-mini).
|
|
85
|
+
|
|
86
|
+
Raises:
|
|
87
|
+
SimulatedResponseError: openai is not installed, OPENAI_API_KEY is unset, or the
|
|
88
|
+
provider call failed.
|
|
89
|
+
"""
|
|
77
90
|
try:
|
|
78
|
-
from openai import OpenAI # noqa: PLC0415
|
|
91
|
+
from openai import OpenAI, OpenAIError # noqa: PLC0415
|
|
79
92
|
except ImportError as exc:
|
|
80
|
-
raise
|
|
93
|
+
raise SimulatedResponseError("openai package is required for generate_simulated_response") from exc
|
|
81
94
|
|
|
82
95
|
api_key = os.environ.get("OPENAI_API_KEY")
|
|
83
96
|
if not api_key:
|
|
84
|
-
raise
|
|
97
|
+
raise SimulatedResponseError("OPENAI_API_KEY environment variable is not set")
|
|
85
98
|
|
|
86
99
|
client = OpenAI(api_key=api_key)
|
|
87
100
|
expected_maql = expected_output.get("maql", "")
|
|
@@ -91,11 +104,15 @@ def generate_simulated_response(agent_message: str, expected_output: dict) -> st
|
|
|
91
104
|
f"The user originally asked to create a metric with MAQL: {expected_maql}. "
|
|
92
105
|
f"Reply briefly as the user, providing any clarification the assistant needs."
|
|
93
106
|
)
|
|
94
|
-
|
|
95
|
-
|
|
96
|
-
|
|
97
|
-
|
|
98
|
-
|
|
107
|
+
try:
|
|
108
|
+
response = client.chat.completions.create(
|
|
109
|
+
model="gpt-4o-mini",
|
|
110
|
+
messages=[{"role": "user", "content": prompt}],
|
|
111
|
+
max_tokens=150,
|
|
112
|
+
temperature=0,
|
|
113
|
+
)
|
|
114
|
+
except OpenAIError as exc:
|
|
115
|
+
raise SimulatedResponseError(f"simulated user reply failed: {exc}") from exc
|
|
99
116
|
return response.choices[0].message.content or "Please proceed."
|
|
100
117
|
|
|
101
118
|
|
|
@@ -165,13 +182,6 @@ def _delete_metric(sdk: GoodDataSdk, workspace_id: str, metric_id: str) -> None:
|
|
|
165
182
|
print(f"[CLEANUP] Failed to delete metric {metric_id}: {exc}")
|
|
166
183
|
|
|
167
184
|
|
|
168
|
-
def _is_asking_clarification(text: str) -> bool:
|
|
169
|
-
if not text:
|
|
170
|
-
return False
|
|
171
|
-
t = text.lower()
|
|
172
|
-
return "?" in t or "could you" in t or "please provide" in t or "clarif" in t
|
|
173
|
-
|
|
174
|
-
|
|
175
185
|
def _execute_single_metric_run(
|
|
176
186
|
client: ChatClient,
|
|
177
187
|
sdk: GoodDataSdk,
|
|
@@ -203,9 +213,14 @@ def _execute_single_metric_run(
|
|
|
203
213
|
metric_id_to_delete = candidate.get("metric_id")
|
|
204
214
|
break
|
|
205
215
|
response_text = (chat_result.text_response or "").strip()
|
|
206
|
-
if
|
|
216
|
+
if not response_text and not chat_result.tool_call_events:
|
|
217
|
+
break
|
|
218
|
+
if _iteration >= max_iterations - 1:
|
|
219
|
+
break
|
|
220
|
+
try:
|
|
207
221
|
current_question = generate_simulated_response(response_text, primary_expected)
|
|
208
|
-
|
|
222
|
+
except SimulatedResponseError as exc:
|
|
223
|
+
print(f"[SIM-USER] Simulated reply failed for conversation {conversation_id}: {exc}")
|
|
209
224
|
break
|
|
210
225
|
|
|
211
226
|
actual_maql = (metric_result or {}).get("maql", "")
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: gooddata-eval
|
|
3
|
-
Version: 1.72.1.
|
|
3
|
+
Version: 1.72.1.dev6
|
|
4
4
|
Summary: Evaluate the GoodData AI agent against your own questions and models.
|
|
5
5
|
Project-URL: Source, https://github.com/gooddata/gooddata-python-sdk
|
|
6
6
|
Author-email: GoodData <support@gooddata.com>
|
|
@@ -17,7 +17,7 @@ Classifier: Topic :: Scientific/Engineering
|
|
|
17
17
|
Classifier: Topic :: Software Development
|
|
18
18
|
Classifier: Typing :: Typed
|
|
19
19
|
Requires-Python: >=3.10
|
|
20
|
-
Requires-Dist: gooddata-sdk~=1.72.1.
|
|
20
|
+
Requires-Dist: gooddata-sdk~=1.72.1.dev6
|
|
21
21
|
Requires-Dist: httpx<1.0,>=0.27
|
|
22
22
|
Requires-Dist: orjson<4.0.0,>=3.9.15
|
|
23
23
|
Requires-Dist: pydantic<3.0,>=2.6
|
|
@@ -13,14 +13,14 @@ gooddata_eval/core/workspace.py,sha256=S8exiy5gpDtcRtT8-i2fIRALF34Pbo0J46xm8X_ld
|
|
|
13
13
|
gooddata_eval/core/agentic/__init__.py,sha256=M35wdSW7Bw5vidfHN4MuL0ZR8L7igUnF5mNWiboglcw,3030
|
|
14
14
|
gooddata_eval/core/agentic/_catalog.py,sha256=eCaeyvLF-FCN0J3UnXSHaKtufU6HvrM2mxnmgyXsY1U,2085
|
|
15
15
|
gooddata_eval/core/agentic/_langfuse.py,sha256=nRI_tcmORBWbuF3AJKAL8VX-mFkLyZBQXagCGSQ6xIE,15102
|
|
16
|
-
gooddata_eval/core/agentic/alert_skill.py,sha256=
|
|
17
|
-
gooddata_eval/core/agentic/conversation.py,sha256=
|
|
16
|
+
gooddata_eval/core/agentic/alert_skill.py,sha256=xFYOmpDX5Qrjj0cVMilf99D7DfLHeVwMyrSftKB1f_c,27224
|
|
17
|
+
gooddata_eval/core/agentic/conversation.py,sha256=kAXkPfF1pYSaZ4HooapqIqVJ52H4BO6u-uowoblqMr8,19442
|
|
18
18
|
gooddata_eval/core/agentic/general_question.py,sha256=Tw2M-FJJWDo0O4leD5AoYW8UBpVQWSZrZJ0fNbHTHW8,8364
|
|
19
19
|
gooddata_eval/core/agentic/guardrail.py,sha256=qE9vADYilDnmqjcwEE-HGF3XsOcGjdvsKFfLSX9IFYQ,8251
|
|
20
|
-
gooddata_eval/core/agentic/kda_skill.py,sha256=
|
|
21
|
-
gooddata_eval/core/agentic/metric_skill.py,sha256=
|
|
20
|
+
gooddata_eval/core/agentic/kda_skill.py,sha256=XsKPhhEHQGJTPC3AufOgantxDTdFUwpjLpqlMfPLLEk,17974
|
|
21
|
+
gooddata_eval/core/agentic/metric_skill.py,sha256=SeFJFPZLzWBLA2UxlH9q8kUDB6Ztb5ylxOApvd2lkCA,14831
|
|
22
22
|
gooddata_eval/core/agentic/search_tool.py,sha256=L6yEs0hJymC_beYmily9ivw3M6tjLgDnF9_6kr7g11s,8020
|
|
23
|
-
gooddata_eval/core/agentic/visualization.py,sha256=
|
|
23
|
+
gooddata_eval/core/agentic/visualization.py,sha256=qhwJ06_SauPjOl1yJ3teZmf42YQjpcSwTy3yuc3kz30,17127
|
|
24
24
|
gooddata_eval/core/chat/__init__.py,sha256=U54lANKm34yjqm7dZL9KkGouPbwAaQWC9wiAmb5B5g4,32
|
|
25
25
|
gooddata_eval/core/chat/sse_client.py,sha256=oYLxMWy6Hy0yoC8InnTkKLWN4Rsg1gSSIiE0FzaVa4o,16062
|
|
26
26
|
gooddata_eval/core/dataset/__init__.py,sha256=U54lANKm34yjqm7dZL9KkGouPbwAaQWC9wiAmb5B5g4,32
|
|
@@ -45,8 +45,8 @@ gooddata_eval/core/reporting/console.py,sha256=kCHt4sCSaSRh0ZWqaSEJOl6_Jgb6uiouS
|
|
|
45
45
|
gooddata_eval/core/reporting/json_report.py,sha256=sKPG9-uxzDsf6EJyBXoaHVIfN1g-E2rZ8fA0wtyBYb4,2943
|
|
46
46
|
gooddata_eval/core/summary/__init__.py,sha256=U54lANKm34yjqm7dZL9KkGouPbwAaQWC9wiAmb5B5g4,32
|
|
47
47
|
gooddata_eval/core/summary/http_client.py,sha256=dPArJ6zoCu1W9xmYXFA1WMDNsSOkhn3gaCpYgCUp3gk,2174
|
|
48
|
-
gooddata_eval-1.72.1.
|
|
49
|
-
gooddata_eval-1.72.1.
|
|
50
|
-
gooddata_eval-1.72.1.
|
|
51
|
-
gooddata_eval-1.72.1.
|
|
52
|
-
gooddata_eval-1.72.1.
|
|
48
|
+
gooddata_eval-1.72.1.dev6.dist-info/METADATA,sha256=b3CWA-37bAb2hXjVrQJ3LiBgj0e5mla8iLp8Ga-ZNLk,9995
|
|
49
|
+
gooddata_eval-1.72.1.dev6.dist-info/WHEEL,sha256=zOwg4jB6zX2kU910N-cMawjivD6tO8NEWvE12je1bVk,87
|
|
50
|
+
gooddata_eval-1.72.1.dev6.dist-info/entry_points.txt,sha256=28nFp5Viknx4haPYgzK9QHlwPFfC6rPF6RmjC2ht8EI,56
|
|
51
|
+
gooddata_eval-1.72.1.dev6.dist-info/licenses/LICENSE.txt,sha256=LVfVlC9maU3K9lgMwdZnClQQsOIbOekdcdVKr9qzJtk,256836
|
|
52
|
+
gooddata_eval-1.72.1.dev6.dist-info/RECORD,,
|
|
File without changes
|
{gooddata_eval-1.72.1.dev4.dist-info → gooddata_eval-1.72.1.dev6.dist-info}/entry_points.txt
RENAMED
|
File without changes
|
{gooddata_eval-1.72.1.dev4.dist-info → gooddata_eval-1.72.1.dev6.dist-info}/licenses/LICENSE.txt
RENAMED
|
File without changes
|