gooddata-eval 1.72.1.dev4__py3-none-any.whl → 1.72.1.dev6__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -250,7 +250,7 @@ def generate_simulated_alert_response(
250
250
  response = openai_client.chat.completions.create(
251
251
  model="gpt-4o",
252
252
  messages=messages,
253
- temperature=0.5,
253
+ temperature=0,
254
254
  )
255
255
  return response.choices[0].message.content or ""
256
256
 
@@ -25,6 +25,8 @@ from gooddata_eval.core.scoring import (
25
25
 
26
26
  _REF_PATTERN = re.compile(r"\$ref:([\w_]+)\.([\w_]+)")
27
27
 
28
+ _DEFAULT_MAX_CLARIFICATION_TURNS = 7
29
+
28
30
 
29
31
  class TurnDefinition(BaseModel):
30
32
  """Definition of a single turn in a multi-turn conversation evaluation."""
@@ -192,13 +194,6 @@ def _check_output_correct(turn: TurnDefinition, chat_result: ChatResult) -> bool
192
194
  return None
193
195
 
194
196
 
195
- def _is_asking_clarification(text: str) -> bool:
196
- if not text:
197
- return False
198
- t = text.lower()
199
- return "?" in t or "could you" in t or "please" in t or "clarif" in t
200
-
201
-
202
197
  def _get_sim_user_response(agent_message: str, turn: TurnDefinition, expected_output: dict | None) -> str:
203
198
  """Generate a simulated user reply to an agent clarification question."""
204
199
  otype = turn.expected_output_type
@@ -252,7 +247,7 @@ def _get_sim_user_response(agent_message: str, turn: TurnDefinition, expected_ou
252
247
  ),
253
248
  },
254
249
  ],
255
- temperature=0.5,
250
+ temperature=0,
256
251
  )
257
252
  content = response.choices[0].message.content
258
253
  return content.strip() if content else "Please proceed with sensible defaults."
@@ -277,7 +272,7 @@ def run_agentic_conversation(
277
272
  token: str,
278
273
  workspace_id: str,
279
274
  fixture: ConversationFixture,
280
- max_clarification_turns: int = 20,
275
+ max_clarification_turns: int = _DEFAULT_MAX_CLARIFICATION_TURNS,
281
276
  initial_conversation_id: str | None = None,
282
277
  reasoning_effort: ReasoningEffort | None = None,
283
278
  ) -> ConversationResult:
@@ -307,8 +302,22 @@ def run_agentic_conversation(
307
302
  owns_conversation = True
308
303
 
309
304
  for turn in fixture.turns:
310
- # Resolve $ref placeholders using outputs captured from prior turns.
311
- resolved_expected = _resolve_refs(turn.expected_output, turn_outputs)
305
+ try:
306
+ resolved_expected = _resolve_refs(turn.expected_output, turn_outputs)
307
+ except ValueError as exc:
308
+ print(f"[SKIP] turn '{turn.turn_id}': {exc}")
309
+ turn_results.append(
310
+ TurnResult(
311
+ turn_id=turn.turn_id,
312
+ expected_skill=turn.expected_skill,
313
+ skill_routing=False,
314
+ output_present=False,
315
+ no_error=False,
316
+ activated_skills=[],
317
+ output_correct=False,
318
+ )
319
+ )
320
+ continue
312
321
  resolved_turn = turn.model_copy(update={"expected_output": resolved_expected})
313
322
 
314
323
  clarification_turns = 0
@@ -327,13 +336,13 @@ def run_agentic_conversation(
327
336
  response_text = (chat_result.text_response or "").strip()
328
337
  if not response_text and chat_result.alert_proposals:
329
338
  response_text = render_alert_proposal(chat_result.alert_proposals[-1])
330
- asking = _is_asking_clarification(response_text) or bool(chat_result.alert_proposals)
331
- if asking and clarification_turns < max_clarification_turns:
332
- clarification_turns += 1
333
- total_clarification_turns += 1
334
- current_message = _get_sim_user_response(response_text, resolved_turn, resolved_expected)
335
- else:
339
+ if not response_text and not chat_result.tool_call_events:
340
+ break
341
+ if clarification_turns >= max_clarification_turns:
336
342
  break
343
+ clarification_turns += 1
344
+ total_clarification_turns += 1
345
+ current_message = _get_sim_user_response(response_text, resolved_turn, resolved_expected)
337
346
 
338
347
  activated = _activated_skills(all_tool_calls)
339
348
  skill_routing = turn.expected_skill in activated if activated else False
@@ -397,7 +406,7 @@ def evaluate_agentic_conversation(
397
406
  token: str,
398
407
  workspace_id: str,
399
408
  fixture: ConversationFixture,
400
- max_clarification_turns: int = 20,
409
+ max_clarification_turns: int = _DEFAULT_MAX_CLARIFICATION_TURNS,
401
410
  initial_conversation_id: str | None = None,
402
411
  langfuse: object | None = None,
403
412
  dataset_item_id: str = "",
@@ -103,6 +103,7 @@ def generate_simulated_kda_response(
103
103
  model="gpt-4o-mini",
104
104
  messages=[{"role": "user", "content": prompt}],
105
105
  max_tokens=150,
106
+ temperature=0,
106
107
  timeout=30,
107
108
  )
108
109
  return response.choices[0].message.content or "Please proceed with either option."
@@ -72,16 +72,29 @@ def _best_maql_match(actual_maql: str, expected_outputs: list[dict]) -> tuple[bo
72
72
  return False, expected_outputs[0].get("maql", "") if expected_outputs else ""
73
73
 
74
74
 
75
+ class SimulatedResponseError(RuntimeError):
76
+ """The simulated user could not reply: openai missing, no API key, or the provider failed.
77
+
78
+ Carries every expected setup/provider failure so callers can end the run without
79
+ swallowing programming errors raised from the same call.
80
+ """
81
+
82
+
75
83
  def generate_simulated_response(agent_message: str, expected_output: dict) -> str:
76
- """Generate a user reply to keep the metric-skill conversation going (gpt-4o-mini)."""
84
+ """Generate a user reply to keep the metric-skill conversation going (gpt-4o-mini).
85
+
86
+ Raises:
87
+ SimulatedResponseError: openai is not installed, OPENAI_API_KEY is unset, or the
88
+ provider call failed.
89
+ """
77
90
  try:
78
- from openai import OpenAI # noqa: PLC0415
91
+ from openai import OpenAI, OpenAIError # noqa: PLC0415
79
92
  except ImportError as exc:
80
- raise RuntimeError("openai package is required for generate_simulated_response") from exc
93
+ raise SimulatedResponseError("openai package is required for generate_simulated_response") from exc
81
94
 
82
95
  api_key = os.environ.get("OPENAI_API_KEY")
83
96
  if not api_key:
84
- raise OSError("OPENAI_API_KEY environment variable is not set")
97
+ raise SimulatedResponseError("OPENAI_API_KEY environment variable is not set")
85
98
 
86
99
  client = OpenAI(api_key=api_key)
87
100
  expected_maql = expected_output.get("maql", "")
@@ -91,11 +104,15 @@ def generate_simulated_response(agent_message: str, expected_output: dict) -> st
91
104
  f"The user originally asked to create a metric with MAQL: {expected_maql}. "
92
105
  f"Reply briefly as the user, providing any clarification the assistant needs."
93
106
  )
94
- response = client.chat.completions.create(
95
- model="gpt-4o-mini",
96
- messages=[{"role": "user", "content": prompt}],
97
- max_tokens=150,
98
- )
107
+ try:
108
+ response = client.chat.completions.create(
109
+ model="gpt-4o-mini",
110
+ messages=[{"role": "user", "content": prompt}],
111
+ max_tokens=150,
112
+ temperature=0,
113
+ )
114
+ except OpenAIError as exc:
115
+ raise SimulatedResponseError(f"simulated user reply failed: {exc}") from exc
99
116
  return response.choices[0].message.content or "Please proceed."
100
117
 
101
118
 
@@ -165,13 +182,6 @@ def _delete_metric(sdk: GoodDataSdk, workspace_id: str, metric_id: str) -> None:
165
182
  print(f"[CLEANUP] Failed to delete metric {metric_id}: {exc}")
166
183
 
167
184
 
168
- def _is_asking_clarification(text: str) -> bool:
169
- if not text:
170
- return False
171
- t = text.lower()
172
- return "?" in t or "could you" in t or "please provide" in t or "clarif" in t
173
-
174
-
175
185
  def _execute_single_metric_run(
176
186
  client: ChatClient,
177
187
  sdk: GoodDataSdk,
@@ -203,9 +213,14 @@ def _execute_single_metric_run(
203
213
  metric_id_to_delete = candidate.get("metric_id")
204
214
  break
205
215
  response_text = (chat_result.text_response or "").strip()
206
- if _is_asking_clarification(response_text):
216
+ if not response_text and not chat_result.tool_call_events:
217
+ break
218
+ if _iteration >= max_iterations - 1:
219
+ break
220
+ try:
207
221
  current_question = generate_simulated_response(response_text, primary_expected)
208
- else:
222
+ except SimulatedResponseError as exc:
223
+ print(f"[SIM-USER] Simulated reply failed for conversation {conversation_id}: {exc}")
209
224
  break
210
225
 
211
226
  actual_maql = (metric_result or {}).get("maql", "")
@@ -141,7 +141,7 @@ def generate_simulated_response(agent_message: str, expected_output: CreatedVisu
141
141
  ),
142
142
  },
143
143
  ],
144
- temperature=0.5,
144
+ temperature=0,
145
145
  )
146
146
  content = response.choices[0].message.content
147
147
  return content.strip() if content else ""
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: gooddata-eval
3
- Version: 1.72.1.dev4
3
+ Version: 1.72.1.dev6
4
4
  Summary: Evaluate the GoodData AI agent against your own questions and models.
5
5
  Project-URL: Source, https://github.com/gooddata/gooddata-python-sdk
6
6
  Author-email: GoodData <support@gooddata.com>
@@ -17,7 +17,7 @@ Classifier: Topic :: Scientific/Engineering
17
17
  Classifier: Topic :: Software Development
18
18
  Classifier: Typing :: Typed
19
19
  Requires-Python: >=3.10
20
- Requires-Dist: gooddata-sdk~=1.72.1.dev4
20
+ Requires-Dist: gooddata-sdk~=1.72.1.dev6
21
21
  Requires-Dist: httpx<1.0,>=0.27
22
22
  Requires-Dist: orjson<4.0.0,>=3.9.15
23
23
  Requires-Dist: pydantic<3.0,>=2.6
@@ -13,14 +13,14 @@ gooddata_eval/core/workspace.py,sha256=S8exiy5gpDtcRtT8-i2fIRALF34Pbo0J46xm8X_ld
13
13
  gooddata_eval/core/agentic/__init__.py,sha256=M35wdSW7Bw5vidfHN4MuL0ZR8L7igUnF5mNWiboglcw,3030
14
14
  gooddata_eval/core/agentic/_catalog.py,sha256=eCaeyvLF-FCN0J3UnXSHaKtufU6HvrM2mxnmgyXsY1U,2085
15
15
  gooddata_eval/core/agentic/_langfuse.py,sha256=nRI_tcmORBWbuF3AJKAL8VX-mFkLyZBQXagCGSQ6xIE,15102
16
- gooddata_eval/core/agentic/alert_skill.py,sha256=Be7rf1sfToQfO_pzg1BwXJcC2MwlY01Vwg8egCxjREI,27226
17
- gooddata_eval/core/agentic/conversation.py,sha256=Yglf68PZ_Yb69z9ujTfX7HQcUrOzwlB85vnpul9SLMk,19073
16
+ gooddata_eval/core/agentic/alert_skill.py,sha256=xFYOmpDX5Qrjj0cVMilf99D7DfLHeVwMyrSftKB1f_c,27224
17
+ gooddata_eval/core/agentic/conversation.py,sha256=kAXkPfF1pYSaZ4HooapqIqVJ52H4BO6u-uowoblqMr8,19442
18
18
  gooddata_eval/core/agentic/general_question.py,sha256=Tw2M-FJJWDo0O4leD5AoYW8UBpVQWSZrZJ0fNbHTHW8,8364
19
19
  gooddata_eval/core/agentic/guardrail.py,sha256=qE9vADYilDnmqjcwEE-HGF3XsOcGjdvsKFfLSX9IFYQ,8251
20
- gooddata_eval/core/agentic/kda_skill.py,sha256=836jOEzvXjv_F-xurm3sUp0QaFPOqu3jZqy1kMSNLxI,17951
21
- gooddata_eval/core/agentic/metric_skill.py,sha256=zZs8cRcFdDLIV_rY6wQKd9o-0VbQYK6EOL8KbmZpk4A,14117
20
+ gooddata_eval/core/agentic/kda_skill.py,sha256=XsKPhhEHQGJTPC3AufOgantxDTdFUwpjLpqlMfPLLEk,17974
21
+ gooddata_eval/core/agentic/metric_skill.py,sha256=SeFJFPZLzWBLA2UxlH9q8kUDB6Ztb5ylxOApvd2lkCA,14831
22
22
  gooddata_eval/core/agentic/search_tool.py,sha256=L6yEs0hJymC_beYmily9ivw3M6tjLgDnF9_6kr7g11s,8020
23
- gooddata_eval/core/agentic/visualization.py,sha256=rZPN6CF_XpMsmPHSwzUP4iisLBL_99g2UIjIww_GbbQ,17129
23
+ gooddata_eval/core/agentic/visualization.py,sha256=qhwJ06_SauPjOl1yJ3teZmf42YQjpcSwTy3yuc3kz30,17127
24
24
  gooddata_eval/core/chat/__init__.py,sha256=U54lANKm34yjqm7dZL9KkGouPbwAaQWC9wiAmb5B5g4,32
25
25
  gooddata_eval/core/chat/sse_client.py,sha256=oYLxMWy6Hy0yoC8InnTkKLWN4Rsg1gSSIiE0FzaVa4o,16062
26
26
  gooddata_eval/core/dataset/__init__.py,sha256=U54lANKm34yjqm7dZL9KkGouPbwAaQWC9wiAmb5B5g4,32
@@ -45,8 +45,8 @@ gooddata_eval/core/reporting/console.py,sha256=kCHt4sCSaSRh0ZWqaSEJOl6_Jgb6uiouS
45
45
  gooddata_eval/core/reporting/json_report.py,sha256=sKPG9-uxzDsf6EJyBXoaHVIfN1g-E2rZ8fA0wtyBYb4,2943
46
46
  gooddata_eval/core/summary/__init__.py,sha256=U54lANKm34yjqm7dZL9KkGouPbwAaQWC9wiAmb5B5g4,32
47
47
  gooddata_eval/core/summary/http_client.py,sha256=dPArJ6zoCu1W9xmYXFA1WMDNsSOkhn3gaCpYgCUp3gk,2174
48
- gooddata_eval-1.72.1.dev4.dist-info/METADATA,sha256=H_8L61m6K2yL9IR9kQPFcQKr001D6-j4Hd1Fz2kSkuc,9995
49
- gooddata_eval-1.72.1.dev4.dist-info/WHEEL,sha256=zOwg4jB6zX2kU910N-cMawjivD6tO8NEWvE12je1bVk,87
50
- gooddata_eval-1.72.1.dev4.dist-info/entry_points.txt,sha256=28nFp5Viknx4haPYgzK9QHlwPFfC6rPF6RmjC2ht8EI,56
51
- gooddata_eval-1.72.1.dev4.dist-info/licenses/LICENSE.txt,sha256=LVfVlC9maU3K9lgMwdZnClQQsOIbOekdcdVKr9qzJtk,256836
52
- gooddata_eval-1.72.1.dev4.dist-info/RECORD,,
48
+ gooddata_eval-1.72.1.dev6.dist-info/METADATA,sha256=b3CWA-37bAb2hXjVrQJ3LiBgj0e5mla8iLp8Ga-ZNLk,9995
49
+ gooddata_eval-1.72.1.dev6.dist-info/WHEEL,sha256=zOwg4jB6zX2kU910N-cMawjivD6tO8NEWvE12je1bVk,87
50
+ gooddata_eval-1.72.1.dev6.dist-info/entry_points.txt,sha256=28nFp5Viknx4haPYgzK9QHlwPFfC6rPF6RmjC2ht8EI,56
51
+ gooddata_eval-1.72.1.dev6.dist-info/licenses/LICENSE.txt,sha256=LVfVlC9maU3K9lgMwdZnClQQsOIbOekdcdVKr9qzJtk,256836
52
+ gooddata_eval-1.72.1.dev6.dist-info/RECORD,,