gooddata-eval 1.72.1.dev3__py3-none-any.whl → 1.72.1.dev5__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -250,7 +250,7 @@ def generate_simulated_alert_response(
250
250
  response = openai_client.chat.completions.create(
251
251
  model="gpt-4o",
252
252
  messages=messages,
253
- temperature=0.5,
253
+ temperature=0,
254
254
  )
255
255
  return response.choices[0].message.content or ""
256
256
 
@@ -252,7 +252,7 @@ def _get_sim_user_response(agent_message: str, turn: TurnDefinition, expected_ou
252
252
  ),
253
253
  },
254
254
  ],
255
- temperature=0.5,
255
+ temperature=0,
256
256
  )
257
257
  content = response.choices[0].message.content
258
258
  return content.strip() if content else "Please proceed with sensible defaults."
@@ -5,7 +5,6 @@ from __future__ import annotations
5
5
 
6
6
  import logging
7
7
  import os
8
- import re
9
8
  from dataclasses import dataclass
10
9
 
11
10
  from gooddata_eval.core.chat.sse_client import ChatClient
@@ -15,36 +14,79 @@ from gooddata_eval.core.models import ToolCallEvent
15
14
  _log = logging.getLogger(__name__)
16
15
 
17
16
  _DEFAULT_K = 1
18
- # Disambiguation safety net only (create+execute always run together in the same
19
- # turn) -- 3 covers metric and period each needing their own clarifying question.
20
- _DEFAULT_MAX_ITERATIONS = 3
17
+ # Disambiguation safety net only (create+execute always run together in the same turn) --
18
+ # 3 real questions' worth (metric, period, +1 slack) since a simulated reply is now sent on
19
+ # every non-final turn (see run_agentic_kda_skill), not just ones classified as a question.
20
+ _DEFAULT_MAX_ITERATIONS = 4
21
21
 
22
22
 
23
- def _is_asking_kda_clarification(text: str) -> bool:
24
- """True if ``text`` reads as the agent asking for input, not a final answer.
25
-
26
- KDA-specific, not shared with metric_skill.py/conversation.py -- each skill's
27
- disambiguation heuristic has already drifted independently. Requires the text to
28
- end on "?" (a "?" anywhere also matches a final answer that merely quotes one).
23
+ def _build_period_hint(expected_output: dict) -> str | None:
24
+ """Build a period hint from whichever of expected_output's Date Attribute/Analyzed
25
+ Period/Reference Period are present -- a question about only one of them (e.g. "which
26
+ date dimension?") must still get an answerable hint, not None just because the other
27
+ two are absent.
28
+ """
29
+ date_attr = expected_output.get("Date Attribute")
30
+ analyzed = expected_output.get("Analyzed Period")
31
+ reference_period = expected_output.get("Reference Period")
32
+ if not (date_attr or analyzed or reference_period):
33
+ return None
34
+ parts = []
35
+ if date_attr:
36
+ parts.append(date_attr)
37
+ if analyzed and reference_period:
38
+ parts.append(f"comparing {analyzed} to {reference_period}")
39
+ elif analyzed:
40
+ parts.append(f"period {analyzed}")
41
+ elif reference_period:
42
+ parts.append(f"compared to {reference_period}")
43
+ return ", ".join(parts)
44
+
45
+
46
+ def _build_clarification_prompt(
47
+ agent_message: str, measure_candidates: dict | list[dict] | None, period_hint: str | None
48
+ ) -> str:
49
+ """Build the simulated-user prompt, referencing only whatever candidates/period-hint
50
+ are actually usable -- an empty/None candidate must drop the "acceptable metric/fact"
51
+ clause entirely rather than assert a literal "None" as if it were a real option.
29
52
  """
30
- if not text:
31
- return False
32
- t = text.strip().lower()
33
- if t.endswith("?"):
34
- return True
35
- # "To clarify, ..." means "in other words" (a final answer), not a request for one --
36
- # strip it first so "clarif" below only matches genuine clarification requests.
37
- t = re.sub(r"^(just )?to clarify,?\s*", "", t)
38
- return "could you" in t or "please provide" in t or "clarif" in t
39
-
40
-
41
- def generate_simulated_kda_response(agent_message: str, measure_candidates: dict | list[dict] | None) -> str:
53
+ candidates = [
54
+ c for c in (measure_candidates if isinstance(measure_candidates, list) else [measure_candidates]) if c
55
+ ]
56
+ reference = ""
57
+ if candidates:
58
+ candidate_desc = "; or ".join(
59
+ f"{c.get('type')} '{c.get('id')}'" + (f" (aggregation {c['aggregation']})" if c.get("aggregation") else "")
60
+ for c in candidates
61
+ )
62
+ reference = f"an acceptable metric/fact is {candidate_desc}"
63
+ if period_hint:
64
+ reference = (
65
+ f"{reference}; the intended time period is {period_hint}"
66
+ if reference
67
+ else f"the intended time period is {period_hint}"
68
+ )
69
+ return (
70
+ f"You are simulating a user in a conversation with a BI assistant that runs key driver "
71
+ f"analysis. The assistant asked: '{agent_message}'. "
72
+ + (f"For reference, {reference}. " if reference else "")
73
+ + "Reply briefly as the user, answering whichever of those the assistant actually asked about."
74
+ )
75
+
76
+
77
+ def generate_simulated_kda_response(
78
+ agent_message: str,
79
+ measure_candidates: dict | list[dict] | None,
80
+ period_hint: str | None = None,
81
+ ) -> str:
42
82
  """Generate a user reply to keep the KDA-skill conversation going (gpt-4o-mini).
43
83
 
44
- Used only when the agent asks a clarifying question instead of triggering KDA
45
- directly. Picks *any* candidate from ``measure_candidates`` -- scope only needs KDA
46
- to trigger, not the resulting measure to be exactly right. Always OpenAI regardless
47
- of the combo's own provider -- this is test-harness plumbing, not the system under test.
84
+ Called on any turn that didn't trigger KDA, whatever the agent's response actually
85
+ said -- most often a clarifying question about the measure, the period, or both, so
86
+ both are given as reference and the reply answers whichever was actually asked.
87
+ Scope only needs KDA to trigger, not the resulting measure/period to be exactly
88
+ right. Always OpenAI regardless of the combo's own provider -- this is
89
+ test-harness plumbing, not the system under test.
48
90
  """
49
91
  try:
50
92
  from openai import OpenAI # noqa: PLC0415
@@ -56,21 +98,12 @@ def generate_simulated_kda_response(agent_message: str, measure_candidates: dict
56
98
  raise OSError("OPENAI_API_KEY environment variable is not set")
57
99
 
58
100
  client = OpenAI(api_key=api_key)
59
- candidates = measure_candidates if isinstance(measure_candidates, list) else [measure_candidates or {}]
60
- candidate_desc = "; or ".join(
61
- f"{c.get('type')} '{c.get('id')}'" + (f" (aggregation {c['aggregation']})" if c.get("aggregation") else "")
62
- for c in candidates
63
- )
64
- prompt = (
65
- f"You are simulating a user in a conversation with a BI assistant that runs key driver "
66
- f"analysis. The assistant said: '{agent_message}'. "
67
- f"The user is happy to proceed with any of the following: {candidate_desc}. "
68
- f"Reply briefly as the user, picking whichever of those the assistant offered."
69
- )
101
+ prompt = _build_clarification_prompt(agent_message, measure_candidates, period_hint)
70
102
  response = client.chat.completions.create(
71
103
  model="gpt-4o-mini",
72
104
  messages=[{"role": "user", "content": prompt}],
73
105
  max_tokens=150,
106
+ temperature=0,
74
107
  timeout=30,
75
108
  )
76
109
  return response.choices[0].message.content or "Please proceed with either option."
@@ -171,9 +204,12 @@ def run_agentic_kda_skill(
171
204
 
172
205
  Each run is normally one message, one turn -- create and execute are always called
173
206
  together in the same turn (the skill's own system prompt: "NO confirmation needed").
174
- The only thing that can extend a run up to ``max_iterations`` turns is the agent
175
- asking a clarifying question instead of triggering KDA directly; a simulated user
176
- reply nudges it forward.
207
+ A run only extends past turn 1, up to ``max_iterations``, when the agent's response
208
+ has no create call and isn't empty; a simulated user reply is then always sent, with
209
+ no attempt to classify whether the text was actually asking for input (matching
210
+ visualization.py/alert_skill.py's own break conditions) -- missing a genuine
211
+ clarifying question hard-fails the run, while sending one after an unrecognized final
212
+ answer only costs one harmless extra turn, so the asymmetry favors never guessing.
177
213
  """
178
214
  if k < 1:
179
215
  # k=0 or negative would otherwise silently run once, indistinguishable from k=1.
@@ -212,17 +248,22 @@ def run_agentic_kda_skill(
212
248
  # execute tool isn't available at all when data-sharing is off for the org).
213
249
  turn_wall_clock_sec = chat_result.turn_wall_clock_sec
214
250
  break
251
+ if not response_text:
252
+ break
215
253
  if iteration >= max_iterations - 1:
216
254
  break
217
- if _is_asking_kda_clarification(response_text):
218
- measure_candidates = expected_output.get("Measure") if isinstance(expected_output, dict) else None
219
- try:
220
- current_question = generate_simulated_kda_response(response_text, measure_candidates)
221
- disambiguated = True
222
- except Exception as exc: # noqa: BLE001 -- safety net, not the assertion; end only this run
223
- _log.warning("Simulated KDA user reply failed for conversation %s: %s", conv_id, exc)
224
- break
225
- else:
255
+ # No text classification -- matches visualization.py/alert_skill.py: break only on
256
+ # the goal signal (create_args set) or an empty response, otherwise always send a
257
+ # simulated reply. A false positive (agent had already given a final answer) costs
258
+ # one harmless extra turn; a false negative (missing a genuine clarifying question)
259
+ # would hard-fail the run, so the asymmetry favors never trying to tell them apart.
260
+ measure_candidates = expected_output.get("Measure") if isinstance(expected_output, dict) else None
261
+ period_hint = _build_period_hint(expected_output) if isinstance(expected_output, dict) else None
262
+ try:
263
+ current_question = generate_simulated_kda_response(response_text, measure_candidates, period_hint)
264
+ disambiguated = True
265
+ except Exception as exc: # noqa: BLE001 -- safety net, not the assertion; end only this run
266
+ _log.warning("Simulated KDA user reply failed for conversation %s: %s", conv_id, exc)
226
267
  break
227
268
 
228
269
  ev = _evaluate_run(create_args, execute_result, turn_completed, disambiguated)
@@ -95,6 +95,7 @@ def generate_simulated_response(agent_message: str, expected_output: dict) -> st
95
95
  model="gpt-4o-mini",
96
96
  messages=[{"role": "user", "content": prompt}],
97
97
  max_tokens=150,
98
+ temperature=0,
98
99
  )
99
100
  return response.choices[0].message.content or "Please proceed."
100
101
 
@@ -141,7 +141,7 @@ def generate_simulated_response(agent_message: str, expected_output: CreatedVisu
141
141
  ),
142
142
  },
143
143
  ],
144
- temperature=0.5,
144
+ temperature=0,
145
145
  )
146
146
  content = response.choices[0].message.content
147
147
  return content.strip() if content else ""
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: gooddata-eval
3
- Version: 1.72.1.dev3
3
+ Version: 1.72.1.dev5
4
4
  Summary: Evaluate the GoodData AI agent against your own questions and models.
5
5
  Project-URL: Source, https://github.com/gooddata/gooddata-python-sdk
6
6
  Author-email: GoodData <support@gooddata.com>
@@ -17,7 +17,7 @@ Classifier: Topic :: Scientific/Engineering
17
17
  Classifier: Topic :: Software Development
18
18
  Classifier: Typing :: Typed
19
19
  Requires-Python: >=3.10
20
- Requires-Dist: gooddata-sdk~=1.72.1.dev3
20
+ Requires-Dist: gooddata-sdk~=1.72.1.dev5
21
21
  Requires-Dist: httpx<1.0,>=0.27
22
22
  Requires-Dist: orjson<4.0.0,>=3.9.15
23
23
  Requires-Dist: pydantic<3.0,>=2.6
@@ -13,14 +13,14 @@ gooddata_eval/core/workspace.py,sha256=S8exiy5gpDtcRtT8-i2fIRALF34Pbo0J46xm8X_ld
13
13
  gooddata_eval/core/agentic/__init__.py,sha256=M35wdSW7Bw5vidfHN4MuL0ZR8L7igUnF5mNWiboglcw,3030
14
14
  gooddata_eval/core/agentic/_catalog.py,sha256=eCaeyvLF-FCN0J3UnXSHaKtufU6HvrM2mxnmgyXsY1U,2085
15
15
  gooddata_eval/core/agentic/_langfuse.py,sha256=nRI_tcmORBWbuF3AJKAL8VX-mFkLyZBQXagCGSQ6xIE,15102
16
- gooddata_eval/core/agentic/alert_skill.py,sha256=Be7rf1sfToQfO_pzg1BwXJcC2MwlY01Vwg8egCxjREI,27226
17
- gooddata_eval/core/agentic/conversation.py,sha256=Yglf68PZ_Yb69z9ujTfX7HQcUrOzwlB85vnpul9SLMk,19073
16
+ gooddata_eval/core/agentic/alert_skill.py,sha256=xFYOmpDX5Qrjj0cVMilf99D7DfLHeVwMyrSftKB1f_c,27224
17
+ gooddata_eval/core/agentic/conversation.py,sha256=c0vj5nk1IsrHFiMAje4997sXo3vayk9oTTv5nkk-D6s,19071
18
18
  gooddata_eval/core/agentic/general_question.py,sha256=Tw2M-FJJWDo0O4leD5AoYW8UBpVQWSZrZJ0fNbHTHW8,8364
19
19
  gooddata_eval/core/agentic/guardrail.py,sha256=qE9vADYilDnmqjcwEE-HGF3XsOcGjdvsKFfLSX9IFYQ,8251
20
- gooddata_eval/core/agentic/kda_skill.py,sha256=d0lY0nUvb9D0Lu-bI_E55EecUFJnnqRS9EkUEaDVJBI,15837
21
- gooddata_eval/core/agentic/metric_skill.py,sha256=zZs8cRcFdDLIV_rY6wQKd9o-0VbQYK6EOL8KbmZpk4A,14117
20
+ gooddata_eval/core/agentic/kda_skill.py,sha256=XsKPhhEHQGJTPC3AufOgantxDTdFUwpjLpqlMfPLLEk,17974
21
+ gooddata_eval/core/agentic/metric_skill.py,sha256=0KmbVliieyL6p3z9AGmMDgMlBDBQlI1oBp1XOgZ-M8U,14140
22
22
  gooddata_eval/core/agentic/search_tool.py,sha256=L6yEs0hJymC_beYmily9ivw3M6tjLgDnF9_6kr7g11s,8020
23
- gooddata_eval/core/agentic/visualization.py,sha256=rZPN6CF_XpMsmPHSwzUP4iisLBL_99g2UIjIww_GbbQ,17129
23
+ gooddata_eval/core/agentic/visualization.py,sha256=qhwJ06_SauPjOl1yJ3teZmf42YQjpcSwTy3yuc3kz30,17127
24
24
  gooddata_eval/core/chat/__init__.py,sha256=U54lANKm34yjqm7dZL9KkGouPbwAaQWC9wiAmb5B5g4,32
25
25
  gooddata_eval/core/chat/sse_client.py,sha256=oYLxMWy6Hy0yoC8InnTkKLWN4Rsg1gSSIiE0FzaVa4o,16062
26
26
  gooddata_eval/core/dataset/__init__.py,sha256=U54lANKm34yjqm7dZL9KkGouPbwAaQWC9wiAmb5B5g4,32
@@ -45,8 +45,8 @@ gooddata_eval/core/reporting/console.py,sha256=kCHt4sCSaSRh0ZWqaSEJOl6_Jgb6uiouS
45
45
  gooddata_eval/core/reporting/json_report.py,sha256=sKPG9-uxzDsf6EJyBXoaHVIfN1g-E2rZ8fA0wtyBYb4,2943
46
46
  gooddata_eval/core/summary/__init__.py,sha256=U54lANKm34yjqm7dZL9KkGouPbwAaQWC9wiAmb5B5g4,32
47
47
  gooddata_eval/core/summary/http_client.py,sha256=dPArJ6zoCu1W9xmYXFA1WMDNsSOkhn3gaCpYgCUp3gk,2174
48
- gooddata_eval-1.72.1.dev3.dist-info/METADATA,sha256=ykwpTF22guapOxGKX7IHM8mZTAncsM7vZf1wCqYLBWM,9995
49
- gooddata_eval-1.72.1.dev3.dist-info/WHEEL,sha256=zOwg4jB6zX2kU910N-cMawjivD6tO8NEWvE12je1bVk,87
50
- gooddata_eval-1.72.1.dev3.dist-info/entry_points.txt,sha256=28nFp5Viknx4haPYgzK9QHlwPFfC6rPF6RmjC2ht8EI,56
51
- gooddata_eval-1.72.1.dev3.dist-info/licenses/LICENSE.txt,sha256=LVfVlC9maU3K9lgMwdZnClQQsOIbOekdcdVKr9qzJtk,256836
52
- gooddata_eval-1.72.1.dev3.dist-info/RECORD,,
48
+ gooddata_eval-1.72.1.dev5.dist-info/METADATA,sha256=8-k1gfi50ned3R52_nOj99ArVhTM-HMdhnD4CJh9HbI,9995
49
+ gooddata_eval-1.72.1.dev5.dist-info/WHEEL,sha256=zOwg4jB6zX2kU910N-cMawjivD6tO8NEWvE12je1bVk,87
50
+ gooddata_eval-1.72.1.dev5.dist-info/entry_points.txt,sha256=28nFp5Viknx4haPYgzK9QHlwPFfC6rPF6RmjC2ht8EI,56
51
+ gooddata_eval-1.72.1.dev5.dist-info/licenses/LICENSE.txt,sha256=LVfVlC9maU3K9lgMwdZnClQQsOIbOekdcdVKr9qzJtk,256836
52
+ gooddata_eval-1.72.1.dev5.dist-info/RECORD,,