gooddata-eval 1.73.0__tar.gz → 1.73.1.dev2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/PKG-INFO +2 -2
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/pyproject.toml +2 -2
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/cli/agentic_runner.py +10 -6
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/agentic/alert_skill.py +65 -6
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/agentic/conversation.py +60 -18
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/agentic/general_question.py +40 -4
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/agentic/guardrail.py +49 -4
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/agentic/kda_skill.py +59 -6
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/agentic/metric_skill.py +67 -13
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/agentic/search_tool.py +40 -5
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/agentic/visualization.py +69 -5
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/chat/sse_client.py +28 -3
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/evaluators/general_question.py +8 -2
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/evaluators/guardrail.py +11 -2
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/evaluators/search_tool.py +4 -1
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/evaluators/visualization.py +38 -18
- gooddata_eval-1.73.1.dev2/src/gooddata_eval/core/models.py +270 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/reporting/json_report.py +3 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/runner.py +19 -1
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/tests/test_agentic_alert_skill.py +35 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/tests/test_agentic_conversation.py +124 -0
- gooddata_eval-1.73.1.dev2/tests/test_agentic_general_question.py +210 -0
- gooddata_eval-1.73.1.dev2/tests/test_agentic_guardrail.py +210 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/tests/test_agentic_kda_skill.py +147 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/tests/test_agentic_metric_skill.py +121 -1
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/tests/test_agentic_runner.py +70 -23
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/tests/test_agentic_search_tool.py +96 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/tests/test_agentic_visualization.py +132 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/tests/test_runner.py +53 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/tests/test_sse_client.py +234 -4
- gooddata_eval-1.73.0/src/gooddata_eval/core/models.py +0 -146
- gooddata_eval-1.73.0/tests/test_agentic_general_question.py +0 -100
- gooddata_eval-1.73.0/tests/test_agentic_guardrail.py +0 -98
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/.gitignore +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/LICENSE.txt +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/Makefile +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/README.md +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/__init__.py +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/_version.py +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/cli/__init__.py +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/cli/main.py +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/__init__.py +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/agentic/__init__.py +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/agentic/_catalog.py +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/agentic/_langfuse.py +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/chat/__init__.py +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/config.py +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/connection.py +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/dataset/__init__.py +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/dataset/langfuse_source.py +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/dataset/local.py +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/evaluators/__init__.py +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/evaluators/_deep_subset.py +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/evaluators/_llm_judge.py +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/evaluators/_text_utils.py +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/evaluators/alert_skill.py +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/evaluators/base.py +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/evaluators/metric_skill.py +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/evaluators/summary.py +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/langfuse/__init__.py +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/langfuse/sink.py +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/reporting/__init__.py +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/reporting/console.py +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/scoring.py +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/summary/__init__.py +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/summary/http_client.py +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/workspace.py +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/tests/__init__.py +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/tests/conftest.py +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/tests/fixtures/sample_dataset/metric_skill_create.json +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/tests/fixtures/sample_dataset/visualization_revenue.json +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/tests/fixtures/sse_visualization_stream.txt +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/tests/test_agentic_langfuse_trace.py +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/tests/test_agentic_run_context.py +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/tests/test_alert_skill_evaluator.py +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/tests/test_cli.py +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/tests/test_connection.py +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/tests/test_deep_subset.py +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/tests/test_langfuse_sink.py +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/tests/test_langfuse_source.py +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/tests/test_llm_judge.py +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/tests/test_local_loader.py +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/tests/test_metric_skill_evaluator.py +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/tests/test_models.py +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/tests/test_reporting.py +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/tests/test_scoring.py +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/tests/test_search_tool_evaluator.py +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/tests/test_summary_client.py +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/tests/test_summary_evaluator.py +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/tests/test_text_evaluators.py +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/tests/test_visualization_evaluator.py +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/tests/test_workspace.py +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/tox.ini +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: gooddata-eval
|
|
3
|
-
Version: 1.73.
|
|
3
|
+
Version: 1.73.1.dev2
|
|
4
4
|
Summary: Evaluate the GoodData AI agent against your own questions and models.
|
|
5
5
|
Project-URL: Source, https://github.com/gooddata/gooddata-python-sdk
|
|
6
6
|
Author-email: GoodData <support@gooddata.com>
|
|
@@ -17,7 +17,7 @@ Classifier: Topic :: Scientific/Engineering
|
|
|
17
17
|
Classifier: Topic :: Software Development
|
|
18
18
|
Classifier: Typing :: Typed
|
|
19
19
|
Requires-Python: >=3.10
|
|
20
|
-
Requires-Dist: gooddata-sdk~=1.73.
|
|
20
|
+
Requires-Dist: gooddata-sdk~=1.73.1.dev2
|
|
21
21
|
Requires-Dist: httpx<1.0,>=0.27
|
|
22
22
|
Requires-Dist: orjson<4.0.0,>=3.9.15
|
|
23
23
|
Requires-Dist: pydantic<3.0,>=2.6
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
# (C) 2026 GoodData Corporation
|
|
2
2
|
[project]
|
|
3
3
|
name = "gooddata-eval"
|
|
4
|
-
version = "1.73.
|
|
4
|
+
version = "1.73.1.dev2"
|
|
5
5
|
description = "Evaluate the GoodData AI agent against your own questions and models."
|
|
6
6
|
readme = "README.md"
|
|
7
7
|
license = "MIT"
|
|
@@ -11,7 +11,7 @@ authors = [
|
|
|
11
11
|
keywords = ["gooddata", "ai", "evaluation", "llm", "analytics", "cli"]
|
|
12
12
|
requires-python = ">=3.10"
|
|
13
13
|
dependencies = [
|
|
14
|
-
"gooddata-sdk~=1.73.
|
|
14
|
+
"gooddata-sdk~=1.73.1.dev2",
|
|
15
15
|
"httpx>=0.27,<1.0",
|
|
16
16
|
"orjson>=3.9.15,<4.0.0",
|
|
17
17
|
"pydantic>=2.6,<3.0",
|
|
@@ -86,12 +86,12 @@ def _dispatch_agentic(
|
|
|
86
86
|
model_version_override: str | None,
|
|
87
87
|
reasoning_effort: ReasoningEffort | None = None,
|
|
88
88
|
agent_id: str | None = None,
|
|
89
|
-
) -> AgenticEvalOutcome
|
|
89
|
+
) -> AgenticEvalOutcome:
|
|
90
90
|
"""Call the appropriate evaluate_agentic_* function for the item's test_kind.
|
|
91
91
|
|
|
92
|
-
|
|
93
|
-
|
|
94
|
-
|
|
92
|
+
Every evaluate_agentic_* function returns an AgenticEvalOutcome (reasoning_steps,
|
|
93
|
+
conversation_id, response_id, detail) on success and attaches the same four attributes
|
|
94
|
+
to its raised *AssertionError on failure -- no kind is exempt.
|
|
95
95
|
"""
|
|
96
96
|
kind = item.test_kind
|
|
97
97
|
eo = item.expected_output
|
|
@@ -174,13 +174,14 @@ def _dispatch_agentic(
|
|
|
174
174
|
**lf_kw,
|
|
175
175
|
)
|
|
176
176
|
elif kind == "agentic_kda_skill":
|
|
177
|
-
evaluate_agentic_kda_skill(
|
|
177
|
+
return evaluate_agentic_kda_skill(
|
|
178
178
|
host=host,
|
|
179
179
|
token=token,
|
|
180
180
|
workspace_id=workspace_id,
|
|
181
181
|
question=item.question,
|
|
182
182
|
expected_output=eo if isinstance(eo, dict) else {},
|
|
183
183
|
k=k,
|
|
184
|
+
agent_id=agent_id,
|
|
184
185
|
**lf_kw,
|
|
185
186
|
)
|
|
186
187
|
elif kind == "agentic_conversation":
|
|
@@ -240,19 +241,22 @@ def run_agentic_items(
|
|
|
240
241
|
reasoning_steps = outcome.reasoning_steps
|
|
241
242
|
conversation_id = outcome.conversation_id
|
|
242
243
|
response_id = outcome.response_id
|
|
244
|
+
detail = outcome.detail
|
|
243
245
|
else:
|
|
244
|
-
reasoning_steps, conversation_id, response_id = outcome, None, None
|
|
246
|
+
reasoning_steps, conversation_id, response_id, detail = outcome, None, None, {}
|
|
245
247
|
item_report.pass_at_k = True
|
|
246
248
|
item_report.runs = k
|
|
247
249
|
item_report.reasoning_steps = reasoning_steps or []
|
|
248
250
|
item_report.conversation_id = conversation_id
|
|
249
251
|
item_report.response_id = response_id
|
|
252
|
+
item_report.best_detail = detail or {}
|
|
250
253
|
except AssertionError as exc:
|
|
251
254
|
item_report.pass_at_k = False
|
|
252
255
|
item_report.runs = k
|
|
253
256
|
item_report.reasoning_steps = getattr(exc, "reasoning_steps", None) or []
|
|
254
257
|
item_report.conversation_id = getattr(exc, "conversation_id", None)
|
|
255
258
|
item_report.response_id = getattr(exc, "response_id", None)
|
|
259
|
+
item_report.best_detail = getattr(exc, "detail", None) or {}
|
|
256
260
|
print(f"[agentic] {item.id} FAIL: {exc}", flush=True)
|
|
257
261
|
except Exception as exc:
|
|
258
262
|
item_report.error = f"{type(exc).__name__}: {exc}"
|
{gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/agentic/alert_skill.py
RENAMED
|
@@ -14,7 +14,7 @@ from gooddata_sdk import GoodDataSdk
|
|
|
14
14
|
from gooddata_eval.core.agentic._catalog import CatalogMetricAlert
|
|
15
15
|
from gooddata_eval.core.chat.sse_client import ChatClient
|
|
16
16
|
from gooddata_eval.core.config import ReasoningEffort
|
|
17
|
-
from gooddata_eval.core.models import AgenticEvalOutcome, ToolCallEvent
|
|
17
|
+
from gooddata_eval.core.models import AgenticEvalOutcome, ReasoningStepEvent, ToolCallEvent, build_latency_breakdown
|
|
18
18
|
|
|
19
19
|
try:
|
|
20
20
|
from openai import OpenAI as _OpenAI
|
|
@@ -160,8 +160,18 @@ def _check_recipients(expected: CatalogMetricAlert, actual_args: dict, sdk: Good
|
|
|
160
160
|
act_recip = []
|
|
161
161
|
if set(expected.recipients) == set(act_recip or []):
|
|
162
162
|
return True
|
|
163
|
-
|
|
164
|
-
|
|
163
|
+
act_internal_raw = actual_args.get("internal_recipients")
|
|
164
|
+
# internal_recipients is declared `anyOf: [array of string, string, null]` in the
|
|
165
|
+
# create_metric_alert tool schema -- a single id as a bare string is schema-legal,
|
|
166
|
+
# not a malformed call, so it needs the same string/list normalization already
|
|
167
|
+
# applied to recipients/external_recipients above.
|
|
168
|
+
if isinstance(act_internal_raw, str):
|
|
169
|
+
act_internal = [act_internal_raw]
|
|
170
|
+
elif isinstance(act_internal_raw, list):
|
|
171
|
+
act_internal = act_internal_raw
|
|
172
|
+
else:
|
|
173
|
+
act_internal = []
|
|
174
|
+
if sdk is not None and act_internal:
|
|
165
175
|
internal_recipient_ids = _resolve_internal_recipient_ids(sdk, expected.recipients)
|
|
166
176
|
if internal_recipient_ids & set(act_internal):
|
|
167
177
|
return True
|
|
@@ -335,6 +345,8 @@ class AlertRunResult:
|
|
|
335
345
|
actual_alert_arguments: dict
|
|
336
346
|
reasoning_steps: list[str] = field(default_factory=list)
|
|
337
347
|
response_id: str | None = None
|
|
348
|
+
tool_call_events: list[ToolCallEvent] = field(default_factory=list)
|
|
349
|
+
reasoning_step_events: list[ReasoningStepEvent] = field(default_factory=list)
|
|
338
350
|
|
|
339
351
|
|
|
340
352
|
@dataclass
|
|
@@ -485,6 +497,11 @@ def run_agentic_alert_skill(
|
|
|
485
497
|
tool_called = False
|
|
486
498
|
reasoning_steps: list[str] = []
|
|
487
499
|
response_id: str | None = None
|
|
500
|
+
all_tool_call_events: list[ToolCallEvent] = []
|
|
501
|
+
all_reasoning_step_events: list[ReasoningStepEvent] = []
|
|
502
|
+
turn_offset = 0.0 # each turn's call_ts/ts restarts near 0 -- shift by prior turns' wall time
|
|
503
|
+
tool_index_offset = 0
|
|
504
|
+
reasoning_index_offset = 0
|
|
488
505
|
# conversation_history stores prior turns for GPT-4o context.
|
|
489
506
|
# Roles follow GPT-4o's perspective: "assistant"=agent text, "user"=sim-user reply.
|
|
490
507
|
conversation_history: list = []
|
|
@@ -494,6 +511,21 @@ def run_agentic_alert_skill(
|
|
|
494
511
|
chat_result = client.send_message(conv_id, current_question)
|
|
495
512
|
reasoning_steps.extend(chat_result.reasoning_steps or [])
|
|
496
513
|
response_id = chat_result.response_id or response_id
|
|
514
|
+
for tc in chat_result.tool_call_events or []:
|
|
515
|
+
if tc.call_ts is not None:
|
|
516
|
+
tc.call_ts += turn_offset
|
|
517
|
+
if tc.result_ts is not None:
|
|
518
|
+
tc.result_ts += turn_offset
|
|
519
|
+
if tc.index is not None:
|
|
520
|
+
tc.index += tool_index_offset
|
|
521
|
+
for rs in chat_result.reasoning_step_events or []:
|
|
522
|
+
rs.ts += turn_offset
|
|
523
|
+
rs.index += reasoning_index_offset
|
|
524
|
+
all_tool_call_events.extend(chat_result.tool_call_events or [])
|
|
525
|
+
all_reasoning_step_events.extend(chat_result.reasoning_step_events or [])
|
|
526
|
+
tool_index_offset += len(chat_result.tool_call_events or [])
|
|
527
|
+
reasoning_index_offset += len(chat_result.reasoning_step_events or [])
|
|
528
|
+
turn_offset += chat_result.turn_wall_clock_sec or 0.0
|
|
497
529
|
alert_id, actual_args, tool_called = _extract_alert_call(chat_result.tool_call_events or [])
|
|
498
530
|
if tool_called:
|
|
499
531
|
alert_id_to_delete = alert_id
|
|
@@ -531,6 +563,8 @@ def run_agentic_alert_skill(
|
|
|
531
563
|
actual_alert_arguments=actual_args,
|
|
532
564
|
reasoning_steps=reasoning_steps,
|
|
533
565
|
response_id=response_id,
|
|
566
|
+
tool_call_events=all_tool_call_events,
|
|
567
|
+
reasoning_step_events=all_reasoning_step_events,
|
|
534
568
|
)
|
|
535
569
|
finally:
|
|
536
570
|
if alert_id_to_delete:
|
|
@@ -584,6 +618,7 @@ class AlertSkillAssertionError(AssertionError):
|
|
|
584
618
|
reasoning_steps: list[str]
|
|
585
619
|
conversation_id: str
|
|
586
620
|
response_id: str | None
|
|
621
|
+
detail: dict
|
|
587
622
|
|
|
588
623
|
|
|
589
624
|
def evaluate_agentic_alert_skill(
|
|
@@ -697,9 +732,33 @@ def evaluate_agentic_alert_skill(
|
|
|
697
732
|
exc.reasoning_steps = best.reasoning_steps
|
|
698
733
|
exc.conversation_id = best.conversation_id
|
|
699
734
|
exc.response_id = best.response_id
|
|
735
|
+
exc.detail = {
|
|
736
|
+
"alert_created": ev.alert_created,
|
|
737
|
+
"operator_correct": ev.operator_correct,
|
|
738
|
+
"threshold_correct": ev.threshold_correct,
|
|
739
|
+
"trigger_correct": ev.trigger_correct,
|
|
740
|
+
"filters_correct": ev.filters_correct,
|
|
741
|
+
"metric_correct": ev.metric_correct,
|
|
742
|
+
"recipients_correct": ev.recipients_correct,
|
|
743
|
+
"actual_alert_arguments": best.actual_alert_arguments,
|
|
744
|
+
"latency_breakdown": build_latency_breakdown(best.tool_call_events, best.reasoning_step_events),
|
|
745
|
+
}
|
|
700
746
|
raise exc
|
|
747
|
+
best = summary.best
|
|
748
|
+
ev = best.eval
|
|
701
749
|
return AgenticEvalOutcome(
|
|
702
|
-
reasoning_steps=
|
|
703
|
-
conversation_id=
|
|
704
|
-
response_id=
|
|
750
|
+
reasoning_steps=best.reasoning_steps,
|
|
751
|
+
conversation_id=best.conversation_id,
|
|
752
|
+
response_id=best.response_id,
|
|
753
|
+
detail={
|
|
754
|
+
"alert_created": ev.alert_created,
|
|
755
|
+
"operator_correct": ev.operator_correct,
|
|
756
|
+
"threshold_correct": ev.threshold_correct,
|
|
757
|
+
"trigger_correct": ev.trigger_correct,
|
|
758
|
+
"filters_correct": ev.filters_correct,
|
|
759
|
+
"metric_correct": ev.metric_correct,
|
|
760
|
+
"recipients_correct": ev.recipients_correct,
|
|
761
|
+
"actual_alert_arguments": best.actual_alert_arguments,
|
|
762
|
+
"latency_breakdown": build_latency_breakdown(best.tool_call_events, best.reasoning_step_events),
|
|
763
|
+
},
|
|
705
764
|
)
|
{gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/agentic/conversation.py
RENAMED
|
@@ -12,10 +12,16 @@ from gooddata_sdk import GoodDataSdk
|
|
|
12
12
|
from pydantic import BaseModel
|
|
13
13
|
|
|
14
14
|
from gooddata_eval.core.agentic.alert_skill import render_alert_proposal
|
|
15
|
-
from gooddata_eval.core.agentic.metric_skill import _delete_metric, _extract_created_metric_ids
|
|
15
|
+
from gooddata_eval.core.agentic.metric_skill import _delete_metric, _extract_created_metric_ids, _extract_metric_result
|
|
16
16
|
from gooddata_eval.core.chat.sse_client import ChatClient
|
|
17
17
|
from gooddata_eval.core.config import ReasoningEffort
|
|
18
|
-
from gooddata_eval.core.models import
|
|
18
|
+
from gooddata_eval.core.models import (
|
|
19
|
+
AgenticEvalOutcome,
|
|
20
|
+
ChatResult,
|
|
21
|
+
ReasoningStepEvent,
|
|
22
|
+
ToolCallEvent,
|
|
23
|
+
build_latency_breakdown,
|
|
24
|
+
)
|
|
19
25
|
from gooddata_eval.core.scoring import (
|
|
20
26
|
check_filters,
|
|
21
27
|
check_viz_type,
|
|
@@ -120,7 +126,7 @@ def _check_output_present(turn: TurnDefinition, chat_result: ChatResult) -> bool
|
|
|
120
126
|
and getattr(chat_result.created_visualizations, "objects", chat_result.created_visualizations)
|
|
121
127
|
)
|
|
122
128
|
if otype == "metric":
|
|
123
|
-
return
|
|
129
|
+
return _extract_metric_result(chat_result.tool_call_events or []) is not None
|
|
124
130
|
if otype == "tool_call":
|
|
125
131
|
expected_tool = turn.expected_tool_name
|
|
126
132
|
if not expected_tool:
|
|
@@ -129,19 +135,6 @@ def _check_output_present(turn: TurnDefinition, chat_result: ChatResult) -> bool
|
|
|
129
135
|
return False
|
|
130
136
|
|
|
131
137
|
|
|
132
|
-
def _extract_metric_from_turn(tool_call_events: list[ToolCallEvent]) -> dict | None:
|
|
133
|
-
"""Extract the result payload from the create_metric tool call, if present."""
|
|
134
|
-
for tc in tool_call_events:
|
|
135
|
-
if tc.function_name != "create_metric":
|
|
136
|
-
continue
|
|
137
|
-
if not tc.result:
|
|
138
|
-
continue
|
|
139
|
-
result_data = tc.parsed_result()
|
|
140
|
-
if result_data is not None:
|
|
141
|
-
return result_data.get("data", result_data)
|
|
142
|
-
return None
|
|
143
|
-
|
|
144
|
-
|
|
145
138
|
def _check_output_correct(turn: TurnDefinition, chat_result: ChatResult) -> bool | None:
|
|
146
139
|
"""Check output correctness against expected_output when defined.
|
|
147
140
|
|
|
@@ -186,7 +179,7 @@ def _check_output_correct(turn: TurnDefinition, chat_result: ChatResult) -> bool
|
|
|
186
179
|
return all(results) if results else None
|
|
187
180
|
|
|
188
181
|
if otype == "metric":
|
|
189
|
-
metric_result =
|
|
182
|
+
metric_result = _extract_metric_result(chat_result.tool_call_events or [])
|
|
190
183
|
if not metric_result:
|
|
191
184
|
return False
|
|
192
185
|
return _normalize_maql(metric_result.get("maql", "")) == _normalize_maql(expected.get("maql", ""))
|
|
@@ -267,6 +260,8 @@ class ConversationResult:
|
|
|
267
260
|
total_clarification_turns: int
|
|
268
261
|
reasoning_steps: list[str] = field(default_factory=list)
|
|
269
262
|
response_id: str | None = None
|
|
263
|
+
tool_call_events: list[ToolCallEvent] = field(default_factory=list)
|
|
264
|
+
reasoning_step_events: list[ReasoningStepEvent] = field(default_factory=list)
|
|
270
265
|
|
|
271
266
|
|
|
272
267
|
def run_agentic_conversation(
|
|
@@ -300,6 +295,14 @@ def run_agentic_conversation(
|
|
|
300
295
|
created_metric_ids: list[str] = []
|
|
301
296
|
reasoning_steps: list[str] = []
|
|
302
297
|
response_id: str | None = None
|
|
298
|
+
conversation_tool_call_events: list[ToolCallEvent] = []
|
|
299
|
+
conversation_reasoning_step_events: list[ReasoningStepEvent] = []
|
|
300
|
+
# Every send_message() call (across every logical turn AND every clarification
|
|
301
|
+
# sub-turn within it) restarts call_ts/ts near 0 -- these run across the whole
|
|
302
|
+
# conversation, not reset per logical turn, so every one of those calls shifts them.
|
|
303
|
+
turn_offset = 0.0
|
|
304
|
+
tool_index_offset = 0
|
|
305
|
+
reasoning_index_offset = 0
|
|
303
306
|
|
|
304
307
|
try:
|
|
305
308
|
if initial_conversation_id is not None:
|
|
@@ -335,7 +338,22 @@ def run_agentic_conversation(
|
|
|
335
338
|
for _iter in range(max_clarification_turns + 1):
|
|
336
339
|
chat_result = client.send_message(conversation_id, current_message)
|
|
337
340
|
final_result = chat_result
|
|
341
|
+
for tc in chat_result.tool_call_events or []:
|
|
342
|
+
if tc.call_ts is not None:
|
|
343
|
+
tc.call_ts += turn_offset
|
|
344
|
+
if tc.result_ts is not None:
|
|
345
|
+
tc.result_ts += turn_offset
|
|
346
|
+
if tc.index is not None:
|
|
347
|
+
tc.index += tool_index_offset
|
|
348
|
+
for rs in chat_result.reasoning_step_events or []:
|
|
349
|
+
rs.ts += turn_offset
|
|
350
|
+
rs.index += reasoning_index_offset
|
|
338
351
|
all_tool_calls.extend(chat_result.tool_call_events or [])
|
|
352
|
+
conversation_tool_call_events.extend(chat_result.tool_call_events or [])
|
|
353
|
+
conversation_reasoning_step_events.extend(chat_result.reasoning_step_events or [])
|
|
354
|
+
tool_index_offset += len(chat_result.tool_call_events or [])
|
|
355
|
+
reasoning_index_offset += len(chat_result.reasoning_step_events or [])
|
|
356
|
+
turn_offset += chat_result.turn_wall_clock_sec or 0.0
|
|
339
357
|
reasoning_steps.extend(chat_result.reasoning_steps or [])
|
|
340
358
|
response_id = chat_result.response_id or response_id
|
|
341
359
|
|
|
@@ -362,7 +380,7 @@ def run_agentic_conversation(
|
|
|
362
380
|
|
|
363
381
|
# Capture metric output for $ref resolution in subsequent turns.
|
|
364
382
|
if final_result and turn.expected_output_type == "metric":
|
|
365
|
-
metric_data =
|
|
383
|
+
metric_data = _extract_metric_result(all_tool_calls)
|
|
366
384
|
if metric_data:
|
|
367
385
|
turn_outputs[turn.turn_id] = metric_data
|
|
368
386
|
|
|
@@ -403,9 +421,30 @@ def run_agentic_conversation(
|
|
|
403
421
|
total_clarification_turns=total_clarification_turns,
|
|
404
422
|
reasoning_steps=reasoning_steps,
|
|
405
423
|
response_id=response_id,
|
|
424
|
+
tool_call_events=conversation_tool_call_events,
|
|
425
|
+
reasoning_step_events=conversation_reasoning_step_events,
|
|
406
426
|
)
|
|
407
427
|
|
|
408
428
|
|
|
429
|
+
def _conversation_detail(result: ConversationResult) -> dict:
|
|
430
|
+
return {
|
|
431
|
+
"full_skill_coverage": result.full_skill_coverage,
|
|
432
|
+
"total_clarification_turns": result.total_clarification_turns,
|
|
433
|
+
"turns": [
|
|
434
|
+
{
|
|
435
|
+
"turn_id": tr.turn_id,
|
|
436
|
+
"expected_skill": tr.expected_skill,
|
|
437
|
+
"skill_routing": tr.skill_routing,
|
|
438
|
+
"output_present": tr.output_present,
|
|
439
|
+
"output_correct": tr.output_correct,
|
|
440
|
+
"activated_skills": tr.activated_skills,
|
|
441
|
+
}
|
|
442
|
+
for tr in result.turn_results
|
|
443
|
+
],
|
|
444
|
+
"latency_breakdown": build_latency_breakdown(result.tool_call_events, result.reasoning_step_events),
|
|
445
|
+
}
|
|
446
|
+
|
|
447
|
+
|
|
409
448
|
class ConversationAssertionError(AssertionError):
|
|
410
449
|
"""Raised when a conversation evaluation fails."""
|
|
411
450
|
|
|
@@ -413,6 +452,7 @@ class ConversationAssertionError(AssertionError):
|
|
|
413
452
|
reasoning_steps: list[str]
|
|
414
453
|
conversation_id: str
|
|
415
454
|
response_id: str | None
|
|
455
|
+
detail: dict
|
|
416
456
|
|
|
417
457
|
|
|
418
458
|
def evaluate_agentic_conversation(
|
|
@@ -524,9 +564,11 @@ def evaluate_agentic_conversation(
|
|
|
524
564
|
exc.reasoning_steps = result.reasoning_steps
|
|
525
565
|
exc.conversation_id = result.conversation_id
|
|
526
566
|
exc.response_id = result.response_id
|
|
567
|
+
exc.detail = _conversation_detail(result)
|
|
527
568
|
raise exc
|
|
528
569
|
return AgenticEvalOutcome(
|
|
529
570
|
reasoning_steps=result.reasoning_steps,
|
|
530
571
|
conversation_id=result.conversation_id,
|
|
531
572
|
response_id=result.response_id,
|
|
573
|
+
detail=_conversation_detail(result),
|
|
532
574
|
)
|
|
@@ -3,11 +3,12 @@
|
|
|
3
3
|
|
|
4
4
|
from __future__ import annotations
|
|
5
5
|
|
|
6
|
-
from dataclasses import dataclass
|
|
6
|
+
from dataclasses import dataclass, field
|
|
7
7
|
|
|
8
8
|
from gooddata_eval.core.chat.sse_client import ChatClient
|
|
9
9
|
from gooddata_eval.core.config import ReasoningEffort
|
|
10
10
|
from gooddata_eval.core.evaluators._llm_judge import LLMJudge
|
|
11
|
+
from gooddata_eval.core.models import AgenticEvalOutcome
|
|
11
12
|
|
|
12
13
|
_DEFAULT_K = 1
|
|
13
14
|
|
|
@@ -52,6 +53,8 @@ class GeneralQuestionResult:
|
|
|
52
53
|
passed: bool
|
|
53
54
|
llm_judge_score: float
|
|
54
55
|
reasoning: str
|
|
56
|
+
reasoning_steps: list[str] = field(default_factory=list)
|
|
57
|
+
response_id: str | None = None
|
|
55
58
|
|
|
56
59
|
|
|
57
60
|
@dataclass
|
|
@@ -98,6 +101,8 @@ def run_agentic_general_question(
|
|
|
98
101
|
passed=passed,
|
|
99
102
|
llm_judge_score=llm_judge_score,
|
|
100
103
|
reasoning=reasoning,
|
|
104
|
+
reasoning_steps=list(chat_result.reasoning_steps or []),
|
|
105
|
+
response_id=chat_result.response_id,
|
|
101
106
|
)
|
|
102
107
|
)
|
|
103
108
|
finally:
|
|
@@ -120,6 +125,8 @@ def run_agentic_general_question(
|
|
|
120
125
|
passed=passed,
|
|
121
126
|
llm_judge_score=llm_judge_score,
|
|
122
127
|
reasoning=reasoning,
|
|
128
|
+
reasoning_steps=list(chat_result.reasoning_steps or []),
|
|
129
|
+
response_id=chat_result.response_id,
|
|
123
130
|
)
|
|
124
131
|
)
|
|
125
132
|
finally:
|
|
@@ -142,6 +149,10 @@ class GeneralQuestionAssertionError(AssertionError):
|
|
|
142
149
|
"""Raised when a general-question evaluation fails."""
|
|
143
150
|
|
|
144
151
|
__tracebackhide__ = True
|
|
152
|
+
reasoning_steps: list[str]
|
|
153
|
+
conversation_id: str
|
|
154
|
+
response_id: str | None
|
|
155
|
+
detail: dict
|
|
145
156
|
|
|
146
157
|
|
|
147
158
|
def evaluate_agentic_general_question(
|
|
@@ -160,8 +171,13 @@ def evaluate_agentic_general_question(
|
|
|
160
171
|
model_version_override: str | None = None,
|
|
161
172
|
run_metadata_extra: dict | None = None,
|
|
162
173
|
reasoning_effort: ReasoningEffort | None = None,
|
|
163
|
-
) ->
|
|
164
|
-
"""Run general-question evaluation, log to Langfuse, and raise on failure.
|
|
174
|
+
) -> AgenticEvalOutcome:
|
|
175
|
+
"""Run general-question evaluation, log to Langfuse, and raise GeneralQuestionAssertionError on failure.
|
|
176
|
+
|
|
177
|
+
Returns the best run's outcome (reasoning_steps, conversation_id, response_id) as an
|
|
178
|
+
AgenticEvalOutcome on success; on failure the same three values are attached to the
|
|
179
|
+
raised exception as ``.reasoning_steps``/``.conversation_id``/``.response_id``.
|
|
180
|
+
"""
|
|
165
181
|
from datetime import datetime as _dt # noqa: PLC0415
|
|
166
182
|
from datetime import timezone as _tz # noqa: PLC0415
|
|
167
183
|
|
|
@@ -223,6 +239,26 @@ def evaluate_agentic_general_question(
|
|
|
223
239
|
|
|
224
240
|
if not summary.pass_at_k:
|
|
225
241
|
best = summary.best
|
|
226
|
-
|
|
242
|
+
exc = GeneralQuestionAssertionError(
|
|
227
243
|
f"General question assertion failed. passed={best.passed}. Reasoning: {best.reasoning}"
|
|
228
244
|
)
|
|
245
|
+
exc.reasoning_steps = best.reasoning_steps
|
|
246
|
+
exc.conversation_id = best.conversation_id
|
|
247
|
+
exc.response_id = best.response_id
|
|
248
|
+
exc.detail = {
|
|
249
|
+
"judge_passed": best.passed,
|
|
250
|
+
"judge_reasoning": best.reasoning,
|
|
251
|
+
"actual_output": best.actual_output,
|
|
252
|
+
}
|
|
253
|
+
raise exc
|
|
254
|
+
best = summary.best
|
|
255
|
+
return AgenticEvalOutcome(
|
|
256
|
+
reasoning_steps=best.reasoning_steps,
|
|
257
|
+
conversation_id=best.conversation_id,
|
|
258
|
+
response_id=best.response_id,
|
|
259
|
+
detail={
|
|
260
|
+
"judge_passed": best.passed,
|
|
261
|
+
"judge_reasoning": best.reasoning,
|
|
262
|
+
"actual_output": best.actual_output,
|
|
263
|
+
},
|
|
264
|
+
)
|
{gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/agentic/guardrail.py
RENAMED
|
@@ -3,11 +3,12 @@
|
|
|
3
3
|
|
|
4
4
|
from __future__ import annotations
|
|
5
5
|
|
|
6
|
-
from dataclasses import dataclass
|
|
6
|
+
from dataclasses import dataclass, field
|
|
7
7
|
|
|
8
8
|
from gooddata_eval.core.chat.sse_client import ChatClient
|
|
9
9
|
from gooddata_eval.core.config import ReasoningEffort
|
|
10
10
|
from gooddata_eval.core.evaluators._llm_judge import LLMJudge
|
|
11
|
+
from gooddata_eval.core.models import AgenticEvalOutcome, ReasoningStepEvent, ToolCallEvent, build_latency_breakdown
|
|
11
12
|
|
|
12
13
|
_DEFAULT_K = 1
|
|
13
14
|
|
|
@@ -49,6 +50,10 @@ class GuardrailResult:
|
|
|
49
50
|
passed: bool
|
|
50
51
|
llm_judge_score: float
|
|
51
52
|
reasoning: str
|
|
53
|
+
reasoning_steps: list[str] = field(default_factory=list)
|
|
54
|
+
response_id: str | None = None
|
|
55
|
+
tool_call_events: list[ToolCallEvent] = field(default_factory=list)
|
|
56
|
+
reasoning_step_events: list[ReasoningStepEvent] = field(default_factory=list)
|
|
52
57
|
|
|
53
58
|
|
|
54
59
|
@dataclass
|
|
@@ -95,6 +100,10 @@ def run_agentic_guardrail(
|
|
|
95
100
|
passed=passed,
|
|
96
101
|
llm_judge_score=llm_judge_score,
|
|
97
102
|
reasoning=reasoning,
|
|
103
|
+
reasoning_steps=list(chat_result.reasoning_steps or []),
|
|
104
|
+
response_id=chat_result.response_id,
|
|
105
|
+
tool_call_events=list(chat_result.tool_call_events or []),
|
|
106
|
+
reasoning_step_events=list(chat_result.reasoning_step_events or []),
|
|
98
107
|
)
|
|
99
108
|
)
|
|
100
109
|
finally:
|
|
@@ -117,6 +126,10 @@ def run_agentic_guardrail(
|
|
|
117
126
|
passed=passed,
|
|
118
127
|
llm_judge_score=llm_judge_score,
|
|
119
128
|
reasoning=reasoning,
|
|
129
|
+
reasoning_steps=list(chat_result.reasoning_steps or []),
|
|
130
|
+
response_id=chat_result.response_id,
|
|
131
|
+
tool_call_events=list(chat_result.tool_call_events or []),
|
|
132
|
+
reasoning_step_events=list(chat_result.reasoning_step_events or []),
|
|
120
133
|
)
|
|
121
134
|
)
|
|
122
135
|
finally:
|
|
@@ -139,6 +152,10 @@ class GuardrailAssertionError(AssertionError):
|
|
|
139
152
|
"""Raised when a guardrail evaluation fails."""
|
|
140
153
|
|
|
141
154
|
__tracebackhide__ = True
|
|
155
|
+
reasoning_steps: list[str]
|
|
156
|
+
conversation_id: str
|
|
157
|
+
response_id: str | None
|
|
158
|
+
detail: dict
|
|
142
159
|
|
|
143
160
|
|
|
144
161
|
def evaluate_agentic_guardrail(
|
|
@@ -157,8 +174,14 @@ def evaluate_agentic_guardrail(
|
|
|
157
174
|
model_version_override: str | None = None,
|
|
158
175
|
run_metadata_extra: dict | None = None,
|
|
159
176
|
reasoning_effort: ReasoningEffort | None = None,
|
|
160
|
-
) ->
|
|
161
|
-
"""Run guardrail evaluation, log to Langfuse, and raise on failure.
|
|
177
|
+
) -> AgenticEvalOutcome:
|
|
178
|
+
"""Run guardrail evaluation, log to Langfuse, and raise GuardrailAssertionError on failure.
|
|
179
|
+
|
|
180
|
+
Returns the best run's outcome (reasoning_steps, conversation_id, response_id) as an
|
|
181
|
+
AgenticEvalOutcome on success; on failure the same three values are attached to the
|
|
182
|
+
raised exception as ``.reasoning_steps``/``.conversation_id``/``.response_id`` (mirrors
|
|
183
|
+
`evaluate_agentic_metric_skill`'s idiom) so callers can retrieve them either way.
|
|
184
|
+
"""
|
|
162
185
|
from datetime import datetime as _dt # noqa: PLC0415
|
|
163
186
|
from datetime import timezone as _tz # noqa: PLC0415
|
|
164
187
|
|
|
@@ -220,4 +243,26 @@ def evaluate_agentic_guardrail(
|
|
|
220
243
|
|
|
221
244
|
if not summary.pass_at_k:
|
|
222
245
|
best = summary.best
|
|
223
|
-
|
|
246
|
+
exc = GuardrailAssertionError(f"Guardrail assertion failed. passed={best.passed}. Reasoning: {best.reasoning}")
|
|
247
|
+
exc.reasoning_steps = best.reasoning_steps
|
|
248
|
+
exc.conversation_id = best.conversation_id
|
|
249
|
+
exc.response_id = best.response_id
|
|
250
|
+
exc.detail = {
|
|
251
|
+
"judge_passed": best.passed,
|
|
252
|
+
"judge_reasoning": best.reasoning,
|
|
253
|
+
"actual_output": best.actual_output,
|
|
254
|
+
"latency_breakdown": build_latency_breakdown(best.tool_call_events, best.reasoning_step_events),
|
|
255
|
+
}
|
|
256
|
+
raise exc
|
|
257
|
+
best = summary.best
|
|
258
|
+
return AgenticEvalOutcome(
|
|
259
|
+
reasoning_steps=best.reasoning_steps,
|
|
260
|
+
conversation_id=best.conversation_id,
|
|
261
|
+
response_id=best.response_id,
|
|
262
|
+
detail={
|
|
263
|
+
"judge_passed": best.passed,
|
|
264
|
+
"judge_reasoning": best.reasoning,
|
|
265
|
+
"actual_output": best.actual_output,
|
|
266
|
+
"latency_breakdown": build_latency_breakdown(best.tool_call_events, best.reasoning_step_events),
|
|
267
|
+
},
|
|
268
|
+
)
|