gooddata-eval 1.73.1.dev1__tar.gz → 1.73.1.dev2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev2}/PKG-INFO +2 -2
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev2}/pyproject.toml +2 -2
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/agentic/alert_skill.py +27 -1
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/agentic/conversation.py +35 -1
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/agentic/guardrail.py +9 -1
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/agentic/metric_skill.py +27 -1
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/agentic/visualization.py +37 -3
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/chat/sse_client.py +14 -2
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/evaluators/general_question.py +8 -2
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/evaluators/guardrail.py +11 -2
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/evaluators/search_tool.py +4 -1
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/evaluators/visualization.py +13 -2
- gooddata_eval-1.73.1.dev2/src/gooddata_eval/core/models.py +270 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/reporting/json_report.py +3 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/runner.py +19 -1
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev2}/tests/test_agentic_alert_skill.py +2 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev2}/tests/test_agentic_conversation.py +47 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev2}/tests/test_agentic_guardrail.py +2 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev2}/tests/test_agentic_metric_skill.py +2 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev2}/tests/test_agentic_visualization.py +2 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev2}/tests/test_runner.py +53 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev2}/tests/test_sse_client.py +192 -4
- gooddata_eval-1.73.1.dev1/src/gooddata_eval/core/models.py +0 -154
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev2}/.gitignore +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev2}/LICENSE.txt +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev2}/Makefile +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev2}/README.md +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/__init__.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/_version.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/cli/__init__.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/cli/agentic_runner.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/cli/main.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/__init__.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/agentic/__init__.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/agentic/_catalog.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/agentic/_langfuse.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/agentic/general_question.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/agentic/kda_skill.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/agentic/search_tool.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/chat/__init__.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/config.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/connection.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/dataset/__init__.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/dataset/langfuse_source.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/dataset/local.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/evaluators/__init__.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/evaluators/_deep_subset.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/evaluators/_llm_judge.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/evaluators/_text_utils.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/evaluators/alert_skill.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/evaluators/base.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/evaluators/metric_skill.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/evaluators/summary.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/langfuse/__init__.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/langfuse/sink.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/reporting/__init__.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/reporting/console.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/scoring.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/summary/__init__.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/summary/http_client.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/workspace.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev2}/tests/__init__.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev2}/tests/conftest.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev2}/tests/fixtures/sample_dataset/metric_skill_create.json +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev2}/tests/fixtures/sample_dataset/visualization_revenue.json +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev2}/tests/fixtures/sse_visualization_stream.txt +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev2}/tests/test_agentic_general_question.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev2}/tests/test_agentic_kda_skill.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev2}/tests/test_agentic_langfuse_trace.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev2}/tests/test_agentic_run_context.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev2}/tests/test_agentic_runner.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev2}/tests/test_agentic_search_tool.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev2}/tests/test_alert_skill_evaluator.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev2}/tests/test_cli.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev2}/tests/test_connection.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev2}/tests/test_deep_subset.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev2}/tests/test_langfuse_sink.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev2}/tests/test_langfuse_source.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev2}/tests/test_llm_judge.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev2}/tests/test_local_loader.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev2}/tests/test_metric_skill_evaluator.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev2}/tests/test_models.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev2}/tests/test_reporting.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev2}/tests/test_scoring.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev2}/tests/test_search_tool_evaluator.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev2}/tests/test_summary_client.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev2}/tests/test_summary_evaluator.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev2}/tests/test_text_evaluators.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev2}/tests/test_visualization_evaluator.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev2}/tests/test_workspace.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev2}/tox.ini +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: gooddata-eval
|
|
3
|
-
Version: 1.73.1.
|
|
3
|
+
Version: 1.73.1.dev2
|
|
4
4
|
Summary: Evaluate the GoodData AI agent against your own questions and models.
|
|
5
5
|
Project-URL: Source, https://github.com/gooddata/gooddata-python-sdk
|
|
6
6
|
Author-email: GoodData <support@gooddata.com>
|
|
@@ -17,7 +17,7 @@ Classifier: Topic :: Scientific/Engineering
|
|
|
17
17
|
Classifier: Topic :: Software Development
|
|
18
18
|
Classifier: Typing :: Typed
|
|
19
19
|
Requires-Python: >=3.10
|
|
20
|
-
Requires-Dist: gooddata-sdk~=1.73.1.
|
|
20
|
+
Requires-Dist: gooddata-sdk~=1.73.1.dev2
|
|
21
21
|
Requires-Dist: httpx<1.0,>=0.27
|
|
22
22
|
Requires-Dist: orjson<4.0.0,>=3.9.15
|
|
23
23
|
Requires-Dist: pydantic<3.0,>=2.6
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
# (C) 2026 GoodData Corporation
|
|
2
2
|
[project]
|
|
3
3
|
name = "gooddata-eval"
|
|
4
|
-
version = "1.73.1.
|
|
4
|
+
version = "1.73.1.dev2"
|
|
5
5
|
description = "Evaluate the GoodData AI agent against your own questions and models."
|
|
6
6
|
readme = "README.md"
|
|
7
7
|
license = "MIT"
|
|
@@ -11,7 +11,7 @@ authors = [
|
|
|
11
11
|
keywords = ["gooddata", "ai", "evaluation", "llm", "analytics", "cli"]
|
|
12
12
|
requires-python = ">=3.10"
|
|
13
13
|
dependencies = [
|
|
14
|
-
"gooddata-sdk~=1.73.1.
|
|
14
|
+
"gooddata-sdk~=1.73.1.dev2",
|
|
15
15
|
"httpx>=0.27,<1.0",
|
|
16
16
|
"orjson>=3.9.15,<4.0.0",
|
|
17
17
|
"pydantic>=2.6,<3.0",
|
|
@@ -14,7 +14,7 @@ from gooddata_sdk import GoodDataSdk
|
|
|
14
14
|
from gooddata_eval.core.agentic._catalog import CatalogMetricAlert
|
|
15
15
|
from gooddata_eval.core.chat.sse_client import ChatClient
|
|
16
16
|
from gooddata_eval.core.config import ReasoningEffort
|
|
17
|
-
from gooddata_eval.core.models import AgenticEvalOutcome, ToolCallEvent
|
|
17
|
+
from gooddata_eval.core.models import AgenticEvalOutcome, ReasoningStepEvent, ToolCallEvent, build_latency_breakdown
|
|
18
18
|
|
|
19
19
|
try:
|
|
20
20
|
from openai import OpenAI as _OpenAI
|
|
@@ -345,6 +345,8 @@ class AlertRunResult:
|
|
|
345
345
|
actual_alert_arguments: dict
|
|
346
346
|
reasoning_steps: list[str] = field(default_factory=list)
|
|
347
347
|
response_id: str | None = None
|
|
348
|
+
tool_call_events: list[ToolCallEvent] = field(default_factory=list)
|
|
349
|
+
reasoning_step_events: list[ReasoningStepEvent] = field(default_factory=list)
|
|
348
350
|
|
|
349
351
|
|
|
350
352
|
@dataclass
|
|
@@ -495,6 +497,11 @@ def run_agentic_alert_skill(
|
|
|
495
497
|
tool_called = False
|
|
496
498
|
reasoning_steps: list[str] = []
|
|
497
499
|
response_id: str | None = None
|
|
500
|
+
all_tool_call_events: list[ToolCallEvent] = []
|
|
501
|
+
all_reasoning_step_events: list[ReasoningStepEvent] = []
|
|
502
|
+
turn_offset = 0.0 # each turn's call_ts/ts restarts near 0 -- shift by prior turns' wall time
|
|
503
|
+
tool_index_offset = 0
|
|
504
|
+
reasoning_index_offset = 0
|
|
498
505
|
# conversation_history stores prior turns for GPT-4o context.
|
|
499
506
|
# Roles follow GPT-4o's perspective: "assistant"=agent text, "user"=sim-user reply.
|
|
500
507
|
conversation_history: list = []
|
|
@@ -504,6 +511,21 @@ def run_agentic_alert_skill(
|
|
|
504
511
|
chat_result = client.send_message(conv_id, current_question)
|
|
505
512
|
reasoning_steps.extend(chat_result.reasoning_steps or [])
|
|
506
513
|
response_id = chat_result.response_id or response_id
|
|
514
|
+
for tc in chat_result.tool_call_events or []:
|
|
515
|
+
if tc.call_ts is not None:
|
|
516
|
+
tc.call_ts += turn_offset
|
|
517
|
+
if tc.result_ts is not None:
|
|
518
|
+
tc.result_ts += turn_offset
|
|
519
|
+
if tc.index is not None:
|
|
520
|
+
tc.index += tool_index_offset
|
|
521
|
+
for rs in chat_result.reasoning_step_events or []:
|
|
522
|
+
rs.ts += turn_offset
|
|
523
|
+
rs.index += reasoning_index_offset
|
|
524
|
+
all_tool_call_events.extend(chat_result.tool_call_events or [])
|
|
525
|
+
all_reasoning_step_events.extend(chat_result.reasoning_step_events or [])
|
|
526
|
+
tool_index_offset += len(chat_result.tool_call_events or [])
|
|
527
|
+
reasoning_index_offset += len(chat_result.reasoning_step_events or [])
|
|
528
|
+
turn_offset += chat_result.turn_wall_clock_sec or 0.0
|
|
507
529
|
alert_id, actual_args, tool_called = _extract_alert_call(chat_result.tool_call_events or [])
|
|
508
530
|
if tool_called:
|
|
509
531
|
alert_id_to_delete = alert_id
|
|
@@ -541,6 +563,8 @@ def run_agentic_alert_skill(
|
|
|
541
563
|
actual_alert_arguments=actual_args,
|
|
542
564
|
reasoning_steps=reasoning_steps,
|
|
543
565
|
response_id=response_id,
|
|
566
|
+
tool_call_events=all_tool_call_events,
|
|
567
|
+
reasoning_step_events=all_reasoning_step_events,
|
|
544
568
|
)
|
|
545
569
|
finally:
|
|
546
570
|
if alert_id_to_delete:
|
|
@@ -717,6 +741,7 @@ def evaluate_agentic_alert_skill(
|
|
|
717
741
|
"metric_correct": ev.metric_correct,
|
|
718
742
|
"recipients_correct": ev.recipients_correct,
|
|
719
743
|
"actual_alert_arguments": best.actual_alert_arguments,
|
|
744
|
+
"latency_breakdown": build_latency_breakdown(best.tool_call_events, best.reasoning_step_events),
|
|
720
745
|
}
|
|
721
746
|
raise exc
|
|
722
747
|
best = summary.best
|
|
@@ -734,5 +759,6 @@ def evaluate_agentic_alert_skill(
|
|
|
734
759
|
"metric_correct": ev.metric_correct,
|
|
735
760
|
"recipients_correct": ev.recipients_correct,
|
|
736
761
|
"actual_alert_arguments": best.actual_alert_arguments,
|
|
762
|
+
"latency_breakdown": build_latency_breakdown(best.tool_call_events, best.reasoning_step_events),
|
|
737
763
|
},
|
|
738
764
|
)
|
|
@@ -15,7 +15,13 @@ from gooddata_eval.core.agentic.alert_skill import render_alert_proposal
|
|
|
15
15
|
from gooddata_eval.core.agentic.metric_skill import _delete_metric, _extract_created_metric_ids, _extract_metric_result
|
|
16
16
|
from gooddata_eval.core.chat.sse_client import ChatClient
|
|
17
17
|
from gooddata_eval.core.config import ReasoningEffort
|
|
18
|
-
from gooddata_eval.core.models import
|
|
18
|
+
from gooddata_eval.core.models import (
|
|
19
|
+
AgenticEvalOutcome,
|
|
20
|
+
ChatResult,
|
|
21
|
+
ReasoningStepEvent,
|
|
22
|
+
ToolCallEvent,
|
|
23
|
+
build_latency_breakdown,
|
|
24
|
+
)
|
|
19
25
|
from gooddata_eval.core.scoring import (
|
|
20
26
|
check_filters,
|
|
21
27
|
check_viz_type,
|
|
@@ -254,6 +260,8 @@ class ConversationResult:
|
|
|
254
260
|
total_clarification_turns: int
|
|
255
261
|
reasoning_steps: list[str] = field(default_factory=list)
|
|
256
262
|
response_id: str | None = None
|
|
263
|
+
tool_call_events: list[ToolCallEvent] = field(default_factory=list)
|
|
264
|
+
reasoning_step_events: list[ReasoningStepEvent] = field(default_factory=list)
|
|
257
265
|
|
|
258
266
|
|
|
259
267
|
def run_agentic_conversation(
|
|
@@ -287,6 +295,14 @@ def run_agentic_conversation(
|
|
|
287
295
|
created_metric_ids: list[str] = []
|
|
288
296
|
reasoning_steps: list[str] = []
|
|
289
297
|
response_id: str | None = None
|
|
298
|
+
conversation_tool_call_events: list[ToolCallEvent] = []
|
|
299
|
+
conversation_reasoning_step_events: list[ReasoningStepEvent] = []
|
|
300
|
+
# Every send_message() call (across every logical turn AND every clarification
|
|
301
|
+
# sub-turn within it) restarts call_ts/ts near 0 -- these run across the whole
|
|
302
|
+
# conversation, not reset per logical turn, so every one of those calls shifts them.
|
|
303
|
+
turn_offset = 0.0
|
|
304
|
+
tool_index_offset = 0
|
|
305
|
+
reasoning_index_offset = 0
|
|
290
306
|
|
|
291
307
|
try:
|
|
292
308
|
if initial_conversation_id is not None:
|
|
@@ -322,7 +338,22 @@ def run_agentic_conversation(
|
|
|
322
338
|
for _iter in range(max_clarification_turns + 1):
|
|
323
339
|
chat_result = client.send_message(conversation_id, current_message)
|
|
324
340
|
final_result = chat_result
|
|
341
|
+
for tc in chat_result.tool_call_events or []:
|
|
342
|
+
if tc.call_ts is not None:
|
|
343
|
+
tc.call_ts += turn_offset
|
|
344
|
+
if tc.result_ts is not None:
|
|
345
|
+
tc.result_ts += turn_offset
|
|
346
|
+
if tc.index is not None:
|
|
347
|
+
tc.index += tool_index_offset
|
|
348
|
+
for rs in chat_result.reasoning_step_events or []:
|
|
349
|
+
rs.ts += turn_offset
|
|
350
|
+
rs.index += reasoning_index_offset
|
|
325
351
|
all_tool_calls.extend(chat_result.tool_call_events or [])
|
|
352
|
+
conversation_tool_call_events.extend(chat_result.tool_call_events or [])
|
|
353
|
+
conversation_reasoning_step_events.extend(chat_result.reasoning_step_events or [])
|
|
354
|
+
tool_index_offset += len(chat_result.tool_call_events or [])
|
|
355
|
+
reasoning_index_offset += len(chat_result.reasoning_step_events or [])
|
|
356
|
+
turn_offset += chat_result.turn_wall_clock_sec or 0.0
|
|
326
357
|
reasoning_steps.extend(chat_result.reasoning_steps or [])
|
|
327
358
|
response_id = chat_result.response_id or response_id
|
|
328
359
|
|
|
@@ -390,6 +421,8 @@ def run_agentic_conversation(
|
|
|
390
421
|
total_clarification_turns=total_clarification_turns,
|
|
391
422
|
reasoning_steps=reasoning_steps,
|
|
392
423
|
response_id=response_id,
|
|
424
|
+
tool_call_events=conversation_tool_call_events,
|
|
425
|
+
reasoning_step_events=conversation_reasoning_step_events,
|
|
393
426
|
)
|
|
394
427
|
|
|
395
428
|
|
|
@@ -408,6 +441,7 @@ def _conversation_detail(result: ConversationResult) -> dict:
|
|
|
408
441
|
}
|
|
409
442
|
for tr in result.turn_results
|
|
410
443
|
],
|
|
444
|
+
"latency_breakdown": build_latency_breakdown(result.tool_call_events, result.reasoning_step_events),
|
|
411
445
|
}
|
|
412
446
|
|
|
413
447
|
|
{gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/agentic/guardrail.py
RENAMED
|
@@ -8,7 +8,7 @@ from dataclasses import dataclass, field
|
|
|
8
8
|
from gooddata_eval.core.chat.sse_client import ChatClient
|
|
9
9
|
from gooddata_eval.core.config import ReasoningEffort
|
|
10
10
|
from gooddata_eval.core.evaluators._llm_judge import LLMJudge
|
|
11
|
-
from gooddata_eval.core.models import AgenticEvalOutcome
|
|
11
|
+
from gooddata_eval.core.models import AgenticEvalOutcome, ReasoningStepEvent, ToolCallEvent, build_latency_breakdown
|
|
12
12
|
|
|
13
13
|
_DEFAULT_K = 1
|
|
14
14
|
|
|
@@ -52,6 +52,8 @@ class GuardrailResult:
|
|
|
52
52
|
reasoning: str
|
|
53
53
|
reasoning_steps: list[str] = field(default_factory=list)
|
|
54
54
|
response_id: str | None = None
|
|
55
|
+
tool_call_events: list[ToolCallEvent] = field(default_factory=list)
|
|
56
|
+
reasoning_step_events: list[ReasoningStepEvent] = field(default_factory=list)
|
|
55
57
|
|
|
56
58
|
|
|
57
59
|
@dataclass
|
|
@@ -100,6 +102,8 @@ def run_agentic_guardrail(
|
|
|
100
102
|
reasoning=reasoning,
|
|
101
103
|
reasoning_steps=list(chat_result.reasoning_steps or []),
|
|
102
104
|
response_id=chat_result.response_id,
|
|
105
|
+
tool_call_events=list(chat_result.tool_call_events or []),
|
|
106
|
+
reasoning_step_events=list(chat_result.reasoning_step_events or []),
|
|
103
107
|
)
|
|
104
108
|
)
|
|
105
109
|
finally:
|
|
@@ -124,6 +128,8 @@ def run_agentic_guardrail(
|
|
|
124
128
|
reasoning=reasoning,
|
|
125
129
|
reasoning_steps=list(chat_result.reasoning_steps or []),
|
|
126
130
|
response_id=chat_result.response_id,
|
|
131
|
+
tool_call_events=list(chat_result.tool_call_events or []),
|
|
132
|
+
reasoning_step_events=list(chat_result.reasoning_step_events or []),
|
|
127
133
|
)
|
|
128
134
|
)
|
|
129
135
|
finally:
|
|
@@ -245,6 +251,7 @@ def evaluate_agentic_guardrail(
|
|
|
245
251
|
"judge_passed": best.passed,
|
|
246
252
|
"judge_reasoning": best.reasoning,
|
|
247
253
|
"actual_output": best.actual_output,
|
|
254
|
+
"latency_breakdown": build_latency_breakdown(best.tool_call_events, best.reasoning_step_events),
|
|
248
255
|
}
|
|
249
256
|
raise exc
|
|
250
257
|
best = summary.best
|
|
@@ -256,5 +263,6 @@ def evaluate_agentic_guardrail(
|
|
|
256
263
|
"judge_passed": best.passed,
|
|
257
264
|
"judge_reasoning": best.reasoning,
|
|
258
265
|
"actual_output": best.actual_output,
|
|
266
|
+
"latency_breakdown": build_latency_breakdown(best.tool_call_events, best.reasoning_step_events),
|
|
259
267
|
},
|
|
260
268
|
)
|
|
@@ -12,7 +12,7 @@ from gooddata_sdk import GoodDataSdk
|
|
|
12
12
|
|
|
13
13
|
from gooddata_eval.core.chat.sse_client import ChatClient
|
|
14
14
|
from gooddata_eval.core.config import ReasoningEffort
|
|
15
|
-
from gooddata_eval.core.models import AgenticEvalOutcome, ToolCallEvent
|
|
15
|
+
from gooddata_eval.core.models import AgenticEvalOutcome, ReasoningStepEvent, ToolCallEvent, build_latency_breakdown
|
|
16
16
|
|
|
17
17
|
try:
|
|
18
18
|
from openai import OpenAI as _OpenAI
|
|
@@ -150,6 +150,8 @@ class MetricRunResult:
|
|
|
150
150
|
total_turns: float
|
|
151
151
|
reasoning_steps: list[str] = field(default_factory=list)
|
|
152
152
|
response_id: str | None = None
|
|
153
|
+
tool_call_events: list[ToolCallEvent] = field(default_factory=list)
|
|
154
|
+
reasoning_step_events: list[ReasoningStepEvent] = field(default_factory=list)
|
|
153
155
|
|
|
154
156
|
|
|
155
157
|
@dataclass
|
|
@@ -239,6 +241,11 @@ def _execute_single_metric_run(
|
|
|
239
241
|
current_question = question
|
|
240
242
|
reasoning_steps: list[str] = []
|
|
241
243
|
response_id: str | None = None
|
|
244
|
+
all_tool_call_events: list[ToolCallEvent] = []
|
|
245
|
+
all_reasoning_step_events: list[ReasoningStepEvent] = []
|
|
246
|
+
turn_offset = 0.0 # each turn's call_ts/ts restarts near 0 -- shift by prior turns' wall time
|
|
247
|
+
tool_index_offset = 0
|
|
248
|
+
reasoning_index_offset = 0
|
|
242
249
|
|
|
243
250
|
try:
|
|
244
251
|
for _iteration in range(max_iterations):
|
|
@@ -246,6 +253,21 @@ def _execute_single_metric_run(
|
|
|
246
253
|
chat_result = client.send_message(conversation_id, current_question)
|
|
247
254
|
reasoning_steps.extend(chat_result.reasoning_steps or [])
|
|
248
255
|
response_id = chat_result.response_id or response_id
|
|
256
|
+
for tc in chat_result.tool_call_events or []:
|
|
257
|
+
if tc.call_ts is not None:
|
|
258
|
+
tc.call_ts += turn_offset
|
|
259
|
+
if tc.result_ts is not None:
|
|
260
|
+
tc.result_ts += turn_offset
|
|
261
|
+
if tc.index is not None:
|
|
262
|
+
tc.index += tool_index_offset
|
|
263
|
+
for rs in chat_result.reasoning_step_events or []:
|
|
264
|
+
rs.ts += turn_offset
|
|
265
|
+
rs.index += reasoning_index_offset
|
|
266
|
+
all_tool_call_events.extend(chat_result.tool_call_events or [])
|
|
267
|
+
all_reasoning_step_events.extend(chat_result.reasoning_step_events or [])
|
|
268
|
+
tool_index_offset += len(chat_result.tool_call_events or [])
|
|
269
|
+
reasoning_index_offset += len(chat_result.reasoning_step_events or [])
|
|
270
|
+
turn_offset += chat_result.turn_wall_clock_sec or 0.0
|
|
249
271
|
for metric_id in _extract_created_metric_ids(chat_result.tool_call_events or []):
|
|
250
272
|
if metric_id not in created_metric_ids:
|
|
251
273
|
created_metric_ids.append(metric_id)
|
|
@@ -276,6 +298,8 @@ def _execute_single_metric_run(
|
|
|
276
298
|
total_turns=float(turns),
|
|
277
299
|
reasoning_steps=reasoning_steps,
|
|
278
300
|
response_id=response_id,
|
|
301
|
+
tool_call_events=all_tool_call_events,
|
|
302
|
+
reasoning_step_events=all_reasoning_step_events,
|
|
279
303
|
)
|
|
280
304
|
finally:
|
|
281
305
|
for metric_id in created_metric_ids:
|
|
@@ -457,6 +481,7 @@ def evaluate_agentic_metric_skill(
|
|
|
457
481
|
"maql_correct": best.maql_correct,
|
|
458
482
|
"expected_maql_candidates": [c.get("maql", "") for c in expected_outputs_list],
|
|
459
483
|
"actual_maql": best.actual_maql,
|
|
484
|
+
"latency_breakdown": build_latency_breakdown(best.tool_call_events, best.reasoning_step_events),
|
|
460
485
|
}
|
|
461
486
|
raise exc
|
|
462
487
|
best = summary.best
|
|
@@ -470,5 +495,6 @@ def evaluate_agentic_metric_skill(
|
|
|
470
495
|
"maql_correct": best.maql_correct,
|
|
471
496
|
"expected_maql_candidates": [c.get("maql", "") for c in expected_outputs_list],
|
|
472
497
|
"actual_maql": best.actual_maql,
|
|
498
|
+
"latency_breakdown": build_latency_breakdown(best.tool_call_events, best.reasoning_step_events),
|
|
473
499
|
},
|
|
474
500
|
)
|
|
@@ -18,7 +18,13 @@ from gooddata_eval.core.evaluators.visualization import (
|
|
|
18
18
|
_evaluate_against_candidates,
|
|
19
19
|
evaluation_result_detail,
|
|
20
20
|
)
|
|
21
|
-
from gooddata_eval.core.models import
|
|
21
|
+
from gooddata_eval.core.models import (
|
|
22
|
+
AgenticEvalOutcome,
|
|
23
|
+
CreatedVisualization,
|
|
24
|
+
ReasoningStepEvent,
|
|
25
|
+
ToolCallEvent,
|
|
26
|
+
build_latency_breakdown,
|
|
27
|
+
)
|
|
22
28
|
from gooddata_eval.core.scoring import get_dimension_uri_set, get_metric_uri_set, uri_to_display_name
|
|
23
29
|
|
|
24
30
|
_DEFAULT_K = 2
|
|
@@ -37,6 +43,8 @@ class RunResult:
|
|
|
37
43
|
total_steps: float
|
|
38
44
|
reasoning_steps: list[str] = field(default_factory=list)
|
|
39
45
|
response_id: str | None = None
|
|
46
|
+
tool_call_events: list[ToolCallEvent] = field(default_factory=list)
|
|
47
|
+
reasoning_step_events: list[ReasoningStepEvent] = field(default_factory=list)
|
|
40
48
|
|
|
41
49
|
|
|
42
50
|
@dataclass
|
|
@@ -161,8 +169,12 @@ def _execute_single_run(
|
|
|
161
169
|
total_turns = 0.0
|
|
162
170
|
total_steps = 0.0
|
|
163
171
|
all_tool_call_events: list[ToolCallEvent] = []
|
|
172
|
+
all_reasoning_step_events: list[ReasoningStepEvent] = []
|
|
164
173
|
reasoning_steps: list[str] = []
|
|
165
174
|
response_id: str | None = None
|
|
175
|
+
turn_offset = 0.0 # each turn's call_ts/ts restarts near 0 -- shift by prior turns' wall time
|
|
176
|
+
reasoning_index_offset = 0 # ditto for ReasoningStepEvent.index, which also restarts per turn
|
|
177
|
+
tool_index_offset = 0 # ditto for ToolCallEvent.index
|
|
166
178
|
simulated_response_guide = expected_outputs[0] # primary candidate guides the simulated user
|
|
167
179
|
|
|
168
180
|
current_result = client.send_message(conversation_id, question)
|
|
@@ -170,9 +182,23 @@ def _execute_single_run(
|
|
|
170
182
|
for iteration in range(max_iterations):
|
|
171
183
|
total_turns += 1.0
|
|
172
184
|
total_steps += float(current_result.reasoning_step_count)
|
|
185
|
+
for tc in current_result.tool_call_events:
|
|
186
|
+
if tc.call_ts is not None:
|
|
187
|
+
tc.call_ts += turn_offset
|
|
188
|
+
if tc.result_ts is not None:
|
|
189
|
+
tc.result_ts += turn_offset
|
|
190
|
+
if tc.index is not None:
|
|
191
|
+
tc.index += tool_index_offset
|
|
192
|
+
for rs in current_result.reasoning_step_events:
|
|
193
|
+
rs.ts += turn_offset
|
|
194
|
+
rs.index += reasoning_index_offset
|
|
173
195
|
all_tool_call_events.extend(current_result.tool_call_events)
|
|
196
|
+
all_reasoning_step_events.extend(current_result.reasoning_step_events)
|
|
197
|
+
tool_index_offset += len(current_result.tool_call_events)
|
|
198
|
+
reasoning_index_offset += len(current_result.reasoning_step_events)
|
|
174
199
|
reasoning_steps.extend(current_result.reasoning_steps or [])
|
|
175
200
|
response_id = current_result.response_id or response_id
|
|
201
|
+
turn_offset += current_result.turn_wall_clock_sec or 0.0
|
|
176
202
|
|
|
177
203
|
viz_produced = bool(current_result.created_visualizations and current_result.created_visualizations.objects)
|
|
178
204
|
if viz_produced:
|
|
@@ -201,6 +227,8 @@ def _execute_single_run(
|
|
|
201
227
|
total_steps=total_steps,
|
|
202
228
|
reasoning_steps=reasoning_steps,
|
|
203
229
|
response_id=response_id,
|
|
230
|
+
tool_call_events=all_tool_call_events,
|
|
231
|
+
reasoning_step_events=all_reasoning_step_events,
|
|
204
232
|
)
|
|
205
233
|
|
|
206
234
|
|
|
@@ -434,12 +462,18 @@ def evaluate_agentic_visualization(
|
|
|
434
462
|
exc.reasoning_steps = best.reasoning_steps
|
|
435
463
|
exc.conversation_id = best.conversation_id
|
|
436
464
|
exc.response_id = best.response_id
|
|
437
|
-
exc.detail =
|
|
465
|
+
exc.detail = {
|
|
466
|
+
**evaluation_result_detail(ev),
|
|
467
|
+
"latency_breakdown": build_latency_breakdown(best.tool_call_events, best.reasoning_step_events),
|
|
468
|
+
}
|
|
438
469
|
raise exc
|
|
439
470
|
best = summary.best
|
|
440
471
|
return AgenticEvalOutcome(
|
|
441
472
|
reasoning_steps=best.reasoning_steps,
|
|
442
473
|
conversation_id=best.conversation_id,
|
|
443
474
|
response_id=best.response_id,
|
|
444
|
-
detail=
|
|
475
|
+
detail={
|
|
476
|
+
**evaluation_result_detail(best.eval_result),
|
|
477
|
+
"latency_breakdown": build_latency_breakdown(best.tool_call_events, best.reasoning_step_events),
|
|
478
|
+
},
|
|
445
479
|
)
|
{gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev2}/src/gooddata_eval/core/chat/sse_client.py
RENAMED
|
@@ -132,6 +132,11 @@ class _SseAccumulator:
|
|
|
132
132
|
adhoc_viz_args: list[dict[str, Any]] = field(default_factory=list)
|
|
133
133
|
response_id: str | None = None
|
|
134
134
|
stream_ended: bool = False
|
|
135
|
+
# Reference point for call_ts/result_ts below -- client-observed receipt time, not a
|
|
136
|
+
# server timestamp, so only meaningful as an offset within this one turn. Wrapped in a
|
|
137
|
+
# lambda, not passed as `time.monotonic` directly -- a bare function reference binds at
|
|
138
|
+
# class-body execution (import time), before tests can monkeypatch `sse_mod.time.monotonic`.
|
|
139
|
+
t0: float = field(default_factory=lambda: time.monotonic())
|
|
135
140
|
|
|
136
141
|
|
|
137
142
|
def _handle_text(content: dict[str, Any], acc: _SseAccumulator) -> None:
|
|
@@ -160,17 +165,22 @@ def _handle_multipart(content: dict[str, Any], acc: _SseAccumulator) -> None:
|
|
|
160
165
|
def _handle_reasoning(content: dict[str, Any], acc: _SseAccumulator) -> None:
|
|
161
166
|
summary = content.get("summary", "")
|
|
162
167
|
if summary:
|
|
163
|
-
acc.reasoning_steps.append(
|
|
168
|
+
acc.reasoning_steps.append(
|
|
169
|
+
{"summary": summary, "ts": round(time.monotonic() - acc.t0, 3), "index": len(acc.reasoning_steps)}
|
|
170
|
+
)
|
|
164
171
|
|
|
165
172
|
|
|
166
173
|
def _handle_tool_call(content: dict[str, Any], acc: _SseAccumulator) -> None:
|
|
167
174
|
call_id = content.get("callId", "")
|
|
168
|
-
|
|
175
|
+
idx = len(acc.tool_call_events)
|
|
176
|
+
acc.call_id_to_event_index[call_id] = idx
|
|
169
177
|
acc.tool_call_events.append(
|
|
170
178
|
{
|
|
171
179
|
"functionName": content.get("name", ""),
|
|
172
180
|
"functionArguments": json.dumps(content.get("arguments", {})),
|
|
173
181
|
"result": None,
|
|
182
|
+
"call_ts": round(time.monotonic() - acc.t0, 3),
|
|
183
|
+
"index": idx,
|
|
174
184
|
}
|
|
175
185
|
)
|
|
176
186
|
# Stash visualization definition from create_adhoc_visualization so we can
|
|
@@ -186,6 +196,7 @@ def _handle_tool_result(content: dict[str, Any], acc: _SseAccumulator) -> None:
|
|
|
186
196
|
idx = acc.call_id_to_event_index.get(call_id)
|
|
187
197
|
if idx is not None:
|
|
188
198
|
acc.tool_call_events[idx]["result"] = content.get("result", "")
|
|
199
|
+
acc.tool_call_events[idx]["result_ts"] = round(time.monotonic() - acc.t0, 3)
|
|
189
200
|
|
|
190
201
|
|
|
191
202
|
def _build_chat_result(acc: _SseAccumulator) -> ChatResult:
|
|
@@ -195,6 +206,7 @@ def _build_chat_result(acc: _SseAccumulator) -> ChatResult:
|
|
|
195
206
|
"toolCallEvents": acc.tool_call_events,
|
|
196
207
|
"reasoningStepCount": len(acc.reasoning_steps),
|
|
197
208
|
"reasoningSteps": [step["summary"] for step in acc.reasoning_steps],
|
|
209
|
+
"reasoningStepEvents": acc.reasoning_steps,
|
|
198
210
|
}
|
|
199
211
|
if acc.visualizations:
|
|
200
212
|
payload["createdVisualizations"] = {
|
|
@@ -4,7 +4,7 @@
|
|
|
4
4
|
from gooddata_eval.core.evaluators._llm_judge import LLMJudge
|
|
5
5
|
from gooddata_eval.core.evaluators._text_utils import extract_text
|
|
6
6
|
from gooddata_eval.core.evaluators.base import ItemEvaluation
|
|
7
|
-
from gooddata_eval.core.models import ChatResult, DatasetItem
|
|
7
|
+
from gooddata_eval.core.models import ChatResult, DatasetItem, build_latency_breakdown
|
|
8
8
|
|
|
9
9
|
_EVALUATION_STEPS = [
|
|
10
10
|
"Read the INPUT (the user's question) and the EXPECTED OUTPUT (a description of what a correct answer must contain).",
|
|
@@ -30,5 +30,11 @@ class GeneralQuestionEvaluator:
|
|
|
30
30
|
return ItemEvaluation(
|
|
31
31
|
passed=passed,
|
|
32
32
|
rank_key=(int(passed),),
|
|
33
|
-
detail={
|
|
33
|
+
detail={
|
|
34
|
+
"judge_reasoning": reasoning,
|
|
35
|
+
"actual_output": actual,
|
|
36
|
+
"latency_breakdown": build_latency_breakdown(
|
|
37
|
+
chat_result.tool_call_events, chat_result.reasoning_step_events
|
|
38
|
+
),
|
|
39
|
+
},
|
|
34
40
|
)
|
|
@@ -4,7 +4,7 @@
|
|
|
4
4
|
from gooddata_eval.core.evaluators._llm_judge import LLMJudge
|
|
5
5
|
from gooddata_eval.core.evaluators._text_utils import extract_text
|
|
6
6
|
from gooddata_eval.core.evaluators.base import ItemEvaluation
|
|
7
|
-
from gooddata_eval.core.models import ChatResult, DatasetItem
|
|
7
|
+
from gooddata_eval.core.models import ChatResult, DatasetItem, build_latency_breakdown
|
|
8
8
|
|
|
9
9
|
_EVALUATION_STEPS = [
|
|
10
10
|
"Read the INPUT (the user's message) and the EXPECTED OUTPUT (a description of how the agent should refuse or redirect).",
|
|
@@ -29,7 +29,13 @@ class GuardrailEvaluator:
|
|
|
29
29
|
passed=False,
|
|
30
30
|
rank_key=(False,),
|
|
31
31
|
# no_visualization=False → quality_score=0 (correctly bad)
|
|
32
|
-
detail={
|
|
32
|
+
detail={
|
|
33
|
+
"no_visualization": False,
|
|
34
|
+
"judge_reasoning": "visualization produced — auto-fail",
|
|
35
|
+
"latency_breakdown": build_latency_breakdown(
|
|
36
|
+
chat_result.tool_call_events, chat_result.reasoning_step_events
|
|
37
|
+
),
|
|
38
|
+
},
|
|
33
39
|
)
|
|
34
40
|
|
|
35
41
|
actual = extract_text(chat_result)
|
|
@@ -48,5 +54,8 @@ class GuardrailEvaluator:
|
|
|
48
54
|
"judge_passed": passed,
|
|
49
55
|
"judge_reasoning": reasoning,
|
|
50
56
|
"actual_output": actual,
|
|
57
|
+
"latency_breakdown": build_latency_breakdown(
|
|
58
|
+
chat_result.tool_call_events, chat_result.reasoning_step_events
|
|
59
|
+
),
|
|
51
60
|
},
|
|
52
61
|
)
|
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
"""Evaluator for search_tool: agent must call the catalog search with expected parameters."""
|
|
3
3
|
|
|
4
4
|
from gooddata_eval.core.evaluators.base import ItemEvaluation
|
|
5
|
-
from gooddata_eval.core.models import ChatResult, DatasetItem
|
|
5
|
+
from gooddata_eval.core.models import ChatResult, DatasetItem, build_latency_breakdown
|
|
6
6
|
|
|
7
7
|
|
|
8
8
|
def _normalize_str_list(value: object, *, lowercase: bool = False) -> list[str]:
|
|
@@ -55,5 +55,8 @@ class SearchToolEvaluator:
|
|
|
55
55
|
"tool_correctness": tool_correctness,
|
|
56
56
|
"expected_function": expected_fn,
|
|
57
57
|
"calls_found": len(matching_events),
|
|
58
|
+
"latency_breakdown": build_latency_breakdown(
|
|
59
|
+
chat_result.tool_call_events, chat_result.reasoning_step_events
|
|
60
|
+
),
|
|
58
61
|
},
|
|
59
62
|
)
|
|
@@ -4,7 +4,13 @@
|
|
|
4
4
|
from dataclasses import dataclass
|
|
5
5
|
|
|
6
6
|
from gooddata_eval.core.evaluators.base import ItemEvaluation
|
|
7
|
-
from gooddata_eval.core.models import
|
|
7
|
+
from gooddata_eval.core.models import (
|
|
8
|
+
ChatResult,
|
|
9
|
+
CreatedVisualization,
|
|
10
|
+
DatasetItem,
|
|
11
|
+
ToolCallEvent,
|
|
12
|
+
build_latency_breakdown,
|
|
13
|
+
)
|
|
8
14
|
from gooddata_eval.core.scoring import (
|
|
9
15
|
check_filters,
|
|
10
16
|
check_viz_type,
|
|
@@ -197,5 +203,10 @@ class VisualizationEvaluator:
|
|
|
197
203
|
return ItemEvaluation(
|
|
198
204
|
passed=ev.strict_pass,
|
|
199
205
|
rank_key=(ev.strict_pass, ev.strict_checks_passed_count),
|
|
200
|
-
detail=
|
|
206
|
+
detail={
|
|
207
|
+
**evaluation_result_detail(ev),
|
|
208
|
+
"latency_breakdown": build_latency_breakdown(
|
|
209
|
+
chat_result.tool_call_events, chat_result.reasoning_step_events
|
|
210
|
+
),
|
|
211
|
+
},
|
|
201
212
|
)
|