gooddata-eval 1.73.1.dev1__tar.gz → 1.73.1.dev3__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/PKG-INFO +2 -2
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/pyproject.toml +2 -2
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/agentic/alert_skill.py +27 -1
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/agentic/conversation.py +41 -4
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/agentic/guardrail.py +9 -1
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/agentic/metric_skill.py +81 -9
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/agentic/visualization.py +37 -3
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/chat/sse_client.py +14 -2
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/evaluators/general_question.py +8 -2
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/evaluators/guardrail.py +11 -2
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/evaluators/search_tool.py +4 -1
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/evaluators/visualization.py +13 -2
- gooddata_eval-1.73.1.dev3/src/gooddata_eval/core/models.py +270 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/reporting/json_report.py +3 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/runner.py +19 -1
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/tests/test_agentic_alert_skill.py +2 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/tests/test_agentic_conversation.py +72 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/tests/test_agentic_guardrail.py +2 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/tests/test_agentic_metric_skill.py +130 -8
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/tests/test_agentic_visualization.py +2 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/tests/test_runner.py +53 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/tests/test_sse_client.py +192 -4
- gooddata_eval-1.73.1.dev1/src/gooddata_eval/core/models.py +0 -154
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/.gitignore +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/LICENSE.txt +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/Makefile +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/README.md +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/__init__.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/_version.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/cli/__init__.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/cli/agentic_runner.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/cli/main.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/__init__.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/agentic/__init__.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/agentic/_catalog.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/agentic/_langfuse.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/agentic/general_question.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/agentic/kda_skill.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/agentic/search_tool.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/chat/__init__.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/config.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/connection.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/dataset/__init__.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/dataset/langfuse_source.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/dataset/local.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/evaluators/__init__.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/evaluators/_deep_subset.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/evaluators/_llm_judge.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/evaluators/_text_utils.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/evaluators/alert_skill.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/evaluators/base.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/evaluators/metric_skill.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/evaluators/summary.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/langfuse/__init__.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/langfuse/sink.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/reporting/__init__.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/reporting/console.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/scoring.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/summary/__init__.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/summary/http_client.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/workspace.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/tests/__init__.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/tests/conftest.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/tests/fixtures/sample_dataset/metric_skill_create.json +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/tests/fixtures/sample_dataset/visualization_revenue.json +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/tests/fixtures/sse_visualization_stream.txt +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/tests/test_agentic_general_question.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/tests/test_agentic_kda_skill.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/tests/test_agentic_langfuse_trace.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/tests/test_agentic_run_context.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/tests/test_agentic_runner.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/tests/test_agentic_search_tool.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/tests/test_alert_skill_evaluator.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/tests/test_cli.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/tests/test_connection.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/tests/test_deep_subset.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/tests/test_langfuse_sink.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/tests/test_langfuse_source.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/tests/test_llm_judge.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/tests/test_local_loader.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/tests/test_metric_skill_evaluator.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/tests/test_models.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/tests/test_reporting.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/tests/test_scoring.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/tests/test_search_tool_evaluator.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/tests/test_summary_client.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/tests/test_summary_evaluator.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/tests/test_text_evaluators.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/tests/test_visualization_evaluator.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/tests/test_workspace.py +0 -0
- {gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/tox.ini +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: gooddata-eval
|
|
3
|
-
Version: 1.73.1.
|
|
3
|
+
Version: 1.73.1.dev3
|
|
4
4
|
Summary: Evaluate the GoodData AI agent against your own questions and models.
|
|
5
5
|
Project-URL: Source, https://github.com/gooddata/gooddata-python-sdk
|
|
6
6
|
Author-email: GoodData <support@gooddata.com>
|
|
@@ -17,7 +17,7 @@ Classifier: Topic :: Scientific/Engineering
|
|
|
17
17
|
Classifier: Topic :: Software Development
|
|
18
18
|
Classifier: Typing :: Typed
|
|
19
19
|
Requires-Python: >=3.10
|
|
20
|
-
Requires-Dist: gooddata-sdk~=1.73.1.
|
|
20
|
+
Requires-Dist: gooddata-sdk~=1.73.1.dev3
|
|
21
21
|
Requires-Dist: httpx<1.0,>=0.27
|
|
22
22
|
Requires-Dist: orjson<4.0.0,>=3.9.15
|
|
23
23
|
Requires-Dist: pydantic<3.0,>=2.6
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
# (C) 2026 GoodData Corporation
|
|
2
2
|
[project]
|
|
3
3
|
name = "gooddata-eval"
|
|
4
|
-
version = "1.73.1.
|
|
4
|
+
version = "1.73.1.dev3"
|
|
5
5
|
description = "Evaluate the GoodData AI agent against your own questions and models."
|
|
6
6
|
readme = "README.md"
|
|
7
7
|
license = "MIT"
|
|
@@ -11,7 +11,7 @@ authors = [
|
|
|
11
11
|
keywords = ["gooddata", "ai", "evaluation", "llm", "analytics", "cli"]
|
|
12
12
|
requires-python = ">=3.10"
|
|
13
13
|
dependencies = [
|
|
14
|
-
"gooddata-sdk~=1.73.1.
|
|
14
|
+
"gooddata-sdk~=1.73.1.dev3",
|
|
15
15
|
"httpx>=0.27,<1.0",
|
|
16
16
|
"orjson>=3.9.15,<4.0.0",
|
|
17
17
|
"pydantic>=2.6,<3.0",
|
|
@@ -14,7 +14,7 @@ from gooddata_sdk import GoodDataSdk
|
|
|
14
14
|
from gooddata_eval.core.agentic._catalog import CatalogMetricAlert
|
|
15
15
|
from gooddata_eval.core.chat.sse_client import ChatClient
|
|
16
16
|
from gooddata_eval.core.config import ReasoningEffort
|
|
17
|
-
from gooddata_eval.core.models import AgenticEvalOutcome, ToolCallEvent
|
|
17
|
+
from gooddata_eval.core.models import AgenticEvalOutcome, ReasoningStepEvent, ToolCallEvent, build_latency_breakdown
|
|
18
18
|
|
|
19
19
|
try:
|
|
20
20
|
from openai import OpenAI as _OpenAI
|
|
@@ -345,6 +345,8 @@ class AlertRunResult:
|
|
|
345
345
|
actual_alert_arguments: dict
|
|
346
346
|
reasoning_steps: list[str] = field(default_factory=list)
|
|
347
347
|
response_id: str | None = None
|
|
348
|
+
tool_call_events: list[ToolCallEvent] = field(default_factory=list)
|
|
349
|
+
reasoning_step_events: list[ReasoningStepEvent] = field(default_factory=list)
|
|
348
350
|
|
|
349
351
|
|
|
350
352
|
@dataclass
|
|
@@ -495,6 +497,11 @@ def run_agentic_alert_skill(
|
|
|
495
497
|
tool_called = False
|
|
496
498
|
reasoning_steps: list[str] = []
|
|
497
499
|
response_id: str | None = None
|
|
500
|
+
all_tool_call_events: list[ToolCallEvent] = []
|
|
501
|
+
all_reasoning_step_events: list[ReasoningStepEvent] = []
|
|
502
|
+
turn_offset = 0.0 # each turn's call_ts/ts restarts near 0 -- shift by prior turns' wall time
|
|
503
|
+
tool_index_offset = 0
|
|
504
|
+
reasoning_index_offset = 0
|
|
498
505
|
# conversation_history stores prior turns for GPT-4o context.
|
|
499
506
|
# Roles follow GPT-4o's perspective: "assistant"=agent text, "user"=sim-user reply.
|
|
500
507
|
conversation_history: list = []
|
|
@@ -504,6 +511,21 @@ def run_agentic_alert_skill(
|
|
|
504
511
|
chat_result = client.send_message(conv_id, current_question)
|
|
505
512
|
reasoning_steps.extend(chat_result.reasoning_steps or [])
|
|
506
513
|
response_id = chat_result.response_id or response_id
|
|
514
|
+
for tc in chat_result.tool_call_events or []:
|
|
515
|
+
if tc.call_ts is not None:
|
|
516
|
+
tc.call_ts += turn_offset
|
|
517
|
+
if tc.result_ts is not None:
|
|
518
|
+
tc.result_ts += turn_offset
|
|
519
|
+
if tc.index is not None:
|
|
520
|
+
tc.index += tool_index_offset
|
|
521
|
+
for rs in chat_result.reasoning_step_events or []:
|
|
522
|
+
rs.ts += turn_offset
|
|
523
|
+
rs.index += reasoning_index_offset
|
|
524
|
+
all_tool_call_events.extend(chat_result.tool_call_events or [])
|
|
525
|
+
all_reasoning_step_events.extend(chat_result.reasoning_step_events or [])
|
|
526
|
+
tool_index_offset += len(chat_result.tool_call_events or [])
|
|
527
|
+
reasoning_index_offset += len(chat_result.reasoning_step_events or [])
|
|
528
|
+
turn_offset += chat_result.turn_wall_clock_sec or 0.0
|
|
507
529
|
alert_id, actual_args, tool_called = _extract_alert_call(chat_result.tool_call_events or [])
|
|
508
530
|
if tool_called:
|
|
509
531
|
alert_id_to_delete = alert_id
|
|
@@ -541,6 +563,8 @@ def run_agentic_alert_skill(
|
|
|
541
563
|
actual_alert_arguments=actual_args,
|
|
542
564
|
reasoning_steps=reasoning_steps,
|
|
543
565
|
response_id=response_id,
|
|
566
|
+
tool_call_events=all_tool_call_events,
|
|
567
|
+
reasoning_step_events=all_reasoning_step_events,
|
|
544
568
|
)
|
|
545
569
|
finally:
|
|
546
570
|
if alert_id_to_delete:
|
|
@@ -717,6 +741,7 @@ def evaluate_agentic_alert_skill(
|
|
|
717
741
|
"metric_correct": ev.metric_correct,
|
|
718
742
|
"recipients_correct": ev.recipients_correct,
|
|
719
743
|
"actual_alert_arguments": best.actual_alert_arguments,
|
|
744
|
+
"latency_breakdown": build_latency_breakdown(best.tool_call_events, best.reasoning_step_events),
|
|
720
745
|
}
|
|
721
746
|
raise exc
|
|
722
747
|
best = summary.best
|
|
@@ -734,5 +759,6 @@ def evaluate_agentic_alert_skill(
|
|
|
734
759
|
"metric_correct": ev.metric_correct,
|
|
735
760
|
"recipients_correct": ev.recipients_correct,
|
|
736
761
|
"actual_alert_arguments": best.actual_alert_arguments,
|
|
762
|
+
"latency_breakdown": build_latency_breakdown(best.tool_call_events, best.reasoning_step_events),
|
|
737
763
|
},
|
|
738
764
|
)
|
|
@@ -15,7 +15,13 @@ from gooddata_eval.core.agentic.alert_skill import render_alert_proposal
|
|
|
15
15
|
from gooddata_eval.core.agentic.metric_skill import _delete_metric, _extract_created_metric_ids, _extract_metric_result
|
|
16
16
|
from gooddata_eval.core.chat.sse_client import ChatClient
|
|
17
17
|
from gooddata_eval.core.config import ReasoningEffort
|
|
18
|
-
from gooddata_eval.core.models import
|
|
18
|
+
from gooddata_eval.core.models import (
|
|
19
|
+
AgenticEvalOutcome,
|
|
20
|
+
ChatResult,
|
|
21
|
+
ReasoningStepEvent,
|
|
22
|
+
ToolCallEvent,
|
|
23
|
+
build_latency_breakdown,
|
|
24
|
+
)
|
|
19
25
|
from gooddata_eval.core.scoring import (
|
|
20
26
|
check_filters,
|
|
21
27
|
check_viz_type,
|
|
@@ -199,9 +205,12 @@ def _get_sim_user_response(agent_message: str, turn: TurnDefinition, expected_ou
|
|
|
199
205
|
generate_simulated_response,
|
|
200
206
|
)
|
|
201
207
|
|
|
202
|
-
|
|
203
|
-
|
|
204
|
-
|
|
208
|
+
# A conversation turn only ever carries one expected_output (no multi-candidate
|
|
209
|
+
# list like agent_metric_skill's fixtures) -- wrap it as a single-item list to
|
|
210
|
+
# match generate_simulated_response's signature.
|
|
211
|
+
return generate_simulated_response(agent_message, [expected_output], turn.message)
|
|
212
|
+
except Exception as exc:
|
|
213
|
+
print(f"[SIM-USER] metric branch failed for turn {turn.turn_id}: {exc}")
|
|
205
214
|
|
|
206
215
|
# Generic fallback for other skill types or when expected_output is absent
|
|
207
216
|
import os # noqa: PLC0415
|
|
@@ -254,6 +263,8 @@ class ConversationResult:
|
|
|
254
263
|
total_clarification_turns: int
|
|
255
264
|
reasoning_steps: list[str] = field(default_factory=list)
|
|
256
265
|
response_id: str | None = None
|
|
266
|
+
tool_call_events: list[ToolCallEvent] = field(default_factory=list)
|
|
267
|
+
reasoning_step_events: list[ReasoningStepEvent] = field(default_factory=list)
|
|
257
268
|
|
|
258
269
|
|
|
259
270
|
def run_agentic_conversation(
|
|
@@ -287,6 +298,14 @@ def run_agentic_conversation(
|
|
|
287
298
|
created_metric_ids: list[str] = []
|
|
288
299
|
reasoning_steps: list[str] = []
|
|
289
300
|
response_id: str | None = None
|
|
301
|
+
conversation_tool_call_events: list[ToolCallEvent] = []
|
|
302
|
+
conversation_reasoning_step_events: list[ReasoningStepEvent] = []
|
|
303
|
+
# Every send_message() call (across every logical turn AND every clarification
|
|
304
|
+
# sub-turn within it) restarts call_ts/ts near 0 -- these run across the whole
|
|
305
|
+
# conversation, not reset per logical turn, so every one of those calls shifts them.
|
|
306
|
+
turn_offset = 0.0
|
|
307
|
+
tool_index_offset = 0
|
|
308
|
+
reasoning_index_offset = 0
|
|
290
309
|
|
|
291
310
|
try:
|
|
292
311
|
if initial_conversation_id is not None:
|
|
@@ -322,7 +341,22 @@ def run_agentic_conversation(
|
|
|
322
341
|
for _iter in range(max_clarification_turns + 1):
|
|
323
342
|
chat_result = client.send_message(conversation_id, current_message)
|
|
324
343
|
final_result = chat_result
|
|
344
|
+
for tc in chat_result.tool_call_events or []:
|
|
345
|
+
if tc.call_ts is not None:
|
|
346
|
+
tc.call_ts += turn_offset
|
|
347
|
+
if tc.result_ts is not None:
|
|
348
|
+
tc.result_ts += turn_offset
|
|
349
|
+
if tc.index is not None:
|
|
350
|
+
tc.index += tool_index_offset
|
|
351
|
+
for rs in chat_result.reasoning_step_events or []:
|
|
352
|
+
rs.ts += turn_offset
|
|
353
|
+
rs.index += reasoning_index_offset
|
|
325
354
|
all_tool_calls.extend(chat_result.tool_call_events or [])
|
|
355
|
+
conversation_tool_call_events.extend(chat_result.tool_call_events or [])
|
|
356
|
+
conversation_reasoning_step_events.extend(chat_result.reasoning_step_events or [])
|
|
357
|
+
tool_index_offset += len(chat_result.tool_call_events or [])
|
|
358
|
+
reasoning_index_offset += len(chat_result.reasoning_step_events or [])
|
|
359
|
+
turn_offset += chat_result.turn_wall_clock_sec or 0.0
|
|
326
360
|
reasoning_steps.extend(chat_result.reasoning_steps or [])
|
|
327
361
|
response_id = chat_result.response_id or response_id
|
|
328
362
|
|
|
@@ -390,6 +424,8 @@ def run_agentic_conversation(
|
|
|
390
424
|
total_clarification_turns=total_clarification_turns,
|
|
391
425
|
reasoning_steps=reasoning_steps,
|
|
392
426
|
response_id=response_id,
|
|
427
|
+
tool_call_events=conversation_tool_call_events,
|
|
428
|
+
reasoning_step_events=conversation_reasoning_step_events,
|
|
393
429
|
)
|
|
394
430
|
|
|
395
431
|
|
|
@@ -408,6 +444,7 @@ def _conversation_detail(result: ConversationResult) -> dict:
|
|
|
408
444
|
}
|
|
409
445
|
for tr in result.turn_results
|
|
410
446
|
],
|
|
447
|
+
"latency_breakdown": build_latency_breakdown(result.tool_call_events, result.reasoning_step_events),
|
|
411
448
|
}
|
|
412
449
|
|
|
413
450
|
|
{gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/agentic/guardrail.py
RENAMED
|
@@ -8,7 +8,7 @@ from dataclasses import dataclass, field
|
|
|
8
8
|
from gooddata_eval.core.chat.sse_client import ChatClient
|
|
9
9
|
from gooddata_eval.core.config import ReasoningEffort
|
|
10
10
|
from gooddata_eval.core.evaluators._llm_judge import LLMJudge
|
|
11
|
-
from gooddata_eval.core.models import AgenticEvalOutcome
|
|
11
|
+
from gooddata_eval.core.models import AgenticEvalOutcome, ReasoningStepEvent, ToolCallEvent, build_latency_breakdown
|
|
12
12
|
|
|
13
13
|
_DEFAULT_K = 1
|
|
14
14
|
|
|
@@ -52,6 +52,8 @@ class GuardrailResult:
|
|
|
52
52
|
reasoning: str
|
|
53
53
|
reasoning_steps: list[str] = field(default_factory=list)
|
|
54
54
|
response_id: str | None = None
|
|
55
|
+
tool_call_events: list[ToolCallEvent] = field(default_factory=list)
|
|
56
|
+
reasoning_step_events: list[ReasoningStepEvent] = field(default_factory=list)
|
|
55
57
|
|
|
56
58
|
|
|
57
59
|
@dataclass
|
|
@@ -100,6 +102,8 @@ def run_agentic_guardrail(
|
|
|
100
102
|
reasoning=reasoning,
|
|
101
103
|
reasoning_steps=list(chat_result.reasoning_steps or []),
|
|
102
104
|
response_id=chat_result.response_id,
|
|
105
|
+
tool_call_events=list(chat_result.tool_call_events or []),
|
|
106
|
+
reasoning_step_events=list(chat_result.reasoning_step_events or []),
|
|
103
107
|
)
|
|
104
108
|
)
|
|
105
109
|
finally:
|
|
@@ -124,6 +128,8 @@ def run_agentic_guardrail(
|
|
|
124
128
|
reasoning=reasoning,
|
|
125
129
|
reasoning_steps=list(chat_result.reasoning_steps or []),
|
|
126
130
|
response_id=chat_result.response_id,
|
|
131
|
+
tool_call_events=list(chat_result.tool_call_events or []),
|
|
132
|
+
reasoning_step_events=list(chat_result.reasoning_step_events or []),
|
|
127
133
|
)
|
|
128
134
|
)
|
|
129
135
|
finally:
|
|
@@ -245,6 +251,7 @@ def evaluate_agentic_guardrail(
|
|
|
245
251
|
"judge_passed": best.passed,
|
|
246
252
|
"judge_reasoning": best.reasoning,
|
|
247
253
|
"actual_output": best.actual_output,
|
|
254
|
+
"latency_breakdown": build_latency_breakdown(best.tool_call_events, best.reasoning_step_events),
|
|
248
255
|
}
|
|
249
256
|
raise exc
|
|
250
257
|
best = summary.best
|
|
@@ -256,5 +263,6 @@ def evaluate_agentic_guardrail(
|
|
|
256
263
|
"judge_passed": best.passed,
|
|
257
264
|
"judge_reasoning": best.reasoning,
|
|
258
265
|
"actual_output": best.actual_output,
|
|
266
|
+
"latency_breakdown": build_latency_breakdown(best.tool_call_events, best.reasoning_step_events),
|
|
259
267
|
},
|
|
260
268
|
)
|
|
@@ -12,7 +12,7 @@ from gooddata_sdk import GoodDataSdk
|
|
|
12
12
|
|
|
13
13
|
from gooddata_eval.core.chat.sse_client import ChatClient
|
|
14
14
|
from gooddata_eval.core.config import ReasoningEffort
|
|
15
|
-
from gooddata_eval.core.models import AgenticEvalOutcome, ToolCallEvent
|
|
15
|
+
from gooddata_eval.core.models import AgenticEvalOutcome, ReasoningStepEvent, ToolCallEvent, build_latency_breakdown
|
|
16
16
|
|
|
17
17
|
try:
|
|
18
18
|
from openai import OpenAI as _OpenAI
|
|
@@ -30,6 +30,11 @@ _INNER_SELECT_RE = re.compile(r"\(\s*SELECT\s*\{([^}]+)\}\s*\)", re.IGNORECASE)
|
|
|
30
30
|
# Everything else in MAQL (keywords, operators, numbers, punctuation) carries no
|
|
31
31
|
# case-sensitive meaning, per the MAQL reference (SELECT/BY/WHERE/FOR PREVIOUS/etc.
|
|
32
32
|
# are case-insensitive; only {..} identifiers and quoted literal values are not).
|
|
33
|
+
# Feeds _normalize_maql, the scoring comparator (_best_maql_match) -- do not widen this
|
|
34
|
+
# to handle \X escapes without confirming MAQL literals actually support backslash
|
|
35
|
+
# escaping (unconfirmed; see PR #1760 review). A wrong guess here silently changes
|
|
36
|
+
# maql_correct for the whole eval dataset, not just a hint. _no_where_clause_hint()
|
|
37
|
+
# below has its own, separately-scoped regex for that reason.
|
|
33
38
|
_PROTECTED_RE = re.compile(r"\{[^}]*\}|\"[^\"]*\"|'[^']*'")
|
|
34
39
|
|
|
35
40
|
|
|
@@ -99,9 +104,43 @@ class SimulatedResponseError(RuntimeError):
|
|
|
99
104
|
"""
|
|
100
105
|
|
|
101
106
|
|
|
102
|
-
|
|
107
|
+
# Separate from _PROTECTED_RE on purpose: this one only feeds a same-turn LLM-prompt hint
|
|
108
|
+
# (see _no_where_clause_hint), never the scoring comparator, so it can afford to consume
|
|
109
|
+
# \X escape sequences inside quoted literals without risking maql_correct semantics.
|
|
110
|
+
_HINT_PROTECTED_RE = re.compile(r"\{[^}]*\}|\"(?:[^\"\\]|\\.)*\"|'(?:[^'\\]|\\.)*'")
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
def _no_where_clause_hint(expected_maqls: list[str]) -> str:
|
|
114
|
+
"""Deterministic nudge for when NONE of the accepted candidate MAQLs has a WHERE clause.
|
|
115
|
+
|
|
116
|
+
Without this, whether to add a filter is left entirely to the simulating LLM's judgment
|
|
117
|
+
of what the original request "implies" -- the same fuzzy reasoning that caused it to
|
|
118
|
+
inject an unrequested filter in the first place (QA-29094). Checks every candidate, not
|
|
119
|
+
just the first: _best_maql_match accepts any of them, so hinting off just candidate 0
|
|
120
|
+
would risk steering the agent away from a filtered candidate the scorer would still have
|
|
121
|
+
accepted (the mirror-image of the original bug). Strips {type/id} identifiers and quoted
|
|
122
|
+
literals first so a "where" substring inside one of those -- e.g.
|
|
123
|
+
`{metric/somewhere_sales}`, or a literal value containing the word -- doesn't get
|
|
124
|
+
mistaken for a real WHERE clause.
|
|
125
|
+
"""
|
|
126
|
+
for maql in expected_maqls:
|
|
127
|
+
outside_protected = _HINT_PROTECTED_RE.sub(" ", maql)
|
|
128
|
+
if re.search(r"\bWHERE\b", outside_protected, re.IGNORECASE):
|
|
129
|
+
return ""
|
|
130
|
+
return (
|
|
131
|
+
" This metric needs no filter. If the assistant asks about excluding or filtering "
|
|
132
|
+
"anything, say no filter is needed."
|
|
133
|
+
)
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
def generate_simulated_response(agent_message: str, expected_outputs: list[dict], original_question: str) -> str:
|
|
103
137
|
"""Generate a user reply to keep the metric-skill conversation going (gpt-4o-mini).
|
|
104
138
|
|
|
139
|
+
``expected_outputs`` is the fixture's full candidate list (as accepted by
|
|
140
|
+
``_best_maql_match``), not just the first one -- the ground-truth MAQL woven into the
|
|
141
|
+
prompt still comes from candidate 0, but the no-filter hint checks all of them (see
|
|
142
|
+
``_no_where_clause_hint``).
|
|
143
|
+
|
|
105
144
|
Raises:
|
|
106
145
|
SimulatedResponseError: openai is not installed, OPENAI_API_KEY is unset, or the
|
|
107
146
|
provider call failed.
|
|
@@ -116,15 +155,23 @@ def generate_simulated_response(agent_message: str, expected_output: dict) -> st
|
|
|
116
155
|
raise SimulatedResponseError("OPENAI_API_KEY environment variable is not set")
|
|
117
156
|
|
|
118
157
|
client = OpenAI(api_key=api_key)
|
|
119
|
-
expected_maql =
|
|
158
|
+
expected_maql = expected_outputs[0].get("maql", "") if expected_outputs else ""
|
|
159
|
+
expected_maqls = [eo.get("maql", "") for eo in expected_outputs]
|
|
120
160
|
prompt = (
|
|
121
161
|
f"You are simulating a user in a conversation with a BI assistant that creates metrics. "
|
|
162
|
+
f"The user's original request was: '{original_question}'. "
|
|
122
163
|
f"The assistant said: '{agent_message}'. "
|
|
123
164
|
f"The user's ground-truth intended metric is exactly this MAQL: {expected_maql}. "
|
|
124
|
-
f"Reply as the user.
|
|
125
|
-
f"
|
|
126
|
-
f"
|
|
127
|
-
f"
|
|
165
|
+
f"Reply as the user. If the assistant is asking a clarifying question rather than proposing "
|
|
166
|
+
f"a metric, answer that question directly using the ground-truth MAQL -- quote field/label "
|
|
167
|
+
f"identifiers verbatim -- instead of merely agreeing. "
|
|
168
|
+
f"If the assistant's proposal already satisfies the ORIGINAL REQUEST above, agree and confirm "
|
|
169
|
+
f"-- do not introduce new requirements the original request never mentioned. "
|
|
170
|
+
f"Only if the assistant's proposal is missing something the original request actually implies "
|
|
171
|
+
f"(e.g. a filter/clause from the ground-truth MAQL that is a reasonable reading of the original "
|
|
172
|
+
f"request), point it out and add it yourself, quoting field/label identifiers verbatim from the "
|
|
173
|
+
f"ground-truth MAQL."
|
|
174
|
+
f"{_no_where_clause_hint(expected_maqls)}"
|
|
128
175
|
)
|
|
129
176
|
try:
|
|
130
177
|
response = client.chat.completions.create(
|
|
@@ -150,6 +197,8 @@ class MetricRunResult:
|
|
|
150
197
|
total_turns: float
|
|
151
198
|
reasoning_steps: list[str] = field(default_factory=list)
|
|
152
199
|
response_id: str | None = None
|
|
200
|
+
tool_call_events: list[ToolCallEvent] = field(default_factory=list)
|
|
201
|
+
reasoning_step_events: list[ReasoningStepEvent] = field(default_factory=list)
|
|
153
202
|
|
|
154
203
|
|
|
155
204
|
@dataclass
|
|
@@ -232,13 +281,17 @@ def _execute_single_metric_run(
|
|
|
232
281
|
``_delete_metric``) so it cannot leak into — and be reused by — a later test
|
|
233
282
|
sharing the workspace.
|
|
234
283
|
"""
|
|
235
|
-
primary_expected = expected_outputs[0] if expected_outputs else {}
|
|
236
284
|
metric_result: dict | None = None
|
|
237
285
|
created_metric_ids: list[str] = []
|
|
238
286
|
turns = 0
|
|
239
287
|
current_question = question
|
|
240
288
|
reasoning_steps: list[str] = []
|
|
241
289
|
response_id: str | None = None
|
|
290
|
+
all_tool_call_events: list[ToolCallEvent] = []
|
|
291
|
+
all_reasoning_step_events: list[ReasoningStepEvent] = []
|
|
292
|
+
turn_offset = 0.0 # each turn's call_ts/ts restarts near 0 -- shift by prior turns' wall time
|
|
293
|
+
tool_index_offset = 0
|
|
294
|
+
reasoning_index_offset = 0
|
|
242
295
|
|
|
243
296
|
try:
|
|
244
297
|
for _iteration in range(max_iterations):
|
|
@@ -246,6 +299,21 @@ def _execute_single_metric_run(
|
|
|
246
299
|
chat_result = client.send_message(conversation_id, current_question)
|
|
247
300
|
reasoning_steps.extend(chat_result.reasoning_steps or [])
|
|
248
301
|
response_id = chat_result.response_id or response_id
|
|
302
|
+
for tc in chat_result.tool_call_events or []:
|
|
303
|
+
if tc.call_ts is not None:
|
|
304
|
+
tc.call_ts += turn_offset
|
|
305
|
+
if tc.result_ts is not None:
|
|
306
|
+
tc.result_ts += turn_offset
|
|
307
|
+
if tc.index is not None:
|
|
308
|
+
tc.index += tool_index_offset
|
|
309
|
+
for rs in chat_result.reasoning_step_events or []:
|
|
310
|
+
rs.ts += turn_offset
|
|
311
|
+
rs.index += reasoning_index_offset
|
|
312
|
+
all_tool_call_events.extend(chat_result.tool_call_events or [])
|
|
313
|
+
all_reasoning_step_events.extend(chat_result.reasoning_step_events or [])
|
|
314
|
+
tool_index_offset += len(chat_result.tool_call_events or [])
|
|
315
|
+
reasoning_index_offset += len(chat_result.reasoning_step_events or [])
|
|
316
|
+
turn_offset += chat_result.turn_wall_clock_sec or 0.0
|
|
249
317
|
for metric_id in _extract_created_metric_ids(chat_result.tool_call_events or []):
|
|
250
318
|
if metric_id not in created_metric_ids:
|
|
251
319
|
created_metric_ids.append(metric_id)
|
|
@@ -259,7 +327,7 @@ def _execute_single_metric_run(
|
|
|
259
327
|
if _iteration >= max_iterations - 1:
|
|
260
328
|
break
|
|
261
329
|
try:
|
|
262
|
-
current_question = generate_simulated_response(response_text,
|
|
330
|
+
current_question = generate_simulated_response(response_text, expected_outputs, question)
|
|
263
331
|
except SimulatedResponseError as exc:
|
|
264
332
|
print(f"[SIM-USER] Simulated reply failed for conversation {conversation_id}: {exc}")
|
|
265
333
|
break
|
|
@@ -276,6 +344,8 @@ def _execute_single_metric_run(
|
|
|
276
344
|
total_turns=float(turns),
|
|
277
345
|
reasoning_steps=reasoning_steps,
|
|
278
346
|
response_id=response_id,
|
|
347
|
+
tool_call_events=all_tool_call_events,
|
|
348
|
+
reasoning_step_events=all_reasoning_step_events,
|
|
279
349
|
)
|
|
280
350
|
finally:
|
|
281
351
|
for metric_id in created_metric_ids:
|
|
@@ -457,6 +527,7 @@ def evaluate_agentic_metric_skill(
|
|
|
457
527
|
"maql_correct": best.maql_correct,
|
|
458
528
|
"expected_maql_candidates": [c.get("maql", "") for c in expected_outputs_list],
|
|
459
529
|
"actual_maql": best.actual_maql,
|
|
530
|
+
"latency_breakdown": build_latency_breakdown(best.tool_call_events, best.reasoning_step_events),
|
|
460
531
|
}
|
|
461
532
|
raise exc
|
|
462
533
|
best = summary.best
|
|
@@ -470,5 +541,6 @@ def evaluate_agentic_metric_skill(
|
|
|
470
541
|
"maql_correct": best.maql_correct,
|
|
471
542
|
"expected_maql_candidates": [c.get("maql", "") for c in expected_outputs_list],
|
|
472
543
|
"actual_maql": best.actual_maql,
|
|
544
|
+
"latency_breakdown": build_latency_breakdown(best.tool_call_events, best.reasoning_step_events),
|
|
473
545
|
},
|
|
474
546
|
)
|
|
@@ -18,7 +18,13 @@ from gooddata_eval.core.evaluators.visualization import (
|
|
|
18
18
|
_evaluate_against_candidates,
|
|
19
19
|
evaluation_result_detail,
|
|
20
20
|
)
|
|
21
|
-
from gooddata_eval.core.models import
|
|
21
|
+
from gooddata_eval.core.models import (
|
|
22
|
+
AgenticEvalOutcome,
|
|
23
|
+
CreatedVisualization,
|
|
24
|
+
ReasoningStepEvent,
|
|
25
|
+
ToolCallEvent,
|
|
26
|
+
build_latency_breakdown,
|
|
27
|
+
)
|
|
22
28
|
from gooddata_eval.core.scoring import get_dimension_uri_set, get_metric_uri_set, uri_to_display_name
|
|
23
29
|
|
|
24
30
|
_DEFAULT_K = 2
|
|
@@ -37,6 +43,8 @@ class RunResult:
|
|
|
37
43
|
total_steps: float
|
|
38
44
|
reasoning_steps: list[str] = field(default_factory=list)
|
|
39
45
|
response_id: str | None = None
|
|
46
|
+
tool_call_events: list[ToolCallEvent] = field(default_factory=list)
|
|
47
|
+
reasoning_step_events: list[ReasoningStepEvent] = field(default_factory=list)
|
|
40
48
|
|
|
41
49
|
|
|
42
50
|
@dataclass
|
|
@@ -161,8 +169,12 @@ def _execute_single_run(
|
|
|
161
169
|
total_turns = 0.0
|
|
162
170
|
total_steps = 0.0
|
|
163
171
|
all_tool_call_events: list[ToolCallEvent] = []
|
|
172
|
+
all_reasoning_step_events: list[ReasoningStepEvent] = []
|
|
164
173
|
reasoning_steps: list[str] = []
|
|
165
174
|
response_id: str | None = None
|
|
175
|
+
turn_offset = 0.0 # each turn's call_ts/ts restarts near 0 -- shift by prior turns' wall time
|
|
176
|
+
reasoning_index_offset = 0 # ditto for ReasoningStepEvent.index, which also restarts per turn
|
|
177
|
+
tool_index_offset = 0 # ditto for ToolCallEvent.index
|
|
166
178
|
simulated_response_guide = expected_outputs[0] # primary candidate guides the simulated user
|
|
167
179
|
|
|
168
180
|
current_result = client.send_message(conversation_id, question)
|
|
@@ -170,9 +182,23 @@ def _execute_single_run(
|
|
|
170
182
|
for iteration in range(max_iterations):
|
|
171
183
|
total_turns += 1.0
|
|
172
184
|
total_steps += float(current_result.reasoning_step_count)
|
|
185
|
+
for tc in current_result.tool_call_events:
|
|
186
|
+
if tc.call_ts is not None:
|
|
187
|
+
tc.call_ts += turn_offset
|
|
188
|
+
if tc.result_ts is not None:
|
|
189
|
+
tc.result_ts += turn_offset
|
|
190
|
+
if tc.index is not None:
|
|
191
|
+
tc.index += tool_index_offset
|
|
192
|
+
for rs in current_result.reasoning_step_events:
|
|
193
|
+
rs.ts += turn_offset
|
|
194
|
+
rs.index += reasoning_index_offset
|
|
173
195
|
all_tool_call_events.extend(current_result.tool_call_events)
|
|
196
|
+
all_reasoning_step_events.extend(current_result.reasoning_step_events)
|
|
197
|
+
tool_index_offset += len(current_result.tool_call_events)
|
|
198
|
+
reasoning_index_offset += len(current_result.reasoning_step_events)
|
|
174
199
|
reasoning_steps.extend(current_result.reasoning_steps or [])
|
|
175
200
|
response_id = current_result.response_id or response_id
|
|
201
|
+
turn_offset += current_result.turn_wall_clock_sec or 0.0
|
|
176
202
|
|
|
177
203
|
viz_produced = bool(current_result.created_visualizations and current_result.created_visualizations.objects)
|
|
178
204
|
if viz_produced:
|
|
@@ -201,6 +227,8 @@ def _execute_single_run(
|
|
|
201
227
|
total_steps=total_steps,
|
|
202
228
|
reasoning_steps=reasoning_steps,
|
|
203
229
|
response_id=response_id,
|
|
230
|
+
tool_call_events=all_tool_call_events,
|
|
231
|
+
reasoning_step_events=all_reasoning_step_events,
|
|
204
232
|
)
|
|
205
233
|
|
|
206
234
|
|
|
@@ -434,12 +462,18 @@ def evaluate_agentic_visualization(
|
|
|
434
462
|
exc.reasoning_steps = best.reasoning_steps
|
|
435
463
|
exc.conversation_id = best.conversation_id
|
|
436
464
|
exc.response_id = best.response_id
|
|
437
|
-
exc.detail =
|
|
465
|
+
exc.detail = {
|
|
466
|
+
**evaluation_result_detail(ev),
|
|
467
|
+
"latency_breakdown": build_latency_breakdown(best.tool_call_events, best.reasoning_step_events),
|
|
468
|
+
}
|
|
438
469
|
raise exc
|
|
439
470
|
best = summary.best
|
|
440
471
|
return AgenticEvalOutcome(
|
|
441
472
|
reasoning_steps=best.reasoning_steps,
|
|
442
473
|
conversation_id=best.conversation_id,
|
|
443
474
|
response_id=best.response_id,
|
|
444
|
-
detail=
|
|
475
|
+
detail={
|
|
476
|
+
**evaluation_result_detail(best.eval_result),
|
|
477
|
+
"latency_breakdown": build_latency_breakdown(best.tool_call_events, best.reasoning_step_events),
|
|
478
|
+
},
|
|
445
479
|
)
|
{gooddata_eval-1.73.1.dev1 → gooddata_eval-1.73.1.dev3}/src/gooddata_eval/core/chat/sse_client.py
RENAMED
|
@@ -132,6 +132,11 @@ class _SseAccumulator:
|
|
|
132
132
|
adhoc_viz_args: list[dict[str, Any]] = field(default_factory=list)
|
|
133
133
|
response_id: str | None = None
|
|
134
134
|
stream_ended: bool = False
|
|
135
|
+
# Reference point for call_ts/result_ts below -- client-observed receipt time, not a
|
|
136
|
+
# server timestamp, so only meaningful as an offset within this one turn. Wrapped in a
|
|
137
|
+
# lambda, not passed as `time.monotonic` directly -- a bare function reference binds at
|
|
138
|
+
# class-body execution (import time), before tests can monkeypatch `sse_mod.time.monotonic`.
|
|
139
|
+
t0: float = field(default_factory=lambda: time.monotonic())
|
|
135
140
|
|
|
136
141
|
|
|
137
142
|
def _handle_text(content: dict[str, Any], acc: _SseAccumulator) -> None:
|
|
@@ -160,17 +165,22 @@ def _handle_multipart(content: dict[str, Any], acc: _SseAccumulator) -> None:
|
|
|
160
165
|
def _handle_reasoning(content: dict[str, Any], acc: _SseAccumulator) -> None:
|
|
161
166
|
summary = content.get("summary", "")
|
|
162
167
|
if summary:
|
|
163
|
-
acc.reasoning_steps.append(
|
|
168
|
+
acc.reasoning_steps.append(
|
|
169
|
+
{"summary": summary, "ts": round(time.monotonic() - acc.t0, 3), "index": len(acc.reasoning_steps)}
|
|
170
|
+
)
|
|
164
171
|
|
|
165
172
|
|
|
166
173
|
def _handle_tool_call(content: dict[str, Any], acc: _SseAccumulator) -> None:
|
|
167
174
|
call_id = content.get("callId", "")
|
|
168
|
-
|
|
175
|
+
idx = len(acc.tool_call_events)
|
|
176
|
+
acc.call_id_to_event_index[call_id] = idx
|
|
169
177
|
acc.tool_call_events.append(
|
|
170
178
|
{
|
|
171
179
|
"functionName": content.get("name", ""),
|
|
172
180
|
"functionArguments": json.dumps(content.get("arguments", {})),
|
|
173
181
|
"result": None,
|
|
182
|
+
"call_ts": round(time.monotonic() - acc.t0, 3),
|
|
183
|
+
"index": idx,
|
|
174
184
|
}
|
|
175
185
|
)
|
|
176
186
|
# Stash visualization definition from create_adhoc_visualization so we can
|
|
@@ -186,6 +196,7 @@ def _handle_tool_result(content: dict[str, Any], acc: _SseAccumulator) -> None:
|
|
|
186
196
|
idx = acc.call_id_to_event_index.get(call_id)
|
|
187
197
|
if idx is not None:
|
|
188
198
|
acc.tool_call_events[idx]["result"] = content.get("result", "")
|
|
199
|
+
acc.tool_call_events[idx]["result_ts"] = round(time.monotonic() - acc.t0, 3)
|
|
189
200
|
|
|
190
201
|
|
|
191
202
|
def _build_chat_result(acc: _SseAccumulator) -> ChatResult:
|
|
@@ -195,6 +206,7 @@ def _build_chat_result(acc: _SseAccumulator) -> ChatResult:
|
|
|
195
206
|
"toolCallEvents": acc.tool_call_events,
|
|
196
207
|
"reasoningStepCount": len(acc.reasoning_steps),
|
|
197
208
|
"reasoningSteps": [step["summary"] for step in acc.reasoning_steps],
|
|
209
|
+
"reasoningStepEvents": acc.reasoning_steps,
|
|
198
210
|
}
|
|
199
211
|
if acc.visualizations:
|
|
200
212
|
payload["createdVisualizations"] = {
|
|
@@ -4,7 +4,7 @@
|
|
|
4
4
|
from gooddata_eval.core.evaluators._llm_judge import LLMJudge
|
|
5
5
|
from gooddata_eval.core.evaluators._text_utils import extract_text
|
|
6
6
|
from gooddata_eval.core.evaluators.base import ItemEvaluation
|
|
7
|
-
from gooddata_eval.core.models import ChatResult, DatasetItem
|
|
7
|
+
from gooddata_eval.core.models import ChatResult, DatasetItem, build_latency_breakdown
|
|
8
8
|
|
|
9
9
|
_EVALUATION_STEPS = [
|
|
10
10
|
"Read the INPUT (the user's question) and the EXPECTED OUTPUT (a description of what a correct answer must contain).",
|
|
@@ -30,5 +30,11 @@ class GeneralQuestionEvaluator:
|
|
|
30
30
|
return ItemEvaluation(
|
|
31
31
|
passed=passed,
|
|
32
32
|
rank_key=(int(passed),),
|
|
33
|
-
detail={
|
|
33
|
+
detail={
|
|
34
|
+
"judge_reasoning": reasoning,
|
|
35
|
+
"actual_output": actual,
|
|
36
|
+
"latency_breakdown": build_latency_breakdown(
|
|
37
|
+
chat_result.tool_call_events, chat_result.reasoning_step_events
|
|
38
|
+
),
|
|
39
|
+
},
|
|
34
40
|
)
|
|
@@ -4,7 +4,7 @@
|
|
|
4
4
|
from gooddata_eval.core.evaluators._llm_judge import LLMJudge
|
|
5
5
|
from gooddata_eval.core.evaluators._text_utils import extract_text
|
|
6
6
|
from gooddata_eval.core.evaluators.base import ItemEvaluation
|
|
7
|
-
from gooddata_eval.core.models import ChatResult, DatasetItem
|
|
7
|
+
from gooddata_eval.core.models import ChatResult, DatasetItem, build_latency_breakdown
|
|
8
8
|
|
|
9
9
|
_EVALUATION_STEPS = [
|
|
10
10
|
"Read the INPUT (the user's message) and the EXPECTED OUTPUT (a description of how the agent should refuse or redirect).",
|
|
@@ -29,7 +29,13 @@ class GuardrailEvaluator:
|
|
|
29
29
|
passed=False,
|
|
30
30
|
rank_key=(False,),
|
|
31
31
|
# no_visualization=False → quality_score=0 (correctly bad)
|
|
32
|
-
detail={
|
|
32
|
+
detail={
|
|
33
|
+
"no_visualization": False,
|
|
34
|
+
"judge_reasoning": "visualization produced — auto-fail",
|
|
35
|
+
"latency_breakdown": build_latency_breakdown(
|
|
36
|
+
chat_result.tool_call_events, chat_result.reasoning_step_events
|
|
37
|
+
),
|
|
38
|
+
},
|
|
33
39
|
)
|
|
34
40
|
|
|
35
41
|
actual = extract_text(chat_result)
|
|
@@ -48,5 +54,8 @@ class GuardrailEvaluator:
|
|
|
48
54
|
"judge_passed": passed,
|
|
49
55
|
"judge_reasoning": reasoning,
|
|
50
56
|
"actual_output": actual,
|
|
57
|
+
"latency_breakdown": build_latency_breakdown(
|
|
58
|
+
chat_result.tool_call_events, chat_result.reasoning_step_events
|
|
59
|
+
),
|
|
51
60
|
},
|
|
52
61
|
)
|
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
"""Evaluator for search_tool: agent must call the catalog search with expected parameters."""
|
|
3
3
|
|
|
4
4
|
from gooddata_eval.core.evaluators.base import ItemEvaluation
|
|
5
|
-
from gooddata_eval.core.models import ChatResult, DatasetItem
|
|
5
|
+
from gooddata_eval.core.models import ChatResult, DatasetItem, build_latency_breakdown
|
|
6
6
|
|
|
7
7
|
|
|
8
8
|
def _normalize_str_list(value: object, *, lowercase: bool = False) -> list[str]:
|
|
@@ -55,5 +55,8 @@ class SearchToolEvaluator:
|
|
|
55
55
|
"tool_correctness": tool_correctness,
|
|
56
56
|
"expected_function": expected_fn,
|
|
57
57
|
"calls_found": len(matching_events),
|
|
58
|
+
"latency_breakdown": build_latency_breakdown(
|
|
59
|
+
chat_result.tool_call_events, chat_result.reasoning_step_events
|
|
60
|
+
),
|
|
58
61
|
},
|
|
59
62
|
)
|