gooddata-eval 1.74.1.dev3__tar.gz → 1.74.1.dev5__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/PKG-INFO +2 -2
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/pyproject.toml +2 -2
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/agentic/alert_skill.py +10 -0
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/agentic/conversation.py +20 -0
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/agentic/kda_skill.py +11 -0
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/agentic/metric_skill.py +8 -2
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/agentic/visualization.py +6 -6
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/tests/test_agentic_alert_skill.py +64 -0
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/tests/test_agentic_conversation.py +104 -0
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/tests/test_agentic_kda_skill.py +85 -0
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/tests/test_agentic_metric_skill.py +88 -1
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/tests/test_agentic_visualization.py +4 -4
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/.gitignore +0 -0
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/AGENTS.md +0 -0
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/CLAUDE.md +0 -0
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/LICENSE.txt +0 -0
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/Makefile +0 -0
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/README.md +0 -0
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/__init__.py +0 -0
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/_version.py +0 -0
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/cli/__init__.py +0 -0
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/cli/agentic_runner.py +0 -0
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/cli/main.py +0 -0
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/__init__.py +0 -0
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/_output.py +0 -0
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/agentic/__init__.py +0 -0
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/agentic/_catalog.py +0 -0
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/agentic/_gate.py +0 -0
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/agentic/_langfuse.py +0 -0
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/agentic/_trace_linker.py +0 -0
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/agentic/general_question.py +0 -0
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/agentic/guardrail.py +0 -0
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/agentic/search_tool.py +0 -0
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/chat/__init__.py +0 -0
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/chat/render.py +0 -0
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/chat/sse_client.py +0 -0
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/config.py +0 -0
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/connection.py +0 -0
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/dataset/__init__.py +0 -0
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/dataset/langfuse_source.py +0 -0
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/dataset/local.py +0 -0
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/evaluators/__init__.py +0 -0
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/evaluators/_deep_subset.py +0 -0
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/evaluators/_llm_judge.py +0 -0
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/evaluators/_maql.py +0 -0
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/evaluators/_text_utils.py +0 -0
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/evaluators/alert_skill.py +0 -0
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/evaluators/base.py +0 -0
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/evaluators/general_question.py +0 -0
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/evaluators/guardrail.py +0 -0
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/evaluators/metric_skill.py +0 -0
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/evaluators/search_tool.py +0 -0
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/evaluators/summary.py +0 -0
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/evaluators/visualization.py +0 -0
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/langfuse/__init__.py +0 -0
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/langfuse/_env.py +0 -0
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/langfuse/client.py +0 -0
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/langfuse/experiment.py +0 -0
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/langfuse/observations.py +0 -0
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/langfuse/otlp.py +0 -0
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/langfuse/sink.py +0 -0
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/models.py +0 -0
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/reporting/__init__.py +0 -0
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/reporting/console.py +0 -0
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/reporting/json_report.py +0 -0
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/runner.py +0 -0
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/scoring.py +0 -0
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/summary/__init__.py +0 -0
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/summary/http_client.py +0 -0
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/timing.py +0 -0
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/workspace.py +0 -0
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/tests/__init__.py +0 -0
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/tests/_fake_langfuse.py +0 -0
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/tests/conftest.py +0 -0
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/tests/fixtures/sample_dataset/metric_skill_create.json +0 -0
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/tests/fixtures/sample_dataset/visualization_revenue.json +0 -0
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/tests/fixtures/sse_visualization_stream.txt +0 -0
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/tests/test_agentic_gate.py +0 -0
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/tests/test_agentic_general_question.py +0 -0
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/tests/test_agentic_guardrail.py +0 -0
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/tests/test_agentic_langfuse_trace.py +0 -0
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/tests/test_agentic_observe_experiment.py +0 -0
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/tests/test_agentic_run_context.py +0 -0
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/tests/test_agentic_runner.py +0 -0
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/tests/test_agentic_search_tool.py +0 -0
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/tests/test_alert_skill_evaluator.py +0 -0
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/tests/test_chat_render.py +0 -0
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/tests/test_cli.py +0 -0
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/tests/test_connection.py +0 -0
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/tests/test_deep_subset.py +0 -0
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/tests/test_fake_langfuse.py +0 -0
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/tests/test_langfuse_client.py +0 -0
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/tests/test_langfuse_e2e_fake_server.py +0 -0
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/tests/test_langfuse_env.py +0 -0
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/tests/test_langfuse_experiment.py +0 -0
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/tests/test_langfuse_observations.py +0 -0
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/tests/test_langfuse_otlp.py +0 -0
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/tests/test_langfuse_sink.py +0 -0
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/tests/test_langfuse_source.py +0 -0
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/tests/test_llm_judge.py +0 -0
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/tests/test_local_loader.py +0 -0
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/tests/test_maql_normalize.py +0 -0
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/tests/test_metric_skill_evaluator.py +0 -0
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/tests/test_models.py +0 -0
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/tests/test_reporting.py +0 -0
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/tests/test_runner.py +0 -0
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/tests/test_scoring.py +0 -0
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/tests/test_search_tool_evaluator.py +0 -0
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/tests/test_sse_client.py +0 -0
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/tests/test_summary_client.py +0 -0
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/tests/test_summary_evaluator.py +0 -0
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/tests/test_text_evaluators.py +0 -0
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/tests/test_timing.py +0 -0
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/tests/test_trace_linker.py +0 -0
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/tests/test_visualization_evaluator.py +0 -0
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/tests/test_workspace.py +0 -0
- {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/tox.ini +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: gooddata-eval
|
|
3
|
-
Version: 1.74.1.
|
|
3
|
+
Version: 1.74.1.dev5
|
|
4
4
|
Summary: Evaluate the GoodData AI agent against your own questions and models.
|
|
5
5
|
Project-URL: Source, https://github.com/gooddata/gooddata-python-sdk
|
|
6
6
|
Author-email: GoodData <support@gooddata.com>
|
|
@@ -17,7 +17,7 @@ Classifier: Topic :: Scientific/Engineering
|
|
|
17
17
|
Classifier: Topic :: Software Development
|
|
18
18
|
Classifier: Typing :: Typed
|
|
19
19
|
Requires-Python: >=3.10
|
|
20
|
-
Requires-Dist: gooddata-sdk~=1.74.1.
|
|
20
|
+
Requires-Dist: gooddata-sdk~=1.74.1.dev5
|
|
21
21
|
Requires-Dist: httpx<1.0,>=0.27
|
|
22
22
|
Requires-Dist: orjson<4.0.0,>=3.9.15
|
|
23
23
|
Requires-Dist: pydantic<3.0,>=2.6
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
# (C) 2026 GoodData Corporation
|
|
2
2
|
[project]
|
|
3
3
|
name = "gooddata-eval"
|
|
4
|
-
version = "1.74.1.
|
|
4
|
+
version = "1.74.1.dev5"
|
|
5
5
|
description = "Evaluate the GoodData AI agent against your own questions and models."
|
|
6
6
|
readme = "README.md"
|
|
7
7
|
license = "MIT"
|
|
@@ -11,7 +11,7 @@ authors = [
|
|
|
11
11
|
keywords = ["gooddata", "ai", "evaluation", "llm", "analytics", "cli"]
|
|
12
12
|
requires-python = ">=3.10"
|
|
13
13
|
dependencies = [
|
|
14
|
-
"gooddata-sdk~=1.74.1.
|
|
14
|
+
"gooddata-sdk~=1.74.1.dev5",
|
|
15
15
|
"httpx>=0.27,<1.0",
|
|
16
16
|
"orjson>=3.9.15,<4.0.0",
|
|
17
17
|
"pydantic>=2.6,<3.0",
|
|
@@ -477,6 +477,8 @@ class AlertRunResult:
|
|
|
477
477
|
alert_id: str | None
|
|
478
478
|
eval: AlertEvaluation
|
|
479
479
|
actual_alert_arguments: dict
|
|
480
|
+
total_turns: int = 0
|
|
481
|
+
total_steps: int = 0
|
|
480
482
|
reasoning_steps: list[str] = field(default_factory=list)
|
|
481
483
|
response_id: str | None = None
|
|
482
484
|
tool_call_events: list[ToolCallEvent] = field(default_factory=list)
|
|
@@ -676,9 +678,13 @@ def run_agentic_alert_skill(
|
|
|
676
678
|
# Roles follow GPT-4o's perspective: "assistant"=agent text, "user"=sim-user reply.
|
|
677
679
|
conversation_history: list = []
|
|
678
680
|
current_question = question
|
|
681
|
+
turns = 0
|
|
682
|
+
steps = 0
|
|
679
683
|
|
|
680
684
|
for _iteration in range(max_iterations):
|
|
681
685
|
chat_result = client.send_message(conv_id, current_question)
|
|
686
|
+
turns += 1
|
|
687
|
+
steps += chat_result.reasoning_step_count
|
|
682
688
|
reasoning_steps.extend(chat_result.reasoning_steps or [])
|
|
683
689
|
response_id = chat_result.response_id or response_id
|
|
684
690
|
turn_offset, tool_index_offset, reasoning_index_offset = shift_and_index_events(
|
|
@@ -728,6 +734,8 @@ def run_agentic_alert_skill(
|
|
|
728
734
|
alert_id=alert_id,
|
|
729
735
|
eval=ev,
|
|
730
736
|
actual_alert_arguments=actual_args,
|
|
737
|
+
total_turns=turns,
|
|
738
|
+
total_steps=steps,
|
|
731
739
|
reasoning_steps=reasoning_steps,
|
|
732
740
|
response_id=response_id,
|
|
733
741
|
tool_call_events=all_tool_call_events,
|
|
@@ -851,6 +859,8 @@ def evaluate_agentic_alert_skill(
|
|
|
851
859
|
with ctx.observe(pt, run_idx, conversation_id=run.conversation_id, output=strict_checks) as tid:
|
|
852
860
|
for score_name, value in strict_checks.items():
|
|
853
861
|
ctx.score(tid, name=score_name, value=float(value), data_type="BOOLEAN")
|
|
862
|
+
ctx.score(tid, name="turns", value=run.total_turns, data_type="NUMERIC")
|
|
863
|
+
ctx.score(tid, name="steps", value=run.total_steps, data_type="NUMERIC")
|
|
854
864
|
log_gate_scores(ctx, tid, gate=gate, pass_at_k=summary.pass_at_k, pass_power_k=summary.pass_power_k)
|
|
855
865
|
ctx.quality(
|
|
856
866
|
tid,
|
|
@@ -336,6 +336,7 @@ class ConversationResult:
|
|
|
336
336
|
full_skill_coverage: bool
|
|
337
337
|
conversation_success: bool
|
|
338
338
|
total_clarification_turns: int
|
|
339
|
+
total_steps: int = 0
|
|
339
340
|
reasoning_steps: list[str] = field(default_factory=list)
|
|
340
341
|
response_id: str | None = None
|
|
341
342
|
tool_call_events: list[ToolCallEvent] = field(default_factory=list)
|
|
@@ -365,6 +366,7 @@ def run_agentic_conversation(
|
|
|
365
366
|
turn_results: list[TurnResult] = []
|
|
366
367
|
turn_outputs: dict[str, dict] = {}
|
|
367
368
|
total_clarification_turns = 0
|
|
369
|
+
total_steps = 0
|
|
368
370
|
conversation_id: str = ""
|
|
369
371
|
owns_conversation = False
|
|
370
372
|
# Metrics created during this conversation, deleted after it completes so they do
|
|
@@ -434,6 +436,7 @@ def run_agentic_conversation(
|
|
|
434
436
|
for _iter in range(max_clarification_turns + 1):
|
|
435
437
|
chat_result = client.send_message(conversation_id, current_message)
|
|
436
438
|
final_result = chat_result
|
|
439
|
+
total_steps += chat_result.reasoning_step_count
|
|
437
440
|
turn_offset, tool_index_offset, reasoning_index_offset = shift_and_index_events(
|
|
438
441
|
chat_result,
|
|
439
442
|
turn_offset=turn_offset,
|
|
@@ -522,6 +525,7 @@ def run_agentic_conversation(
|
|
|
522
525
|
full_skill_coverage=full_skill_coverage,
|
|
523
526
|
conversation_success=conversation_success,
|
|
524
527
|
total_clarification_turns=total_clarification_turns,
|
|
528
|
+
total_steps=total_steps,
|
|
525
529
|
reasoning_steps=reasoning_steps,
|
|
526
530
|
response_id=response_id,
|
|
527
531
|
tool_call_events=conversation_tool_call_events,
|
|
@@ -611,6 +615,22 @@ def evaluate_agentic_conversation(
|
|
|
611
615
|
value=float(result.full_skill_coverage),
|
|
612
616
|
data_type="BOOLEAN",
|
|
613
617
|
)
|
|
618
|
+
# One turn per fixture turn, plus every simulated-user round the agent triggered.
|
|
619
|
+
# The clarification count alone hides how much of the conversation the fixture
|
|
620
|
+
# asked for, so the comparison needs the total.
|
|
621
|
+
ctx.score(
|
|
622
|
+
tid,
|
|
623
|
+
name="turns",
|
|
624
|
+
value=len(result.turn_results) + result.total_clarification_turns,
|
|
625
|
+
data_type="NUMERIC",
|
|
626
|
+
)
|
|
627
|
+
ctx.score(tid, name="steps", value=result.total_steps, data_type="NUMERIC")
|
|
628
|
+
ctx.score(
|
|
629
|
+
tid,
|
|
630
|
+
name="clarification_turns",
|
|
631
|
+
value=result.total_clarification_turns,
|
|
632
|
+
data_type="NUMERIC",
|
|
633
|
+
)
|
|
614
634
|
for tr in result.turn_results:
|
|
615
635
|
ctx.score(
|
|
616
636
|
tid,
|
{gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/agentic/kda_skill.py
RENAMED
|
@@ -185,6 +185,8 @@ class KdaRunResult:
|
|
|
185
185
|
# Wall-clock time of the turn that called create (None if create never happened) --
|
|
186
186
|
# not any earlier disambiguation turn. See run_agentic_kda_skill's _run_once.
|
|
187
187
|
turn_wall_clock_sec: float | None = None
|
|
188
|
+
total_turns: int = 0
|
|
189
|
+
total_steps: int = 0
|
|
188
190
|
reasoning_steps: list[str] = field(default_factory=list)
|
|
189
191
|
response_id: str | None = None
|
|
190
192
|
tool_call_events: list[ToolCallEvent] = field(default_factory=list)
|
|
@@ -276,6 +278,9 @@ def run_agentic_kda_skill(
|
|
|
276
278
|
all_tool_call_events.extend(result.tool_call_events or [])
|
|
277
279
|
all_reasoning_step_events.extend(result.reasoning_step_events or [])
|
|
278
280
|
|
|
281
|
+
turns = 0
|
|
282
|
+
steps = 0
|
|
283
|
+
|
|
279
284
|
for iteration in range(max_iterations):
|
|
280
285
|
try:
|
|
281
286
|
chat_result = client.send_message(conv_id, current_question)
|
|
@@ -291,6 +296,8 @@ def run_agentic_kda_skill(
|
|
|
291
296
|
turn_wall_clock_sec = partial.turn_wall_clock_sec
|
|
292
297
|
turn_completed = False
|
|
293
298
|
break
|
|
299
|
+
turns += 1
|
|
300
|
+
steps += chat_result.reasoning_step_count
|
|
294
301
|
reasoning_steps.extend(chat_result.reasoning_steps or [])
|
|
295
302
|
response_id = chat_result.response_id or response_id
|
|
296
303
|
_accumulate(chat_result)
|
|
@@ -330,6 +337,8 @@ def run_agentic_kda_skill(
|
|
|
330
337
|
actual_create_args=create_args,
|
|
331
338
|
actual_execute_result=execute_result,
|
|
332
339
|
turn_wall_clock_sec=turn_wall_clock_sec,
|
|
340
|
+
total_turns=turns,
|
|
341
|
+
total_steps=steps,
|
|
333
342
|
reasoning_steps=reasoning_steps,
|
|
334
343
|
response_id=response_id,
|
|
335
344
|
tool_call_events=all_tool_call_events,
|
|
@@ -442,6 +451,8 @@ def evaluate_agentic_kda_skill(
|
|
|
442
451
|
for score_name, value in strict_checks.items():
|
|
443
452
|
ctx.score(tid, name=score_name, value=float(value), data_type="BOOLEAN")
|
|
444
453
|
ctx.score(tid, name="kda_disambiguated", value=float(ev.disambiguated), data_type="BOOLEAN")
|
|
454
|
+
ctx.score(tid, name="turns", value=run.total_turns, data_type="NUMERIC")
|
|
455
|
+
ctx.score(tid, name="steps", value=run.total_steps, data_type="NUMERIC")
|
|
445
456
|
if turn_wall_clock_sec is not None:
|
|
446
457
|
# combo_report.py reads this score directly -- no trace re-resolution needed.
|
|
447
458
|
ctx.score(
|
|
@@ -162,7 +162,8 @@ class MetricRunResult:
|
|
|
162
162
|
metric_created: bool
|
|
163
163
|
actual_maql: str
|
|
164
164
|
maql_correct: bool
|
|
165
|
-
total_turns:
|
|
165
|
+
total_turns: int
|
|
166
|
+
total_steps: int = 0
|
|
166
167
|
reasoning_steps: list[str] = field(default_factory=list)
|
|
167
168
|
response_id: str | None = None
|
|
168
169
|
tool_call_events: list[ToolCallEvent] = field(default_factory=list)
|
|
@@ -253,6 +254,7 @@ def _execute_single_metric_run(
|
|
|
253
254
|
metric_result: dict | None = None
|
|
254
255
|
created_metric_ids: list[str] = []
|
|
255
256
|
turns = 0
|
|
257
|
+
steps = 0
|
|
256
258
|
current_question = question
|
|
257
259
|
reasoning_steps: list[str] = []
|
|
258
260
|
response_id: str | None = None
|
|
@@ -280,6 +282,7 @@ def _execute_single_metric_run(
|
|
|
280
282
|
)
|
|
281
283
|
all_tool_call_events.extend(chat_result.tool_call_events or [])
|
|
282
284
|
all_reasoning_step_events.extend(chat_result.reasoning_step_events or [])
|
|
285
|
+
steps += chat_result.reasoning_step_count
|
|
283
286
|
for metric_id in _extract_created_metric_ids(chat_result.tool_call_events or []):
|
|
284
287
|
if metric_id not in created_metric_ids:
|
|
285
288
|
created_metric_ids.append(metric_id)
|
|
@@ -330,7 +333,8 @@ def _execute_single_metric_run(
|
|
|
330
333
|
metric_created=metric_created,
|
|
331
334
|
actual_maql=actual_maql,
|
|
332
335
|
maql_correct=maql_correct,
|
|
333
|
-
total_turns=
|
|
336
|
+
total_turns=turns,
|
|
337
|
+
total_steps=steps,
|
|
334
338
|
reasoning_steps=reasoning_steps,
|
|
335
339
|
response_id=response_id,
|
|
336
340
|
tool_call_events=all_tool_call_events,
|
|
@@ -466,6 +470,8 @@ def evaluate_agentic_metric_skill(
|
|
|
466
470
|
) as tid:
|
|
467
471
|
ctx.score(tid, name="metric_created", value=float(run.metric_created), data_type="BOOLEAN")
|
|
468
472
|
ctx.score(tid, name="maql_correct", value=float(run.maql_correct), data_type="BOOLEAN")
|
|
473
|
+
ctx.score(tid, name="turns", value=run.total_turns, data_type="NUMERIC")
|
|
474
|
+
ctx.score(tid, name="steps", value=run.total_steps, data_type="NUMERIC")
|
|
469
475
|
log_gate_scores(ctx, tid, gate=gate, pass_at_k=summary.pass_at_k, pass_power_k=summary.pass_power_k)
|
|
470
476
|
ctx.quality(
|
|
471
477
|
tid,
|
|
@@ -59,8 +59,8 @@ class RunResult:
|
|
|
59
59
|
actual_output: CreatedVisualization | None
|
|
60
60
|
eval_result: EvaluationResult
|
|
61
61
|
best_expected: CreatedVisualization
|
|
62
|
-
total_turns:
|
|
63
|
-
total_steps:
|
|
62
|
+
total_turns: int
|
|
63
|
+
total_steps: int
|
|
64
64
|
reasoning_steps: list[str] = field(default_factory=list)
|
|
65
65
|
response_id: str | None = None
|
|
66
66
|
tool_call_events: list[ToolCallEvent] = field(default_factory=list)
|
|
@@ -186,8 +186,8 @@ def _execute_single_run(
|
|
|
186
186
|
max_iterations: int = _DEFAULT_MAX_ITERATIONS,
|
|
187
187
|
) -> RunResult:
|
|
188
188
|
"""Drive one full multi-turn conversation and evaluate the result."""
|
|
189
|
-
total_turns = 0
|
|
190
|
-
total_steps = 0
|
|
189
|
+
total_turns = 0
|
|
190
|
+
total_steps = 0
|
|
191
191
|
all_tool_call_events: list[ToolCallEvent] = []
|
|
192
192
|
all_reasoning_step_events: list[ReasoningStepEvent] = []
|
|
193
193
|
reasoning_steps: list[str] = []
|
|
@@ -200,8 +200,8 @@ def _execute_single_run(
|
|
|
200
200
|
current_result = client.send_message(conversation_id, question)
|
|
201
201
|
|
|
202
202
|
for iteration in range(max_iterations):
|
|
203
|
-
total_turns += 1
|
|
204
|
-
total_steps +=
|
|
203
|
+
total_turns += 1
|
|
204
|
+
total_steps += current_result.reasoning_step_count
|
|
205
205
|
turn_offset, tool_index_offset, reasoning_index_offset = shift_and_index_events(
|
|
206
206
|
current_result,
|
|
207
207
|
turn_offset=turn_offset,
|
|
@@ -996,3 +996,67 @@ def test_every_gen_ai_interval_is_accepted():
|
|
|
996
996
|
assert AnomalyDetectionGranularity.parse(value.lower()) is AnomalyDetectionGranularity(value)
|
|
997
997
|
assert AnomalyDetectionGranularity.parse(None) is None
|
|
998
998
|
assert AnomalyDetectionGranularity.parse(" ") is None
|
|
999
|
+
|
|
1000
|
+
|
|
1001
|
+
def test_run_agentic_alert_skill_counts_the_turns_and_reasoning_steps_it_used():
|
|
1002
|
+
"""QA-29110: the effort comparison reads these. A refusal still took a turn, and the turn
|
|
1003
|
+
count is what separates a wrong answer from a run max_iterations cut short."""
|
|
1004
|
+
mock_client = MagicMock()
|
|
1005
|
+
mock_client.create_conversation.return_value = "conv-1"
|
|
1006
|
+
mock_client.send_message.return_value = _no_alert_chat_result()
|
|
1007
|
+
mock_client._base = "http://host/api/v1/actions/workspaces/ws1/ai"
|
|
1008
|
+
mock_client._auth = {"Authorization": "Bearer tok"}
|
|
1009
|
+
|
|
1010
|
+
with _patched(mock_client, simulated_reply="Yes please"):
|
|
1011
|
+
summary = run_agentic_alert_skill(
|
|
1012
|
+
host="http://host/api/v1/actions/workspaces/ws1/ai",
|
|
1013
|
+
token="tok",
|
|
1014
|
+
workspace_id="ws1",
|
|
1015
|
+
question="Create alert",
|
|
1016
|
+
expected_output={"operator": "GREATER_THAN", "threshold": 100},
|
|
1017
|
+
k=1,
|
|
1018
|
+
max_iterations=2,
|
|
1019
|
+
)
|
|
1020
|
+
|
|
1021
|
+
# _no_alert_chat_result has no tool calls and non-empty text, so the run replies once and
|
|
1022
|
+
# stops at max_iterations: 2 turns, 1 reasoning step each.
|
|
1023
|
+
assert summary.best.total_turns == 2
|
|
1024
|
+
assert summary.best.total_steps == 2
|
|
1025
|
+
|
|
1026
|
+
|
|
1027
|
+
def test_alert_skill_writes_the_turn_and_step_counts_to_langfuse():
|
|
1028
|
+
"""The counters exist to reach Langfuse; asserting only the dataclass would pass even if
|
|
1029
|
+
the scores were never written."""
|
|
1030
|
+
mock_client = MagicMock()
|
|
1031
|
+
mock_client.create_conversation.return_value = "conv-1"
|
|
1032
|
+
mock_client.send_message.return_value = _no_alert_chat_result()
|
|
1033
|
+
mock_client._base = "http://host/api/v1/actions/workspaces/ws1/ai"
|
|
1034
|
+
mock_client._auth = {"Authorization": "Bearer tok"}
|
|
1035
|
+
captured = {}
|
|
1036
|
+
|
|
1037
|
+
def _capture(_submit, _identity, **kwargs):
|
|
1038
|
+
captured["write_scores"] = kwargs["write_scores"]
|
|
1039
|
+
|
|
1040
|
+
with (
|
|
1041
|
+
_patched(mock_client),
|
|
1042
|
+
patch("gooddata_eval.core.agentic.alert_skill.submit_trace_scoring", _capture),
|
|
1043
|
+
pytest.raises(AlertSkillAssertionError),
|
|
1044
|
+
):
|
|
1045
|
+
evaluate_agentic_alert_skill(
|
|
1046
|
+
host="http://host/api/v1/actions/workspaces/ws1/ai",
|
|
1047
|
+
token="tok",
|
|
1048
|
+
workspace_id="ws1",
|
|
1049
|
+
question="Create alert",
|
|
1050
|
+
expected_output={"operator": "GREATER_THAN", "threshold": 100},
|
|
1051
|
+
k=1,
|
|
1052
|
+
max_iterations=1,
|
|
1053
|
+
langfuse=MagicMock(),
|
|
1054
|
+
dataset_item_id="item-1",
|
|
1055
|
+
)
|
|
1056
|
+
|
|
1057
|
+
ctx = MagicMock()
|
|
1058
|
+
captured["write_scores"](ctx)
|
|
1059
|
+
scores = {c.kwargs["name"]: c.kwargs["value"] for c in ctx.score.call_args_list}
|
|
1060
|
+
|
|
1061
|
+
assert scores["turns"] == 1
|
|
1062
|
+
assert scores["steps"] == 1
|
|
@@ -1025,3 +1025,107 @@ def test_evaluate_agentic_conversation_attaches_reasoning_steps_to_exception_on_
|
|
|
1025
1025
|
],
|
|
1026
1026
|
"latency_breakdown": [],
|
|
1027
1027
|
}
|
|
1028
|
+
|
|
1029
|
+
|
|
1030
|
+
def test_run_agentic_conversation_sums_the_reasoning_steps_of_every_turn():
|
|
1031
|
+
"""QA-29110: the effort comparison reads `steps`. A clarification round is part of the
|
|
1032
|
+
work the effort setting changes, so its steps count with the rest."""
|
|
1033
|
+
proposal_turn = ChatResult.model_validate(
|
|
1034
|
+
{
|
|
1035
|
+
"text_response": None,
|
|
1036
|
+
"alertProposals": [{"cta": "Should I create this alert?", "recipients": [{"email": "a@b.com"}]}],
|
|
1037
|
+
"reasoningStepCount": 2,
|
|
1038
|
+
"toolCallEvents": [
|
|
1039
|
+
{"functionName": "set_skills", "functionArguments": '{"skills": ["alert"]}', "result": None},
|
|
1040
|
+
{"functionName": "prepare_metric_alert_proposal", "functionArguments": "{}", "result": None},
|
|
1041
|
+
],
|
|
1042
|
+
}
|
|
1043
|
+
)
|
|
1044
|
+
created_turn = ChatResult.model_validate(
|
|
1045
|
+
{
|
|
1046
|
+
"text_response": "Alert created.",
|
|
1047
|
+
"reasoningStepCount": 3,
|
|
1048
|
+
"toolCallEvents": [
|
|
1049
|
+
{"functionName": "create_metric_alert", "functionArguments": "{}", "result": '{"id": "alert-1"}'}
|
|
1050
|
+
],
|
|
1051
|
+
}
|
|
1052
|
+
)
|
|
1053
|
+
mock_client = MagicMock()
|
|
1054
|
+
mock_client.create_conversation.return_value = "conv-1"
|
|
1055
|
+
mock_client.send_message.side_effect = [proposal_turn, created_turn]
|
|
1056
|
+
|
|
1057
|
+
with (
|
|
1058
|
+
patch("gooddata_eval.core.agentic.conversation.ChatClient", return_value=mock_client),
|
|
1059
|
+
patch("gooddata_eval.core.agentic.conversation.GoodDataSdk"),
|
|
1060
|
+
patch(
|
|
1061
|
+
"gooddata_eval.core.agentic.conversation._get_sim_user_response",
|
|
1062
|
+
return_value="Yes, please create it.",
|
|
1063
|
+
),
|
|
1064
|
+
):
|
|
1065
|
+
result = run_agentic_conversation(
|
|
1066
|
+
host="http://host/api/v1/actions/workspaces/ws1/ai",
|
|
1067
|
+
token="tok",
|
|
1068
|
+
workspace_id="ws1",
|
|
1069
|
+
fixture=_alert_turn_fixture(),
|
|
1070
|
+
)
|
|
1071
|
+
|
|
1072
|
+
assert result.total_steps == 5
|
|
1073
|
+
assert result.total_clarification_turns == 1
|
|
1074
|
+
|
|
1075
|
+
|
|
1076
|
+
def test_conversation_writes_the_turn_step_and_clarification_counts_to_langfuse():
|
|
1077
|
+
"""`turns` is not the clarification count: it is one per fixture turn plus every
|
|
1078
|
+
simulated-user round, so a test has to pin the sum rather than either half."""
|
|
1079
|
+
proposal_turn = ChatResult.model_validate(
|
|
1080
|
+
{
|
|
1081
|
+
"text_response": None,
|
|
1082
|
+
"alertProposals": [{"cta": "Should I create this alert?", "recipients": [{"email": "a@b.com"}]}],
|
|
1083
|
+
"reasoningStepCount": 2,
|
|
1084
|
+
"toolCallEvents": [
|
|
1085
|
+
{"functionName": "set_skills", "functionArguments": '{"skills": ["alert"]}', "result": None},
|
|
1086
|
+
{"functionName": "prepare_metric_alert_proposal", "functionArguments": "{}", "result": None},
|
|
1087
|
+
],
|
|
1088
|
+
}
|
|
1089
|
+
)
|
|
1090
|
+
created_turn = ChatResult.model_validate(
|
|
1091
|
+
{
|
|
1092
|
+
"text_response": "Alert created.",
|
|
1093
|
+
"reasoningStepCount": 3,
|
|
1094
|
+
"toolCallEvents": [
|
|
1095
|
+
{"functionName": "create_metric_alert", "functionArguments": "{}", "result": '{"id": "alert-1"}'}
|
|
1096
|
+
],
|
|
1097
|
+
}
|
|
1098
|
+
)
|
|
1099
|
+
mock_client = MagicMock()
|
|
1100
|
+
mock_client.create_conversation.return_value = "conv-1"
|
|
1101
|
+
mock_client.send_message.side_effect = [proposal_turn, created_turn]
|
|
1102
|
+
captured = {}
|
|
1103
|
+
|
|
1104
|
+
def _capture(_submit, _identity, **kwargs):
|
|
1105
|
+
captured["write_scores"] = kwargs["write_scores"]
|
|
1106
|
+
|
|
1107
|
+
with (
|
|
1108
|
+
patch("gooddata_eval.core.agentic.conversation.ChatClient", return_value=mock_client),
|
|
1109
|
+
patch("gooddata_eval.core.agentic.conversation.GoodDataSdk"),
|
|
1110
|
+
patch("gooddata_eval.core.agentic.conversation.submit_trace_scoring", _capture),
|
|
1111
|
+
patch(
|
|
1112
|
+
"gooddata_eval.core.agentic.conversation._get_sim_user_response",
|
|
1113
|
+
return_value="Yes, please create it.",
|
|
1114
|
+
),
|
|
1115
|
+
):
|
|
1116
|
+
evaluate_agentic_conversation(
|
|
1117
|
+
host="http://host/api/v1/actions/workspaces/ws1/ai",
|
|
1118
|
+
token="tok",
|
|
1119
|
+
workspace_id="ws1",
|
|
1120
|
+
fixture=_alert_turn_fixture(),
|
|
1121
|
+
langfuse=MagicMock(),
|
|
1122
|
+
dataset_item_id="item-1",
|
|
1123
|
+
)
|
|
1124
|
+
|
|
1125
|
+
ctx = MagicMock()
|
|
1126
|
+
captured["write_scores"](ctx)
|
|
1127
|
+
scores = {c.kwargs["name"]: c.kwargs["value"] for c in ctx.score.call_args_list}
|
|
1128
|
+
|
|
1129
|
+
assert scores["clarification_turns"] == 1
|
|
1130
|
+
assert scores["turns"] == 2 # 1 fixture turn + 1 clarification round
|
|
1131
|
+
assert scores["steps"] == 5
|
|
@@ -1195,3 +1195,88 @@ def test_evaluate_agentic_kda_skill_preserves_reasoning_from_a_chat_error_partia
|
|
|
1195
1195
|
|
|
1196
1196
|
assert exc_info.value.reasoning_steps == ["analyzing before cutoff"]
|
|
1197
1197
|
assert exc_info.value.response_id == "resp-3"
|
|
1198
|
+
|
|
1199
|
+
|
|
1200
|
+
def test_run_agentic_kda_skill_counts_the_turns_and_reasoning_steps_it_used():
|
|
1201
|
+
"""QA-29110: the effort comparison reads these. A binary pass/fail cannot separate two
|
|
1202
|
+
efforts on a nightly's sample, while the reasoning-step count moves with the effort."""
|
|
1203
|
+
mock_client = MagicMock()
|
|
1204
|
+
mock_client.create_conversation.return_value = "conv-1"
|
|
1205
|
+
# Turn 1 asks for clarification, turn 2 runs the analysis: 2 turns, 1 step each.
|
|
1206
|
+
mock_client.send_message.side_effect = [
|
|
1207
|
+
_no_kda_chat_result("Which metric did you mean?"),
|
|
1208
|
+
_kda_chat_result(success=True),
|
|
1209
|
+
]
|
|
1210
|
+
|
|
1211
|
+
with (
|
|
1212
|
+
patch("gooddata_eval.core.agentic.kda_skill.ChatClient", return_value=mock_client),
|
|
1213
|
+
patch("gooddata_eval.core.agentic.kda_skill.generate_simulated_kda_response", return_value="Revenue"),
|
|
1214
|
+
):
|
|
1215
|
+
summary = run_agentic_kda_skill(
|
|
1216
|
+
host="http://host/api/v1/actions/workspaces/ws1/ai",
|
|
1217
|
+
token="tok",
|
|
1218
|
+
workspace_id="ws1",
|
|
1219
|
+
question="What drove the change?",
|
|
1220
|
+
expected_output=_EXPECTED,
|
|
1221
|
+
k=1,
|
|
1222
|
+
max_iterations=2,
|
|
1223
|
+
)
|
|
1224
|
+
|
|
1225
|
+
assert summary.best.total_turns == 2
|
|
1226
|
+
assert summary.best.total_steps == 2
|
|
1227
|
+
|
|
1228
|
+
|
|
1229
|
+
def test_run_agentic_kda_skill_reports_no_turns_when_the_first_send_fails():
|
|
1230
|
+
"""A run that never got a reply must not report a turn it did not take."""
|
|
1231
|
+
mock_client = MagicMock()
|
|
1232
|
+
mock_client.create_conversation.return_value = "conv-1"
|
|
1233
|
+
mock_client.send_message.side_effect = RuntimeError("stream died")
|
|
1234
|
+
|
|
1235
|
+
with patch("gooddata_eval.core.agentic.kda_skill.ChatClient", return_value=mock_client):
|
|
1236
|
+
summary = run_agentic_kda_skill(
|
|
1237
|
+
host="http://host/api/v1/actions/workspaces/ws1/ai",
|
|
1238
|
+
token="tok",
|
|
1239
|
+
workspace_id="ws1",
|
|
1240
|
+
question="What drove the change?",
|
|
1241
|
+
expected_output=_EXPECTED,
|
|
1242
|
+
k=1,
|
|
1243
|
+
max_iterations=1,
|
|
1244
|
+
)
|
|
1245
|
+
|
|
1246
|
+
assert summary.best.total_turns == 0
|
|
1247
|
+
assert summary.best.total_steps == 0
|
|
1248
|
+
|
|
1249
|
+
|
|
1250
|
+
def test_kda_skill_writes_the_turn_and_step_counts_to_langfuse():
|
|
1251
|
+
"""The counters exist to reach Langfuse; asserting only the dataclass would pass even if
|
|
1252
|
+
the scores were never written."""
|
|
1253
|
+
mock_client = MagicMock()
|
|
1254
|
+
mock_client.create_conversation.return_value = "conv-1"
|
|
1255
|
+
mock_client.send_message.return_value = _kda_chat_result(success=True)
|
|
1256
|
+
captured = {}
|
|
1257
|
+
|
|
1258
|
+
def _capture(_submit, _identity, **kwargs):
|
|
1259
|
+
captured["write_scores"] = kwargs["write_scores"]
|
|
1260
|
+
|
|
1261
|
+
with (
|
|
1262
|
+
patch("gooddata_eval.core.agentic.kda_skill.ChatClient", return_value=mock_client),
|
|
1263
|
+
patch("gooddata_eval.core.agentic.kda_skill.submit_trace_scoring", _capture),
|
|
1264
|
+
):
|
|
1265
|
+
evaluate_agentic_kda_skill(
|
|
1266
|
+
host="http://host/api/v1/actions/workspaces/ws1/ai",
|
|
1267
|
+
token="tok",
|
|
1268
|
+
workspace_id="ws1",
|
|
1269
|
+
question="What drove the change?",
|
|
1270
|
+
expected_output=_EXPECTED,
|
|
1271
|
+
k=1,
|
|
1272
|
+
max_iterations=1,
|
|
1273
|
+
langfuse=MagicMock(),
|
|
1274
|
+
dataset_item_id="item-1",
|
|
1275
|
+
)
|
|
1276
|
+
|
|
1277
|
+
ctx = MagicMock()
|
|
1278
|
+
captured["write_scores"](ctx)
|
|
1279
|
+
scores = {c.kwargs["name"]: c.kwargs["value"] for c in ctx.score.call_args_list}
|
|
1280
|
+
|
|
1281
|
+
assert scores["turns"] == 1
|
|
1282
|
+
assert scores["steps"] == 1
|
|
@@ -582,7 +582,7 @@ def test_run_agentic_metric_skill_fails_the_run_when_the_simulated_reply_cannot_
|
|
|
582
582
|
|
|
583
583
|
assert summary.pass_at_k is False
|
|
584
584
|
assert summary.best.metric_created is False
|
|
585
|
-
assert summary.best.total_turns == 1
|
|
585
|
+
assert summary.best.total_turns == 1
|
|
586
586
|
mock_client.close.assert_called_once()
|
|
587
587
|
mock_sim.assert_called_once_with(
|
|
588
588
|
"Which brand field should I count?", [{"maql": "SELECT {metric/foo}"}], "Create metric foo"
|
|
@@ -764,3 +764,90 @@ def test_no_timer_output_by_default(monkeypatch, capsys):
|
|
|
764
764
|
assert "[timer]" not in capsys.readouterr().out
|
|
765
765
|
# Silenced, not un-measured.
|
|
766
766
|
assert summary.run_results[0].timings.agent_s == 3.0
|
|
767
|
+
|
|
768
|
+
|
|
769
|
+
def test_run_agentic_metric_skill_counts_the_turns_and_reasoning_steps_it_used():
|
|
770
|
+
"""QA-29110: the effort comparison reads these. A clarification round is part of the work
|
|
771
|
+
the effort setting changes, so its steps count with the rest."""
|
|
772
|
+
clarify_turn = ChatResult.model_validate(
|
|
773
|
+
{"textResponse": "Which foo?", "toolCallEvents": [], "reasoningStepCount": 2}
|
|
774
|
+
)
|
|
775
|
+
created_turn = ChatResult.model_validate(
|
|
776
|
+
{
|
|
777
|
+
"textResponse": "done",
|
|
778
|
+
"reasoningStepCount": 3,
|
|
779
|
+
"toolCallEvents": [
|
|
780
|
+
{
|
|
781
|
+
"functionName": "create_metric",
|
|
782
|
+
"functionArguments": "{}",
|
|
783
|
+
"result": '{"data": {"maql": "SELECT {metric/foo}"}}',
|
|
784
|
+
}
|
|
785
|
+
],
|
|
786
|
+
}
|
|
787
|
+
)
|
|
788
|
+
mock_client = _client()
|
|
789
|
+
mock_client.send_message.side_effect = [clarify_turn, created_turn]
|
|
790
|
+
|
|
791
|
+
with _patched(mock_client, simulated_reply="It's foo"):
|
|
792
|
+
summary = run_agentic_metric_skill(
|
|
793
|
+
host="http://host/api/v1/actions/workspaces/ws1/ai",
|
|
794
|
+
token="tok",
|
|
795
|
+
workspace_id="ws1",
|
|
796
|
+
question="Create metric foo",
|
|
797
|
+
expected_output={"maql": "SELECT {metric/foo}"},
|
|
798
|
+
k=1,
|
|
799
|
+
max_iterations=2,
|
|
800
|
+
)
|
|
801
|
+
|
|
802
|
+
assert summary.best.total_turns == 2
|
|
803
|
+
assert summary.best.total_steps == 5
|
|
804
|
+
|
|
805
|
+
|
|
806
|
+
def test_metric_skill_writes_the_turn_and_step_counts_to_langfuse():
|
|
807
|
+
"""The counters exist to reach Langfuse; asserting only the dataclass would pass even if
|
|
808
|
+
the scores were never written."""
|
|
809
|
+
mock_client = _client()
|
|
810
|
+
mock_client.send_message.return_value = ChatResult.model_validate(
|
|
811
|
+
{
|
|
812
|
+
"textResponse": "done",
|
|
813
|
+
"reasoningStepCount": 4,
|
|
814
|
+
"toolCallEvents": [
|
|
815
|
+
{
|
|
816
|
+
"functionName": "create_metric",
|
|
817
|
+
"functionArguments": "{}",
|
|
818
|
+
"result": '{"data": {"maql": "SELECT {metric/foo}"}}',
|
|
819
|
+
}
|
|
820
|
+
],
|
|
821
|
+
}
|
|
822
|
+
)
|
|
823
|
+
captured = {}
|
|
824
|
+
|
|
825
|
+
def _capture(_submit, _identity, **kwargs):
|
|
826
|
+
captured["write_scores"] = kwargs["write_scores"]
|
|
827
|
+
|
|
828
|
+
with (
|
|
829
|
+
_patched(mock_client),
|
|
830
|
+
patch("gooddata_eval.core.agentic.metric_skill.submit_trace_scoring", _capture),
|
|
831
|
+
):
|
|
832
|
+
evaluate_agentic_metric_skill(
|
|
833
|
+
host="http://host/api/v1/actions/workspaces/ws1/ai",
|
|
834
|
+
token="tok",
|
|
835
|
+
workspace_id="ws1",
|
|
836
|
+
question="Create metric foo",
|
|
837
|
+
expected_output={"maql": "SELECT {metric/foo}"},
|
|
838
|
+
k=1,
|
|
839
|
+
max_iterations=1,
|
|
840
|
+
langfuse=MagicMock(),
|
|
841
|
+
dataset_item_id="item-1",
|
|
842
|
+
)
|
|
843
|
+
|
|
844
|
+
ctx = MagicMock()
|
|
845
|
+
captured["write_scores"](ctx)
|
|
846
|
+
scores = {c.kwargs["name"]: c.kwargs["value"] for c in ctx.score.call_args_list}
|
|
847
|
+
|
|
848
|
+
assert scores["turns"] == 1
|
|
849
|
+
assert scores["steps"] == 4
|
|
850
|
+
# `==` does not separate 1 from 1.0, so the counts need their type pinned separately:
|
|
851
|
+
# they are counts, and a float reads as though a fraction of a turn were possible.
|
|
852
|
+
assert isinstance(scores["turns"], int)
|
|
853
|
+
assert isinstance(scores["steps"], int)
|
|
@@ -71,8 +71,8 @@ def test_execute_single_run_viz_on_first_turn():
|
|
|
71
71
|
|
|
72
72
|
assert result.eval_result.visualization_created is True
|
|
73
73
|
assert result.eval_result.strict_pass is True
|
|
74
|
-
assert result.total_turns == 1
|
|
75
|
-
assert result.total_steps == 2
|
|
74
|
+
assert result.total_turns == 1
|
|
75
|
+
assert result.total_steps == 2
|
|
76
76
|
assert result.conversation_id == "conv-1"
|
|
77
77
|
client.send_message.assert_called_once_with("conv-1", "Show revenue")
|
|
78
78
|
|
|
@@ -93,7 +93,7 @@ def test_execute_single_run_clarification_then_viz(monkeypatch):
|
|
|
93
93
|
result = _execute_single_run(client, "conv-1", "Show me a chart", [_expected()])
|
|
94
94
|
|
|
95
95
|
assert result.eval_result.visualization_created is True
|
|
96
|
-
assert result.total_turns == 2
|
|
96
|
+
assert result.total_turns == 2
|
|
97
97
|
assert client.send_message.call_count == 2
|
|
98
98
|
assert client.send_message.call_args_list[1] == call("conv-1", "Revenue please")
|
|
99
99
|
|
|
@@ -111,7 +111,7 @@ def test_execute_single_run_no_viz_no_text():
|
|
|
111
111
|
result = _execute_single_run(client, "conv-1", "Show revenue", [_expected()])
|
|
112
112
|
|
|
113
113
|
assert result.eval_result.visualization_created is False
|
|
114
|
-
assert result.total_turns == 1
|
|
114
|
+
assert result.total_turns == 1
|
|
115
115
|
|
|
116
116
|
|
|
117
117
|
def test_execute_single_run_max_iterations_stops_loop(monkeypatch):
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/cli/agentic_runner.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/agentic/__init__.py
RENAMED
|
File without changes
|
{gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/agentic/_catalog.py
RENAMED
|
File without changes
|
{gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/agentic/_gate.py
RENAMED
|
File without changes
|
{gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/agentic/_langfuse.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
{gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/agentic/guardrail.py
RENAMED
|
File without changes
|
|
File without changes
|
{gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/chat/__init__.py
RENAMED
|
File without changes
|
{gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/chat/render.py
RENAMED
|
File without changes
|
{gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/chat/sse_client.py
RENAMED
|
File without changes
|
|
File without changes
|
{gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/connection.py
RENAMED
|
File without changes
|
{gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/dataset/__init__.py
RENAMED
|
File without changes
|
|
File without changes
|
{gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/dataset/local.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/evaluators/_maql.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
{gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/evaluators/base.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/evaluators/summary.py
RENAMED
|
File without changes
|
|
File without changes
|
{gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/langfuse/__init__.py
RENAMED
|
File without changes
|
{gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/langfuse/_env.py
RENAMED
|
File without changes
|
{gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/langfuse/client.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
{gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/langfuse/otlp.py
RENAMED
|
File without changes
|
{gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/langfuse/sink.py
RENAMED
|
File without changes
|
|
File without changes
|
{gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/reporting/__init__.py
RENAMED
|
File without changes
|
{gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/reporting/console.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/summary/__init__.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/tests/fixtures/sse_visualization_stream.txt
RENAMED
|
File without changes
|
|
File without changes
|
{gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/tests/test_agentic_general_question.py
RENAMED
|
File without changes
|
|
File without changes
|
{gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/tests/test_agentic_langfuse_trace.py
RENAMED
|
File without changes
|
{gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/tests/test_agentic_observe_experiment.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/tests/test_langfuse_e2e_fake_server.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/tests/test_metric_skill_evaluator.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/tests/test_visualization_evaluator.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|