gooddata-eval 1.75.1.dev1__tar.gz → 1.75.1.dev2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/PKG-INFO +2 -2
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/pyproject.toml +2 -2
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/scripts/verify_guardrail_refusal_criteria.py +3 -1
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/cli/agentic_runner.py +13 -0
- gooddata_eval-1.75.1.dev2/src/gooddata_eval/core/agentic/what_if.py +593 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/models.py +2 -2
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/tests/test_agentic_runner.py +1 -0
- gooddata_eval-1.75.1.dev2/tests/test_agentic_what_if.py +439 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/tests/test_from_insights.py +2 -2
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/tests/test_scoring.py +9 -9
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/tests/test_trace_linker.py +1 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/.gitignore +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/AGENTS.md +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/CLAUDE.md +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/LICENSE.txt +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/Makefile +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/README.md +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/__init__.py +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/_version.py +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/cli/__init__.py +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/cli/main.py +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/__init__.py +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/_output.py +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/agentic/__init__.py +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/agentic/_catalog.py +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/agentic/_gate.py +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/agentic/_langfuse.py +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/agentic/_trace_linker.py +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/agentic/alert_skill.py +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/agentic/conversation.py +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/agentic/dashboard_skill.py +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/agentic/general_question.py +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/agentic/guardrail.py +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/agentic/kda_skill.py +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/agentic/metric_skill.py +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/agentic/search_tool.py +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/agentic/visualization.py +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/chat/__init__.py +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/chat/render.py +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/chat/sse_client.py +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/config.py +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/connection.py +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/dataset/__init__.py +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/dataset/from_insights.py +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/dataset/langfuse_source.py +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/dataset/local.py +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/evaluators/__init__.py +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/evaluators/_deep_subset.py +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/evaluators/_guardrail_criteria.py +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/evaluators/_llm_judge.py +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/evaluators/_maql.py +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/evaluators/_text_utils.py +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/evaluators/alert_skill.py +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/evaluators/base.py +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/evaluators/general_question.py +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/evaluators/guardrail.py +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/evaluators/metric_skill.py +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/evaluators/search_tool.py +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/evaluators/summary.py +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/evaluators/visualization.py +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/granularity.py +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/langfuse/__init__.py +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/langfuse/_env.py +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/langfuse/client.py +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/langfuse/experiment.py +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/langfuse/observations.py +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/langfuse/otlp.py +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/langfuse/sink.py +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/reporting/__init__.py +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/reporting/console.py +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/reporting/html_report.py +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/reporting/json_report.py +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/reporting/report_template.html +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/runner.py +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/scoring.py +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/summary/__init__.py +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/summary/http_client.py +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/timing.py +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/workspace.py +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/tests/__init__.py +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/tests/_fake_langfuse.py +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/tests/conftest.py +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/tests/fixtures/sample_dataset/metric_skill_create.json +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/tests/fixtures/sample_dataset/visualization_revenue.json +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/tests/fixtures/sse_visualization_stream.txt +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/tests/test_agentic_alert_skill.py +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/tests/test_agentic_conversation.py +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/tests/test_agentic_dashboard_skill.py +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/tests/test_agentic_gate.py +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/tests/test_agentic_general_question.py +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/tests/test_agentic_guardrail.py +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/tests/test_agentic_kda_skill.py +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/tests/test_agentic_langfuse_trace.py +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/tests/test_agentic_metric_skill.py +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/tests/test_agentic_observe_experiment.py +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/tests/test_agentic_run_context.py +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/tests/test_agentic_search_tool.py +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/tests/test_agentic_visualization.py +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/tests/test_alert_skill_evaluator.py +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/tests/test_chat_render.py +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/tests/test_cli.py +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/tests/test_connection.py +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/tests/test_deep_subset.py +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/tests/test_fake_langfuse.py +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/tests/test_guardrail_criteria.py +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/tests/test_html_report.py +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/tests/test_langfuse_client.py +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/tests/test_langfuse_e2e_fake_server.py +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/tests/test_langfuse_env.py +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/tests/test_langfuse_experiment.py +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/tests/test_langfuse_observations.py +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/tests/test_langfuse_otlp.py +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/tests/test_langfuse_sink.py +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/tests/test_langfuse_source.py +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/tests/test_llm_judge.py +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/tests/test_local_loader.py +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/tests/test_maql_normalize.py +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/tests/test_metric_skill_evaluator.py +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/tests/test_models.py +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/tests/test_reporting.py +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/tests/test_runner.py +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/tests/test_search_tool_evaluator.py +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/tests/test_sse_client.py +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/tests/test_summary_client.py +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/tests/test_summary_evaluator.py +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/tests/test_text_evaluators.py +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/tests/test_timing.py +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/tests/test_visualization_evaluator.py +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/tests/test_workspace.py +0 -0
- {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/tox.ini +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: gooddata-eval
|
|
3
|
-
Version: 1.75.1.
|
|
3
|
+
Version: 1.75.1.dev2
|
|
4
4
|
Summary: Evaluate the GoodData AI agent against your own questions and models.
|
|
5
5
|
Project-URL: Source, https://github.com/gooddata/gooddata-python-sdk
|
|
6
6
|
Author-email: GoodData <support@gooddata.com>
|
|
@@ -17,7 +17,7 @@ Classifier: Topic :: Scientific/Engineering
|
|
|
17
17
|
Classifier: Topic :: Software Development
|
|
18
18
|
Classifier: Typing :: Typed
|
|
19
19
|
Requires-Python: >=3.10
|
|
20
|
-
Requires-Dist: gooddata-sdk~=1.75.1.
|
|
20
|
+
Requires-Dist: gooddata-sdk~=1.75.1.dev2
|
|
21
21
|
Requires-Dist: httpx<1.0,>=0.27
|
|
22
22
|
Requires-Dist: orjson<4.0.0,>=3.9.15
|
|
23
23
|
Requires-Dist: pydantic<3.0,>=2.6
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
# (C) 2026 GoodData Corporation
|
|
2
2
|
[project]
|
|
3
3
|
name = "gooddata-eval"
|
|
4
|
-
version = "1.75.1.
|
|
4
|
+
version = "1.75.1.dev2"
|
|
5
5
|
description = "Evaluate the GoodData AI agent against your own questions and models."
|
|
6
6
|
readme = "README.md"
|
|
7
7
|
license = "MIT"
|
|
@@ -11,7 +11,7 @@ authors = [
|
|
|
11
11
|
keywords = ["gooddata", "ai", "evaluation", "llm", "analytics", "cli"]
|
|
12
12
|
requires-python = ">=3.10"
|
|
13
13
|
dependencies = [
|
|
14
|
-
"gooddata-sdk~=1.75.1.
|
|
14
|
+
"gooddata-sdk~=1.75.1.dev2",
|
|
15
15
|
"httpx>=0.27,<1.0",
|
|
16
16
|
"orjson>=3.9.15,<4.0.0",
|
|
17
17
|
"pydantic>=2.6,<3.0",
|
{gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/scripts/verify_guardrail_refusal_criteria.py
RENAMED
|
@@ -39,7 +39,9 @@ import os
|
|
|
39
39
|
|
|
40
40
|
from dotenv import load_dotenv
|
|
41
41
|
|
|
42
|
-
|
|
42
|
+
# Whatever .env the caller points at, defaulting to the working directory. It used to be an
|
|
43
|
+
# absolute path, which made the script runnable on exactly one machine.
|
|
44
|
+
load_dotenv(os.environ.get("GD_EVAL_ENV_FILE", ".env"))
|
|
43
45
|
|
|
44
46
|
from gooddata_eval.core.agentic.guardrail import _GUARDRAIL_EVALUATION_STEPS # noqa: E402
|
|
45
47
|
from gooddata_eval.core.evaluators._guardrail_criteria import ( # noqa: E402
|
{gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/cli/agentic_runner.py
RENAMED
|
@@ -20,6 +20,7 @@ from gooddata_eval.core.agentic.kda_skill import evaluate_agentic_kda_skill
|
|
|
20
20
|
from gooddata_eval.core.agentic.metric_skill import evaluate_agentic_metric_skill
|
|
21
21
|
from gooddata_eval.core.agentic.search_tool import evaluate_agentic_search_tool
|
|
22
22
|
from gooddata_eval.core.agentic.visualization import evaluate_agentic_visualization
|
|
23
|
+
from gooddata_eval.core.agentic.what_if import evaluate_agentic_what_if
|
|
23
24
|
from gooddata_eval.core.config import ReasoningEffort
|
|
24
25
|
from gooddata_eval.core.models import AgenticEvalOutcome, CreatedVisualization, DatasetItem
|
|
25
26
|
from gooddata_eval.core.runner import EvalReport, ItemReport
|
|
@@ -47,6 +48,7 @@ AGENTIC_TEST_KINDS = frozenset(
|
|
|
47
48
|
"agentic_guardrail",
|
|
48
49
|
"agentic_conversation",
|
|
49
50
|
"agentic_kda_skill",
|
|
51
|
+
"agentic_what_if",
|
|
50
52
|
}
|
|
51
53
|
)
|
|
52
54
|
|
|
@@ -265,6 +267,17 @@ def _dispatch_agentic(
|
|
|
265
267
|
agent_id=agent_id,
|
|
266
268
|
**lf_kw,
|
|
267
269
|
)
|
|
270
|
+
elif kind == "agentic_what_if":
|
|
271
|
+
return evaluate_agentic_what_if(
|
|
272
|
+
host=host,
|
|
273
|
+
token=token,
|
|
274
|
+
workspace_id=workspace_id,
|
|
275
|
+
question=item.question,
|
|
276
|
+
expected_output=eo if isinstance(eo, dict) else {},
|
|
277
|
+
k=k,
|
|
278
|
+
agent_id=agent_id,
|
|
279
|
+
**lf_kw,
|
|
280
|
+
)
|
|
268
281
|
elif kind == "agentic_conversation":
|
|
269
282
|
fixture_data = eo.get("fixture") or eo if isinstance(eo, dict) else {}
|
|
270
283
|
return evaluate_agentic_conversation(
|
|
@@ -0,0 +1,593 @@
|
|
|
1
|
+
# (C) 2026 GoodData Corporation. All rights reserved.
|
|
2
|
+
"""Agentic what-if-analysis skill evaluation runner.
|
|
3
|
+
|
|
4
|
+
The skill builds a scenario spec and executes it:
|
|
5
|
+
|
|
6
|
+
create_what_if_scenario(visualization_ref, scenarios[], include_baseline)
|
|
7
|
+
execute_what_if_scenario(scenario_ref) -> one result per scenario, plus the baseline
|
|
8
|
+
|
|
9
|
+
Each scenario carries adjustments of the form ``{metric_id, metric_type, scenario_maql}``,
|
|
10
|
+
where ``scenario_maql`` is the adjusted expression -- a 10% uplift on a revenue metric
|
|
11
|
+
defined as ``SELECT SUM({fact/price} * {fact/quantity})`` becomes
|
|
12
|
+
``SELECT SUM({fact/price} * 1.10 * {fact/quantity})``.
|
|
13
|
+
|
|
14
|
+
That makes this the most checkable of the analysis skills: the adjustment is MAQL, and
|
|
15
|
+
MAQL already has a comparator here (``evaluators._maql.normalize_maql``, used by
|
|
16
|
+
metric_skill), so "did it apply the right adjustment" is answerable without a judge. What
|
|
17
|
+
the adjustment produced is not checked -- that is the platform's arithmetic, not the
|
|
18
|
+
agent's -- only that the agent asked for the right thing and the execution succeeded.
|
|
19
|
+
"""
|
|
20
|
+
|
|
21
|
+
from __future__ import annotations
|
|
22
|
+
|
|
23
|
+
import logging
|
|
24
|
+
import os
|
|
25
|
+
from dataclasses import dataclass, field
|
|
26
|
+
from typing import Any
|
|
27
|
+
|
|
28
|
+
from gooddata_eval.core.agentic._trace_linker import (
|
|
29
|
+
RunIdentity,
|
|
30
|
+
RunTraceContext,
|
|
31
|
+
SubmitTraceLink,
|
|
32
|
+
open_trace_window,
|
|
33
|
+
run_trace_link_inline,
|
|
34
|
+
submit_trace_scoring,
|
|
35
|
+
utc_now,
|
|
36
|
+
)
|
|
37
|
+
from gooddata_eval.core.chat.render import render_answer_text
|
|
38
|
+
from gooddata_eval.core.chat.sse_client import ChatClient
|
|
39
|
+
from gooddata_eval.core.config import ReasoningEffort
|
|
40
|
+
from gooddata_eval.core.evaluators._maql import normalize_maql
|
|
41
|
+
from gooddata_eval.core.models import (
|
|
42
|
+
AgenticAssertionError,
|
|
43
|
+
AgenticEvalOutcome,
|
|
44
|
+
ChatResult,
|
|
45
|
+
ReasoningStepEvent,
|
|
46
|
+
ToolCallEvent,
|
|
47
|
+
build_latency_breakdown,
|
|
48
|
+
shift_and_index_events,
|
|
49
|
+
)
|
|
50
|
+
|
|
51
|
+
_log = logging.getLogger(__name__)
|
|
52
|
+
|
|
53
|
+
_DEFAULT_K = 1
|
|
54
|
+
# The agent asks which measure to adjust before building anything (observed live: "I need
|
|
55
|
+
# to confirm which 'Spend' calculation you want to adjust"), so the budget covers a couple
|
|
56
|
+
# of disambiguation rounds plus slack.
|
|
57
|
+
_DEFAULT_MAX_ITERATIONS = 4
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def _build_clarification_prompt(agent_message: str, expected_output: dict) -> str:
|
|
61
|
+
"""The simulated-user reply, mentioning only the hints the fixture actually supplies."""
|
|
62
|
+
hints: list[str] = []
|
|
63
|
+
metric = expected_output.get("metric_id")
|
|
64
|
+
if metric:
|
|
65
|
+
hints.append(f"the measure to adjust is '{metric}'")
|
|
66
|
+
change = expected_output.get("change")
|
|
67
|
+
if change:
|
|
68
|
+
hints.append(f"the adjustment is {change}")
|
|
69
|
+
period = expected_output.get("period")
|
|
70
|
+
if period:
|
|
71
|
+
hints.append(f"the time period is {period}")
|
|
72
|
+
reference = "; ".join(hints)
|
|
73
|
+
return (
|
|
74
|
+
f"You are simulating a user in a conversation with a BI assistant that runs what-if "
|
|
75
|
+
f"scenario analysis. The assistant asked: '{agent_message}'. "
|
|
76
|
+
+ (f"For reference, {reference}. " if reference else "")
|
|
77
|
+
+ "Reply briefly as the user, answering whichever of those the assistant actually asked about."
|
|
78
|
+
)
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
def generate_simulated_what_if_response(agent_message: str, expected_output: dict) -> str:
|
|
82
|
+
"""Generate a user reply to keep the what-if conversation going (gpt-4o-mini).
|
|
83
|
+
|
|
84
|
+
Always OpenAI regardless of the workspace's own model: harness plumbing, not the system
|
|
85
|
+
under test.
|
|
86
|
+
"""
|
|
87
|
+
try:
|
|
88
|
+
from openai import OpenAI # noqa: PLC0415
|
|
89
|
+
except ImportError as exc:
|
|
90
|
+
raise RuntimeError("openai package is required for generate_simulated_what_if_response") from exc
|
|
91
|
+
|
|
92
|
+
api_key = os.environ.get("OPENAI_API_KEY")
|
|
93
|
+
if not api_key:
|
|
94
|
+
raise OSError("OPENAI_API_KEY environment variable is not set")
|
|
95
|
+
|
|
96
|
+
client = OpenAI(api_key=api_key)
|
|
97
|
+
response = client.chat.completions.create(
|
|
98
|
+
model="gpt-4o-mini",
|
|
99
|
+
messages=[{"role": "user", "content": _build_clarification_prompt(agent_message, expected_output)}],
|
|
100
|
+
max_tokens=150,
|
|
101
|
+
temperature=0,
|
|
102
|
+
timeout=30,
|
|
103
|
+
)
|
|
104
|
+
return response.choices[0].message.content or "Please proceed with the most complete option."
|
|
105
|
+
|
|
106
|
+
|
|
107
|
+
def _extract_what_if_calls(tool_call_events: list[ToolCallEvent]) -> tuple[dict | None, dict | None]:
|
|
108
|
+
"""Return (create_args, execute_result) for the LAST create/execute pair.
|
|
109
|
+
|
|
110
|
+
A new create_what_if_scenario clears any earlier execute result: that result belongs to
|
|
111
|
+
the spec it followed. Picking the last of each independently would score a fresh
|
|
112
|
+
scenario against a stale execution.
|
|
113
|
+
"""
|
|
114
|
+
create_args: dict | None = None
|
|
115
|
+
execute_result: dict | None = None
|
|
116
|
+
for tc in tool_call_events:
|
|
117
|
+
if tc.function_name == "create_what_if_scenario":
|
|
118
|
+
create_args = tc.parsed_arguments()
|
|
119
|
+
execute_result = None
|
|
120
|
+
elif tc.function_name == "execute_what_if_scenario" and tc.result:
|
|
121
|
+
execute_result = tc.parsed_result()
|
|
122
|
+
return create_args, execute_result
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
def _adjustments(create_args: dict | None) -> list[dict]:
|
|
126
|
+
"""Every adjustment across every scenario, flattened.
|
|
127
|
+
|
|
128
|
+
Scenario grouping does not matter to the checks below -- a fixture asserts that the
|
|
129
|
+
right measure was adjusted the right way, not which scenario label it landed under.
|
|
130
|
+
"""
|
|
131
|
+
scenarios = (create_args or {}).get("scenarios")
|
|
132
|
+
if not isinstance(scenarios, list):
|
|
133
|
+
return []
|
|
134
|
+
out: list[dict] = []
|
|
135
|
+
for scenario in scenarios:
|
|
136
|
+
if not isinstance(scenario, dict):
|
|
137
|
+
continue
|
|
138
|
+
out.extend(a for a in scenario.get("adjustments") or [] if isinstance(a, dict))
|
|
139
|
+
return out
|
|
140
|
+
|
|
141
|
+
|
|
142
|
+
def _maql_matches(actual_maql: str, expected: str | list[str]) -> bool:
|
|
143
|
+
"""Whether the adjustment matches any accepted expression, compared as MAQL.
|
|
144
|
+
|
|
145
|
+
Uses metric_skill's normalizer, so whitespace and casing differences do not decide a
|
|
146
|
+
verdict. A list is a candidate set: several expressions can be equally correct
|
|
147
|
+
adjustments (``* 1.1`` and ``* 1.10``, or a rewrite that reaches the same value).
|
|
148
|
+
"""
|
|
149
|
+
candidates = [expected] if isinstance(expected, str) else list(expected)
|
|
150
|
+
normalized = normalize_maql(actual_maql)
|
|
151
|
+
return any(normalized == normalize_maql(c) for c in candidates)
|
|
152
|
+
|
|
153
|
+
|
|
154
|
+
@dataclass
|
|
155
|
+
class WhatIfEvaluation:
|
|
156
|
+
"""Scores for a single what-if run.
|
|
157
|
+
|
|
158
|
+
``triggered``/``executed``/``success``/``turn_completed`` are the shared process checks.
|
|
159
|
+
``metric_correct``, ``maql_correct``, ``scenario_count_correct`` and ``baseline_correct``
|
|
160
|
+
are content checks, each True when the fixture did not pin it; ``asserted`` records
|
|
161
|
+
which ones it did, so a run that verified nothing is not reported as a full pass.
|
|
162
|
+
"""
|
|
163
|
+
|
|
164
|
+
triggered: bool
|
|
165
|
+
executed: bool
|
|
166
|
+
success: bool
|
|
167
|
+
turn_completed: bool
|
|
168
|
+
metric_correct: bool
|
|
169
|
+
maql_correct: bool
|
|
170
|
+
scenario_count_correct: bool
|
|
171
|
+
baseline_correct: bool
|
|
172
|
+
asserted: list[str] = field(default_factory=list)
|
|
173
|
+
disambiguated: bool = False
|
|
174
|
+
|
|
175
|
+
@property
|
|
176
|
+
def strict_pass(self) -> bool:
|
|
177
|
+
return all(
|
|
178
|
+
[
|
|
179
|
+
self.triggered,
|
|
180
|
+
self.executed,
|
|
181
|
+
self.success,
|
|
182
|
+
self.turn_completed,
|
|
183
|
+
self.metric_correct,
|
|
184
|
+
self.maql_correct,
|
|
185
|
+
self.scenario_count_correct,
|
|
186
|
+
self.baseline_correct,
|
|
187
|
+
]
|
|
188
|
+
)
|
|
189
|
+
|
|
190
|
+
|
|
191
|
+
@dataclass
|
|
192
|
+
class WhatIfRunResult:
|
|
193
|
+
"""Outcome of one run (one conversation, up to max_iterations messages)."""
|
|
194
|
+
|
|
195
|
+
conversation_id: str
|
|
196
|
+
evaluation: WhatIfEvaluation
|
|
197
|
+
actual_create_args: dict | None
|
|
198
|
+
actual_execute_result: dict | None
|
|
199
|
+
turn_wall_clock_sec: float | None = None
|
|
200
|
+
reasoning_steps: list[str] = field(default_factory=list)
|
|
201
|
+
response_id: str | None = None
|
|
202
|
+
tool_call_events: list[ToolCallEvent] = field(default_factory=list)
|
|
203
|
+
reasoning_step_events: list[ReasoningStepEvent] = field(default_factory=list)
|
|
204
|
+
|
|
205
|
+
|
|
206
|
+
@dataclass
|
|
207
|
+
class AgenticWhatIfSummary:
|
|
208
|
+
"""Aggregated outcome of K runs for one what-if item."""
|
|
209
|
+
|
|
210
|
+
run_results: list[WhatIfRunResult]
|
|
211
|
+
pass_at_k: bool
|
|
212
|
+
pass_power_k: bool
|
|
213
|
+
best: WhatIfRunResult
|
|
214
|
+
|
|
215
|
+
|
|
216
|
+
def _evaluate_run(
|
|
217
|
+
create_args: dict | None,
|
|
218
|
+
execute_result: dict | None,
|
|
219
|
+
expected_output: dict,
|
|
220
|
+
turn_completed: bool,
|
|
221
|
+
disambiguated: bool = False,
|
|
222
|
+
) -> WhatIfEvaluation:
|
|
223
|
+
triggered = create_args is not None
|
|
224
|
+
executed = execute_result is not None
|
|
225
|
+
success = executed and execute_result.get("success") is True
|
|
226
|
+
adjustments = _adjustments(create_args)
|
|
227
|
+
asserted: list[str] = []
|
|
228
|
+
|
|
229
|
+
expected_metric = expected_output.get("metric_id")
|
|
230
|
+
if not expected_metric:
|
|
231
|
+
wanted: set[str] | None = None
|
|
232
|
+
metric_correct = True
|
|
233
|
+
else:
|
|
234
|
+
asserted.append("metric_id")
|
|
235
|
+
wanted = {expected_metric} if isinstance(expected_metric, str) else set(expected_metric)
|
|
236
|
+
metric_correct = any(a.get("metric_id") in wanted for a in adjustments)
|
|
237
|
+
|
|
238
|
+
expected_maql = expected_output.get("scenario_maql")
|
|
239
|
+
if not expected_maql:
|
|
240
|
+
maql_correct = True
|
|
241
|
+
else:
|
|
242
|
+
asserted.append("scenario_maql")
|
|
243
|
+
# Only adjustments on the expected measure. Searching every adjustment independently
|
|
244
|
+
# lets a wrong adjustment on the right metric and a right adjustment on the wrong
|
|
245
|
+
# metric satisfy the two checks between them -- two failures scoring as a pass.
|
|
246
|
+
candidates = adjustments if wanted is None else [a for a in adjustments if a.get("metric_id") in wanted]
|
|
247
|
+
maql_correct = any(
|
|
248
|
+
isinstance(a.get("scenario_maql"), str) and _maql_matches(a["scenario_maql"], expected_maql)
|
|
249
|
+
for a in candidates
|
|
250
|
+
)
|
|
251
|
+
|
|
252
|
+
expected_scenarios = expected_output.get("scenarios")
|
|
253
|
+
if expected_scenarios is None:
|
|
254
|
+
scenario_count_correct = True
|
|
255
|
+
else:
|
|
256
|
+
asserted.append("scenarios")
|
|
257
|
+
actual = (create_args or {}).get("scenarios")
|
|
258
|
+
scenario_count_correct = isinstance(actual, list) and len(actual) == expected_scenarios
|
|
259
|
+
|
|
260
|
+
expected_baseline = expected_output.get("include_baseline")
|
|
261
|
+
if expected_baseline is None:
|
|
262
|
+
baseline_correct = True
|
|
263
|
+
else:
|
|
264
|
+
asserted.append("include_baseline")
|
|
265
|
+
# The tool defaults include_baseline to true, so an absent argument means true --
|
|
266
|
+
# `.get(..., True)` would be wrong only if the agent sent an explicit null, which
|
|
267
|
+
# the `is None` fallback below also treats as the default.
|
|
268
|
+
actual_baseline = (create_args or {}).get("include_baseline")
|
|
269
|
+
baseline_correct = (True if actual_baseline is None else bool(actual_baseline)) == bool(expected_baseline)
|
|
270
|
+
|
|
271
|
+
return WhatIfEvaluation(
|
|
272
|
+
triggered=triggered,
|
|
273
|
+
executed=executed,
|
|
274
|
+
success=success,
|
|
275
|
+
turn_completed=turn_completed,
|
|
276
|
+
metric_correct=metric_correct,
|
|
277
|
+
maql_correct=maql_correct,
|
|
278
|
+
scenario_count_correct=scenario_count_correct,
|
|
279
|
+
baseline_correct=baseline_correct,
|
|
280
|
+
asserted=asserted,
|
|
281
|
+
disambiguated=disambiguated,
|
|
282
|
+
)
|
|
283
|
+
|
|
284
|
+
|
|
285
|
+
def run_agentic_what_if(
|
|
286
|
+
host: str,
|
|
287
|
+
token: str,
|
|
288
|
+
workspace_id: str,
|
|
289
|
+
question: str,
|
|
290
|
+
expected_output: dict,
|
|
291
|
+
k: int = _DEFAULT_K,
|
|
292
|
+
max_iterations: int = _DEFAULT_MAX_ITERATIONS,
|
|
293
|
+
initial_conversation_id: str | None = None,
|
|
294
|
+
reasoning_effort: ReasoningEffort | None = None,
|
|
295
|
+
agent_id: str | None = None,
|
|
296
|
+
) -> AgenticWhatIfSummary:
|
|
297
|
+
"""Run the what-if agentic evaluation K times and return a summary.
|
|
298
|
+
|
|
299
|
+
A run ends when execute_what_if_scenario returns. Short of that it keeps sending
|
|
300
|
+
simulated replies up to ``max_iterations``, without trying to classify whether the
|
|
301
|
+
agent's text was a question: missing a genuine one hard-fails the run, while answering
|
|
302
|
+
a final answer costs one harmless extra turn.
|
|
303
|
+
"""
|
|
304
|
+
if k < 1:
|
|
305
|
+
raise ValueError(f"k must be >= 1, got {k}")
|
|
306
|
+
run_results: list[WhatIfRunResult] = []
|
|
307
|
+
client = ChatClient(
|
|
308
|
+
host=host, token=token, workspace_id=workspace_id, reasoning_effort=reasoning_effort, agent_id=agent_id
|
|
309
|
+
)
|
|
310
|
+
|
|
311
|
+
def _run_once(conv_id: str) -> WhatIfRunResult:
|
|
312
|
+
create_args: dict | None = None
|
|
313
|
+
execute_result: dict | None = None
|
|
314
|
+
turn_wall_clock_sec: float | None = None
|
|
315
|
+
turn_completed = False
|
|
316
|
+
disambiguated = False
|
|
317
|
+
current_question = question
|
|
318
|
+
reasoning_steps: list[str] = []
|
|
319
|
+
response_id: str | None = None
|
|
320
|
+
all_tool_call_events: list[ToolCallEvent] = []
|
|
321
|
+
all_reasoning_step_events: list[ReasoningStepEvent] = []
|
|
322
|
+
turn_offset = 0.0 # each turn's call_ts/ts restarts near 0 -- shift by prior turns' wall time
|
|
323
|
+
tool_index_offset = 0
|
|
324
|
+
reasoning_index_offset = 0
|
|
325
|
+
|
|
326
|
+
def _accumulate(result: ChatResult) -> None:
|
|
327
|
+
nonlocal turn_offset, tool_index_offset, reasoning_index_offset
|
|
328
|
+
turn_offset, tool_index_offset, reasoning_index_offset = shift_and_index_events(
|
|
329
|
+
result,
|
|
330
|
+
turn_offset=turn_offset,
|
|
331
|
+
tool_index_offset=tool_index_offset,
|
|
332
|
+
reasoning_index_offset=reasoning_index_offset,
|
|
333
|
+
)
|
|
334
|
+
all_tool_call_events.extend(result.tool_call_events or [])
|
|
335
|
+
all_reasoning_step_events.extend(result.reasoning_step_events or [])
|
|
336
|
+
|
|
337
|
+
for iteration in range(max_iterations):
|
|
338
|
+
try:
|
|
339
|
+
chat_result = client.send_message(conv_id, current_question)
|
|
340
|
+
except Exception as exc: # noqa: BLE001 -- end this run, not the whole item
|
|
341
|
+
_log.warning("What-if send_message failed for conversation %s: %s", conv_id, exc)
|
|
342
|
+
partial = getattr(exc, "partial_result", None)
|
|
343
|
+
if partial is not None:
|
|
344
|
+
reasoning_steps.extend(partial.reasoning_steps or [])
|
|
345
|
+
response_id = partial.response_id or response_id
|
|
346
|
+
_accumulate(partial)
|
|
347
|
+
create_args, execute_result = _extract_what_if_calls(all_tool_call_events)
|
|
348
|
+
turn_completed = False
|
|
349
|
+
break
|
|
350
|
+
reasoning_steps.extend(chat_result.reasoning_steps or [])
|
|
351
|
+
response_id = chat_result.response_id or response_id
|
|
352
|
+
_accumulate(chat_result)
|
|
353
|
+
# Over every turn so far, not just this one: the agent may build the spec on
|
|
354
|
+
# one turn and execute it on the next, and reading a single turn would drop the
|
|
355
|
+
# scenario the execution actually ran.
|
|
356
|
+
create_args, execute_result = _extract_what_if_calls(all_tool_call_events)
|
|
357
|
+
response_text = render_answer_text(chat_result)
|
|
358
|
+
turn_completed = chat_result.stream_ended and bool(response_text)
|
|
359
|
+
if execute_result is not None:
|
|
360
|
+
# The turn that ran the scenario, not an earlier disambiguation turn.
|
|
361
|
+
turn_wall_clock_sec = chat_result.turn_wall_clock_sec
|
|
362
|
+
break
|
|
363
|
+
if not response_text:
|
|
364
|
+
break
|
|
365
|
+
if iteration >= max_iterations - 1:
|
|
366
|
+
break
|
|
367
|
+
try:
|
|
368
|
+
current_question = generate_simulated_what_if_response(response_text, expected_output)
|
|
369
|
+
disambiguated = True
|
|
370
|
+
except Exception as exc: # noqa: BLE001 -- harness-side fault; end only this run
|
|
371
|
+
_log.warning("Simulated what-if user reply failed for conversation %s: %s", conv_id, exc)
|
|
372
|
+
break
|
|
373
|
+
|
|
374
|
+
return WhatIfRunResult(
|
|
375
|
+
conversation_id=conv_id,
|
|
376
|
+
evaluation=_evaluate_run(create_args, execute_result, expected_output, turn_completed, disambiguated),
|
|
377
|
+
actual_create_args=create_args,
|
|
378
|
+
actual_execute_result=execute_result,
|
|
379
|
+
turn_wall_clock_sec=turn_wall_clock_sec,
|
|
380
|
+
reasoning_steps=reasoning_steps,
|
|
381
|
+
response_id=response_id,
|
|
382
|
+
tool_call_events=all_tool_call_events,
|
|
383
|
+
reasoning_step_events=all_reasoning_step_events,
|
|
384
|
+
)
|
|
385
|
+
|
|
386
|
+
try:
|
|
387
|
+
conv_id_0 = initial_conversation_id if initial_conversation_id is not None else client.create_conversation()
|
|
388
|
+
try:
|
|
389
|
+
run_results.append(_run_once(conv_id_0))
|
|
390
|
+
finally:
|
|
391
|
+
if initial_conversation_id is None: # only delete conversations we created
|
|
392
|
+
client.delete_conversation(conv_id_0)
|
|
393
|
+
|
|
394
|
+
for _ in range(1, k):
|
|
395
|
+
conv_id = client.create_conversation()
|
|
396
|
+
try:
|
|
397
|
+
run_results.append(_run_once(conv_id))
|
|
398
|
+
finally:
|
|
399
|
+
client.delete_conversation(conv_id)
|
|
400
|
+
finally:
|
|
401
|
+
client.close()
|
|
402
|
+
|
|
403
|
+
pass_at_k = any(r.evaluation.strict_pass for r in run_results)
|
|
404
|
+
pass_power_k = all(r.evaluation.strict_pass for r in run_results)
|
|
405
|
+
best = max(
|
|
406
|
+
run_results,
|
|
407
|
+
key=lambda r: sum(
|
|
408
|
+
[
|
|
409
|
+
r.evaluation.triggered,
|
|
410
|
+
r.evaluation.executed,
|
|
411
|
+
r.evaluation.success,
|
|
412
|
+
r.evaluation.turn_completed,
|
|
413
|
+
r.evaluation.metric_correct,
|
|
414
|
+
r.evaluation.maql_correct,
|
|
415
|
+
r.evaluation.scenario_count_correct,
|
|
416
|
+
r.evaluation.baseline_correct,
|
|
417
|
+
]
|
|
418
|
+
),
|
|
419
|
+
)
|
|
420
|
+
return AgenticWhatIfSummary(
|
|
421
|
+
run_results=run_results,
|
|
422
|
+
pass_at_k=pass_at_k,
|
|
423
|
+
pass_power_k=pass_power_k,
|
|
424
|
+
best=best,
|
|
425
|
+
)
|
|
426
|
+
|
|
427
|
+
|
|
428
|
+
class WhatIfAssertionError(AgenticAssertionError):
|
|
429
|
+
"""Raised when a what-if evaluation fails."""
|
|
430
|
+
|
|
431
|
+
|
|
432
|
+
def _detail(best: WhatIfRunResult) -> dict[str, Any]:
|
|
433
|
+
ev = best.evaluation
|
|
434
|
+
adjustments = _adjustments(best.actual_create_args)
|
|
435
|
+
return {
|
|
436
|
+
"triggered": ev.triggered,
|
|
437
|
+
"executed": ev.executed,
|
|
438
|
+
"success": ev.success,
|
|
439
|
+
"turn_completed": ev.turn_completed,
|
|
440
|
+
"metric_correct": ev.metric_correct,
|
|
441
|
+
"maql_correct": ev.maql_correct,
|
|
442
|
+
"scenario_count_correct": ev.scenario_count_correct,
|
|
443
|
+
"baseline_correct": ev.baseline_correct,
|
|
444
|
+
# Which content checks the fixture pinned -- without it a run that verified nothing
|
|
445
|
+
# reads the same as one where everything matched.
|
|
446
|
+
"asserted": ev.asserted,
|
|
447
|
+
"disambiguated": ev.disambiguated,
|
|
448
|
+
"actual_adjustments": adjustments,
|
|
449
|
+
"actual_scenario_labels": [
|
|
450
|
+
s.get("label") for s in ((best.actual_create_args or {}).get("scenarios") or []) if isinstance(s, dict)
|
|
451
|
+
],
|
|
452
|
+
"actual_execute_result": best.actual_execute_result,
|
|
453
|
+
"latency_breakdown": build_latency_breakdown(best.tool_call_events, best.reasoning_step_events),
|
|
454
|
+
}
|
|
455
|
+
|
|
456
|
+
|
|
457
|
+
def evaluate_agentic_what_if(
|
|
458
|
+
host: str,
|
|
459
|
+
token: str,
|
|
460
|
+
workspace_id: str,
|
|
461
|
+
question: str,
|
|
462
|
+
expected_output: dict,
|
|
463
|
+
k: int = _DEFAULT_K,
|
|
464
|
+
max_iterations: int = _DEFAULT_MAX_ITERATIONS,
|
|
465
|
+
initial_conversation_id: str | None = None,
|
|
466
|
+
agent_id: str | None = None,
|
|
467
|
+
langfuse: object | None = None,
|
|
468
|
+
dataset_item_id: str = "",
|
|
469
|
+
dataset_name: str = "what_if_analysis",
|
|
470
|
+
run_timestamp: str | None = None,
|
|
471
|
+
model_version_override: str | None = None,
|
|
472
|
+
run_metadata_extra: dict | None = None,
|
|
473
|
+
reasoning_effort: ReasoningEffort | None = None,
|
|
474
|
+
submit_trace_link: SubmitTraceLink = run_trace_link_inline,
|
|
475
|
+
) -> AgenticEvalOutcome:
|
|
476
|
+
"""Run what-if evaluation, log to Langfuse, and raise WhatIfAssertionError on failure."""
|
|
477
|
+
langfuse, window_start = open_trace_window(langfuse)
|
|
478
|
+
summary = run_agentic_what_if(
|
|
479
|
+
host=host,
|
|
480
|
+
token=token,
|
|
481
|
+
workspace_id=workspace_id,
|
|
482
|
+
question=question,
|
|
483
|
+
expected_output=expected_output,
|
|
484
|
+
k=k,
|
|
485
|
+
max_iterations=max_iterations,
|
|
486
|
+
initial_conversation_id=initial_conversation_id,
|
|
487
|
+
reasoning_effort=reasoning_effort,
|
|
488
|
+
agent_id=agent_id,
|
|
489
|
+
)
|
|
490
|
+
|
|
491
|
+
if langfuse is not None and dataset_item_id:
|
|
492
|
+
# Pinned on the calling thread: a deferred poll must not widen its query window.
|
|
493
|
+
window_end = utc_now()
|
|
494
|
+
|
|
495
|
+
def _write_scores(ctx: RunTraceContext) -> None:
|
|
496
|
+
for run_idx, run in enumerate(summary.run_results):
|
|
497
|
+
pt = ctx.trace(run.conversation_id)
|
|
498
|
+
ev = run.evaluation
|
|
499
|
+
strict_checks = {
|
|
500
|
+
"what_if_triggered": ev.triggered,
|
|
501
|
+
"what_if_executed": ev.executed,
|
|
502
|
+
"what_if_success": ev.success,
|
|
503
|
+
"what_if_turn_completed": ev.turn_completed,
|
|
504
|
+
}
|
|
505
|
+
# Only the content checks the fixture actually pinned. An unasserted check
|
|
506
|
+
# is True internally so it cannot fail a run, but publishing that as a
|
|
507
|
+
# BOOLEAN 1 would claim the evaluator verified something it never looked at.
|
|
508
|
+
strict_checks.update(
|
|
509
|
+
{
|
|
510
|
+
key: value
|
|
511
|
+
for name, key, value in (
|
|
512
|
+
("metric_id", "what_if_metric_correct", ev.metric_correct),
|
|
513
|
+
("scenario_maql", "what_if_maql_correct", ev.maql_correct),
|
|
514
|
+
("scenarios", "what_if_scenario_count_correct", ev.scenario_count_correct),
|
|
515
|
+
("include_baseline", "what_if_baseline_correct", ev.baseline_correct),
|
|
516
|
+
)
|
|
517
|
+
if name in ev.asserted
|
|
518
|
+
}
|
|
519
|
+
)
|
|
520
|
+
with ctx.observe(pt, run_idx) as tid:
|
|
521
|
+
for score_name, value in strict_checks.items():
|
|
522
|
+
ctx.score(tid, name=score_name, value=float(value), data_type="BOOLEAN")
|
|
523
|
+
ctx.quality(
|
|
524
|
+
tid,
|
|
525
|
+
strict_checks=strict_checks,
|
|
526
|
+
# pt.latency covers the whole conversation, which is the item's real
|
|
527
|
+
# elapsed cost when the agent needed clarification turns to get
|
|
528
|
+
# there; turn_wall_clock_sec (the goal turn alone) is the fallback.
|
|
529
|
+
# This is what 7 of the 8 existing kinds do -- kda_skill is the
|
|
530
|
+
# outlier and documents its own reason. Cost is not gated on
|
|
531
|
+
# ev.triggered: a run that answered without ever reaching the tool
|
|
532
|
+
# still spent tokens, and hiding that understates what the item cost.
|
|
533
|
+
latency_sec=pt.latency if pt else run.turn_wall_clock_sec,
|
|
534
|
+
cost_usd=pt.total_cost if pt else None,
|
|
535
|
+
)
|
|
536
|
+
|
|
537
|
+
# Before the pass@K raise: a failing item's scores are the ones worth having.
|
|
538
|
+
submit_trace_scoring(
|
|
539
|
+
submit_trace_link,
|
|
540
|
+
RunIdentity(
|
|
541
|
+
host,
|
|
542
|
+
token,
|
|
543
|
+
workspace_id,
|
|
544
|
+
dataset_name,
|
|
545
|
+
run_timestamp,
|
|
546
|
+
model_version_override,
|
|
547
|
+
run_metadata_extra,
|
|
548
|
+
reasoning_effort,
|
|
549
|
+
),
|
|
550
|
+
langfuse=langfuse,
|
|
551
|
+
dataset_item_id=dataset_item_id,
|
|
552
|
+
conversation_ids=[r.conversation_id for r in summary.run_results],
|
|
553
|
+
window_start=window_start,
|
|
554
|
+
window_end=window_end,
|
|
555
|
+
suffix_runs=len(summary.run_results) > 1,
|
|
556
|
+
write_scores=_write_scores,
|
|
557
|
+
# The question this run answered, so a score is readable without resolving the
|
|
558
|
+
# conversation back to its item.
|
|
559
|
+
item_input=question,
|
|
560
|
+
)
|
|
561
|
+
|
|
562
|
+
best = summary.best
|
|
563
|
+
ev = best.evaluation
|
|
564
|
+
detail = _detail(best)
|
|
565
|
+
runs_passed = sum(1 for r in summary.run_results if r.evaluation.strict_pass)
|
|
566
|
+
|
|
567
|
+
if not summary.pass_at_k:
|
|
568
|
+
message = (
|
|
569
|
+
f"What-if assertion failed. strict_pass={ev.strict_pass} "
|
|
570
|
+
f"(triggered={ev.triggered}, executed={ev.executed}, success={ev.success}, "
|
|
571
|
+
f"turn_completed={ev.turn_completed}, metric_correct={ev.metric_correct}, "
|
|
572
|
+
f"maql_correct={ev.maql_correct}, scenario_count_correct={ev.scenario_count_correct}, "
|
|
573
|
+
f"baseline_correct={ev.baseline_correct}). "
|
|
574
|
+
f"Actual adjustments: {detail['actual_adjustments']}. "
|
|
575
|
+
f"Actual execute result: {best.actual_execute_result}."
|
|
576
|
+
)
|
|
577
|
+
exc = WhatIfAssertionError(message)
|
|
578
|
+
exc.reasoning_steps = best.reasoning_steps
|
|
579
|
+
exc.conversation_id = best.conversation_id
|
|
580
|
+
exc.response_id = best.response_id
|
|
581
|
+
exc.detail = detail
|
|
582
|
+
exc.runs_passed = runs_passed
|
|
583
|
+
exc.runs_effective = len(summary.run_results)
|
|
584
|
+
raise exc
|
|
585
|
+
|
|
586
|
+
return AgenticEvalOutcome(
|
|
587
|
+
runs_passed=runs_passed,
|
|
588
|
+
runs_effective=len(summary.run_results),
|
|
589
|
+
reasoning_steps=best.reasoning_steps,
|
|
590
|
+
conversation_id=best.conversation_id,
|
|
591
|
+
response_id=best.response_id,
|
|
592
|
+
detail=detail,
|
|
593
|
+
)
|
|
@@ -134,8 +134,8 @@ class ReasoningStepEvent(BaseModel):
|
|
|
134
134
|
|
|
135
135
|
# Reasoning summaries are full paragraphs, e.g. "**Identifying analytics needs**\n\nI'm
|
|
136
136
|
# analyzing..." -- using the whole thing as a latency_breakdown label would make every
|
|
137
|
-
# entry an unreadable wall of text.
|
|
138
|
-
#
|
|
137
|
+
# entry an unreadable wall of text. The bolded title is the summary's own heading, and
|
|
138
|
+
# downstream reporting keys off it for the same reason.
|
|
139
139
|
_REASONING_TITLE_RE = re.compile(r"^\*\*(.+?)\*\*")
|
|
140
140
|
_REASONING_LABEL_MAX_LEN = 60
|
|
141
141
|
|