gooddata-eval 1.75.1.dev1__tar.gz → 1.75.1.dev2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (130) hide show
  1. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/PKG-INFO +2 -2
  2. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/pyproject.toml +2 -2
  3. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/scripts/verify_guardrail_refusal_criteria.py +3 -1
  4. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/cli/agentic_runner.py +13 -0
  5. gooddata_eval-1.75.1.dev2/src/gooddata_eval/core/agentic/what_if.py +593 -0
  6. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/models.py +2 -2
  7. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/tests/test_agentic_runner.py +1 -0
  8. gooddata_eval-1.75.1.dev2/tests/test_agentic_what_if.py +439 -0
  9. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/tests/test_from_insights.py +2 -2
  10. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/tests/test_scoring.py +9 -9
  11. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/tests/test_trace_linker.py +1 -0
  12. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/.gitignore +0 -0
  13. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/AGENTS.md +0 -0
  14. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/CLAUDE.md +0 -0
  15. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/LICENSE.txt +0 -0
  16. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/Makefile +0 -0
  17. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/README.md +0 -0
  18. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/__init__.py +0 -0
  19. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/_version.py +0 -0
  20. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/cli/__init__.py +0 -0
  21. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/cli/main.py +0 -0
  22. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/__init__.py +0 -0
  23. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/_output.py +0 -0
  24. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/agentic/__init__.py +0 -0
  25. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/agentic/_catalog.py +0 -0
  26. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/agentic/_gate.py +0 -0
  27. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/agentic/_langfuse.py +0 -0
  28. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/agentic/_trace_linker.py +0 -0
  29. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/agentic/alert_skill.py +0 -0
  30. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/agentic/conversation.py +0 -0
  31. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/agentic/dashboard_skill.py +0 -0
  32. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/agentic/general_question.py +0 -0
  33. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/agentic/guardrail.py +0 -0
  34. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/agentic/kda_skill.py +0 -0
  35. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/agentic/metric_skill.py +0 -0
  36. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/agentic/search_tool.py +0 -0
  37. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/agentic/visualization.py +0 -0
  38. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/chat/__init__.py +0 -0
  39. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/chat/render.py +0 -0
  40. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/chat/sse_client.py +0 -0
  41. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/config.py +0 -0
  42. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/connection.py +0 -0
  43. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/dataset/__init__.py +0 -0
  44. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/dataset/from_insights.py +0 -0
  45. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/dataset/langfuse_source.py +0 -0
  46. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/dataset/local.py +0 -0
  47. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/evaluators/__init__.py +0 -0
  48. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/evaluators/_deep_subset.py +0 -0
  49. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/evaluators/_guardrail_criteria.py +0 -0
  50. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/evaluators/_llm_judge.py +0 -0
  51. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/evaluators/_maql.py +0 -0
  52. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/evaluators/_text_utils.py +0 -0
  53. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/evaluators/alert_skill.py +0 -0
  54. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/evaluators/base.py +0 -0
  55. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/evaluators/general_question.py +0 -0
  56. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/evaluators/guardrail.py +0 -0
  57. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/evaluators/metric_skill.py +0 -0
  58. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/evaluators/search_tool.py +0 -0
  59. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/evaluators/summary.py +0 -0
  60. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/evaluators/visualization.py +0 -0
  61. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/granularity.py +0 -0
  62. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/langfuse/__init__.py +0 -0
  63. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/langfuse/_env.py +0 -0
  64. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/langfuse/client.py +0 -0
  65. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/langfuse/experiment.py +0 -0
  66. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/langfuse/observations.py +0 -0
  67. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/langfuse/otlp.py +0 -0
  68. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/langfuse/sink.py +0 -0
  69. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/reporting/__init__.py +0 -0
  70. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/reporting/console.py +0 -0
  71. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/reporting/html_report.py +0 -0
  72. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/reporting/json_report.py +0 -0
  73. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/reporting/report_template.html +0 -0
  74. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/runner.py +0 -0
  75. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/scoring.py +0 -0
  76. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/summary/__init__.py +0 -0
  77. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/summary/http_client.py +0 -0
  78. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/timing.py +0 -0
  79. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/src/gooddata_eval/core/workspace.py +0 -0
  80. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/tests/__init__.py +0 -0
  81. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/tests/_fake_langfuse.py +0 -0
  82. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/tests/conftest.py +0 -0
  83. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/tests/fixtures/sample_dataset/metric_skill_create.json +0 -0
  84. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/tests/fixtures/sample_dataset/visualization_revenue.json +0 -0
  85. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/tests/fixtures/sse_visualization_stream.txt +0 -0
  86. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/tests/test_agentic_alert_skill.py +0 -0
  87. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/tests/test_agentic_conversation.py +0 -0
  88. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/tests/test_agentic_dashboard_skill.py +0 -0
  89. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/tests/test_agentic_gate.py +0 -0
  90. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/tests/test_agentic_general_question.py +0 -0
  91. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/tests/test_agentic_guardrail.py +0 -0
  92. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/tests/test_agentic_kda_skill.py +0 -0
  93. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/tests/test_agentic_langfuse_trace.py +0 -0
  94. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/tests/test_agentic_metric_skill.py +0 -0
  95. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/tests/test_agentic_observe_experiment.py +0 -0
  96. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/tests/test_agentic_run_context.py +0 -0
  97. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/tests/test_agentic_search_tool.py +0 -0
  98. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/tests/test_agentic_visualization.py +0 -0
  99. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/tests/test_alert_skill_evaluator.py +0 -0
  100. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/tests/test_chat_render.py +0 -0
  101. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/tests/test_cli.py +0 -0
  102. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/tests/test_connection.py +0 -0
  103. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/tests/test_deep_subset.py +0 -0
  104. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/tests/test_fake_langfuse.py +0 -0
  105. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/tests/test_guardrail_criteria.py +0 -0
  106. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/tests/test_html_report.py +0 -0
  107. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/tests/test_langfuse_client.py +0 -0
  108. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/tests/test_langfuse_e2e_fake_server.py +0 -0
  109. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/tests/test_langfuse_env.py +0 -0
  110. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/tests/test_langfuse_experiment.py +0 -0
  111. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/tests/test_langfuse_observations.py +0 -0
  112. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/tests/test_langfuse_otlp.py +0 -0
  113. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/tests/test_langfuse_sink.py +0 -0
  114. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/tests/test_langfuse_source.py +0 -0
  115. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/tests/test_llm_judge.py +0 -0
  116. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/tests/test_local_loader.py +0 -0
  117. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/tests/test_maql_normalize.py +0 -0
  118. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/tests/test_metric_skill_evaluator.py +0 -0
  119. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/tests/test_models.py +0 -0
  120. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/tests/test_reporting.py +0 -0
  121. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/tests/test_runner.py +0 -0
  122. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/tests/test_search_tool_evaluator.py +0 -0
  123. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/tests/test_sse_client.py +0 -0
  124. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/tests/test_summary_client.py +0 -0
  125. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/tests/test_summary_evaluator.py +0 -0
  126. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/tests/test_text_evaluators.py +0 -0
  127. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/tests/test_timing.py +0 -0
  128. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/tests/test_visualization_evaluator.py +0 -0
  129. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/tests/test_workspace.py +0 -0
  130. {gooddata_eval-1.75.1.dev1 → gooddata_eval-1.75.1.dev2}/tox.ini +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: gooddata-eval
3
- Version: 1.75.1.dev1
3
+ Version: 1.75.1.dev2
4
4
  Summary: Evaluate the GoodData AI agent against your own questions and models.
5
5
  Project-URL: Source, https://github.com/gooddata/gooddata-python-sdk
6
6
  Author-email: GoodData <support@gooddata.com>
@@ -17,7 +17,7 @@ Classifier: Topic :: Scientific/Engineering
17
17
  Classifier: Topic :: Software Development
18
18
  Classifier: Typing :: Typed
19
19
  Requires-Python: >=3.10
20
- Requires-Dist: gooddata-sdk~=1.75.1.dev1
20
+ Requires-Dist: gooddata-sdk~=1.75.1.dev2
21
21
  Requires-Dist: httpx<1.0,>=0.27
22
22
  Requires-Dist: orjson<4.0.0,>=3.9.15
23
23
  Requires-Dist: pydantic<3.0,>=2.6
@@ -1,7 +1,7 @@
1
1
  # (C) 2026 GoodData Corporation
2
2
  [project]
3
3
  name = "gooddata-eval"
4
- version = "1.75.1.dev1"
4
+ version = "1.75.1.dev2"
5
5
  description = "Evaluate the GoodData AI agent against your own questions and models."
6
6
  readme = "README.md"
7
7
  license = "MIT"
@@ -11,7 +11,7 @@ authors = [
11
11
  keywords = ["gooddata", "ai", "evaluation", "llm", "analytics", "cli"]
12
12
  requires-python = ">=3.10"
13
13
  dependencies = [
14
- "gooddata-sdk~=1.75.1.dev1",
14
+ "gooddata-sdk~=1.75.1.dev2",
15
15
  "httpx>=0.27,<1.0",
16
16
  "orjson>=3.9.15,<4.0.0",
17
17
  "pydantic>=2.6,<3.0",
@@ -39,7 +39,9 @@ import os
39
39
 
40
40
  from dotenv import load_dotenv
41
41
 
42
- load_dotenv("/Users/petertomko/gdc-mic-ai-evaluation/.env")
42
+ # Whatever .env the caller points at, defaulting to the working directory. It used to be an
43
+ # absolute path, which made the script runnable on exactly one machine.
44
+ load_dotenv(os.environ.get("GD_EVAL_ENV_FILE", ".env"))
43
45
 
44
46
  from gooddata_eval.core.agentic.guardrail import _GUARDRAIL_EVALUATION_STEPS # noqa: E402
45
47
  from gooddata_eval.core.evaluators._guardrail_criteria import ( # noqa: E402
@@ -20,6 +20,7 @@ from gooddata_eval.core.agentic.kda_skill import evaluate_agentic_kda_skill
20
20
  from gooddata_eval.core.agentic.metric_skill import evaluate_agentic_metric_skill
21
21
  from gooddata_eval.core.agentic.search_tool import evaluate_agentic_search_tool
22
22
  from gooddata_eval.core.agentic.visualization import evaluate_agentic_visualization
23
+ from gooddata_eval.core.agentic.what_if import evaluate_agentic_what_if
23
24
  from gooddata_eval.core.config import ReasoningEffort
24
25
  from gooddata_eval.core.models import AgenticEvalOutcome, CreatedVisualization, DatasetItem
25
26
  from gooddata_eval.core.runner import EvalReport, ItemReport
@@ -47,6 +48,7 @@ AGENTIC_TEST_KINDS = frozenset(
47
48
  "agentic_guardrail",
48
49
  "agentic_conversation",
49
50
  "agentic_kda_skill",
51
+ "agentic_what_if",
50
52
  }
51
53
  )
52
54
 
@@ -265,6 +267,17 @@ def _dispatch_agentic(
265
267
  agent_id=agent_id,
266
268
  **lf_kw,
267
269
  )
270
+ elif kind == "agentic_what_if":
271
+ return evaluate_agentic_what_if(
272
+ host=host,
273
+ token=token,
274
+ workspace_id=workspace_id,
275
+ question=item.question,
276
+ expected_output=eo if isinstance(eo, dict) else {},
277
+ k=k,
278
+ agent_id=agent_id,
279
+ **lf_kw,
280
+ )
268
281
  elif kind == "agentic_conversation":
269
282
  fixture_data = eo.get("fixture") or eo if isinstance(eo, dict) else {}
270
283
  return evaluate_agentic_conversation(
@@ -0,0 +1,593 @@
1
+ # (C) 2026 GoodData Corporation. All rights reserved.
2
+ """Agentic what-if-analysis skill evaluation runner.
3
+
4
+ The skill builds a scenario spec and executes it:
5
+
6
+ create_what_if_scenario(visualization_ref, scenarios[], include_baseline)
7
+ execute_what_if_scenario(scenario_ref) -> one result per scenario, plus the baseline
8
+
9
+ Each scenario carries adjustments of the form ``{metric_id, metric_type, scenario_maql}``,
10
+ where ``scenario_maql`` is the adjusted expression -- a 10% uplift on a revenue metric
11
+ defined as ``SELECT SUM({fact/price} * {fact/quantity})`` becomes
12
+ ``SELECT SUM({fact/price} * 1.10 * {fact/quantity})``.
13
+
14
+ That makes this the most checkable of the analysis skills: the adjustment is MAQL, and
15
+ MAQL already has a comparator here (``evaluators._maql.normalize_maql``, used by
16
+ metric_skill), so "did it apply the right adjustment" is answerable without a judge. What
17
+ the adjustment produced is not checked -- that is the platform's arithmetic, not the
18
+ agent's -- only that the agent asked for the right thing and the execution succeeded.
19
+ """
20
+
21
+ from __future__ import annotations
22
+
23
+ import logging
24
+ import os
25
+ from dataclasses import dataclass, field
26
+ from typing import Any
27
+
28
+ from gooddata_eval.core.agentic._trace_linker import (
29
+ RunIdentity,
30
+ RunTraceContext,
31
+ SubmitTraceLink,
32
+ open_trace_window,
33
+ run_trace_link_inline,
34
+ submit_trace_scoring,
35
+ utc_now,
36
+ )
37
+ from gooddata_eval.core.chat.render import render_answer_text
38
+ from gooddata_eval.core.chat.sse_client import ChatClient
39
+ from gooddata_eval.core.config import ReasoningEffort
40
+ from gooddata_eval.core.evaluators._maql import normalize_maql
41
+ from gooddata_eval.core.models import (
42
+ AgenticAssertionError,
43
+ AgenticEvalOutcome,
44
+ ChatResult,
45
+ ReasoningStepEvent,
46
+ ToolCallEvent,
47
+ build_latency_breakdown,
48
+ shift_and_index_events,
49
+ )
50
+
51
+ _log = logging.getLogger(__name__)
52
+
53
+ _DEFAULT_K = 1
54
+ # The agent asks which measure to adjust before building anything (observed live: "I need
55
+ # to confirm which 'Spend' calculation you want to adjust"), so the budget covers a couple
56
+ # of disambiguation rounds plus slack.
57
+ _DEFAULT_MAX_ITERATIONS = 4
58
+
59
+
60
+ def _build_clarification_prompt(agent_message: str, expected_output: dict) -> str:
61
+ """The simulated-user reply, mentioning only the hints the fixture actually supplies."""
62
+ hints: list[str] = []
63
+ metric = expected_output.get("metric_id")
64
+ if metric:
65
+ hints.append(f"the measure to adjust is '{metric}'")
66
+ change = expected_output.get("change")
67
+ if change:
68
+ hints.append(f"the adjustment is {change}")
69
+ period = expected_output.get("period")
70
+ if period:
71
+ hints.append(f"the time period is {period}")
72
+ reference = "; ".join(hints)
73
+ return (
74
+ f"You are simulating a user in a conversation with a BI assistant that runs what-if "
75
+ f"scenario analysis. The assistant asked: '{agent_message}'. "
76
+ + (f"For reference, {reference}. " if reference else "")
77
+ + "Reply briefly as the user, answering whichever of those the assistant actually asked about."
78
+ )
79
+
80
+
81
+ def generate_simulated_what_if_response(agent_message: str, expected_output: dict) -> str:
82
+ """Generate a user reply to keep the what-if conversation going (gpt-4o-mini).
83
+
84
+ Always OpenAI regardless of the workspace's own model: harness plumbing, not the system
85
+ under test.
86
+ """
87
+ try:
88
+ from openai import OpenAI # noqa: PLC0415
89
+ except ImportError as exc:
90
+ raise RuntimeError("openai package is required for generate_simulated_what_if_response") from exc
91
+
92
+ api_key = os.environ.get("OPENAI_API_KEY")
93
+ if not api_key:
94
+ raise OSError("OPENAI_API_KEY environment variable is not set")
95
+
96
+ client = OpenAI(api_key=api_key)
97
+ response = client.chat.completions.create(
98
+ model="gpt-4o-mini",
99
+ messages=[{"role": "user", "content": _build_clarification_prompt(agent_message, expected_output)}],
100
+ max_tokens=150,
101
+ temperature=0,
102
+ timeout=30,
103
+ )
104
+ return response.choices[0].message.content or "Please proceed with the most complete option."
105
+
106
+
107
+ def _extract_what_if_calls(tool_call_events: list[ToolCallEvent]) -> tuple[dict | None, dict | None]:
108
+ """Return (create_args, execute_result) for the LAST create/execute pair.
109
+
110
+ A new create_what_if_scenario clears any earlier execute result: that result belongs to
111
+ the spec it followed. Picking the last of each independently would score a fresh
112
+ scenario against a stale execution.
113
+ """
114
+ create_args: dict | None = None
115
+ execute_result: dict | None = None
116
+ for tc in tool_call_events:
117
+ if tc.function_name == "create_what_if_scenario":
118
+ create_args = tc.parsed_arguments()
119
+ execute_result = None
120
+ elif tc.function_name == "execute_what_if_scenario" and tc.result:
121
+ execute_result = tc.parsed_result()
122
+ return create_args, execute_result
123
+
124
+
125
+ def _adjustments(create_args: dict | None) -> list[dict]:
126
+ """Every adjustment across every scenario, flattened.
127
+
128
+ Scenario grouping does not matter to the checks below -- a fixture asserts that the
129
+ right measure was adjusted the right way, not which scenario label it landed under.
130
+ """
131
+ scenarios = (create_args or {}).get("scenarios")
132
+ if not isinstance(scenarios, list):
133
+ return []
134
+ out: list[dict] = []
135
+ for scenario in scenarios:
136
+ if not isinstance(scenario, dict):
137
+ continue
138
+ out.extend(a for a in scenario.get("adjustments") or [] if isinstance(a, dict))
139
+ return out
140
+
141
+
142
+ def _maql_matches(actual_maql: str, expected: str | list[str]) -> bool:
143
+ """Whether the adjustment matches any accepted expression, compared as MAQL.
144
+
145
+ Uses metric_skill's normalizer, so whitespace and casing differences do not decide a
146
+ verdict. A list is a candidate set: several expressions can be equally correct
147
+ adjustments (``* 1.1`` and ``* 1.10``, or a rewrite that reaches the same value).
148
+ """
149
+ candidates = [expected] if isinstance(expected, str) else list(expected)
150
+ normalized = normalize_maql(actual_maql)
151
+ return any(normalized == normalize_maql(c) for c in candidates)
152
+
153
+
154
+ @dataclass
155
+ class WhatIfEvaluation:
156
+ """Scores for a single what-if run.
157
+
158
+ ``triggered``/``executed``/``success``/``turn_completed`` are the shared process checks.
159
+ ``metric_correct``, ``maql_correct``, ``scenario_count_correct`` and ``baseline_correct``
160
+ are content checks, each True when the fixture did not pin it; ``asserted`` records
161
+ which ones it did, so a run that verified nothing is not reported as a full pass.
162
+ """
163
+
164
+ triggered: bool
165
+ executed: bool
166
+ success: bool
167
+ turn_completed: bool
168
+ metric_correct: bool
169
+ maql_correct: bool
170
+ scenario_count_correct: bool
171
+ baseline_correct: bool
172
+ asserted: list[str] = field(default_factory=list)
173
+ disambiguated: bool = False
174
+
175
+ @property
176
+ def strict_pass(self) -> bool:
177
+ return all(
178
+ [
179
+ self.triggered,
180
+ self.executed,
181
+ self.success,
182
+ self.turn_completed,
183
+ self.metric_correct,
184
+ self.maql_correct,
185
+ self.scenario_count_correct,
186
+ self.baseline_correct,
187
+ ]
188
+ )
189
+
190
+
191
+ @dataclass
192
+ class WhatIfRunResult:
193
+ """Outcome of one run (one conversation, up to max_iterations messages)."""
194
+
195
+ conversation_id: str
196
+ evaluation: WhatIfEvaluation
197
+ actual_create_args: dict | None
198
+ actual_execute_result: dict | None
199
+ turn_wall_clock_sec: float | None = None
200
+ reasoning_steps: list[str] = field(default_factory=list)
201
+ response_id: str | None = None
202
+ tool_call_events: list[ToolCallEvent] = field(default_factory=list)
203
+ reasoning_step_events: list[ReasoningStepEvent] = field(default_factory=list)
204
+
205
+
206
+ @dataclass
207
+ class AgenticWhatIfSummary:
208
+ """Aggregated outcome of K runs for one what-if item."""
209
+
210
+ run_results: list[WhatIfRunResult]
211
+ pass_at_k: bool
212
+ pass_power_k: bool
213
+ best: WhatIfRunResult
214
+
215
+
216
+ def _evaluate_run(
217
+ create_args: dict | None,
218
+ execute_result: dict | None,
219
+ expected_output: dict,
220
+ turn_completed: bool,
221
+ disambiguated: bool = False,
222
+ ) -> WhatIfEvaluation:
223
+ triggered = create_args is not None
224
+ executed = execute_result is not None
225
+ success = executed and execute_result.get("success") is True
226
+ adjustments = _adjustments(create_args)
227
+ asserted: list[str] = []
228
+
229
+ expected_metric = expected_output.get("metric_id")
230
+ if not expected_metric:
231
+ wanted: set[str] | None = None
232
+ metric_correct = True
233
+ else:
234
+ asserted.append("metric_id")
235
+ wanted = {expected_metric} if isinstance(expected_metric, str) else set(expected_metric)
236
+ metric_correct = any(a.get("metric_id") in wanted for a in adjustments)
237
+
238
+ expected_maql = expected_output.get("scenario_maql")
239
+ if not expected_maql:
240
+ maql_correct = True
241
+ else:
242
+ asserted.append("scenario_maql")
243
+ # Only adjustments on the expected measure. Searching every adjustment independently
244
+ # lets a wrong adjustment on the right metric and a right adjustment on the wrong
245
+ # metric satisfy the two checks between them -- two failures scoring as a pass.
246
+ candidates = adjustments if wanted is None else [a for a in adjustments if a.get("metric_id") in wanted]
247
+ maql_correct = any(
248
+ isinstance(a.get("scenario_maql"), str) and _maql_matches(a["scenario_maql"], expected_maql)
249
+ for a in candidates
250
+ )
251
+
252
+ expected_scenarios = expected_output.get("scenarios")
253
+ if expected_scenarios is None:
254
+ scenario_count_correct = True
255
+ else:
256
+ asserted.append("scenarios")
257
+ actual = (create_args or {}).get("scenarios")
258
+ scenario_count_correct = isinstance(actual, list) and len(actual) == expected_scenarios
259
+
260
+ expected_baseline = expected_output.get("include_baseline")
261
+ if expected_baseline is None:
262
+ baseline_correct = True
263
+ else:
264
+ asserted.append("include_baseline")
265
+ # The tool defaults include_baseline to true, so an absent argument means true --
266
+ # `.get(..., True)` would be wrong only if the agent sent an explicit null, which
267
+ # the `is None` fallback below also treats as the default.
268
+ actual_baseline = (create_args or {}).get("include_baseline")
269
+ baseline_correct = (True if actual_baseline is None else bool(actual_baseline)) == bool(expected_baseline)
270
+
271
+ return WhatIfEvaluation(
272
+ triggered=triggered,
273
+ executed=executed,
274
+ success=success,
275
+ turn_completed=turn_completed,
276
+ metric_correct=metric_correct,
277
+ maql_correct=maql_correct,
278
+ scenario_count_correct=scenario_count_correct,
279
+ baseline_correct=baseline_correct,
280
+ asserted=asserted,
281
+ disambiguated=disambiguated,
282
+ )
283
+
284
+
285
+ def run_agentic_what_if(
286
+ host: str,
287
+ token: str,
288
+ workspace_id: str,
289
+ question: str,
290
+ expected_output: dict,
291
+ k: int = _DEFAULT_K,
292
+ max_iterations: int = _DEFAULT_MAX_ITERATIONS,
293
+ initial_conversation_id: str | None = None,
294
+ reasoning_effort: ReasoningEffort | None = None,
295
+ agent_id: str | None = None,
296
+ ) -> AgenticWhatIfSummary:
297
+ """Run the what-if agentic evaluation K times and return a summary.
298
+
299
+ A run ends when execute_what_if_scenario returns. Short of that it keeps sending
300
+ simulated replies up to ``max_iterations``, without trying to classify whether the
301
+ agent's text was a question: missing a genuine one hard-fails the run, while answering
302
+ a final answer costs one harmless extra turn.
303
+ """
304
+ if k < 1:
305
+ raise ValueError(f"k must be >= 1, got {k}")
306
+ run_results: list[WhatIfRunResult] = []
307
+ client = ChatClient(
308
+ host=host, token=token, workspace_id=workspace_id, reasoning_effort=reasoning_effort, agent_id=agent_id
309
+ )
310
+
311
+ def _run_once(conv_id: str) -> WhatIfRunResult:
312
+ create_args: dict | None = None
313
+ execute_result: dict | None = None
314
+ turn_wall_clock_sec: float | None = None
315
+ turn_completed = False
316
+ disambiguated = False
317
+ current_question = question
318
+ reasoning_steps: list[str] = []
319
+ response_id: str | None = None
320
+ all_tool_call_events: list[ToolCallEvent] = []
321
+ all_reasoning_step_events: list[ReasoningStepEvent] = []
322
+ turn_offset = 0.0 # each turn's call_ts/ts restarts near 0 -- shift by prior turns' wall time
323
+ tool_index_offset = 0
324
+ reasoning_index_offset = 0
325
+
326
+ def _accumulate(result: ChatResult) -> None:
327
+ nonlocal turn_offset, tool_index_offset, reasoning_index_offset
328
+ turn_offset, tool_index_offset, reasoning_index_offset = shift_and_index_events(
329
+ result,
330
+ turn_offset=turn_offset,
331
+ tool_index_offset=tool_index_offset,
332
+ reasoning_index_offset=reasoning_index_offset,
333
+ )
334
+ all_tool_call_events.extend(result.tool_call_events or [])
335
+ all_reasoning_step_events.extend(result.reasoning_step_events or [])
336
+
337
+ for iteration in range(max_iterations):
338
+ try:
339
+ chat_result = client.send_message(conv_id, current_question)
340
+ except Exception as exc: # noqa: BLE001 -- end this run, not the whole item
341
+ _log.warning("What-if send_message failed for conversation %s: %s", conv_id, exc)
342
+ partial = getattr(exc, "partial_result", None)
343
+ if partial is not None:
344
+ reasoning_steps.extend(partial.reasoning_steps or [])
345
+ response_id = partial.response_id or response_id
346
+ _accumulate(partial)
347
+ create_args, execute_result = _extract_what_if_calls(all_tool_call_events)
348
+ turn_completed = False
349
+ break
350
+ reasoning_steps.extend(chat_result.reasoning_steps or [])
351
+ response_id = chat_result.response_id or response_id
352
+ _accumulate(chat_result)
353
+ # Over every turn so far, not just this one: the agent may build the spec on
354
+ # one turn and execute it on the next, and reading a single turn would drop the
355
+ # scenario the execution actually ran.
356
+ create_args, execute_result = _extract_what_if_calls(all_tool_call_events)
357
+ response_text = render_answer_text(chat_result)
358
+ turn_completed = chat_result.stream_ended and bool(response_text)
359
+ if execute_result is not None:
360
+ # The turn that ran the scenario, not an earlier disambiguation turn.
361
+ turn_wall_clock_sec = chat_result.turn_wall_clock_sec
362
+ break
363
+ if not response_text:
364
+ break
365
+ if iteration >= max_iterations - 1:
366
+ break
367
+ try:
368
+ current_question = generate_simulated_what_if_response(response_text, expected_output)
369
+ disambiguated = True
370
+ except Exception as exc: # noqa: BLE001 -- harness-side fault; end only this run
371
+ _log.warning("Simulated what-if user reply failed for conversation %s: %s", conv_id, exc)
372
+ break
373
+
374
+ return WhatIfRunResult(
375
+ conversation_id=conv_id,
376
+ evaluation=_evaluate_run(create_args, execute_result, expected_output, turn_completed, disambiguated),
377
+ actual_create_args=create_args,
378
+ actual_execute_result=execute_result,
379
+ turn_wall_clock_sec=turn_wall_clock_sec,
380
+ reasoning_steps=reasoning_steps,
381
+ response_id=response_id,
382
+ tool_call_events=all_tool_call_events,
383
+ reasoning_step_events=all_reasoning_step_events,
384
+ )
385
+
386
+ try:
387
+ conv_id_0 = initial_conversation_id if initial_conversation_id is not None else client.create_conversation()
388
+ try:
389
+ run_results.append(_run_once(conv_id_0))
390
+ finally:
391
+ if initial_conversation_id is None: # only delete conversations we created
392
+ client.delete_conversation(conv_id_0)
393
+
394
+ for _ in range(1, k):
395
+ conv_id = client.create_conversation()
396
+ try:
397
+ run_results.append(_run_once(conv_id))
398
+ finally:
399
+ client.delete_conversation(conv_id)
400
+ finally:
401
+ client.close()
402
+
403
+ pass_at_k = any(r.evaluation.strict_pass for r in run_results)
404
+ pass_power_k = all(r.evaluation.strict_pass for r in run_results)
405
+ best = max(
406
+ run_results,
407
+ key=lambda r: sum(
408
+ [
409
+ r.evaluation.triggered,
410
+ r.evaluation.executed,
411
+ r.evaluation.success,
412
+ r.evaluation.turn_completed,
413
+ r.evaluation.metric_correct,
414
+ r.evaluation.maql_correct,
415
+ r.evaluation.scenario_count_correct,
416
+ r.evaluation.baseline_correct,
417
+ ]
418
+ ),
419
+ )
420
+ return AgenticWhatIfSummary(
421
+ run_results=run_results,
422
+ pass_at_k=pass_at_k,
423
+ pass_power_k=pass_power_k,
424
+ best=best,
425
+ )
426
+
427
+
428
+ class WhatIfAssertionError(AgenticAssertionError):
429
+ """Raised when a what-if evaluation fails."""
430
+
431
+
432
+ def _detail(best: WhatIfRunResult) -> dict[str, Any]:
433
+ ev = best.evaluation
434
+ adjustments = _adjustments(best.actual_create_args)
435
+ return {
436
+ "triggered": ev.triggered,
437
+ "executed": ev.executed,
438
+ "success": ev.success,
439
+ "turn_completed": ev.turn_completed,
440
+ "metric_correct": ev.metric_correct,
441
+ "maql_correct": ev.maql_correct,
442
+ "scenario_count_correct": ev.scenario_count_correct,
443
+ "baseline_correct": ev.baseline_correct,
444
+ # Which content checks the fixture pinned -- without it a run that verified nothing
445
+ # reads the same as one where everything matched.
446
+ "asserted": ev.asserted,
447
+ "disambiguated": ev.disambiguated,
448
+ "actual_adjustments": adjustments,
449
+ "actual_scenario_labels": [
450
+ s.get("label") for s in ((best.actual_create_args or {}).get("scenarios") or []) if isinstance(s, dict)
451
+ ],
452
+ "actual_execute_result": best.actual_execute_result,
453
+ "latency_breakdown": build_latency_breakdown(best.tool_call_events, best.reasoning_step_events),
454
+ }
455
+
456
+
457
+ def evaluate_agentic_what_if(
458
+ host: str,
459
+ token: str,
460
+ workspace_id: str,
461
+ question: str,
462
+ expected_output: dict,
463
+ k: int = _DEFAULT_K,
464
+ max_iterations: int = _DEFAULT_MAX_ITERATIONS,
465
+ initial_conversation_id: str | None = None,
466
+ agent_id: str | None = None,
467
+ langfuse: object | None = None,
468
+ dataset_item_id: str = "",
469
+ dataset_name: str = "what_if_analysis",
470
+ run_timestamp: str | None = None,
471
+ model_version_override: str | None = None,
472
+ run_metadata_extra: dict | None = None,
473
+ reasoning_effort: ReasoningEffort | None = None,
474
+ submit_trace_link: SubmitTraceLink = run_trace_link_inline,
475
+ ) -> AgenticEvalOutcome:
476
+ """Run what-if evaluation, log to Langfuse, and raise WhatIfAssertionError on failure."""
477
+ langfuse, window_start = open_trace_window(langfuse)
478
+ summary = run_agentic_what_if(
479
+ host=host,
480
+ token=token,
481
+ workspace_id=workspace_id,
482
+ question=question,
483
+ expected_output=expected_output,
484
+ k=k,
485
+ max_iterations=max_iterations,
486
+ initial_conversation_id=initial_conversation_id,
487
+ reasoning_effort=reasoning_effort,
488
+ agent_id=agent_id,
489
+ )
490
+
491
+ if langfuse is not None and dataset_item_id:
492
+ # Pinned on the calling thread: a deferred poll must not widen its query window.
493
+ window_end = utc_now()
494
+
495
+ def _write_scores(ctx: RunTraceContext) -> None:
496
+ for run_idx, run in enumerate(summary.run_results):
497
+ pt = ctx.trace(run.conversation_id)
498
+ ev = run.evaluation
499
+ strict_checks = {
500
+ "what_if_triggered": ev.triggered,
501
+ "what_if_executed": ev.executed,
502
+ "what_if_success": ev.success,
503
+ "what_if_turn_completed": ev.turn_completed,
504
+ }
505
+ # Only the content checks the fixture actually pinned. An unasserted check
506
+ # is True internally so it cannot fail a run, but publishing that as a
507
+ # BOOLEAN 1 would claim the evaluator verified something it never looked at.
508
+ strict_checks.update(
509
+ {
510
+ key: value
511
+ for name, key, value in (
512
+ ("metric_id", "what_if_metric_correct", ev.metric_correct),
513
+ ("scenario_maql", "what_if_maql_correct", ev.maql_correct),
514
+ ("scenarios", "what_if_scenario_count_correct", ev.scenario_count_correct),
515
+ ("include_baseline", "what_if_baseline_correct", ev.baseline_correct),
516
+ )
517
+ if name in ev.asserted
518
+ }
519
+ )
520
+ with ctx.observe(pt, run_idx) as tid:
521
+ for score_name, value in strict_checks.items():
522
+ ctx.score(tid, name=score_name, value=float(value), data_type="BOOLEAN")
523
+ ctx.quality(
524
+ tid,
525
+ strict_checks=strict_checks,
526
+ # pt.latency covers the whole conversation, which is the item's real
527
+ # elapsed cost when the agent needed clarification turns to get
528
+ # there; turn_wall_clock_sec (the goal turn alone) is the fallback.
529
+ # This is what 7 of the 8 existing kinds do -- kda_skill is the
530
+ # outlier and documents its own reason. Cost is not gated on
531
+ # ev.triggered: a run that answered without ever reaching the tool
532
+ # still spent tokens, and hiding that understates what the item cost.
533
+ latency_sec=pt.latency if pt else run.turn_wall_clock_sec,
534
+ cost_usd=pt.total_cost if pt else None,
535
+ )
536
+
537
+ # Before the pass@K raise: a failing item's scores are the ones worth having.
538
+ submit_trace_scoring(
539
+ submit_trace_link,
540
+ RunIdentity(
541
+ host,
542
+ token,
543
+ workspace_id,
544
+ dataset_name,
545
+ run_timestamp,
546
+ model_version_override,
547
+ run_metadata_extra,
548
+ reasoning_effort,
549
+ ),
550
+ langfuse=langfuse,
551
+ dataset_item_id=dataset_item_id,
552
+ conversation_ids=[r.conversation_id for r in summary.run_results],
553
+ window_start=window_start,
554
+ window_end=window_end,
555
+ suffix_runs=len(summary.run_results) > 1,
556
+ write_scores=_write_scores,
557
+ # The question this run answered, so a score is readable without resolving the
558
+ # conversation back to its item.
559
+ item_input=question,
560
+ )
561
+
562
+ best = summary.best
563
+ ev = best.evaluation
564
+ detail = _detail(best)
565
+ runs_passed = sum(1 for r in summary.run_results if r.evaluation.strict_pass)
566
+
567
+ if not summary.pass_at_k:
568
+ message = (
569
+ f"What-if assertion failed. strict_pass={ev.strict_pass} "
570
+ f"(triggered={ev.triggered}, executed={ev.executed}, success={ev.success}, "
571
+ f"turn_completed={ev.turn_completed}, metric_correct={ev.metric_correct}, "
572
+ f"maql_correct={ev.maql_correct}, scenario_count_correct={ev.scenario_count_correct}, "
573
+ f"baseline_correct={ev.baseline_correct}). "
574
+ f"Actual adjustments: {detail['actual_adjustments']}. "
575
+ f"Actual execute result: {best.actual_execute_result}."
576
+ )
577
+ exc = WhatIfAssertionError(message)
578
+ exc.reasoning_steps = best.reasoning_steps
579
+ exc.conversation_id = best.conversation_id
580
+ exc.response_id = best.response_id
581
+ exc.detail = detail
582
+ exc.runs_passed = runs_passed
583
+ exc.runs_effective = len(summary.run_results)
584
+ raise exc
585
+
586
+ return AgenticEvalOutcome(
587
+ runs_passed=runs_passed,
588
+ runs_effective=len(summary.run_results),
589
+ reasoning_steps=best.reasoning_steps,
590
+ conversation_id=best.conversation_id,
591
+ response_id=best.response_id,
592
+ detail=detail,
593
+ )
@@ -134,8 +134,8 @@ class ReasoningStepEvent(BaseModel):
134
134
 
135
135
  # Reasoning summaries are full paragraphs, e.g. "**Identifying analytics needs**\n\nI'm
136
136
  # analyzing..." -- using the whole thing as a latency_breakdown label would make every
137
- # entry an unreadable wall of text. Same bolded-title convention this repo's own reasoning
138
- # tooling already keys off of (see gdc-mic-ai-evaluation's generate_dashboard_summary.py).
137
+ # entry an unreadable wall of text. The bolded title is the summary's own heading, and
138
+ # downstream reporting keys off it for the same reason.
139
139
  _REASONING_TITLE_RE = re.compile(r"^\*\*(.+?)\*\*")
140
140
  _REASONING_LABEL_MAX_LEN = 60
141
141