gooddata-eval 1.72.1.dev2__tar.gz → 1.72.1.dev4__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (89) hide show
  1. {gooddata_eval-1.72.1.dev2 → gooddata_eval-1.72.1.dev4}/PKG-INFO +3 -3
  2. {gooddata_eval-1.72.1.dev2 → gooddata_eval-1.72.1.dev4}/pyproject.toml +2 -2
  3. {gooddata_eval-1.72.1.dev2 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/cli/agentic_runner.py +12 -0
  4. {gooddata_eval-1.72.1.dev2 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/agentic/__init__.py +14 -0
  5. gooddata_eval-1.72.1.dev4/src/gooddata_eval/core/agentic/kda_skill.py +422 -0
  6. {gooddata_eval-1.72.1.dev2 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/chat/sse_client.py +60 -7
  7. {gooddata_eval-1.72.1.dev2 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/models.py +4 -0
  8. gooddata_eval-1.72.1.dev4/tests/test_agentic_kda_skill.py +1061 -0
  9. gooddata_eval-1.72.1.dev4/tests/test_agentic_langfuse_trace.py +26 -0
  10. {gooddata_eval-1.72.1.dev2 → gooddata_eval-1.72.1.dev4}/tests/test_sse_client.py +150 -0
  11. {gooddata_eval-1.72.1.dev2 → gooddata_eval-1.72.1.dev4}/.gitignore +0 -0
  12. {gooddata_eval-1.72.1.dev2 → gooddata_eval-1.72.1.dev4}/LICENSE.txt +0 -0
  13. {gooddata_eval-1.72.1.dev2 → gooddata_eval-1.72.1.dev4}/Makefile +0 -0
  14. {gooddata_eval-1.72.1.dev2 → gooddata_eval-1.72.1.dev4}/README.md +0 -0
  15. {gooddata_eval-1.72.1.dev2 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/__init__.py +0 -0
  16. {gooddata_eval-1.72.1.dev2 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/_version.py +0 -0
  17. {gooddata_eval-1.72.1.dev2 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/cli/__init__.py +0 -0
  18. {gooddata_eval-1.72.1.dev2 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/cli/main.py +0 -0
  19. {gooddata_eval-1.72.1.dev2 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/__init__.py +0 -0
  20. {gooddata_eval-1.72.1.dev2 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/agentic/_catalog.py +0 -0
  21. {gooddata_eval-1.72.1.dev2 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/agentic/_langfuse.py +0 -0
  22. {gooddata_eval-1.72.1.dev2 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/agentic/alert_skill.py +0 -0
  23. {gooddata_eval-1.72.1.dev2 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/agentic/conversation.py +0 -0
  24. {gooddata_eval-1.72.1.dev2 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/agentic/general_question.py +0 -0
  25. {gooddata_eval-1.72.1.dev2 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/agentic/guardrail.py +0 -0
  26. {gooddata_eval-1.72.1.dev2 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/agentic/metric_skill.py +0 -0
  27. {gooddata_eval-1.72.1.dev2 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/agentic/search_tool.py +0 -0
  28. {gooddata_eval-1.72.1.dev2 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/agentic/visualization.py +0 -0
  29. {gooddata_eval-1.72.1.dev2 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/chat/__init__.py +0 -0
  30. {gooddata_eval-1.72.1.dev2 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/config.py +0 -0
  31. {gooddata_eval-1.72.1.dev2 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/connection.py +0 -0
  32. {gooddata_eval-1.72.1.dev2 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/dataset/__init__.py +0 -0
  33. {gooddata_eval-1.72.1.dev2 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/dataset/langfuse_source.py +0 -0
  34. {gooddata_eval-1.72.1.dev2 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/dataset/local.py +0 -0
  35. {gooddata_eval-1.72.1.dev2 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/evaluators/__init__.py +0 -0
  36. {gooddata_eval-1.72.1.dev2 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/evaluators/_deep_subset.py +0 -0
  37. {gooddata_eval-1.72.1.dev2 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/evaluators/_llm_judge.py +0 -0
  38. {gooddata_eval-1.72.1.dev2 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/evaluators/_text_utils.py +0 -0
  39. {gooddata_eval-1.72.1.dev2 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/evaluators/alert_skill.py +0 -0
  40. {gooddata_eval-1.72.1.dev2 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/evaluators/base.py +0 -0
  41. {gooddata_eval-1.72.1.dev2 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/evaluators/general_question.py +0 -0
  42. {gooddata_eval-1.72.1.dev2 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/evaluators/guardrail.py +0 -0
  43. {gooddata_eval-1.72.1.dev2 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/evaluators/metric_skill.py +0 -0
  44. {gooddata_eval-1.72.1.dev2 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/evaluators/search_tool.py +0 -0
  45. {gooddata_eval-1.72.1.dev2 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/evaluators/summary.py +0 -0
  46. {gooddata_eval-1.72.1.dev2 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/evaluators/visualization.py +0 -0
  47. {gooddata_eval-1.72.1.dev2 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/langfuse/__init__.py +0 -0
  48. {gooddata_eval-1.72.1.dev2 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/langfuse/sink.py +0 -0
  49. {gooddata_eval-1.72.1.dev2 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/reporting/__init__.py +0 -0
  50. {gooddata_eval-1.72.1.dev2 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/reporting/console.py +0 -0
  51. {gooddata_eval-1.72.1.dev2 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/reporting/json_report.py +0 -0
  52. {gooddata_eval-1.72.1.dev2 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/runner.py +0 -0
  53. {gooddata_eval-1.72.1.dev2 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/scoring.py +0 -0
  54. {gooddata_eval-1.72.1.dev2 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/summary/__init__.py +0 -0
  55. {gooddata_eval-1.72.1.dev2 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/summary/http_client.py +0 -0
  56. {gooddata_eval-1.72.1.dev2 → gooddata_eval-1.72.1.dev4}/src/gooddata_eval/core/workspace.py +0 -0
  57. {gooddata_eval-1.72.1.dev2 → gooddata_eval-1.72.1.dev4}/tests/__init__.py +0 -0
  58. {gooddata_eval-1.72.1.dev2 → gooddata_eval-1.72.1.dev4}/tests/conftest.py +0 -0
  59. {gooddata_eval-1.72.1.dev2 → gooddata_eval-1.72.1.dev4}/tests/fixtures/sample_dataset/metric_skill_create.json +0 -0
  60. {gooddata_eval-1.72.1.dev2 → gooddata_eval-1.72.1.dev4}/tests/fixtures/sample_dataset/visualization_revenue.json +0 -0
  61. {gooddata_eval-1.72.1.dev2 → gooddata_eval-1.72.1.dev4}/tests/fixtures/sse_visualization_stream.txt +0 -0
  62. {gooddata_eval-1.72.1.dev2 → gooddata_eval-1.72.1.dev4}/tests/test_agentic_alert_skill.py +0 -0
  63. {gooddata_eval-1.72.1.dev2 → gooddata_eval-1.72.1.dev4}/tests/test_agentic_conversation.py +0 -0
  64. {gooddata_eval-1.72.1.dev2 → gooddata_eval-1.72.1.dev4}/tests/test_agentic_general_question.py +0 -0
  65. {gooddata_eval-1.72.1.dev2 → gooddata_eval-1.72.1.dev4}/tests/test_agentic_guardrail.py +0 -0
  66. {gooddata_eval-1.72.1.dev2 → gooddata_eval-1.72.1.dev4}/tests/test_agentic_metric_skill.py +0 -0
  67. {gooddata_eval-1.72.1.dev2 → gooddata_eval-1.72.1.dev4}/tests/test_agentic_run_context.py +0 -0
  68. {gooddata_eval-1.72.1.dev2 → gooddata_eval-1.72.1.dev4}/tests/test_agentic_search_tool.py +0 -0
  69. {gooddata_eval-1.72.1.dev2 → gooddata_eval-1.72.1.dev4}/tests/test_agentic_visualization.py +0 -0
  70. {gooddata_eval-1.72.1.dev2 → gooddata_eval-1.72.1.dev4}/tests/test_alert_skill_evaluator.py +0 -0
  71. {gooddata_eval-1.72.1.dev2 → gooddata_eval-1.72.1.dev4}/tests/test_cli.py +0 -0
  72. {gooddata_eval-1.72.1.dev2 → gooddata_eval-1.72.1.dev4}/tests/test_connection.py +0 -0
  73. {gooddata_eval-1.72.1.dev2 → gooddata_eval-1.72.1.dev4}/tests/test_deep_subset.py +0 -0
  74. {gooddata_eval-1.72.1.dev2 → gooddata_eval-1.72.1.dev4}/tests/test_langfuse_sink.py +0 -0
  75. {gooddata_eval-1.72.1.dev2 → gooddata_eval-1.72.1.dev4}/tests/test_langfuse_source.py +0 -0
  76. {gooddata_eval-1.72.1.dev2 → gooddata_eval-1.72.1.dev4}/tests/test_llm_judge.py +0 -0
  77. {gooddata_eval-1.72.1.dev2 → gooddata_eval-1.72.1.dev4}/tests/test_local_loader.py +0 -0
  78. {gooddata_eval-1.72.1.dev2 → gooddata_eval-1.72.1.dev4}/tests/test_metric_skill_evaluator.py +0 -0
  79. {gooddata_eval-1.72.1.dev2 → gooddata_eval-1.72.1.dev4}/tests/test_models.py +0 -0
  80. {gooddata_eval-1.72.1.dev2 → gooddata_eval-1.72.1.dev4}/tests/test_reporting.py +0 -0
  81. {gooddata_eval-1.72.1.dev2 → gooddata_eval-1.72.1.dev4}/tests/test_runner.py +0 -0
  82. {gooddata_eval-1.72.1.dev2 → gooddata_eval-1.72.1.dev4}/tests/test_scoring.py +0 -0
  83. {gooddata_eval-1.72.1.dev2 → gooddata_eval-1.72.1.dev4}/tests/test_search_tool_evaluator.py +0 -0
  84. {gooddata_eval-1.72.1.dev2 → gooddata_eval-1.72.1.dev4}/tests/test_summary_client.py +0 -0
  85. {gooddata_eval-1.72.1.dev2 → gooddata_eval-1.72.1.dev4}/tests/test_summary_evaluator.py +0 -0
  86. {gooddata_eval-1.72.1.dev2 → gooddata_eval-1.72.1.dev4}/tests/test_text_evaluators.py +0 -0
  87. {gooddata_eval-1.72.1.dev2 → gooddata_eval-1.72.1.dev4}/tests/test_visualization_evaluator.py +0 -0
  88. {gooddata_eval-1.72.1.dev2 → gooddata_eval-1.72.1.dev4}/tests/test_workspace.py +0 -0
  89. {gooddata_eval-1.72.1.dev2 → gooddata_eval-1.72.1.dev4}/tox.ini +0 -0
@@ -1,6 +1,6 @@
1
- Metadata-Version: 2.4
1
+ Metadata-Version: 2.5
2
2
  Name: gooddata-eval
3
- Version: 1.72.1.dev2
3
+ Version: 1.72.1.dev4
4
4
  Summary: Evaluate the GoodData AI agent against your own questions and models.
5
5
  Project-URL: Source, https://github.com/gooddata/gooddata-python-sdk
6
6
  Author-email: GoodData <support@gooddata.com>
@@ -17,7 +17,7 @@ Classifier: Topic :: Scientific/Engineering
17
17
  Classifier: Topic :: Software Development
18
18
  Classifier: Typing :: Typed
19
19
  Requires-Python: >=3.10
20
- Requires-Dist: gooddata-sdk~=1.72.1.dev2
20
+ Requires-Dist: gooddata-sdk~=1.72.1.dev4
21
21
  Requires-Dist: httpx<1.0,>=0.27
22
22
  Requires-Dist: orjson<4.0.0,>=3.9.15
23
23
  Requires-Dist: pydantic<3.0,>=2.6
@@ -1,7 +1,7 @@
1
1
  # (C) 2026 GoodData Corporation
2
2
  [project]
3
3
  name = "gooddata-eval"
4
- version = "1.72.1.dev2"
4
+ version = "1.72.1.dev4"
5
5
  description = "Evaluate the GoodData AI agent against your own questions and models."
6
6
  readme = "README.md"
7
7
  license = "MIT"
@@ -11,7 +11,7 @@ authors = [
11
11
  keywords = ["gooddata", "ai", "evaluation", "llm", "analytics", "cli"]
12
12
  requires-python = ">=3.10"
13
13
  dependencies = [
14
- "gooddata-sdk~=1.72.1.dev2",
14
+ "gooddata-sdk~=1.72.1.dev4",
15
15
  "httpx>=0.27,<1.0",
16
16
  "orjson>=3.9.15,<4.0.0",
17
17
  "pydantic>=2.6,<3.0",
@@ -11,6 +11,7 @@ from gooddata_eval.core.agentic.alert_skill import evaluate_agentic_alert_skill
11
11
  from gooddata_eval.core.agentic.conversation import ConversationFixture, evaluate_agentic_conversation
12
12
  from gooddata_eval.core.agentic.general_question import evaluate_agentic_general_question
13
13
  from gooddata_eval.core.agentic.guardrail import evaluate_agentic_guardrail
14
+ from gooddata_eval.core.agentic.kda_skill import evaluate_agentic_kda_skill
14
15
  from gooddata_eval.core.agentic.metric_skill import evaluate_agentic_metric_skill
15
16
  from gooddata_eval.core.agentic.search_tool import evaluate_agentic_search_tool
16
17
  from gooddata_eval.core.agentic.visualization import evaluate_agentic_visualization
@@ -38,6 +39,7 @@ AGENTIC_TEST_KINDS = frozenset(
38
39
  "agentic_general_question",
39
40
  "agentic_guardrail",
40
41
  "agentic_conversation",
42
+ "agentic_kda_skill",
41
43
  }
42
44
  )
43
45
 
@@ -159,6 +161,16 @@ def _dispatch_agentic(
159
161
  k=k,
160
162
  **lf_kw,
161
163
  )
164
+ elif kind == "agentic_kda_skill":
165
+ evaluate_agentic_kda_skill(
166
+ host=host,
167
+ token=token,
168
+ workspace_id=workspace_id,
169
+ question=item.question,
170
+ expected_output=eo if isinstance(eo, dict) else {},
171
+ k=k,
172
+ **lf_kw,
173
+ )
162
174
  elif kind == "agentic_conversation":
163
175
  fixture_data = eo.get("fixture") or eo if isinstance(eo, dict) else {}
164
176
  evaluate_agentic_conversation(
@@ -30,6 +30,14 @@ from gooddata_eval.core.agentic.guardrail import (
30
30
  evaluate_agentic_guardrail,
31
31
  run_agentic_guardrail,
32
32
  )
33
+ from gooddata_eval.core.agentic.kda_skill import (
34
+ AgenticKdaSummary,
35
+ KdaEvaluation,
36
+ KdaRunResult,
37
+ KdaSkillAssertionError,
38
+ evaluate_agentic_kda_skill,
39
+ run_agentic_kda_skill,
40
+ )
33
41
  from gooddata_eval.core.agentic.metric_skill import (
34
42
  AgenticMetricSummary,
35
43
  MetricRunResult,
@@ -56,6 +64,7 @@ __all__ = [
56
64
  "AgenticAlertSummary",
57
65
  "AgenticGeneralQuestionSummary",
58
66
  "AgenticGuardrailSummary",
67
+ "AgenticKdaSummary",
59
68
  "AgenticMetricSummary",
60
69
  "AgenticSearchSummary",
61
70
  "AgenticRunSummary",
@@ -69,6 +78,9 @@ __all__ = [
69
78
  "GeneralQuestionResult",
70
79
  "GuardrailAssertionError",
71
80
  "GuardrailResult",
81
+ "KdaEvaluation",
82
+ "KdaRunResult",
83
+ "KdaSkillAssertionError",
72
84
  "MetricRunResult",
73
85
  "MetricSkillAssertionError",
74
86
  "RunResult",
@@ -81,6 +93,7 @@ __all__ = [
81
93
  "evaluate_agentic_conversation",
82
94
  "evaluate_agentic_general_question",
83
95
  "evaluate_agentic_guardrail",
96
+ "evaluate_agentic_kda_skill",
84
97
  "evaluate_agentic_metric_skill",
85
98
  "evaluate_agentic_search_tool",
86
99
  "evaluate_agentic_visualization",
@@ -88,6 +101,7 @@ __all__ = [
88
101
  "run_agentic_conversation",
89
102
  "run_agentic_general_question",
90
103
  "run_agentic_guardrail",
104
+ "run_agentic_kda_skill",
91
105
  "run_agentic_metric_skill",
92
106
  "run_agentic_search_tool",
93
107
  "run_agentic_visualization",
@@ -0,0 +1,422 @@
1
+ # (C) 2026 GoodData Corporation. All rights reserved.
2
+ """Agentic KDA (Key Driver Analysis)-skill evaluation runner."""
3
+
4
+ from __future__ import annotations
5
+
6
+ import logging
7
+ import os
8
+ from dataclasses import dataclass
9
+
10
+ from gooddata_eval.core.chat.sse_client import ChatClient
11
+ from gooddata_eval.core.config import ReasoningEffort
12
+ from gooddata_eval.core.models import ToolCallEvent
13
+
14
+ _log = logging.getLogger(__name__)
15
+
16
+ _DEFAULT_K = 1
17
+ # Disambiguation safety net only (create+execute always run together in the same turn) --
18
+ # 3 real questions' worth (metric, period, +1 slack) since a simulated reply is now sent on
19
+ # every non-final turn (see run_agentic_kda_skill), not just ones classified as a question.
20
+ _DEFAULT_MAX_ITERATIONS = 4
21
+
22
+
23
+ def _build_period_hint(expected_output: dict) -> str | None:
24
+ """Build a period hint from whichever of expected_output's Date Attribute/Analyzed
25
+ Period/Reference Period are present -- a question about only one of them (e.g. "which
26
+ date dimension?") must still get an answerable hint, not None just because the other
27
+ two are absent.
28
+ """
29
+ date_attr = expected_output.get("Date Attribute")
30
+ analyzed = expected_output.get("Analyzed Period")
31
+ reference_period = expected_output.get("Reference Period")
32
+ if not (date_attr or analyzed or reference_period):
33
+ return None
34
+ parts = []
35
+ if date_attr:
36
+ parts.append(date_attr)
37
+ if analyzed and reference_period:
38
+ parts.append(f"comparing {analyzed} to {reference_period}")
39
+ elif analyzed:
40
+ parts.append(f"period {analyzed}")
41
+ elif reference_period:
42
+ parts.append(f"compared to {reference_period}")
43
+ return ", ".join(parts)
44
+
45
+
46
+ def _build_clarification_prompt(
47
+ agent_message: str, measure_candidates: dict | list[dict] | None, period_hint: str | None
48
+ ) -> str:
49
+ """Build the simulated-user prompt, referencing only whatever candidates/period-hint
50
+ are actually usable -- an empty/None candidate must drop the "acceptable metric/fact"
51
+ clause entirely rather than assert a literal "None" as if it were a real option.
52
+ """
53
+ candidates = [
54
+ c for c in (measure_candidates if isinstance(measure_candidates, list) else [measure_candidates]) if c
55
+ ]
56
+ reference = ""
57
+ if candidates:
58
+ candidate_desc = "; or ".join(
59
+ f"{c.get('type')} '{c.get('id')}'" + (f" (aggregation {c['aggregation']})" if c.get("aggregation") else "")
60
+ for c in candidates
61
+ )
62
+ reference = f"an acceptable metric/fact is {candidate_desc}"
63
+ if period_hint:
64
+ reference = (
65
+ f"{reference}; the intended time period is {period_hint}"
66
+ if reference
67
+ else f"the intended time period is {period_hint}"
68
+ )
69
+ return (
70
+ f"You are simulating a user in a conversation with a BI assistant that runs key driver "
71
+ f"analysis. The assistant asked: '{agent_message}'. "
72
+ + (f"For reference, {reference}. " if reference else "")
73
+ + "Reply briefly as the user, answering whichever of those the assistant actually asked about."
74
+ )
75
+
76
+
77
+ def generate_simulated_kda_response(
78
+ agent_message: str,
79
+ measure_candidates: dict | list[dict] | None,
80
+ period_hint: str | None = None,
81
+ ) -> str:
82
+ """Generate a user reply to keep the KDA-skill conversation going (gpt-4o-mini).
83
+
84
+ Called on any turn that didn't trigger KDA, whatever the agent's response actually
85
+ said -- most often a clarifying question about the measure, the period, or both, so
86
+ both are given as reference and the reply answers whichever was actually asked.
87
+ Scope only needs KDA to trigger, not the resulting measure/period to be exactly
88
+ right. Always OpenAI regardless of the combo's own provider -- this is
89
+ test-harness plumbing, not the system under test.
90
+ """
91
+ try:
92
+ from openai import OpenAI # noqa: PLC0415
93
+ except ImportError as exc:
94
+ raise RuntimeError("openai package is required for generate_simulated_kda_response") from exc
95
+
96
+ api_key = os.environ.get("OPENAI_API_KEY")
97
+ if not api_key:
98
+ raise OSError("OPENAI_API_KEY environment variable is not set")
99
+
100
+ client = OpenAI(api_key=api_key)
101
+ prompt = _build_clarification_prompt(agent_message, measure_candidates, period_hint)
102
+ response = client.chat.completions.create(
103
+ model="gpt-4o-mini",
104
+ messages=[{"role": "user", "content": prompt}],
105
+ max_tokens=150,
106
+ timeout=30,
107
+ )
108
+ return response.choices[0].message.content or "Please proceed with either option."
109
+
110
+
111
+ def _extract_kda_calls(tool_call_events: list[ToolCallEvent]) -> tuple[dict | None, dict | None]:
112
+ """Return (create_args, execute_result) for the LAST create/execute pair in this turn's
113
+ tool calls -- not the last create and last execute picked independently. A new create
114
+ call clears any earlier execute_result -- it belongs to the create it followed, not to
115
+ this one.
116
+ """
117
+ create_args: dict | None = None
118
+ execute_result: dict | None = None
119
+ for tc in tool_call_events:
120
+ if tc.function_name == "create_key_driver_analysis":
121
+ create_args = tc.parsed_arguments()
122
+ execute_result = None
123
+ elif tc.function_name == "execute_key_driver_analysis" and tc.result:
124
+ execute_result = tc.parsed_result()
125
+ return create_args, execute_result
126
+
127
+
128
+ @dataclass
129
+ class KdaEvaluation:
130
+ """Evaluation scores for a single KDA-skill run.
131
+
132
+ Scope: asserts only that the KDA process runs to completion -- the tool chain
133
+ triggers, executes successfully, and the chat turn ends cleanly with a non-empty
134
+ response (``turn_completed`` requires both gen-ai's stream-ended signal and a
135
+ non-empty ``text_response`` -- a stream that ends cleanly but delivers nothing to the
136
+ user isn't a completed turn either).
137
+ """
138
+
139
+ triggered: bool
140
+ executed: bool
141
+ success: bool
142
+ turn_completed: bool
143
+ disambiguated: bool = False
144
+
145
+ @property
146
+ def strict_pass(self) -> bool:
147
+ return all([self.triggered, self.executed, self.success, self.turn_completed])
148
+
149
+
150
+ @dataclass
151
+ class KdaRunResult:
152
+ """Outcome of one run (one conversation, up to max_iterations messages) for a KDA case."""
153
+
154
+ conversation_id: str
155
+ evaluation: KdaEvaluation
156
+ actual_create_args: dict | None
157
+ actual_execute_result: dict | None
158
+ # Wall-clock time of the turn that called create (None if create never happened) --
159
+ # not any earlier disambiguation turn. See run_agentic_kda_skill's _run_once.
160
+ turn_wall_clock_sec: float | None = None
161
+
162
+
163
+ @dataclass
164
+ class AgenticKdaSummary:
165
+ """Aggregated outcome of K runs for a KDA case."""
166
+
167
+ run_results: list[KdaRunResult]
168
+ pass_at_k: bool
169
+ pass_power_k: bool
170
+ best: KdaRunResult
171
+
172
+
173
+ def _evaluate_run(
174
+ create_args: dict | None,
175
+ execute_result: dict | None,
176
+ turn_completed: bool,
177
+ disambiguated: bool = False,
178
+ ) -> KdaEvaluation:
179
+ triggered = create_args is not None
180
+ executed = execute_result is not None
181
+ success = executed and execute_result.get("success") is True
182
+ return KdaEvaluation(
183
+ triggered=triggered,
184
+ executed=executed,
185
+ success=success,
186
+ turn_completed=turn_completed,
187
+ disambiguated=disambiguated,
188
+ )
189
+
190
+
191
+ def run_agentic_kda_skill(
192
+ host: str,
193
+ token: str,
194
+ workspace_id: str,
195
+ question: str,
196
+ expected_output: dict,
197
+ k: int = _DEFAULT_K,
198
+ max_iterations: int = _DEFAULT_MAX_ITERATIONS,
199
+ initial_conversation_id: str | None = None,
200
+ reasoning_effort: ReasoningEffort | None = None,
201
+ ) -> AgenticKdaSummary:
202
+ """Run the KDA-skill agentic evaluation K times and return a summary.
203
+
204
+ Each run is normally one message, one turn -- create and execute are always called
205
+ together in the same turn (the skill's own system prompt: "NO confirmation needed").
206
+ A run only extends past turn 1, up to ``max_iterations``, when the agent's response
207
+ has no create call and isn't empty; a simulated user reply is then always sent, with
208
+ no attempt to classify whether the text was actually asking for input (matching
209
+ visualization.py/alert_skill.py's own break conditions) -- missing a genuine
210
+ clarifying question hard-fails the run, while sending one after an unrecognized final
211
+ answer only costs one harmless extra turn, so the asymmetry favors never guessing.
212
+ """
213
+ if k < 1:
214
+ # k=0 or negative would otherwise silently run once, indistinguishable from k=1.
215
+ raise ValueError(f"k must be >= 1, got {k}")
216
+ run_results: list[KdaRunResult] = []
217
+ client = ChatClient(host=host, token=token, workspace_id=workspace_id, reasoning_effort=reasoning_effort)
218
+
219
+ def _run_once(conv_id: str) -> KdaRunResult:
220
+ create_args: dict | None = None
221
+ execute_result: dict | None = None
222
+ turn_wall_clock_sec: float | None = None
223
+ turn_completed = False
224
+ disambiguated = False
225
+ current_question = question
226
+
227
+ for iteration in range(max_iterations):
228
+ try:
229
+ chat_result = client.send_message(conv_id, current_question)
230
+ except Exception as exc: # noqa: BLE001 -- end this run, not the whole assertion
231
+ _log.warning("KDA send_message failed for conversation %s: %s", conv_id, exc)
232
+ partial = getattr(exc, "partial_result", None)
233
+ if partial is not None:
234
+ create_args, execute_result = _extract_kda_calls(partial.tool_call_events or [])
235
+ if create_args is not None:
236
+ turn_wall_clock_sec = partial.turn_wall_clock_sec
237
+ turn_completed = False
238
+ break
239
+ create_args, execute_result = _extract_kda_calls(chat_result.tool_call_events or [])
240
+ response_text = (chat_result.text_response or "").strip()
241
+ turn_completed = chat_result.stream_ended and bool(response_text)
242
+ if create_args is not None:
243
+ # This turn's own time -- the turn that called create, not any earlier
244
+ # disambiguation turn or the simulated-reply generation. create and execute
245
+ # are always called together in the same turn (or not at all), so this is
246
+ # final either way -- execute_result may still be None (e.g. the skill's
247
+ # execute tool isn't available at all when data-sharing is off for the org).
248
+ turn_wall_clock_sec = chat_result.turn_wall_clock_sec
249
+ break
250
+ if not response_text:
251
+ break
252
+ if iteration >= max_iterations - 1:
253
+ break
254
+ # No text classification -- matches visualization.py/alert_skill.py: break only on
255
+ # the goal signal (create_args set) or an empty response, otherwise always send a
256
+ # simulated reply. A false positive (agent had already given a final answer) costs
257
+ # one harmless extra turn; a false negative (missing a genuine clarifying question)
258
+ # would hard-fail the run, so the asymmetry favors never trying to tell them apart.
259
+ measure_candidates = expected_output.get("Measure") if isinstance(expected_output, dict) else None
260
+ period_hint = _build_period_hint(expected_output) if isinstance(expected_output, dict) else None
261
+ try:
262
+ current_question = generate_simulated_kda_response(response_text, measure_candidates, period_hint)
263
+ disambiguated = True
264
+ except Exception as exc: # noqa: BLE001 -- safety net, not the assertion; end only this run
265
+ _log.warning("Simulated KDA user reply failed for conversation %s: %s", conv_id, exc)
266
+ break
267
+
268
+ ev = _evaluate_run(create_args, execute_result, turn_completed, disambiguated)
269
+ return KdaRunResult(
270
+ conversation_id=conv_id,
271
+ evaluation=ev,
272
+ actual_create_args=create_args,
273
+ actual_execute_result=execute_result,
274
+ turn_wall_clock_sec=turn_wall_clock_sec,
275
+ )
276
+
277
+ try:
278
+ conv_id_0 = initial_conversation_id if initial_conversation_id is not None else client.create_conversation()
279
+ try:
280
+ run_results.append(_run_once(conv_id_0))
281
+ finally:
282
+ if initial_conversation_id is None: # only delete conversations we created
283
+ client.delete_conversation(conv_id_0)
284
+
285
+ for _ in range(1, k):
286
+ conv_id = client.create_conversation()
287
+ try:
288
+ run_results.append(_run_once(conv_id))
289
+ finally:
290
+ client.delete_conversation(conv_id)
291
+ finally:
292
+ client.close()
293
+
294
+ pass_at_k = any(r.evaluation.strict_pass for r in run_results)
295
+ pass_power_k = all(r.evaluation.strict_pass for r in run_results)
296
+ best = max(
297
+ run_results,
298
+ key=lambda r: sum(
299
+ [r.evaluation.triggered, r.evaluation.executed, r.evaluation.success, r.evaluation.turn_completed]
300
+ ),
301
+ )
302
+ return AgenticKdaSummary(
303
+ run_results=run_results,
304
+ pass_at_k=pass_at_k,
305
+ pass_power_k=pass_power_k,
306
+ best=best,
307
+ )
308
+
309
+
310
+ class KdaSkillAssertionError(AssertionError):
311
+ """Raised when a KDA-skill evaluation fails."""
312
+
313
+ __tracebackhide__ = True
314
+
315
+
316
+ def evaluate_agentic_kda_skill(
317
+ host: str,
318
+ token: str,
319
+ workspace_id: str,
320
+ question: str,
321
+ expected_output: dict,
322
+ k: int = _DEFAULT_K,
323
+ max_iterations: int = _DEFAULT_MAX_ITERATIONS,
324
+ initial_conversation_id: str | None = None,
325
+ langfuse: object | None = None,
326
+ dataset_item_id: str = "",
327
+ dataset_name: str = "kda_skill",
328
+ run_timestamp: str | None = None,
329
+ model_version_override: str | None = None,
330
+ run_metadata_extra: dict | None = None,
331
+ reasoning_effort: ReasoningEffort | None = None,
332
+ ) -> None:
333
+ """Run KDA-skill evaluation, log to Langfuse, and raise KdaSkillAssertionError on failure."""
334
+ from datetime import datetime as _dt # noqa: PLC0415
335
+ from datetime import timezone as _tz # noqa: PLC0415
336
+
337
+ from gooddata_eval.core.agentic._langfuse import try_make_langfuse_client # noqa: PLC0415
338
+
339
+ if langfuse is None:
340
+ langfuse = try_make_langfuse_client()
341
+ window_start = _dt.now(_tz.utc)
342
+ summary = run_agentic_kda_skill(
343
+ host=host,
344
+ token=token,
345
+ workspace_id=workspace_id,
346
+ question=question,
347
+ expected_output=expected_output,
348
+ k=k,
349
+ max_iterations=max_iterations,
350
+ initial_conversation_id=initial_conversation_id,
351
+ reasoning_effort=reasoning_effort,
352
+ )
353
+
354
+ if langfuse is not None and dataset_item_id:
355
+ from gooddata_eval.core.agentic._langfuse import ( # noqa: PLC0415
356
+ build_run_context,
357
+ find_traces_per_conversation,
358
+ log_quality_and_value_scores,
359
+ observe,
360
+ score_safe,
361
+ )
362
+
363
+ run_name_base, run_metadata = build_run_context(
364
+ host,
365
+ token,
366
+ workspace_id,
367
+ dataset_name,
368
+ run_timestamp,
369
+ model_version_override,
370
+ run_metadata_extra,
371
+ reasoning_effort,
372
+ )
373
+ # No custom selector -- same default (max-latency) as every other skill; harmless
374
+ # here since latency comes from run.turn_wall_clock_sec below, not this trace.
375
+ traces_by_conv = find_traces_per_conversation(
376
+ langfuse,
377
+ [r.conversation_id for r in summary.run_results],
378
+ window_start,
379
+ )
380
+ suffix_needed = len(summary.run_results) > 1
381
+ for run_idx, run in enumerate(summary.run_results):
382
+ pt = traces_by_conv.get(run.conversation_id)
383
+ run_name = f"{run_name_base}_run{run_idx}" if suffix_needed else run_name_base
384
+ ev = run.evaluation
385
+ # Gates strict_pass -- current scope is completion only (see KdaEvaluation docstring).
386
+ strict_checks = {
387
+ "kda_triggered": ev.triggered,
388
+ "kda_executed": ev.executed,
389
+ "kda_success": ev.success,
390
+ "kda_turn_completed": ev.turn_completed,
391
+ }
392
+ # Not pt.latency: pt can be any trace of the conversation, not necessarily the KDA turn.
393
+ turn_wall_clock_sec = run.turn_wall_clock_sec
394
+ _log.info("[kda-report] %s: strict_pass=%s latency_sec=%s", run_name, ev.strict_pass, turn_wall_clock_sec)
395
+ with observe(langfuse, pt.id if pt else None, dataset_item_id, run_name, run_metadata) as tid:
396
+ for score_name, value in strict_checks.items():
397
+ score_safe(langfuse, tid, name=score_name, value=float(value), data_type="BOOLEAN")
398
+ score_safe(langfuse, tid, name="kda_disambiguated", value=float(ev.disambiguated), data_type="BOOLEAN")
399
+ if turn_wall_clock_sec is not None:
400
+ # combo_report.py reads this score directly -- no trace re-resolution needed.
401
+ score_safe(
402
+ langfuse, tid, name="kda_turn_wall_clock_sec", value=turn_wall_clock_sec, data_type="NUMERIC"
403
+ )
404
+ log_quality_and_value_scores(
405
+ langfuse,
406
+ tid,
407
+ strict_checks=strict_checks,
408
+ latency_sec=turn_wall_clock_sec,
409
+ cost_usd=pt.total_cost if pt and ev.triggered else None,
410
+ )
411
+
412
+ if not summary.pass_at_k:
413
+ best = summary.best
414
+ ev = best.evaluation
415
+ message = (
416
+ f"KDA skill assertion failed. strict_pass={ev.strict_pass} "
417
+ f"(triggered={ev.triggered}, executed={ev.executed}, "
418
+ f"success={ev.success}, turn_completed={ev.turn_completed}). "
419
+ f"Actual create args: {best.actual_create_args}. "
420
+ f"Actual execute result: {best.actual_execute_result}."
421
+ )
422
+ raise KdaSkillAssertionError(message)
@@ -28,18 +28,34 @@ from gooddata_eval.core.models import ChatResult, DatasetItem
28
28
  _log = logging.getLogger(__name__)
29
29
 
30
30
  SSE_DATA_PREFIX = "data: "
31
+ SSE_EVENT_PREFIX = "event: "
32
+ # gen-ai's last event, only if at least one item was already emitted (conversations_controller.py).
33
+ _RESPONSE_ENDED_EVENT = "response_ended"
31
34
 
32
35
  _RETRYABLE_STATUS_CODES: frozenset[int] = frozenset({429, 502, 503, 504})
33
36
  _METADATA_SYNC_MARKER = "METADATA_SYNC_IN_PROGRESS"
34
37
 
35
38
 
36
39
  class ChatError(RuntimeError):
37
- """Non-retryable error reported by the chat SSE stream."""
40
+ """Non-retryable error reported by the chat SSE stream.
38
41
 
39
- def __init__(self, message: str, *, status_code: int | None = None, detail: str | None = None) -> None:
42
+ ``partial_result`` carries whatever the accumulator captured before the error fired
43
+ (tool calls included). Callers must not assume it's complete -- fields like
44
+ ``stream_ended`` reflect the state at the moment of the error, not a finished turn.
45
+ """
46
+
47
+ def __init__(
48
+ self,
49
+ message: str,
50
+ *,
51
+ status_code: int | None = None,
52
+ detail: str | None = None,
53
+ partial_result: ChatResult | None = None,
54
+ ) -> None:
40
55
  super().__init__(message)
41
56
  self.status_code = status_code
42
57
  self.detail = detail
58
+ self.partial_result = partial_result
43
59
 
44
60
 
45
61
  class TransientChatError(ChatError):
@@ -109,6 +125,7 @@ class _SseAccumulator:
109
125
  reasoning_steps: list[dict[str, Any]] = field(default_factory=list)
110
126
  adhoc_viz_args: list[dict[str, Any]] = field(default_factory=list)
111
127
  response_id: str | None = None
128
+ stream_ended: bool = False
112
129
 
113
130
 
114
131
  def _handle_text(content: dict[str, Any], acc: _SseAccumulator) -> None:
@@ -187,15 +204,37 @@ def _build_chat_result(acc: _SseAccumulator) -> ChatResult:
187
204
  }
188
205
  result = ChatResult.model_validate(payload)
189
206
  result.response_id = acc.response_id
207
+ result.stream_ended = acc.stream_ended
190
208
  return result
191
209
 
192
210
 
193
211
  def parse_sse_lines(lines: Iterable[str]) -> ChatResult:
194
212
  """Parse an SSE stream (iterable of decoded lines) into a ChatResult."""
195
213
  acc = _SseAccumulator()
196
- for raw_line in lines:
214
+ current_event = "message" # SSE default in the absence of an explicit "event: " line
215
+ it = iter(lines)
216
+ while True:
217
+ try:
218
+ raw_line = next(it)
219
+ except StopIteration:
220
+ break
221
+ except Exception as exc:
222
+ # Only a transport-level failure (e.g. connection drop mid-stream) is rescued
223
+ # here -- a bug in the processing below must propagate uncaught, not get
224
+ # mislabeled as a network error.
225
+ raise ChatError(f"SSE stream error: {exc}", partial_result=_build_chat_result(acc)) from exc
197
226
  line = raw_line.decode("utf-8") if isinstance(raw_line, bytes) else raw_line
198
- if not line or line.startswith("event: ") or not line.startswith(SSE_DATA_PREFIX):
227
+ if not line:
228
+ current_event = "message" # blank line ends one event block per the SSE spec
229
+ continue
230
+ if line.startswith(SSE_EVENT_PREFIX):
231
+ current_event = line[len(SSE_EVENT_PREFIX) :].strip()
232
+ if current_event == _RESPONSE_ENDED_EVENT:
233
+ acc.stream_ended = True
234
+ continue
235
+ if not line.startswith(SSE_DATA_PREFIX):
236
+ continue
237
+ if current_event == _RESPONSE_ENDED_EVENT:
199
238
  continue
200
239
  data_str = line[len(SSE_DATA_PREFIX) :]
201
240
  if _METADATA_SYNC_MARKER in data_str:
@@ -203,6 +242,7 @@ def parse_sse_lines(lines: Iterable[str]) -> ChatResult:
203
242
  f"SSE transient error: {_METADATA_SYNC_MARKER}",
204
243
  status_code=None,
205
244
  detail=None,
245
+ partial_result=_build_chat_result(acc),
206
246
  )
207
247
  try:
208
248
  event_data = json.loads(data_str)
@@ -213,8 +253,10 @@ def parse_sse_lines(lines: Iterable[str]) -> ChatResult:
213
253
  detail = event_data.get("detail")
214
254
  message = f"SSE error {code}: {detail}"
215
255
  if code in _RETRYABLE_STATUS_CODES:
216
- raise TransientChatError(message, status_code=code, detail=detail)
217
- raise ChatError(message, status_code=code, detail=detail)
256
+ raise TransientChatError(
257
+ message, status_code=code, detail=detail, partial_result=_build_chat_result(acc)
258
+ )
259
+ raise ChatError(message, status_code=code, detail=detail, partial_result=_build_chat_result(acc))
218
260
  if event_data.get("responseId") and not acc.response_id:
219
261
  acc.response_id = event_data["responseId"]
220
262
  item = event_data.get("item")
@@ -293,9 +335,20 @@ class ChatClient:
293
335
  body["options"] = {"reasoningEffort": self._reasoning_effort}
294
336
 
295
337
  def _do() -> ChatResult:
338
+ # Set fresh on every retry attempt (before opening this attempt's stream, so its
339
+ # own connection setup time counts) -- excludes not just the sleep backoff between
340
+ # attempts, but the entire duration of any earlier failed attempt.
341
+ t0 = time.monotonic()
296
342
  with self._client.stream("POST", url, json=body, headers=headers) as resp:
297
343
  resp.raise_for_status()
298
- return parse_sse_lines(resp.iter_lines())
344
+ try:
345
+ result = parse_sse_lines(resp.iter_lines())
346
+ except ChatError as exc:
347
+ if exc.partial_result is not None:
348
+ exc.partial_result.turn_wall_clock_sec = time.monotonic() - t0
349
+ raise
350
+ result.turn_wall_clock_sec = time.monotonic() - t0
351
+ return result
299
352
 
300
353
  return _retry_transient(_do, is_retryable=_is_retryable_exc)
301
354
 
@@ -100,6 +100,10 @@ class ChatResult(BaseModel):
100
100
  reasoning_step_count: int = Field(default=0, alias="reasoningStepCount")
101
101
  conversation_id: str | None = Field(default=None, alias="conversationId")
102
102
  response_id: str | None = Field(default=None, alias="responseId")
103
+ # True once gen-ai's response_ended event arrived.
104
+ stream_ended: bool = False
105
+ # Wall-clock seconds for the whole chat turn, timed by the client.
106
+ turn_wall_clock_sec: float | None = None
103
107
 
104
108
 
105
109
  class SummaryInput(BaseModel):