gooddata-eval 1.72.1.dev1__tar.gz → 1.72.1.dev3__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (89) hide show
  1. {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/PKG-INFO +3 -3
  2. {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/pyproject.toml +2 -2
  3. {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/src/gooddata_eval/cli/agentic_runner.py +12 -0
  4. {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/src/gooddata_eval/core/agentic/__init__.py +14 -0
  5. gooddata_eval-1.72.1.dev3/src/gooddata_eval/core/agentic/kda_skill.py +382 -0
  6. {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/src/gooddata_eval/core/chat/sse_client.py +60 -7
  7. {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/src/gooddata_eval/core/models.py +4 -0
  8. gooddata_eval-1.72.1.dev3/tests/test_agentic_kda_skill.py +930 -0
  9. gooddata_eval-1.72.1.dev3/tests/test_agentic_langfuse_trace.py +26 -0
  10. {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/tests/test_sse_client.py +150 -0
  11. {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/.gitignore +0 -0
  12. {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/LICENSE.txt +0 -0
  13. {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/Makefile +0 -0
  14. {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/README.md +0 -0
  15. {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/src/gooddata_eval/__init__.py +0 -0
  16. {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/src/gooddata_eval/_version.py +0 -0
  17. {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/src/gooddata_eval/cli/__init__.py +0 -0
  18. {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/src/gooddata_eval/cli/main.py +0 -0
  19. {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/src/gooddata_eval/core/__init__.py +0 -0
  20. {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/src/gooddata_eval/core/agentic/_catalog.py +0 -0
  21. {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/src/gooddata_eval/core/agentic/_langfuse.py +0 -0
  22. {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/src/gooddata_eval/core/agentic/alert_skill.py +0 -0
  23. {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/src/gooddata_eval/core/agentic/conversation.py +0 -0
  24. {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/src/gooddata_eval/core/agentic/general_question.py +0 -0
  25. {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/src/gooddata_eval/core/agentic/guardrail.py +0 -0
  26. {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/src/gooddata_eval/core/agentic/metric_skill.py +0 -0
  27. {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/src/gooddata_eval/core/agentic/search_tool.py +0 -0
  28. {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/src/gooddata_eval/core/agentic/visualization.py +0 -0
  29. {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/src/gooddata_eval/core/chat/__init__.py +0 -0
  30. {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/src/gooddata_eval/core/config.py +0 -0
  31. {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/src/gooddata_eval/core/connection.py +0 -0
  32. {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/src/gooddata_eval/core/dataset/__init__.py +0 -0
  33. {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/src/gooddata_eval/core/dataset/langfuse_source.py +0 -0
  34. {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/src/gooddata_eval/core/dataset/local.py +0 -0
  35. {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/src/gooddata_eval/core/evaluators/__init__.py +0 -0
  36. {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/src/gooddata_eval/core/evaluators/_deep_subset.py +0 -0
  37. {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/src/gooddata_eval/core/evaluators/_llm_judge.py +0 -0
  38. {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/src/gooddata_eval/core/evaluators/_text_utils.py +0 -0
  39. {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/src/gooddata_eval/core/evaluators/alert_skill.py +0 -0
  40. {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/src/gooddata_eval/core/evaluators/base.py +0 -0
  41. {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/src/gooddata_eval/core/evaluators/general_question.py +0 -0
  42. {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/src/gooddata_eval/core/evaluators/guardrail.py +0 -0
  43. {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/src/gooddata_eval/core/evaluators/metric_skill.py +0 -0
  44. {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/src/gooddata_eval/core/evaluators/search_tool.py +0 -0
  45. {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/src/gooddata_eval/core/evaluators/summary.py +0 -0
  46. {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/src/gooddata_eval/core/evaluators/visualization.py +0 -0
  47. {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/src/gooddata_eval/core/langfuse/__init__.py +0 -0
  48. {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/src/gooddata_eval/core/langfuse/sink.py +0 -0
  49. {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/src/gooddata_eval/core/reporting/__init__.py +0 -0
  50. {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/src/gooddata_eval/core/reporting/console.py +0 -0
  51. {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/src/gooddata_eval/core/reporting/json_report.py +0 -0
  52. {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/src/gooddata_eval/core/runner.py +0 -0
  53. {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/src/gooddata_eval/core/scoring.py +0 -0
  54. {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/src/gooddata_eval/core/summary/__init__.py +0 -0
  55. {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/src/gooddata_eval/core/summary/http_client.py +0 -0
  56. {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/src/gooddata_eval/core/workspace.py +0 -0
  57. {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/tests/__init__.py +0 -0
  58. {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/tests/conftest.py +0 -0
  59. {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/tests/fixtures/sample_dataset/metric_skill_create.json +0 -0
  60. {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/tests/fixtures/sample_dataset/visualization_revenue.json +0 -0
  61. {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/tests/fixtures/sse_visualization_stream.txt +0 -0
  62. {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/tests/test_agentic_alert_skill.py +0 -0
  63. {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/tests/test_agentic_conversation.py +0 -0
  64. {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/tests/test_agentic_general_question.py +0 -0
  65. {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/tests/test_agentic_guardrail.py +0 -0
  66. {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/tests/test_agentic_metric_skill.py +0 -0
  67. {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/tests/test_agentic_run_context.py +0 -0
  68. {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/tests/test_agentic_search_tool.py +0 -0
  69. {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/tests/test_agentic_visualization.py +0 -0
  70. {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/tests/test_alert_skill_evaluator.py +0 -0
  71. {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/tests/test_cli.py +0 -0
  72. {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/tests/test_connection.py +0 -0
  73. {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/tests/test_deep_subset.py +0 -0
  74. {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/tests/test_langfuse_sink.py +0 -0
  75. {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/tests/test_langfuse_source.py +0 -0
  76. {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/tests/test_llm_judge.py +0 -0
  77. {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/tests/test_local_loader.py +0 -0
  78. {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/tests/test_metric_skill_evaluator.py +0 -0
  79. {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/tests/test_models.py +0 -0
  80. {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/tests/test_reporting.py +0 -0
  81. {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/tests/test_runner.py +0 -0
  82. {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/tests/test_scoring.py +0 -0
  83. {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/tests/test_search_tool_evaluator.py +0 -0
  84. {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/tests/test_summary_client.py +0 -0
  85. {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/tests/test_summary_evaluator.py +0 -0
  86. {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/tests/test_text_evaluators.py +0 -0
  87. {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/tests/test_visualization_evaluator.py +0 -0
  88. {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/tests/test_workspace.py +0 -0
  89. {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/tox.ini +0 -0
@@ -1,6 +1,6 @@
1
- Metadata-Version: 2.4
1
+ Metadata-Version: 2.5
2
2
  Name: gooddata-eval
3
- Version: 1.72.1.dev1
3
+ Version: 1.72.1.dev3
4
4
  Summary: Evaluate the GoodData AI agent against your own questions and models.
5
5
  Project-URL: Source, https://github.com/gooddata/gooddata-python-sdk
6
6
  Author-email: GoodData <support@gooddata.com>
@@ -17,7 +17,7 @@ Classifier: Topic :: Scientific/Engineering
17
17
  Classifier: Topic :: Software Development
18
18
  Classifier: Typing :: Typed
19
19
  Requires-Python: >=3.10
20
- Requires-Dist: gooddata-sdk~=1.72.1.dev1
20
+ Requires-Dist: gooddata-sdk~=1.72.1.dev3
21
21
  Requires-Dist: httpx<1.0,>=0.27
22
22
  Requires-Dist: orjson<4.0.0,>=3.9.15
23
23
  Requires-Dist: pydantic<3.0,>=2.6
@@ -1,7 +1,7 @@
1
1
  # (C) 2026 GoodData Corporation
2
2
  [project]
3
3
  name = "gooddata-eval"
4
- version = "1.72.1.dev1"
4
+ version = "1.72.1.dev3"
5
5
  description = "Evaluate the GoodData AI agent against your own questions and models."
6
6
  readme = "README.md"
7
7
  license = "MIT"
@@ -11,7 +11,7 @@ authors = [
11
11
  keywords = ["gooddata", "ai", "evaluation", "llm", "analytics", "cli"]
12
12
  requires-python = ">=3.10"
13
13
  dependencies = [
14
- "gooddata-sdk~=1.72.1.dev1",
14
+ "gooddata-sdk~=1.72.1.dev3",
15
15
  "httpx>=0.27,<1.0",
16
16
  "orjson>=3.9.15,<4.0.0",
17
17
  "pydantic>=2.6,<3.0",
@@ -11,6 +11,7 @@ from gooddata_eval.core.agentic.alert_skill import evaluate_agentic_alert_skill
11
11
  from gooddata_eval.core.agentic.conversation import ConversationFixture, evaluate_agentic_conversation
12
12
  from gooddata_eval.core.agentic.general_question import evaluate_agentic_general_question
13
13
  from gooddata_eval.core.agentic.guardrail import evaluate_agentic_guardrail
14
+ from gooddata_eval.core.agentic.kda_skill import evaluate_agentic_kda_skill
14
15
  from gooddata_eval.core.agentic.metric_skill import evaluate_agentic_metric_skill
15
16
  from gooddata_eval.core.agentic.search_tool import evaluate_agentic_search_tool
16
17
  from gooddata_eval.core.agentic.visualization import evaluate_agentic_visualization
@@ -38,6 +39,7 @@ AGENTIC_TEST_KINDS = frozenset(
38
39
  "agentic_general_question",
39
40
  "agentic_guardrail",
40
41
  "agentic_conversation",
42
+ "agentic_kda_skill",
41
43
  }
42
44
  )
43
45
 
@@ -159,6 +161,16 @@ def _dispatch_agentic(
159
161
  k=k,
160
162
  **lf_kw,
161
163
  )
164
+ elif kind == "agentic_kda_skill":
165
+ evaluate_agentic_kda_skill(
166
+ host=host,
167
+ token=token,
168
+ workspace_id=workspace_id,
169
+ question=item.question,
170
+ expected_output=eo if isinstance(eo, dict) else {},
171
+ k=k,
172
+ **lf_kw,
173
+ )
162
174
  elif kind == "agentic_conversation":
163
175
  fixture_data = eo.get("fixture") or eo if isinstance(eo, dict) else {}
164
176
  evaluate_agentic_conversation(
@@ -30,6 +30,14 @@ from gooddata_eval.core.agentic.guardrail import (
30
30
  evaluate_agentic_guardrail,
31
31
  run_agentic_guardrail,
32
32
  )
33
+ from gooddata_eval.core.agentic.kda_skill import (
34
+ AgenticKdaSummary,
35
+ KdaEvaluation,
36
+ KdaRunResult,
37
+ KdaSkillAssertionError,
38
+ evaluate_agentic_kda_skill,
39
+ run_agentic_kda_skill,
40
+ )
33
41
  from gooddata_eval.core.agentic.metric_skill import (
34
42
  AgenticMetricSummary,
35
43
  MetricRunResult,
@@ -56,6 +64,7 @@ __all__ = [
56
64
  "AgenticAlertSummary",
57
65
  "AgenticGeneralQuestionSummary",
58
66
  "AgenticGuardrailSummary",
67
+ "AgenticKdaSummary",
59
68
  "AgenticMetricSummary",
60
69
  "AgenticSearchSummary",
61
70
  "AgenticRunSummary",
@@ -69,6 +78,9 @@ __all__ = [
69
78
  "GeneralQuestionResult",
70
79
  "GuardrailAssertionError",
71
80
  "GuardrailResult",
81
+ "KdaEvaluation",
82
+ "KdaRunResult",
83
+ "KdaSkillAssertionError",
72
84
  "MetricRunResult",
73
85
  "MetricSkillAssertionError",
74
86
  "RunResult",
@@ -81,6 +93,7 @@ __all__ = [
81
93
  "evaluate_agentic_conversation",
82
94
  "evaluate_agentic_general_question",
83
95
  "evaluate_agentic_guardrail",
96
+ "evaluate_agentic_kda_skill",
84
97
  "evaluate_agentic_metric_skill",
85
98
  "evaluate_agentic_search_tool",
86
99
  "evaluate_agentic_visualization",
@@ -88,6 +101,7 @@ __all__ = [
88
101
  "run_agentic_conversation",
89
102
  "run_agentic_general_question",
90
103
  "run_agentic_guardrail",
104
+ "run_agentic_kda_skill",
91
105
  "run_agentic_metric_skill",
92
106
  "run_agentic_search_tool",
93
107
  "run_agentic_visualization",
@@ -0,0 +1,382 @@
1
+ # (C) 2026 GoodData Corporation. All rights reserved.
2
+ """Agentic KDA (Key Driver Analysis)-skill evaluation runner."""
3
+
4
+ from __future__ import annotations
5
+
6
+ import logging
7
+ import os
8
+ import re
9
+ from dataclasses import dataclass
10
+
11
+ from gooddata_eval.core.chat.sse_client import ChatClient
12
+ from gooddata_eval.core.config import ReasoningEffort
13
+ from gooddata_eval.core.models import ToolCallEvent
14
+
15
+ _log = logging.getLogger(__name__)
16
+
17
+ _DEFAULT_K = 1
18
+ # Disambiguation safety net only (create+execute always run together in the same
19
+ # turn) -- 3 covers metric and period each needing their own clarifying question.
20
+ _DEFAULT_MAX_ITERATIONS = 3
21
+
22
+
23
+ def _is_asking_kda_clarification(text: str) -> bool:
24
+ """True if ``text`` reads as the agent asking for input, not a final answer.
25
+
26
+ KDA-specific, not shared with metric_skill.py/conversation.py -- each skill's
27
+ disambiguation heuristic has already drifted independently. Requires the text to
28
+ end on "?" (a "?" anywhere also matches a final answer that merely quotes one).
29
+ """
30
+ if not text:
31
+ return False
32
+ t = text.strip().lower()
33
+ if t.endswith("?"):
34
+ return True
35
+ # "To clarify, ..." means "in other words" (a final answer), not a request for one --
36
+ # strip it first so "clarif" below only matches genuine clarification requests.
37
+ t = re.sub(r"^(just )?to clarify,?\s*", "", t)
38
+ return "could you" in t or "please provide" in t or "clarif" in t
39
+
40
+
41
+ def generate_simulated_kda_response(agent_message: str, measure_candidates: dict | list[dict] | None) -> str:
42
+ """Generate a user reply to keep the KDA-skill conversation going (gpt-4o-mini).
43
+
44
+ Used only when the agent asks a clarifying question instead of triggering KDA
45
+ directly. Picks *any* candidate from ``measure_candidates`` -- scope only needs KDA
46
+ to trigger, not the resulting measure to be exactly right. Always OpenAI regardless
47
+ of the combo's own provider -- this is test-harness plumbing, not the system under test.
48
+ """
49
+ try:
50
+ from openai import OpenAI # noqa: PLC0415
51
+ except ImportError as exc:
52
+ raise RuntimeError("openai package is required for generate_simulated_kda_response") from exc
53
+
54
+ api_key = os.environ.get("OPENAI_API_KEY")
55
+ if not api_key:
56
+ raise OSError("OPENAI_API_KEY environment variable is not set")
57
+
58
+ client = OpenAI(api_key=api_key)
59
+ candidates = measure_candidates if isinstance(measure_candidates, list) else [measure_candidates or {}]
60
+ candidate_desc = "; or ".join(
61
+ f"{c.get('type')} '{c.get('id')}'" + (f" (aggregation {c['aggregation']})" if c.get("aggregation") else "")
62
+ for c in candidates
63
+ )
64
+ prompt = (
65
+ f"You are simulating a user in a conversation with a BI assistant that runs key driver "
66
+ f"analysis. The assistant said: '{agent_message}'. "
67
+ f"The user is happy to proceed with any of the following: {candidate_desc}. "
68
+ f"Reply briefly as the user, picking whichever of those the assistant offered."
69
+ )
70
+ response = client.chat.completions.create(
71
+ model="gpt-4o-mini",
72
+ messages=[{"role": "user", "content": prompt}],
73
+ max_tokens=150,
74
+ timeout=30,
75
+ )
76
+ return response.choices[0].message.content or "Please proceed with either option."
77
+
78
+
79
+ def _extract_kda_calls(tool_call_events: list[ToolCallEvent]) -> tuple[dict | None, dict | None]:
80
+ """Return (create_args, execute_result) for the LAST create/execute pair in this turn's
81
+ tool calls -- not the last create and last execute picked independently. A new create
82
+ call clears any earlier execute_result -- it belongs to the create it followed, not to
83
+ this one.
84
+ """
85
+ create_args: dict | None = None
86
+ execute_result: dict | None = None
87
+ for tc in tool_call_events:
88
+ if tc.function_name == "create_key_driver_analysis":
89
+ create_args = tc.parsed_arguments()
90
+ execute_result = None
91
+ elif tc.function_name == "execute_key_driver_analysis" and tc.result:
92
+ execute_result = tc.parsed_result()
93
+ return create_args, execute_result
94
+
95
+
96
+ @dataclass
97
+ class KdaEvaluation:
98
+ """Evaluation scores for a single KDA-skill run.
99
+
100
+ Scope: asserts only that the KDA process runs to completion -- the tool chain
101
+ triggers, executes successfully, and the chat turn ends cleanly with a non-empty
102
+ response (``turn_completed`` requires both gen-ai's stream-ended signal and a
103
+ non-empty ``text_response`` -- a stream that ends cleanly but delivers nothing to the
104
+ user isn't a completed turn either).
105
+ """
106
+
107
+ triggered: bool
108
+ executed: bool
109
+ success: bool
110
+ turn_completed: bool
111
+ disambiguated: bool = False
112
+
113
+ @property
114
+ def strict_pass(self) -> bool:
115
+ return all([self.triggered, self.executed, self.success, self.turn_completed])
116
+
117
+
118
+ @dataclass
119
+ class KdaRunResult:
120
+ """Outcome of one run (one conversation, up to max_iterations messages) for a KDA case."""
121
+
122
+ conversation_id: str
123
+ evaluation: KdaEvaluation
124
+ actual_create_args: dict | None
125
+ actual_execute_result: dict | None
126
+ # Wall-clock time of the turn that called create (None if create never happened) --
127
+ # not any earlier disambiguation turn. See run_agentic_kda_skill's _run_once.
128
+ turn_wall_clock_sec: float | None = None
129
+
130
+
131
+ @dataclass
132
+ class AgenticKdaSummary:
133
+ """Aggregated outcome of K runs for a KDA case."""
134
+
135
+ run_results: list[KdaRunResult]
136
+ pass_at_k: bool
137
+ pass_power_k: bool
138
+ best: KdaRunResult
139
+
140
+
141
+ def _evaluate_run(
142
+ create_args: dict | None,
143
+ execute_result: dict | None,
144
+ turn_completed: bool,
145
+ disambiguated: bool = False,
146
+ ) -> KdaEvaluation:
147
+ triggered = create_args is not None
148
+ executed = execute_result is not None
149
+ success = executed and execute_result.get("success") is True
150
+ return KdaEvaluation(
151
+ triggered=triggered,
152
+ executed=executed,
153
+ success=success,
154
+ turn_completed=turn_completed,
155
+ disambiguated=disambiguated,
156
+ )
157
+
158
+
159
+ def run_agentic_kda_skill(
160
+ host: str,
161
+ token: str,
162
+ workspace_id: str,
163
+ question: str,
164
+ expected_output: dict,
165
+ k: int = _DEFAULT_K,
166
+ max_iterations: int = _DEFAULT_MAX_ITERATIONS,
167
+ initial_conversation_id: str | None = None,
168
+ reasoning_effort: ReasoningEffort | None = None,
169
+ ) -> AgenticKdaSummary:
170
+ """Run the KDA-skill agentic evaluation K times and return a summary.
171
+
172
+ Each run is normally one message, one turn -- create and execute are always called
173
+ together in the same turn (the skill's own system prompt: "NO confirmation needed").
174
+ The only thing that can extend a run up to ``max_iterations`` turns is the agent
175
+ asking a clarifying question instead of triggering KDA directly; a simulated user
176
+ reply nudges it forward.
177
+ """
178
+ if k < 1:
179
+ # k=0 or negative would otherwise silently run once, indistinguishable from k=1.
180
+ raise ValueError(f"k must be >= 1, got {k}")
181
+ run_results: list[KdaRunResult] = []
182
+ client = ChatClient(host=host, token=token, workspace_id=workspace_id, reasoning_effort=reasoning_effort)
183
+
184
+ def _run_once(conv_id: str) -> KdaRunResult:
185
+ create_args: dict | None = None
186
+ execute_result: dict | None = None
187
+ turn_wall_clock_sec: float | None = None
188
+ turn_completed = False
189
+ disambiguated = False
190
+ current_question = question
191
+
192
+ for iteration in range(max_iterations):
193
+ try:
194
+ chat_result = client.send_message(conv_id, current_question)
195
+ except Exception as exc: # noqa: BLE001 -- end this run, not the whole assertion
196
+ _log.warning("KDA send_message failed for conversation %s: %s", conv_id, exc)
197
+ partial = getattr(exc, "partial_result", None)
198
+ if partial is not None:
199
+ create_args, execute_result = _extract_kda_calls(partial.tool_call_events or [])
200
+ if create_args is not None:
201
+ turn_wall_clock_sec = partial.turn_wall_clock_sec
202
+ turn_completed = False
203
+ break
204
+ create_args, execute_result = _extract_kda_calls(chat_result.tool_call_events or [])
205
+ response_text = (chat_result.text_response or "").strip()
206
+ turn_completed = chat_result.stream_ended and bool(response_text)
207
+ if create_args is not None:
208
+ # This turn's own time -- the turn that called create, not any earlier
209
+ # disambiguation turn or the simulated-reply generation. create and execute
210
+ # are always called together in the same turn (or not at all), so this is
211
+ # final either way -- execute_result may still be None (e.g. the skill's
212
+ # execute tool isn't available at all when data-sharing is off for the org).
213
+ turn_wall_clock_sec = chat_result.turn_wall_clock_sec
214
+ break
215
+ if iteration >= max_iterations - 1:
216
+ break
217
+ if _is_asking_kda_clarification(response_text):
218
+ measure_candidates = expected_output.get("Measure") if isinstance(expected_output, dict) else None
219
+ try:
220
+ current_question = generate_simulated_kda_response(response_text, measure_candidates)
221
+ disambiguated = True
222
+ except Exception as exc: # noqa: BLE001 -- safety net, not the assertion; end only this run
223
+ _log.warning("Simulated KDA user reply failed for conversation %s: %s", conv_id, exc)
224
+ break
225
+ else:
226
+ break
227
+
228
+ ev = _evaluate_run(create_args, execute_result, turn_completed, disambiguated)
229
+ return KdaRunResult(
230
+ conversation_id=conv_id,
231
+ evaluation=ev,
232
+ actual_create_args=create_args,
233
+ actual_execute_result=execute_result,
234
+ turn_wall_clock_sec=turn_wall_clock_sec,
235
+ )
236
+
237
+ try:
238
+ conv_id_0 = initial_conversation_id if initial_conversation_id is not None else client.create_conversation()
239
+ try:
240
+ run_results.append(_run_once(conv_id_0))
241
+ finally:
242
+ if initial_conversation_id is None: # only delete conversations we created
243
+ client.delete_conversation(conv_id_0)
244
+
245
+ for _ in range(1, k):
246
+ conv_id = client.create_conversation()
247
+ try:
248
+ run_results.append(_run_once(conv_id))
249
+ finally:
250
+ client.delete_conversation(conv_id)
251
+ finally:
252
+ client.close()
253
+
254
+ pass_at_k = any(r.evaluation.strict_pass for r in run_results)
255
+ pass_power_k = all(r.evaluation.strict_pass for r in run_results)
256
+ best = max(
257
+ run_results,
258
+ key=lambda r: sum(
259
+ [r.evaluation.triggered, r.evaluation.executed, r.evaluation.success, r.evaluation.turn_completed]
260
+ ),
261
+ )
262
+ return AgenticKdaSummary(
263
+ run_results=run_results,
264
+ pass_at_k=pass_at_k,
265
+ pass_power_k=pass_power_k,
266
+ best=best,
267
+ )
268
+
269
+
270
+ class KdaSkillAssertionError(AssertionError):
271
+ """Raised when a KDA-skill evaluation fails."""
272
+
273
+ __tracebackhide__ = True
274
+
275
+
276
+ def evaluate_agentic_kda_skill(
277
+ host: str,
278
+ token: str,
279
+ workspace_id: str,
280
+ question: str,
281
+ expected_output: dict,
282
+ k: int = _DEFAULT_K,
283
+ max_iterations: int = _DEFAULT_MAX_ITERATIONS,
284
+ initial_conversation_id: str | None = None,
285
+ langfuse: object | None = None,
286
+ dataset_item_id: str = "",
287
+ dataset_name: str = "kda_skill",
288
+ run_timestamp: str | None = None,
289
+ model_version_override: str | None = None,
290
+ run_metadata_extra: dict | None = None,
291
+ reasoning_effort: ReasoningEffort | None = None,
292
+ ) -> None:
293
+ """Run KDA-skill evaluation, log to Langfuse, and raise KdaSkillAssertionError on failure."""
294
+ from datetime import datetime as _dt # noqa: PLC0415
295
+ from datetime import timezone as _tz # noqa: PLC0415
296
+
297
+ from gooddata_eval.core.agentic._langfuse import try_make_langfuse_client # noqa: PLC0415
298
+
299
+ if langfuse is None:
300
+ langfuse = try_make_langfuse_client()
301
+ window_start = _dt.now(_tz.utc)
302
+ summary = run_agentic_kda_skill(
303
+ host=host,
304
+ token=token,
305
+ workspace_id=workspace_id,
306
+ question=question,
307
+ expected_output=expected_output,
308
+ k=k,
309
+ max_iterations=max_iterations,
310
+ initial_conversation_id=initial_conversation_id,
311
+ reasoning_effort=reasoning_effort,
312
+ )
313
+
314
+ if langfuse is not None and dataset_item_id:
315
+ from gooddata_eval.core.agentic._langfuse import ( # noqa: PLC0415
316
+ build_run_context,
317
+ find_traces_per_conversation,
318
+ log_quality_and_value_scores,
319
+ observe,
320
+ score_safe,
321
+ )
322
+
323
+ run_name_base, run_metadata = build_run_context(
324
+ host,
325
+ token,
326
+ workspace_id,
327
+ dataset_name,
328
+ run_timestamp,
329
+ model_version_override,
330
+ run_metadata_extra,
331
+ reasoning_effort,
332
+ )
333
+ # No custom selector -- same default (max-latency) as every other skill; harmless
334
+ # here since latency comes from run.turn_wall_clock_sec below, not this trace.
335
+ traces_by_conv = find_traces_per_conversation(
336
+ langfuse,
337
+ [r.conversation_id for r in summary.run_results],
338
+ window_start,
339
+ )
340
+ suffix_needed = len(summary.run_results) > 1
341
+ for run_idx, run in enumerate(summary.run_results):
342
+ pt = traces_by_conv.get(run.conversation_id)
343
+ run_name = f"{run_name_base}_run{run_idx}" if suffix_needed else run_name_base
344
+ ev = run.evaluation
345
+ # Gates strict_pass -- current scope is completion only (see KdaEvaluation docstring).
346
+ strict_checks = {
347
+ "kda_triggered": ev.triggered,
348
+ "kda_executed": ev.executed,
349
+ "kda_success": ev.success,
350
+ "kda_turn_completed": ev.turn_completed,
351
+ }
352
+ # Not pt.latency: pt can be any trace of the conversation, not necessarily the KDA turn.
353
+ turn_wall_clock_sec = run.turn_wall_clock_sec
354
+ _log.info("[kda-report] %s: strict_pass=%s latency_sec=%s", run_name, ev.strict_pass, turn_wall_clock_sec)
355
+ with observe(langfuse, pt.id if pt else None, dataset_item_id, run_name, run_metadata) as tid:
356
+ for score_name, value in strict_checks.items():
357
+ score_safe(langfuse, tid, name=score_name, value=float(value), data_type="BOOLEAN")
358
+ score_safe(langfuse, tid, name="kda_disambiguated", value=float(ev.disambiguated), data_type="BOOLEAN")
359
+ if turn_wall_clock_sec is not None:
360
+ # combo_report.py reads this score directly -- no trace re-resolution needed.
361
+ score_safe(
362
+ langfuse, tid, name="kda_turn_wall_clock_sec", value=turn_wall_clock_sec, data_type="NUMERIC"
363
+ )
364
+ log_quality_and_value_scores(
365
+ langfuse,
366
+ tid,
367
+ strict_checks=strict_checks,
368
+ latency_sec=turn_wall_clock_sec,
369
+ cost_usd=pt.total_cost if pt and ev.triggered else None,
370
+ )
371
+
372
+ if not summary.pass_at_k:
373
+ best = summary.best
374
+ ev = best.evaluation
375
+ message = (
376
+ f"KDA skill assertion failed. strict_pass={ev.strict_pass} "
377
+ f"(triggered={ev.triggered}, executed={ev.executed}, "
378
+ f"success={ev.success}, turn_completed={ev.turn_completed}). "
379
+ f"Actual create args: {best.actual_create_args}. "
380
+ f"Actual execute result: {best.actual_execute_result}."
381
+ )
382
+ raise KdaSkillAssertionError(message)
@@ -28,18 +28,34 @@ from gooddata_eval.core.models import ChatResult, DatasetItem
28
28
  _log = logging.getLogger(__name__)
29
29
 
30
30
  SSE_DATA_PREFIX = "data: "
31
+ SSE_EVENT_PREFIX = "event: "
32
+ # gen-ai's last event, only if at least one item was already emitted (conversations_controller.py).
33
+ _RESPONSE_ENDED_EVENT = "response_ended"
31
34
 
32
35
  _RETRYABLE_STATUS_CODES: frozenset[int] = frozenset({429, 502, 503, 504})
33
36
  _METADATA_SYNC_MARKER = "METADATA_SYNC_IN_PROGRESS"
34
37
 
35
38
 
36
39
  class ChatError(RuntimeError):
37
- """Non-retryable error reported by the chat SSE stream."""
40
+ """Non-retryable error reported by the chat SSE stream.
38
41
 
39
- def __init__(self, message: str, *, status_code: int | None = None, detail: str | None = None) -> None:
42
+ ``partial_result`` carries whatever the accumulator captured before the error fired
43
+ (tool calls included). Callers must not assume it's complete -- fields like
44
+ ``stream_ended`` reflect the state at the moment of the error, not a finished turn.
45
+ """
46
+
47
+ def __init__(
48
+ self,
49
+ message: str,
50
+ *,
51
+ status_code: int | None = None,
52
+ detail: str | None = None,
53
+ partial_result: ChatResult | None = None,
54
+ ) -> None:
40
55
  super().__init__(message)
41
56
  self.status_code = status_code
42
57
  self.detail = detail
58
+ self.partial_result = partial_result
43
59
 
44
60
 
45
61
  class TransientChatError(ChatError):
@@ -109,6 +125,7 @@ class _SseAccumulator:
109
125
  reasoning_steps: list[dict[str, Any]] = field(default_factory=list)
110
126
  adhoc_viz_args: list[dict[str, Any]] = field(default_factory=list)
111
127
  response_id: str | None = None
128
+ stream_ended: bool = False
112
129
 
113
130
 
114
131
  def _handle_text(content: dict[str, Any], acc: _SseAccumulator) -> None:
@@ -187,15 +204,37 @@ def _build_chat_result(acc: _SseAccumulator) -> ChatResult:
187
204
  }
188
205
  result = ChatResult.model_validate(payload)
189
206
  result.response_id = acc.response_id
207
+ result.stream_ended = acc.stream_ended
190
208
  return result
191
209
 
192
210
 
193
211
  def parse_sse_lines(lines: Iterable[str]) -> ChatResult:
194
212
  """Parse an SSE stream (iterable of decoded lines) into a ChatResult."""
195
213
  acc = _SseAccumulator()
196
- for raw_line in lines:
214
+ current_event = "message" # SSE default in the absence of an explicit "event: " line
215
+ it = iter(lines)
216
+ while True:
217
+ try:
218
+ raw_line = next(it)
219
+ except StopIteration:
220
+ break
221
+ except Exception as exc:
222
+ # Only a transport-level failure (e.g. connection drop mid-stream) is rescued
223
+ # here -- a bug in the processing below must propagate uncaught, not get
224
+ # mislabeled as a network error.
225
+ raise ChatError(f"SSE stream error: {exc}", partial_result=_build_chat_result(acc)) from exc
197
226
  line = raw_line.decode("utf-8") if isinstance(raw_line, bytes) else raw_line
198
- if not line or line.startswith("event: ") or not line.startswith(SSE_DATA_PREFIX):
227
+ if not line:
228
+ current_event = "message" # blank line ends one event block per the SSE spec
229
+ continue
230
+ if line.startswith(SSE_EVENT_PREFIX):
231
+ current_event = line[len(SSE_EVENT_PREFIX) :].strip()
232
+ if current_event == _RESPONSE_ENDED_EVENT:
233
+ acc.stream_ended = True
234
+ continue
235
+ if not line.startswith(SSE_DATA_PREFIX):
236
+ continue
237
+ if current_event == _RESPONSE_ENDED_EVENT:
199
238
  continue
200
239
  data_str = line[len(SSE_DATA_PREFIX) :]
201
240
  if _METADATA_SYNC_MARKER in data_str:
@@ -203,6 +242,7 @@ def parse_sse_lines(lines: Iterable[str]) -> ChatResult:
203
242
  f"SSE transient error: {_METADATA_SYNC_MARKER}",
204
243
  status_code=None,
205
244
  detail=None,
245
+ partial_result=_build_chat_result(acc),
206
246
  )
207
247
  try:
208
248
  event_data = json.loads(data_str)
@@ -213,8 +253,10 @@ def parse_sse_lines(lines: Iterable[str]) -> ChatResult:
213
253
  detail = event_data.get("detail")
214
254
  message = f"SSE error {code}: {detail}"
215
255
  if code in _RETRYABLE_STATUS_CODES:
216
- raise TransientChatError(message, status_code=code, detail=detail)
217
- raise ChatError(message, status_code=code, detail=detail)
256
+ raise TransientChatError(
257
+ message, status_code=code, detail=detail, partial_result=_build_chat_result(acc)
258
+ )
259
+ raise ChatError(message, status_code=code, detail=detail, partial_result=_build_chat_result(acc))
218
260
  if event_data.get("responseId") and not acc.response_id:
219
261
  acc.response_id = event_data["responseId"]
220
262
  item = event_data.get("item")
@@ -293,9 +335,20 @@ class ChatClient:
293
335
  body["options"] = {"reasoningEffort": self._reasoning_effort}
294
336
 
295
337
  def _do() -> ChatResult:
338
+ # Set fresh on every retry attempt (before opening this attempt's stream, so its
339
+ # own connection setup time counts) -- excludes not just the sleep backoff between
340
+ # attempts, but the entire duration of any earlier failed attempt.
341
+ t0 = time.monotonic()
296
342
  with self._client.stream("POST", url, json=body, headers=headers) as resp:
297
343
  resp.raise_for_status()
298
- return parse_sse_lines(resp.iter_lines())
344
+ try:
345
+ result = parse_sse_lines(resp.iter_lines())
346
+ except ChatError as exc:
347
+ if exc.partial_result is not None:
348
+ exc.partial_result.turn_wall_clock_sec = time.monotonic() - t0
349
+ raise
350
+ result.turn_wall_clock_sec = time.monotonic() - t0
351
+ return result
299
352
 
300
353
  return _retry_transient(_do, is_retryable=_is_retryable_exc)
301
354
 
@@ -100,6 +100,10 @@ class ChatResult(BaseModel):
100
100
  reasoning_step_count: int = Field(default=0, alias="reasoningStepCount")
101
101
  conversation_id: str | None = Field(default=None, alias="conversationId")
102
102
  response_id: str | None = Field(default=None, alias="responseId")
103
+ # True once gen-ai's response_ended event arrived.
104
+ stream_ended: bool = False
105
+ # Wall-clock seconds for the whole chat turn, timed by the client.
106
+ turn_wall_clock_sec: float | None = None
103
107
 
104
108
 
105
109
  class SummaryInput(BaseModel):