gooddata-eval 1.72.1.dev1__tar.gz → 1.72.1.dev3__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/PKG-INFO +3 -3
- {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/pyproject.toml +2 -2
- {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/src/gooddata_eval/cli/agentic_runner.py +12 -0
- {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/src/gooddata_eval/core/agentic/__init__.py +14 -0
- gooddata_eval-1.72.1.dev3/src/gooddata_eval/core/agentic/kda_skill.py +382 -0
- {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/src/gooddata_eval/core/chat/sse_client.py +60 -7
- {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/src/gooddata_eval/core/models.py +4 -0
- gooddata_eval-1.72.1.dev3/tests/test_agentic_kda_skill.py +930 -0
- gooddata_eval-1.72.1.dev3/tests/test_agentic_langfuse_trace.py +26 -0
- {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/tests/test_sse_client.py +150 -0
- {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/.gitignore +0 -0
- {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/LICENSE.txt +0 -0
- {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/Makefile +0 -0
- {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/README.md +0 -0
- {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/src/gooddata_eval/__init__.py +0 -0
- {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/src/gooddata_eval/_version.py +0 -0
- {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/src/gooddata_eval/cli/__init__.py +0 -0
- {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/src/gooddata_eval/cli/main.py +0 -0
- {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/src/gooddata_eval/core/__init__.py +0 -0
- {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/src/gooddata_eval/core/agentic/_catalog.py +0 -0
- {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/src/gooddata_eval/core/agentic/_langfuse.py +0 -0
- {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/src/gooddata_eval/core/agentic/alert_skill.py +0 -0
- {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/src/gooddata_eval/core/agentic/conversation.py +0 -0
- {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/src/gooddata_eval/core/agentic/general_question.py +0 -0
- {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/src/gooddata_eval/core/agentic/guardrail.py +0 -0
- {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/src/gooddata_eval/core/agentic/metric_skill.py +0 -0
- {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/src/gooddata_eval/core/agentic/search_tool.py +0 -0
- {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/src/gooddata_eval/core/agentic/visualization.py +0 -0
- {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/src/gooddata_eval/core/chat/__init__.py +0 -0
- {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/src/gooddata_eval/core/config.py +0 -0
- {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/src/gooddata_eval/core/connection.py +0 -0
- {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/src/gooddata_eval/core/dataset/__init__.py +0 -0
- {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/src/gooddata_eval/core/dataset/langfuse_source.py +0 -0
- {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/src/gooddata_eval/core/dataset/local.py +0 -0
- {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/src/gooddata_eval/core/evaluators/__init__.py +0 -0
- {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/src/gooddata_eval/core/evaluators/_deep_subset.py +0 -0
- {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/src/gooddata_eval/core/evaluators/_llm_judge.py +0 -0
- {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/src/gooddata_eval/core/evaluators/_text_utils.py +0 -0
- {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/src/gooddata_eval/core/evaluators/alert_skill.py +0 -0
- {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/src/gooddata_eval/core/evaluators/base.py +0 -0
- {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/src/gooddata_eval/core/evaluators/general_question.py +0 -0
- {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/src/gooddata_eval/core/evaluators/guardrail.py +0 -0
- {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/src/gooddata_eval/core/evaluators/metric_skill.py +0 -0
- {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/src/gooddata_eval/core/evaluators/search_tool.py +0 -0
- {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/src/gooddata_eval/core/evaluators/summary.py +0 -0
- {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/src/gooddata_eval/core/evaluators/visualization.py +0 -0
- {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/src/gooddata_eval/core/langfuse/__init__.py +0 -0
- {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/src/gooddata_eval/core/langfuse/sink.py +0 -0
- {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/src/gooddata_eval/core/reporting/__init__.py +0 -0
- {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/src/gooddata_eval/core/reporting/console.py +0 -0
- {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/src/gooddata_eval/core/reporting/json_report.py +0 -0
- {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/src/gooddata_eval/core/runner.py +0 -0
- {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/src/gooddata_eval/core/scoring.py +0 -0
- {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/src/gooddata_eval/core/summary/__init__.py +0 -0
- {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/src/gooddata_eval/core/summary/http_client.py +0 -0
- {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/src/gooddata_eval/core/workspace.py +0 -0
- {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/tests/__init__.py +0 -0
- {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/tests/conftest.py +0 -0
- {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/tests/fixtures/sample_dataset/metric_skill_create.json +0 -0
- {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/tests/fixtures/sample_dataset/visualization_revenue.json +0 -0
- {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/tests/fixtures/sse_visualization_stream.txt +0 -0
- {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/tests/test_agentic_alert_skill.py +0 -0
- {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/tests/test_agentic_conversation.py +0 -0
- {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/tests/test_agentic_general_question.py +0 -0
- {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/tests/test_agentic_guardrail.py +0 -0
- {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/tests/test_agentic_metric_skill.py +0 -0
- {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/tests/test_agentic_run_context.py +0 -0
- {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/tests/test_agentic_search_tool.py +0 -0
- {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/tests/test_agentic_visualization.py +0 -0
- {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/tests/test_alert_skill_evaluator.py +0 -0
- {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/tests/test_cli.py +0 -0
- {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/tests/test_connection.py +0 -0
- {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/tests/test_deep_subset.py +0 -0
- {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/tests/test_langfuse_sink.py +0 -0
- {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/tests/test_langfuse_source.py +0 -0
- {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/tests/test_llm_judge.py +0 -0
- {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/tests/test_local_loader.py +0 -0
- {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/tests/test_metric_skill_evaluator.py +0 -0
- {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/tests/test_models.py +0 -0
- {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/tests/test_reporting.py +0 -0
- {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/tests/test_runner.py +0 -0
- {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/tests/test_scoring.py +0 -0
- {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/tests/test_search_tool_evaluator.py +0 -0
- {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/tests/test_summary_client.py +0 -0
- {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/tests/test_summary_evaluator.py +0 -0
- {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/tests/test_text_evaluators.py +0 -0
- {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/tests/test_visualization_evaluator.py +0 -0
- {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/tests/test_workspace.py +0 -0
- {gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/tox.ini +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
|
-
Metadata-Version: 2.
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
2
|
Name: gooddata-eval
|
|
3
|
-
Version: 1.72.1.
|
|
3
|
+
Version: 1.72.1.dev3
|
|
4
4
|
Summary: Evaluate the GoodData AI agent against your own questions and models.
|
|
5
5
|
Project-URL: Source, https://github.com/gooddata/gooddata-python-sdk
|
|
6
6
|
Author-email: GoodData <support@gooddata.com>
|
|
@@ -17,7 +17,7 @@ Classifier: Topic :: Scientific/Engineering
|
|
|
17
17
|
Classifier: Topic :: Software Development
|
|
18
18
|
Classifier: Typing :: Typed
|
|
19
19
|
Requires-Python: >=3.10
|
|
20
|
-
Requires-Dist: gooddata-sdk~=1.72.1.
|
|
20
|
+
Requires-Dist: gooddata-sdk~=1.72.1.dev3
|
|
21
21
|
Requires-Dist: httpx<1.0,>=0.27
|
|
22
22
|
Requires-Dist: orjson<4.0.0,>=3.9.15
|
|
23
23
|
Requires-Dist: pydantic<3.0,>=2.6
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
# (C) 2026 GoodData Corporation
|
|
2
2
|
[project]
|
|
3
3
|
name = "gooddata-eval"
|
|
4
|
-
version = "1.72.1.
|
|
4
|
+
version = "1.72.1.dev3"
|
|
5
5
|
description = "Evaluate the GoodData AI agent against your own questions and models."
|
|
6
6
|
readme = "README.md"
|
|
7
7
|
license = "MIT"
|
|
@@ -11,7 +11,7 @@ authors = [
|
|
|
11
11
|
keywords = ["gooddata", "ai", "evaluation", "llm", "analytics", "cli"]
|
|
12
12
|
requires-python = ">=3.10"
|
|
13
13
|
dependencies = [
|
|
14
|
-
"gooddata-sdk~=1.72.1.
|
|
14
|
+
"gooddata-sdk~=1.72.1.dev3",
|
|
15
15
|
"httpx>=0.27,<1.0",
|
|
16
16
|
"orjson>=3.9.15,<4.0.0",
|
|
17
17
|
"pydantic>=2.6,<3.0",
|
{gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/src/gooddata_eval/cli/agentic_runner.py
RENAMED
|
@@ -11,6 +11,7 @@ from gooddata_eval.core.agentic.alert_skill import evaluate_agentic_alert_skill
|
|
|
11
11
|
from gooddata_eval.core.agentic.conversation import ConversationFixture, evaluate_agentic_conversation
|
|
12
12
|
from gooddata_eval.core.agentic.general_question import evaluate_agentic_general_question
|
|
13
13
|
from gooddata_eval.core.agentic.guardrail import evaluate_agentic_guardrail
|
|
14
|
+
from gooddata_eval.core.agentic.kda_skill import evaluate_agentic_kda_skill
|
|
14
15
|
from gooddata_eval.core.agentic.metric_skill import evaluate_agentic_metric_skill
|
|
15
16
|
from gooddata_eval.core.agentic.search_tool import evaluate_agentic_search_tool
|
|
16
17
|
from gooddata_eval.core.agentic.visualization import evaluate_agentic_visualization
|
|
@@ -38,6 +39,7 @@ AGENTIC_TEST_KINDS = frozenset(
|
|
|
38
39
|
"agentic_general_question",
|
|
39
40
|
"agentic_guardrail",
|
|
40
41
|
"agentic_conversation",
|
|
42
|
+
"agentic_kda_skill",
|
|
41
43
|
}
|
|
42
44
|
)
|
|
43
45
|
|
|
@@ -159,6 +161,16 @@ def _dispatch_agentic(
|
|
|
159
161
|
k=k,
|
|
160
162
|
**lf_kw,
|
|
161
163
|
)
|
|
164
|
+
elif kind == "agentic_kda_skill":
|
|
165
|
+
evaluate_agentic_kda_skill(
|
|
166
|
+
host=host,
|
|
167
|
+
token=token,
|
|
168
|
+
workspace_id=workspace_id,
|
|
169
|
+
question=item.question,
|
|
170
|
+
expected_output=eo if isinstance(eo, dict) else {},
|
|
171
|
+
k=k,
|
|
172
|
+
**lf_kw,
|
|
173
|
+
)
|
|
162
174
|
elif kind == "agentic_conversation":
|
|
163
175
|
fixture_data = eo.get("fixture") or eo if isinstance(eo, dict) else {}
|
|
164
176
|
evaluate_agentic_conversation(
|
{gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/src/gooddata_eval/core/agentic/__init__.py
RENAMED
|
@@ -30,6 +30,14 @@ from gooddata_eval.core.agentic.guardrail import (
|
|
|
30
30
|
evaluate_agentic_guardrail,
|
|
31
31
|
run_agentic_guardrail,
|
|
32
32
|
)
|
|
33
|
+
from gooddata_eval.core.agentic.kda_skill import (
|
|
34
|
+
AgenticKdaSummary,
|
|
35
|
+
KdaEvaluation,
|
|
36
|
+
KdaRunResult,
|
|
37
|
+
KdaSkillAssertionError,
|
|
38
|
+
evaluate_agentic_kda_skill,
|
|
39
|
+
run_agentic_kda_skill,
|
|
40
|
+
)
|
|
33
41
|
from gooddata_eval.core.agentic.metric_skill import (
|
|
34
42
|
AgenticMetricSummary,
|
|
35
43
|
MetricRunResult,
|
|
@@ -56,6 +64,7 @@ __all__ = [
|
|
|
56
64
|
"AgenticAlertSummary",
|
|
57
65
|
"AgenticGeneralQuestionSummary",
|
|
58
66
|
"AgenticGuardrailSummary",
|
|
67
|
+
"AgenticKdaSummary",
|
|
59
68
|
"AgenticMetricSummary",
|
|
60
69
|
"AgenticSearchSummary",
|
|
61
70
|
"AgenticRunSummary",
|
|
@@ -69,6 +78,9 @@ __all__ = [
|
|
|
69
78
|
"GeneralQuestionResult",
|
|
70
79
|
"GuardrailAssertionError",
|
|
71
80
|
"GuardrailResult",
|
|
81
|
+
"KdaEvaluation",
|
|
82
|
+
"KdaRunResult",
|
|
83
|
+
"KdaSkillAssertionError",
|
|
72
84
|
"MetricRunResult",
|
|
73
85
|
"MetricSkillAssertionError",
|
|
74
86
|
"RunResult",
|
|
@@ -81,6 +93,7 @@ __all__ = [
|
|
|
81
93
|
"evaluate_agentic_conversation",
|
|
82
94
|
"evaluate_agentic_general_question",
|
|
83
95
|
"evaluate_agentic_guardrail",
|
|
96
|
+
"evaluate_agentic_kda_skill",
|
|
84
97
|
"evaluate_agentic_metric_skill",
|
|
85
98
|
"evaluate_agentic_search_tool",
|
|
86
99
|
"evaluate_agentic_visualization",
|
|
@@ -88,6 +101,7 @@ __all__ = [
|
|
|
88
101
|
"run_agentic_conversation",
|
|
89
102
|
"run_agentic_general_question",
|
|
90
103
|
"run_agentic_guardrail",
|
|
104
|
+
"run_agentic_kda_skill",
|
|
91
105
|
"run_agentic_metric_skill",
|
|
92
106
|
"run_agentic_search_tool",
|
|
93
107
|
"run_agentic_visualization",
|
|
@@ -0,0 +1,382 @@
|
|
|
1
|
+
# (C) 2026 GoodData Corporation. All rights reserved.
|
|
2
|
+
"""Agentic KDA (Key Driver Analysis)-skill evaluation runner."""
|
|
3
|
+
|
|
4
|
+
from __future__ import annotations
|
|
5
|
+
|
|
6
|
+
import logging
|
|
7
|
+
import os
|
|
8
|
+
import re
|
|
9
|
+
from dataclasses import dataclass
|
|
10
|
+
|
|
11
|
+
from gooddata_eval.core.chat.sse_client import ChatClient
|
|
12
|
+
from gooddata_eval.core.config import ReasoningEffort
|
|
13
|
+
from gooddata_eval.core.models import ToolCallEvent
|
|
14
|
+
|
|
15
|
+
_log = logging.getLogger(__name__)
|
|
16
|
+
|
|
17
|
+
_DEFAULT_K = 1
|
|
18
|
+
# Disambiguation safety net only (create+execute always run together in the same
|
|
19
|
+
# turn) -- 3 covers metric and period each needing their own clarifying question.
|
|
20
|
+
_DEFAULT_MAX_ITERATIONS = 3
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def _is_asking_kda_clarification(text: str) -> bool:
|
|
24
|
+
"""True if ``text`` reads as the agent asking for input, not a final answer.
|
|
25
|
+
|
|
26
|
+
KDA-specific, not shared with metric_skill.py/conversation.py -- each skill's
|
|
27
|
+
disambiguation heuristic has already drifted independently. Requires the text to
|
|
28
|
+
end on "?" (a "?" anywhere also matches a final answer that merely quotes one).
|
|
29
|
+
"""
|
|
30
|
+
if not text:
|
|
31
|
+
return False
|
|
32
|
+
t = text.strip().lower()
|
|
33
|
+
if t.endswith("?"):
|
|
34
|
+
return True
|
|
35
|
+
# "To clarify, ..." means "in other words" (a final answer), not a request for one --
|
|
36
|
+
# strip it first so "clarif" below only matches genuine clarification requests.
|
|
37
|
+
t = re.sub(r"^(just )?to clarify,?\s*", "", t)
|
|
38
|
+
return "could you" in t or "please provide" in t or "clarif" in t
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def generate_simulated_kda_response(agent_message: str, measure_candidates: dict | list[dict] | None) -> str:
|
|
42
|
+
"""Generate a user reply to keep the KDA-skill conversation going (gpt-4o-mini).
|
|
43
|
+
|
|
44
|
+
Used only when the agent asks a clarifying question instead of triggering KDA
|
|
45
|
+
directly. Picks *any* candidate from ``measure_candidates`` -- scope only needs KDA
|
|
46
|
+
to trigger, not the resulting measure to be exactly right. Always OpenAI regardless
|
|
47
|
+
of the combo's own provider -- this is test-harness plumbing, not the system under test.
|
|
48
|
+
"""
|
|
49
|
+
try:
|
|
50
|
+
from openai import OpenAI # noqa: PLC0415
|
|
51
|
+
except ImportError as exc:
|
|
52
|
+
raise RuntimeError("openai package is required for generate_simulated_kda_response") from exc
|
|
53
|
+
|
|
54
|
+
api_key = os.environ.get("OPENAI_API_KEY")
|
|
55
|
+
if not api_key:
|
|
56
|
+
raise OSError("OPENAI_API_KEY environment variable is not set")
|
|
57
|
+
|
|
58
|
+
client = OpenAI(api_key=api_key)
|
|
59
|
+
candidates = measure_candidates if isinstance(measure_candidates, list) else [measure_candidates or {}]
|
|
60
|
+
candidate_desc = "; or ".join(
|
|
61
|
+
f"{c.get('type')} '{c.get('id')}'" + (f" (aggregation {c['aggregation']})" if c.get("aggregation") else "")
|
|
62
|
+
for c in candidates
|
|
63
|
+
)
|
|
64
|
+
prompt = (
|
|
65
|
+
f"You are simulating a user in a conversation with a BI assistant that runs key driver "
|
|
66
|
+
f"analysis. The assistant said: '{agent_message}'. "
|
|
67
|
+
f"The user is happy to proceed with any of the following: {candidate_desc}. "
|
|
68
|
+
f"Reply briefly as the user, picking whichever of those the assistant offered."
|
|
69
|
+
)
|
|
70
|
+
response = client.chat.completions.create(
|
|
71
|
+
model="gpt-4o-mini",
|
|
72
|
+
messages=[{"role": "user", "content": prompt}],
|
|
73
|
+
max_tokens=150,
|
|
74
|
+
timeout=30,
|
|
75
|
+
)
|
|
76
|
+
return response.choices[0].message.content or "Please proceed with either option."
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
def _extract_kda_calls(tool_call_events: list[ToolCallEvent]) -> tuple[dict | None, dict | None]:
|
|
80
|
+
"""Return (create_args, execute_result) for the LAST create/execute pair in this turn's
|
|
81
|
+
tool calls -- not the last create and last execute picked independently. A new create
|
|
82
|
+
call clears any earlier execute_result -- it belongs to the create it followed, not to
|
|
83
|
+
this one.
|
|
84
|
+
"""
|
|
85
|
+
create_args: dict | None = None
|
|
86
|
+
execute_result: dict | None = None
|
|
87
|
+
for tc in tool_call_events:
|
|
88
|
+
if tc.function_name == "create_key_driver_analysis":
|
|
89
|
+
create_args = tc.parsed_arguments()
|
|
90
|
+
execute_result = None
|
|
91
|
+
elif tc.function_name == "execute_key_driver_analysis" and tc.result:
|
|
92
|
+
execute_result = tc.parsed_result()
|
|
93
|
+
return create_args, execute_result
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
@dataclass
|
|
97
|
+
class KdaEvaluation:
|
|
98
|
+
"""Evaluation scores for a single KDA-skill run.
|
|
99
|
+
|
|
100
|
+
Scope: asserts only that the KDA process runs to completion -- the tool chain
|
|
101
|
+
triggers, executes successfully, and the chat turn ends cleanly with a non-empty
|
|
102
|
+
response (``turn_completed`` requires both gen-ai's stream-ended signal and a
|
|
103
|
+
non-empty ``text_response`` -- a stream that ends cleanly but delivers nothing to the
|
|
104
|
+
user isn't a completed turn either).
|
|
105
|
+
"""
|
|
106
|
+
|
|
107
|
+
triggered: bool
|
|
108
|
+
executed: bool
|
|
109
|
+
success: bool
|
|
110
|
+
turn_completed: bool
|
|
111
|
+
disambiguated: bool = False
|
|
112
|
+
|
|
113
|
+
@property
|
|
114
|
+
def strict_pass(self) -> bool:
|
|
115
|
+
return all([self.triggered, self.executed, self.success, self.turn_completed])
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
@dataclass
|
|
119
|
+
class KdaRunResult:
|
|
120
|
+
"""Outcome of one run (one conversation, up to max_iterations messages) for a KDA case."""
|
|
121
|
+
|
|
122
|
+
conversation_id: str
|
|
123
|
+
evaluation: KdaEvaluation
|
|
124
|
+
actual_create_args: dict | None
|
|
125
|
+
actual_execute_result: dict | None
|
|
126
|
+
# Wall-clock time of the turn that called create (None if create never happened) --
|
|
127
|
+
# not any earlier disambiguation turn. See run_agentic_kda_skill's _run_once.
|
|
128
|
+
turn_wall_clock_sec: float | None = None
|
|
129
|
+
|
|
130
|
+
|
|
131
|
+
@dataclass
|
|
132
|
+
class AgenticKdaSummary:
|
|
133
|
+
"""Aggregated outcome of K runs for a KDA case."""
|
|
134
|
+
|
|
135
|
+
run_results: list[KdaRunResult]
|
|
136
|
+
pass_at_k: bool
|
|
137
|
+
pass_power_k: bool
|
|
138
|
+
best: KdaRunResult
|
|
139
|
+
|
|
140
|
+
|
|
141
|
+
def _evaluate_run(
|
|
142
|
+
create_args: dict | None,
|
|
143
|
+
execute_result: dict | None,
|
|
144
|
+
turn_completed: bool,
|
|
145
|
+
disambiguated: bool = False,
|
|
146
|
+
) -> KdaEvaluation:
|
|
147
|
+
triggered = create_args is not None
|
|
148
|
+
executed = execute_result is not None
|
|
149
|
+
success = executed and execute_result.get("success") is True
|
|
150
|
+
return KdaEvaluation(
|
|
151
|
+
triggered=triggered,
|
|
152
|
+
executed=executed,
|
|
153
|
+
success=success,
|
|
154
|
+
turn_completed=turn_completed,
|
|
155
|
+
disambiguated=disambiguated,
|
|
156
|
+
)
|
|
157
|
+
|
|
158
|
+
|
|
159
|
+
def run_agentic_kda_skill(
|
|
160
|
+
host: str,
|
|
161
|
+
token: str,
|
|
162
|
+
workspace_id: str,
|
|
163
|
+
question: str,
|
|
164
|
+
expected_output: dict,
|
|
165
|
+
k: int = _DEFAULT_K,
|
|
166
|
+
max_iterations: int = _DEFAULT_MAX_ITERATIONS,
|
|
167
|
+
initial_conversation_id: str | None = None,
|
|
168
|
+
reasoning_effort: ReasoningEffort | None = None,
|
|
169
|
+
) -> AgenticKdaSummary:
|
|
170
|
+
"""Run the KDA-skill agentic evaluation K times and return a summary.
|
|
171
|
+
|
|
172
|
+
Each run is normally one message, one turn -- create and execute are always called
|
|
173
|
+
together in the same turn (the skill's own system prompt: "NO confirmation needed").
|
|
174
|
+
The only thing that can extend a run up to ``max_iterations`` turns is the agent
|
|
175
|
+
asking a clarifying question instead of triggering KDA directly; a simulated user
|
|
176
|
+
reply nudges it forward.
|
|
177
|
+
"""
|
|
178
|
+
if k < 1:
|
|
179
|
+
# k=0 or negative would otherwise silently run once, indistinguishable from k=1.
|
|
180
|
+
raise ValueError(f"k must be >= 1, got {k}")
|
|
181
|
+
run_results: list[KdaRunResult] = []
|
|
182
|
+
client = ChatClient(host=host, token=token, workspace_id=workspace_id, reasoning_effort=reasoning_effort)
|
|
183
|
+
|
|
184
|
+
def _run_once(conv_id: str) -> KdaRunResult:
|
|
185
|
+
create_args: dict | None = None
|
|
186
|
+
execute_result: dict | None = None
|
|
187
|
+
turn_wall_clock_sec: float | None = None
|
|
188
|
+
turn_completed = False
|
|
189
|
+
disambiguated = False
|
|
190
|
+
current_question = question
|
|
191
|
+
|
|
192
|
+
for iteration in range(max_iterations):
|
|
193
|
+
try:
|
|
194
|
+
chat_result = client.send_message(conv_id, current_question)
|
|
195
|
+
except Exception as exc: # noqa: BLE001 -- end this run, not the whole assertion
|
|
196
|
+
_log.warning("KDA send_message failed for conversation %s: %s", conv_id, exc)
|
|
197
|
+
partial = getattr(exc, "partial_result", None)
|
|
198
|
+
if partial is not None:
|
|
199
|
+
create_args, execute_result = _extract_kda_calls(partial.tool_call_events or [])
|
|
200
|
+
if create_args is not None:
|
|
201
|
+
turn_wall_clock_sec = partial.turn_wall_clock_sec
|
|
202
|
+
turn_completed = False
|
|
203
|
+
break
|
|
204
|
+
create_args, execute_result = _extract_kda_calls(chat_result.tool_call_events or [])
|
|
205
|
+
response_text = (chat_result.text_response or "").strip()
|
|
206
|
+
turn_completed = chat_result.stream_ended and bool(response_text)
|
|
207
|
+
if create_args is not None:
|
|
208
|
+
# This turn's own time -- the turn that called create, not any earlier
|
|
209
|
+
# disambiguation turn or the simulated-reply generation. create and execute
|
|
210
|
+
# are always called together in the same turn (or not at all), so this is
|
|
211
|
+
# final either way -- execute_result may still be None (e.g. the skill's
|
|
212
|
+
# execute tool isn't available at all when data-sharing is off for the org).
|
|
213
|
+
turn_wall_clock_sec = chat_result.turn_wall_clock_sec
|
|
214
|
+
break
|
|
215
|
+
if iteration >= max_iterations - 1:
|
|
216
|
+
break
|
|
217
|
+
if _is_asking_kda_clarification(response_text):
|
|
218
|
+
measure_candidates = expected_output.get("Measure") if isinstance(expected_output, dict) else None
|
|
219
|
+
try:
|
|
220
|
+
current_question = generate_simulated_kda_response(response_text, measure_candidates)
|
|
221
|
+
disambiguated = True
|
|
222
|
+
except Exception as exc: # noqa: BLE001 -- safety net, not the assertion; end only this run
|
|
223
|
+
_log.warning("Simulated KDA user reply failed for conversation %s: %s", conv_id, exc)
|
|
224
|
+
break
|
|
225
|
+
else:
|
|
226
|
+
break
|
|
227
|
+
|
|
228
|
+
ev = _evaluate_run(create_args, execute_result, turn_completed, disambiguated)
|
|
229
|
+
return KdaRunResult(
|
|
230
|
+
conversation_id=conv_id,
|
|
231
|
+
evaluation=ev,
|
|
232
|
+
actual_create_args=create_args,
|
|
233
|
+
actual_execute_result=execute_result,
|
|
234
|
+
turn_wall_clock_sec=turn_wall_clock_sec,
|
|
235
|
+
)
|
|
236
|
+
|
|
237
|
+
try:
|
|
238
|
+
conv_id_0 = initial_conversation_id if initial_conversation_id is not None else client.create_conversation()
|
|
239
|
+
try:
|
|
240
|
+
run_results.append(_run_once(conv_id_0))
|
|
241
|
+
finally:
|
|
242
|
+
if initial_conversation_id is None: # only delete conversations we created
|
|
243
|
+
client.delete_conversation(conv_id_0)
|
|
244
|
+
|
|
245
|
+
for _ in range(1, k):
|
|
246
|
+
conv_id = client.create_conversation()
|
|
247
|
+
try:
|
|
248
|
+
run_results.append(_run_once(conv_id))
|
|
249
|
+
finally:
|
|
250
|
+
client.delete_conversation(conv_id)
|
|
251
|
+
finally:
|
|
252
|
+
client.close()
|
|
253
|
+
|
|
254
|
+
pass_at_k = any(r.evaluation.strict_pass for r in run_results)
|
|
255
|
+
pass_power_k = all(r.evaluation.strict_pass for r in run_results)
|
|
256
|
+
best = max(
|
|
257
|
+
run_results,
|
|
258
|
+
key=lambda r: sum(
|
|
259
|
+
[r.evaluation.triggered, r.evaluation.executed, r.evaluation.success, r.evaluation.turn_completed]
|
|
260
|
+
),
|
|
261
|
+
)
|
|
262
|
+
return AgenticKdaSummary(
|
|
263
|
+
run_results=run_results,
|
|
264
|
+
pass_at_k=pass_at_k,
|
|
265
|
+
pass_power_k=pass_power_k,
|
|
266
|
+
best=best,
|
|
267
|
+
)
|
|
268
|
+
|
|
269
|
+
|
|
270
|
+
class KdaSkillAssertionError(AssertionError):
|
|
271
|
+
"""Raised when a KDA-skill evaluation fails."""
|
|
272
|
+
|
|
273
|
+
__tracebackhide__ = True
|
|
274
|
+
|
|
275
|
+
|
|
276
|
+
def evaluate_agentic_kda_skill(
|
|
277
|
+
host: str,
|
|
278
|
+
token: str,
|
|
279
|
+
workspace_id: str,
|
|
280
|
+
question: str,
|
|
281
|
+
expected_output: dict,
|
|
282
|
+
k: int = _DEFAULT_K,
|
|
283
|
+
max_iterations: int = _DEFAULT_MAX_ITERATIONS,
|
|
284
|
+
initial_conversation_id: str | None = None,
|
|
285
|
+
langfuse: object | None = None,
|
|
286
|
+
dataset_item_id: str = "",
|
|
287
|
+
dataset_name: str = "kda_skill",
|
|
288
|
+
run_timestamp: str | None = None,
|
|
289
|
+
model_version_override: str | None = None,
|
|
290
|
+
run_metadata_extra: dict | None = None,
|
|
291
|
+
reasoning_effort: ReasoningEffort | None = None,
|
|
292
|
+
) -> None:
|
|
293
|
+
"""Run KDA-skill evaluation, log to Langfuse, and raise KdaSkillAssertionError on failure."""
|
|
294
|
+
from datetime import datetime as _dt # noqa: PLC0415
|
|
295
|
+
from datetime import timezone as _tz # noqa: PLC0415
|
|
296
|
+
|
|
297
|
+
from gooddata_eval.core.agentic._langfuse import try_make_langfuse_client # noqa: PLC0415
|
|
298
|
+
|
|
299
|
+
if langfuse is None:
|
|
300
|
+
langfuse = try_make_langfuse_client()
|
|
301
|
+
window_start = _dt.now(_tz.utc)
|
|
302
|
+
summary = run_agentic_kda_skill(
|
|
303
|
+
host=host,
|
|
304
|
+
token=token,
|
|
305
|
+
workspace_id=workspace_id,
|
|
306
|
+
question=question,
|
|
307
|
+
expected_output=expected_output,
|
|
308
|
+
k=k,
|
|
309
|
+
max_iterations=max_iterations,
|
|
310
|
+
initial_conversation_id=initial_conversation_id,
|
|
311
|
+
reasoning_effort=reasoning_effort,
|
|
312
|
+
)
|
|
313
|
+
|
|
314
|
+
if langfuse is not None and dataset_item_id:
|
|
315
|
+
from gooddata_eval.core.agentic._langfuse import ( # noqa: PLC0415
|
|
316
|
+
build_run_context,
|
|
317
|
+
find_traces_per_conversation,
|
|
318
|
+
log_quality_and_value_scores,
|
|
319
|
+
observe,
|
|
320
|
+
score_safe,
|
|
321
|
+
)
|
|
322
|
+
|
|
323
|
+
run_name_base, run_metadata = build_run_context(
|
|
324
|
+
host,
|
|
325
|
+
token,
|
|
326
|
+
workspace_id,
|
|
327
|
+
dataset_name,
|
|
328
|
+
run_timestamp,
|
|
329
|
+
model_version_override,
|
|
330
|
+
run_metadata_extra,
|
|
331
|
+
reasoning_effort,
|
|
332
|
+
)
|
|
333
|
+
# No custom selector -- same default (max-latency) as every other skill; harmless
|
|
334
|
+
# here since latency comes from run.turn_wall_clock_sec below, not this trace.
|
|
335
|
+
traces_by_conv = find_traces_per_conversation(
|
|
336
|
+
langfuse,
|
|
337
|
+
[r.conversation_id for r in summary.run_results],
|
|
338
|
+
window_start,
|
|
339
|
+
)
|
|
340
|
+
suffix_needed = len(summary.run_results) > 1
|
|
341
|
+
for run_idx, run in enumerate(summary.run_results):
|
|
342
|
+
pt = traces_by_conv.get(run.conversation_id)
|
|
343
|
+
run_name = f"{run_name_base}_run{run_idx}" if suffix_needed else run_name_base
|
|
344
|
+
ev = run.evaluation
|
|
345
|
+
# Gates strict_pass -- current scope is completion only (see KdaEvaluation docstring).
|
|
346
|
+
strict_checks = {
|
|
347
|
+
"kda_triggered": ev.triggered,
|
|
348
|
+
"kda_executed": ev.executed,
|
|
349
|
+
"kda_success": ev.success,
|
|
350
|
+
"kda_turn_completed": ev.turn_completed,
|
|
351
|
+
}
|
|
352
|
+
# Not pt.latency: pt can be any trace of the conversation, not necessarily the KDA turn.
|
|
353
|
+
turn_wall_clock_sec = run.turn_wall_clock_sec
|
|
354
|
+
_log.info("[kda-report] %s: strict_pass=%s latency_sec=%s", run_name, ev.strict_pass, turn_wall_clock_sec)
|
|
355
|
+
with observe(langfuse, pt.id if pt else None, dataset_item_id, run_name, run_metadata) as tid:
|
|
356
|
+
for score_name, value in strict_checks.items():
|
|
357
|
+
score_safe(langfuse, tid, name=score_name, value=float(value), data_type="BOOLEAN")
|
|
358
|
+
score_safe(langfuse, tid, name="kda_disambiguated", value=float(ev.disambiguated), data_type="BOOLEAN")
|
|
359
|
+
if turn_wall_clock_sec is not None:
|
|
360
|
+
# combo_report.py reads this score directly -- no trace re-resolution needed.
|
|
361
|
+
score_safe(
|
|
362
|
+
langfuse, tid, name="kda_turn_wall_clock_sec", value=turn_wall_clock_sec, data_type="NUMERIC"
|
|
363
|
+
)
|
|
364
|
+
log_quality_and_value_scores(
|
|
365
|
+
langfuse,
|
|
366
|
+
tid,
|
|
367
|
+
strict_checks=strict_checks,
|
|
368
|
+
latency_sec=turn_wall_clock_sec,
|
|
369
|
+
cost_usd=pt.total_cost if pt and ev.triggered else None,
|
|
370
|
+
)
|
|
371
|
+
|
|
372
|
+
if not summary.pass_at_k:
|
|
373
|
+
best = summary.best
|
|
374
|
+
ev = best.evaluation
|
|
375
|
+
message = (
|
|
376
|
+
f"KDA skill assertion failed. strict_pass={ev.strict_pass} "
|
|
377
|
+
f"(triggered={ev.triggered}, executed={ev.executed}, "
|
|
378
|
+
f"success={ev.success}, turn_completed={ev.turn_completed}). "
|
|
379
|
+
f"Actual create args: {best.actual_create_args}. "
|
|
380
|
+
f"Actual execute result: {best.actual_execute_result}."
|
|
381
|
+
)
|
|
382
|
+
raise KdaSkillAssertionError(message)
|
{gooddata_eval-1.72.1.dev1 → gooddata_eval-1.72.1.dev3}/src/gooddata_eval/core/chat/sse_client.py
RENAMED
|
@@ -28,18 +28,34 @@ from gooddata_eval.core.models import ChatResult, DatasetItem
|
|
|
28
28
|
_log = logging.getLogger(__name__)
|
|
29
29
|
|
|
30
30
|
SSE_DATA_PREFIX = "data: "
|
|
31
|
+
SSE_EVENT_PREFIX = "event: "
|
|
32
|
+
# gen-ai's last event, only if at least one item was already emitted (conversations_controller.py).
|
|
33
|
+
_RESPONSE_ENDED_EVENT = "response_ended"
|
|
31
34
|
|
|
32
35
|
_RETRYABLE_STATUS_CODES: frozenset[int] = frozenset({429, 502, 503, 504})
|
|
33
36
|
_METADATA_SYNC_MARKER = "METADATA_SYNC_IN_PROGRESS"
|
|
34
37
|
|
|
35
38
|
|
|
36
39
|
class ChatError(RuntimeError):
|
|
37
|
-
"""Non-retryable error reported by the chat SSE stream.
|
|
40
|
+
"""Non-retryable error reported by the chat SSE stream.
|
|
38
41
|
|
|
39
|
-
|
|
42
|
+
``partial_result`` carries whatever the accumulator captured before the error fired
|
|
43
|
+
(tool calls included). Callers must not assume it's complete -- fields like
|
|
44
|
+
``stream_ended`` reflect the state at the moment of the error, not a finished turn.
|
|
45
|
+
"""
|
|
46
|
+
|
|
47
|
+
def __init__(
|
|
48
|
+
self,
|
|
49
|
+
message: str,
|
|
50
|
+
*,
|
|
51
|
+
status_code: int | None = None,
|
|
52
|
+
detail: str | None = None,
|
|
53
|
+
partial_result: ChatResult | None = None,
|
|
54
|
+
) -> None:
|
|
40
55
|
super().__init__(message)
|
|
41
56
|
self.status_code = status_code
|
|
42
57
|
self.detail = detail
|
|
58
|
+
self.partial_result = partial_result
|
|
43
59
|
|
|
44
60
|
|
|
45
61
|
class TransientChatError(ChatError):
|
|
@@ -109,6 +125,7 @@ class _SseAccumulator:
|
|
|
109
125
|
reasoning_steps: list[dict[str, Any]] = field(default_factory=list)
|
|
110
126
|
adhoc_viz_args: list[dict[str, Any]] = field(default_factory=list)
|
|
111
127
|
response_id: str | None = None
|
|
128
|
+
stream_ended: bool = False
|
|
112
129
|
|
|
113
130
|
|
|
114
131
|
def _handle_text(content: dict[str, Any], acc: _SseAccumulator) -> None:
|
|
@@ -187,15 +204,37 @@ def _build_chat_result(acc: _SseAccumulator) -> ChatResult:
|
|
|
187
204
|
}
|
|
188
205
|
result = ChatResult.model_validate(payload)
|
|
189
206
|
result.response_id = acc.response_id
|
|
207
|
+
result.stream_ended = acc.stream_ended
|
|
190
208
|
return result
|
|
191
209
|
|
|
192
210
|
|
|
193
211
|
def parse_sse_lines(lines: Iterable[str]) -> ChatResult:
|
|
194
212
|
"""Parse an SSE stream (iterable of decoded lines) into a ChatResult."""
|
|
195
213
|
acc = _SseAccumulator()
|
|
196
|
-
|
|
214
|
+
current_event = "message" # SSE default in the absence of an explicit "event: " line
|
|
215
|
+
it = iter(lines)
|
|
216
|
+
while True:
|
|
217
|
+
try:
|
|
218
|
+
raw_line = next(it)
|
|
219
|
+
except StopIteration:
|
|
220
|
+
break
|
|
221
|
+
except Exception as exc:
|
|
222
|
+
# Only a transport-level failure (e.g. connection drop mid-stream) is rescued
|
|
223
|
+
# here -- a bug in the processing below must propagate uncaught, not get
|
|
224
|
+
# mislabeled as a network error.
|
|
225
|
+
raise ChatError(f"SSE stream error: {exc}", partial_result=_build_chat_result(acc)) from exc
|
|
197
226
|
line = raw_line.decode("utf-8") if isinstance(raw_line, bytes) else raw_line
|
|
198
|
-
if not line
|
|
227
|
+
if not line:
|
|
228
|
+
current_event = "message" # blank line ends one event block per the SSE spec
|
|
229
|
+
continue
|
|
230
|
+
if line.startswith(SSE_EVENT_PREFIX):
|
|
231
|
+
current_event = line[len(SSE_EVENT_PREFIX) :].strip()
|
|
232
|
+
if current_event == _RESPONSE_ENDED_EVENT:
|
|
233
|
+
acc.stream_ended = True
|
|
234
|
+
continue
|
|
235
|
+
if not line.startswith(SSE_DATA_PREFIX):
|
|
236
|
+
continue
|
|
237
|
+
if current_event == _RESPONSE_ENDED_EVENT:
|
|
199
238
|
continue
|
|
200
239
|
data_str = line[len(SSE_DATA_PREFIX) :]
|
|
201
240
|
if _METADATA_SYNC_MARKER in data_str:
|
|
@@ -203,6 +242,7 @@ def parse_sse_lines(lines: Iterable[str]) -> ChatResult:
|
|
|
203
242
|
f"SSE transient error: {_METADATA_SYNC_MARKER}",
|
|
204
243
|
status_code=None,
|
|
205
244
|
detail=None,
|
|
245
|
+
partial_result=_build_chat_result(acc),
|
|
206
246
|
)
|
|
207
247
|
try:
|
|
208
248
|
event_data = json.loads(data_str)
|
|
@@ -213,8 +253,10 @@ def parse_sse_lines(lines: Iterable[str]) -> ChatResult:
|
|
|
213
253
|
detail = event_data.get("detail")
|
|
214
254
|
message = f"SSE error {code}: {detail}"
|
|
215
255
|
if code in _RETRYABLE_STATUS_CODES:
|
|
216
|
-
raise TransientChatError(
|
|
217
|
-
|
|
256
|
+
raise TransientChatError(
|
|
257
|
+
message, status_code=code, detail=detail, partial_result=_build_chat_result(acc)
|
|
258
|
+
)
|
|
259
|
+
raise ChatError(message, status_code=code, detail=detail, partial_result=_build_chat_result(acc))
|
|
218
260
|
if event_data.get("responseId") and not acc.response_id:
|
|
219
261
|
acc.response_id = event_data["responseId"]
|
|
220
262
|
item = event_data.get("item")
|
|
@@ -293,9 +335,20 @@ class ChatClient:
|
|
|
293
335
|
body["options"] = {"reasoningEffort": self._reasoning_effort}
|
|
294
336
|
|
|
295
337
|
def _do() -> ChatResult:
|
|
338
|
+
# Set fresh on every retry attempt (before opening this attempt's stream, so its
|
|
339
|
+
# own connection setup time counts) -- excludes not just the sleep backoff between
|
|
340
|
+
# attempts, but the entire duration of any earlier failed attempt.
|
|
341
|
+
t0 = time.monotonic()
|
|
296
342
|
with self._client.stream("POST", url, json=body, headers=headers) as resp:
|
|
297
343
|
resp.raise_for_status()
|
|
298
|
-
|
|
344
|
+
try:
|
|
345
|
+
result = parse_sse_lines(resp.iter_lines())
|
|
346
|
+
except ChatError as exc:
|
|
347
|
+
if exc.partial_result is not None:
|
|
348
|
+
exc.partial_result.turn_wall_clock_sec = time.monotonic() - t0
|
|
349
|
+
raise
|
|
350
|
+
result.turn_wall_clock_sec = time.monotonic() - t0
|
|
351
|
+
return result
|
|
299
352
|
|
|
300
353
|
return _retry_transient(_do, is_retryable=_is_retryable_exc)
|
|
301
354
|
|
|
@@ -100,6 +100,10 @@ class ChatResult(BaseModel):
|
|
|
100
100
|
reasoning_step_count: int = Field(default=0, alias="reasoningStepCount")
|
|
101
101
|
conversation_id: str | None = Field(default=None, alias="conversationId")
|
|
102
102
|
response_id: str | None = Field(default=None, alias="responseId")
|
|
103
|
+
# True once gen-ai's response_ended event arrived.
|
|
104
|
+
stream_ended: bool = False
|
|
105
|
+
# Wall-clock seconds for the whole chat turn, timed by the client.
|
|
106
|
+
turn_wall_clock_sec: float | None = None
|
|
103
107
|
|
|
104
108
|
|
|
105
109
|
class SummaryInput(BaseModel):
|