gooddata-eval 1.73.0__tar.gz → 1.73.1.dev1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/PKG-INFO +2 -2
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/pyproject.toml +2 -2
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/cli/agentic_runner.py +10 -6
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/agentic/alert_skill.py +38 -5
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/agentic/conversation.py +25 -17
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/agentic/general_question.py +40 -4
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/agentic/guardrail.py +41 -4
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/agentic/kda_skill.py +59 -6
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/agentic/metric_skill.py +40 -12
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/agentic/search_tool.py +40 -5
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/agentic/visualization.py +35 -5
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/chat/sse_client.py +14 -1
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/evaluators/visualization.py +28 -19
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/models.py +9 -1
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/tests/test_agentic_alert_skill.py +33 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/tests/test_agentic_conversation.py +77 -0
- gooddata_eval-1.73.1.dev1/tests/test_agentic_general_question.py +210 -0
- gooddata_eval-1.73.1.dev1/tests/test_agentic_guardrail.py +208 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/tests/test_agentic_kda_skill.py +147 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/tests/test_agentic_metric_skill.py +119 -1
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/tests/test_agentic_runner.py +70 -23
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/tests/test_agentic_search_tool.py +96 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/tests/test_agentic_visualization.py +130 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/tests/test_sse_client.py +42 -0
- gooddata_eval-1.73.0/tests/test_agentic_general_question.py +0 -100
- gooddata_eval-1.73.0/tests/test_agentic_guardrail.py +0 -98
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/.gitignore +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/LICENSE.txt +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/Makefile +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/README.md +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/__init__.py +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/_version.py +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/cli/__init__.py +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/cli/main.py +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/__init__.py +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/agentic/__init__.py +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/agentic/_catalog.py +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/agentic/_langfuse.py +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/chat/__init__.py +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/config.py +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/connection.py +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/dataset/__init__.py +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/dataset/langfuse_source.py +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/dataset/local.py +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/evaluators/__init__.py +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/evaluators/_deep_subset.py +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/evaluators/_llm_judge.py +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/evaluators/_text_utils.py +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/evaluators/alert_skill.py +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/evaluators/base.py +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/evaluators/general_question.py +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/evaluators/guardrail.py +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/evaluators/metric_skill.py +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/evaluators/search_tool.py +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/evaluators/summary.py +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/langfuse/__init__.py +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/langfuse/sink.py +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/reporting/__init__.py +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/reporting/console.py +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/reporting/json_report.py +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/runner.py +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/scoring.py +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/summary/__init__.py +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/summary/http_client.py +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/workspace.py +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/tests/__init__.py +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/tests/conftest.py +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/tests/fixtures/sample_dataset/metric_skill_create.json +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/tests/fixtures/sample_dataset/visualization_revenue.json +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/tests/fixtures/sse_visualization_stream.txt +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/tests/test_agentic_langfuse_trace.py +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/tests/test_agentic_run_context.py +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/tests/test_alert_skill_evaluator.py +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/tests/test_cli.py +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/tests/test_connection.py +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/tests/test_deep_subset.py +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/tests/test_langfuse_sink.py +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/tests/test_langfuse_source.py +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/tests/test_llm_judge.py +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/tests/test_local_loader.py +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/tests/test_metric_skill_evaluator.py +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/tests/test_models.py +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/tests/test_reporting.py +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/tests/test_runner.py +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/tests/test_scoring.py +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/tests/test_search_tool_evaluator.py +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/tests/test_summary_client.py +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/tests/test_summary_evaluator.py +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/tests/test_text_evaluators.py +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/tests/test_visualization_evaluator.py +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/tests/test_workspace.py +0 -0
- {gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/tox.ini +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: gooddata-eval
|
|
3
|
-
Version: 1.73.
|
|
3
|
+
Version: 1.73.1.dev1
|
|
4
4
|
Summary: Evaluate the GoodData AI agent against your own questions and models.
|
|
5
5
|
Project-URL: Source, https://github.com/gooddata/gooddata-python-sdk
|
|
6
6
|
Author-email: GoodData <support@gooddata.com>
|
|
@@ -17,7 +17,7 @@ Classifier: Topic :: Scientific/Engineering
|
|
|
17
17
|
Classifier: Topic :: Software Development
|
|
18
18
|
Classifier: Typing :: Typed
|
|
19
19
|
Requires-Python: >=3.10
|
|
20
|
-
Requires-Dist: gooddata-sdk~=1.73.
|
|
20
|
+
Requires-Dist: gooddata-sdk~=1.73.1.dev1
|
|
21
21
|
Requires-Dist: httpx<1.0,>=0.27
|
|
22
22
|
Requires-Dist: orjson<4.0.0,>=3.9.15
|
|
23
23
|
Requires-Dist: pydantic<3.0,>=2.6
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
# (C) 2026 GoodData Corporation
|
|
2
2
|
[project]
|
|
3
3
|
name = "gooddata-eval"
|
|
4
|
-
version = "1.73.
|
|
4
|
+
version = "1.73.1.dev1"
|
|
5
5
|
description = "Evaluate the GoodData AI agent against your own questions and models."
|
|
6
6
|
readme = "README.md"
|
|
7
7
|
license = "MIT"
|
|
@@ -11,7 +11,7 @@ authors = [
|
|
|
11
11
|
keywords = ["gooddata", "ai", "evaluation", "llm", "analytics", "cli"]
|
|
12
12
|
requires-python = ">=3.10"
|
|
13
13
|
dependencies = [
|
|
14
|
-
"gooddata-sdk~=1.73.
|
|
14
|
+
"gooddata-sdk~=1.73.1.dev1",
|
|
15
15
|
"httpx>=0.27,<1.0",
|
|
16
16
|
"orjson>=3.9.15,<4.0.0",
|
|
17
17
|
"pydantic>=2.6,<3.0",
|
|
@@ -86,12 +86,12 @@ def _dispatch_agentic(
|
|
|
86
86
|
model_version_override: str | None,
|
|
87
87
|
reasoning_effort: ReasoningEffort | None = None,
|
|
88
88
|
agent_id: str | None = None,
|
|
89
|
-
) -> AgenticEvalOutcome
|
|
89
|
+
) -> AgenticEvalOutcome:
|
|
90
90
|
"""Call the appropriate evaluate_agentic_* function for the item's test_kind.
|
|
91
91
|
|
|
92
|
-
|
|
93
|
-
|
|
94
|
-
|
|
92
|
+
Every evaluate_agentic_* function returns an AgenticEvalOutcome (reasoning_steps,
|
|
93
|
+
conversation_id, response_id, detail) on success and attaches the same four attributes
|
|
94
|
+
to its raised *AssertionError on failure -- no kind is exempt.
|
|
95
95
|
"""
|
|
96
96
|
kind = item.test_kind
|
|
97
97
|
eo = item.expected_output
|
|
@@ -174,13 +174,14 @@ def _dispatch_agentic(
|
|
|
174
174
|
**lf_kw,
|
|
175
175
|
)
|
|
176
176
|
elif kind == "agentic_kda_skill":
|
|
177
|
-
evaluate_agentic_kda_skill(
|
|
177
|
+
return evaluate_agentic_kda_skill(
|
|
178
178
|
host=host,
|
|
179
179
|
token=token,
|
|
180
180
|
workspace_id=workspace_id,
|
|
181
181
|
question=item.question,
|
|
182
182
|
expected_output=eo if isinstance(eo, dict) else {},
|
|
183
183
|
k=k,
|
|
184
|
+
agent_id=agent_id,
|
|
184
185
|
**lf_kw,
|
|
185
186
|
)
|
|
186
187
|
elif kind == "agentic_conversation":
|
|
@@ -240,19 +241,22 @@ def run_agentic_items(
|
|
|
240
241
|
reasoning_steps = outcome.reasoning_steps
|
|
241
242
|
conversation_id = outcome.conversation_id
|
|
242
243
|
response_id = outcome.response_id
|
|
244
|
+
detail = outcome.detail
|
|
243
245
|
else:
|
|
244
|
-
reasoning_steps, conversation_id, response_id = outcome, None, None
|
|
246
|
+
reasoning_steps, conversation_id, response_id, detail = outcome, None, None, {}
|
|
245
247
|
item_report.pass_at_k = True
|
|
246
248
|
item_report.runs = k
|
|
247
249
|
item_report.reasoning_steps = reasoning_steps or []
|
|
248
250
|
item_report.conversation_id = conversation_id
|
|
249
251
|
item_report.response_id = response_id
|
|
252
|
+
item_report.best_detail = detail or {}
|
|
250
253
|
except AssertionError as exc:
|
|
251
254
|
item_report.pass_at_k = False
|
|
252
255
|
item_report.runs = k
|
|
253
256
|
item_report.reasoning_steps = getattr(exc, "reasoning_steps", None) or []
|
|
254
257
|
item_report.conversation_id = getattr(exc, "conversation_id", None)
|
|
255
258
|
item_report.response_id = getattr(exc, "response_id", None)
|
|
259
|
+
item_report.best_detail = getattr(exc, "detail", None) or {}
|
|
256
260
|
print(f"[agentic] {item.id} FAIL: {exc}", flush=True)
|
|
257
261
|
except Exception as exc:
|
|
258
262
|
item_report.error = f"{type(exc).__name__}: {exc}"
|
{gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/agentic/alert_skill.py
RENAMED
|
@@ -160,8 +160,18 @@ def _check_recipients(expected: CatalogMetricAlert, actual_args: dict, sdk: Good
|
|
|
160
160
|
act_recip = []
|
|
161
161
|
if set(expected.recipients) == set(act_recip or []):
|
|
162
162
|
return True
|
|
163
|
-
|
|
164
|
-
|
|
163
|
+
act_internal_raw = actual_args.get("internal_recipients")
|
|
164
|
+
# internal_recipients is declared `anyOf: [array of string, string, null]` in the
|
|
165
|
+
# create_metric_alert tool schema -- a single id as a bare string is schema-legal,
|
|
166
|
+
# not a malformed call, so it needs the same string/list normalization already
|
|
167
|
+
# applied to recipients/external_recipients above.
|
|
168
|
+
if isinstance(act_internal_raw, str):
|
|
169
|
+
act_internal = [act_internal_raw]
|
|
170
|
+
elif isinstance(act_internal_raw, list):
|
|
171
|
+
act_internal = act_internal_raw
|
|
172
|
+
else:
|
|
173
|
+
act_internal = []
|
|
174
|
+
if sdk is not None and act_internal:
|
|
165
175
|
internal_recipient_ids = _resolve_internal_recipient_ids(sdk, expected.recipients)
|
|
166
176
|
if internal_recipient_ids & set(act_internal):
|
|
167
177
|
return True
|
|
@@ -584,6 +594,7 @@ class AlertSkillAssertionError(AssertionError):
|
|
|
584
594
|
reasoning_steps: list[str]
|
|
585
595
|
conversation_id: str
|
|
586
596
|
response_id: str | None
|
|
597
|
+
detail: dict
|
|
587
598
|
|
|
588
599
|
|
|
589
600
|
def evaluate_agentic_alert_skill(
|
|
@@ -697,9 +708,31 @@ def evaluate_agentic_alert_skill(
|
|
|
697
708
|
exc.reasoning_steps = best.reasoning_steps
|
|
698
709
|
exc.conversation_id = best.conversation_id
|
|
699
710
|
exc.response_id = best.response_id
|
|
711
|
+
exc.detail = {
|
|
712
|
+
"alert_created": ev.alert_created,
|
|
713
|
+
"operator_correct": ev.operator_correct,
|
|
714
|
+
"threshold_correct": ev.threshold_correct,
|
|
715
|
+
"trigger_correct": ev.trigger_correct,
|
|
716
|
+
"filters_correct": ev.filters_correct,
|
|
717
|
+
"metric_correct": ev.metric_correct,
|
|
718
|
+
"recipients_correct": ev.recipients_correct,
|
|
719
|
+
"actual_alert_arguments": best.actual_alert_arguments,
|
|
720
|
+
}
|
|
700
721
|
raise exc
|
|
722
|
+
best = summary.best
|
|
723
|
+
ev = best.eval
|
|
701
724
|
return AgenticEvalOutcome(
|
|
702
|
-
reasoning_steps=
|
|
703
|
-
conversation_id=
|
|
704
|
-
response_id=
|
|
725
|
+
reasoning_steps=best.reasoning_steps,
|
|
726
|
+
conversation_id=best.conversation_id,
|
|
727
|
+
response_id=best.response_id,
|
|
728
|
+
detail={
|
|
729
|
+
"alert_created": ev.alert_created,
|
|
730
|
+
"operator_correct": ev.operator_correct,
|
|
731
|
+
"threshold_correct": ev.threshold_correct,
|
|
732
|
+
"trigger_correct": ev.trigger_correct,
|
|
733
|
+
"filters_correct": ev.filters_correct,
|
|
734
|
+
"metric_correct": ev.metric_correct,
|
|
735
|
+
"recipients_correct": ev.recipients_correct,
|
|
736
|
+
"actual_alert_arguments": best.actual_alert_arguments,
|
|
737
|
+
},
|
|
705
738
|
)
|
{gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/agentic/conversation.py
RENAMED
|
@@ -12,7 +12,7 @@ from gooddata_sdk import GoodDataSdk
|
|
|
12
12
|
from pydantic import BaseModel
|
|
13
13
|
|
|
14
14
|
from gooddata_eval.core.agentic.alert_skill import render_alert_proposal
|
|
15
|
-
from gooddata_eval.core.agentic.metric_skill import _delete_metric, _extract_created_metric_ids
|
|
15
|
+
from gooddata_eval.core.agentic.metric_skill import _delete_metric, _extract_created_metric_ids, _extract_metric_result
|
|
16
16
|
from gooddata_eval.core.chat.sse_client import ChatClient
|
|
17
17
|
from gooddata_eval.core.config import ReasoningEffort
|
|
18
18
|
from gooddata_eval.core.models import AgenticEvalOutcome, ChatResult, ToolCallEvent
|
|
@@ -120,7 +120,7 @@ def _check_output_present(turn: TurnDefinition, chat_result: ChatResult) -> bool
|
|
|
120
120
|
and getattr(chat_result.created_visualizations, "objects", chat_result.created_visualizations)
|
|
121
121
|
)
|
|
122
122
|
if otype == "metric":
|
|
123
|
-
return
|
|
123
|
+
return _extract_metric_result(chat_result.tool_call_events or []) is not None
|
|
124
124
|
if otype == "tool_call":
|
|
125
125
|
expected_tool = turn.expected_tool_name
|
|
126
126
|
if not expected_tool:
|
|
@@ -129,19 +129,6 @@ def _check_output_present(turn: TurnDefinition, chat_result: ChatResult) -> bool
|
|
|
129
129
|
return False
|
|
130
130
|
|
|
131
131
|
|
|
132
|
-
def _extract_metric_from_turn(tool_call_events: list[ToolCallEvent]) -> dict | None:
|
|
133
|
-
"""Extract the result payload from the create_metric tool call, if present."""
|
|
134
|
-
for tc in tool_call_events:
|
|
135
|
-
if tc.function_name != "create_metric":
|
|
136
|
-
continue
|
|
137
|
-
if not tc.result:
|
|
138
|
-
continue
|
|
139
|
-
result_data = tc.parsed_result()
|
|
140
|
-
if result_data is not None:
|
|
141
|
-
return result_data.get("data", result_data)
|
|
142
|
-
return None
|
|
143
|
-
|
|
144
|
-
|
|
145
132
|
def _check_output_correct(turn: TurnDefinition, chat_result: ChatResult) -> bool | None:
|
|
146
133
|
"""Check output correctness against expected_output when defined.
|
|
147
134
|
|
|
@@ -186,7 +173,7 @@ def _check_output_correct(turn: TurnDefinition, chat_result: ChatResult) -> bool
|
|
|
186
173
|
return all(results) if results else None
|
|
187
174
|
|
|
188
175
|
if otype == "metric":
|
|
189
|
-
metric_result =
|
|
176
|
+
metric_result = _extract_metric_result(chat_result.tool_call_events or [])
|
|
190
177
|
if not metric_result:
|
|
191
178
|
return False
|
|
192
179
|
return _normalize_maql(metric_result.get("maql", "")) == _normalize_maql(expected.get("maql", ""))
|
|
@@ -362,7 +349,7 @@ def run_agentic_conversation(
|
|
|
362
349
|
|
|
363
350
|
# Capture metric output for $ref resolution in subsequent turns.
|
|
364
351
|
if final_result and turn.expected_output_type == "metric":
|
|
365
|
-
metric_data =
|
|
352
|
+
metric_data = _extract_metric_result(all_tool_calls)
|
|
366
353
|
if metric_data:
|
|
367
354
|
turn_outputs[turn.turn_id] = metric_data
|
|
368
355
|
|
|
@@ -406,6 +393,24 @@ def run_agentic_conversation(
|
|
|
406
393
|
)
|
|
407
394
|
|
|
408
395
|
|
|
396
|
+
def _conversation_detail(result: ConversationResult) -> dict:
|
|
397
|
+
return {
|
|
398
|
+
"full_skill_coverage": result.full_skill_coverage,
|
|
399
|
+
"total_clarification_turns": result.total_clarification_turns,
|
|
400
|
+
"turns": [
|
|
401
|
+
{
|
|
402
|
+
"turn_id": tr.turn_id,
|
|
403
|
+
"expected_skill": tr.expected_skill,
|
|
404
|
+
"skill_routing": tr.skill_routing,
|
|
405
|
+
"output_present": tr.output_present,
|
|
406
|
+
"output_correct": tr.output_correct,
|
|
407
|
+
"activated_skills": tr.activated_skills,
|
|
408
|
+
}
|
|
409
|
+
for tr in result.turn_results
|
|
410
|
+
],
|
|
411
|
+
}
|
|
412
|
+
|
|
413
|
+
|
|
409
414
|
class ConversationAssertionError(AssertionError):
|
|
410
415
|
"""Raised when a conversation evaluation fails."""
|
|
411
416
|
|
|
@@ -413,6 +418,7 @@ class ConversationAssertionError(AssertionError):
|
|
|
413
418
|
reasoning_steps: list[str]
|
|
414
419
|
conversation_id: str
|
|
415
420
|
response_id: str | None
|
|
421
|
+
detail: dict
|
|
416
422
|
|
|
417
423
|
|
|
418
424
|
def evaluate_agentic_conversation(
|
|
@@ -524,9 +530,11 @@ def evaluate_agentic_conversation(
|
|
|
524
530
|
exc.reasoning_steps = result.reasoning_steps
|
|
525
531
|
exc.conversation_id = result.conversation_id
|
|
526
532
|
exc.response_id = result.response_id
|
|
533
|
+
exc.detail = _conversation_detail(result)
|
|
527
534
|
raise exc
|
|
528
535
|
return AgenticEvalOutcome(
|
|
529
536
|
reasoning_steps=result.reasoning_steps,
|
|
530
537
|
conversation_id=result.conversation_id,
|
|
531
538
|
response_id=result.response_id,
|
|
539
|
+
detail=_conversation_detail(result),
|
|
532
540
|
)
|
|
@@ -3,11 +3,12 @@
|
|
|
3
3
|
|
|
4
4
|
from __future__ import annotations
|
|
5
5
|
|
|
6
|
-
from dataclasses import dataclass
|
|
6
|
+
from dataclasses import dataclass, field
|
|
7
7
|
|
|
8
8
|
from gooddata_eval.core.chat.sse_client import ChatClient
|
|
9
9
|
from gooddata_eval.core.config import ReasoningEffort
|
|
10
10
|
from gooddata_eval.core.evaluators._llm_judge import LLMJudge
|
|
11
|
+
from gooddata_eval.core.models import AgenticEvalOutcome
|
|
11
12
|
|
|
12
13
|
_DEFAULT_K = 1
|
|
13
14
|
|
|
@@ -52,6 +53,8 @@ class GeneralQuestionResult:
|
|
|
52
53
|
passed: bool
|
|
53
54
|
llm_judge_score: float
|
|
54
55
|
reasoning: str
|
|
56
|
+
reasoning_steps: list[str] = field(default_factory=list)
|
|
57
|
+
response_id: str | None = None
|
|
55
58
|
|
|
56
59
|
|
|
57
60
|
@dataclass
|
|
@@ -98,6 +101,8 @@ def run_agentic_general_question(
|
|
|
98
101
|
passed=passed,
|
|
99
102
|
llm_judge_score=llm_judge_score,
|
|
100
103
|
reasoning=reasoning,
|
|
104
|
+
reasoning_steps=list(chat_result.reasoning_steps or []),
|
|
105
|
+
response_id=chat_result.response_id,
|
|
101
106
|
)
|
|
102
107
|
)
|
|
103
108
|
finally:
|
|
@@ -120,6 +125,8 @@ def run_agentic_general_question(
|
|
|
120
125
|
passed=passed,
|
|
121
126
|
llm_judge_score=llm_judge_score,
|
|
122
127
|
reasoning=reasoning,
|
|
128
|
+
reasoning_steps=list(chat_result.reasoning_steps or []),
|
|
129
|
+
response_id=chat_result.response_id,
|
|
123
130
|
)
|
|
124
131
|
)
|
|
125
132
|
finally:
|
|
@@ -142,6 +149,10 @@ class GeneralQuestionAssertionError(AssertionError):
|
|
|
142
149
|
"""Raised when a general-question evaluation fails."""
|
|
143
150
|
|
|
144
151
|
__tracebackhide__ = True
|
|
152
|
+
reasoning_steps: list[str]
|
|
153
|
+
conversation_id: str
|
|
154
|
+
response_id: str | None
|
|
155
|
+
detail: dict
|
|
145
156
|
|
|
146
157
|
|
|
147
158
|
def evaluate_agentic_general_question(
|
|
@@ -160,8 +171,13 @@ def evaluate_agentic_general_question(
|
|
|
160
171
|
model_version_override: str | None = None,
|
|
161
172
|
run_metadata_extra: dict | None = None,
|
|
162
173
|
reasoning_effort: ReasoningEffort | None = None,
|
|
163
|
-
) ->
|
|
164
|
-
"""Run general-question evaluation, log to Langfuse, and raise on failure.
|
|
174
|
+
) -> AgenticEvalOutcome:
|
|
175
|
+
"""Run general-question evaluation, log to Langfuse, and raise GeneralQuestionAssertionError on failure.
|
|
176
|
+
|
|
177
|
+
Returns the best run's outcome (reasoning_steps, conversation_id, response_id) as an
|
|
178
|
+
AgenticEvalOutcome on success; on failure the same three values are attached to the
|
|
179
|
+
raised exception as ``.reasoning_steps``/``.conversation_id``/``.response_id``.
|
|
180
|
+
"""
|
|
165
181
|
from datetime import datetime as _dt # noqa: PLC0415
|
|
166
182
|
from datetime import timezone as _tz # noqa: PLC0415
|
|
167
183
|
|
|
@@ -223,6 +239,26 @@ def evaluate_agentic_general_question(
|
|
|
223
239
|
|
|
224
240
|
if not summary.pass_at_k:
|
|
225
241
|
best = summary.best
|
|
226
|
-
|
|
242
|
+
exc = GeneralQuestionAssertionError(
|
|
227
243
|
f"General question assertion failed. passed={best.passed}. Reasoning: {best.reasoning}"
|
|
228
244
|
)
|
|
245
|
+
exc.reasoning_steps = best.reasoning_steps
|
|
246
|
+
exc.conversation_id = best.conversation_id
|
|
247
|
+
exc.response_id = best.response_id
|
|
248
|
+
exc.detail = {
|
|
249
|
+
"judge_passed": best.passed,
|
|
250
|
+
"judge_reasoning": best.reasoning,
|
|
251
|
+
"actual_output": best.actual_output,
|
|
252
|
+
}
|
|
253
|
+
raise exc
|
|
254
|
+
best = summary.best
|
|
255
|
+
return AgenticEvalOutcome(
|
|
256
|
+
reasoning_steps=best.reasoning_steps,
|
|
257
|
+
conversation_id=best.conversation_id,
|
|
258
|
+
response_id=best.response_id,
|
|
259
|
+
detail={
|
|
260
|
+
"judge_passed": best.passed,
|
|
261
|
+
"judge_reasoning": best.reasoning,
|
|
262
|
+
"actual_output": best.actual_output,
|
|
263
|
+
},
|
|
264
|
+
)
|
{gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/agentic/guardrail.py
RENAMED
|
@@ -3,11 +3,12 @@
|
|
|
3
3
|
|
|
4
4
|
from __future__ import annotations
|
|
5
5
|
|
|
6
|
-
from dataclasses import dataclass
|
|
6
|
+
from dataclasses import dataclass, field
|
|
7
7
|
|
|
8
8
|
from gooddata_eval.core.chat.sse_client import ChatClient
|
|
9
9
|
from gooddata_eval.core.config import ReasoningEffort
|
|
10
10
|
from gooddata_eval.core.evaluators._llm_judge import LLMJudge
|
|
11
|
+
from gooddata_eval.core.models import AgenticEvalOutcome
|
|
11
12
|
|
|
12
13
|
_DEFAULT_K = 1
|
|
13
14
|
|
|
@@ -49,6 +50,8 @@ class GuardrailResult:
|
|
|
49
50
|
passed: bool
|
|
50
51
|
llm_judge_score: float
|
|
51
52
|
reasoning: str
|
|
53
|
+
reasoning_steps: list[str] = field(default_factory=list)
|
|
54
|
+
response_id: str | None = None
|
|
52
55
|
|
|
53
56
|
|
|
54
57
|
@dataclass
|
|
@@ -95,6 +98,8 @@ def run_agentic_guardrail(
|
|
|
95
98
|
passed=passed,
|
|
96
99
|
llm_judge_score=llm_judge_score,
|
|
97
100
|
reasoning=reasoning,
|
|
101
|
+
reasoning_steps=list(chat_result.reasoning_steps or []),
|
|
102
|
+
response_id=chat_result.response_id,
|
|
98
103
|
)
|
|
99
104
|
)
|
|
100
105
|
finally:
|
|
@@ -117,6 +122,8 @@ def run_agentic_guardrail(
|
|
|
117
122
|
passed=passed,
|
|
118
123
|
llm_judge_score=llm_judge_score,
|
|
119
124
|
reasoning=reasoning,
|
|
125
|
+
reasoning_steps=list(chat_result.reasoning_steps or []),
|
|
126
|
+
response_id=chat_result.response_id,
|
|
120
127
|
)
|
|
121
128
|
)
|
|
122
129
|
finally:
|
|
@@ -139,6 +146,10 @@ class GuardrailAssertionError(AssertionError):
|
|
|
139
146
|
"""Raised when a guardrail evaluation fails."""
|
|
140
147
|
|
|
141
148
|
__tracebackhide__ = True
|
|
149
|
+
reasoning_steps: list[str]
|
|
150
|
+
conversation_id: str
|
|
151
|
+
response_id: str | None
|
|
152
|
+
detail: dict
|
|
142
153
|
|
|
143
154
|
|
|
144
155
|
def evaluate_agentic_guardrail(
|
|
@@ -157,8 +168,14 @@ def evaluate_agentic_guardrail(
|
|
|
157
168
|
model_version_override: str | None = None,
|
|
158
169
|
run_metadata_extra: dict | None = None,
|
|
159
170
|
reasoning_effort: ReasoningEffort | None = None,
|
|
160
|
-
) ->
|
|
161
|
-
"""Run guardrail evaluation, log to Langfuse, and raise on failure.
|
|
171
|
+
) -> AgenticEvalOutcome:
|
|
172
|
+
"""Run guardrail evaluation, log to Langfuse, and raise GuardrailAssertionError on failure.
|
|
173
|
+
|
|
174
|
+
Returns the best run's outcome (reasoning_steps, conversation_id, response_id) as an
|
|
175
|
+
AgenticEvalOutcome on success; on failure the same three values are attached to the
|
|
176
|
+
raised exception as ``.reasoning_steps``/``.conversation_id``/``.response_id`` (mirrors
|
|
177
|
+
`evaluate_agentic_metric_skill`'s idiom) so callers can retrieve them either way.
|
|
178
|
+
"""
|
|
162
179
|
from datetime import datetime as _dt # noqa: PLC0415
|
|
163
180
|
from datetime import timezone as _tz # noqa: PLC0415
|
|
164
181
|
|
|
@@ -220,4 +237,24 @@ def evaluate_agentic_guardrail(
|
|
|
220
237
|
|
|
221
238
|
if not summary.pass_at_k:
|
|
222
239
|
best = summary.best
|
|
223
|
-
|
|
240
|
+
exc = GuardrailAssertionError(f"Guardrail assertion failed. passed={best.passed}. Reasoning: {best.reasoning}")
|
|
241
|
+
exc.reasoning_steps = best.reasoning_steps
|
|
242
|
+
exc.conversation_id = best.conversation_id
|
|
243
|
+
exc.response_id = best.response_id
|
|
244
|
+
exc.detail = {
|
|
245
|
+
"judge_passed": best.passed,
|
|
246
|
+
"judge_reasoning": best.reasoning,
|
|
247
|
+
"actual_output": best.actual_output,
|
|
248
|
+
}
|
|
249
|
+
raise exc
|
|
250
|
+
best = summary.best
|
|
251
|
+
return AgenticEvalOutcome(
|
|
252
|
+
reasoning_steps=best.reasoning_steps,
|
|
253
|
+
conversation_id=best.conversation_id,
|
|
254
|
+
response_id=best.response_id,
|
|
255
|
+
detail={
|
|
256
|
+
"judge_passed": best.passed,
|
|
257
|
+
"judge_reasoning": best.reasoning,
|
|
258
|
+
"actual_output": best.actual_output,
|
|
259
|
+
},
|
|
260
|
+
)
|
{gooddata_eval-1.73.0 → gooddata_eval-1.73.1.dev1}/src/gooddata_eval/core/agentic/kda_skill.py
RENAMED
|
@@ -5,11 +5,11 @@ from __future__ import annotations
|
|
|
5
5
|
|
|
6
6
|
import logging
|
|
7
7
|
import os
|
|
8
|
-
from dataclasses import dataclass
|
|
8
|
+
from dataclasses import dataclass, field
|
|
9
9
|
|
|
10
10
|
from gooddata_eval.core.chat.sse_client import ChatClient
|
|
11
11
|
from gooddata_eval.core.config import ReasoningEffort
|
|
12
|
-
from gooddata_eval.core.models import ToolCallEvent
|
|
12
|
+
from gooddata_eval.core.models import AgenticEvalOutcome, ToolCallEvent
|
|
13
13
|
|
|
14
14
|
_log = logging.getLogger(__name__)
|
|
15
15
|
|
|
@@ -159,6 +159,8 @@ class KdaRunResult:
|
|
|
159
159
|
# Wall-clock time of the turn that called create (None if create never happened) --
|
|
160
160
|
# not any earlier disambiguation turn. See run_agentic_kda_skill's _run_once.
|
|
161
161
|
turn_wall_clock_sec: float | None = None
|
|
162
|
+
reasoning_steps: list[str] = field(default_factory=list)
|
|
163
|
+
response_id: str | None = None
|
|
162
164
|
|
|
163
165
|
|
|
164
166
|
@dataclass
|
|
@@ -199,6 +201,7 @@ def run_agentic_kda_skill(
|
|
|
199
201
|
max_iterations: int = _DEFAULT_MAX_ITERATIONS,
|
|
200
202
|
initial_conversation_id: str | None = None,
|
|
201
203
|
reasoning_effort: ReasoningEffort | None = None,
|
|
204
|
+
agent_id: str | None = None,
|
|
202
205
|
) -> AgenticKdaSummary:
|
|
203
206
|
"""Run the KDA-skill agentic evaluation K times and return a summary.
|
|
204
207
|
|
|
@@ -215,7 +218,9 @@ def run_agentic_kda_skill(
|
|
|
215
218
|
# k=0 or negative would otherwise silently run once, indistinguishable from k=1.
|
|
216
219
|
raise ValueError(f"k must be >= 1, got {k}")
|
|
217
220
|
run_results: list[KdaRunResult] = []
|
|
218
|
-
client = ChatClient(
|
|
221
|
+
client = ChatClient(
|
|
222
|
+
host=host, token=token, workspace_id=workspace_id, reasoning_effort=reasoning_effort, agent_id=agent_id
|
|
223
|
+
)
|
|
219
224
|
|
|
220
225
|
def _run_once(conv_id: str) -> KdaRunResult:
|
|
221
226
|
create_args: dict | None = None
|
|
@@ -224,6 +229,8 @@ def run_agentic_kda_skill(
|
|
|
224
229
|
turn_completed = False
|
|
225
230
|
disambiguated = False
|
|
226
231
|
current_question = question
|
|
232
|
+
reasoning_steps: list[str] = []
|
|
233
|
+
response_id: str | None = None
|
|
227
234
|
|
|
228
235
|
for iteration in range(max_iterations):
|
|
229
236
|
try:
|
|
@@ -232,11 +239,15 @@ def run_agentic_kda_skill(
|
|
|
232
239
|
_log.warning("KDA send_message failed for conversation %s: %s", conv_id, exc)
|
|
233
240
|
partial = getattr(exc, "partial_result", None)
|
|
234
241
|
if partial is not None:
|
|
242
|
+
reasoning_steps.extend(partial.reasoning_steps or [])
|
|
243
|
+
response_id = partial.response_id or response_id
|
|
235
244
|
create_args, execute_result = _extract_kda_calls(partial.tool_call_events or [])
|
|
236
245
|
if create_args is not None:
|
|
237
246
|
turn_wall_clock_sec = partial.turn_wall_clock_sec
|
|
238
247
|
turn_completed = False
|
|
239
248
|
break
|
|
249
|
+
reasoning_steps.extend(chat_result.reasoning_steps or [])
|
|
250
|
+
response_id = chat_result.response_id or response_id
|
|
240
251
|
create_args, execute_result = _extract_kda_calls(chat_result.tool_call_events or [])
|
|
241
252
|
response_text = (chat_result.text_response or "").strip()
|
|
242
253
|
turn_completed = chat_result.stream_ended and bool(response_text)
|
|
@@ -273,6 +284,8 @@ def run_agentic_kda_skill(
|
|
|
273
284
|
actual_create_args=create_args,
|
|
274
285
|
actual_execute_result=execute_result,
|
|
275
286
|
turn_wall_clock_sec=turn_wall_clock_sec,
|
|
287
|
+
reasoning_steps=reasoning_steps,
|
|
288
|
+
response_id=response_id,
|
|
276
289
|
)
|
|
277
290
|
|
|
278
291
|
try:
|
|
@@ -312,6 +325,10 @@ class KdaSkillAssertionError(AssertionError):
|
|
|
312
325
|
"""Raised when a KDA-skill evaluation fails."""
|
|
313
326
|
|
|
314
327
|
__tracebackhide__ = True
|
|
328
|
+
reasoning_steps: list[str]
|
|
329
|
+
conversation_id: str
|
|
330
|
+
response_id: str | None
|
|
331
|
+
detail: dict
|
|
315
332
|
|
|
316
333
|
|
|
317
334
|
def evaluate_agentic_kda_skill(
|
|
@@ -323,6 +340,7 @@ def evaluate_agentic_kda_skill(
|
|
|
323
340
|
k: int = _DEFAULT_K,
|
|
324
341
|
max_iterations: int = _DEFAULT_MAX_ITERATIONS,
|
|
325
342
|
initial_conversation_id: str | None = None,
|
|
343
|
+
agent_id: str | None = None,
|
|
326
344
|
langfuse: object | None = None,
|
|
327
345
|
dataset_item_id: str = "",
|
|
328
346
|
dataset_name: str = "kda_skill",
|
|
@@ -330,8 +348,13 @@ def evaluate_agentic_kda_skill(
|
|
|
330
348
|
model_version_override: str | None = None,
|
|
331
349
|
run_metadata_extra: dict | None = None,
|
|
332
350
|
reasoning_effort: ReasoningEffort | None = None,
|
|
333
|
-
) ->
|
|
334
|
-
"""Run KDA-skill evaluation, log to Langfuse, and raise KdaSkillAssertionError on failure.
|
|
351
|
+
) -> AgenticEvalOutcome:
|
|
352
|
+
"""Run KDA-skill evaluation, log to Langfuse, and raise KdaSkillAssertionError on failure.
|
|
353
|
+
|
|
354
|
+
Returns the best run's outcome (reasoning_steps, conversation_id, response_id) as an
|
|
355
|
+
AgenticEvalOutcome on success; on failure the same three values are attached to the
|
|
356
|
+
raised exception as ``.reasoning_steps``/``.conversation_id``/``.response_id``.
|
|
357
|
+
"""
|
|
335
358
|
from datetime import datetime as _dt # noqa: PLC0415
|
|
336
359
|
from datetime import timezone as _tz # noqa: PLC0415
|
|
337
360
|
|
|
@@ -350,6 +373,7 @@ def evaluate_agentic_kda_skill(
|
|
|
350
373
|
max_iterations=max_iterations,
|
|
351
374
|
initial_conversation_id=initial_conversation_id,
|
|
352
375
|
reasoning_effort=reasoning_effort,
|
|
376
|
+
agent_id=agent_id,
|
|
353
377
|
)
|
|
354
378
|
|
|
355
379
|
if langfuse is not None and dataset_item_id:
|
|
@@ -420,4 +444,33 @@ def evaluate_agentic_kda_skill(
|
|
|
420
444
|
f"Actual create args: {best.actual_create_args}. "
|
|
421
445
|
f"Actual execute result: {best.actual_execute_result}."
|
|
422
446
|
)
|
|
423
|
-
|
|
447
|
+
exc = KdaSkillAssertionError(message)
|
|
448
|
+
exc.reasoning_steps = best.reasoning_steps
|
|
449
|
+
exc.conversation_id = best.conversation_id
|
|
450
|
+
exc.response_id = best.response_id
|
|
451
|
+
exc.detail = {
|
|
452
|
+
"triggered": ev.triggered,
|
|
453
|
+
"executed": ev.executed,
|
|
454
|
+
"success": ev.success,
|
|
455
|
+
"turn_completed": ev.turn_completed,
|
|
456
|
+
"disambiguated": ev.disambiguated,
|
|
457
|
+
"actual_create_args": best.actual_create_args,
|
|
458
|
+
"actual_execute_result": best.actual_execute_result,
|
|
459
|
+
}
|
|
460
|
+
raise exc
|
|
461
|
+
best = summary.best
|
|
462
|
+
ev = best.evaluation
|
|
463
|
+
return AgenticEvalOutcome(
|
|
464
|
+
reasoning_steps=best.reasoning_steps,
|
|
465
|
+
conversation_id=best.conversation_id,
|
|
466
|
+
response_id=best.response_id,
|
|
467
|
+
detail={
|
|
468
|
+
"triggered": ev.triggered,
|
|
469
|
+
"executed": ev.executed,
|
|
470
|
+
"success": ev.success,
|
|
471
|
+
"turn_completed": ev.turn_completed,
|
|
472
|
+
"disambiguated": ev.disambiguated,
|
|
473
|
+
"actual_create_args": best.actual_create_args,
|
|
474
|
+
"actual_execute_result": best.actual_execute_result,
|
|
475
|
+
},
|
|
476
|
+
)
|