gooddata-eval 1.71.0__tar.gz → 1.71.1.dev2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/PKG-INFO +4 -3
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/README.md +2 -1
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/pyproject.toml +2 -2
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/cli/agentic_runner.py +6 -1
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/cli/main.py +17 -1
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/agentic/_langfuse.py +12 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/agentic/alert_skill.py +28 -6
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/agentic/conversation.py +11 -2
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/agentic/general_question.py +6 -1
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/agentic/guardrail.py +6 -1
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/agentic/metric_skill.py +6 -1
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/agentic/search_tool.py +6 -1
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/agentic/visualization.py +6 -1
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/chat/sse_client.py +28 -2
- gooddata_eval-1.71.1.dev2/src/gooddata_eval/core/config.py +46 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/langfuse/sink.py +21 -2
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/models.py +4 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/tests/test_agentic_alert_skill.py +92 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/tests/test_agentic_conversation.py +62 -1
- gooddata_eval-1.71.1.dev2/tests/test_agentic_run_context.py +73 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/tests/test_cli.py +77 -3
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/tests/test_sse_client.py +115 -2
- gooddata_eval-1.71.0/src/gooddata_eval/core/config.py +0 -22
- gooddata_eval-1.71.0/tests/test_agentic_run_context.py +0 -32
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/.gitignore +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/LICENSE.txt +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/Makefile +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/__init__.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/_version.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/cli/__init__.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/__init__.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/agentic/__init__.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/agentic/_catalog.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/chat/__init__.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/connection.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/dataset/__init__.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/dataset/langfuse_source.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/dataset/local.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/evaluators/__init__.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/evaluators/_deep_subset.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/evaluators/_llm_judge.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/evaluators/_text_utils.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/evaluators/alert_skill.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/evaluators/base.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/evaluators/general_question.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/evaluators/guardrail.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/evaluators/metric_skill.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/evaluators/search_tool.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/evaluators/summary.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/evaluators/visualization.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/langfuse/__init__.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/reporting/__init__.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/reporting/console.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/reporting/json_report.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/runner.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/scoring.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/summary/__init__.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/summary/http_client.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/workspace.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/tests/__init__.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/tests/conftest.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/tests/fixtures/sample_dataset/metric_skill_create.json +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/tests/fixtures/sample_dataset/visualization_revenue.json +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/tests/fixtures/sse_visualization_stream.txt +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/tests/test_agentic_general_question.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/tests/test_agentic_guardrail.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/tests/test_agentic_metric_skill.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/tests/test_agentic_search_tool.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/tests/test_agentic_visualization.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/tests/test_alert_skill_evaluator.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/tests/test_connection.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/tests/test_deep_subset.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/tests/test_langfuse_sink.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/tests/test_langfuse_source.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/tests/test_llm_judge.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/tests/test_local_loader.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/tests/test_metric_skill_evaluator.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/tests/test_models.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/tests/test_reporting.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/tests/test_runner.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/tests/test_scoring.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/tests/test_search_tool_evaluator.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/tests/test_summary_client.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/tests/test_summary_evaluator.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/tests/test_text_evaluators.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/tests/test_visualization_evaluator.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/tests/test_workspace.py +0 -0
- {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/tox.ini +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: gooddata-eval
|
|
3
|
-
Version: 1.71.
|
|
3
|
+
Version: 1.71.1.dev2
|
|
4
4
|
Summary: Evaluate the GoodData AI agent against your own questions and models.
|
|
5
5
|
Project-URL: Source, https://github.com/gooddata/gooddata-python-sdk
|
|
6
6
|
Author-email: GoodData <support@gooddata.com>
|
|
@@ -17,7 +17,7 @@ Classifier: Topic :: Scientific/Engineering
|
|
|
17
17
|
Classifier: Topic :: Software Development
|
|
18
18
|
Classifier: Typing :: Typed
|
|
19
19
|
Requires-Python: >=3.10
|
|
20
|
-
Requires-Dist: gooddata-sdk~=1.71.
|
|
20
|
+
Requires-Dist: gooddata-sdk~=1.71.1.dev2
|
|
21
21
|
Requires-Dist: httpx<1.0,>=0.27
|
|
22
22
|
Requires-Dist: orjson<4.0.0,>=3.9.15
|
|
23
23
|
Requires-Dist: pydantic<3.0,>=2.6
|
|
@@ -120,6 +120,7 @@ Both provider name and provider id are accepted as the prefix.
|
|
|
120
120
|
|---|---|---|
|
|
121
121
|
| `--runs K` | `2` | Independent runs per item (pass@K). An item passes if any run passes. |
|
|
122
122
|
| `--concurrency K` | `1` | Number of items evaluated concurrently. `1` = sequential (default). Increase to load-test the agent under simultaneous requests. Progress output interleaves when K > 1. |
|
|
123
|
+
| `--reasoning-effort LEVEL` | server default | `LOW`, `MEDIUM` or `HIGH`, sent as `options.reasoningEffort` on every chat message. Requires the `enableGenAiReasoningEffort` feature flag on the target organization — without it the server ignores the value. Applies to chat items only; `dashboard_summary` items go through the summary endpoint, which has no such option. |
|
|
123
124
|
|
|
124
125
|
#### Output
|
|
125
126
|
|
|
@@ -132,7 +133,7 @@ Both provider name and provider id are accepted as the prefix.
|
|
|
132
133
|
|
|
133
134
|
| Flag | Description |
|
|
134
135
|
|---|---|
|
|
135
|
-
| `--langfuse` | Log scores and traces to Langfuse after each item. Requires `--langfuse-dataset`. Creates one named experiment run per model (`gd-eval-{timestamp}-{model}`). Requires `LANGFUSE_PUBLIC_KEY`, `LANGFUSE_SECRET_KEY`, `LANGFUSE_HOST`. |
|
|
136
|
+
| `--langfuse` | Log scores and traces to Langfuse after each item. Requires `--langfuse-dataset`. Creates one named experiment run per model (`gd-eval-{timestamp}-{model}`, suffixed `-effort-{level}` when `--reasoning-effort` is set so runs differing only by effort stay separate). Requires `LANGFUSE_PUBLIC_KEY`, `LANGFUSE_SECRET_KEY`, `LANGFUSE_HOST`. |
|
|
136
137
|
|
|
137
138
|
### JSON report shape
|
|
138
139
|
|
|
@@ -92,6 +92,7 @@ Both provider name and provider id are accepted as the prefix.
|
|
|
92
92
|
|---|---|---|
|
|
93
93
|
| `--runs K` | `2` | Independent runs per item (pass@K). An item passes if any run passes. |
|
|
94
94
|
| `--concurrency K` | `1` | Number of items evaluated concurrently. `1` = sequential (default). Increase to load-test the agent under simultaneous requests. Progress output interleaves when K > 1. |
|
|
95
|
+
| `--reasoning-effort LEVEL` | server default | `LOW`, `MEDIUM` or `HIGH`, sent as `options.reasoningEffort` on every chat message. Requires the `enableGenAiReasoningEffort` feature flag on the target organization — without it the server ignores the value. Applies to chat items only; `dashboard_summary` items go through the summary endpoint, which has no such option. |
|
|
95
96
|
|
|
96
97
|
#### Output
|
|
97
98
|
|
|
@@ -104,7 +105,7 @@ Both provider name and provider id are accepted as the prefix.
|
|
|
104
105
|
|
|
105
106
|
| Flag | Description |
|
|
106
107
|
|---|---|
|
|
107
|
-
| `--langfuse` | Log scores and traces to Langfuse after each item. Requires `--langfuse-dataset`. Creates one named experiment run per model (`gd-eval-{timestamp}-{model}`). Requires `LANGFUSE_PUBLIC_KEY`, `LANGFUSE_SECRET_KEY`, `LANGFUSE_HOST`. |
|
|
108
|
+
| `--langfuse` | Log scores and traces to Langfuse after each item. Requires `--langfuse-dataset`. Creates one named experiment run per model (`gd-eval-{timestamp}-{model}`, suffixed `-effort-{level}` when `--reasoning-effort` is set so runs differing only by effort stay separate). Requires `LANGFUSE_PUBLIC_KEY`, `LANGFUSE_SECRET_KEY`, `LANGFUSE_HOST`. |
|
|
108
109
|
|
|
109
110
|
### JSON report shape
|
|
110
111
|
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
# (C) 2026 GoodData Corporation
|
|
2
2
|
[project]
|
|
3
3
|
name = "gooddata-eval"
|
|
4
|
-
version = "1.71.
|
|
4
|
+
version = "1.71.1.dev2"
|
|
5
5
|
description = "Evaluate the GoodData AI agent against your own questions and models."
|
|
6
6
|
readme = "README.md"
|
|
7
7
|
license = "MIT"
|
|
@@ -11,7 +11,7 @@ authors = [
|
|
|
11
11
|
keywords = ["gooddata", "ai", "evaluation", "llm", "analytics", "cli"]
|
|
12
12
|
requires-python = ">=3.10"
|
|
13
13
|
dependencies = [
|
|
14
|
-
"gooddata-sdk~=1.71.
|
|
14
|
+
"gooddata-sdk~=1.71.1.dev2",
|
|
15
15
|
"httpx>=0.27,<1.0",
|
|
16
16
|
"orjson>=3.9.15,<4.0.0",
|
|
17
17
|
"pydantic>=2.6,<3.0",
|
|
@@ -14,6 +14,7 @@ from gooddata_eval.core.agentic.guardrail import evaluate_agentic_guardrail
|
|
|
14
14
|
from gooddata_eval.core.agentic.metric_skill import evaluate_agentic_metric_skill
|
|
15
15
|
from gooddata_eval.core.agentic.search_tool import evaluate_agentic_search_tool
|
|
16
16
|
from gooddata_eval.core.agentic.visualization import evaluate_agentic_visualization
|
|
17
|
+
from gooddata_eval.core.config import ReasoningEffort
|
|
17
18
|
from gooddata_eval.core.models import CreatedVisualization, DatasetItem
|
|
18
19
|
from gooddata_eval.core.runner import EvalReport, ItemReport
|
|
19
20
|
|
|
@@ -24,6 +25,7 @@ class _LfKw(TypedDict, total=False):
|
|
|
24
25
|
dataset_name: str
|
|
25
26
|
run_timestamp: str
|
|
26
27
|
model_version_override: str | None
|
|
28
|
+
reasoning_effort: ReasoningEffort | None
|
|
27
29
|
|
|
28
30
|
|
|
29
31
|
AGENTIC_TEST_KINDS = frozenset(
|
|
@@ -80,6 +82,7 @@ def _dispatch_agentic(
|
|
|
80
82
|
langfuse: Any,
|
|
81
83
|
run_ts: str,
|
|
82
84
|
model_version_override: str | None,
|
|
85
|
+
reasoning_effort: ReasoningEffort | None = None,
|
|
83
86
|
) -> None:
|
|
84
87
|
"""Call the appropriate evaluate_agentic_* function for the item's test_kind."""
|
|
85
88
|
kind = item.test_kind
|
|
@@ -90,6 +93,7 @@ def _dispatch_agentic(
|
|
|
90
93
|
"dataset_name": item.dataset_name,
|
|
91
94
|
"run_timestamp": run_ts,
|
|
92
95
|
"model_version_override": model_version_override,
|
|
96
|
+
"reasoning_effort": reasoning_effort,
|
|
93
97
|
}
|
|
94
98
|
|
|
95
99
|
if kind in ("vis_agentic", "agentic_visualization"):
|
|
@@ -176,6 +180,7 @@ def run_agentic_items(
|
|
|
176
180
|
*,
|
|
177
181
|
k: int = 2,
|
|
178
182
|
model_version: str | None = None,
|
|
183
|
+
reasoning_effort: ReasoningEffort | None = None,
|
|
179
184
|
use_langfuse: bool = False,
|
|
180
185
|
run_ts: str,
|
|
181
186
|
on_item_start: Any = None,
|
|
@@ -202,7 +207,7 @@ def run_agentic_items(
|
|
|
202
207
|
)
|
|
203
208
|
t0 = time.perf_counter()
|
|
204
209
|
try:
|
|
205
|
-
_dispatch_agentic(item, host, token, workspace_id, k, langfuse, run_ts, model_version)
|
|
210
|
+
_dispatch_agentic(item, host, token, workspace_id, k, langfuse, run_ts, model_version, reasoning_effort)
|
|
206
211
|
item_report.pass_at_k = True
|
|
207
212
|
item_report.runs = k
|
|
208
213
|
except AssertionError as exc:
|
|
@@ -6,6 +6,7 @@ import sys
|
|
|
6
6
|
import threading
|
|
7
7
|
from datetime import datetime, timezone
|
|
8
8
|
from pathlib import Path
|
|
9
|
+
from typing import get_args
|
|
9
10
|
|
|
10
11
|
import httpx
|
|
11
12
|
from gooddata_api_client.exceptions import ApiException
|
|
@@ -14,7 +15,7 @@ from rich.table import Table
|
|
|
14
15
|
|
|
15
16
|
from gooddata_eval.cli.agentic_runner import AGENTIC_TEST_KINDS, run_agentic_items
|
|
16
17
|
from gooddata_eval.core.chat.sse_client import ChatClient
|
|
17
|
-
from gooddata_eval.core.config import RunConfig
|
|
18
|
+
from gooddata_eval.core.config import ReasoningEffort, RunConfig
|
|
18
19
|
from gooddata_eval.core.connection import ConnectionError_, resolve_connection
|
|
19
20
|
from gooddata_eval.core.dataset.local import load_local_dataset
|
|
20
21
|
from gooddata_eval.core.langfuse.sink import LangfuseSink
|
|
@@ -104,6 +105,13 @@ def _build_parser() -> argparse.ArgumentParser:
|
|
|
104
105
|
dest="preserve_failed",
|
|
105
106
|
help="Keep failed conversations on the server for post-mortem inspection.",
|
|
106
107
|
)
|
|
108
|
+
run.add_argument(
|
|
109
|
+
"--reasoning-effort",
|
|
110
|
+
dest="reasoning_effort",
|
|
111
|
+
choices=list(get_args(ReasoningEffort)),
|
|
112
|
+
help="Reasoning effort requested per message. Requires the enableGenAiReasoningEffort "
|
|
113
|
+
"feature flag on the target organization; without it the server ignores the value.",
|
|
114
|
+
)
|
|
107
115
|
run.add_argument(
|
|
108
116
|
"--langfuse",
|
|
109
117
|
action="store_true",
|
|
@@ -292,6 +300,10 @@ def _run(config: RunConfig) -> int:
|
|
|
292
300
|
progress_console.print(f"Provider={provider_display}, model={resolved.model_id}{switched}")
|
|
293
301
|
|
|
294
302
|
run_name = f"gd-eval-{run_ts}-{resolved.model_id}"
|
|
303
|
+
if config.reasoning_effort:
|
|
304
|
+
# Without this two runs differing only by effort share a name and are
|
|
305
|
+
# indistinguishable in the report, which is the comparison this exists for.
|
|
306
|
+
run_name = f"{run_name}-effort-{config.reasoning_effort.lower()}"
|
|
295
307
|
if progress_console and config.log_to_langfuse:
|
|
296
308
|
progress_console.print(f"Logging to Langfuse run '{run_name}'...")
|
|
297
309
|
|
|
@@ -307,6 +319,7 @@ def _run(config: RunConfig) -> int:
|
|
|
307
319
|
run_name=run_name,
|
|
308
320
|
model_id=resolved.model_id,
|
|
309
321
|
provider_type=resolved.provider_type,
|
|
322
|
+
reasoning_effort=config.reasoning_effort,
|
|
310
323
|
)
|
|
311
324
|
|
|
312
325
|
def on_langfuse_item_done(
|
|
@@ -329,6 +342,7 @@ def _run(config: RunConfig) -> int:
|
|
|
329
342
|
workspace_id=config.workspace_id,
|
|
330
343
|
k=config.runs,
|
|
331
344
|
model_version=resolved.model_id,
|
|
345
|
+
reasoning_effort=config.reasoning_effort,
|
|
332
346
|
use_langfuse=config.log_to_langfuse,
|
|
333
347
|
run_ts=run_ts,
|
|
334
348
|
on_item_start=on_item_start,
|
|
@@ -342,6 +356,7 @@ def _run(config: RunConfig) -> int:
|
|
|
342
356
|
token=config.token,
|
|
343
357
|
workspace_id=config.workspace_id,
|
|
344
358
|
preserve_failed=config.preserve_failed,
|
|
359
|
+
reasoning_effort=config.reasoning_effort,
|
|
345
360
|
),
|
|
346
361
|
SummaryClient(host=config.host, token=config.token, workspace_id=config.workspace_id),
|
|
347
362
|
)
|
|
@@ -433,6 +448,7 @@ def main(argv: list[str] | None = None) -> int:
|
|
|
433
448
|
quiet=args.quiet,
|
|
434
449
|
kind=args.kind,
|
|
435
450
|
preserve_failed=args.preserve_failed,
|
|
451
|
+
reasoning_effort=args.reasoning_effort,
|
|
436
452
|
)
|
|
437
453
|
return _run(config)
|
|
438
454
|
except (
|
{gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/agentic/_langfuse.py
RENAMED
|
@@ -15,6 +15,8 @@ from typing import Any
|
|
|
15
15
|
|
|
16
16
|
import httpx
|
|
17
17
|
|
|
18
|
+
from gooddata_eval.core.config import ReasoningEffort, normalize_reasoning_effort
|
|
19
|
+
|
|
18
20
|
_log = logging.getLogger(__name__)
|
|
19
21
|
|
|
20
22
|
# ---------------------------------------------------------------------------
|
|
@@ -384,6 +386,7 @@ def build_run_context(
|
|
|
384
386
|
run_timestamp: str | None,
|
|
385
387
|
model_version_override: str | None,
|
|
386
388
|
run_metadata_extra: dict[str, Any] | None = None,
|
|
389
|
+
reasoning_effort: ReasoningEffort | None = None,
|
|
387
390
|
) -> tuple[str, dict[str, Any]]:
|
|
388
391
|
"""Return (run_name_base, run_metadata) with model version resolved from workspace API.
|
|
389
392
|
|
|
@@ -399,15 +402,24 @@ def build_run_context(
|
|
|
399
402
|
(e.g. a testing-framework tag or a CI run id for scoping). Default None keeps
|
|
400
403
|
behavior unchanged. The SDK-derived model_version is applied last and cannot
|
|
401
404
|
be overwritten by this dict.
|
|
405
|
+
reasoning_effort: Effort the run requested, stamped into both the run name and
|
|
406
|
+
the metadata so effort-varying runs stay comparable side by side.
|
|
402
407
|
"""
|
|
408
|
+
effort = normalize_reasoning_effort(reasoning_effort)
|
|
403
409
|
model = get_model_version(host, token, workspace_id, model_version_override)
|
|
404
410
|
ts = run_timestamp or datetime.now().strftime("%Y-%m-%d_%H-%M-%S")
|
|
405
411
|
base = f"{dataset_name}_{ts}"
|
|
406
412
|
if model:
|
|
407
413
|
base = f"{base}_{model}"
|
|
414
|
+
# Part of the run name, not just metadata: two runs that differ only by effort would
|
|
415
|
+
# otherwise collide on the same name and be indistinguishable in the report.
|
|
416
|
+
if effort:
|
|
417
|
+
base = f"{base}_effort-{effort.lower()}"
|
|
408
418
|
# Caller supplies its own run tags (e.g. testing_framework); model_version is applied
|
|
409
419
|
# last so the SDK-derived value cannot be overwritten by run_metadata_extra.
|
|
410
420
|
metadata: dict[str, Any] = dict(run_metadata_extra) if run_metadata_extra else {}
|
|
421
|
+
if effort:
|
|
422
|
+
metadata["reasoning_effort"] = effort
|
|
411
423
|
if model:
|
|
412
424
|
metadata["model_version"] = model
|
|
413
425
|
return base, metadata
|
{gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/agentic/alert_skill.py
RENAMED
|
@@ -13,6 +13,7 @@ from gooddata_sdk import GoodDataSdk
|
|
|
13
13
|
|
|
14
14
|
from gooddata_eval.core.agentic._catalog import CatalogMetricAlert
|
|
15
15
|
from gooddata_eval.core.chat.sse_client import ChatClient
|
|
16
|
+
from gooddata_eval.core.config import ReasoningEffort
|
|
16
17
|
from gooddata_eval.core.models import ToolCallEvent
|
|
17
18
|
|
|
18
19
|
try:
|
|
@@ -311,11 +312,26 @@ def _extract_alert_call(tool_call_events: list[ToolCallEvent]) -> tuple[str | No
|
|
|
311
312
|
return None, {}, False
|
|
312
313
|
|
|
313
314
|
|
|
314
|
-
def
|
|
315
|
-
|
|
316
|
-
|
|
317
|
-
|
|
318
|
-
|
|
315
|
+
def render_alert_proposal(proposal: dict) -> str:
|
|
316
|
+
"""Render an alert-proposal part as the text the simulated user reacts to.
|
|
317
|
+
|
|
318
|
+
The alert skill's confirmation step deliberately emits no text part (GDAI-2032) — the
|
|
319
|
+
prompt and the CTA live only in the proposal payload, which the frontend renders as a
|
|
320
|
+
widget. Dumping the payload (rather than prose) keeps recipients, condition, trigger and
|
|
321
|
+
dashboard visible so the simulated user can still verify them against its goal, and does
|
|
322
|
+
not need updating whenever ``AlertProposal`` grows a field.
|
|
323
|
+
"""
|
|
324
|
+
cta = proposal.get("cta") or "Should I create this alert?"
|
|
325
|
+
summary = {k: v for k, v in proposal.items() if k != "cta"}
|
|
326
|
+
alert = dict(summary.get("alert") or {})
|
|
327
|
+
# The AFM execution block is opaque wire dicts — noise that would crowd out the fields
|
|
328
|
+
# the simulated user actually has to check.
|
|
329
|
+
alert.pop("execution", None)
|
|
330
|
+
if "alert" in summary:
|
|
331
|
+
# Key off presence, not truthiness: an alert whose only key was `execution` must
|
|
332
|
+
# still be replaced, otherwise the original (execution-bearing) dict survives.
|
|
333
|
+
summary["alert"] = alert
|
|
334
|
+
return f"{cta}\n\nAlert proposal:\n{json.dumps(summary, indent=2, sort_keys=True)}"
|
|
319
335
|
|
|
320
336
|
|
|
321
337
|
def run_agentic_alert_skill(
|
|
@@ -327,11 +343,12 @@ def run_agentic_alert_skill(
|
|
|
327
343
|
k: int = _DEFAULT_K,
|
|
328
344
|
max_iterations: int = _DEFAULT_MAX_ITERATIONS,
|
|
329
345
|
initial_conversation_id: str | None = None,
|
|
346
|
+
reasoning_effort: ReasoningEffort | None = None,
|
|
330
347
|
) -> AgenticAlertSummary:
|
|
331
348
|
"""Run the alert-skill agentic evaluation K times and return a summary."""
|
|
332
349
|
expected = _normalize_expected_output(expected_output)
|
|
333
350
|
run_results: list[AlertRunResult] = []
|
|
334
|
-
client = ChatClient(host=host, token=token, workspace_id=workspace_id)
|
|
351
|
+
client = ChatClient(host=host, token=token, workspace_id=workspace_id, reasoning_effort=reasoning_effort)
|
|
335
352
|
sdk = GoodDataSdk.create(host, token)
|
|
336
353
|
|
|
337
354
|
def _run_once(conv_id: str) -> AlertRunResult:
|
|
@@ -352,6 +369,8 @@ def run_agentic_alert_skill(
|
|
|
352
369
|
alert_id_to_delete = alert_id
|
|
353
370
|
break
|
|
354
371
|
response_text = (chat_result.text_response or "").strip()
|
|
372
|
+
if not response_text and chat_result.alert_proposals:
|
|
373
|
+
response_text = render_alert_proposal(chat_result.alert_proposals[-1])
|
|
355
374
|
# Stop if agent gave a completely empty response (stuck)
|
|
356
375
|
if not response_text and not chat_result.tool_call_events:
|
|
357
376
|
break
|
|
@@ -445,6 +464,7 @@ def evaluate_agentic_alert_skill(
|
|
|
445
464
|
run_timestamp: str | None = None,
|
|
446
465
|
model_version_override: str | None = None,
|
|
447
466
|
run_metadata_extra: dict | None = None,
|
|
467
|
+
reasoning_effort: ReasoningEffort | None = None,
|
|
448
468
|
) -> None:
|
|
449
469
|
"""Run alert-skill evaluation, log to Langfuse, and raise AlertSkillAssertionError on failure."""
|
|
450
470
|
from datetime import datetime as _dt # noqa: PLC0415
|
|
@@ -464,6 +484,7 @@ def evaluate_agentic_alert_skill(
|
|
|
464
484
|
k=k,
|
|
465
485
|
max_iterations=max_iterations,
|
|
466
486
|
initial_conversation_id=initial_conversation_id,
|
|
487
|
+
reasoning_effort=reasoning_effort,
|
|
467
488
|
)
|
|
468
489
|
|
|
469
490
|
if langfuse is not None and dataset_item_id:
|
|
@@ -483,6 +504,7 @@ def evaluate_agentic_alert_skill(
|
|
|
483
504
|
run_timestamp,
|
|
484
505
|
model_version_override,
|
|
485
506
|
run_metadata_extra,
|
|
507
|
+
reasoning_effort,
|
|
486
508
|
)
|
|
487
509
|
traces_by_conv = find_traces_per_conversation(
|
|
488
510
|
langfuse,
|
{gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/agentic/conversation.py
RENAMED
|
@@ -11,8 +11,10 @@ from typing import Literal
|
|
|
11
11
|
from gooddata_sdk import GoodDataSdk
|
|
12
12
|
from pydantic import BaseModel
|
|
13
13
|
|
|
14
|
+
from gooddata_eval.core.agentic.alert_skill import render_alert_proposal
|
|
14
15
|
from gooddata_eval.core.agentic.metric_skill import _delete_metric, _extract_created_metric_ids
|
|
15
16
|
from gooddata_eval.core.chat.sse_client import ChatClient
|
|
17
|
+
from gooddata_eval.core.config import ReasoningEffort
|
|
16
18
|
from gooddata_eval.core.models import ChatResult, ToolCallEvent
|
|
17
19
|
from gooddata_eval.core.scoring import (
|
|
18
20
|
check_filters,
|
|
@@ -277,6 +279,7 @@ def run_agentic_conversation(
|
|
|
277
279
|
fixture: ConversationFixture,
|
|
278
280
|
max_clarification_turns: int = 20,
|
|
279
281
|
initial_conversation_id: str | None = None,
|
|
282
|
+
reasoning_effort: ReasoningEffort | None = None,
|
|
280
283
|
) -> ConversationResult:
|
|
281
284
|
"""Run a multi-turn, multi-skill conversation evaluation (no K-runs).
|
|
282
285
|
|
|
@@ -284,7 +287,7 @@ def run_agentic_conversation(
|
|
|
284
287
|
trigger up to *max_clarification_turns* additional rounds of simulated-user
|
|
285
288
|
replies before the agent produces the expected output.
|
|
286
289
|
"""
|
|
287
|
-
client = ChatClient(host=host, token=token, workspace_id=workspace_id)
|
|
290
|
+
client = ChatClient(host=host, token=token, workspace_id=workspace_id, reasoning_effort=reasoning_effort)
|
|
288
291
|
sdk = GoodDataSdk.create(host, token)
|
|
289
292
|
turn_results: list[TurnResult] = []
|
|
290
293
|
turn_outputs: dict[str, dict] = {}
|
|
@@ -322,7 +325,10 @@ def run_agentic_conversation(
|
|
|
322
325
|
break
|
|
323
326
|
|
|
324
327
|
response_text = (chat_result.text_response or "").strip()
|
|
325
|
-
if
|
|
328
|
+
if not response_text and chat_result.alert_proposals:
|
|
329
|
+
response_text = render_alert_proposal(chat_result.alert_proposals[-1])
|
|
330
|
+
asking = _is_asking_clarification(response_text) or bool(chat_result.alert_proposals)
|
|
331
|
+
if asking and clarification_turns < max_clarification_turns:
|
|
326
332
|
clarification_turns += 1
|
|
327
333
|
total_clarification_turns += 1
|
|
328
334
|
current_message = _get_sim_user_response(response_text, resolved_turn, resolved_expected)
|
|
@@ -399,6 +405,7 @@ def evaluate_agentic_conversation(
|
|
|
399
405
|
run_timestamp: str | None = None,
|
|
400
406
|
model_version_override: str | None = None,
|
|
401
407
|
run_metadata_extra: dict | None = None,
|
|
408
|
+
reasoning_effort: ReasoningEffort | None = None,
|
|
402
409
|
) -> None:
|
|
403
410
|
"""Run conversation evaluation, log to Langfuse, and raise on failure."""
|
|
404
411
|
from datetime import datetime as _dt # noqa: PLC0415
|
|
@@ -416,6 +423,7 @@ def evaluate_agentic_conversation(
|
|
|
416
423
|
fixture=fixture,
|
|
417
424
|
max_clarification_turns=max_clarification_turns,
|
|
418
425
|
initial_conversation_id=initial_conversation_id,
|
|
426
|
+
reasoning_effort=reasoning_effort,
|
|
419
427
|
)
|
|
420
428
|
|
|
421
429
|
if langfuse is not None and dataset_item_id:
|
|
@@ -435,6 +443,7 @@ def evaluate_agentic_conversation(
|
|
|
435
443
|
run_timestamp,
|
|
436
444
|
model_version_override,
|
|
437
445
|
run_metadata_extra,
|
|
446
|
+
reasoning_effort,
|
|
438
447
|
)
|
|
439
448
|
traces_by_conv = find_traces_per_conversation(
|
|
440
449
|
langfuse,
|
|
@@ -6,6 +6,7 @@ from __future__ import annotations
|
|
|
6
6
|
from dataclasses import dataclass
|
|
7
7
|
|
|
8
8
|
from gooddata_eval.core.chat.sse_client import ChatClient
|
|
9
|
+
from gooddata_eval.core.config import ReasoningEffort
|
|
9
10
|
from gooddata_eval.core.evaluators._llm_judge import LLMJudge
|
|
10
11
|
|
|
11
12
|
_DEFAULT_K = 1
|
|
@@ -71,10 +72,11 @@ def run_agentic_general_question(
|
|
|
71
72
|
expected_output: str,
|
|
72
73
|
k: int = _DEFAULT_K,
|
|
73
74
|
initial_conversation_id: str | None = None,
|
|
75
|
+
reasoning_effort: ReasoningEffort | None = None,
|
|
74
76
|
) -> AgenticGeneralQuestionSummary:
|
|
75
77
|
"""Run the general-question agentic evaluation K times and return a summary."""
|
|
76
78
|
run_results: list[GeneralQuestionResult] = []
|
|
77
|
-
client = ChatClient(host=host, token=token, workspace_id=workspace_id)
|
|
79
|
+
client = ChatClient(host=host, token=token, workspace_id=workspace_id, reasoning_effort=reasoning_effort)
|
|
78
80
|
judge = LLMJudge(_GENERAL_QUESTION_EVALUATION_STEPS, model="gpt-4o")
|
|
79
81
|
|
|
80
82
|
try:
|
|
@@ -153,6 +155,7 @@ def evaluate_agentic_general_question(
|
|
|
153
155
|
run_timestamp: str | None = None,
|
|
154
156
|
model_version_override: str | None = None,
|
|
155
157
|
run_metadata_extra: dict | None = None,
|
|
158
|
+
reasoning_effort: ReasoningEffort | None = None,
|
|
156
159
|
) -> None:
|
|
157
160
|
"""Run general-question evaluation, log to Langfuse, and raise on failure."""
|
|
158
161
|
from datetime import datetime as _dt # noqa: PLC0415
|
|
@@ -171,6 +174,7 @@ def evaluate_agentic_general_question(
|
|
|
171
174
|
expected_output=expected_output,
|
|
172
175
|
k=k,
|
|
173
176
|
initial_conversation_id=initial_conversation_id,
|
|
177
|
+
reasoning_effort=reasoning_effort,
|
|
174
178
|
)
|
|
175
179
|
|
|
176
180
|
if langfuse is not None and dataset_item_id:
|
|
@@ -190,6 +194,7 @@ def evaluate_agentic_general_question(
|
|
|
190
194
|
run_timestamp,
|
|
191
195
|
model_version_override,
|
|
192
196
|
run_metadata_extra,
|
|
197
|
+
reasoning_effort,
|
|
193
198
|
)
|
|
194
199
|
traces_by_conv = find_traces_per_conversation(
|
|
195
200
|
langfuse,
|
{gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/agentic/guardrail.py
RENAMED
|
@@ -6,6 +6,7 @@ from __future__ import annotations
|
|
|
6
6
|
from dataclasses import dataclass
|
|
7
7
|
|
|
8
8
|
from gooddata_eval.core.chat.sse_client import ChatClient
|
|
9
|
+
from gooddata_eval.core.config import ReasoningEffort
|
|
9
10
|
from gooddata_eval.core.evaluators._llm_judge import LLMJudge
|
|
10
11
|
|
|
11
12
|
_DEFAULT_K = 1
|
|
@@ -68,10 +69,11 @@ def run_agentic_guardrail(
|
|
|
68
69
|
expected_output: str,
|
|
69
70
|
k: int = _DEFAULT_K,
|
|
70
71
|
initial_conversation_id: str | None = None,
|
|
72
|
+
reasoning_effort: ReasoningEffort | None = None,
|
|
71
73
|
) -> AgenticGuardrailSummary:
|
|
72
74
|
"""Run the guardrail agentic evaluation K times and return a summary."""
|
|
73
75
|
run_results: list[GuardrailResult] = []
|
|
74
|
-
client = ChatClient(host=host, token=token, workspace_id=workspace_id)
|
|
76
|
+
client = ChatClient(host=host, token=token, workspace_id=workspace_id, reasoning_effort=reasoning_effort)
|
|
75
77
|
judge = LLMJudge(_GUARDRAIL_EVALUATION_STEPS, model="gpt-4o")
|
|
76
78
|
|
|
77
79
|
try:
|
|
@@ -150,6 +152,7 @@ def evaluate_agentic_guardrail(
|
|
|
150
152
|
run_timestamp: str | None = None,
|
|
151
153
|
model_version_override: str | None = None,
|
|
152
154
|
run_metadata_extra: dict | None = None,
|
|
155
|
+
reasoning_effort: ReasoningEffort | None = None,
|
|
153
156
|
) -> None:
|
|
154
157
|
"""Run guardrail evaluation, log to Langfuse, and raise on failure."""
|
|
155
158
|
from datetime import datetime as _dt # noqa: PLC0415
|
|
@@ -168,6 +171,7 @@ def evaluate_agentic_guardrail(
|
|
|
168
171
|
expected_output=expected_output,
|
|
169
172
|
k=k,
|
|
170
173
|
initial_conversation_id=initial_conversation_id,
|
|
174
|
+
reasoning_effort=reasoning_effort,
|
|
171
175
|
)
|
|
172
176
|
|
|
173
177
|
if langfuse is not None and dataset_item_id:
|
|
@@ -187,6 +191,7 @@ def evaluate_agentic_guardrail(
|
|
|
187
191
|
run_timestamp,
|
|
188
192
|
model_version_override,
|
|
189
193
|
run_metadata_extra,
|
|
194
|
+
reasoning_effort,
|
|
190
195
|
)
|
|
191
196
|
traces_by_conv = find_traces_per_conversation(
|
|
192
197
|
langfuse,
|
{gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/agentic/metric_skill.py
RENAMED
|
@@ -11,6 +11,7 @@ from typing import Any
|
|
|
11
11
|
from gooddata_sdk import GoodDataSdk
|
|
12
12
|
|
|
13
13
|
from gooddata_eval.core.chat.sse_client import ChatClient
|
|
14
|
+
from gooddata_eval.core.config import ReasoningEffort
|
|
14
15
|
from gooddata_eval.core.models import ToolCallEvent
|
|
15
16
|
|
|
16
17
|
try:
|
|
@@ -232,6 +233,7 @@ def run_agentic_metric_skill(
|
|
|
232
233
|
k: int = _DEFAULT_K,
|
|
233
234
|
max_iterations: int = _DEFAULT_MAX_ITERATIONS,
|
|
234
235
|
initial_conversation_id: str | None = None,
|
|
236
|
+
reasoning_effort: ReasoningEffort | None = None,
|
|
235
237
|
) -> AgenticMetricSummary:
|
|
236
238
|
"""Run the metric-skill agentic evaluation K times and return a summary.
|
|
237
239
|
|
|
@@ -240,7 +242,7 @@ def run_agentic_metric_skill(
|
|
|
240
242
|
"""
|
|
241
243
|
expected_outputs: list[dict] = expected_output if isinstance(expected_output, list) else [expected_output]
|
|
242
244
|
run_results: list[MetricRunResult] = []
|
|
243
|
-
client = ChatClient(host=host, token=token, workspace_id=workspace_id)
|
|
245
|
+
client = ChatClient(host=host, token=token, workspace_id=workspace_id, reasoning_effort=reasoning_effort)
|
|
244
246
|
sdk = GoodDataSdk.create(host, token)
|
|
245
247
|
|
|
246
248
|
try:
|
|
@@ -300,6 +302,7 @@ def evaluate_agentic_metric_skill(
|
|
|
300
302
|
run_timestamp: str | None = None,
|
|
301
303
|
model_version_override: str | None = None,
|
|
302
304
|
run_metadata_extra: dict | None = None,
|
|
305
|
+
reasoning_effort: ReasoningEffort | None = None,
|
|
303
306
|
) -> None:
|
|
304
307
|
"""Run metric-skill evaluation, log to Langfuse, and raise MetricSkillAssertionError on failure."""
|
|
305
308
|
from datetime import datetime as _dt # noqa: PLC0415
|
|
@@ -319,6 +322,7 @@ def evaluate_agentic_metric_skill(
|
|
|
319
322
|
k=k,
|
|
320
323
|
max_iterations=max_iterations,
|
|
321
324
|
initial_conversation_id=initial_conversation_id,
|
|
325
|
+
reasoning_effort=reasoning_effort,
|
|
322
326
|
)
|
|
323
327
|
|
|
324
328
|
if langfuse is not None and dataset_item_id:
|
|
@@ -338,6 +342,7 @@ def evaluate_agentic_metric_skill(
|
|
|
338
342
|
run_timestamp,
|
|
339
343
|
model_version_override,
|
|
340
344
|
run_metadata_extra,
|
|
345
|
+
reasoning_effort,
|
|
341
346
|
)
|
|
342
347
|
traces_by_conv = find_traces_per_conversation(
|
|
343
348
|
langfuse,
|
{gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/agentic/search_tool.py
RENAMED
|
@@ -6,6 +6,7 @@ from __future__ import annotations
|
|
|
6
6
|
from dataclasses import dataclass
|
|
7
7
|
|
|
8
8
|
from gooddata_eval.core.chat.sse_client import ChatClient
|
|
9
|
+
from gooddata_eval.core.config import ReasoningEffort
|
|
9
10
|
from gooddata_eval.core.models import ToolCallEvent
|
|
10
11
|
|
|
11
12
|
_DEFAULT_K = 1
|
|
@@ -67,11 +68,12 @@ def run_agentic_search_tool(
|
|
|
67
68
|
expected_tool_call: dict,
|
|
68
69
|
k: int = _DEFAULT_K,
|
|
69
70
|
initial_conversation_id: str | None = None,
|
|
71
|
+
reasoning_effort: ReasoningEffort | None = None,
|
|
70
72
|
) -> AgenticSearchSummary:
|
|
71
73
|
"""Run the search-tool agentic evaluation K times (single-turn each)."""
|
|
72
74
|
run_results: list[SearchResult] = []
|
|
73
75
|
|
|
74
|
-
client = ChatClient(host=host, token=token, workspace_id=workspace_id)
|
|
76
|
+
client = ChatClient(host=host, token=token, workspace_id=workspace_id, reasoning_effort=reasoning_effort)
|
|
75
77
|
try:
|
|
76
78
|
conv_id_0 = initial_conversation_id if initial_conversation_id is not None else client.create_conversation()
|
|
77
79
|
try:
|
|
@@ -144,6 +146,7 @@ def evaluate_agentic_search_tool(
|
|
|
144
146
|
run_timestamp: str | None = None,
|
|
145
147
|
model_version_override: str | None = None,
|
|
146
148
|
run_metadata_extra: dict | None = None,
|
|
149
|
+
reasoning_effort: ReasoningEffort | None = None,
|
|
147
150
|
) -> None:
|
|
148
151
|
"""Run search-tool evaluation, log to Langfuse, and raise SearchToolAssertionError on failure."""
|
|
149
152
|
from datetime import datetime as _dt # noqa: PLC0415
|
|
@@ -162,6 +165,7 @@ def evaluate_agentic_search_tool(
|
|
|
162
165
|
expected_tool_call=expected_tool_call,
|
|
163
166
|
k=k,
|
|
164
167
|
initial_conversation_id=initial_conversation_id,
|
|
168
|
+
reasoning_effort=reasoning_effort,
|
|
165
169
|
)
|
|
166
170
|
|
|
167
171
|
if langfuse is not None and dataset_item_id:
|
|
@@ -181,6 +185,7 @@ def evaluate_agentic_search_tool(
|
|
|
181
185
|
run_timestamp,
|
|
182
186
|
model_version_override,
|
|
183
187
|
run_metadata_extra,
|
|
188
|
+
reasoning_effort,
|
|
184
189
|
)
|
|
185
190
|
traces_by_conv = find_traces_per_conversation(
|
|
186
191
|
langfuse,
|
{gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/agentic/visualization.py
RENAMED
|
@@ -11,6 +11,7 @@ import os
|
|
|
11
11
|
from dataclasses import dataclass
|
|
12
12
|
|
|
13
13
|
from gooddata_eval.core.chat.sse_client import ChatClient
|
|
14
|
+
from gooddata_eval.core.config import ReasoningEffort
|
|
14
15
|
from gooddata_eval.core.evaluators.visualization import (
|
|
15
16
|
EvaluationResult,
|
|
16
17
|
_check_visualization_skill_activated,
|
|
@@ -203,6 +204,7 @@ def run_agentic_visualization(
|
|
|
203
204
|
k: int = _DEFAULT_K,
|
|
204
205
|
max_iterations: int = _DEFAULT_MAX_ITERATIONS,
|
|
205
206
|
initial_conversation_id: str | None = None,
|
|
207
|
+
reasoning_effort: ReasoningEffort | None = None,
|
|
206
208
|
) -> AgenticRunSummary:
|
|
207
209
|
"""Run K independent conversations and return evaluation results.
|
|
208
210
|
|
|
@@ -211,7 +213,7 @@ def run_agentic_visualization(
|
|
|
211
213
|
fresh conversations. Caller-supplied conversations are not deleted; all
|
|
212
214
|
conversations created by this function are deleted on completion.
|
|
213
215
|
"""
|
|
214
|
-
client = ChatClient(host=host, token=token, workspace_id=workspace_id)
|
|
216
|
+
client = ChatClient(host=host, token=token, workspace_id=workspace_id, reasoning_effort=reasoning_effort)
|
|
215
217
|
run_results: list[RunResult] = []
|
|
216
218
|
|
|
217
219
|
try:
|
|
@@ -265,6 +267,7 @@ def evaluate_agentic_visualization(
|
|
|
265
267
|
model_version_override: str | None = None,
|
|
266
268
|
run_metadata_extra: dict | None = None,
|
|
267
269
|
record_output_path: str | None = None,
|
|
270
|
+
reasoning_effort: ReasoningEffort | None = None,
|
|
268
271
|
) -> None:
|
|
269
272
|
"""Run visualization evaluation, log to Langfuse, and raise VisualizationAssertionError on failure."""
|
|
270
273
|
import json as _json # noqa: PLC0415
|
|
@@ -285,6 +288,7 @@ def evaluate_agentic_visualization(
|
|
|
285
288
|
k=k,
|
|
286
289
|
max_iterations=max_iterations,
|
|
287
290
|
initial_conversation_id=initial_conversation_id,
|
|
291
|
+
reasoning_effort=reasoning_effort,
|
|
288
292
|
)
|
|
289
293
|
|
|
290
294
|
if langfuse is not None and dataset_item_id:
|
|
@@ -304,6 +308,7 @@ def evaluate_agentic_visualization(
|
|
|
304
308
|
run_timestamp,
|
|
305
309
|
model_version_override,
|
|
306
310
|
run_metadata_extra,
|
|
311
|
+
reasoning_effort,
|
|
307
312
|
)
|
|
308
313
|
K = len(summary.run_results)
|
|
309
314
|
traces_by_conv = find_traces_per_conversation(
|