gooddata-eval 1.71.1.dev1__tar.gz → 1.71.1.dev2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/PKG-INFO +4 -3
- {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/README.md +2 -1
- {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/pyproject.toml +2 -2
- {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/cli/agentic_runner.py +6 -1
- {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/cli/main.py +17 -1
- {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/agentic/_langfuse.py +12 -0
- {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/agentic/alert_skill.py +6 -1
- {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/agentic/conversation.py +6 -1
- {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/agentic/general_question.py +6 -1
- {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/agentic/guardrail.py +6 -1
- {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/agentic/metric_skill.py +6 -1
- {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/agentic/search_tool.py +6 -1
- {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/agentic/visualization.py +6 -1
- {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/chat/sse_client.py +21 -2
- gooddata_eval-1.71.1.dev2/src/gooddata_eval/core/config.py +46 -0
- {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/langfuse/sink.py +21 -2
- gooddata_eval-1.71.1.dev2/tests/test_agentic_run_context.py +73 -0
- {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/tests/test_cli.py +77 -3
- {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/tests/test_sse_client.py +77 -2
- gooddata_eval-1.71.1.dev1/src/gooddata_eval/core/config.py +0 -22
- gooddata_eval-1.71.1.dev1/tests/test_agentic_run_context.py +0 -32
- {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/.gitignore +0 -0
- {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/LICENSE.txt +0 -0
- {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/Makefile +0 -0
- {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/__init__.py +0 -0
- {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/_version.py +0 -0
- {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/cli/__init__.py +0 -0
- {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/__init__.py +0 -0
- {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/agentic/__init__.py +0 -0
- {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/agentic/_catalog.py +0 -0
- {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/chat/__init__.py +0 -0
- {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/connection.py +0 -0
- {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/dataset/__init__.py +0 -0
- {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/dataset/langfuse_source.py +0 -0
- {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/dataset/local.py +0 -0
- {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/evaluators/__init__.py +0 -0
- {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/evaluators/_deep_subset.py +0 -0
- {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/evaluators/_llm_judge.py +0 -0
- {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/evaluators/_text_utils.py +0 -0
- {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/evaluators/alert_skill.py +0 -0
- {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/evaluators/base.py +0 -0
- {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/evaluators/general_question.py +0 -0
- {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/evaluators/guardrail.py +0 -0
- {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/evaluators/metric_skill.py +0 -0
- {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/evaluators/search_tool.py +0 -0
- {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/evaluators/summary.py +0 -0
- {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/evaluators/visualization.py +0 -0
- {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/langfuse/__init__.py +0 -0
- {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/models.py +0 -0
- {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/reporting/__init__.py +0 -0
- {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/reporting/console.py +0 -0
- {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/reporting/json_report.py +0 -0
- {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/runner.py +0 -0
- {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/scoring.py +0 -0
- {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/summary/__init__.py +0 -0
- {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/summary/http_client.py +0 -0
- {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/workspace.py +0 -0
- {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/tests/__init__.py +0 -0
- {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/tests/conftest.py +0 -0
- {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/tests/fixtures/sample_dataset/metric_skill_create.json +0 -0
- {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/tests/fixtures/sample_dataset/visualization_revenue.json +0 -0
- {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/tests/fixtures/sse_visualization_stream.txt +0 -0
- {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/tests/test_agentic_alert_skill.py +0 -0
- {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/tests/test_agentic_conversation.py +0 -0
- {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/tests/test_agentic_general_question.py +0 -0
- {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/tests/test_agentic_guardrail.py +0 -0
- {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/tests/test_agentic_metric_skill.py +0 -0
- {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/tests/test_agentic_search_tool.py +0 -0
- {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/tests/test_agentic_visualization.py +0 -0
- {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/tests/test_alert_skill_evaluator.py +0 -0
- {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/tests/test_connection.py +0 -0
- {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/tests/test_deep_subset.py +0 -0
- {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/tests/test_langfuse_sink.py +0 -0
- {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/tests/test_langfuse_source.py +0 -0
- {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/tests/test_llm_judge.py +0 -0
- {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/tests/test_local_loader.py +0 -0
- {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/tests/test_metric_skill_evaluator.py +0 -0
- {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/tests/test_models.py +0 -0
- {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/tests/test_reporting.py +0 -0
- {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/tests/test_runner.py +0 -0
- {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/tests/test_scoring.py +0 -0
- {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/tests/test_search_tool_evaluator.py +0 -0
- {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/tests/test_summary_client.py +0 -0
- {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/tests/test_summary_evaluator.py +0 -0
- {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/tests/test_text_evaluators.py +0 -0
- {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/tests/test_visualization_evaluator.py +0 -0
- {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/tests/test_workspace.py +0 -0
- {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/tox.ini +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: gooddata-eval
|
|
3
|
-
Version: 1.71.1.
|
|
3
|
+
Version: 1.71.1.dev2
|
|
4
4
|
Summary: Evaluate the GoodData AI agent against your own questions and models.
|
|
5
5
|
Project-URL: Source, https://github.com/gooddata/gooddata-python-sdk
|
|
6
6
|
Author-email: GoodData <support@gooddata.com>
|
|
@@ -17,7 +17,7 @@ Classifier: Topic :: Scientific/Engineering
|
|
|
17
17
|
Classifier: Topic :: Software Development
|
|
18
18
|
Classifier: Typing :: Typed
|
|
19
19
|
Requires-Python: >=3.10
|
|
20
|
-
Requires-Dist: gooddata-sdk~=1.71.1.
|
|
20
|
+
Requires-Dist: gooddata-sdk~=1.71.1.dev2
|
|
21
21
|
Requires-Dist: httpx<1.0,>=0.27
|
|
22
22
|
Requires-Dist: orjson<4.0.0,>=3.9.15
|
|
23
23
|
Requires-Dist: pydantic<3.0,>=2.6
|
|
@@ -120,6 +120,7 @@ Both provider name and provider id are accepted as the prefix.
|
|
|
120
120
|
|---|---|---|
|
|
121
121
|
| `--runs K` | `2` | Independent runs per item (pass@K). An item passes if any run passes. |
|
|
122
122
|
| `--concurrency K` | `1` | Number of items evaluated concurrently. `1` = sequential (default). Increase to load-test the agent under simultaneous requests. Progress output interleaves when K > 1. |
|
|
123
|
+
| `--reasoning-effort LEVEL` | server default | `LOW`, `MEDIUM` or `HIGH`, sent as `options.reasoningEffort` on every chat message. Requires the `enableGenAiReasoningEffort` feature flag on the target organization — without it the server ignores the value. Applies to chat items only; `dashboard_summary` items go through the summary endpoint, which has no such option. |
|
|
123
124
|
|
|
124
125
|
#### Output
|
|
125
126
|
|
|
@@ -132,7 +133,7 @@ Both provider name and provider id are accepted as the prefix.
|
|
|
132
133
|
|
|
133
134
|
| Flag | Description |
|
|
134
135
|
|---|---|
|
|
135
|
-
| `--langfuse` | Log scores and traces to Langfuse after each item. Requires `--langfuse-dataset`. Creates one named experiment run per model (`gd-eval-{timestamp}-{model}`). Requires `LANGFUSE_PUBLIC_KEY`, `LANGFUSE_SECRET_KEY`, `LANGFUSE_HOST`. |
|
|
136
|
+
| `--langfuse` | Log scores and traces to Langfuse after each item. Requires `--langfuse-dataset`. Creates one named experiment run per model (`gd-eval-{timestamp}-{model}`, suffixed `-effort-{level}` when `--reasoning-effort` is set so runs differing only by effort stay separate). Requires `LANGFUSE_PUBLIC_KEY`, `LANGFUSE_SECRET_KEY`, `LANGFUSE_HOST`. |
|
|
136
137
|
|
|
137
138
|
### JSON report shape
|
|
138
139
|
|
|
@@ -92,6 +92,7 @@ Both provider name and provider id are accepted as the prefix.
|
|
|
92
92
|
|---|---|---|
|
|
93
93
|
| `--runs K` | `2` | Independent runs per item (pass@K). An item passes if any run passes. |
|
|
94
94
|
| `--concurrency K` | `1` | Number of items evaluated concurrently. `1` = sequential (default). Increase to load-test the agent under simultaneous requests. Progress output interleaves when K > 1. |
|
|
95
|
+
| `--reasoning-effort LEVEL` | server default | `LOW`, `MEDIUM` or `HIGH`, sent as `options.reasoningEffort` on every chat message. Requires the `enableGenAiReasoningEffort` feature flag on the target organization — without it the server ignores the value. Applies to chat items only; `dashboard_summary` items go through the summary endpoint, which has no such option. |
|
|
95
96
|
|
|
96
97
|
#### Output
|
|
97
98
|
|
|
@@ -104,7 +105,7 @@ Both provider name and provider id are accepted as the prefix.
|
|
|
104
105
|
|
|
105
106
|
| Flag | Description |
|
|
106
107
|
|---|---|
|
|
107
|
-
| `--langfuse` | Log scores and traces to Langfuse after each item. Requires `--langfuse-dataset`. Creates one named experiment run per model (`gd-eval-{timestamp}-{model}`). Requires `LANGFUSE_PUBLIC_KEY`, `LANGFUSE_SECRET_KEY`, `LANGFUSE_HOST`. |
|
|
108
|
+
| `--langfuse` | Log scores and traces to Langfuse after each item. Requires `--langfuse-dataset`. Creates one named experiment run per model (`gd-eval-{timestamp}-{model}`, suffixed `-effort-{level}` when `--reasoning-effort` is set so runs differing only by effort stay separate). Requires `LANGFUSE_PUBLIC_KEY`, `LANGFUSE_SECRET_KEY`, `LANGFUSE_HOST`. |
|
|
108
109
|
|
|
109
110
|
### JSON report shape
|
|
110
111
|
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
# (C) 2026 GoodData Corporation
|
|
2
2
|
[project]
|
|
3
3
|
name = "gooddata-eval"
|
|
4
|
-
version = "1.71.1.
|
|
4
|
+
version = "1.71.1.dev2"
|
|
5
5
|
description = "Evaluate the GoodData AI agent against your own questions and models."
|
|
6
6
|
readme = "README.md"
|
|
7
7
|
license = "MIT"
|
|
@@ -11,7 +11,7 @@ authors = [
|
|
|
11
11
|
keywords = ["gooddata", "ai", "evaluation", "llm", "analytics", "cli"]
|
|
12
12
|
requires-python = ">=3.10"
|
|
13
13
|
dependencies = [
|
|
14
|
-
"gooddata-sdk~=1.71.1.
|
|
14
|
+
"gooddata-sdk~=1.71.1.dev2",
|
|
15
15
|
"httpx>=0.27,<1.0",
|
|
16
16
|
"orjson>=3.9.15,<4.0.0",
|
|
17
17
|
"pydantic>=2.6,<3.0",
|
{gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/cli/agentic_runner.py
RENAMED
|
@@ -14,6 +14,7 @@ from gooddata_eval.core.agentic.guardrail import evaluate_agentic_guardrail
|
|
|
14
14
|
from gooddata_eval.core.agentic.metric_skill import evaluate_agentic_metric_skill
|
|
15
15
|
from gooddata_eval.core.agentic.search_tool import evaluate_agentic_search_tool
|
|
16
16
|
from gooddata_eval.core.agentic.visualization import evaluate_agentic_visualization
|
|
17
|
+
from gooddata_eval.core.config import ReasoningEffort
|
|
17
18
|
from gooddata_eval.core.models import CreatedVisualization, DatasetItem
|
|
18
19
|
from gooddata_eval.core.runner import EvalReport, ItemReport
|
|
19
20
|
|
|
@@ -24,6 +25,7 @@ class _LfKw(TypedDict, total=False):
|
|
|
24
25
|
dataset_name: str
|
|
25
26
|
run_timestamp: str
|
|
26
27
|
model_version_override: str | None
|
|
28
|
+
reasoning_effort: ReasoningEffort | None
|
|
27
29
|
|
|
28
30
|
|
|
29
31
|
AGENTIC_TEST_KINDS = frozenset(
|
|
@@ -80,6 +82,7 @@ def _dispatch_agentic(
|
|
|
80
82
|
langfuse: Any,
|
|
81
83
|
run_ts: str,
|
|
82
84
|
model_version_override: str | None,
|
|
85
|
+
reasoning_effort: ReasoningEffort | None = None,
|
|
83
86
|
) -> None:
|
|
84
87
|
"""Call the appropriate evaluate_agentic_* function for the item's test_kind."""
|
|
85
88
|
kind = item.test_kind
|
|
@@ -90,6 +93,7 @@ def _dispatch_agentic(
|
|
|
90
93
|
"dataset_name": item.dataset_name,
|
|
91
94
|
"run_timestamp": run_ts,
|
|
92
95
|
"model_version_override": model_version_override,
|
|
96
|
+
"reasoning_effort": reasoning_effort,
|
|
93
97
|
}
|
|
94
98
|
|
|
95
99
|
if kind in ("vis_agentic", "agentic_visualization"):
|
|
@@ -176,6 +180,7 @@ def run_agentic_items(
|
|
|
176
180
|
*,
|
|
177
181
|
k: int = 2,
|
|
178
182
|
model_version: str | None = None,
|
|
183
|
+
reasoning_effort: ReasoningEffort | None = None,
|
|
179
184
|
use_langfuse: bool = False,
|
|
180
185
|
run_ts: str,
|
|
181
186
|
on_item_start: Any = None,
|
|
@@ -202,7 +207,7 @@ def run_agentic_items(
|
|
|
202
207
|
)
|
|
203
208
|
t0 = time.perf_counter()
|
|
204
209
|
try:
|
|
205
|
-
_dispatch_agentic(item, host, token, workspace_id, k, langfuse, run_ts, model_version)
|
|
210
|
+
_dispatch_agentic(item, host, token, workspace_id, k, langfuse, run_ts, model_version, reasoning_effort)
|
|
206
211
|
item_report.pass_at_k = True
|
|
207
212
|
item_report.runs = k
|
|
208
213
|
except AssertionError as exc:
|
|
@@ -6,6 +6,7 @@ import sys
|
|
|
6
6
|
import threading
|
|
7
7
|
from datetime import datetime, timezone
|
|
8
8
|
from pathlib import Path
|
|
9
|
+
from typing import get_args
|
|
9
10
|
|
|
10
11
|
import httpx
|
|
11
12
|
from gooddata_api_client.exceptions import ApiException
|
|
@@ -14,7 +15,7 @@ from rich.table import Table
|
|
|
14
15
|
|
|
15
16
|
from gooddata_eval.cli.agentic_runner import AGENTIC_TEST_KINDS, run_agentic_items
|
|
16
17
|
from gooddata_eval.core.chat.sse_client import ChatClient
|
|
17
|
-
from gooddata_eval.core.config import RunConfig
|
|
18
|
+
from gooddata_eval.core.config import ReasoningEffort, RunConfig
|
|
18
19
|
from gooddata_eval.core.connection import ConnectionError_, resolve_connection
|
|
19
20
|
from gooddata_eval.core.dataset.local import load_local_dataset
|
|
20
21
|
from gooddata_eval.core.langfuse.sink import LangfuseSink
|
|
@@ -104,6 +105,13 @@ def _build_parser() -> argparse.ArgumentParser:
|
|
|
104
105
|
dest="preserve_failed",
|
|
105
106
|
help="Keep failed conversations on the server for post-mortem inspection.",
|
|
106
107
|
)
|
|
108
|
+
run.add_argument(
|
|
109
|
+
"--reasoning-effort",
|
|
110
|
+
dest="reasoning_effort",
|
|
111
|
+
choices=list(get_args(ReasoningEffort)),
|
|
112
|
+
help="Reasoning effort requested per message. Requires the enableGenAiReasoningEffort "
|
|
113
|
+
"feature flag on the target organization; without it the server ignores the value.",
|
|
114
|
+
)
|
|
107
115
|
run.add_argument(
|
|
108
116
|
"--langfuse",
|
|
109
117
|
action="store_true",
|
|
@@ -292,6 +300,10 @@ def _run(config: RunConfig) -> int:
|
|
|
292
300
|
progress_console.print(f"Provider={provider_display}, model={resolved.model_id}{switched}")
|
|
293
301
|
|
|
294
302
|
run_name = f"gd-eval-{run_ts}-{resolved.model_id}"
|
|
303
|
+
if config.reasoning_effort:
|
|
304
|
+
# Without this two runs differing only by effort share a name and are
|
|
305
|
+
# indistinguishable in the report, which is the comparison this exists for.
|
|
306
|
+
run_name = f"{run_name}-effort-{config.reasoning_effort.lower()}"
|
|
295
307
|
if progress_console and config.log_to_langfuse:
|
|
296
308
|
progress_console.print(f"Logging to Langfuse run '{run_name}'...")
|
|
297
309
|
|
|
@@ -307,6 +319,7 @@ def _run(config: RunConfig) -> int:
|
|
|
307
319
|
run_name=run_name,
|
|
308
320
|
model_id=resolved.model_id,
|
|
309
321
|
provider_type=resolved.provider_type,
|
|
322
|
+
reasoning_effort=config.reasoning_effort,
|
|
310
323
|
)
|
|
311
324
|
|
|
312
325
|
def on_langfuse_item_done(
|
|
@@ -329,6 +342,7 @@ def _run(config: RunConfig) -> int:
|
|
|
329
342
|
workspace_id=config.workspace_id,
|
|
330
343
|
k=config.runs,
|
|
331
344
|
model_version=resolved.model_id,
|
|
345
|
+
reasoning_effort=config.reasoning_effort,
|
|
332
346
|
use_langfuse=config.log_to_langfuse,
|
|
333
347
|
run_ts=run_ts,
|
|
334
348
|
on_item_start=on_item_start,
|
|
@@ -342,6 +356,7 @@ def _run(config: RunConfig) -> int:
|
|
|
342
356
|
token=config.token,
|
|
343
357
|
workspace_id=config.workspace_id,
|
|
344
358
|
preserve_failed=config.preserve_failed,
|
|
359
|
+
reasoning_effort=config.reasoning_effort,
|
|
345
360
|
),
|
|
346
361
|
SummaryClient(host=config.host, token=config.token, workspace_id=config.workspace_id),
|
|
347
362
|
)
|
|
@@ -433,6 +448,7 @@ def main(argv: list[str] | None = None) -> int:
|
|
|
433
448
|
quiet=args.quiet,
|
|
434
449
|
kind=args.kind,
|
|
435
450
|
preserve_failed=args.preserve_failed,
|
|
451
|
+
reasoning_effort=args.reasoning_effort,
|
|
436
452
|
)
|
|
437
453
|
return _run(config)
|
|
438
454
|
except (
|
{gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/agentic/_langfuse.py
RENAMED
|
@@ -15,6 +15,8 @@ from typing import Any
|
|
|
15
15
|
|
|
16
16
|
import httpx
|
|
17
17
|
|
|
18
|
+
from gooddata_eval.core.config import ReasoningEffort, normalize_reasoning_effort
|
|
19
|
+
|
|
18
20
|
_log = logging.getLogger(__name__)
|
|
19
21
|
|
|
20
22
|
# ---------------------------------------------------------------------------
|
|
@@ -384,6 +386,7 @@ def build_run_context(
|
|
|
384
386
|
run_timestamp: str | None,
|
|
385
387
|
model_version_override: str | None,
|
|
386
388
|
run_metadata_extra: dict[str, Any] | None = None,
|
|
389
|
+
reasoning_effort: ReasoningEffort | None = None,
|
|
387
390
|
) -> tuple[str, dict[str, Any]]:
|
|
388
391
|
"""Return (run_name_base, run_metadata) with model version resolved from workspace API.
|
|
389
392
|
|
|
@@ -399,15 +402,24 @@ def build_run_context(
|
|
|
399
402
|
(e.g. a testing-framework tag or a CI run id for scoping). Default None keeps
|
|
400
403
|
behavior unchanged. The SDK-derived model_version is applied last and cannot
|
|
401
404
|
be overwritten by this dict.
|
|
405
|
+
reasoning_effort: Effort the run requested, stamped into both the run name and
|
|
406
|
+
the metadata so effort-varying runs stay comparable side by side.
|
|
402
407
|
"""
|
|
408
|
+
effort = normalize_reasoning_effort(reasoning_effort)
|
|
403
409
|
model = get_model_version(host, token, workspace_id, model_version_override)
|
|
404
410
|
ts = run_timestamp or datetime.now().strftime("%Y-%m-%d_%H-%M-%S")
|
|
405
411
|
base = f"{dataset_name}_{ts}"
|
|
406
412
|
if model:
|
|
407
413
|
base = f"{base}_{model}"
|
|
414
|
+
# Part of the run name, not just metadata: two runs that differ only by effort would
|
|
415
|
+
# otherwise collide on the same name and be indistinguishable in the report.
|
|
416
|
+
if effort:
|
|
417
|
+
base = f"{base}_effort-{effort.lower()}"
|
|
408
418
|
# Caller supplies its own run tags (e.g. testing_framework); model_version is applied
|
|
409
419
|
# last so the SDK-derived value cannot be overwritten by run_metadata_extra.
|
|
410
420
|
metadata: dict[str, Any] = dict(run_metadata_extra) if run_metadata_extra else {}
|
|
421
|
+
if effort:
|
|
422
|
+
metadata["reasoning_effort"] = effort
|
|
411
423
|
if model:
|
|
412
424
|
metadata["model_version"] = model
|
|
413
425
|
return base, metadata
|
|
@@ -13,6 +13,7 @@ from gooddata_sdk import GoodDataSdk
|
|
|
13
13
|
|
|
14
14
|
from gooddata_eval.core.agentic._catalog import CatalogMetricAlert
|
|
15
15
|
from gooddata_eval.core.chat.sse_client import ChatClient
|
|
16
|
+
from gooddata_eval.core.config import ReasoningEffort
|
|
16
17
|
from gooddata_eval.core.models import ToolCallEvent
|
|
17
18
|
|
|
18
19
|
try:
|
|
@@ -342,11 +343,12 @@ def run_agentic_alert_skill(
|
|
|
342
343
|
k: int = _DEFAULT_K,
|
|
343
344
|
max_iterations: int = _DEFAULT_MAX_ITERATIONS,
|
|
344
345
|
initial_conversation_id: str | None = None,
|
|
346
|
+
reasoning_effort: ReasoningEffort | None = None,
|
|
345
347
|
) -> AgenticAlertSummary:
|
|
346
348
|
"""Run the alert-skill agentic evaluation K times and return a summary."""
|
|
347
349
|
expected = _normalize_expected_output(expected_output)
|
|
348
350
|
run_results: list[AlertRunResult] = []
|
|
349
|
-
client = ChatClient(host=host, token=token, workspace_id=workspace_id)
|
|
351
|
+
client = ChatClient(host=host, token=token, workspace_id=workspace_id, reasoning_effort=reasoning_effort)
|
|
350
352
|
sdk = GoodDataSdk.create(host, token)
|
|
351
353
|
|
|
352
354
|
def _run_once(conv_id: str) -> AlertRunResult:
|
|
@@ -462,6 +464,7 @@ def evaluate_agentic_alert_skill(
|
|
|
462
464
|
run_timestamp: str | None = None,
|
|
463
465
|
model_version_override: str | None = None,
|
|
464
466
|
run_metadata_extra: dict | None = None,
|
|
467
|
+
reasoning_effort: ReasoningEffort | None = None,
|
|
465
468
|
) -> None:
|
|
466
469
|
"""Run alert-skill evaluation, log to Langfuse, and raise AlertSkillAssertionError on failure."""
|
|
467
470
|
from datetime import datetime as _dt # noqa: PLC0415
|
|
@@ -481,6 +484,7 @@ def evaluate_agentic_alert_skill(
|
|
|
481
484
|
k=k,
|
|
482
485
|
max_iterations=max_iterations,
|
|
483
486
|
initial_conversation_id=initial_conversation_id,
|
|
487
|
+
reasoning_effort=reasoning_effort,
|
|
484
488
|
)
|
|
485
489
|
|
|
486
490
|
if langfuse is not None and dataset_item_id:
|
|
@@ -500,6 +504,7 @@ def evaluate_agentic_alert_skill(
|
|
|
500
504
|
run_timestamp,
|
|
501
505
|
model_version_override,
|
|
502
506
|
run_metadata_extra,
|
|
507
|
+
reasoning_effort,
|
|
503
508
|
)
|
|
504
509
|
traces_by_conv = find_traces_per_conversation(
|
|
505
510
|
langfuse,
|
|
@@ -14,6 +14,7 @@ from pydantic import BaseModel
|
|
|
14
14
|
from gooddata_eval.core.agentic.alert_skill import render_alert_proposal
|
|
15
15
|
from gooddata_eval.core.agentic.metric_skill import _delete_metric, _extract_created_metric_ids
|
|
16
16
|
from gooddata_eval.core.chat.sse_client import ChatClient
|
|
17
|
+
from gooddata_eval.core.config import ReasoningEffort
|
|
17
18
|
from gooddata_eval.core.models import ChatResult, ToolCallEvent
|
|
18
19
|
from gooddata_eval.core.scoring import (
|
|
19
20
|
check_filters,
|
|
@@ -278,6 +279,7 @@ def run_agentic_conversation(
|
|
|
278
279
|
fixture: ConversationFixture,
|
|
279
280
|
max_clarification_turns: int = 20,
|
|
280
281
|
initial_conversation_id: str | None = None,
|
|
282
|
+
reasoning_effort: ReasoningEffort | None = None,
|
|
281
283
|
) -> ConversationResult:
|
|
282
284
|
"""Run a multi-turn, multi-skill conversation evaluation (no K-runs).
|
|
283
285
|
|
|
@@ -285,7 +287,7 @@ def run_agentic_conversation(
|
|
|
285
287
|
trigger up to *max_clarification_turns* additional rounds of simulated-user
|
|
286
288
|
replies before the agent produces the expected output.
|
|
287
289
|
"""
|
|
288
|
-
client = ChatClient(host=host, token=token, workspace_id=workspace_id)
|
|
290
|
+
client = ChatClient(host=host, token=token, workspace_id=workspace_id, reasoning_effort=reasoning_effort)
|
|
289
291
|
sdk = GoodDataSdk.create(host, token)
|
|
290
292
|
turn_results: list[TurnResult] = []
|
|
291
293
|
turn_outputs: dict[str, dict] = {}
|
|
@@ -403,6 +405,7 @@ def evaluate_agentic_conversation(
|
|
|
403
405
|
run_timestamp: str | None = None,
|
|
404
406
|
model_version_override: str | None = None,
|
|
405
407
|
run_metadata_extra: dict | None = None,
|
|
408
|
+
reasoning_effort: ReasoningEffort | None = None,
|
|
406
409
|
) -> None:
|
|
407
410
|
"""Run conversation evaluation, log to Langfuse, and raise on failure."""
|
|
408
411
|
from datetime import datetime as _dt # noqa: PLC0415
|
|
@@ -420,6 +423,7 @@ def evaluate_agentic_conversation(
|
|
|
420
423
|
fixture=fixture,
|
|
421
424
|
max_clarification_turns=max_clarification_turns,
|
|
422
425
|
initial_conversation_id=initial_conversation_id,
|
|
426
|
+
reasoning_effort=reasoning_effort,
|
|
423
427
|
)
|
|
424
428
|
|
|
425
429
|
if langfuse is not None and dataset_item_id:
|
|
@@ -439,6 +443,7 @@ def evaluate_agentic_conversation(
|
|
|
439
443
|
run_timestamp,
|
|
440
444
|
model_version_override,
|
|
441
445
|
run_metadata_extra,
|
|
446
|
+
reasoning_effort,
|
|
442
447
|
)
|
|
443
448
|
traces_by_conv = find_traces_per_conversation(
|
|
444
449
|
langfuse,
|
|
@@ -6,6 +6,7 @@ from __future__ import annotations
|
|
|
6
6
|
from dataclasses import dataclass
|
|
7
7
|
|
|
8
8
|
from gooddata_eval.core.chat.sse_client import ChatClient
|
|
9
|
+
from gooddata_eval.core.config import ReasoningEffort
|
|
9
10
|
from gooddata_eval.core.evaluators._llm_judge import LLMJudge
|
|
10
11
|
|
|
11
12
|
_DEFAULT_K = 1
|
|
@@ -71,10 +72,11 @@ def run_agentic_general_question(
|
|
|
71
72
|
expected_output: str,
|
|
72
73
|
k: int = _DEFAULT_K,
|
|
73
74
|
initial_conversation_id: str | None = None,
|
|
75
|
+
reasoning_effort: ReasoningEffort | None = None,
|
|
74
76
|
) -> AgenticGeneralQuestionSummary:
|
|
75
77
|
"""Run the general-question agentic evaluation K times and return a summary."""
|
|
76
78
|
run_results: list[GeneralQuestionResult] = []
|
|
77
|
-
client = ChatClient(host=host, token=token, workspace_id=workspace_id)
|
|
79
|
+
client = ChatClient(host=host, token=token, workspace_id=workspace_id, reasoning_effort=reasoning_effort)
|
|
78
80
|
judge = LLMJudge(_GENERAL_QUESTION_EVALUATION_STEPS, model="gpt-4o")
|
|
79
81
|
|
|
80
82
|
try:
|
|
@@ -153,6 +155,7 @@ def evaluate_agentic_general_question(
|
|
|
153
155
|
run_timestamp: str | None = None,
|
|
154
156
|
model_version_override: str | None = None,
|
|
155
157
|
run_metadata_extra: dict | None = None,
|
|
158
|
+
reasoning_effort: ReasoningEffort | None = None,
|
|
156
159
|
) -> None:
|
|
157
160
|
"""Run general-question evaluation, log to Langfuse, and raise on failure."""
|
|
158
161
|
from datetime import datetime as _dt # noqa: PLC0415
|
|
@@ -171,6 +174,7 @@ def evaluate_agentic_general_question(
|
|
|
171
174
|
expected_output=expected_output,
|
|
172
175
|
k=k,
|
|
173
176
|
initial_conversation_id=initial_conversation_id,
|
|
177
|
+
reasoning_effort=reasoning_effort,
|
|
174
178
|
)
|
|
175
179
|
|
|
176
180
|
if langfuse is not None and dataset_item_id:
|
|
@@ -190,6 +194,7 @@ def evaluate_agentic_general_question(
|
|
|
190
194
|
run_timestamp,
|
|
191
195
|
model_version_override,
|
|
192
196
|
run_metadata_extra,
|
|
197
|
+
reasoning_effort,
|
|
193
198
|
)
|
|
194
199
|
traces_by_conv = find_traces_per_conversation(
|
|
195
200
|
langfuse,
|
{gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/agentic/guardrail.py
RENAMED
|
@@ -6,6 +6,7 @@ from __future__ import annotations
|
|
|
6
6
|
from dataclasses import dataclass
|
|
7
7
|
|
|
8
8
|
from gooddata_eval.core.chat.sse_client import ChatClient
|
|
9
|
+
from gooddata_eval.core.config import ReasoningEffort
|
|
9
10
|
from gooddata_eval.core.evaluators._llm_judge import LLMJudge
|
|
10
11
|
|
|
11
12
|
_DEFAULT_K = 1
|
|
@@ -68,10 +69,11 @@ def run_agentic_guardrail(
|
|
|
68
69
|
expected_output: str,
|
|
69
70
|
k: int = _DEFAULT_K,
|
|
70
71
|
initial_conversation_id: str | None = None,
|
|
72
|
+
reasoning_effort: ReasoningEffort | None = None,
|
|
71
73
|
) -> AgenticGuardrailSummary:
|
|
72
74
|
"""Run the guardrail agentic evaluation K times and return a summary."""
|
|
73
75
|
run_results: list[GuardrailResult] = []
|
|
74
|
-
client = ChatClient(host=host, token=token, workspace_id=workspace_id)
|
|
76
|
+
client = ChatClient(host=host, token=token, workspace_id=workspace_id, reasoning_effort=reasoning_effort)
|
|
75
77
|
judge = LLMJudge(_GUARDRAIL_EVALUATION_STEPS, model="gpt-4o")
|
|
76
78
|
|
|
77
79
|
try:
|
|
@@ -150,6 +152,7 @@ def evaluate_agentic_guardrail(
|
|
|
150
152
|
run_timestamp: str | None = None,
|
|
151
153
|
model_version_override: str | None = None,
|
|
152
154
|
run_metadata_extra: dict | None = None,
|
|
155
|
+
reasoning_effort: ReasoningEffort | None = None,
|
|
153
156
|
) -> None:
|
|
154
157
|
"""Run guardrail evaluation, log to Langfuse, and raise on failure."""
|
|
155
158
|
from datetime import datetime as _dt # noqa: PLC0415
|
|
@@ -168,6 +171,7 @@ def evaluate_agentic_guardrail(
|
|
|
168
171
|
expected_output=expected_output,
|
|
169
172
|
k=k,
|
|
170
173
|
initial_conversation_id=initial_conversation_id,
|
|
174
|
+
reasoning_effort=reasoning_effort,
|
|
171
175
|
)
|
|
172
176
|
|
|
173
177
|
if langfuse is not None and dataset_item_id:
|
|
@@ -187,6 +191,7 @@ def evaluate_agentic_guardrail(
|
|
|
187
191
|
run_timestamp,
|
|
188
192
|
model_version_override,
|
|
189
193
|
run_metadata_extra,
|
|
194
|
+
reasoning_effort,
|
|
190
195
|
)
|
|
191
196
|
traces_by_conv = find_traces_per_conversation(
|
|
192
197
|
langfuse,
|
|
@@ -11,6 +11,7 @@ from typing import Any
|
|
|
11
11
|
from gooddata_sdk import GoodDataSdk
|
|
12
12
|
|
|
13
13
|
from gooddata_eval.core.chat.sse_client import ChatClient
|
|
14
|
+
from gooddata_eval.core.config import ReasoningEffort
|
|
14
15
|
from gooddata_eval.core.models import ToolCallEvent
|
|
15
16
|
|
|
16
17
|
try:
|
|
@@ -232,6 +233,7 @@ def run_agentic_metric_skill(
|
|
|
232
233
|
k: int = _DEFAULT_K,
|
|
233
234
|
max_iterations: int = _DEFAULT_MAX_ITERATIONS,
|
|
234
235
|
initial_conversation_id: str | None = None,
|
|
236
|
+
reasoning_effort: ReasoningEffort | None = None,
|
|
235
237
|
) -> AgenticMetricSummary:
|
|
236
238
|
"""Run the metric-skill agentic evaluation K times and return a summary.
|
|
237
239
|
|
|
@@ -240,7 +242,7 @@ def run_agentic_metric_skill(
|
|
|
240
242
|
"""
|
|
241
243
|
expected_outputs: list[dict] = expected_output if isinstance(expected_output, list) else [expected_output]
|
|
242
244
|
run_results: list[MetricRunResult] = []
|
|
243
|
-
client = ChatClient(host=host, token=token, workspace_id=workspace_id)
|
|
245
|
+
client = ChatClient(host=host, token=token, workspace_id=workspace_id, reasoning_effort=reasoning_effort)
|
|
244
246
|
sdk = GoodDataSdk.create(host, token)
|
|
245
247
|
|
|
246
248
|
try:
|
|
@@ -300,6 +302,7 @@ def evaluate_agentic_metric_skill(
|
|
|
300
302
|
run_timestamp: str | None = None,
|
|
301
303
|
model_version_override: str | None = None,
|
|
302
304
|
run_metadata_extra: dict | None = None,
|
|
305
|
+
reasoning_effort: ReasoningEffort | None = None,
|
|
303
306
|
) -> None:
|
|
304
307
|
"""Run metric-skill evaluation, log to Langfuse, and raise MetricSkillAssertionError on failure."""
|
|
305
308
|
from datetime import datetime as _dt # noqa: PLC0415
|
|
@@ -319,6 +322,7 @@ def evaluate_agentic_metric_skill(
|
|
|
319
322
|
k=k,
|
|
320
323
|
max_iterations=max_iterations,
|
|
321
324
|
initial_conversation_id=initial_conversation_id,
|
|
325
|
+
reasoning_effort=reasoning_effort,
|
|
322
326
|
)
|
|
323
327
|
|
|
324
328
|
if langfuse is not None and dataset_item_id:
|
|
@@ -338,6 +342,7 @@ def evaluate_agentic_metric_skill(
|
|
|
338
342
|
run_timestamp,
|
|
339
343
|
model_version_override,
|
|
340
344
|
run_metadata_extra,
|
|
345
|
+
reasoning_effort,
|
|
341
346
|
)
|
|
342
347
|
traces_by_conv = find_traces_per_conversation(
|
|
343
348
|
langfuse,
|
|
@@ -6,6 +6,7 @@ from __future__ import annotations
|
|
|
6
6
|
from dataclasses import dataclass
|
|
7
7
|
|
|
8
8
|
from gooddata_eval.core.chat.sse_client import ChatClient
|
|
9
|
+
from gooddata_eval.core.config import ReasoningEffort
|
|
9
10
|
from gooddata_eval.core.models import ToolCallEvent
|
|
10
11
|
|
|
11
12
|
_DEFAULT_K = 1
|
|
@@ -67,11 +68,12 @@ def run_agentic_search_tool(
|
|
|
67
68
|
expected_tool_call: dict,
|
|
68
69
|
k: int = _DEFAULT_K,
|
|
69
70
|
initial_conversation_id: str | None = None,
|
|
71
|
+
reasoning_effort: ReasoningEffort | None = None,
|
|
70
72
|
) -> AgenticSearchSummary:
|
|
71
73
|
"""Run the search-tool agentic evaluation K times (single-turn each)."""
|
|
72
74
|
run_results: list[SearchResult] = []
|
|
73
75
|
|
|
74
|
-
client = ChatClient(host=host, token=token, workspace_id=workspace_id)
|
|
76
|
+
client = ChatClient(host=host, token=token, workspace_id=workspace_id, reasoning_effort=reasoning_effort)
|
|
75
77
|
try:
|
|
76
78
|
conv_id_0 = initial_conversation_id if initial_conversation_id is not None else client.create_conversation()
|
|
77
79
|
try:
|
|
@@ -144,6 +146,7 @@ def evaluate_agentic_search_tool(
|
|
|
144
146
|
run_timestamp: str | None = None,
|
|
145
147
|
model_version_override: str | None = None,
|
|
146
148
|
run_metadata_extra: dict | None = None,
|
|
149
|
+
reasoning_effort: ReasoningEffort | None = None,
|
|
147
150
|
) -> None:
|
|
148
151
|
"""Run search-tool evaluation, log to Langfuse, and raise SearchToolAssertionError on failure."""
|
|
149
152
|
from datetime import datetime as _dt # noqa: PLC0415
|
|
@@ -162,6 +165,7 @@ def evaluate_agentic_search_tool(
|
|
|
162
165
|
expected_tool_call=expected_tool_call,
|
|
163
166
|
k=k,
|
|
164
167
|
initial_conversation_id=initial_conversation_id,
|
|
168
|
+
reasoning_effort=reasoning_effort,
|
|
165
169
|
)
|
|
166
170
|
|
|
167
171
|
if langfuse is not None and dataset_item_id:
|
|
@@ -181,6 +185,7 @@ def evaluate_agentic_search_tool(
|
|
|
181
185
|
run_timestamp,
|
|
182
186
|
model_version_override,
|
|
183
187
|
run_metadata_extra,
|
|
188
|
+
reasoning_effort,
|
|
184
189
|
)
|
|
185
190
|
traces_by_conv = find_traces_per_conversation(
|
|
186
191
|
langfuse,
|
|
@@ -11,6 +11,7 @@ import os
|
|
|
11
11
|
from dataclasses import dataclass
|
|
12
12
|
|
|
13
13
|
from gooddata_eval.core.chat.sse_client import ChatClient
|
|
14
|
+
from gooddata_eval.core.config import ReasoningEffort
|
|
14
15
|
from gooddata_eval.core.evaluators.visualization import (
|
|
15
16
|
EvaluationResult,
|
|
16
17
|
_check_visualization_skill_activated,
|
|
@@ -203,6 +204,7 @@ def run_agentic_visualization(
|
|
|
203
204
|
k: int = _DEFAULT_K,
|
|
204
205
|
max_iterations: int = _DEFAULT_MAX_ITERATIONS,
|
|
205
206
|
initial_conversation_id: str | None = None,
|
|
207
|
+
reasoning_effort: ReasoningEffort | None = None,
|
|
206
208
|
) -> AgenticRunSummary:
|
|
207
209
|
"""Run K independent conversations and return evaluation results.
|
|
208
210
|
|
|
@@ -211,7 +213,7 @@ def run_agentic_visualization(
|
|
|
211
213
|
fresh conversations. Caller-supplied conversations are not deleted; all
|
|
212
214
|
conversations created by this function are deleted on completion.
|
|
213
215
|
"""
|
|
214
|
-
client = ChatClient(host=host, token=token, workspace_id=workspace_id)
|
|
216
|
+
client = ChatClient(host=host, token=token, workspace_id=workspace_id, reasoning_effort=reasoning_effort)
|
|
215
217
|
run_results: list[RunResult] = []
|
|
216
218
|
|
|
217
219
|
try:
|
|
@@ -265,6 +267,7 @@ def evaluate_agentic_visualization(
|
|
|
265
267
|
model_version_override: str | None = None,
|
|
266
268
|
run_metadata_extra: dict | None = None,
|
|
267
269
|
record_output_path: str | None = None,
|
|
270
|
+
reasoning_effort: ReasoningEffort | None = None,
|
|
268
271
|
) -> None:
|
|
269
272
|
"""Run visualization evaluation, log to Langfuse, and raise VisualizationAssertionError on failure."""
|
|
270
273
|
import json as _json # noqa: PLC0415
|
|
@@ -285,6 +288,7 @@ def evaluate_agentic_visualization(
|
|
|
285
288
|
k=k,
|
|
286
289
|
max_iterations=max_iterations,
|
|
287
290
|
initial_conversation_id=initial_conversation_id,
|
|
291
|
+
reasoning_effort=reasoning_effort,
|
|
288
292
|
)
|
|
289
293
|
|
|
290
294
|
if langfuse is not None and dataset_item_id:
|
|
@@ -304,6 +308,7 @@ def evaluate_agentic_visualization(
|
|
|
304
308
|
run_timestamp,
|
|
305
309
|
model_version_override,
|
|
306
310
|
run_metadata_extra,
|
|
311
|
+
reasoning_effort,
|
|
307
312
|
)
|
|
308
313
|
K = len(summary.run_results)
|
|
309
314
|
traces_by_conv = find_traces_per_conversation(
|
{gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/chat/sse_client.py
RENAMED
|
@@ -22,6 +22,7 @@ from typing import Any, Callable, Iterable, TypeVar
|
|
|
22
22
|
|
|
23
23
|
import httpx
|
|
24
24
|
|
|
25
|
+
from gooddata_eval.core.config import ReasoningEffort, normalize_reasoning_effort
|
|
25
26
|
from gooddata_eval.core.models import ChatResult, DatasetItem
|
|
26
27
|
|
|
27
28
|
_log = logging.getLogger(__name__)
|
|
@@ -242,12 +243,28 @@ class ChatClient:
|
|
|
242
243
|
"""Single-turn AI chat client over the GoodData AI conversation endpoints."""
|
|
243
244
|
|
|
244
245
|
def __init__(
|
|
245
|
-
self,
|
|
246
|
+
self,
|
|
247
|
+
host: str,
|
|
248
|
+
token: str,
|
|
249
|
+
workspace_id: str,
|
|
250
|
+
*,
|
|
251
|
+
timeout: float = 300.0,
|
|
252
|
+
preserve_failed: bool = False,
|
|
253
|
+
reasoning_effort: ReasoningEffort | None = None,
|
|
246
254
|
):
|
|
255
|
+
"""Create a chat client bound to one workspace.
|
|
256
|
+
|
|
257
|
+
``reasoning_effort`` (``LOW``/``MEDIUM``/``HIGH``) is sent as
|
|
258
|
+
``options.reasoningEffort`` on every message; when None the key is omitted
|
|
259
|
+
entirely and the server keeps its own default. The server honours it only
|
|
260
|
+
while the ``enableGenAiReasoningEffort`` feature flag is on for the
|
|
261
|
+
organization, so setting it is a request rather than a guarantee.
|
|
262
|
+
"""
|
|
247
263
|
self._base = f"{host.rstrip('/')}/api/v1/ai/workspaces/{workspace_id}/chat/conversations"
|
|
248
264
|
self._auth = {"Authorization": f"Bearer {token}"}
|
|
249
265
|
self._client = httpx.Client(timeout=timeout)
|
|
250
266
|
self._preserve_failed = preserve_failed
|
|
267
|
+
self._reasoning_effort = normalize_reasoning_effort(reasoning_effort)
|
|
251
268
|
|
|
252
269
|
def create_conversation(self) -> str:
|
|
253
270
|
def _do() -> str:
|
|
@@ -271,7 +288,9 @@ class ChatClient:
|
|
|
271
288
|
def send_message(self, conversation_id: str, question: str) -> ChatResult:
|
|
272
289
|
url = f"{self._base}/{conversation_id}/messages"
|
|
273
290
|
headers = {**self._auth, "Accept": "text/event-stream", "Content-Type": "application/json"}
|
|
274
|
-
body = {"item": {"role": "user", "content": {"type": "text", "text": question}}}
|
|
291
|
+
body: dict[str, Any] = {"item": {"role": "user", "content": {"type": "text", "text": question}}}
|
|
292
|
+
if self._reasoning_effort is not None:
|
|
293
|
+
body["options"] = {"reasoningEffort": self._reasoning_effort}
|
|
275
294
|
|
|
276
295
|
def _do() -> ChatResult:
|
|
277
296
|
with self._client.stream("POST", url, json=body, headers=headers) as resp:
|
|
@@ -0,0 +1,46 @@
|
|
|
1
|
+
# (C) 2026 GoodData Corporation
|
|
2
|
+
"""Validated run configuration produced by the CLI and consumed by the runner."""
|
|
3
|
+
|
|
4
|
+
from dataclasses import dataclass, field
|
|
5
|
+
from pathlib import Path
|
|
6
|
+
from typing import Literal, cast, get_args
|
|
7
|
+
|
|
8
|
+
ReasoningEffort = Literal["LOW", "MEDIUM", "HIGH"]
|
|
9
|
+
"""Effort values the AI chat endpoint accepts, uppercase as the server enum requires."""
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
def normalize_reasoning_effort(value: str | None) -> ReasoningEffort | None:
|
|
13
|
+
"""Canonical effort, or None when unset.
|
|
14
|
+
|
|
15
|
+
The `Literal` above only constrains static callers, so normalize once at the
|
|
16
|
+
boundary: without it a lowercase value reaches the endpoint and is rejected as
|
|
17
|
+
an out-of-enum request, while an empty string is sent yet skipped by the
|
|
18
|
+
truthiness checks in the Langfuse writers — leaving a run whose recorded
|
|
19
|
+
identity disagrees with what it actually requested.
|
|
20
|
+
"""
|
|
21
|
+
if value is None:
|
|
22
|
+
return None
|
|
23
|
+
candidate = value.strip().upper()
|
|
24
|
+
if not candidate:
|
|
25
|
+
return None
|
|
26
|
+
if candidate not in get_args(ReasoningEffort):
|
|
27
|
+
raise ValueError(f"Invalid reasoning effort {value!r}; expected one of {', '.join(get_args(ReasoningEffort))}.")
|
|
28
|
+
return cast("ReasoningEffort", candidate)
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
@dataclass
|
|
32
|
+
class RunConfig:
|
|
33
|
+
host: str
|
|
34
|
+
token: str
|
|
35
|
+
workspace_id: str
|
|
36
|
+
dataset_folder: Path | None = None
|
|
37
|
+
langfuse_dataset: str | None = None
|
|
38
|
+
models: list[str] = field(default_factory=list)
|
|
39
|
+
runs: int = 2
|
|
40
|
+
concurrency: int = 1
|
|
41
|
+
json_path: Path | None = None
|
|
42
|
+
log_to_langfuse: bool = False
|
|
43
|
+
quiet: bool = False
|
|
44
|
+
kind: str = "visualization"
|
|
45
|
+
preserve_failed: bool = False
|
|
46
|
+
reasoning_effort: ReasoningEffort | None = None
|