gooddata-eval 1.71.1.dev1__tar.gz → 1.71.1.dev2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (88) hide show
  1. {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/PKG-INFO +4 -3
  2. {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/README.md +2 -1
  3. {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/pyproject.toml +2 -2
  4. {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/cli/agentic_runner.py +6 -1
  5. {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/cli/main.py +17 -1
  6. {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/agentic/_langfuse.py +12 -0
  7. {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/agentic/alert_skill.py +6 -1
  8. {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/agentic/conversation.py +6 -1
  9. {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/agentic/general_question.py +6 -1
  10. {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/agentic/guardrail.py +6 -1
  11. {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/agentic/metric_skill.py +6 -1
  12. {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/agentic/search_tool.py +6 -1
  13. {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/agentic/visualization.py +6 -1
  14. {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/chat/sse_client.py +21 -2
  15. gooddata_eval-1.71.1.dev2/src/gooddata_eval/core/config.py +46 -0
  16. {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/langfuse/sink.py +21 -2
  17. gooddata_eval-1.71.1.dev2/tests/test_agentic_run_context.py +73 -0
  18. {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/tests/test_cli.py +77 -3
  19. {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/tests/test_sse_client.py +77 -2
  20. gooddata_eval-1.71.1.dev1/src/gooddata_eval/core/config.py +0 -22
  21. gooddata_eval-1.71.1.dev1/tests/test_agentic_run_context.py +0 -32
  22. {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/.gitignore +0 -0
  23. {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/LICENSE.txt +0 -0
  24. {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/Makefile +0 -0
  25. {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/__init__.py +0 -0
  26. {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/_version.py +0 -0
  27. {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/cli/__init__.py +0 -0
  28. {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/__init__.py +0 -0
  29. {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/agentic/__init__.py +0 -0
  30. {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/agentic/_catalog.py +0 -0
  31. {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/chat/__init__.py +0 -0
  32. {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/connection.py +0 -0
  33. {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/dataset/__init__.py +0 -0
  34. {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/dataset/langfuse_source.py +0 -0
  35. {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/dataset/local.py +0 -0
  36. {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/evaluators/__init__.py +0 -0
  37. {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/evaluators/_deep_subset.py +0 -0
  38. {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/evaluators/_llm_judge.py +0 -0
  39. {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/evaluators/_text_utils.py +0 -0
  40. {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/evaluators/alert_skill.py +0 -0
  41. {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/evaluators/base.py +0 -0
  42. {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/evaluators/general_question.py +0 -0
  43. {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/evaluators/guardrail.py +0 -0
  44. {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/evaluators/metric_skill.py +0 -0
  45. {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/evaluators/search_tool.py +0 -0
  46. {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/evaluators/summary.py +0 -0
  47. {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/evaluators/visualization.py +0 -0
  48. {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/langfuse/__init__.py +0 -0
  49. {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/models.py +0 -0
  50. {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/reporting/__init__.py +0 -0
  51. {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/reporting/console.py +0 -0
  52. {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/reporting/json_report.py +0 -0
  53. {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/runner.py +0 -0
  54. {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/scoring.py +0 -0
  55. {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/summary/__init__.py +0 -0
  56. {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/summary/http_client.py +0 -0
  57. {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/workspace.py +0 -0
  58. {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/tests/__init__.py +0 -0
  59. {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/tests/conftest.py +0 -0
  60. {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/tests/fixtures/sample_dataset/metric_skill_create.json +0 -0
  61. {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/tests/fixtures/sample_dataset/visualization_revenue.json +0 -0
  62. {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/tests/fixtures/sse_visualization_stream.txt +0 -0
  63. {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/tests/test_agentic_alert_skill.py +0 -0
  64. {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/tests/test_agentic_conversation.py +0 -0
  65. {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/tests/test_agentic_general_question.py +0 -0
  66. {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/tests/test_agentic_guardrail.py +0 -0
  67. {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/tests/test_agentic_metric_skill.py +0 -0
  68. {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/tests/test_agentic_search_tool.py +0 -0
  69. {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/tests/test_agentic_visualization.py +0 -0
  70. {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/tests/test_alert_skill_evaluator.py +0 -0
  71. {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/tests/test_connection.py +0 -0
  72. {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/tests/test_deep_subset.py +0 -0
  73. {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/tests/test_langfuse_sink.py +0 -0
  74. {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/tests/test_langfuse_source.py +0 -0
  75. {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/tests/test_llm_judge.py +0 -0
  76. {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/tests/test_local_loader.py +0 -0
  77. {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/tests/test_metric_skill_evaluator.py +0 -0
  78. {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/tests/test_models.py +0 -0
  79. {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/tests/test_reporting.py +0 -0
  80. {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/tests/test_runner.py +0 -0
  81. {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/tests/test_scoring.py +0 -0
  82. {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/tests/test_search_tool_evaluator.py +0 -0
  83. {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/tests/test_summary_client.py +0 -0
  84. {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/tests/test_summary_evaluator.py +0 -0
  85. {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/tests/test_text_evaluators.py +0 -0
  86. {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/tests/test_visualization_evaluator.py +0 -0
  87. {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/tests/test_workspace.py +0 -0
  88. {gooddata_eval-1.71.1.dev1 → gooddata_eval-1.71.1.dev2}/tox.ini +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: gooddata-eval
3
- Version: 1.71.1.dev1
3
+ Version: 1.71.1.dev2
4
4
  Summary: Evaluate the GoodData AI agent against your own questions and models.
5
5
  Project-URL: Source, https://github.com/gooddata/gooddata-python-sdk
6
6
  Author-email: GoodData <support@gooddata.com>
@@ -17,7 +17,7 @@ Classifier: Topic :: Scientific/Engineering
17
17
  Classifier: Topic :: Software Development
18
18
  Classifier: Typing :: Typed
19
19
  Requires-Python: >=3.10
20
- Requires-Dist: gooddata-sdk~=1.71.1.dev1
20
+ Requires-Dist: gooddata-sdk~=1.71.1.dev2
21
21
  Requires-Dist: httpx<1.0,>=0.27
22
22
  Requires-Dist: orjson<4.0.0,>=3.9.15
23
23
  Requires-Dist: pydantic<3.0,>=2.6
@@ -120,6 +120,7 @@ Both provider name and provider id are accepted as the prefix.
120
120
  |---|---|---|
121
121
  | `--runs K` | `2` | Independent runs per item (pass@K). An item passes if any run passes. |
122
122
  | `--concurrency K` | `1` | Number of items evaluated concurrently. `1` = sequential (default). Increase to load-test the agent under simultaneous requests. Progress output interleaves when K > 1. |
123
+ | `--reasoning-effort LEVEL` | server default | `LOW`, `MEDIUM` or `HIGH`, sent as `options.reasoningEffort` on every chat message. Requires the `enableGenAiReasoningEffort` feature flag on the target organization — without it the server ignores the value. Applies to chat items only; `dashboard_summary` items go through the summary endpoint, which has no such option. |
123
124
 
124
125
  #### Output
125
126
 
@@ -132,7 +133,7 @@ Both provider name and provider id are accepted as the prefix.
132
133
 
133
134
  | Flag | Description |
134
135
  |---|---|
135
- | `--langfuse` | Log scores and traces to Langfuse after each item. Requires `--langfuse-dataset`. Creates one named experiment run per model (`gd-eval-{timestamp}-{model}`). Requires `LANGFUSE_PUBLIC_KEY`, `LANGFUSE_SECRET_KEY`, `LANGFUSE_HOST`. |
136
+ | `--langfuse` | Log scores and traces to Langfuse after each item. Requires `--langfuse-dataset`. Creates one named experiment run per model (`gd-eval-{timestamp}-{model}`, suffixed `-effort-{level}` when `--reasoning-effort` is set so runs differing only by effort stay separate). Requires `LANGFUSE_PUBLIC_KEY`, `LANGFUSE_SECRET_KEY`, `LANGFUSE_HOST`. |
136
137
 
137
138
  ### JSON report shape
138
139
 
@@ -92,6 +92,7 @@ Both provider name and provider id are accepted as the prefix.
92
92
  |---|---|---|
93
93
  | `--runs K` | `2` | Independent runs per item (pass@K). An item passes if any run passes. |
94
94
  | `--concurrency K` | `1` | Number of items evaluated concurrently. `1` = sequential (default). Increase to load-test the agent under simultaneous requests. Progress output interleaves when K > 1. |
95
+ | `--reasoning-effort LEVEL` | server default | `LOW`, `MEDIUM` or `HIGH`, sent as `options.reasoningEffort` on every chat message. Requires the `enableGenAiReasoningEffort` feature flag on the target organization — without it the server ignores the value. Applies to chat items only; `dashboard_summary` items go through the summary endpoint, which has no such option. |
95
96
 
96
97
  #### Output
97
98
 
@@ -104,7 +105,7 @@ Both provider name and provider id are accepted as the prefix.
104
105
 
105
106
  | Flag | Description |
106
107
  |---|---|
107
- | `--langfuse` | Log scores and traces to Langfuse after each item. Requires `--langfuse-dataset`. Creates one named experiment run per model (`gd-eval-{timestamp}-{model}`). Requires `LANGFUSE_PUBLIC_KEY`, `LANGFUSE_SECRET_KEY`, `LANGFUSE_HOST`. |
108
+ | `--langfuse` | Log scores and traces to Langfuse after each item. Requires `--langfuse-dataset`. Creates one named experiment run per model (`gd-eval-{timestamp}-{model}`, suffixed `-effort-{level}` when `--reasoning-effort` is set so runs differing only by effort stay separate). Requires `LANGFUSE_PUBLIC_KEY`, `LANGFUSE_SECRET_KEY`, `LANGFUSE_HOST`. |
108
109
 
109
110
  ### JSON report shape
110
111
 
@@ -1,7 +1,7 @@
1
1
  # (C) 2026 GoodData Corporation
2
2
  [project]
3
3
  name = "gooddata-eval"
4
- version = "1.71.1.dev1"
4
+ version = "1.71.1.dev2"
5
5
  description = "Evaluate the GoodData AI agent against your own questions and models."
6
6
  readme = "README.md"
7
7
  license = "MIT"
@@ -11,7 +11,7 @@ authors = [
11
11
  keywords = ["gooddata", "ai", "evaluation", "llm", "analytics", "cli"]
12
12
  requires-python = ">=3.10"
13
13
  dependencies = [
14
- "gooddata-sdk~=1.71.1.dev1",
14
+ "gooddata-sdk~=1.71.1.dev2",
15
15
  "httpx>=0.27,<1.0",
16
16
  "orjson>=3.9.15,<4.0.0",
17
17
  "pydantic>=2.6,<3.0",
@@ -14,6 +14,7 @@ from gooddata_eval.core.agentic.guardrail import evaluate_agentic_guardrail
14
14
  from gooddata_eval.core.agentic.metric_skill import evaluate_agentic_metric_skill
15
15
  from gooddata_eval.core.agentic.search_tool import evaluate_agentic_search_tool
16
16
  from gooddata_eval.core.agentic.visualization import evaluate_agentic_visualization
17
+ from gooddata_eval.core.config import ReasoningEffort
17
18
  from gooddata_eval.core.models import CreatedVisualization, DatasetItem
18
19
  from gooddata_eval.core.runner import EvalReport, ItemReport
19
20
 
@@ -24,6 +25,7 @@ class _LfKw(TypedDict, total=False):
24
25
  dataset_name: str
25
26
  run_timestamp: str
26
27
  model_version_override: str | None
28
+ reasoning_effort: ReasoningEffort | None
27
29
 
28
30
 
29
31
  AGENTIC_TEST_KINDS = frozenset(
@@ -80,6 +82,7 @@ def _dispatch_agentic(
80
82
  langfuse: Any,
81
83
  run_ts: str,
82
84
  model_version_override: str | None,
85
+ reasoning_effort: ReasoningEffort | None = None,
83
86
  ) -> None:
84
87
  """Call the appropriate evaluate_agentic_* function for the item's test_kind."""
85
88
  kind = item.test_kind
@@ -90,6 +93,7 @@ def _dispatch_agentic(
90
93
  "dataset_name": item.dataset_name,
91
94
  "run_timestamp": run_ts,
92
95
  "model_version_override": model_version_override,
96
+ "reasoning_effort": reasoning_effort,
93
97
  }
94
98
 
95
99
  if kind in ("vis_agentic", "agentic_visualization"):
@@ -176,6 +180,7 @@ def run_agentic_items(
176
180
  *,
177
181
  k: int = 2,
178
182
  model_version: str | None = None,
183
+ reasoning_effort: ReasoningEffort | None = None,
179
184
  use_langfuse: bool = False,
180
185
  run_ts: str,
181
186
  on_item_start: Any = None,
@@ -202,7 +207,7 @@ def run_agentic_items(
202
207
  )
203
208
  t0 = time.perf_counter()
204
209
  try:
205
- _dispatch_agentic(item, host, token, workspace_id, k, langfuse, run_ts, model_version)
210
+ _dispatch_agentic(item, host, token, workspace_id, k, langfuse, run_ts, model_version, reasoning_effort)
206
211
  item_report.pass_at_k = True
207
212
  item_report.runs = k
208
213
  except AssertionError as exc:
@@ -6,6 +6,7 @@ import sys
6
6
  import threading
7
7
  from datetime import datetime, timezone
8
8
  from pathlib import Path
9
+ from typing import get_args
9
10
 
10
11
  import httpx
11
12
  from gooddata_api_client.exceptions import ApiException
@@ -14,7 +15,7 @@ from rich.table import Table
14
15
 
15
16
  from gooddata_eval.cli.agentic_runner import AGENTIC_TEST_KINDS, run_agentic_items
16
17
  from gooddata_eval.core.chat.sse_client import ChatClient
17
- from gooddata_eval.core.config import RunConfig
18
+ from gooddata_eval.core.config import ReasoningEffort, RunConfig
18
19
  from gooddata_eval.core.connection import ConnectionError_, resolve_connection
19
20
  from gooddata_eval.core.dataset.local import load_local_dataset
20
21
  from gooddata_eval.core.langfuse.sink import LangfuseSink
@@ -104,6 +105,13 @@ def _build_parser() -> argparse.ArgumentParser:
104
105
  dest="preserve_failed",
105
106
  help="Keep failed conversations on the server for post-mortem inspection.",
106
107
  )
108
+ run.add_argument(
109
+ "--reasoning-effort",
110
+ dest="reasoning_effort",
111
+ choices=list(get_args(ReasoningEffort)),
112
+ help="Reasoning effort requested per message. Requires the enableGenAiReasoningEffort "
113
+ "feature flag on the target organization; without it the server ignores the value.",
114
+ )
107
115
  run.add_argument(
108
116
  "--langfuse",
109
117
  action="store_true",
@@ -292,6 +300,10 @@ def _run(config: RunConfig) -> int:
292
300
  progress_console.print(f"Provider={provider_display}, model={resolved.model_id}{switched}")
293
301
 
294
302
  run_name = f"gd-eval-{run_ts}-{resolved.model_id}"
303
+ if config.reasoning_effort:
304
+ # Without this two runs differing only by effort share a name and are
305
+ # indistinguishable in the report, which is the comparison this exists for.
306
+ run_name = f"{run_name}-effort-{config.reasoning_effort.lower()}"
295
307
  if progress_console and config.log_to_langfuse:
296
308
  progress_console.print(f"Logging to Langfuse run '{run_name}'...")
297
309
 
@@ -307,6 +319,7 @@ def _run(config: RunConfig) -> int:
307
319
  run_name=run_name,
308
320
  model_id=resolved.model_id,
309
321
  provider_type=resolved.provider_type,
322
+ reasoning_effort=config.reasoning_effort,
310
323
  )
311
324
 
312
325
  def on_langfuse_item_done(
@@ -329,6 +342,7 @@ def _run(config: RunConfig) -> int:
329
342
  workspace_id=config.workspace_id,
330
343
  k=config.runs,
331
344
  model_version=resolved.model_id,
345
+ reasoning_effort=config.reasoning_effort,
332
346
  use_langfuse=config.log_to_langfuse,
333
347
  run_ts=run_ts,
334
348
  on_item_start=on_item_start,
@@ -342,6 +356,7 @@ def _run(config: RunConfig) -> int:
342
356
  token=config.token,
343
357
  workspace_id=config.workspace_id,
344
358
  preserve_failed=config.preserve_failed,
359
+ reasoning_effort=config.reasoning_effort,
345
360
  ),
346
361
  SummaryClient(host=config.host, token=config.token, workspace_id=config.workspace_id),
347
362
  )
@@ -433,6 +448,7 @@ def main(argv: list[str] | None = None) -> int:
433
448
  quiet=args.quiet,
434
449
  kind=args.kind,
435
450
  preserve_failed=args.preserve_failed,
451
+ reasoning_effort=args.reasoning_effort,
436
452
  )
437
453
  return _run(config)
438
454
  except (
@@ -15,6 +15,8 @@ from typing import Any
15
15
 
16
16
  import httpx
17
17
 
18
+ from gooddata_eval.core.config import ReasoningEffort, normalize_reasoning_effort
19
+
18
20
  _log = logging.getLogger(__name__)
19
21
 
20
22
  # ---------------------------------------------------------------------------
@@ -384,6 +386,7 @@ def build_run_context(
384
386
  run_timestamp: str | None,
385
387
  model_version_override: str | None,
386
388
  run_metadata_extra: dict[str, Any] | None = None,
389
+ reasoning_effort: ReasoningEffort | None = None,
387
390
  ) -> tuple[str, dict[str, Any]]:
388
391
  """Return (run_name_base, run_metadata) with model version resolved from workspace API.
389
392
 
@@ -399,15 +402,24 @@ def build_run_context(
399
402
  (e.g. a testing-framework tag or a CI run id for scoping). Default None keeps
400
403
  behavior unchanged. The SDK-derived model_version is applied last and cannot
401
404
  be overwritten by this dict.
405
+ reasoning_effort: Effort the run requested, stamped into both the run name and
406
+ the metadata so effort-varying runs stay comparable side by side.
402
407
  """
408
+ effort = normalize_reasoning_effort(reasoning_effort)
403
409
  model = get_model_version(host, token, workspace_id, model_version_override)
404
410
  ts = run_timestamp or datetime.now().strftime("%Y-%m-%d_%H-%M-%S")
405
411
  base = f"{dataset_name}_{ts}"
406
412
  if model:
407
413
  base = f"{base}_{model}"
414
+ # Part of the run name, not just metadata: two runs that differ only by effort would
415
+ # otherwise collide on the same name and be indistinguishable in the report.
416
+ if effort:
417
+ base = f"{base}_effort-{effort.lower()}"
408
418
  # Caller supplies its own run tags (e.g. testing_framework); model_version is applied
409
419
  # last so the SDK-derived value cannot be overwritten by run_metadata_extra.
410
420
  metadata: dict[str, Any] = dict(run_metadata_extra) if run_metadata_extra else {}
421
+ if effort:
422
+ metadata["reasoning_effort"] = effort
411
423
  if model:
412
424
  metadata["model_version"] = model
413
425
  return base, metadata
@@ -13,6 +13,7 @@ from gooddata_sdk import GoodDataSdk
13
13
 
14
14
  from gooddata_eval.core.agentic._catalog import CatalogMetricAlert
15
15
  from gooddata_eval.core.chat.sse_client import ChatClient
16
+ from gooddata_eval.core.config import ReasoningEffort
16
17
  from gooddata_eval.core.models import ToolCallEvent
17
18
 
18
19
  try:
@@ -342,11 +343,12 @@ def run_agentic_alert_skill(
342
343
  k: int = _DEFAULT_K,
343
344
  max_iterations: int = _DEFAULT_MAX_ITERATIONS,
344
345
  initial_conversation_id: str | None = None,
346
+ reasoning_effort: ReasoningEffort | None = None,
345
347
  ) -> AgenticAlertSummary:
346
348
  """Run the alert-skill agentic evaluation K times and return a summary."""
347
349
  expected = _normalize_expected_output(expected_output)
348
350
  run_results: list[AlertRunResult] = []
349
- client = ChatClient(host=host, token=token, workspace_id=workspace_id)
351
+ client = ChatClient(host=host, token=token, workspace_id=workspace_id, reasoning_effort=reasoning_effort)
350
352
  sdk = GoodDataSdk.create(host, token)
351
353
 
352
354
  def _run_once(conv_id: str) -> AlertRunResult:
@@ -462,6 +464,7 @@ def evaluate_agentic_alert_skill(
462
464
  run_timestamp: str | None = None,
463
465
  model_version_override: str | None = None,
464
466
  run_metadata_extra: dict | None = None,
467
+ reasoning_effort: ReasoningEffort | None = None,
465
468
  ) -> None:
466
469
  """Run alert-skill evaluation, log to Langfuse, and raise AlertSkillAssertionError on failure."""
467
470
  from datetime import datetime as _dt # noqa: PLC0415
@@ -481,6 +484,7 @@ def evaluate_agentic_alert_skill(
481
484
  k=k,
482
485
  max_iterations=max_iterations,
483
486
  initial_conversation_id=initial_conversation_id,
487
+ reasoning_effort=reasoning_effort,
484
488
  )
485
489
 
486
490
  if langfuse is not None and dataset_item_id:
@@ -500,6 +504,7 @@ def evaluate_agentic_alert_skill(
500
504
  run_timestamp,
501
505
  model_version_override,
502
506
  run_metadata_extra,
507
+ reasoning_effort,
503
508
  )
504
509
  traces_by_conv = find_traces_per_conversation(
505
510
  langfuse,
@@ -14,6 +14,7 @@ from pydantic import BaseModel
14
14
  from gooddata_eval.core.agentic.alert_skill import render_alert_proposal
15
15
  from gooddata_eval.core.agentic.metric_skill import _delete_metric, _extract_created_metric_ids
16
16
  from gooddata_eval.core.chat.sse_client import ChatClient
17
+ from gooddata_eval.core.config import ReasoningEffort
17
18
  from gooddata_eval.core.models import ChatResult, ToolCallEvent
18
19
  from gooddata_eval.core.scoring import (
19
20
  check_filters,
@@ -278,6 +279,7 @@ def run_agentic_conversation(
278
279
  fixture: ConversationFixture,
279
280
  max_clarification_turns: int = 20,
280
281
  initial_conversation_id: str | None = None,
282
+ reasoning_effort: ReasoningEffort | None = None,
281
283
  ) -> ConversationResult:
282
284
  """Run a multi-turn, multi-skill conversation evaluation (no K-runs).
283
285
 
@@ -285,7 +287,7 @@ def run_agentic_conversation(
285
287
  trigger up to *max_clarification_turns* additional rounds of simulated-user
286
288
  replies before the agent produces the expected output.
287
289
  """
288
- client = ChatClient(host=host, token=token, workspace_id=workspace_id)
290
+ client = ChatClient(host=host, token=token, workspace_id=workspace_id, reasoning_effort=reasoning_effort)
289
291
  sdk = GoodDataSdk.create(host, token)
290
292
  turn_results: list[TurnResult] = []
291
293
  turn_outputs: dict[str, dict] = {}
@@ -403,6 +405,7 @@ def evaluate_agentic_conversation(
403
405
  run_timestamp: str | None = None,
404
406
  model_version_override: str | None = None,
405
407
  run_metadata_extra: dict | None = None,
408
+ reasoning_effort: ReasoningEffort | None = None,
406
409
  ) -> None:
407
410
  """Run conversation evaluation, log to Langfuse, and raise on failure."""
408
411
  from datetime import datetime as _dt # noqa: PLC0415
@@ -420,6 +423,7 @@ def evaluate_agentic_conversation(
420
423
  fixture=fixture,
421
424
  max_clarification_turns=max_clarification_turns,
422
425
  initial_conversation_id=initial_conversation_id,
426
+ reasoning_effort=reasoning_effort,
423
427
  )
424
428
 
425
429
  if langfuse is not None and dataset_item_id:
@@ -439,6 +443,7 @@ def evaluate_agentic_conversation(
439
443
  run_timestamp,
440
444
  model_version_override,
441
445
  run_metadata_extra,
446
+ reasoning_effort,
442
447
  )
443
448
  traces_by_conv = find_traces_per_conversation(
444
449
  langfuse,
@@ -6,6 +6,7 @@ from __future__ import annotations
6
6
  from dataclasses import dataclass
7
7
 
8
8
  from gooddata_eval.core.chat.sse_client import ChatClient
9
+ from gooddata_eval.core.config import ReasoningEffort
9
10
  from gooddata_eval.core.evaluators._llm_judge import LLMJudge
10
11
 
11
12
  _DEFAULT_K = 1
@@ -71,10 +72,11 @@ def run_agentic_general_question(
71
72
  expected_output: str,
72
73
  k: int = _DEFAULT_K,
73
74
  initial_conversation_id: str | None = None,
75
+ reasoning_effort: ReasoningEffort | None = None,
74
76
  ) -> AgenticGeneralQuestionSummary:
75
77
  """Run the general-question agentic evaluation K times and return a summary."""
76
78
  run_results: list[GeneralQuestionResult] = []
77
- client = ChatClient(host=host, token=token, workspace_id=workspace_id)
79
+ client = ChatClient(host=host, token=token, workspace_id=workspace_id, reasoning_effort=reasoning_effort)
78
80
  judge = LLMJudge(_GENERAL_QUESTION_EVALUATION_STEPS, model="gpt-4o")
79
81
 
80
82
  try:
@@ -153,6 +155,7 @@ def evaluate_agentic_general_question(
153
155
  run_timestamp: str | None = None,
154
156
  model_version_override: str | None = None,
155
157
  run_metadata_extra: dict | None = None,
158
+ reasoning_effort: ReasoningEffort | None = None,
156
159
  ) -> None:
157
160
  """Run general-question evaluation, log to Langfuse, and raise on failure."""
158
161
  from datetime import datetime as _dt # noqa: PLC0415
@@ -171,6 +174,7 @@ def evaluate_agentic_general_question(
171
174
  expected_output=expected_output,
172
175
  k=k,
173
176
  initial_conversation_id=initial_conversation_id,
177
+ reasoning_effort=reasoning_effort,
174
178
  )
175
179
 
176
180
  if langfuse is not None and dataset_item_id:
@@ -190,6 +194,7 @@ def evaluate_agentic_general_question(
190
194
  run_timestamp,
191
195
  model_version_override,
192
196
  run_metadata_extra,
197
+ reasoning_effort,
193
198
  )
194
199
  traces_by_conv = find_traces_per_conversation(
195
200
  langfuse,
@@ -6,6 +6,7 @@ from __future__ import annotations
6
6
  from dataclasses import dataclass
7
7
 
8
8
  from gooddata_eval.core.chat.sse_client import ChatClient
9
+ from gooddata_eval.core.config import ReasoningEffort
9
10
  from gooddata_eval.core.evaluators._llm_judge import LLMJudge
10
11
 
11
12
  _DEFAULT_K = 1
@@ -68,10 +69,11 @@ def run_agentic_guardrail(
68
69
  expected_output: str,
69
70
  k: int = _DEFAULT_K,
70
71
  initial_conversation_id: str | None = None,
72
+ reasoning_effort: ReasoningEffort | None = None,
71
73
  ) -> AgenticGuardrailSummary:
72
74
  """Run the guardrail agentic evaluation K times and return a summary."""
73
75
  run_results: list[GuardrailResult] = []
74
- client = ChatClient(host=host, token=token, workspace_id=workspace_id)
76
+ client = ChatClient(host=host, token=token, workspace_id=workspace_id, reasoning_effort=reasoning_effort)
75
77
  judge = LLMJudge(_GUARDRAIL_EVALUATION_STEPS, model="gpt-4o")
76
78
 
77
79
  try:
@@ -150,6 +152,7 @@ def evaluate_agentic_guardrail(
150
152
  run_timestamp: str | None = None,
151
153
  model_version_override: str | None = None,
152
154
  run_metadata_extra: dict | None = None,
155
+ reasoning_effort: ReasoningEffort | None = None,
153
156
  ) -> None:
154
157
  """Run guardrail evaluation, log to Langfuse, and raise on failure."""
155
158
  from datetime import datetime as _dt # noqa: PLC0415
@@ -168,6 +171,7 @@ def evaluate_agentic_guardrail(
168
171
  expected_output=expected_output,
169
172
  k=k,
170
173
  initial_conversation_id=initial_conversation_id,
174
+ reasoning_effort=reasoning_effort,
171
175
  )
172
176
 
173
177
  if langfuse is not None and dataset_item_id:
@@ -187,6 +191,7 @@ def evaluate_agentic_guardrail(
187
191
  run_timestamp,
188
192
  model_version_override,
189
193
  run_metadata_extra,
194
+ reasoning_effort,
190
195
  )
191
196
  traces_by_conv = find_traces_per_conversation(
192
197
  langfuse,
@@ -11,6 +11,7 @@ from typing import Any
11
11
  from gooddata_sdk import GoodDataSdk
12
12
 
13
13
  from gooddata_eval.core.chat.sse_client import ChatClient
14
+ from gooddata_eval.core.config import ReasoningEffort
14
15
  from gooddata_eval.core.models import ToolCallEvent
15
16
 
16
17
  try:
@@ -232,6 +233,7 @@ def run_agentic_metric_skill(
232
233
  k: int = _DEFAULT_K,
233
234
  max_iterations: int = _DEFAULT_MAX_ITERATIONS,
234
235
  initial_conversation_id: str | None = None,
236
+ reasoning_effort: ReasoningEffort | None = None,
235
237
  ) -> AgenticMetricSummary:
236
238
  """Run the metric-skill agentic evaluation K times and return a summary.
237
239
 
@@ -240,7 +242,7 @@ def run_agentic_metric_skill(
240
242
  """
241
243
  expected_outputs: list[dict] = expected_output if isinstance(expected_output, list) else [expected_output]
242
244
  run_results: list[MetricRunResult] = []
243
- client = ChatClient(host=host, token=token, workspace_id=workspace_id)
245
+ client = ChatClient(host=host, token=token, workspace_id=workspace_id, reasoning_effort=reasoning_effort)
244
246
  sdk = GoodDataSdk.create(host, token)
245
247
 
246
248
  try:
@@ -300,6 +302,7 @@ def evaluate_agentic_metric_skill(
300
302
  run_timestamp: str | None = None,
301
303
  model_version_override: str | None = None,
302
304
  run_metadata_extra: dict | None = None,
305
+ reasoning_effort: ReasoningEffort | None = None,
303
306
  ) -> None:
304
307
  """Run metric-skill evaluation, log to Langfuse, and raise MetricSkillAssertionError on failure."""
305
308
  from datetime import datetime as _dt # noqa: PLC0415
@@ -319,6 +322,7 @@ def evaluate_agentic_metric_skill(
319
322
  k=k,
320
323
  max_iterations=max_iterations,
321
324
  initial_conversation_id=initial_conversation_id,
325
+ reasoning_effort=reasoning_effort,
322
326
  )
323
327
 
324
328
  if langfuse is not None and dataset_item_id:
@@ -338,6 +342,7 @@ def evaluate_agentic_metric_skill(
338
342
  run_timestamp,
339
343
  model_version_override,
340
344
  run_metadata_extra,
345
+ reasoning_effort,
341
346
  )
342
347
  traces_by_conv = find_traces_per_conversation(
343
348
  langfuse,
@@ -6,6 +6,7 @@ from __future__ import annotations
6
6
  from dataclasses import dataclass
7
7
 
8
8
  from gooddata_eval.core.chat.sse_client import ChatClient
9
+ from gooddata_eval.core.config import ReasoningEffort
9
10
  from gooddata_eval.core.models import ToolCallEvent
10
11
 
11
12
  _DEFAULT_K = 1
@@ -67,11 +68,12 @@ def run_agentic_search_tool(
67
68
  expected_tool_call: dict,
68
69
  k: int = _DEFAULT_K,
69
70
  initial_conversation_id: str | None = None,
71
+ reasoning_effort: ReasoningEffort | None = None,
70
72
  ) -> AgenticSearchSummary:
71
73
  """Run the search-tool agentic evaluation K times (single-turn each)."""
72
74
  run_results: list[SearchResult] = []
73
75
 
74
- client = ChatClient(host=host, token=token, workspace_id=workspace_id)
76
+ client = ChatClient(host=host, token=token, workspace_id=workspace_id, reasoning_effort=reasoning_effort)
75
77
  try:
76
78
  conv_id_0 = initial_conversation_id if initial_conversation_id is not None else client.create_conversation()
77
79
  try:
@@ -144,6 +146,7 @@ def evaluate_agentic_search_tool(
144
146
  run_timestamp: str | None = None,
145
147
  model_version_override: str | None = None,
146
148
  run_metadata_extra: dict | None = None,
149
+ reasoning_effort: ReasoningEffort | None = None,
147
150
  ) -> None:
148
151
  """Run search-tool evaluation, log to Langfuse, and raise SearchToolAssertionError on failure."""
149
152
  from datetime import datetime as _dt # noqa: PLC0415
@@ -162,6 +165,7 @@ def evaluate_agentic_search_tool(
162
165
  expected_tool_call=expected_tool_call,
163
166
  k=k,
164
167
  initial_conversation_id=initial_conversation_id,
168
+ reasoning_effort=reasoning_effort,
165
169
  )
166
170
 
167
171
  if langfuse is not None and dataset_item_id:
@@ -181,6 +185,7 @@ def evaluate_agentic_search_tool(
181
185
  run_timestamp,
182
186
  model_version_override,
183
187
  run_metadata_extra,
188
+ reasoning_effort,
184
189
  )
185
190
  traces_by_conv = find_traces_per_conversation(
186
191
  langfuse,
@@ -11,6 +11,7 @@ import os
11
11
  from dataclasses import dataclass
12
12
 
13
13
  from gooddata_eval.core.chat.sse_client import ChatClient
14
+ from gooddata_eval.core.config import ReasoningEffort
14
15
  from gooddata_eval.core.evaluators.visualization import (
15
16
  EvaluationResult,
16
17
  _check_visualization_skill_activated,
@@ -203,6 +204,7 @@ def run_agentic_visualization(
203
204
  k: int = _DEFAULT_K,
204
205
  max_iterations: int = _DEFAULT_MAX_ITERATIONS,
205
206
  initial_conversation_id: str | None = None,
207
+ reasoning_effort: ReasoningEffort | None = None,
206
208
  ) -> AgenticRunSummary:
207
209
  """Run K independent conversations and return evaluation results.
208
210
 
@@ -211,7 +213,7 @@ def run_agentic_visualization(
211
213
  fresh conversations. Caller-supplied conversations are not deleted; all
212
214
  conversations created by this function are deleted on completion.
213
215
  """
214
- client = ChatClient(host=host, token=token, workspace_id=workspace_id)
216
+ client = ChatClient(host=host, token=token, workspace_id=workspace_id, reasoning_effort=reasoning_effort)
215
217
  run_results: list[RunResult] = []
216
218
 
217
219
  try:
@@ -265,6 +267,7 @@ def evaluate_agentic_visualization(
265
267
  model_version_override: str | None = None,
266
268
  run_metadata_extra: dict | None = None,
267
269
  record_output_path: str | None = None,
270
+ reasoning_effort: ReasoningEffort | None = None,
268
271
  ) -> None:
269
272
  """Run visualization evaluation, log to Langfuse, and raise VisualizationAssertionError on failure."""
270
273
  import json as _json # noqa: PLC0415
@@ -285,6 +288,7 @@ def evaluate_agentic_visualization(
285
288
  k=k,
286
289
  max_iterations=max_iterations,
287
290
  initial_conversation_id=initial_conversation_id,
291
+ reasoning_effort=reasoning_effort,
288
292
  )
289
293
 
290
294
  if langfuse is not None and dataset_item_id:
@@ -304,6 +308,7 @@ def evaluate_agentic_visualization(
304
308
  run_timestamp,
305
309
  model_version_override,
306
310
  run_metadata_extra,
311
+ reasoning_effort,
307
312
  )
308
313
  K = len(summary.run_results)
309
314
  traces_by_conv = find_traces_per_conversation(
@@ -22,6 +22,7 @@ from typing import Any, Callable, Iterable, TypeVar
22
22
 
23
23
  import httpx
24
24
 
25
+ from gooddata_eval.core.config import ReasoningEffort, normalize_reasoning_effort
25
26
  from gooddata_eval.core.models import ChatResult, DatasetItem
26
27
 
27
28
  _log = logging.getLogger(__name__)
@@ -242,12 +243,28 @@ class ChatClient:
242
243
  """Single-turn AI chat client over the GoodData AI conversation endpoints."""
243
244
 
244
245
  def __init__(
245
- self, host: str, token: str, workspace_id: str, *, timeout: float = 300.0, preserve_failed: bool = False
246
+ self,
247
+ host: str,
248
+ token: str,
249
+ workspace_id: str,
250
+ *,
251
+ timeout: float = 300.0,
252
+ preserve_failed: bool = False,
253
+ reasoning_effort: ReasoningEffort | None = None,
246
254
  ):
255
+ """Create a chat client bound to one workspace.
256
+
257
+ ``reasoning_effort`` (``LOW``/``MEDIUM``/``HIGH``) is sent as
258
+ ``options.reasoningEffort`` on every message; when None the key is omitted
259
+ entirely and the server keeps its own default. The server honours it only
260
+ while the ``enableGenAiReasoningEffort`` feature flag is on for the
261
+ organization, so setting it is a request rather than a guarantee.
262
+ """
247
263
  self._base = f"{host.rstrip('/')}/api/v1/ai/workspaces/{workspace_id}/chat/conversations"
248
264
  self._auth = {"Authorization": f"Bearer {token}"}
249
265
  self._client = httpx.Client(timeout=timeout)
250
266
  self._preserve_failed = preserve_failed
267
+ self._reasoning_effort = normalize_reasoning_effort(reasoning_effort)
251
268
 
252
269
  def create_conversation(self) -> str:
253
270
  def _do() -> str:
@@ -271,7 +288,9 @@ class ChatClient:
271
288
  def send_message(self, conversation_id: str, question: str) -> ChatResult:
272
289
  url = f"{self._base}/{conversation_id}/messages"
273
290
  headers = {**self._auth, "Accept": "text/event-stream", "Content-Type": "application/json"}
274
- body = {"item": {"role": "user", "content": {"type": "text", "text": question}}}
291
+ body: dict[str, Any] = {"item": {"role": "user", "content": {"type": "text", "text": question}}}
292
+ if self._reasoning_effort is not None:
293
+ body["options"] = {"reasoningEffort": self._reasoning_effort}
275
294
 
276
295
  def _do() -> ChatResult:
277
296
  with self._client.stream("POST", url, json=body, headers=headers) as resp:
@@ -0,0 +1,46 @@
1
+ # (C) 2026 GoodData Corporation
2
+ """Validated run configuration produced by the CLI and consumed by the runner."""
3
+
4
+ from dataclasses import dataclass, field
5
+ from pathlib import Path
6
+ from typing import Literal, cast, get_args
7
+
8
+ ReasoningEffort = Literal["LOW", "MEDIUM", "HIGH"]
9
+ """Effort values the AI chat endpoint accepts, uppercase as the server enum requires."""
10
+
11
+
12
+ def normalize_reasoning_effort(value: str | None) -> ReasoningEffort | None:
13
+ """Canonical effort, or None when unset.
14
+
15
+ The `Literal` above only constrains static callers, so normalize once at the
16
+ boundary: without it a lowercase value reaches the endpoint and is rejected as
17
+ an out-of-enum request, while an empty string is sent yet skipped by the
18
+ truthiness checks in the Langfuse writers — leaving a run whose recorded
19
+ identity disagrees with what it actually requested.
20
+ """
21
+ if value is None:
22
+ return None
23
+ candidate = value.strip().upper()
24
+ if not candidate:
25
+ return None
26
+ if candidate not in get_args(ReasoningEffort):
27
+ raise ValueError(f"Invalid reasoning effort {value!r}; expected one of {', '.join(get_args(ReasoningEffort))}.")
28
+ return cast("ReasoningEffort", candidate)
29
+
30
+
31
+ @dataclass
32
+ class RunConfig:
33
+ host: str
34
+ token: str
35
+ workspace_id: str
36
+ dataset_folder: Path | None = None
37
+ langfuse_dataset: str | None = None
38
+ models: list[str] = field(default_factory=list)
39
+ runs: int = 2
40
+ concurrency: int = 1
41
+ json_path: Path | None = None
42
+ log_to_langfuse: bool = False
43
+ quiet: bool = False
44
+ kind: str = "visualization"
45
+ preserve_failed: bool = False
46
+ reasoning_effort: ReasoningEffort | None = None