gooddata-eval 1.71.0__tar.gz → 1.71.1.dev2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (88) hide show
  1. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/PKG-INFO +4 -3
  2. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/README.md +2 -1
  3. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/pyproject.toml +2 -2
  4. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/cli/agentic_runner.py +6 -1
  5. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/cli/main.py +17 -1
  6. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/agentic/_langfuse.py +12 -0
  7. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/agentic/alert_skill.py +28 -6
  8. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/agentic/conversation.py +11 -2
  9. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/agentic/general_question.py +6 -1
  10. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/agentic/guardrail.py +6 -1
  11. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/agentic/metric_skill.py +6 -1
  12. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/agentic/search_tool.py +6 -1
  13. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/agentic/visualization.py +6 -1
  14. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/chat/sse_client.py +28 -2
  15. gooddata_eval-1.71.1.dev2/src/gooddata_eval/core/config.py +46 -0
  16. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/langfuse/sink.py +21 -2
  17. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/models.py +4 -0
  18. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/tests/test_agentic_alert_skill.py +92 -0
  19. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/tests/test_agentic_conversation.py +62 -1
  20. gooddata_eval-1.71.1.dev2/tests/test_agentic_run_context.py +73 -0
  21. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/tests/test_cli.py +77 -3
  22. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/tests/test_sse_client.py +115 -2
  23. gooddata_eval-1.71.0/src/gooddata_eval/core/config.py +0 -22
  24. gooddata_eval-1.71.0/tests/test_agentic_run_context.py +0 -32
  25. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/.gitignore +0 -0
  26. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/LICENSE.txt +0 -0
  27. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/Makefile +0 -0
  28. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/__init__.py +0 -0
  29. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/_version.py +0 -0
  30. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/cli/__init__.py +0 -0
  31. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/__init__.py +0 -0
  32. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/agentic/__init__.py +0 -0
  33. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/agentic/_catalog.py +0 -0
  34. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/chat/__init__.py +0 -0
  35. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/connection.py +0 -0
  36. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/dataset/__init__.py +0 -0
  37. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/dataset/langfuse_source.py +0 -0
  38. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/dataset/local.py +0 -0
  39. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/evaluators/__init__.py +0 -0
  40. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/evaluators/_deep_subset.py +0 -0
  41. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/evaluators/_llm_judge.py +0 -0
  42. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/evaluators/_text_utils.py +0 -0
  43. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/evaluators/alert_skill.py +0 -0
  44. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/evaluators/base.py +0 -0
  45. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/evaluators/general_question.py +0 -0
  46. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/evaluators/guardrail.py +0 -0
  47. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/evaluators/metric_skill.py +0 -0
  48. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/evaluators/search_tool.py +0 -0
  49. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/evaluators/summary.py +0 -0
  50. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/evaluators/visualization.py +0 -0
  51. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/langfuse/__init__.py +0 -0
  52. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/reporting/__init__.py +0 -0
  53. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/reporting/console.py +0 -0
  54. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/reporting/json_report.py +0 -0
  55. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/runner.py +0 -0
  56. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/scoring.py +0 -0
  57. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/summary/__init__.py +0 -0
  58. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/summary/http_client.py +0 -0
  59. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/src/gooddata_eval/core/workspace.py +0 -0
  60. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/tests/__init__.py +0 -0
  61. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/tests/conftest.py +0 -0
  62. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/tests/fixtures/sample_dataset/metric_skill_create.json +0 -0
  63. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/tests/fixtures/sample_dataset/visualization_revenue.json +0 -0
  64. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/tests/fixtures/sse_visualization_stream.txt +0 -0
  65. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/tests/test_agentic_general_question.py +0 -0
  66. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/tests/test_agentic_guardrail.py +0 -0
  67. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/tests/test_agentic_metric_skill.py +0 -0
  68. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/tests/test_agentic_search_tool.py +0 -0
  69. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/tests/test_agentic_visualization.py +0 -0
  70. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/tests/test_alert_skill_evaluator.py +0 -0
  71. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/tests/test_connection.py +0 -0
  72. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/tests/test_deep_subset.py +0 -0
  73. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/tests/test_langfuse_sink.py +0 -0
  74. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/tests/test_langfuse_source.py +0 -0
  75. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/tests/test_llm_judge.py +0 -0
  76. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/tests/test_local_loader.py +0 -0
  77. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/tests/test_metric_skill_evaluator.py +0 -0
  78. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/tests/test_models.py +0 -0
  79. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/tests/test_reporting.py +0 -0
  80. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/tests/test_runner.py +0 -0
  81. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/tests/test_scoring.py +0 -0
  82. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/tests/test_search_tool_evaluator.py +0 -0
  83. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/tests/test_summary_client.py +0 -0
  84. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/tests/test_summary_evaluator.py +0 -0
  85. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/tests/test_text_evaluators.py +0 -0
  86. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/tests/test_visualization_evaluator.py +0 -0
  87. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/tests/test_workspace.py +0 -0
  88. {gooddata_eval-1.71.0 → gooddata_eval-1.71.1.dev2}/tox.ini +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: gooddata-eval
3
- Version: 1.71.0
3
+ Version: 1.71.1.dev2
4
4
  Summary: Evaluate the GoodData AI agent against your own questions and models.
5
5
  Project-URL: Source, https://github.com/gooddata/gooddata-python-sdk
6
6
  Author-email: GoodData <support@gooddata.com>
@@ -17,7 +17,7 @@ Classifier: Topic :: Scientific/Engineering
17
17
  Classifier: Topic :: Software Development
18
18
  Classifier: Typing :: Typed
19
19
  Requires-Python: >=3.10
20
- Requires-Dist: gooddata-sdk~=1.71.0
20
+ Requires-Dist: gooddata-sdk~=1.71.1.dev2
21
21
  Requires-Dist: httpx<1.0,>=0.27
22
22
  Requires-Dist: orjson<4.0.0,>=3.9.15
23
23
  Requires-Dist: pydantic<3.0,>=2.6
@@ -120,6 +120,7 @@ Both provider name and provider id are accepted as the prefix.
120
120
  |---|---|---|
121
121
  | `--runs K` | `2` | Independent runs per item (pass@K). An item passes if any run passes. |
122
122
  | `--concurrency K` | `1` | Number of items evaluated concurrently. `1` = sequential (default). Increase to load-test the agent under simultaneous requests. Progress output interleaves when K > 1. |
123
+ | `--reasoning-effort LEVEL` | server default | `LOW`, `MEDIUM` or `HIGH`, sent as `options.reasoningEffort` on every chat message. Requires the `enableGenAiReasoningEffort` feature flag on the target organization — without it the server ignores the value. Applies to chat items only; `dashboard_summary` items go through the summary endpoint, which has no such option. |
123
124
 
124
125
  #### Output
125
126
 
@@ -132,7 +133,7 @@ Both provider name and provider id are accepted as the prefix.
132
133
 
133
134
  | Flag | Description |
134
135
  |---|---|
135
- | `--langfuse` | Log scores and traces to Langfuse after each item. Requires `--langfuse-dataset`. Creates one named experiment run per model (`gd-eval-{timestamp}-{model}`). Requires `LANGFUSE_PUBLIC_KEY`, `LANGFUSE_SECRET_KEY`, `LANGFUSE_HOST`. |
136
+ | `--langfuse` | Log scores and traces to Langfuse after each item. Requires `--langfuse-dataset`. Creates one named experiment run per model (`gd-eval-{timestamp}-{model}`, suffixed `-effort-{level}` when `--reasoning-effort` is set so runs differing only by effort stay separate). Requires `LANGFUSE_PUBLIC_KEY`, `LANGFUSE_SECRET_KEY`, `LANGFUSE_HOST`. |
136
137
 
137
138
  ### JSON report shape
138
139
 
@@ -92,6 +92,7 @@ Both provider name and provider id are accepted as the prefix.
92
92
  |---|---|---|
93
93
  | `--runs K` | `2` | Independent runs per item (pass@K). An item passes if any run passes. |
94
94
  | `--concurrency K` | `1` | Number of items evaluated concurrently. `1` = sequential (default). Increase to load-test the agent under simultaneous requests. Progress output interleaves when K > 1. |
95
+ | `--reasoning-effort LEVEL` | server default | `LOW`, `MEDIUM` or `HIGH`, sent as `options.reasoningEffort` on every chat message. Requires the `enableGenAiReasoningEffort` feature flag on the target organization — without it the server ignores the value. Applies to chat items only; `dashboard_summary` items go through the summary endpoint, which has no such option. |
95
96
 
96
97
  #### Output
97
98
 
@@ -104,7 +105,7 @@ Both provider name and provider id are accepted as the prefix.
104
105
 
105
106
  | Flag | Description |
106
107
  |---|---|
107
- | `--langfuse` | Log scores and traces to Langfuse after each item. Requires `--langfuse-dataset`. Creates one named experiment run per model (`gd-eval-{timestamp}-{model}`). Requires `LANGFUSE_PUBLIC_KEY`, `LANGFUSE_SECRET_KEY`, `LANGFUSE_HOST`. |
108
+ | `--langfuse` | Log scores and traces to Langfuse after each item. Requires `--langfuse-dataset`. Creates one named experiment run per model (`gd-eval-{timestamp}-{model}`, suffixed `-effort-{level}` when `--reasoning-effort` is set so runs differing only by effort stay separate). Requires `LANGFUSE_PUBLIC_KEY`, `LANGFUSE_SECRET_KEY`, `LANGFUSE_HOST`. |
108
109
 
109
110
  ### JSON report shape
110
111
 
@@ -1,7 +1,7 @@
1
1
  # (C) 2026 GoodData Corporation
2
2
  [project]
3
3
  name = "gooddata-eval"
4
- version = "1.71.0"
4
+ version = "1.71.1.dev2"
5
5
  description = "Evaluate the GoodData AI agent against your own questions and models."
6
6
  readme = "README.md"
7
7
  license = "MIT"
@@ -11,7 +11,7 @@ authors = [
11
11
  keywords = ["gooddata", "ai", "evaluation", "llm", "analytics", "cli"]
12
12
  requires-python = ">=3.10"
13
13
  dependencies = [
14
- "gooddata-sdk~=1.71.0",
14
+ "gooddata-sdk~=1.71.1.dev2",
15
15
  "httpx>=0.27,<1.0",
16
16
  "orjson>=3.9.15,<4.0.0",
17
17
  "pydantic>=2.6,<3.0",
@@ -14,6 +14,7 @@ from gooddata_eval.core.agentic.guardrail import evaluate_agentic_guardrail
14
14
  from gooddata_eval.core.agentic.metric_skill import evaluate_agentic_metric_skill
15
15
  from gooddata_eval.core.agentic.search_tool import evaluate_agentic_search_tool
16
16
  from gooddata_eval.core.agentic.visualization import evaluate_agentic_visualization
17
+ from gooddata_eval.core.config import ReasoningEffort
17
18
  from gooddata_eval.core.models import CreatedVisualization, DatasetItem
18
19
  from gooddata_eval.core.runner import EvalReport, ItemReport
19
20
 
@@ -24,6 +25,7 @@ class _LfKw(TypedDict, total=False):
24
25
  dataset_name: str
25
26
  run_timestamp: str
26
27
  model_version_override: str | None
28
+ reasoning_effort: ReasoningEffort | None
27
29
 
28
30
 
29
31
  AGENTIC_TEST_KINDS = frozenset(
@@ -80,6 +82,7 @@ def _dispatch_agentic(
80
82
  langfuse: Any,
81
83
  run_ts: str,
82
84
  model_version_override: str | None,
85
+ reasoning_effort: ReasoningEffort | None = None,
83
86
  ) -> None:
84
87
  """Call the appropriate evaluate_agentic_* function for the item's test_kind."""
85
88
  kind = item.test_kind
@@ -90,6 +93,7 @@ def _dispatch_agentic(
90
93
  "dataset_name": item.dataset_name,
91
94
  "run_timestamp": run_ts,
92
95
  "model_version_override": model_version_override,
96
+ "reasoning_effort": reasoning_effort,
93
97
  }
94
98
 
95
99
  if kind in ("vis_agentic", "agentic_visualization"):
@@ -176,6 +180,7 @@ def run_agentic_items(
176
180
  *,
177
181
  k: int = 2,
178
182
  model_version: str | None = None,
183
+ reasoning_effort: ReasoningEffort | None = None,
179
184
  use_langfuse: bool = False,
180
185
  run_ts: str,
181
186
  on_item_start: Any = None,
@@ -202,7 +207,7 @@ def run_agentic_items(
202
207
  )
203
208
  t0 = time.perf_counter()
204
209
  try:
205
- _dispatch_agentic(item, host, token, workspace_id, k, langfuse, run_ts, model_version)
210
+ _dispatch_agentic(item, host, token, workspace_id, k, langfuse, run_ts, model_version, reasoning_effort)
206
211
  item_report.pass_at_k = True
207
212
  item_report.runs = k
208
213
  except AssertionError as exc:
@@ -6,6 +6,7 @@ import sys
6
6
  import threading
7
7
  from datetime import datetime, timezone
8
8
  from pathlib import Path
9
+ from typing import get_args
9
10
 
10
11
  import httpx
11
12
  from gooddata_api_client.exceptions import ApiException
@@ -14,7 +15,7 @@ from rich.table import Table
14
15
 
15
16
  from gooddata_eval.cli.agentic_runner import AGENTIC_TEST_KINDS, run_agentic_items
16
17
  from gooddata_eval.core.chat.sse_client import ChatClient
17
- from gooddata_eval.core.config import RunConfig
18
+ from gooddata_eval.core.config import ReasoningEffort, RunConfig
18
19
  from gooddata_eval.core.connection import ConnectionError_, resolve_connection
19
20
  from gooddata_eval.core.dataset.local import load_local_dataset
20
21
  from gooddata_eval.core.langfuse.sink import LangfuseSink
@@ -104,6 +105,13 @@ def _build_parser() -> argparse.ArgumentParser:
104
105
  dest="preserve_failed",
105
106
  help="Keep failed conversations on the server for post-mortem inspection.",
106
107
  )
108
+ run.add_argument(
109
+ "--reasoning-effort",
110
+ dest="reasoning_effort",
111
+ choices=list(get_args(ReasoningEffort)),
112
+ help="Reasoning effort requested per message. Requires the enableGenAiReasoningEffort "
113
+ "feature flag on the target organization; without it the server ignores the value.",
114
+ )
107
115
  run.add_argument(
108
116
  "--langfuse",
109
117
  action="store_true",
@@ -292,6 +300,10 @@ def _run(config: RunConfig) -> int:
292
300
  progress_console.print(f"Provider={provider_display}, model={resolved.model_id}{switched}")
293
301
 
294
302
  run_name = f"gd-eval-{run_ts}-{resolved.model_id}"
303
+ if config.reasoning_effort:
304
+ # Without this two runs differing only by effort share a name and are
305
+ # indistinguishable in the report, which is the comparison this exists for.
306
+ run_name = f"{run_name}-effort-{config.reasoning_effort.lower()}"
295
307
  if progress_console and config.log_to_langfuse:
296
308
  progress_console.print(f"Logging to Langfuse run '{run_name}'...")
297
309
 
@@ -307,6 +319,7 @@ def _run(config: RunConfig) -> int:
307
319
  run_name=run_name,
308
320
  model_id=resolved.model_id,
309
321
  provider_type=resolved.provider_type,
322
+ reasoning_effort=config.reasoning_effort,
310
323
  )
311
324
 
312
325
  def on_langfuse_item_done(
@@ -329,6 +342,7 @@ def _run(config: RunConfig) -> int:
329
342
  workspace_id=config.workspace_id,
330
343
  k=config.runs,
331
344
  model_version=resolved.model_id,
345
+ reasoning_effort=config.reasoning_effort,
332
346
  use_langfuse=config.log_to_langfuse,
333
347
  run_ts=run_ts,
334
348
  on_item_start=on_item_start,
@@ -342,6 +356,7 @@ def _run(config: RunConfig) -> int:
342
356
  token=config.token,
343
357
  workspace_id=config.workspace_id,
344
358
  preserve_failed=config.preserve_failed,
359
+ reasoning_effort=config.reasoning_effort,
345
360
  ),
346
361
  SummaryClient(host=config.host, token=config.token, workspace_id=config.workspace_id),
347
362
  )
@@ -433,6 +448,7 @@ def main(argv: list[str] | None = None) -> int:
433
448
  quiet=args.quiet,
434
449
  kind=args.kind,
435
450
  preserve_failed=args.preserve_failed,
451
+ reasoning_effort=args.reasoning_effort,
436
452
  )
437
453
  return _run(config)
438
454
  except (
@@ -15,6 +15,8 @@ from typing import Any
15
15
 
16
16
  import httpx
17
17
 
18
+ from gooddata_eval.core.config import ReasoningEffort, normalize_reasoning_effort
19
+
18
20
  _log = logging.getLogger(__name__)
19
21
 
20
22
  # ---------------------------------------------------------------------------
@@ -384,6 +386,7 @@ def build_run_context(
384
386
  run_timestamp: str | None,
385
387
  model_version_override: str | None,
386
388
  run_metadata_extra: dict[str, Any] | None = None,
389
+ reasoning_effort: ReasoningEffort | None = None,
387
390
  ) -> tuple[str, dict[str, Any]]:
388
391
  """Return (run_name_base, run_metadata) with model version resolved from workspace API.
389
392
 
@@ -399,15 +402,24 @@ def build_run_context(
399
402
  (e.g. a testing-framework tag or a CI run id for scoping). Default None keeps
400
403
  behavior unchanged. The SDK-derived model_version is applied last and cannot
401
404
  be overwritten by this dict.
405
+ reasoning_effort: Effort the run requested, stamped into both the run name and
406
+ the metadata so effort-varying runs stay comparable side by side.
402
407
  """
408
+ effort = normalize_reasoning_effort(reasoning_effort)
403
409
  model = get_model_version(host, token, workspace_id, model_version_override)
404
410
  ts = run_timestamp or datetime.now().strftime("%Y-%m-%d_%H-%M-%S")
405
411
  base = f"{dataset_name}_{ts}"
406
412
  if model:
407
413
  base = f"{base}_{model}"
414
+ # Part of the run name, not just metadata: two runs that differ only by effort would
415
+ # otherwise collide on the same name and be indistinguishable in the report.
416
+ if effort:
417
+ base = f"{base}_effort-{effort.lower()}"
408
418
  # Caller supplies its own run tags (e.g. testing_framework); model_version is applied
409
419
  # last so the SDK-derived value cannot be overwritten by run_metadata_extra.
410
420
  metadata: dict[str, Any] = dict(run_metadata_extra) if run_metadata_extra else {}
421
+ if effort:
422
+ metadata["reasoning_effort"] = effort
411
423
  if model:
412
424
  metadata["model_version"] = model
413
425
  return base, metadata
@@ -13,6 +13,7 @@ from gooddata_sdk import GoodDataSdk
13
13
 
14
14
  from gooddata_eval.core.agentic._catalog import CatalogMetricAlert
15
15
  from gooddata_eval.core.chat.sse_client import ChatClient
16
+ from gooddata_eval.core.config import ReasoningEffort
16
17
  from gooddata_eval.core.models import ToolCallEvent
17
18
 
18
19
  try:
@@ -311,11 +312,26 @@ def _extract_alert_call(tool_call_events: list[ToolCallEvent]) -> tuple[str | No
311
312
  return None, {}, False
312
313
 
313
314
 
314
- def _is_asking_clarification(text: str) -> bool:
315
- if not text:
316
- return False
317
- t = text.lower()
318
- return "?" in t or "could you" in t or "please" in t or "clarif" in t
315
+ def render_alert_proposal(proposal: dict) -> str:
316
+ """Render an alert-proposal part as the text the simulated user reacts to.
317
+
318
+ The alert skill's confirmation step deliberately emits no text part (GDAI-2032) — the
319
+ prompt and the CTA live only in the proposal payload, which the frontend renders as a
320
+ widget. Dumping the payload (rather than prose) keeps recipients, condition, trigger and
321
+ dashboard visible so the simulated user can still verify them against its goal, and does
322
+ not need updating whenever ``AlertProposal`` grows a field.
323
+ """
324
+ cta = proposal.get("cta") or "Should I create this alert?"
325
+ summary = {k: v for k, v in proposal.items() if k != "cta"}
326
+ alert = dict(summary.get("alert") or {})
327
+ # The AFM execution block is opaque wire dicts — noise that would crowd out the fields
328
+ # the simulated user actually has to check.
329
+ alert.pop("execution", None)
330
+ if "alert" in summary:
331
+ # Key off presence, not truthiness: an alert whose only key was `execution` must
332
+ # still be replaced, otherwise the original (execution-bearing) dict survives.
333
+ summary["alert"] = alert
334
+ return f"{cta}\n\nAlert proposal:\n{json.dumps(summary, indent=2, sort_keys=True)}"
319
335
 
320
336
 
321
337
  def run_agentic_alert_skill(
@@ -327,11 +343,12 @@ def run_agentic_alert_skill(
327
343
  k: int = _DEFAULT_K,
328
344
  max_iterations: int = _DEFAULT_MAX_ITERATIONS,
329
345
  initial_conversation_id: str | None = None,
346
+ reasoning_effort: ReasoningEffort | None = None,
330
347
  ) -> AgenticAlertSummary:
331
348
  """Run the alert-skill agentic evaluation K times and return a summary."""
332
349
  expected = _normalize_expected_output(expected_output)
333
350
  run_results: list[AlertRunResult] = []
334
- client = ChatClient(host=host, token=token, workspace_id=workspace_id)
351
+ client = ChatClient(host=host, token=token, workspace_id=workspace_id, reasoning_effort=reasoning_effort)
335
352
  sdk = GoodDataSdk.create(host, token)
336
353
 
337
354
  def _run_once(conv_id: str) -> AlertRunResult:
@@ -352,6 +369,8 @@ def run_agentic_alert_skill(
352
369
  alert_id_to_delete = alert_id
353
370
  break
354
371
  response_text = (chat_result.text_response or "").strip()
372
+ if not response_text and chat_result.alert_proposals:
373
+ response_text = render_alert_proposal(chat_result.alert_proposals[-1])
355
374
  # Stop if agent gave a completely empty response (stuck)
356
375
  if not response_text and not chat_result.tool_call_events:
357
376
  break
@@ -445,6 +464,7 @@ def evaluate_agentic_alert_skill(
445
464
  run_timestamp: str | None = None,
446
465
  model_version_override: str | None = None,
447
466
  run_metadata_extra: dict | None = None,
467
+ reasoning_effort: ReasoningEffort | None = None,
448
468
  ) -> None:
449
469
  """Run alert-skill evaluation, log to Langfuse, and raise AlertSkillAssertionError on failure."""
450
470
  from datetime import datetime as _dt # noqa: PLC0415
@@ -464,6 +484,7 @@ def evaluate_agentic_alert_skill(
464
484
  k=k,
465
485
  max_iterations=max_iterations,
466
486
  initial_conversation_id=initial_conversation_id,
487
+ reasoning_effort=reasoning_effort,
467
488
  )
468
489
 
469
490
  if langfuse is not None and dataset_item_id:
@@ -483,6 +504,7 @@ def evaluate_agentic_alert_skill(
483
504
  run_timestamp,
484
505
  model_version_override,
485
506
  run_metadata_extra,
507
+ reasoning_effort,
486
508
  )
487
509
  traces_by_conv = find_traces_per_conversation(
488
510
  langfuse,
@@ -11,8 +11,10 @@ from typing import Literal
11
11
  from gooddata_sdk import GoodDataSdk
12
12
  from pydantic import BaseModel
13
13
 
14
+ from gooddata_eval.core.agentic.alert_skill import render_alert_proposal
14
15
  from gooddata_eval.core.agentic.metric_skill import _delete_metric, _extract_created_metric_ids
15
16
  from gooddata_eval.core.chat.sse_client import ChatClient
17
+ from gooddata_eval.core.config import ReasoningEffort
16
18
  from gooddata_eval.core.models import ChatResult, ToolCallEvent
17
19
  from gooddata_eval.core.scoring import (
18
20
  check_filters,
@@ -277,6 +279,7 @@ def run_agentic_conversation(
277
279
  fixture: ConversationFixture,
278
280
  max_clarification_turns: int = 20,
279
281
  initial_conversation_id: str | None = None,
282
+ reasoning_effort: ReasoningEffort | None = None,
280
283
  ) -> ConversationResult:
281
284
  """Run a multi-turn, multi-skill conversation evaluation (no K-runs).
282
285
 
@@ -284,7 +287,7 @@ def run_agentic_conversation(
284
287
  trigger up to *max_clarification_turns* additional rounds of simulated-user
285
288
  replies before the agent produces the expected output.
286
289
  """
287
- client = ChatClient(host=host, token=token, workspace_id=workspace_id)
290
+ client = ChatClient(host=host, token=token, workspace_id=workspace_id, reasoning_effort=reasoning_effort)
288
291
  sdk = GoodDataSdk.create(host, token)
289
292
  turn_results: list[TurnResult] = []
290
293
  turn_outputs: dict[str, dict] = {}
@@ -322,7 +325,10 @@ def run_agentic_conversation(
322
325
  break
323
326
 
324
327
  response_text = (chat_result.text_response or "").strip()
325
- if _is_asking_clarification(response_text) and clarification_turns < max_clarification_turns:
328
+ if not response_text and chat_result.alert_proposals:
329
+ response_text = render_alert_proposal(chat_result.alert_proposals[-1])
330
+ asking = _is_asking_clarification(response_text) or bool(chat_result.alert_proposals)
331
+ if asking and clarification_turns < max_clarification_turns:
326
332
  clarification_turns += 1
327
333
  total_clarification_turns += 1
328
334
  current_message = _get_sim_user_response(response_text, resolved_turn, resolved_expected)
@@ -399,6 +405,7 @@ def evaluate_agentic_conversation(
399
405
  run_timestamp: str | None = None,
400
406
  model_version_override: str | None = None,
401
407
  run_metadata_extra: dict | None = None,
408
+ reasoning_effort: ReasoningEffort | None = None,
402
409
  ) -> None:
403
410
  """Run conversation evaluation, log to Langfuse, and raise on failure."""
404
411
  from datetime import datetime as _dt # noqa: PLC0415
@@ -416,6 +423,7 @@ def evaluate_agentic_conversation(
416
423
  fixture=fixture,
417
424
  max_clarification_turns=max_clarification_turns,
418
425
  initial_conversation_id=initial_conversation_id,
426
+ reasoning_effort=reasoning_effort,
419
427
  )
420
428
 
421
429
  if langfuse is not None and dataset_item_id:
@@ -435,6 +443,7 @@ def evaluate_agentic_conversation(
435
443
  run_timestamp,
436
444
  model_version_override,
437
445
  run_metadata_extra,
446
+ reasoning_effort,
438
447
  )
439
448
  traces_by_conv = find_traces_per_conversation(
440
449
  langfuse,
@@ -6,6 +6,7 @@ from __future__ import annotations
6
6
  from dataclasses import dataclass
7
7
 
8
8
  from gooddata_eval.core.chat.sse_client import ChatClient
9
+ from gooddata_eval.core.config import ReasoningEffort
9
10
  from gooddata_eval.core.evaluators._llm_judge import LLMJudge
10
11
 
11
12
  _DEFAULT_K = 1
@@ -71,10 +72,11 @@ def run_agentic_general_question(
71
72
  expected_output: str,
72
73
  k: int = _DEFAULT_K,
73
74
  initial_conversation_id: str | None = None,
75
+ reasoning_effort: ReasoningEffort | None = None,
74
76
  ) -> AgenticGeneralQuestionSummary:
75
77
  """Run the general-question agentic evaluation K times and return a summary."""
76
78
  run_results: list[GeneralQuestionResult] = []
77
- client = ChatClient(host=host, token=token, workspace_id=workspace_id)
79
+ client = ChatClient(host=host, token=token, workspace_id=workspace_id, reasoning_effort=reasoning_effort)
78
80
  judge = LLMJudge(_GENERAL_QUESTION_EVALUATION_STEPS, model="gpt-4o")
79
81
 
80
82
  try:
@@ -153,6 +155,7 @@ def evaluate_agentic_general_question(
153
155
  run_timestamp: str | None = None,
154
156
  model_version_override: str | None = None,
155
157
  run_metadata_extra: dict | None = None,
158
+ reasoning_effort: ReasoningEffort | None = None,
156
159
  ) -> None:
157
160
  """Run general-question evaluation, log to Langfuse, and raise on failure."""
158
161
  from datetime import datetime as _dt # noqa: PLC0415
@@ -171,6 +174,7 @@ def evaluate_agentic_general_question(
171
174
  expected_output=expected_output,
172
175
  k=k,
173
176
  initial_conversation_id=initial_conversation_id,
177
+ reasoning_effort=reasoning_effort,
174
178
  )
175
179
 
176
180
  if langfuse is not None and dataset_item_id:
@@ -190,6 +194,7 @@ def evaluate_agentic_general_question(
190
194
  run_timestamp,
191
195
  model_version_override,
192
196
  run_metadata_extra,
197
+ reasoning_effort,
193
198
  )
194
199
  traces_by_conv = find_traces_per_conversation(
195
200
  langfuse,
@@ -6,6 +6,7 @@ from __future__ import annotations
6
6
  from dataclasses import dataclass
7
7
 
8
8
  from gooddata_eval.core.chat.sse_client import ChatClient
9
+ from gooddata_eval.core.config import ReasoningEffort
9
10
  from gooddata_eval.core.evaluators._llm_judge import LLMJudge
10
11
 
11
12
  _DEFAULT_K = 1
@@ -68,10 +69,11 @@ def run_agentic_guardrail(
68
69
  expected_output: str,
69
70
  k: int = _DEFAULT_K,
70
71
  initial_conversation_id: str | None = None,
72
+ reasoning_effort: ReasoningEffort | None = None,
71
73
  ) -> AgenticGuardrailSummary:
72
74
  """Run the guardrail agentic evaluation K times and return a summary."""
73
75
  run_results: list[GuardrailResult] = []
74
- client = ChatClient(host=host, token=token, workspace_id=workspace_id)
76
+ client = ChatClient(host=host, token=token, workspace_id=workspace_id, reasoning_effort=reasoning_effort)
75
77
  judge = LLMJudge(_GUARDRAIL_EVALUATION_STEPS, model="gpt-4o")
76
78
 
77
79
  try:
@@ -150,6 +152,7 @@ def evaluate_agentic_guardrail(
150
152
  run_timestamp: str | None = None,
151
153
  model_version_override: str | None = None,
152
154
  run_metadata_extra: dict | None = None,
155
+ reasoning_effort: ReasoningEffort | None = None,
153
156
  ) -> None:
154
157
  """Run guardrail evaluation, log to Langfuse, and raise on failure."""
155
158
  from datetime import datetime as _dt # noqa: PLC0415
@@ -168,6 +171,7 @@ def evaluate_agentic_guardrail(
168
171
  expected_output=expected_output,
169
172
  k=k,
170
173
  initial_conversation_id=initial_conversation_id,
174
+ reasoning_effort=reasoning_effort,
171
175
  )
172
176
 
173
177
  if langfuse is not None and dataset_item_id:
@@ -187,6 +191,7 @@ def evaluate_agentic_guardrail(
187
191
  run_timestamp,
188
192
  model_version_override,
189
193
  run_metadata_extra,
194
+ reasoning_effort,
190
195
  )
191
196
  traces_by_conv = find_traces_per_conversation(
192
197
  langfuse,
@@ -11,6 +11,7 @@ from typing import Any
11
11
  from gooddata_sdk import GoodDataSdk
12
12
 
13
13
  from gooddata_eval.core.chat.sse_client import ChatClient
14
+ from gooddata_eval.core.config import ReasoningEffort
14
15
  from gooddata_eval.core.models import ToolCallEvent
15
16
 
16
17
  try:
@@ -232,6 +233,7 @@ def run_agentic_metric_skill(
232
233
  k: int = _DEFAULT_K,
233
234
  max_iterations: int = _DEFAULT_MAX_ITERATIONS,
234
235
  initial_conversation_id: str | None = None,
236
+ reasoning_effort: ReasoningEffort | None = None,
235
237
  ) -> AgenticMetricSummary:
236
238
  """Run the metric-skill agentic evaluation K times and return a summary.
237
239
 
@@ -240,7 +242,7 @@ def run_agentic_metric_skill(
240
242
  """
241
243
  expected_outputs: list[dict] = expected_output if isinstance(expected_output, list) else [expected_output]
242
244
  run_results: list[MetricRunResult] = []
243
- client = ChatClient(host=host, token=token, workspace_id=workspace_id)
245
+ client = ChatClient(host=host, token=token, workspace_id=workspace_id, reasoning_effort=reasoning_effort)
244
246
  sdk = GoodDataSdk.create(host, token)
245
247
 
246
248
  try:
@@ -300,6 +302,7 @@ def evaluate_agentic_metric_skill(
300
302
  run_timestamp: str | None = None,
301
303
  model_version_override: str | None = None,
302
304
  run_metadata_extra: dict | None = None,
305
+ reasoning_effort: ReasoningEffort | None = None,
303
306
  ) -> None:
304
307
  """Run metric-skill evaluation, log to Langfuse, and raise MetricSkillAssertionError on failure."""
305
308
  from datetime import datetime as _dt # noqa: PLC0415
@@ -319,6 +322,7 @@ def evaluate_agentic_metric_skill(
319
322
  k=k,
320
323
  max_iterations=max_iterations,
321
324
  initial_conversation_id=initial_conversation_id,
325
+ reasoning_effort=reasoning_effort,
322
326
  )
323
327
 
324
328
  if langfuse is not None and dataset_item_id:
@@ -338,6 +342,7 @@ def evaluate_agentic_metric_skill(
338
342
  run_timestamp,
339
343
  model_version_override,
340
344
  run_metadata_extra,
345
+ reasoning_effort,
341
346
  )
342
347
  traces_by_conv = find_traces_per_conversation(
343
348
  langfuse,
@@ -6,6 +6,7 @@ from __future__ import annotations
6
6
  from dataclasses import dataclass
7
7
 
8
8
  from gooddata_eval.core.chat.sse_client import ChatClient
9
+ from gooddata_eval.core.config import ReasoningEffort
9
10
  from gooddata_eval.core.models import ToolCallEvent
10
11
 
11
12
  _DEFAULT_K = 1
@@ -67,11 +68,12 @@ def run_agentic_search_tool(
67
68
  expected_tool_call: dict,
68
69
  k: int = _DEFAULT_K,
69
70
  initial_conversation_id: str | None = None,
71
+ reasoning_effort: ReasoningEffort | None = None,
70
72
  ) -> AgenticSearchSummary:
71
73
  """Run the search-tool agentic evaluation K times (single-turn each)."""
72
74
  run_results: list[SearchResult] = []
73
75
 
74
- client = ChatClient(host=host, token=token, workspace_id=workspace_id)
76
+ client = ChatClient(host=host, token=token, workspace_id=workspace_id, reasoning_effort=reasoning_effort)
75
77
  try:
76
78
  conv_id_0 = initial_conversation_id if initial_conversation_id is not None else client.create_conversation()
77
79
  try:
@@ -144,6 +146,7 @@ def evaluate_agentic_search_tool(
144
146
  run_timestamp: str | None = None,
145
147
  model_version_override: str | None = None,
146
148
  run_metadata_extra: dict | None = None,
149
+ reasoning_effort: ReasoningEffort | None = None,
147
150
  ) -> None:
148
151
  """Run search-tool evaluation, log to Langfuse, and raise SearchToolAssertionError on failure."""
149
152
  from datetime import datetime as _dt # noqa: PLC0415
@@ -162,6 +165,7 @@ def evaluate_agentic_search_tool(
162
165
  expected_tool_call=expected_tool_call,
163
166
  k=k,
164
167
  initial_conversation_id=initial_conversation_id,
168
+ reasoning_effort=reasoning_effort,
165
169
  )
166
170
 
167
171
  if langfuse is not None and dataset_item_id:
@@ -181,6 +185,7 @@ def evaluate_agentic_search_tool(
181
185
  run_timestamp,
182
186
  model_version_override,
183
187
  run_metadata_extra,
188
+ reasoning_effort,
184
189
  )
185
190
  traces_by_conv = find_traces_per_conversation(
186
191
  langfuse,
@@ -11,6 +11,7 @@ import os
11
11
  from dataclasses import dataclass
12
12
 
13
13
  from gooddata_eval.core.chat.sse_client import ChatClient
14
+ from gooddata_eval.core.config import ReasoningEffort
14
15
  from gooddata_eval.core.evaluators.visualization import (
15
16
  EvaluationResult,
16
17
  _check_visualization_skill_activated,
@@ -203,6 +204,7 @@ def run_agentic_visualization(
203
204
  k: int = _DEFAULT_K,
204
205
  max_iterations: int = _DEFAULT_MAX_ITERATIONS,
205
206
  initial_conversation_id: str | None = None,
207
+ reasoning_effort: ReasoningEffort | None = None,
206
208
  ) -> AgenticRunSummary:
207
209
  """Run K independent conversations and return evaluation results.
208
210
 
@@ -211,7 +213,7 @@ def run_agentic_visualization(
211
213
  fresh conversations. Caller-supplied conversations are not deleted; all
212
214
  conversations created by this function are deleted on completion.
213
215
  """
214
- client = ChatClient(host=host, token=token, workspace_id=workspace_id)
216
+ client = ChatClient(host=host, token=token, workspace_id=workspace_id, reasoning_effort=reasoning_effort)
215
217
  run_results: list[RunResult] = []
216
218
 
217
219
  try:
@@ -265,6 +267,7 @@ def evaluate_agentic_visualization(
265
267
  model_version_override: str | None = None,
266
268
  run_metadata_extra: dict | None = None,
267
269
  record_output_path: str | None = None,
270
+ reasoning_effort: ReasoningEffort | None = None,
268
271
  ) -> None:
269
272
  """Run visualization evaluation, log to Langfuse, and raise VisualizationAssertionError on failure."""
270
273
  import json as _json # noqa: PLC0415
@@ -285,6 +288,7 @@ def evaluate_agentic_visualization(
285
288
  k=k,
286
289
  max_iterations=max_iterations,
287
290
  initial_conversation_id=initial_conversation_id,
291
+ reasoning_effort=reasoning_effort,
288
292
  )
289
293
 
290
294
  if langfuse is not None and dataset_item_id:
@@ -304,6 +308,7 @@ def evaluate_agentic_visualization(
304
308
  run_timestamp,
305
309
  model_version_override,
306
310
  run_metadata_extra,
311
+ reasoning_effort,
307
312
  )
308
313
  K = len(summary.run_results)
309
314
  traces_by_conv = find_traces_per_conversation(