gooddata-eval 1.74.1.dev2__tar.gz → 1.74.1.dev3__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (120) hide show
  1. {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/AGENTS.md +4 -4
  2. {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/PKG-INFO +58 -14
  3. {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/README.md +56 -12
  4. {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/pyproject.toml +2 -2
  5. {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/cli/agentic_runner.py +32 -2
  6. {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/cli/main.py +62 -11
  7. gooddata_eval-1.74.1.dev3/src/gooddata_eval/core/agentic/_gate.py +70 -0
  8. {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/agentic/_langfuse.py +199 -240
  9. {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/agentic/_trace_linker.py +24 -5
  10. {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/agentic/alert_skill.py +16 -3
  11. {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/agentic/conversation.py +10 -1
  12. {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/agentic/general_question.py +18 -3
  13. {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/agentic/guardrail.py +20 -3
  14. {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/agentic/kda_skill.py +16 -3
  15. {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/agentic/metric_skill.py +21 -3
  16. {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/agentic/search_tool.py +18 -3
  17. {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/agentic/visualization.py +26 -9
  18. {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/config.py +21 -0
  19. {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/dataset/langfuse_source.py +4 -14
  20. gooddata_eval-1.74.1.dev3/src/gooddata_eval/core/langfuse/_env.py +39 -0
  21. gooddata_eval-1.74.1.dev3/src/gooddata_eval/core/langfuse/client.py +205 -0
  22. gooddata_eval-1.74.1.dev3/src/gooddata_eval/core/langfuse/experiment.py +156 -0
  23. gooddata_eval-1.74.1.dev3/src/gooddata_eval/core/langfuse/observations.py +125 -0
  24. gooddata_eval-1.74.1.dev3/src/gooddata_eval/core/langfuse/otlp.py +164 -0
  25. gooddata_eval-1.74.1.dev3/src/gooddata_eval/core/langfuse/sink.py +184 -0
  26. {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/reporting/console.py +12 -2
  27. {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/reporting/json_report.py +7 -1
  28. {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/runner.py +20 -3
  29. gooddata_eval-1.74.1.dev3/tests/_fake_langfuse.py +273 -0
  30. gooddata_eval-1.74.1.dev3/tests/conftest.py +22 -0
  31. gooddata_eval-1.74.1.dev3/tests/test_agentic_gate.py +381 -0
  32. {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/tests/test_agentic_kda_skill.py +5 -7
  33. {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/tests/test_agentic_langfuse_trace.py +76 -86
  34. gooddata_eval-1.74.1.dev3/tests/test_agentic_observe_experiment.py +325 -0
  35. {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/tests/test_cli.py +2 -2
  36. gooddata_eval-1.74.1.dev3/tests/test_fake_langfuse.py +140 -0
  37. gooddata_eval-1.74.1.dev3/tests/test_langfuse_client.py +379 -0
  38. gooddata_eval-1.74.1.dev3/tests/test_langfuse_e2e_fake_server.py +385 -0
  39. gooddata_eval-1.74.1.dev3/tests/test_langfuse_env.py +70 -0
  40. gooddata_eval-1.74.1.dev3/tests/test_langfuse_experiment.py +206 -0
  41. gooddata_eval-1.74.1.dev3/tests/test_langfuse_observations.py +264 -0
  42. gooddata_eval-1.74.1.dev3/tests/test_langfuse_otlp.py +197 -0
  43. gooddata_eval-1.74.1.dev3/tests/test_langfuse_sink.py +251 -0
  44. {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/tests/test_langfuse_source.py +19 -1
  45. {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/tests/test_trace_linker.py +16 -2
  46. gooddata_eval-1.74.1.dev2/src/gooddata_eval/core/langfuse/sink.py +0 -197
  47. gooddata_eval-1.74.1.dev2/tests/conftest.py +0 -9
  48. gooddata_eval-1.74.1.dev2/tests/test_langfuse_sink.py +0 -165
  49. {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/.gitignore +0 -0
  50. {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/CLAUDE.md +0 -0
  51. {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/LICENSE.txt +0 -0
  52. {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/Makefile +0 -0
  53. {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/__init__.py +0 -0
  54. {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/_version.py +0 -0
  55. {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/cli/__init__.py +0 -0
  56. {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/__init__.py +0 -0
  57. {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/_output.py +0 -0
  58. {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/agentic/__init__.py +0 -0
  59. {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/agentic/_catalog.py +0 -0
  60. {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/chat/__init__.py +0 -0
  61. {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/chat/render.py +0 -0
  62. {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/chat/sse_client.py +0 -0
  63. {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/connection.py +0 -0
  64. {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/dataset/__init__.py +0 -0
  65. {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/dataset/local.py +0 -0
  66. {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/evaluators/__init__.py +0 -0
  67. {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/evaluators/_deep_subset.py +0 -0
  68. {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/evaluators/_llm_judge.py +0 -0
  69. {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/evaluators/_maql.py +0 -0
  70. {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/evaluators/_text_utils.py +0 -0
  71. {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/evaluators/alert_skill.py +0 -0
  72. {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/evaluators/base.py +0 -0
  73. {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/evaluators/general_question.py +0 -0
  74. {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/evaluators/guardrail.py +0 -0
  75. {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/evaluators/metric_skill.py +0 -0
  76. {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/evaluators/search_tool.py +0 -0
  77. {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/evaluators/summary.py +0 -0
  78. {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/evaluators/visualization.py +0 -0
  79. {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/langfuse/__init__.py +0 -0
  80. {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/models.py +0 -0
  81. {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/reporting/__init__.py +0 -0
  82. {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/scoring.py +0 -0
  83. {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/summary/__init__.py +0 -0
  84. {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/summary/http_client.py +0 -0
  85. {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/timing.py +0 -0
  86. {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/workspace.py +0 -0
  87. {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/tests/__init__.py +0 -0
  88. {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/tests/fixtures/sample_dataset/metric_skill_create.json +0 -0
  89. {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/tests/fixtures/sample_dataset/visualization_revenue.json +0 -0
  90. {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/tests/fixtures/sse_visualization_stream.txt +0 -0
  91. {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/tests/test_agentic_alert_skill.py +0 -0
  92. {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/tests/test_agentic_conversation.py +0 -0
  93. {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/tests/test_agentic_general_question.py +0 -0
  94. {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/tests/test_agentic_guardrail.py +0 -0
  95. {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/tests/test_agentic_metric_skill.py +0 -0
  96. {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/tests/test_agentic_run_context.py +0 -0
  97. {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/tests/test_agentic_runner.py +0 -0
  98. {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/tests/test_agentic_search_tool.py +0 -0
  99. {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/tests/test_agentic_visualization.py +0 -0
  100. {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/tests/test_alert_skill_evaluator.py +0 -0
  101. {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/tests/test_chat_render.py +0 -0
  102. {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/tests/test_connection.py +0 -0
  103. {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/tests/test_deep_subset.py +0 -0
  104. {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/tests/test_llm_judge.py +0 -0
  105. {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/tests/test_local_loader.py +0 -0
  106. {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/tests/test_maql_normalize.py +0 -0
  107. {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/tests/test_metric_skill_evaluator.py +0 -0
  108. {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/tests/test_models.py +0 -0
  109. {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/tests/test_reporting.py +0 -0
  110. {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/tests/test_runner.py +0 -0
  111. {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/tests/test_scoring.py +0 -0
  112. {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/tests/test_search_tool_evaluator.py +0 -0
  113. {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/tests/test_sse_client.py +0 -0
  114. {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/tests/test_summary_client.py +0 -0
  115. {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/tests/test_summary_evaluator.py +0 -0
  116. {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/tests/test_text_evaluators.py +0 -0
  117. {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/tests/test_timing.py +0 -0
  118. {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/tests/test_visualization_evaluator.py +0 -0
  119. {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/tests/test_workspace.py +0 -0
  120. {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/tox.ini +0 -0
@@ -4,15 +4,15 @@
4
4
  `gdc-nas`) through a dataset of natural-language questions and scores what comes back,
5
5
  including side-by-side comparison across models. Each dataset item is a JSON envelope
6
6
  loaded from a local folder or pulled from a Langfuse dataset. Results are aggregated into
7
- pass@K / pass^K reports and optionally pushed to Langfuse as scored traces tied to a
8
- dataset run. The newest and most actively developed package in the repo.
7
+ pass@K / pass^K reports and optionally pushed to Langfuse as scored traces tied to an
8
+ experiment. The newest and most actively developed package in the repo.
9
9
 
10
10
  ## Owns
11
11
 
12
12
  - The `gd-eval` CLI (`gd-eval run`, `gd-eval models`)
13
13
  - Dataset loading and the evaluation run loop
14
14
  - Per-capability evaluators and their scoring
15
- - Result reporting, and pushing runs, scores and trace links to Langfuse
15
+ - Result reporting, and pushing experiments, scores and trace links to Langfuse
16
16
 
17
17
  ## Does NOT Own
18
18
 
@@ -29,7 +29,7 @@ dataset run. The newest and most actively developed package in the repo.
29
29
  | `core/summary/` | HTTP client for the dedicated dashboard-summary endpoint — a single-shot chat backend, not reporting |
30
30
  | `core/dataset/` | dataset format and loading |
31
31
  | `core/evaluators/` | single-shot evaluators and their registry |
32
- | `core/langfuse/` | `sink.py` only — pushes single-turn scores and dataset-run items |
32
+ | `core/langfuse/` | the whole Langfuse v4 client: `_env` (base URL + credentials), `otlp` (OTLP/JSON encoding), `experiment` (root-span construction, score targets), `observations` (trace reads), `client` (httpx calls), `sink` (single-shot results as experiments) |
33
33
  | `core/reporting/` | console and JSON output rendering |
34
34
  | `core/scoring.py`, `core/runner.py` | scoring and orchestration |
35
35
  | `core/models.py` | `DatasetItem`, `ChatResult`, `ItemReport` and friends |
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: gooddata-eval
3
- Version: 1.74.1.dev2
3
+ Version: 1.74.1.dev3
4
4
  Summary: Evaluate the GoodData AI agent against your own questions and models.
5
5
  Project-URL: Source, https://github.com/gooddata/gooddata-python-sdk
6
6
  Author-email: GoodData <support@gooddata.com>
@@ -17,7 +17,7 @@ Classifier: Topic :: Scientific/Engineering
17
17
  Classifier: Topic :: Software Development
18
18
  Classifier: Typing :: Typed
19
19
  Requires-Python: >=3.10
20
- Requires-Dist: gooddata-sdk~=1.74.1.dev2
20
+ Requires-Dist: gooddata-sdk~=1.74.1.dev3
21
21
  Requires-Dist: httpx<1.0,>=0.27
22
22
  Requires-Dist: orjson<4.0.0,>=3.9.15
23
23
  Requires-Dist: pydantic<3.0,>=2.6
@@ -141,7 +141,7 @@ gd-eval run \
141
141
  | Flag | Description |
142
142
  |---|---|
143
143
  | `--dataset PATH` | Flat folder of JSON files — one question per file. |
144
- | `--langfuse-dataset NAME` | Pull items by name from a Langfuse dataset. Requires `LANGFUSE_PUBLIC_KEY`, `LANGFUSE_SECRET_KEY`, `LANGFUSE_HOST`. |
144
+ | `--langfuse-dataset NAME` | Pull items by name from a Langfuse dataset. Requires `LANGFUSE_PUBLIC_KEY`, `LANGFUSE_SECRET_KEY` and `LANGFUSE_BASE_URL` (or the legacy `LANGFUSE_HOST`). |
145
145
  | `--kind TEST_KIND` | Fallback `test_kind` for dataset items that do not embed one. Defaults to `visualization`; use e.g. `agentic_metric_skill` for multi-turn agentic evaluation. Items that declare their own `test_kind` ignore this. |
146
146
 
147
147
  #### Model selection
@@ -154,7 +154,8 @@ gd-eval run \
154
154
 
155
155
  | Flag | Default | Description |
156
156
  |---|---|---|
157
- | `--runs K` | `2` | Independent runs per item (pass@K). An item passes if any run passes. |
157
+ | `--runs K` | `2` | Independent runs per item. |
158
+ | `--gate` | `any` | Which verdict decides an item: `any` = pass@K (a run passing is enough), `power` = pass^K (every run must pass, so the verdict measures stability). Identical at `--runs 1`. Kinds that repeat K runs only — `power` is refused when the dataset also has non-agentic items or `agentic_conversation`, which are always decided on pass@K. |
158
159
  | `--concurrency K` | `1` | Number of items evaluated concurrently. `1` = sequential (default). Increase to load-test the agent under simultaneous requests — see *Concurrency and workspace safety* below. |
159
160
  | `--judge-model MODEL` | `gpt-4o` | Model used for LLM-as-judge scoring — `agentic_general_question`, `agentic_guardrail`, `general_question`, `guardrail` and `dashboard_summary`. Also settable via `GD_EVAL_JUDGE_MODEL`. Two things to weigh before changing it: the gpt-5 family rejects `temperature=0`, so verdicts stop being reproducible (the run warns when this happens); and choosing the same model the agent runs means the judge grades its own family's output. |
160
161
  | `--reasoning-effort LEVEL` | server default | `LOW`, `MEDIUM` or `HIGH`, sent as `options.reasoningEffort` on every chat message. Requires the `enableGenAiReasoningEffort` feature flag on the target organization — without it the server ignores the value. Applies to chat items only; `dashboard_summary` items go through the summary endpoint, which has no such option. |
@@ -180,15 +181,39 @@ interleaves when K > 1, and per-item latencies rise, so they stop being clean si
180
181
 
181
182
  | Flag | Description |
182
183
  |---|---|
183
- | `--langfuse` | Log scores and traces to Langfuse after each item. Requires `--langfuse-dataset`. Names each experiment run `{dataset_name}_{timestamp}_{model}`, suffixed `_effort-{level}` when `--reasoning-effort` is set (so runs differing only by effort stay separate) and `_run{N}` per run when `--runs` > 1 — e.g. `general_question_2026-09-02-11-13_gpt-5.2_run0`. Requires `LANGFUSE_PUBLIC_KEY`, `LANGFUSE_SECRET_KEY`, `LANGFUSE_HOST`. |
184
+ | `--langfuse` | Log scores and traces to Langfuse after each item. Requires `--langfuse-dataset`. Names each experiment run `{dataset_name}_{timestamp}_{model}`, suffixed `_effort-{level}` when `--reasoning-effort` is set (so runs differing only by effort stay separate) and `_run{N}` per run when `--runs` > 1 — e.g. `general_question_2026-09-02-11-13_gpt-5.2_run0`. Requires `LANGFUSE_PUBLIC_KEY`, `LANGFUSE_SECRET_KEY` and `LANGFUSE_BASE_URL` (or the legacy `LANGFUSE_HOST`). |
184
185
 
185
- Set `TAVERN_E2E_SKIP_TRACE_LINK=1` to skip trace lookup entirely (scores are then orphaned; the run says so).
186
+ ##### Langfuse v4
186
187
 
187
- **A local `--dataset` cannot be attached to a Langfuse run.** `--langfuse` is refused alongside `--dataset`
188
- because a local folder's item ids are not Langfuse dataset item ids. But trace linking does not depend on that
189
- flag — each `evaluate_agentic_*` builds its own client whenever `LANGFUSE_PUBLIC_KEY`/`LANGFUSE_SECRET_KEY` are
190
- exported — so a local run still finds its traces and writes its scores onto them, and only the per-run grouping
191
- fails, with one `404 from dataset-run-items` reported per run. The run warns about this before it starts. Use
188
+ One run is one Langfuse **experiment**. Each evaluated item becomes its own trace whose root span carries the
189
+ experiment and dataset-item attributes (`langfuse.experiment.name`, `langfuse.experiment.dataset.id`,
190
+ `langfuse.experiment.item.id`), and the four scores attach to that root observation. gd-eval speaks to Langfuse
191
+ over four REST endpoints and uses no Langfuse SDK, so it runs on every Python version the package supports:
192
+
193
+ | Endpoint | Used for |
194
+ |---|---|
195
+ | `POST /api/public/otel/v1/traces` | exporting the experiment root span as OTLP/HTTP JSON |
196
+ | `POST /api/public/scores` | one score per write, on a trace or on a single observation inside it |
197
+ | `GET /api/public/v2/observations` | finding the agent's gen-ai trace for a conversation |
198
+ | `GET /api/public/dataset-items` | loading `--langfuse-dataset` items, and resolving an item's dataset id |
199
+
200
+ Two consequences of v4's immutable observations. gd-eval sets `version` only on its own experiment span, never on
201
+ the agent's gen-ai trace — filter on the gd-eval experiment's `langfuse.version` to compare models. And on the
202
+ agentic kinds the latency in `value_score` is the gen-ai trace's root generation latency, read from the
203
+ observations endpoint; the single-shot `--langfuse` sink keeps using the item's own measured average latency.
204
+
205
+ Langfuse Cloud drops v3 on **2026-11-16**; a self-hosted Langfuse must be on v4 for any of this to work.
206
+
207
+ Set `TAVERN_E2E_SKIP_TRACE_LINK=1` to turn the whole **agentic** Langfuse write path off — no trace lookup, no
208
+ span export and no scores for `agentic_*` items. The run says so once. It does not reach the `--langfuse` sink,
209
+ which still writes a span and four scores for every single-shot item; drop `--langfuse` to silence that too.
210
+
211
+ **A local `--dataset` cannot be attached to a Langfuse experiment.** `--langfuse` is refused alongside
212
+ `--dataset` because a local folder's item ids are not Langfuse dataset item ids. But trace linking does not
213
+ depend on that flag — each `evaluate_agentic_*` builds its own client whenever
214
+ `LANGFUSE_PUBLIC_KEY`/`LANGFUSE_SECRET_KEY` are exported — so a local run still finds its traces and writes its
215
+ scores onto them, and only the per-run grouping fails: the dataset-item lookup 404s and the run reports the item
216
+ as one that does not exist in Langfuse, once. The run also warns about this before it starts. Use
192
217
  `--langfuse-dataset` when you want runs that are comparable across models, or `TAVERN_E2E_SKIP_TRACE_LINK=1` to
193
218
  skip linking altogether.
194
219
 
@@ -235,10 +260,12 @@ Winner is selected by **pass rate → quality score → latency** (lower latency
235
260
  Each item reports **how many of its runs passed**, not only whether one did:
236
261
 
237
262
  ```json
238
- "runs": 5, "runs_passed": 4, "pass_at_k": true, "pass_power_k": false
263
+ "runs": 5, "runs_passed": 4, "pass_at_k": true, "pass_power_k": false, "gate_passed": true
239
264
  ```
240
265
 
241
- `pass_at_k` is "did any run pass" and is what `passed` counts. `runs_passed` is the fact that separates a
266
+ `pass_at_k` is "did any run pass" — always literal, whatever the gate. `gate_passed` is the verdict the item
267
+ was decided on and is what `passed` counts; the two differ only under `--gate power`, where an item that
268
+ passed 4 of 5 runs is `"pass_at_k": true, "gate_passed": false`. `runs_passed` is the fact that separates a
242
269
  reliable item from a coin-flip — without it a 5/5 item and a 1/5 item are identical in every field, because
243
270
  `quality_score` is derived from the best run alone. `pass_power_k` is true only when every run passed, and the
244
271
  run summary carries `passed_all_runs` beside `passed`; a large gap between the two means the model is
@@ -248,6 +275,14 @@ stays quiet for a unanimous one, and its summary line reads `3/4 passed, 1 on ev
248
275
  `runs` is what the item actually ran, which is not always the requested `--runs`: `agentic_conversation` takes
249
276
  no K and drives its fixture exactly once.
250
277
 
278
+ Which of the two decides pass/fail is `--gate`: `any` (default) gates on `pass_at_k`, `power` gates on
279
+ `pass_power_k`. The run records it as a top-level `gate`, and a failure under `power` says so —
280
+ `Gate pass^3 failed: 2/3 runs passed — unstable, not a clean failure` — because the message body describes
281
+ the best run, which under pass^K can be a run that passed, and the console repeats the count in `Notes` for
282
+ the same reason. When some runs were ungraded the note says so instead of calling the remainder unstable.
283
+ `agentic_conversation` has no K gate, so `--gate power` is refused for a dataset containing one rather than
284
+ labelling a report `power` that only part of the dataset was decided under.
285
+
251
286
  Each item additionally carries a per-phase breakdown:
252
287
 
253
288
  ```json
@@ -408,10 +443,19 @@ Without `[llm-judge]`, those items are **skipped**.
408
443
 
409
444
  ## Scores (in JSON report and Langfuse)
410
445
 
446
+ In Langfuse every score is written to the experiment run's root observation — `traceId` plus `observationId` of
447
+ the item's own root span. On the agentic path each score is mirrored onto the agent's gen-ai trace as well
448
+ (`traceId` only), so a score survives even when one of the two traces is missing.
449
+
411
450
  | Score | Description |
412
451
  |---|---|
413
- | `pass_at_k` | 1 if any of the K runs passed strict checks, else 0. |
452
+ | `pass_at_k` | 1 if **any** of the K runs passed strict checks, else 0. |
453
+ | `pass_power_k` | 1 only if **every** one of the K runs passed. Agentic kinds only. |
454
+ | `gate_passed` | The verdict that decided the item: `pass_at_k` under `--gate any`, `pass_power_k` under `--gate power`. Agentic kinds only. |
414
455
  | `quality_score` | Fraction of strict check flags that are `True` (0.0–1.0). Shown in CLI as a percentage. |
415
456
  | `value_score` | Weighted blend: 0.6 × quality + 0.2 × speed (speed = max(0, 1 − latency/60s)). |
416
457
  | `latency_s` | Average per-run latency in seconds. |
417
458
  | `provider_type` | Model vendor + gateway label (e.g. `ANTHROPIC`, `BEDROCK/ANTHROPIC`, `AZURE/OPENAI`). Stored in Langfuse trace metadata and tags. |
459
+
460
+ Score names carry no K; K and the gate are on the dataset-run metadata as `eval_k` and `eval_gate`.
461
+ `agentic_visualization` also still writes its historic `pass_at_{K}` / `pass_power_{K}` pair.
@@ -113,7 +113,7 @@ gd-eval run \
113
113
  | Flag | Description |
114
114
  |---|---|
115
115
  | `--dataset PATH` | Flat folder of JSON files — one question per file. |
116
- | `--langfuse-dataset NAME` | Pull items by name from a Langfuse dataset. Requires `LANGFUSE_PUBLIC_KEY`, `LANGFUSE_SECRET_KEY`, `LANGFUSE_HOST`. |
116
+ | `--langfuse-dataset NAME` | Pull items by name from a Langfuse dataset. Requires `LANGFUSE_PUBLIC_KEY`, `LANGFUSE_SECRET_KEY` and `LANGFUSE_BASE_URL` (or the legacy `LANGFUSE_HOST`). |
117
117
  | `--kind TEST_KIND` | Fallback `test_kind` for dataset items that do not embed one. Defaults to `visualization`; use e.g. `agentic_metric_skill` for multi-turn agentic evaluation. Items that declare their own `test_kind` ignore this. |
118
118
 
119
119
  #### Model selection
@@ -126,7 +126,8 @@ gd-eval run \
126
126
 
127
127
  | Flag | Default | Description |
128
128
  |---|---|---|
129
- | `--runs K` | `2` | Independent runs per item (pass@K). An item passes if any run passes. |
129
+ | `--runs K` | `2` | Independent runs per item. |
130
+ | `--gate` | `any` | Which verdict decides an item: `any` = pass@K (a run passing is enough), `power` = pass^K (every run must pass, so the verdict measures stability). Identical at `--runs 1`. Kinds that repeat K runs only — `power` is refused when the dataset also has non-agentic items or `agentic_conversation`, which are always decided on pass@K. |
130
131
  | `--concurrency K` | `1` | Number of items evaluated concurrently. `1` = sequential (default). Increase to load-test the agent under simultaneous requests — see *Concurrency and workspace safety* below. |
131
132
  | `--judge-model MODEL` | `gpt-4o` | Model used for LLM-as-judge scoring — `agentic_general_question`, `agentic_guardrail`, `general_question`, `guardrail` and `dashboard_summary`. Also settable via `GD_EVAL_JUDGE_MODEL`. Two things to weigh before changing it: the gpt-5 family rejects `temperature=0`, so verdicts stop being reproducible (the run warns when this happens); and choosing the same model the agent runs means the judge grades its own family's output. |
132
133
  | `--reasoning-effort LEVEL` | server default | `LOW`, `MEDIUM` or `HIGH`, sent as `options.reasoningEffort` on every chat message. Requires the `enableGenAiReasoningEffort` feature flag on the target organization — without it the server ignores the value. Applies to chat items only; `dashboard_summary` items go through the summary endpoint, which has no such option. |
@@ -152,15 +153,39 @@ interleaves when K > 1, and per-item latencies rise, so they stop being clean si
152
153
 
153
154
  | Flag | Description |
154
155
  |---|---|
155
- | `--langfuse` | Log scores and traces to Langfuse after each item. Requires `--langfuse-dataset`. Names each experiment run `{dataset_name}_{timestamp}_{model}`, suffixed `_effort-{level}` when `--reasoning-effort` is set (so runs differing only by effort stay separate) and `_run{N}` per run when `--runs` > 1 — e.g. `general_question_2026-09-02-11-13_gpt-5.2_run0`. Requires `LANGFUSE_PUBLIC_KEY`, `LANGFUSE_SECRET_KEY`, `LANGFUSE_HOST`. |
156
+ | `--langfuse` | Log scores and traces to Langfuse after each item. Requires `--langfuse-dataset`. Names each experiment run `{dataset_name}_{timestamp}_{model}`, suffixed `_effort-{level}` when `--reasoning-effort` is set (so runs differing only by effort stay separate) and `_run{N}` per run when `--runs` > 1 — e.g. `general_question_2026-09-02-11-13_gpt-5.2_run0`. Requires `LANGFUSE_PUBLIC_KEY`, `LANGFUSE_SECRET_KEY` and `LANGFUSE_BASE_URL` (or the legacy `LANGFUSE_HOST`). |
156
157
 
157
- Set `TAVERN_E2E_SKIP_TRACE_LINK=1` to skip trace lookup entirely (scores are then orphaned; the run says so).
158
+ ##### Langfuse v4
158
159
 
159
- **A local `--dataset` cannot be attached to a Langfuse run.** `--langfuse` is refused alongside `--dataset`
160
- because a local folder's item ids are not Langfuse dataset item ids. But trace linking does not depend on that
161
- flag — each `evaluate_agentic_*` builds its own client whenever `LANGFUSE_PUBLIC_KEY`/`LANGFUSE_SECRET_KEY` are
162
- exported — so a local run still finds its traces and writes its scores onto them, and only the per-run grouping
163
- fails, with one `404 from dataset-run-items` reported per run. The run warns about this before it starts. Use
160
+ One run is one Langfuse **experiment**. Each evaluated item becomes its own trace whose root span carries the
161
+ experiment and dataset-item attributes (`langfuse.experiment.name`, `langfuse.experiment.dataset.id`,
162
+ `langfuse.experiment.item.id`), and the four scores attach to that root observation. gd-eval speaks to Langfuse
163
+ over four REST endpoints and uses no Langfuse SDK, so it runs on every Python version the package supports:
164
+
165
+ | Endpoint | Used for |
166
+ |---|---|
167
+ | `POST /api/public/otel/v1/traces` | exporting the experiment root span as OTLP/HTTP JSON |
168
+ | `POST /api/public/scores` | one score per write, on a trace or on a single observation inside it |
169
+ | `GET /api/public/v2/observations` | finding the agent's gen-ai trace for a conversation |
170
+ | `GET /api/public/dataset-items` | loading `--langfuse-dataset` items, and resolving an item's dataset id |
171
+
172
+ Two consequences of v4's immutable observations. gd-eval sets `version` only on its own experiment span, never on
173
+ the agent's gen-ai trace — filter on the gd-eval experiment's `langfuse.version` to compare models. And on the
174
+ agentic kinds the latency in `value_score` is the gen-ai trace's root generation latency, read from the
175
+ observations endpoint; the single-shot `--langfuse` sink keeps using the item's own measured average latency.
176
+
177
+ Langfuse Cloud drops v3 on **2026-11-16**; a self-hosted Langfuse must be on v4 for any of this to work.
178
+
179
+ Set `TAVERN_E2E_SKIP_TRACE_LINK=1` to turn the whole **agentic** Langfuse write path off — no trace lookup, no
180
+ span export and no scores for `agentic_*` items. The run says so once. It does not reach the `--langfuse` sink,
181
+ which still writes a span and four scores for every single-shot item; drop `--langfuse` to silence that too.
182
+
183
+ **A local `--dataset` cannot be attached to a Langfuse experiment.** `--langfuse` is refused alongside
184
+ `--dataset` because a local folder's item ids are not Langfuse dataset item ids. But trace linking does not
185
+ depend on that flag — each `evaluate_agentic_*` builds its own client whenever
186
+ `LANGFUSE_PUBLIC_KEY`/`LANGFUSE_SECRET_KEY` are exported — so a local run still finds its traces and writes its
187
+ scores onto them, and only the per-run grouping fails: the dataset-item lookup 404s and the run reports the item
188
+ as one that does not exist in Langfuse, once. The run also warns about this before it starts. Use
164
189
  `--langfuse-dataset` when you want runs that are comparable across models, or `TAVERN_E2E_SKIP_TRACE_LINK=1` to
165
190
  skip linking altogether.
166
191
 
@@ -207,10 +232,12 @@ Winner is selected by **pass rate → quality score → latency** (lower latency
207
232
  Each item reports **how many of its runs passed**, not only whether one did:
208
233
 
209
234
  ```json
210
- "runs": 5, "runs_passed": 4, "pass_at_k": true, "pass_power_k": false
235
+ "runs": 5, "runs_passed": 4, "pass_at_k": true, "pass_power_k": false, "gate_passed": true
211
236
  ```
212
237
 
213
- `pass_at_k` is "did any run pass" and is what `passed` counts. `runs_passed` is the fact that separates a
238
+ `pass_at_k` is "did any run pass" — always literal, whatever the gate. `gate_passed` is the verdict the item
239
+ was decided on and is what `passed` counts; the two differ only under `--gate power`, where an item that
240
+ passed 4 of 5 runs is `"pass_at_k": true, "gate_passed": false`. `runs_passed` is the fact that separates a
214
241
  reliable item from a coin-flip — without it a 5/5 item and a 1/5 item are identical in every field, because
215
242
  `quality_score` is derived from the best run alone. `pass_power_k` is true only when every run passed, and the
216
243
  run summary carries `passed_all_runs` beside `passed`; a large gap between the two means the model is
@@ -220,6 +247,14 @@ stays quiet for a unanimous one, and its summary line reads `3/4 passed, 1 on ev
220
247
  `runs` is what the item actually ran, which is not always the requested `--runs`: `agentic_conversation` takes
221
248
  no K and drives its fixture exactly once.
222
249
 
250
+ Which of the two decides pass/fail is `--gate`: `any` (default) gates on `pass_at_k`, `power` gates on
251
+ `pass_power_k`. The run records it as a top-level `gate`, and a failure under `power` says so —
252
+ `Gate pass^3 failed: 2/3 runs passed — unstable, not a clean failure` — because the message body describes
253
+ the best run, which under pass^K can be a run that passed, and the console repeats the count in `Notes` for
254
+ the same reason. When some runs were ungraded the note says so instead of calling the remainder unstable.
255
+ `agentic_conversation` has no K gate, so `--gate power` is refused for a dataset containing one rather than
256
+ labelling a report `power` that only part of the dataset was decided under.
257
+
223
258
  Each item additionally carries a per-phase breakdown:
224
259
 
225
260
  ```json
@@ -380,10 +415,19 @@ Without `[llm-judge]`, those items are **skipped**.
380
415
 
381
416
  ## Scores (in JSON report and Langfuse)
382
417
 
418
+ In Langfuse every score is written to the experiment run's root observation — `traceId` plus `observationId` of
419
+ the item's own root span. On the agentic path each score is mirrored onto the agent's gen-ai trace as well
420
+ (`traceId` only), so a score survives even when one of the two traces is missing.
421
+
383
422
  | Score | Description |
384
423
  |---|---|
385
- | `pass_at_k` | 1 if any of the K runs passed strict checks, else 0. |
424
+ | `pass_at_k` | 1 if **any** of the K runs passed strict checks, else 0. |
425
+ | `pass_power_k` | 1 only if **every** one of the K runs passed. Agentic kinds only. |
426
+ | `gate_passed` | The verdict that decided the item: `pass_at_k` under `--gate any`, `pass_power_k` under `--gate power`. Agentic kinds only. |
386
427
  | `quality_score` | Fraction of strict check flags that are `True` (0.0–1.0). Shown in CLI as a percentage. |
387
428
  | `value_score` | Weighted blend: 0.6 × quality + 0.2 × speed (speed = max(0, 1 − latency/60s)). |
388
429
  | `latency_s` | Average per-run latency in seconds. |
389
430
  | `provider_type` | Model vendor + gateway label (e.g. `ANTHROPIC`, `BEDROCK/ANTHROPIC`, `AZURE/OPENAI`). Stored in Langfuse trace metadata and tags. |
431
+
432
+ Score names carry no K; K and the gate are on the dataset-run metadata as `eval_k` and `eval_gate`.
433
+ `agentic_visualization` also still writes its historic `pass_at_{K}` / `pass_power_{K}` pair.
@@ -1,7 +1,7 @@
1
1
  # (C) 2026 GoodData Corporation
2
2
  [project]
3
3
  name = "gooddata-eval"
4
- version = "1.74.1.dev2"
4
+ version = "1.74.1.dev3"
5
5
  description = "Evaluate the GoodData AI agent against your own questions and models."
6
6
  readme = "README.md"
7
7
  license = "MIT"
@@ -11,7 +11,7 @@ authors = [
11
11
  keywords = ["gooddata", "ai", "evaluation", "llm", "analytics", "cli"]
12
12
  requires-python = ">=3.10"
13
13
  dependencies = [
14
- "gooddata-sdk~=1.74.1.dev2",
14
+ "gooddata-sdk~=1.74.1.dev3",
15
15
  "httpx>=0.27,<1.0",
16
16
  "orjson>=3.9.15,<4.0.0",
17
17
  "pydantic>=2.6,<3.0",
@@ -8,6 +8,7 @@ import time
8
8
  from concurrent.futures import ThreadPoolExecutor, as_completed
9
9
  from typing import Any, TypedDict
10
10
 
11
+ from gooddata_eval.core.agentic._gate import DEFAULT_GATE, EvalGate, normalize_gate
11
12
  from gooddata_eval.core.agentic._langfuse import make_langfuse_client
12
13
  from gooddata_eval.core.agentic._trace_linker import BackgroundTraceLinker, SubmitTraceLink, run_trace_link_inline
13
14
  from gooddata_eval.core.agentic.alert_skill import evaluate_agentic_alert_skill
@@ -48,6 +49,13 @@ AGENTIC_TEST_KINDS = frozenset(
48
49
  )
49
50
 
50
51
 
52
+ # Agentic kinds that no gate applies to: they drive their fixture exactly once, so there is
53
+ # no K to take pass@K or pass^K over. Named here rather than inline in _dispatch_agentic so
54
+ # the CLI can refuse --gate power for a dataset containing one instead of labelling the whole
55
+ # report `power` when part of it was never gated.
56
+ UNGATED_AGENTIC_TEST_KINDS = frozenset({"agentic_conversation"})
57
+
58
+
51
59
  # Kinds cleared to run several at a time. An EXPLICIT allowlist, not a subtraction: nothing
52
60
  # in this package can prove a kind is read-only, because the mutation happens server-side in
53
61
  # whichever tools the agent decides to call. So each entry here is a reviewed judgement, and
@@ -128,9 +136,13 @@ def _dispatch_agentic(
128
136
  reasoning_effort: ReasoningEffort | None = None,
129
137
  agent_id: str | None = None,
130
138
  submit_trace_link: SubmitTraceLink = run_trace_link_inline,
139
+ gate: EvalGate = DEFAULT_GATE,
131
140
  ) -> AgenticEvalOutcome:
132
141
  """Call the appropriate evaluate_agentic_* function for the item's test_kind.
133
142
 
143
+ `gate` reaches every kind except those in UNGATED_AGENTIC_TEST_KINDS, which have no K
144
+ to gate over; the CLI refuses --gate power for a dataset containing one.
145
+
134
146
  Every evaluate_agentic_* function returns an AgenticEvalOutcome (reasoning_steps,
135
147
  conversation_id, response_id, detail) on success and attaches the same four attributes
136
148
  to its raised *AssertionError on failure -- no kind is exempt.
@@ -155,6 +167,7 @@ def _dispatch_agentic(
155
167
  question=item.question,
156
168
  expected_outputs=_parse_visualization_expected(eo),
157
169
  k=k,
170
+ gate=gate,
158
171
  agent_id=agent_id,
159
172
  **lf_kw,
160
173
  )
@@ -166,6 +179,7 @@ def _dispatch_agentic(
166
179
  question=item.question,
167
180
  expected_output=eo if isinstance(eo, (dict, list)) else {},
168
181
  k=k,
182
+ gate=gate,
169
183
  agent_id=agent_id,
170
184
  **lf_kw,
171
185
  )
@@ -177,6 +191,7 @@ def _dispatch_agentic(
177
191
  question=item.question,
178
192
  expected_output=eo if isinstance(eo, dict) else {},
179
193
  k=k,
194
+ gate=gate,
180
195
  agent_id=agent_id,
181
196
  **lf_kw,
182
197
  )
@@ -191,6 +206,7 @@ def _dispatch_agentic(
191
206
  question=item.question,
192
207
  expected_tool_call=expected_args,
193
208
  k=k,
209
+ gate=gate,
194
210
  agent_id=agent_id,
195
211
  **lf_kw,
196
212
  )
@@ -202,6 +218,7 @@ def _dispatch_agentic(
202
218
  question=item.question,
203
219
  expected_output=eo if isinstance(eo, str) else str(eo),
204
220
  k=k,
221
+ gate=gate,
205
222
  agent_id=agent_id,
206
223
  user_context=item.user_context,
207
224
  **lf_kw,
@@ -214,6 +231,7 @@ def _dispatch_agentic(
214
231
  question=item.question,
215
232
  expected_output=eo if isinstance(eo, str) else str(eo),
216
233
  k=k,
234
+ gate=gate,
217
235
  agent_id=agent_id,
218
236
  **lf_kw,
219
237
  )
@@ -225,6 +243,7 @@ def _dispatch_agentic(
225
243
  question=item.question,
226
244
  expected_output=eo if isinstance(eo, dict) else {},
227
245
  k=k,
246
+ gate=gate,
228
247
  agent_id=agent_id,
229
248
  **lf_kw,
230
249
  )
@@ -291,6 +310,7 @@ def run_agentic_items(
291
310
  on_item_done: Any = None,
292
311
  agent_id: str | None = None,
293
312
  concurrency: int = 1,
313
+ gate: EvalGate = DEFAULT_GATE,
294
314
  ) -> EvalReport:
295
315
  """Run agentic items through evaluate_agentic_* and return an EvalReport.
296
316
 
@@ -303,7 +323,7 @@ def run_agentic_items(
303
323
  """
304
324
  langfuse = make_langfuse_client() if use_langfuse else None
305
325
 
306
- report = EvalReport(model=model_version)
326
+ report = EvalReport(model=model_version, gate=normalize_gate(gate))
307
327
  total = len(items)
308
328
  # Trace linking runs here rather than inside each evaluate_agentic_*, so an item's
309
329
  # Langfuse poll overlaps the NEXT item's agent call instead of extending its own
@@ -325,6 +345,9 @@ def run_agentic_items(
325
345
  test_kind=item.test_kind,
326
346
  question=item.question,
327
347
  )
348
+ # None, not False, for the kinds _dispatch_agentic passes no gate to: ItemReport.passed
349
+ # then falls back to pass_at_k and gate_passed keeps meaning "a gate ran".
350
+ gated = item.test_kind not in UNGATED_AGENTIC_TEST_KINDS
328
351
  t0 = time.perf_counter()
329
352
  try:
330
353
  outcome = _dispatch_agentic(
@@ -339,6 +362,7 @@ def run_agentic_items(
339
362
  reasoning_effort,
340
363
  agent_id,
341
364
  submit_trace_link=linker.submit,
365
+ gate=gate,
342
366
  )
343
367
  if isinstance(outcome, AgenticEvalOutcome):
344
368
  reasoning_steps = outcome.reasoning_steps
@@ -347,6 +371,8 @@ def run_agentic_items(
347
371
  detail = outcome.detail
348
372
  else:
349
373
  reasoning_steps, conversation_id, response_id, detail = outcome, None, None, {}
374
+ item_report.gate_passed = True if gated else None
375
+ # Whichever gate decided the item, clearing it means at least one run passed.
350
376
  item_report.pass_at_k = True
351
377
  item_report.runs = k
352
378
  item_report.reasoning_steps = reasoning_steps or []
@@ -356,7 +382,7 @@ def run_agentic_items(
356
382
  _apply_timings(item_report, getattr(outcome, "timings", None))
357
383
  _apply_run_counts(item_report, outcome)
358
384
  except AssertionError as exc:
359
- item_report.pass_at_k = False
385
+ item_report.gate_passed = False if gated else None
360
386
  item_report.runs = k
361
387
  item_report.reasoning_steps = getattr(exc, "reasoning_steps", None) or []
362
388
  item_report.conversation_id = getattr(exc, "conversation_id", None)
@@ -364,6 +390,10 @@ def run_agentic_items(
364
390
  item_report.best_detail = getattr(exc, "detail", None) or {}
365
391
  _apply_timings(item_report, getattr(exc, "timings", None))
366
392
  _apply_run_counts(item_report, exc)
393
+ # Read off the counts, not off the gate: pass^K fails items where runs did pass,
394
+ # and reporting those as pass_at_k False would contradict the Langfuse score of
395
+ # the same name. Kinds that report no count read as 0, i.e. a clean failure.
396
+ item_report.pass_at_k = item_report.runs_passed > 0
367
397
  print(f"[agentic] {item.id} FAIL: {exc}", flush=True)
368
398
  except Exception as exc:
369
399
  item_report.error = f"{type(exc).__name__}: {exc}"
@@ -14,11 +14,20 @@ from gooddata_api_client.exceptions import ApiException
14
14
  from rich.console import Console
15
15
  from rich.table import Table
16
16
 
17
- from gooddata_eval.cli.agentic_runner import AGENTIC_TEST_KINDS, run_agentic_items
17
+ from gooddata_eval.cli.agentic_runner import AGENTIC_TEST_KINDS, UNGATED_AGENTIC_TEST_KINDS, run_agentic_items
18
18
  from gooddata_eval.core.chat.sse_client import ChatClient
19
- from gooddata_eval.core.config import DEFAULT_JUDGE_MODEL, JUDGE_MODEL_ENV_VAR, ReasoningEffort, RunConfig
19
+ from gooddata_eval.core.config import (
20
+ DEFAULT_GATE,
21
+ DEFAULT_JUDGE_MODEL,
22
+ JUDGE_MODEL_ENV_VAR,
23
+ EvalGate,
24
+ ReasoningEffort,
25
+ RunConfig,
26
+ normalize_gate,
27
+ )
20
28
  from gooddata_eval.core.connection import ConnectionError_, resolve_connection
21
29
  from gooddata_eval.core.dataset.local import load_local_dataset
30
+ from gooddata_eval.core.evaluators import supported_test_kinds
22
31
  from gooddata_eval.core.langfuse.sink import LangfuseSink
23
32
  from gooddata_eval.core.models import ChatResult, DatasetItem
24
33
  from gooddata_eval.core.reporting.console import render_comparison, render_console
@@ -91,7 +100,15 @@ def _build_parser() -> argparse.ArgumentParser:
91
100
  "Default: workspace's current active model."
92
101
  ),
93
102
  )
94
- run.add_argument("--runs", type=int, default=2, help="Independent runs per item (pass@K). Default 2.")
103
+ run.add_argument("--runs", type=int, default=2, help="Independent runs per item. Default 2.")
104
+ run.add_argument(
105
+ "--gate",
106
+ choices=get_args(EvalGate),
107
+ default=DEFAULT_GATE,
108
+ help="Which verdict decides an item: 'any' = pass@K (a run passing is enough, the "
109
+ "default and historic behaviour), 'power' = pass^K (every run must pass, so the verdict "
110
+ "measures stability). Identical at --runs 1. Agentic kinds only.",
111
+ )
95
112
  run.add_argument(
96
113
  "--concurrency",
97
114
  type=int,
@@ -180,16 +197,46 @@ def _apply_timer_flag(enabled: bool) -> None:
180
197
  os.environ[TIMERS_ENV_VAR] = "1"
181
198
 
182
199
 
200
+ def _reject_power_gate_on_ungated_items(config: RunConfig, items: list) -> None:
201
+ """Refuse a pass^K request the run cannot honour for every item.
202
+
203
+ Two kinds of item are never gated: everything on the non-agentic path, because
204
+ `run_items` has no gate and always decides on pass@K, and agentic_conversation, which
205
+ drives its fixture once whatever --runs says and so has no K to gate over. Running a
206
+ mixed dataset anyway would decide part of it under each rule and label the whole report
207
+ `power`. test_kind is resolved per item, so a dataset does not have to be homogeneous.
208
+
209
+ Kinds no evaluator supports are not counted: those items are skipped rather than
210
+ decided, so refusing on them would make --gate power fail where --gate any runs.
211
+ """
212
+ if normalize_gate(config.gate) != "power":
213
+ return
214
+ supported = supported_test_kinds()
215
+ ungated = [
216
+ i
217
+ for i in items
218
+ if i.test_kind in UNGATED_AGENTIC_TEST_KINDS
219
+ or (i.test_kind not in AGENTIC_TEST_KINDS and i.test_kind in supported)
220
+ ]
221
+ if not ungated:
222
+ return
223
+ kinds = sorted({i.test_kind for i in ungated})
224
+ raise ValueError(
225
+ f"--gate power applies to kinds that repeat K runs, but this dataset has {len(ungated)} "
226
+ f"item(s) of kind {kinds}, which are always decided on pass@K. Run them separately, or "
227
+ f"use --gate any."
228
+ )
229
+
230
+
183
231
  def _warn_if_local_dataset_cannot_link(config: RunConfig, agentic_items: list) -> None:
184
- """Say up front that dataset-run assembly will fail, rather than after the run.
232
+ """Say up front that experiment assembly will fail, rather than after the run.
185
233
 
186
234
  --langfuse is refused outright with a local dataset because local item ids cannot be
187
235
  linked. But every evaluate_agentic_* falls back to try_make_langfuse_client() when the
188
- caller passes none, so with LANGFUSE_* exported the linking runs anyway and each
189
- conversation earns a 404 from dataset-run-items -- arriving in a block at the very end
190
- of the run, long after the flag that would have prevented it could be changed. The
191
- fallback is deliberate (direct library and tavern callers rely on it), so this warns
192
- instead of disabling it.
236
+ caller passes none, so with LANGFUSE_* exported the linking runs anyway and every
237
+ dataset-item lookup 404s -- arriving in a block at the very end of the run, long after
238
+ the flag that would have prevented it could be changed. The fallback is deliberate
239
+ (direct library and tavern callers rely on it), so this warns instead of disabling it.
193
240
  """
194
241
  from gooddata_eval.core.agentic._langfuse import SKIP_ENV_VAR, langfuse_credentials_present # noqa: PLC0415
195
242
  from gooddata_eval.core.config import env_flag # noqa: PLC0415
@@ -201,7 +248,7 @@ def _warn_if_local_dataset_cannot_link(config: RunConfig, agentic_items: list) -
201
248
  print(
202
249
  f"warning: --dataset is a local folder, so its item ids are not Langfuse dataset item ids. "
203
250
  f"Traces will be found and scored, but the per-run grouping that makes models comparable "
204
- f"cannot be created and each conversation will report a 404 from dataset-run-items. "
251
+ f"cannot be created and each conversation will report that its item does not exist in Langfuse. "
205
252
  f"Use --langfuse-dataset for comparable runs, or set {SKIP_ENV_VAR}=1 to skip trace linking.",
206
253
  file=sys.stderr,
207
254
  )
@@ -253,7 +300,7 @@ def _make_progress_callbacks(console: Console):
253
300
  tag = "[yellow]SKIP[/yellow]"
254
301
  elif report.error:
255
302
  tag = "[red]ERR [/red]"
256
- elif report.pass_at_k:
303
+ elif report.passed:
257
304
  tag = "[green]PASS[/green]"
258
305
  else:
259
306
  tag = "[red]FAIL[/red]"
@@ -343,6 +390,7 @@ def _run(config: RunConfig) -> int:
343
390
  items = _load_dataset(config)
344
391
  agentic_items = [i for i in items if i.test_kind in AGENTIC_TEST_KINDS]
345
392
  non_agentic_items = [i for i in items if i.test_kind not in AGENTIC_TEST_KINDS]
393
+ _reject_power_gate_on_ungated_items(config, items)
346
394
  _warn_if_local_dataset_cannot_link(config, agentic_items)
347
395
  models = config.models or []
348
396
  run_ts = datetime.now(timezone.utc).strftime("%Y-%m-%d-%H-%M")
@@ -417,6 +465,7 @@ def _run(config: RunConfig) -> int:
417
465
  token=config.token,
418
466
  workspace_id=config.workspace_id,
419
467
  k=config.runs,
468
+ gate=config.gate,
420
469
  model_version=resolved.model_id,
421
470
  reasoning_effort=config.reasoning_effort,
422
471
  use_langfuse=config.log_to_langfuse,
@@ -466,6 +515,7 @@ def _run(config: RunConfig) -> int:
466
515
  provider_name=resolved.provider_name or resolved.provider_id,
467
516
  provider_type=resolved.provider_type,
468
517
  workspace_id=config.workspace_id,
518
+ gate=config.gate,
469
519
  )
470
520
  if agentic_report is not None:
471
521
  report.items.extend(agentic_report.items)
@@ -530,6 +580,7 @@ def main(argv: list[str] | None = None) -> int:
530
580
  kind=args.kind,
531
581
  preserve_failed=args.preserve_failed,
532
582
  reasoning_effort=args.reasoning_effort,
583
+ gate=normalize_gate(args.gate),
533
584
  agent_id=args.agent_id or os.environ.get("GD_EVAL_AGENT_ID"),
534
585
  )
535
586
  return _run(config)