gooddata-eval 1.74.0__tar.gz → 1.74.1.dev1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (105) hide show
  1. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/PKG-INFO +111 -5
  2. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/README.md +108 -2
  3. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/pyproject.toml +3 -3
  4. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/src/gooddata_eval/cli/agentic_runner.py +188 -4
  5. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/src/gooddata_eval/cli/main.py +70 -2
  6. gooddata_eval-1.74.1.dev1/src/gooddata_eval/core/_output.py +18 -0
  7. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/src/gooddata_eval/core/agentic/_langfuse.py +185 -29
  8. gooddata_eval-1.74.1.dev1/src/gooddata_eval/core/agentic/_trace_linker.py +302 -0
  9. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/src/gooddata_eval/core/agentic/alert_skill.py +88 -87
  10. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/src/gooddata_eval/core/agentic/conversation.py +79 -66
  11. gooddata_eval-1.74.1.dev1/src/gooddata_eval/core/agentic/general_question.py +332 -0
  12. gooddata_eval-1.74.1.dev1/src/gooddata_eval/core/agentic/guardrail.py +305 -0
  13. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/src/gooddata_eval/core/agentic/kda_skill.py +95 -92
  14. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/src/gooddata_eval/core/agentic/metric_skill.py +107 -69
  15. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/src/gooddata_eval/core/agentic/search_tool.py +65 -63
  16. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/src/gooddata_eval/core/agentic/visualization.py +83 -81
  17. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/src/gooddata_eval/core/chat/sse_client.py +8 -2
  18. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/src/gooddata_eval/core/config.py +25 -0
  19. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/src/gooddata_eval/core/dataset/langfuse_source.py +53 -22
  20. gooddata_eval-1.74.1.dev1/src/gooddata_eval/core/evaluators/_llm_judge.py +311 -0
  21. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/src/gooddata_eval/core/evaluators/general_question.py +15 -13
  22. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/src/gooddata_eval/core/evaluators/guardrail.py +20 -17
  23. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/src/gooddata_eval/core/evaluators/summary.py +56 -13
  24. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/src/gooddata_eval/core/models.py +38 -0
  25. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/src/gooddata_eval/core/reporting/console.py +34 -10
  26. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/src/gooddata_eval/core/reporting/json_report.py +26 -2
  27. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/src/gooddata_eval/core/runner.py +68 -2
  28. gooddata_eval-1.74.1.dev1/src/gooddata_eval/core/timing.py +74 -0
  29. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/tests/test_agentic_alert_skill.py +49 -59
  30. gooddata_eval-1.74.1.dev1/tests/test_agentic_general_question.py +689 -0
  31. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/tests/test_agentic_guardrail.py +108 -82
  32. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/tests/test_agentic_kda_skill.py +120 -208
  33. gooddata_eval-1.74.1.dev1/tests/test_agentic_langfuse_trace.py +552 -0
  34. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/tests/test_agentic_metric_skill.py +178 -54
  35. gooddata_eval-1.74.1.dev1/tests/test_agentic_runner.py +768 -0
  36. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/tests/test_cli.py +188 -0
  37. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/tests/test_connection.py +5 -0
  38. gooddata_eval-1.74.1.dev1/tests/test_langfuse_source.py +198 -0
  39. gooddata_eval-1.74.1.dev1/tests/test_llm_judge.py +616 -0
  40. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/tests/test_models.py +27 -0
  41. gooddata_eval-1.74.1.dev1/tests/test_reporting.py +444 -0
  42. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/tests/test_runner.py +59 -0
  43. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/tests/test_sse_client.py +54 -0
  44. gooddata_eval-1.74.1.dev1/tests/test_summary_evaluator.py +192 -0
  45. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/tests/test_text_evaluators.py +32 -0
  46. gooddata_eval-1.74.1.dev1/tests/test_timing.py +93 -0
  47. gooddata_eval-1.74.1.dev1/tests/test_trace_linker.py +568 -0
  48. gooddata_eval-1.74.0/src/gooddata_eval/core/agentic/general_question.py +0 -264
  49. gooddata_eval-1.74.0/src/gooddata_eval/core/agentic/guardrail.py +0 -268
  50. gooddata_eval-1.74.0/src/gooddata_eval/core/evaluators/_llm_judge.py +0 -66
  51. gooddata_eval-1.74.0/tests/test_agentic_general_question.py +0 -210
  52. gooddata_eval-1.74.0/tests/test_agentic_langfuse_trace.py +0 -26
  53. gooddata_eval-1.74.0/tests/test_agentic_runner.py +0 -220
  54. gooddata_eval-1.74.0/tests/test_langfuse_source.py +0 -105
  55. gooddata_eval-1.74.0/tests/test_llm_judge.py +0 -45
  56. gooddata_eval-1.74.0/tests/test_reporting.py +0 -194
  57. gooddata_eval-1.74.0/tests/test_summary_evaluator.py +0 -87
  58. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/.gitignore +0 -0
  59. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/LICENSE.txt +0 -0
  60. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/Makefile +0 -0
  61. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/src/gooddata_eval/__init__.py +0 -0
  62. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/src/gooddata_eval/_version.py +0 -0
  63. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/src/gooddata_eval/cli/__init__.py +0 -0
  64. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/src/gooddata_eval/core/__init__.py +0 -0
  65. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/src/gooddata_eval/core/agentic/__init__.py +0 -0
  66. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/src/gooddata_eval/core/agentic/_catalog.py +0 -0
  67. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/src/gooddata_eval/core/chat/__init__.py +0 -0
  68. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/src/gooddata_eval/core/connection.py +0 -0
  69. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/src/gooddata_eval/core/dataset/__init__.py +0 -0
  70. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/src/gooddata_eval/core/dataset/local.py +0 -0
  71. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/src/gooddata_eval/core/evaluators/__init__.py +0 -0
  72. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/src/gooddata_eval/core/evaluators/_deep_subset.py +0 -0
  73. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/src/gooddata_eval/core/evaluators/_text_utils.py +0 -0
  74. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/src/gooddata_eval/core/evaluators/alert_skill.py +0 -0
  75. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/src/gooddata_eval/core/evaluators/base.py +0 -0
  76. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/src/gooddata_eval/core/evaluators/metric_skill.py +0 -0
  77. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/src/gooddata_eval/core/evaluators/search_tool.py +0 -0
  78. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/src/gooddata_eval/core/evaluators/visualization.py +0 -0
  79. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/src/gooddata_eval/core/langfuse/__init__.py +0 -0
  80. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/src/gooddata_eval/core/langfuse/sink.py +0 -0
  81. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/src/gooddata_eval/core/reporting/__init__.py +0 -0
  82. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/src/gooddata_eval/core/scoring.py +0 -0
  83. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/src/gooddata_eval/core/summary/__init__.py +0 -0
  84. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/src/gooddata_eval/core/summary/http_client.py +0 -0
  85. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/src/gooddata_eval/core/workspace.py +0 -0
  86. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/tests/__init__.py +0 -0
  87. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/tests/conftest.py +0 -0
  88. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/tests/fixtures/sample_dataset/metric_skill_create.json +0 -0
  89. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/tests/fixtures/sample_dataset/visualization_revenue.json +0 -0
  90. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/tests/fixtures/sse_visualization_stream.txt +0 -0
  91. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/tests/test_agentic_conversation.py +0 -0
  92. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/tests/test_agentic_run_context.py +0 -0
  93. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/tests/test_agentic_search_tool.py +0 -0
  94. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/tests/test_agentic_visualization.py +0 -0
  95. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/tests/test_alert_skill_evaluator.py +0 -0
  96. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/tests/test_deep_subset.py +0 -0
  97. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/tests/test_langfuse_sink.py +0 -0
  98. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/tests/test_local_loader.py +0 -0
  99. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/tests/test_metric_skill_evaluator.py +0 -0
  100. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/tests/test_scoring.py +0 -0
  101. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/tests/test_search_tool_evaluator.py +0 -0
  102. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/tests/test_summary_client.py +0 -0
  103. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/tests/test_visualization_evaluator.py +0 -0
  104. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/tests/test_workspace.py +0 -0
  105. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/tox.ini +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: gooddata-eval
3
- Version: 1.74.0
3
+ Version: 1.74.1.dev1
4
4
  Summary: Evaluate the GoodData AI agent against your own questions and models.
5
5
  Project-URL: Source, https://github.com/gooddata/gooddata-python-sdk
6
6
  Author-email: GoodData <support@gooddata.com>
@@ -17,13 +17,13 @@ Classifier: Topic :: Scientific/Engineering
17
17
  Classifier: Topic :: Software Development
18
18
  Classifier: Typing :: Typed
19
19
  Requires-Python: >=3.10
20
- Requires-Dist: gooddata-sdk~=1.74.0
20
+ Requires-Dist: gooddata-sdk~=1.74.1.dev1
21
21
  Requires-Dist: httpx<1.0,>=0.27
22
22
  Requires-Dist: orjson<4.0.0,>=3.9.15
23
23
  Requires-Dist: pydantic<3.0,>=2.6
24
24
  Requires-Dist: rich<15.0,>=13.0
25
25
  Provides-Extra: llm-judge
26
- Requires-Dist: openai<2.0,>=1.40; extra == 'llm-judge'
26
+ Requires-Dist: openai<2.0,>=1.45; extra == 'llm-judge'
27
27
  Description-Content-Type: text/markdown
28
28
 
29
29
  # gooddata-eval
@@ -142,6 +142,7 @@ gd-eval run \
142
142
  |---|---|
143
143
  | `--dataset PATH` | Flat folder of JSON files — one question per file. |
144
144
  | `--langfuse-dataset NAME` | Pull items by name from a Langfuse dataset. Requires `LANGFUSE_PUBLIC_KEY`, `LANGFUSE_SECRET_KEY`, `LANGFUSE_HOST`. |
145
+ | `--kind TEST_KIND` | Fallback `test_kind` for dataset items that do not embed one. Defaults to `visualization`; use e.g. `agentic_metric_skill` for multi-turn agentic evaluation. Items that declare their own `test_kind` ignore this. |
145
146
 
146
147
  #### Model selection
147
148
 
@@ -154,21 +155,62 @@ gd-eval run \
154
155
  | Flag | Default | Description |
155
156
  |---|---|---|
156
157
  | `--runs K` | `2` | Independent runs per item (pass@K). An item passes if any run passes. |
157
- | `--concurrency K` | `1` | Number of items evaluated concurrently. `1` = sequential (default). Increase to load-test the agent under simultaneous requests. Progress output interleaves when K > 1. |
158
+ | `--concurrency K` | `1` | Number of items evaluated concurrently. `1` = sequential (default). Increase to load-test the agent under simultaneous requests — see *Concurrency and workspace safety* below. |
159
+ | `--judge-model MODEL` | `gpt-4o` | Model used for LLM-as-judge scoring — `agentic_general_question`, `agentic_guardrail`, `general_question`, `guardrail` and `dashboard_summary`. Also settable via `GD_EVAL_JUDGE_MODEL`. Two things to weigh before changing it: the gpt-5 family rejects `temperature=0`, so verdicts stop being reproducible (the run warns when this happens); and choosing the same model the agent runs means the judge grades its own family's output. |
158
160
  | `--reasoning-effort LEVEL` | server default | `LOW`, `MEDIUM` or `HIGH`, sent as `options.reasoningEffort` on every chat message. Requires the `enableGenAiReasoningEffort` feature flag on the target organization — without it the server ignores the value. Applies to chat items only; `dashboard_summary` items go through the summary endpoint, which has no such option. |
159
161
 
162
+ **Concurrency and workspace safety.** Agentic kinds that create workspace objects
163
+ (`agentic_metric_skill`, `agentic_alert_skill`, `agentic_conversation`, `agentic_kda_skill`) always run one at a
164
+ time whatever `--concurrency` says — a metric or alert created and dropped mid-run would otherwise be visible to
165
+ another item reading the same catalog. **That protection is for the agentic kinds only:** the single-turn
166
+ `metric_skill` and `alert_skill` kinds are still fanned out and the agent performs the same server-side writes on
167
+ that path, so avoid raising `--concurrency` on a dataset of those against a shared workspace. Progress output
168
+ interleaves when K > 1, and per-item latencies rise, so they stop being clean single-request measurements.
169
+
160
170
  #### Output
161
171
 
162
172
  | Flag | Description |
163
173
  |---|---|
164
174
  | `--json PATH` | Write a JSON report to this path. Always uses the nested `{models, runs, comparison}` shape even for a single model. |
165
175
  | `--quiet` | Suppress per-item progress. Per-model result tables and the comparison summary are still printed. |
176
+ | `--preserve-failed` | Keep failed conversations on the server instead of deleting them, so they can be inspected afterwards. Applies to the single-turn chat path; agentic kinds manage their own conversation lifecycle. |
177
+ | `--timers` | Print per-turn `[timer]` diagnostics — GoodData response, judge, and simulated-user seconds as they happen. Off by default: an 18-item `--runs 2` run emits ~72 lines and buries the progress output. The same measurements are always in the JSON report's `latency_breakdown_s`, so this only adds a live view. Also settable via `GD_EVAL_TIMERS=1`. |
166
178
 
167
179
  #### Langfuse sink
168
180
 
169
181
  | Flag | Description |
170
182
  |---|---|
171
- | `--langfuse` | Log scores and traces to Langfuse after each item. Requires `--langfuse-dataset`. Creates one named experiment run per model (`gd-eval-{timestamp}-{model}`, suffixed `-effort-{level}` when `--reasoning-effort` is set so runs differing only by effort stay separate). Requires `LANGFUSE_PUBLIC_KEY`, `LANGFUSE_SECRET_KEY`, `LANGFUSE_HOST`. |
183
+ | `--langfuse` | Log scores and traces to Langfuse after each item. Requires `--langfuse-dataset`. Names each experiment run `{dataset_name}_{timestamp}_{model}`, suffixed `_effort-{level}` when `--reasoning-effort` is set (so runs differing only by effort stay separate) and `_run{N}` per run when `--runs` > 1 — e.g. `general_question_2026-09-02-11-13_gpt-5.2_run0`. Requires `LANGFUSE_PUBLIC_KEY`, `LANGFUSE_SECRET_KEY`, `LANGFUSE_HOST`. |
184
+
185
+ Set `TAVERN_E2E_SKIP_TRACE_LINK=1` to skip trace lookup entirely (scores are then orphaned; the run says so).
186
+
187
+ **A local `--dataset` cannot be attached to a Langfuse run.** `--langfuse` is refused alongside `--dataset`
188
+ because a local folder's item ids are not Langfuse dataset item ids. But trace linking does not depend on that
189
+ flag — each `evaluate_agentic_*` builds its own client whenever `LANGFUSE_PUBLIC_KEY`/`LANGFUSE_SECRET_KEY` are
190
+ exported — so a local run still finds its traces and writes its scores onto them, and only the per-run grouping
191
+ fails, with one `404 from dataset-run-items` reported per run. The run warns about this before it starts. Use
192
+ `--langfuse-dataset` when you want runs that are comparable across models, or `TAVERN_E2E_SKIP_TRACE_LINK=1` to
193
+ skip linking altogether.
194
+
195
+ **When trace linking happens.** Finding a gen-ai trace means polling until Langfuse has ingested it, which is
196
+ lag measured in seconds to minutes. That work produces no verdict — the pass/fail is already decided — so it
197
+ does not run inline per item. Every item's Langfuse block is queued and the whole batch runs *after* the agent
198
+ phase, draining before any report is written. Two consequences worth knowing:
199
+
200
+ - **No item's `latency_s` includes trace linking.** Its cost is reported separately as
201
+ `latency_breakdown_s.langfuse_s`, and the run prints
202
+ `[langfuse] trace linking finished in Xs for N item(s); slowest Ys`. If `slowest` approaches the **120s**
203
+ batched retry budget, links are timing out and scores are being orphaned — look for
204
+ `[langfuse] WARNING: no trace found for conversation ...`.
205
+ - **The budget depends on who is waiting.** 120s is affordable only because the batch blocks nobody. A direct
206
+ library caller (`evaluate_agentic_*` without a `submit_trace_link`) polls inline, on its own critical path, and
207
+ gets **35s** instead — the same cost as before batching existed, so no inline caller pays for a budget raised
208
+ on the CLI's behalf. Either way a trace that is already ingested costs nothing: the loop looks before it sleeps.
209
+ Scores are always final before the command exits — the run blocks on the batch. Interrupting with Ctrl-C drops
210
+ whatever is still queued rather than making you wait it out: both the queued trace links and, under
211
+ `--concurrency`, the items that have not started. The handful of items already in flight still have to finish —
212
+ worker threads are joined at exit and an in-progress agent call cannot be cancelled — so expect to wait up to one
213
+ `--concurrency`-wide wave, not the rest of the dataset.
172
214
 
173
215
  ### JSON report shape
174
216
 
@@ -190,6 +232,70 @@ The JSON report always uses the nested multi-model shape:
190
232
 
191
233
  Winner is selected by **pass rate → quality score → latency** (lower latency wins all-equal ties).
192
234
 
235
+ Each item reports **how many of its runs passed**, not only whether one did:
236
+
237
+ ```json
238
+ "runs": 5, "runs_passed": 4, "pass_at_k": true, "pass_power_k": false
239
+ ```
240
+
241
+ `pass_at_k` is "did any run pass" and is what `passed` counts. `runs_passed` is the fact that separates a
242
+ reliable item from a coin-flip — without it a 5/5 item and a 1/5 item are identical in every field, because
243
+ `quality_score` is derived from the best run alone. `pass_power_k` is true only when every run passed, and the
244
+ run summary carries `passed_all_runs` beside `passed`; a large gap between the two means the model is
245
+ inconsistent rather than wrong. The console shows `4/5 runs passed` in `Notes` for a non-unanimous pass and
246
+ stays quiet for a unanimous one, and its summary line reads `3/4 passed, 1 on every run`.
247
+
248
+ `runs` is what the item actually ran, which is not always the requested `--runs`: `agentic_conversation` takes
249
+ no K and drives its fixture exactly once.
250
+
251
+ Each item additionally carries a per-phase breakdown:
252
+
253
+ ```json
254
+ "latency_breakdown_s": {
255
+ "agent_s": 4.02, // GoodData's own response time — the system under test
256
+ "judge_s": 1.31, // LLM-as-judge scoring, post-hoc
257
+ "simulated_user_s": 0.0, // our simulated user composing the next turn (multi-turn kinds)
258
+ "langfuse_s": 5.70 // trace lookup + score writing, off the critical path
259
+ }
260
+ ```
261
+
262
+ Every item reports `runs_ungraded` beside `runs_passed`: runs the agent answered but the LLM judge returned
263
+ nothing readable for. Agentic items also list them as `unscored_runs` / `judge_errors` in their `detail`, and a
264
+ `dashboard_summary` item carries `ungraded_criteria`. Such a run — or, for `dashboard_summary`, such a
265
+ criterion — is excluded from pass@K and from the quality score rather than counted as a failure: scoring it 0
266
+ would be indistinguishable from the judge genuinely failing the answer, which is the confusion
267
+ `JudgeResponseError` exists to end. `pass@K` still holds on the runs that *were* graded, so an item can pass with
268
+ `runs_ungraded` set; `pass^K` cannot, because a run nobody graded leaves "all K passed" unverified. For
269
+ `dashboard_summary` an ungraded `must_include` / `must_not_include` criterion likewise cannot carry a pass —
270
+ "the judge could not tell" is not evidence the fact is present — while an ungraded `rubric` line only narrows
271
+ the quality score. When *no* run could be graded (for `dashboard_summary`: no gating criterion on any run) the
272
+ item errors instead of reporting failures. A non-zero count means pass@K was computed over fewer runs than
273
+ `--runs` asked for, so treat the result as weaker evidence and check the judge (`GD_EVAL_JUDGE_DIAGNOSTICS=1`,
274
+ or raise `JUDGE_MAX_COMPLETION_TOKENS` if the cause is `finish_reason=length`). The console says so in `Notes`
275
+ (`1 run(s) ungraded`, `2 criterion(s) ungraded`).
276
+
277
+ `agent_s` + `judge_s` + `simulated_user_s` are the instrumented parts of the item's `latency_s`; they do not add
278
+ up to it exactly, because `latency_s` is wall-clock around the whole item and also covers the conversation
279
+ create/delete round trips, SDK construction and any cleanup. `langfuse_s` sits **beside** `latency_s`, never
280
+ inside it, because trace linking runs outside every item's critical path (see above) — summing all four would
281
+ re-inflate exactly what that design removes.
282
+
283
+ A phase that a kind does not have reports `0.0` rather than an invented number, so read the zeroes as "not
284
+ applicable here", not "instant". Today:
285
+
286
+ | Field | Populated by |
287
+ |---|---|
288
+ | `agent_s` | `agentic_general_question`, `agentic_metric_skill` |
289
+ | `judge_s` | `agentic_general_question` only — `agentic_metric_skill` compares MAQL by string, it has no LLM judge |
290
+ | `simulated_user_s` | `agentic_metric_skill` only — `agentic_general_question` is single-turn, it has no simulated user |
291
+ | `langfuse_s` | every agentic kind, but only on the `gd-eval` path and only when Langfuse credentials are present |
292
+
293
+ The other six agentic kinds report `0.0` for the first three. Trace linking itself happens whenever
294
+ `LANGFUSE_PUBLIC_KEY`/`LANGFUSE_SECRET_KEY` are exported, with or without `--langfuse`, because each
295
+ `evaluate_agentic_*` falls back to `try_make_langfuse_client()`. But its *duration* is measured by the CLI
296
+ runner rather than by `evaluate_agentic_*`, so a direct library caller sees `langfuse_s: 0.0` even though its
297
+ linking ran. Pass `TAVERN_E2E_SKIP_TRACE_LINK=1` to opt out of linking altogether.
298
+
193
299
  ---
194
300
 
195
301
  ## `gd-eval models`
@@ -114,6 +114,7 @@ gd-eval run \
114
114
  |---|---|
115
115
  | `--dataset PATH` | Flat folder of JSON files — one question per file. |
116
116
  | `--langfuse-dataset NAME` | Pull items by name from a Langfuse dataset. Requires `LANGFUSE_PUBLIC_KEY`, `LANGFUSE_SECRET_KEY`, `LANGFUSE_HOST`. |
117
+ | `--kind TEST_KIND` | Fallback `test_kind` for dataset items that do not embed one. Defaults to `visualization`; use e.g. `agentic_metric_skill` for multi-turn agentic evaluation. Items that declare their own `test_kind` ignore this. |
117
118
 
118
119
  #### Model selection
119
120
 
@@ -126,21 +127,62 @@ gd-eval run \
126
127
  | Flag | Default | Description |
127
128
  |---|---|---|
128
129
  | `--runs K` | `2` | Independent runs per item (pass@K). An item passes if any run passes. |
129
- | `--concurrency K` | `1` | Number of items evaluated concurrently. `1` = sequential (default). Increase to load-test the agent under simultaneous requests. Progress output interleaves when K > 1. |
130
+ | `--concurrency K` | `1` | Number of items evaluated concurrently. `1` = sequential (default). Increase to load-test the agent under simultaneous requests — see *Concurrency and workspace safety* below. |
131
+ | `--judge-model MODEL` | `gpt-4o` | Model used for LLM-as-judge scoring — `agentic_general_question`, `agentic_guardrail`, `general_question`, `guardrail` and `dashboard_summary`. Also settable via `GD_EVAL_JUDGE_MODEL`. Two things to weigh before changing it: the gpt-5 family rejects `temperature=0`, so verdicts stop being reproducible (the run warns when this happens); and choosing the same model the agent runs means the judge grades its own family's output. |
130
132
  | `--reasoning-effort LEVEL` | server default | `LOW`, `MEDIUM` or `HIGH`, sent as `options.reasoningEffort` on every chat message. Requires the `enableGenAiReasoningEffort` feature flag on the target organization — without it the server ignores the value. Applies to chat items only; `dashboard_summary` items go through the summary endpoint, which has no such option. |
131
133
 
134
+ **Concurrency and workspace safety.** Agentic kinds that create workspace objects
135
+ (`agentic_metric_skill`, `agentic_alert_skill`, `agentic_conversation`, `agentic_kda_skill`) always run one at a
136
+ time whatever `--concurrency` says — a metric or alert created and dropped mid-run would otherwise be visible to
137
+ another item reading the same catalog. **That protection is for the agentic kinds only:** the single-turn
138
+ `metric_skill` and `alert_skill` kinds are still fanned out and the agent performs the same server-side writes on
139
+ that path, so avoid raising `--concurrency` on a dataset of those against a shared workspace. Progress output
140
+ interleaves when K > 1, and per-item latencies rise, so they stop being clean single-request measurements.
141
+
132
142
  #### Output
133
143
 
134
144
  | Flag | Description |
135
145
  |---|---|
136
146
  | `--json PATH` | Write a JSON report to this path. Always uses the nested `{models, runs, comparison}` shape even for a single model. |
137
147
  | `--quiet` | Suppress per-item progress. Per-model result tables and the comparison summary are still printed. |
148
+ | `--preserve-failed` | Keep failed conversations on the server instead of deleting them, so they can be inspected afterwards. Applies to the single-turn chat path; agentic kinds manage their own conversation lifecycle. |
149
+ | `--timers` | Print per-turn `[timer]` diagnostics — GoodData response, judge, and simulated-user seconds as they happen. Off by default: an 18-item `--runs 2` run emits ~72 lines and buries the progress output. The same measurements are always in the JSON report's `latency_breakdown_s`, so this only adds a live view. Also settable via `GD_EVAL_TIMERS=1`. |
138
150
 
139
151
  #### Langfuse sink
140
152
 
141
153
  | Flag | Description |
142
154
  |---|---|
143
- | `--langfuse` | Log scores and traces to Langfuse after each item. Requires `--langfuse-dataset`. Creates one named experiment run per model (`gd-eval-{timestamp}-{model}`, suffixed `-effort-{level}` when `--reasoning-effort` is set so runs differing only by effort stay separate). Requires `LANGFUSE_PUBLIC_KEY`, `LANGFUSE_SECRET_KEY`, `LANGFUSE_HOST`. |
155
+ | `--langfuse` | Log scores and traces to Langfuse after each item. Requires `--langfuse-dataset`. Names each experiment run `{dataset_name}_{timestamp}_{model}`, suffixed `_effort-{level}` when `--reasoning-effort` is set (so runs differing only by effort stay separate) and `_run{N}` per run when `--runs` > 1 — e.g. `general_question_2026-09-02-11-13_gpt-5.2_run0`. Requires `LANGFUSE_PUBLIC_KEY`, `LANGFUSE_SECRET_KEY`, `LANGFUSE_HOST`. |
156
+
157
+ Set `TAVERN_E2E_SKIP_TRACE_LINK=1` to skip trace lookup entirely (scores are then orphaned; the run says so).
158
+
159
+ **A local `--dataset` cannot be attached to a Langfuse run.** `--langfuse` is refused alongside `--dataset`
160
+ because a local folder's item ids are not Langfuse dataset item ids. But trace linking does not depend on that
161
+ flag — each `evaluate_agentic_*` builds its own client whenever `LANGFUSE_PUBLIC_KEY`/`LANGFUSE_SECRET_KEY` are
162
+ exported — so a local run still finds its traces and writes its scores onto them, and only the per-run grouping
163
+ fails, with one `404 from dataset-run-items` reported per run. The run warns about this before it starts. Use
164
+ `--langfuse-dataset` when you want runs that are comparable across models, or `TAVERN_E2E_SKIP_TRACE_LINK=1` to
165
+ skip linking altogether.
166
+
167
+ **When trace linking happens.** Finding a gen-ai trace means polling until Langfuse has ingested it, which is
168
+ lag measured in seconds to minutes. That work produces no verdict — the pass/fail is already decided — so it
169
+ does not run inline per item. Every item's Langfuse block is queued and the whole batch runs *after* the agent
170
+ phase, draining before any report is written. Two consequences worth knowing:
171
+
172
+ - **No item's `latency_s` includes trace linking.** Its cost is reported separately as
173
+ `latency_breakdown_s.langfuse_s`, and the run prints
174
+ `[langfuse] trace linking finished in Xs for N item(s); slowest Ys`. If `slowest` approaches the **120s**
175
+ batched retry budget, links are timing out and scores are being orphaned — look for
176
+ `[langfuse] WARNING: no trace found for conversation ...`.
177
+ - **The budget depends on who is waiting.** 120s is affordable only because the batch blocks nobody. A direct
178
+ library caller (`evaluate_agentic_*` without a `submit_trace_link`) polls inline, on its own critical path, and
179
+ gets **35s** instead — the same cost as before batching existed, so no inline caller pays for a budget raised
180
+ on the CLI's behalf. Either way a trace that is already ingested costs nothing: the loop looks before it sleeps.
181
+ Scores are always final before the command exits — the run blocks on the batch. Interrupting with Ctrl-C drops
182
+ whatever is still queued rather than making you wait it out: both the queued trace links and, under
183
+ `--concurrency`, the items that have not started. The handful of items already in flight still have to finish —
184
+ worker threads are joined at exit and an in-progress agent call cannot be cancelled — so expect to wait up to one
185
+ `--concurrency`-wide wave, not the rest of the dataset.
144
186
 
145
187
  ### JSON report shape
146
188
 
@@ -162,6 +204,70 @@ The JSON report always uses the nested multi-model shape:
162
204
 
163
205
  Winner is selected by **pass rate → quality score → latency** (lower latency wins all-equal ties).
164
206
 
207
+ Each item reports **how many of its runs passed**, not only whether one did:
208
+
209
+ ```json
210
+ "runs": 5, "runs_passed": 4, "pass_at_k": true, "pass_power_k": false
211
+ ```
212
+
213
+ `pass_at_k` is "did any run pass" and is what `passed` counts. `runs_passed` is the fact that separates a
214
+ reliable item from a coin-flip — without it a 5/5 item and a 1/5 item are identical in every field, because
215
+ `quality_score` is derived from the best run alone. `pass_power_k` is true only when every run passed, and the
216
+ run summary carries `passed_all_runs` beside `passed`; a large gap between the two means the model is
217
+ inconsistent rather than wrong. The console shows `4/5 runs passed` in `Notes` for a non-unanimous pass and
218
+ stays quiet for a unanimous one, and its summary line reads `3/4 passed, 1 on every run`.
219
+
220
+ `runs` is what the item actually ran, which is not always the requested `--runs`: `agentic_conversation` takes
221
+ no K and drives its fixture exactly once.
222
+
223
+ Each item additionally carries a per-phase breakdown:
224
+
225
+ ```json
226
+ "latency_breakdown_s": {
227
+ "agent_s": 4.02, // GoodData's own response time — the system under test
228
+ "judge_s": 1.31, // LLM-as-judge scoring, post-hoc
229
+ "simulated_user_s": 0.0, // our simulated user composing the next turn (multi-turn kinds)
230
+ "langfuse_s": 5.70 // trace lookup + score writing, off the critical path
231
+ }
232
+ ```
233
+
234
+ Every item reports `runs_ungraded` beside `runs_passed`: runs the agent answered but the LLM judge returned
235
+ nothing readable for. Agentic items also list them as `unscored_runs` / `judge_errors` in their `detail`, and a
236
+ `dashboard_summary` item carries `ungraded_criteria`. Such a run — or, for `dashboard_summary`, such a
237
+ criterion — is excluded from pass@K and from the quality score rather than counted as a failure: scoring it 0
238
+ would be indistinguishable from the judge genuinely failing the answer, which is the confusion
239
+ `JudgeResponseError` exists to end. `pass@K` still holds on the runs that *were* graded, so an item can pass with
240
+ `runs_ungraded` set; `pass^K` cannot, because a run nobody graded leaves "all K passed" unverified. For
241
+ `dashboard_summary` an ungraded `must_include` / `must_not_include` criterion likewise cannot carry a pass —
242
+ "the judge could not tell" is not evidence the fact is present — while an ungraded `rubric` line only narrows
243
+ the quality score. When *no* run could be graded (for `dashboard_summary`: no gating criterion on any run) the
244
+ item errors instead of reporting failures. A non-zero count means pass@K was computed over fewer runs than
245
+ `--runs` asked for, so treat the result as weaker evidence and check the judge (`GD_EVAL_JUDGE_DIAGNOSTICS=1`,
246
+ or raise `JUDGE_MAX_COMPLETION_TOKENS` if the cause is `finish_reason=length`). The console says so in `Notes`
247
+ (`1 run(s) ungraded`, `2 criterion(s) ungraded`).
248
+
249
+ `agent_s` + `judge_s` + `simulated_user_s` are the instrumented parts of the item's `latency_s`; they do not add
250
+ up to it exactly, because `latency_s` is wall-clock around the whole item and also covers the conversation
251
+ create/delete round trips, SDK construction and any cleanup. `langfuse_s` sits **beside** `latency_s`, never
252
+ inside it, because trace linking runs outside every item's critical path (see above) — summing all four would
253
+ re-inflate exactly what that design removes.
254
+
255
+ A phase that a kind does not have reports `0.0` rather than an invented number, so read the zeroes as "not
256
+ applicable here", not "instant". Today:
257
+
258
+ | Field | Populated by |
259
+ |---|---|
260
+ | `agent_s` | `agentic_general_question`, `agentic_metric_skill` |
261
+ | `judge_s` | `agentic_general_question` only — `agentic_metric_skill` compares MAQL by string, it has no LLM judge |
262
+ | `simulated_user_s` | `agentic_metric_skill` only — `agentic_general_question` is single-turn, it has no simulated user |
263
+ | `langfuse_s` | every agentic kind, but only on the `gd-eval` path and only when Langfuse credentials are present |
264
+
265
+ The other six agentic kinds report `0.0` for the first three. Trace linking itself happens whenever
266
+ `LANGFUSE_PUBLIC_KEY`/`LANGFUSE_SECRET_KEY` are exported, with or without `--langfuse`, because each
267
+ `evaluate_agentic_*` falls back to `try_make_langfuse_client()`. But its *duration* is measured by the CLI
268
+ runner rather than by `evaluate_agentic_*`, so a direct library caller sees `langfuse_s: 0.0` even though its
269
+ linking ran. Pass `TAVERN_E2E_SKIP_TRACE_LINK=1` to opt out of linking altogether.
270
+
165
271
  ---
166
272
 
167
273
  ## `gd-eval models`
@@ -1,7 +1,7 @@
1
1
  # (C) 2026 GoodData Corporation
2
2
  [project]
3
3
  name = "gooddata-eval"
4
- version = "1.74.0"
4
+ version = "1.74.1.dev1"
5
5
  description = "Evaluate the GoodData AI agent against your own questions and models."
6
6
  readme = "README.md"
7
7
  license = "MIT"
@@ -11,7 +11,7 @@ authors = [
11
11
  keywords = ["gooddata", "ai", "evaluation", "llm", "analytics", "cli"]
12
12
  requires-python = ">=3.10"
13
13
  dependencies = [
14
- "gooddata-sdk~=1.74.0",
14
+ "gooddata-sdk~=1.74.1.dev1",
15
15
  "httpx>=0.27,<1.0",
16
16
  "orjson>=3.9.15,<4.0.0",
17
17
  "pydantic>=2.6,<3.0",
@@ -30,7 +30,7 @@ classifiers = [
30
30
  ]
31
31
 
32
32
  [project.optional-dependencies]
33
- llm-judge = ["openai>=1.40,<2.0"]
33
+ llm-judge = ["openai>=1.45,<2.0"]
34
34
 
35
35
  [project.scripts]
36
36
  gd-eval = "gooddata_eval.cli.main:main"
@@ -3,10 +3,13 @@
3
3
 
4
4
  from __future__ import annotations
5
5
 
6
+ import sys
6
7
  import time
8
+ from concurrent.futures import ThreadPoolExecutor, as_completed
7
9
  from typing import Any, TypedDict
8
10
 
9
11
  from gooddata_eval.core.agentic._langfuse import make_langfuse_client
12
+ from gooddata_eval.core.agentic._trace_linker import BackgroundTraceLinker, SubmitTraceLink, run_trace_link_inline
10
13
  from gooddata_eval.core.agentic.alert_skill import evaluate_agentic_alert_skill
11
14
  from gooddata_eval.core.agentic.conversation import ConversationFixture, evaluate_agentic_conversation
12
15
  from gooddata_eval.core.agentic.general_question import evaluate_agentic_general_question
@@ -27,6 +30,7 @@ class _LfKw(TypedDict, total=False):
27
30
  run_timestamp: str
28
31
  model_version_override: str | None
29
32
  reasoning_effort: ReasoningEffort | None
33
+ submit_trace_link: SubmitTraceLink
30
34
 
31
35
 
32
36
  AGENTIC_TEST_KINDS = frozenset(
@@ -44,6 +48,43 @@ AGENTIC_TEST_KINDS = frozenset(
44
48
  )
45
49
 
46
50
 
51
+ # Kinds cleared to run several at a time. An EXPLICIT allowlist, not a subtraction: nothing
52
+ # in this package can prove a kind is read-only, because the mutation happens server-side in
53
+ # whichever tools the agent decides to call. So each entry here is a reviewed judgement, and
54
+ # anything absent -- including a kind added later -- runs serially. Slow is a recoverable
55
+ # mistake; two runs sharing a workspace mid-mutation corrupts eval results silently and
56
+ # reads like a model regression.
57
+ #
58
+ # agentic_general_question, agentic_guardrail answer questions only, no tool writes
59
+ # agentic_search search_objects, read-only by definition
60
+ # vis_agentic, agentic_visualization visualizations come back as AAC proposals
61
+ # in the chat response; nothing is persisted
62
+ # and neither module has cleanup code
63
+ PARALLEL_SAFE_TEST_KINDS = frozenset(
64
+ {
65
+ "agentic_general_question",
66
+ "agentic_guardrail",
67
+ "agentic_search",
68
+ "vis_agentic",
69
+ "agentic_visualization",
70
+ }
71
+ )
72
+
73
+ # Everything else. metric_skill and alert_skill demonstrably create workspace objects (they
74
+ # carry delete_entity_metrics / delete_entity_automations cleanup) and metric_skill._delete_metric
75
+ # records that a leaked metric gets reused by a later test. agentic_conversation drives the
76
+ # metric skill. agentic_kda_skill is here on suspicion rather than proof: it triggers
77
+ # create_key_driver_analysis with no cleanup, and while the evaluator only ever reads that
78
+ # call's ARGUMENTS -- never a created object id -- whether the platform persists anything is
79
+ # unverified. Move it to the allowlist once someone confirms it does not.
80
+ WORKSPACE_MUTATING_TEST_KINDS = frozenset(AGENTIC_TEST_KINDS) - PARALLEL_SAFE_TEST_KINDS
81
+
82
+
83
+ def runs_in_parallel(test_kind: str) -> bool:
84
+ """True only for kinds explicitly cleared for concurrent execution."""
85
+ return test_kind in PARALLEL_SAFE_TEST_KINDS
86
+
87
+
47
88
  def _parse_visualization_expected(expected_output: Any) -> list[CreatedVisualization]:
48
89
  """Parse expected_output into a list of CreatedVisualization candidates.
49
90
 
@@ -86,6 +127,7 @@ def _dispatch_agentic(
86
127
  model_version_override: str | None,
87
128
  reasoning_effort: ReasoningEffort | None = None,
88
129
  agent_id: str | None = None,
130
+ submit_trace_link: SubmitTraceLink = run_trace_link_inline,
89
131
  ) -> AgenticEvalOutcome:
90
132
  """Call the appropriate evaluate_agentic_* function for the item's test_kind.
91
133
 
@@ -102,6 +144,7 @@ def _dispatch_agentic(
102
144
  "run_timestamp": run_ts,
103
145
  "model_version_override": model_version_override,
104
146
  "reasoning_effort": reasoning_effort,
147
+ "submit_trace_link": submit_trace_link,
105
148
  }
106
149
 
107
150
  if kind in ("vis_agentic", "agentic_visualization"):
@@ -160,6 +203,7 @@ def _dispatch_agentic(
160
203
  expected_output=eo if isinstance(eo, str) else str(eo),
161
204
  k=k,
162
205
  agent_id=agent_id,
206
+ user_context=item.user_context,
163
207
  **lf_kw,
164
208
  )
165
209
  elif kind == "agentic_guardrail":
@@ -198,6 +242,40 @@ def _dispatch_agentic(
198
242
  raise ValueError(f"Unknown agentic test kind: {kind!r}")
199
243
 
200
244
 
245
+ def _apply_run_counts(item_report: ItemReport, source: Any) -> None:
246
+ """Copy how many runs passed, and how many actually ran, onto the item report.
247
+
248
+ Kinds that report neither keep the requested K and a 0 count, which reads as "not
249
+ instrumented" rather than "nothing passed" because ``pass_power_k`` is only consulted
250
+ for an item that already passed.
251
+ """
252
+ runs_passed = getattr(source, "runs_passed", None)
253
+ if runs_passed is not None:
254
+ item_report.runs_passed = runs_passed
255
+ effective = getattr(source, "runs_effective", None)
256
+ if effective:
257
+ # Only when the kind knows better than K -- agentic_conversation runs once.
258
+ item_report.runs_effective = effective
259
+ # The agentic kinds record their unscored runs in the detail; the report field is the
260
+ # one place every kind's count is read from.
261
+ unscored = (getattr(source, "detail", None) or {}).get("unscored_runs")
262
+ if isinstance(unscored, int):
263
+ item_report.runs_ungraded = unscored
264
+
265
+
266
+ def _apply_timings(item_report: ItemReport, timings: Any) -> None:
267
+ """Copy an outcome's phase breakdown onto the item report, if the kind recorded one.
268
+
269
+ Kinds with no phase instrumentation pass None and keep their 0.0 defaults rather than
270
+ reporting invented numbers.
271
+ """
272
+ if timings is None:
273
+ return
274
+ item_report.agent_latency_s = timings.agent_s
275
+ item_report.judge_latency_s = timings.judge_s
276
+ item_report.simulated_user_latency_s = timings.simulated_user_s
277
+
278
+
201
279
  def run_agentic_items(
202
280
  items: list[DatasetItem],
203
281
  host: str,
@@ -212,14 +290,29 @@ def run_agentic_items(
212
290
  on_item_start: Any = None,
213
291
  on_item_done: Any = None,
214
292
  agent_id: str | None = None,
293
+ concurrency: int = 1,
215
294
  ) -> EvalReport:
216
- """Run agentic items through evaluate_agentic_* and return an EvalReport."""
295
+ """Run agentic items through evaluate_agentic_* and return an EvalReport.
296
+
297
+ ``concurrency`` > 1 runs PARALLEL_SAFE_TEST_KINDS items simultaneously.
298
+ WORKSPACE_MUTATING_TEST_KINDS items always run one at a time, and in a separate phase
299
+ from the parallel ones -- a metric being created and dropped mid-run would otherwise be
300
+ visible to a catalog-reading item running alongside it.
301
+
302
+ Results are collected in dataset order regardless of completion order.
303
+ """
217
304
  langfuse = make_langfuse_client() if use_langfuse else None
218
305
 
219
306
  report = EvalReport(model=model_version)
220
307
  total = len(items)
308
+ # Trace linking runs here rather than inside each evaluate_agentic_*, so an item's
309
+ # Langfuse poll overlaps the NEXT item's agent call instead of extending its own
310
+ # latency. Drained below before this function returns, so every score is written
311
+ # before the caller renders a report or decides an exit code.
312
+ linker = BackgroundTraceLinker()
313
+ _t0 = time.perf_counter()
221
314
 
222
- for index, item in enumerate(items, start=1):
315
+ def _process_item(index: int, item: DatasetItem) -> ItemReport:
223
316
  if on_item_start is not None:
224
317
  try:
225
318
  on_item_start(index, total, item)
@@ -235,7 +328,17 @@ def run_agentic_items(
235
328
  t0 = time.perf_counter()
236
329
  try:
237
330
  outcome = _dispatch_agentic(
238
- item, host, token, workspace_id, k, langfuse, run_ts, model_version, reasoning_effort, agent_id
331
+ item,
332
+ host,
333
+ token,
334
+ workspace_id,
335
+ k,
336
+ langfuse,
337
+ run_ts,
338
+ model_version,
339
+ reasoning_effort,
340
+ agent_id,
341
+ submit_trace_link=linker.submit,
239
342
  )
240
343
  if isinstance(outcome, AgenticEvalOutcome):
241
344
  reasoning_steps = outcome.reasoning_steps
@@ -250,6 +353,8 @@ def run_agentic_items(
250
353
  item_report.conversation_id = conversation_id
251
354
  item_report.response_id = response_id
252
355
  item_report.best_detail = detail or {}
356
+ _apply_timings(item_report, getattr(outcome, "timings", None))
357
+ _apply_run_counts(item_report, outcome)
253
358
  except AssertionError as exc:
254
359
  item_report.pass_at_k = False
255
360
  item_report.runs = k
@@ -257,10 +362,17 @@ def run_agentic_items(
257
362
  item_report.conversation_id = getattr(exc, "conversation_id", None)
258
363
  item_report.response_id = getattr(exc, "response_id", None)
259
364
  item_report.best_detail = getattr(exc, "detail", None) or {}
365
+ _apply_timings(item_report, getattr(exc, "timings", None))
366
+ _apply_run_counts(item_report, exc)
260
367
  print(f"[agentic] {item.id} FAIL: {exc}", flush=True)
261
368
  except Exception as exc:
262
369
  item_report.error = f"{type(exc).__name__}: {exc}"
263
370
  item_report.runs = 0
371
+ # An item that errored still measured whatever it got through, and those are
372
+ # the most useful numbers on the report -- an item unevaluable because its
373
+ # judge broke should not also report the agent as costing 0s. Kinds that
374
+ # attach no timings to the exception keep their 0.0 defaults.
375
+ _apply_timings(item_report, getattr(exc, "timings", None))
264
376
  finally:
265
377
  item_report.latency_s = time.perf_counter() - t0
266
378
 
@@ -270,7 +382,79 @@ def run_agentic_items(
270
382
  except Exception:
271
383
  pass
272
384
 
273
- report.items.append(item_report)
385
+ return item_report
386
+
387
+ concurrency = max(1, concurrency)
388
+ indexed = list(enumerate(items, start=1))
389
+ if concurrency > 1:
390
+ parallel = [(i, it) for i, it in indexed if runs_in_parallel(it.test_kind)]
391
+ serial = [(i, it) for i, it in indexed if not runs_in_parallel(it.test_kind)]
392
+ if not parallel and serial:
393
+ blocked = ", ".join(sorted({it.test_kind for _, it in serial}))
394
+ print(
395
+ f"warning: --concurrency {concurrency} has no effect here; every item is a "
396
+ f"workspace-mutating kind ({blocked}) and those always run one at a time.",
397
+ file=sys.stderr,
398
+ )
399
+ else:
400
+ parallel, serial = [], indexed
401
+
402
+ results: dict[int, ItemReport] = {}
403
+ try:
404
+ # Two phases, never interleaved: a mutating item creating and dropping a metric
405
+ # mid-run would otherwise be visible to a catalog-reading item beside it.
406
+ if parallel:
407
+ # NOT a `with` block. ThreadPoolExecutor.__exit__ is shutdown(wait=True) with
408
+ # cancel_futures left False, so an interrupt raised in this thread while it
409
+ # waits on as_completed runs every QUEUED item to completion before the
410
+ # KeyboardInterrupt is honoured. cancel_futures drops whatever has not started; the handful already in flight
411
+ # cannot be cancelled (the interpreter joins those worker threads at exit
412
+ # regardless), so this bounds the wait at one wave rather than the dataset.
413
+ pool = ThreadPoolExecutor(max_workers=concurrency, thread_name_prefix="agentic")
414
+ try:
415
+ futures = {pool.submit(_process_item, i, it): i for i, it in parallel}
416
+ for future in as_completed(futures):
417
+ results[futures[future]] = future.result()
418
+ except BaseException:
419
+ pool.shutdown(wait=False, cancel_futures=True)
420
+ raise
421
+ else:
422
+ pool.shutdown(wait=True)
423
+ for i, it in serial:
424
+ results[i] = _process_item(i, it)
425
+ report.items.extend(results[i] for i in sorted(results))
426
+ except BaseException:
427
+ # Ctrl-C, or anything else escaping the loop: drop the queued polls rather than
428
+ # make the user sit through them (see BackgroundTraceLinker.abandon).
429
+ linker.abandon()
430
+ raise
431
+
432
+ # Blocks until every deferred trace link has finished: "async" here means the poll
433
+ # overlaps other items' work, never that the command finishes before scores are final.
434
+ if linker.pending:
435
+ # Said before the wait, not after. The batch runs once the last item is done and
436
+ # can take tens of seconds waiting on Langfuse ingestion; without this the
437
+ # terminal sits silent right after the final item and looks hung.
438
+ print(
439
+ f"[langfuse] linking traces for {linker.pending} item(s); waiting on Langfuse ingestion...",
440
+ flush=True,
441
+ )
442
+ _link_t0 = time.perf_counter()
443
+ linker.drain()
444
+ _link_elapsed = time.perf_counter() - _link_t0
445
+ for finished in report.items:
446
+ finished.langfuse_latency_s = linker.durations.get(finished.id, 0.0)
447
+ if linker.durations:
448
+ # Surfaces whether the retry budget is binding. A slowest close to
449
+ # _langfuse._LINK_BUDGET_SEC means links are timing out and scores are being
450
+ # orphaned -- read it together with any "no trace found for conversation" warnings.
451
+ slowest = max(linker.durations.values())
452
+ print(
453
+ f"[langfuse] trace linking finished in {_link_elapsed:.1f}s "
454
+ f"for {len(linker.durations)} item(s); slowest {slowest:.1f}s",
455
+ flush=True,
456
+ )
457
+ report.wall_clock_s = time.perf_counter() - _t0
274
458
 
275
459
  if langfuse is not None:
276
460
  try: