gooddata-eval 1.74.0__tar.gz → 1.74.1.dev1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/PKG-INFO +111 -5
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/README.md +108 -2
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/pyproject.toml +3 -3
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/src/gooddata_eval/cli/agentic_runner.py +188 -4
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/src/gooddata_eval/cli/main.py +70 -2
- gooddata_eval-1.74.1.dev1/src/gooddata_eval/core/_output.py +18 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/src/gooddata_eval/core/agentic/_langfuse.py +185 -29
- gooddata_eval-1.74.1.dev1/src/gooddata_eval/core/agentic/_trace_linker.py +302 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/src/gooddata_eval/core/agentic/alert_skill.py +88 -87
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/src/gooddata_eval/core/agentic/conversation.py +79 -66
- gooddata_eval-1.74.1.dev1/src/gooddata_eval/core/agentic/general_question.py +332 -0
- gooddata_eval-1.74.1.dev1/src/gooddata_eval/core/agentic/guardrail.py +305 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/src/gooddata_eval/core/agentic/kda_skill.py +95 -92
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/src/gooddata_eval/core/agentic/metric_skill.py +107 -69
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/src/gooddata_eval/core/agentic/search_tool.py +65 -63
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/src/gooddata_eval/core/agentic/visualization.py +83 -81
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/src/gooddata_eval/core/chat/sse_client.py +8 -2
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/src/gooddata_eval/core/config.py +25 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/src/gooddata_eval/core/dataset/langfuse_source.py +53 -22
- gooddata_eval-1.74.1.dev1/src/gooddata_eval/core/evaluators/_llm_judge.py +311 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/src/gooddata_eval/core/evaluators/general_question.py +15 -13
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/src/gooddata_eval/core/evaluators/guardrail.py +20 -17
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/src/gooddata_eval/core/evaluators/summary.py +56 -13
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/src/gooddata_eval/core/models.py +38 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/src/gooddata_eval/core/reporting/console.py +34 -10
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/src/gooddata_eval/core/reporting/json_report.py +26 -2
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/src/gooddata_eval/core/runner.py +68 -2
- gooddata_eval-1.74.1.dev1/src/gooddata_eval/core/timing.py +74 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/tests/test_agentic_alert_skill.py +49 -59
- gooddata_eval-1.74.1.dev1/tests/test_agentic_general_question.py +689 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/tests/test_agentic_guardrail.py +108 -82
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/tests/test_agentic_kda_skill.py +120 -208
- gooddata_eval-1.74.1.dev1/tests/test_agentic_langfuse_trace.py +552 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/tests/test_agentic_metric_skill.py +178 -54
- gooddata_eval-1.74.1.dev1/tests/test_agentic_runner.py +768 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/tests/test_cli.py +188 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/tests/test_connection.py +5 -0
- gooddata_eval-1.74.1.dev1/tests/test_langfuse_source.py +198 -0
- gooddata_eval-1.74.1.dev1/tests/test_llm_judge.py +616 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/tests/test_models.py +27 -0
- gooddata_eval-1.74.1.dev1/tests/test_reporting.py +444 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/tests/test_runner.py +59 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/tests/test_sse_client.py +54 -0
- gooddata_eval-1.74.1.dev1/tests/test_summary_evaluator.py +192 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/tests/test_text_evaluators.py +32 -0
- gooddata_eval-1.74.1.dev1/tests/test_timing.py +93 -0
- gooddata_eval-1.74.1.dev1/tests/test_trace_linker.py +568 -0
- gooddata_eval-1.74.0/src/gooddata_eval/core/agentic/general_question.py +0 -264
- gooddata_eval-1.74.0/src/gooddata_eval/core/agentic/guardrail.py +0 -268
- gooddata_eval-1.74.0/src/gooddata_eval/core/evaluators/_llm_judge.py +0 -66
- gooddata_eval-1.74.0/tests/test_agentic_general_question.py +0 -210
- gooddata_eval-1.74.0/tests/test_agentic_langfuse_trace.py +0 -26
- gooddata_eval-1.74.0/tests/test_agentic_runner.py +0 -220
- gooddata_eval-1.74.0/tests/test_langfuse_source.py +0 -105
- gooddata_eval-1.74.0/tests/test_llm_judge.py +0 -45
- gooddata_eval-1.74.0/tests/test_reporting.py +0 -194
- gooddata_eval-1.74.0/tests/test_summary_evaluator.py +0 -87
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/.gitignore +0 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/LICENSE.txt +0 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/Makefile +0 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/src/gooddata_eval/__init__.py +0 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/src/gooddata_eval/_version.py +0 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/src/gooddata_eval/cli/__init__.py +0 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/src/gooddata_eval/core/__init__.py +0 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/src/gooddata_eval/core/agentic/__init__.py +0 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/src/gooddata_eval/core/agentic/_catalog.py +0 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/src/gooddata_eval/core/chat/__init__.py +0 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/src/gooddata_eval/core/connection.py +0 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/src/gooddata_eval/core/dataset/__init__.py +0 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/src/gooddata_eval/core/dataset/local.py +0 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/src/gooddata_eval/core/evaluators/__init__.py +0 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/src/gooddata_eval/core/evaluators/_deep_subset.py +0 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/src/gooddata_eval/core/evaluators/_text_utils.py +0 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/src/gooddata_eval/core/evaluators/alert_skill.py +0 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/src/gooddata_eval/core/evaluators/base.py +0 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/src/gooddata_eval/core/evaluators/metric_skill.py +0 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/src/gooddata_eval/core/evaluators/search_tool.py +0 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/src/gooddata_eval/core/evaluators/visualization.py +0 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/src/gooddata_eval/core/langfuse/__init__.py +0 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/src/gooddata_eval/core/langfuse/sink.py +0 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/src/gooddata_eval/core/reporting/__init__.py +0 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/src/gooddata_eval/core/scoring.py +0 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/src/gooddata_eval/core/summary/__init__.py +0 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/src/gooddata_eval/core/summary/http_client.py +0 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/src/gooddata_eval/core/workspace.py +0 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/tests/__init__.py +0 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/tests/conftest.py +0 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/tests/fixtures/sample_dataset/metric_skill_create.json +0 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/tests/fixtures/sample_dataset/visualization_revenue.json +0 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/tests/fixtures/sse_visualization_stream.txt +0 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/tests/test_agentic_conversation.py +0 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/tests/test_agentic_run_context.py +0 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/tests/test_agentic_search_tool.py +0 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/tests/test_agentic_visualization.py +0 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/tests/test_alert_skill_evaluator.py +0 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/tests/test_deep_subset.py +0 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/tests/test_langfuse_sink.py +0 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/tests/test_local_loader.py +0 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/tests/test_metric_skill_evaluator.py +0 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/tests/test_scoring.py +0 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/tests/test_search_tool_evaluator.py +0 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/tests/test_summary_client.py +0 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/tests/test_visualization_evaluator.py +0 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/tests/test_workspace.py +0 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev1}/tox.ini +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: gooddata-eval
|
|
3
|
-
Version: 1.74.
|
|
3
|
+
Version: 1.74.1.dev1
|
|
4
4
|
Summary: Evaluate the GoodData AI agent against your own questions and models.
|
|
5
5
|
Project-URL: Source, https://github.com/gooddata/gooddata-python-sdk
|
|
6
6
|
Author-email: GoodData <support@gooddata.com>
|
|
@@ -17,13 +17,13 @@ Classifier: Topic :: Scientific/Engineering
|
|
|
17
17
|
Classifier: Topic :: Software Development
|
|
18
18
|
Classifier: Typing :: Typed
|
|
19
19
|
Requires-Python: >=3.10
|
|
20
|
-
Requires-Dist: gooddata-sdk~=1.74.
|
|
20
|
+
Requires-Dist: gooddata-sdk~=1.74.1.dev1
|
|
21
21
|
Requires-Dist: httpx<1.0,>=0.27
|
|
22
22
|
Requires-Dist: orjson<4.0.0,>=3.9.15
|
|
23
23
|
Requires-Dist: pydantic<3.0,>=2.6
|
|
24
24
|
Requires-Dist: rich<15.0,>=13.0
|
|
25
25
|
Provides-Extra: llm-judge
|
|
26
|
-
Requires-Dist: openai<2.0,>=1.
|
|
26
|
+
Requires-Dist: openai<2.0,>=1.45; extra == 'llm-judge'
|
|
27
27
|
Description-Content-Type: text/markdown
|
|
28
28
|
|
|
29
29
|
# gooddata-eval
|
|
@@ -142,6 +142,7 @@ gd-eval run \
|
|
|
142
142
|
|---|---|
|
|
143
143
|
| `--dataset PATH` | Flat folder of JSON files — one question per file. |
|
|
144
144
|
| `--langfuse-dataset NAME` | Pull items by name from a Langfuse dataset. Requires `LANGFUSE_PUBLIC_KEY`, `LANGFUSE_SECRET_KEY`, `LANGFUSE_HOST`. |
|
|
145
|
+
| `--kind TEST_KIND` | Fallback `test_kind` for dataset items that do not embed one. Defaults to `visualization`; use e.g. `agentic_metric_skill` for multi-turn agentic evaluation. Items that declare their own `test_kind` ignore this. |
|
|
145
146
|
|
|
146
147
|
#### Model selection
|
|
147
148
|
|
|
@@ -154,21 +155,62 @@ gd-eval run \
|
|
|
154
155
|
| Flag | Default | Description |
|
|
155
156
|
|---|---|---|
|
|
156
157
|
| `--runs K` | `2` | Independent runs per item (pass@K). An item passes if any run passes. |
|
|
157
|
-
| `--concurrency K` | `1` | Number of items evaluated concurrently. `1` = sequential (default). Increase to load-test the agent under simultaneous requests
|
|
158
|
+
| `--concurrency K` | `1` | Number of items evaluated concurrently. `1` = sequential (default). Increase to load-test the agent under simultaneous requests — see *Concurrency and workspace safety* below. |
|
|
159
|
+
| `--judge-model MODEL` | `gpt-4o` | Model used for LLM-as-judge scoring — `agentic_general_question`, `agentic_guardrail`, `general_question`, `guardrail` and `dashboard_summary`. Also settable via `GD_EVAL_JUDGE_MODEL`. Two things to weigh before changing it: the gpt-5 family rejects `temperature=0`, so verdicts stop being reproducible (the run warns when this happens); and choosing the same model the agent runs means the judge grades its own family's output. |
|
|
158
160
|
| `--reasoning-effort LEVEL` | server default | `LOW`, `MEDIUM` or `HIGH`, sent as `options.reasoningEffort` on every chat message. Requires the `enableGenAiReasoningEffort` feature flag on the target organization — without it the server ignores the value. Applies to chat items only; `dashboard_summary` items go through the summary endpoint, which has no such option. |
|
|
159
161
|
|
|
162
|
+
**Concurrency and workspace safety.** Agentic kinds that create workspace objects
|
|
163
|
+
(`agentic_metric_skill`, `agentic_alert_skill`, `agentic_conversation`, `agentic_kda_skill`) always run one at a
|
|
164
|
+
time whatever `--concurrency` says — a metric or alert created and dropped mid-run would otherwise be visible to
|
|
165
|
+
another item reading the same catalog. **That protection is for the agentic kinds only:** the single-turn
|
|
166
|
+
`metric_skill` and `alert_skill` kinds are still fanned out and the agent performs the same server-side writes on
|
|
167
|
+
that path, so avoid raising `--concurrency` on a dataset of those against a shared workspace. Progress output
|
|
168
|
+
interleaves when K > 1, and per-item latencies rise, so they stop being clean single-request measurements.
|
|
169
|
+
|
|
160
170
|
#### Output
|
|
161
171
|
|
|
162
172
|
| Flag | Description |
|
|
163
173
|
|---|---|
|
|
164
174
|
| `--json PATH` | Write a JSON report to this path. Always uses the nested `{models, runs, comparison}` shape even for a single model. |
|
|
165
175
|
| `--quiet` | Suppress per-item progress. Per-model result tables and the comparison summary are still printed. |
|
|
176
|
+
| `--preserve-failed` | Keep failed conversations on the server instead of deleting them, so they can be inspected afterwards. Applies to the single-turn chat path; agentic kinds manage their own conversation lifecycle. |
|
|
177
|
+
| `--timers` | Print per-turn `[timer]` diagnostics — GoodData response, judge, and simulated-user seconds as they happen. Off by default: an 18-item `--runs 2` run emits ~72 lines and buries the progress output. The same measurements are always in the JSON report's `latency_breakdown_s`, so this only adds a live view. Also settable via `GD_EVAL_TIMERS=1`. |
|
|
166
178
|
|
|
167
179
|
#### Langfuse sink
|
|
168
180
|
|
|
169
181
|
| Flag | Description |
|
|
170
182
|
|---|---|
|
|
171
|
-
| `--langfuse` | Log scores and traces to Langfuse after each item. Requires `--langfuse-dataset`.
|
|
183
|
+
| `--langfuse` | Log scores and traces to Langfuse after each item. Requires `--langfuse-dataset`. Names each experiment run `{dataset_name}_{timestamp}_{model}`, suffixed `_effort-{level}` when `--reasoning-effort` is set (so runs differing only by effort stay separate) and `_run{N}` per run when `--runs` > 1 — e.g. `general_question_2026-09-02-11-13_gpt-5.2_run0`. Requires `LANGFUSE_PUBLIC_KEY`, `LANGFUSE_SECRET_KEY`, `LANGFUSE_HOST`. |
|
|
184
|
+
|
|
185
|
+
Set `TAVERN_E2E_SKIP_TRACE_LINK=1` to skip trace lookup entirely (scores are then orphaned; the run says so).
|
|
186
|
+
|
|
187
|
+
**A local `--dataset` cannot be attached to a Langfuse run.** `--langfuse` is refused alongside `--dataset`
|
|
188
|
+
because a local folder's item ids are not Langfuse dataset item ids. But trace linking does not depend on that
|
|
189
|
+
flag — each `evaluate_agentic_*` builds its own client whenever `LANGFUSE_PUBLIC_KEY`/`LANGFUSE_SECRET_KEY` are
|
|
190
|
+
exported — so a local run still finds its traces and writes its scores onto them, and only the per-run grouping
|
|
191
|
+
fails, with one `404 from dataset-run-items` reported per run. The run warns about this before it starts. Use
|
|
192
|
+
`--langfuse-dataset` when you want runs that are comparable across models, or `TAVERN_E2E_SKIP_TRACE_LINK=1` to
|
|
193
|
+
skip linking altogether.
|
|
194
|
+
|
|
195
|
+
**When trace linking happens.** Finding a gen-ai trace means polling until Langfuse has ingested it, which is
|
|
196
|
+
lag measured in seconds to minutes. That work produces no verdict — the pass/fail is already decided — so it
|
|
197
|
+
does not run inline per item. Every item's Langfuse block is queued and the whole batch runs *after* the agent
|
|
198
|
+
phase, draining before any report is written. Two consequences worth knowing:
|
|
199
|
+
|
|
200
|
+
- **No item's `latency_s` includes trace linking.** Its cost is reported separately as
|
|
201
|
+
`latency_breakdown_s.langfuse_s`, and the run prints
|
|
202
|
+
`[langfuse] trace linking finished in Xs for N item(s); slowest Ys`. If `slowest` approaches the **120s**
|
|
203
|
+
batched retry budget, links are timing out and scores are being orphaned — look for
|
|
204
|
+
`[langfuse] WARNING: no trace found for conversation ...`.
|
|
205
|
+
- **The budget depends on who is waiting.** 120s is affordable only because the batch blocks nobody. A direct
|
|
206
|
+
library caller (`evaluate_agentic_*` without a `submit_trace_link`) polls inline, on its own critical path, and
|
|
207
|
+
gets **35s** instead — the same cost as before batching existed, so no inline caller pays for a budget raised
|
|
208
|
+
on the CLI's behalf. Either way a trace that is already ingested costs nothing: the loop looks before it sleeps.
|
|
209
|
+
Scores are always final before the command exits — the run blocks on the batch. Interrupting with Ctrl-C drops
|
|
210
|
+
whatever is still queued rather than making you wait it out: both the queued trace links and, under
|
|
211
|
+
`--concurrency`, the items that have not started. The handful of items already in flight still have to finish —
|
|
212
|
+
worker threads are joined at exit and an in-progress agent call cannot be cancelled — so expect to wait up to one
|
|
213
|
+
`--concurrency`-wide wave, not the rest of the dataset.
|
|
172
214
|
|
|
173
215
|
### JSON report shape
|
|
174
216
|
|
|
@@ -190,6 +232,70 @@ The JSON report always uses the nested multi-model shape:
|
|
|
190
232
|
|
|
191
233
|
Winner is selected by **pass rate → quality score → latency** (lower latency wins all-equal ties).
|
|
192
234
|
|
|
235
|
+
Each item reports **how many of its runs passed**, not only whether one did:
|
|
236
|
+
|
|
237
|
+
```json
|
|
238
|
+
"runs": 5, "runs_passed": 4, "pass_at_k": true, "pass_power_k": false
|
|
239
|
+
```
|
|
240
|
+
|
|
241
|
+
`pass_at_k` is "did any run pass" and is what `passed` counts. `runs_passed` is the fact that separates a
|
|
242
|
+
reliable item from a coin-flip — without it a 5/5 item and a 1/5 item are identical in every field, because
|
|
243
|
+
`quality_score` is derived from the best run alone. `pass_power_k` is true only when every run passed, and the
|
|
244
|
+
run summary carries `passed_all_runs` beside `passed`; a large gap between the two means the model is
|
|
245
|
+
inconsistent rather than wrong. The console shows `4/5 runs passed` in `Notes` for a non-unanimous pass and
|
|
246
|
+
stays quiet for a unanimous one, and its summary line reads `3/4 passed, 1 on every run`.
|
|
247
|
+
|
|
248
|
+
`runs` is what the item actually ran, which is not always the requested `--runs`: `agentic_conversation` takes
|
|
249
|
+
no K and drives its fixture exactly once.
|
|
250
|
+
|
|
251
|
+
Each item additionally carries a per-phase breakdown:
|
|
252
|
+
|
|
253
|
+
```json
|
|
254
|
+
"latency_breakdown_s": {
|
|
255
|
+
"agent_s": 4.02, // GoodData's own response time — the system under test
|
|
256
|
+
"judge_s": 1.31, // LLM-as-judge scoring, post-hoc
|
|
257
|
+
"simulated_user_s": 0.0, // our simulated user composing the next turn (multi-turn kinds)
|
|
258
|
+
"langfuse_s": 5.70 // trace lookup + score writing, off the critical path
|
|
259
|
+
}
|
|
260
|
+
```
|
|
261
|
+
|
|
262
|
+
Every item reports `runs_ungraded` beside `runs_passed`: runs the agent answered but the LLM judge returned
|
|
263
|
+
nothing readable for. Agentic items also list them as `unscored_runs` / `judge_errors` in their `detail`, and a
|
|
264
|
+
`dashboard_summary` item carries `ungraded_criteria`. Such a run — or, for `dashboard_summary`, such a
|
|
265
|
+
criterion — is excluded from pass@K and from the quality score rather than counted as a failure: scoring it 0
|
|
266
|
+
would be indistinguishable from the judge genuinely failing the answer, which is the confusion
|
|
267
|
+
`JudgeResponseError` exists to end. `pass@K` still holds on the runs that *were* graded, so an item can pass with
|
|
268
|
+
`runs_ungraded` set; `pass^K` cannot, because a run nobody graded leaves "all K passed" unverified. For
|
|
269
|
+
`dashboard_summary` an ungraded `must_include` / `must_not_include` criterion likewise cannot carry a pass —
|
|
270
|
+
"the judge could not tell" is not evidence the fact is present — while an ungraded `rubric` line only narrows
|
|
271
|
+
the quality score. When *no* run could be graded (for `dashboard_summary`: no gating criterion on any run) the
|
|
272
|
+
item errors instead of reporting failures. A non-zero count means pass@K was computed over fewer runs than
|
|
273
|
+
`--runs` asked for, so treat the result as weaker evidence and check the judge (`GD_EVAL_JUDGE_DIAGNOSTICS=1`,
|
|
274
|
+
or raise `JUDGE_MAX_COMPLETION_TOKENS` if the cause is `finish_reason=length`). The console says so in `Notes`
|
|
275
|
+
(`1 run(s) ungraded`, `2 criterion(s) ungraded`).
|
|
276
|
+
|
|
277
|
+
`agent_s` + `judge_s` + `simulated_user_s` are the instrumented parts of the item's `latency_s`; they do not add
|
|
278
|
+
up to it exactly, because `latency_s` is wall-clock around the whole item and also covers the conversation
|
|
279
|
+
create/delete round trips, SDK construction and any cleanup. `langfuse_s` sits **beside** `latency_s`, never
|
|
280
|
+
inside it, because trace linking runs outside every item's critical path (see above) — summing all four would
|
|
281
|
+
re-inflate exactly what that design removes.
|
|
282
|
+
|
|
283
|
+
A phase that a kind does not have reports `0.0` rather than an invented number, so read the zeroes as "not
|
|
284
|
+
applicable here", not "instant". Today:
|
|
285
|
+
|
|
286
|
+
| Field | Populated by |
|
|
287
|
+
|---|---|
|
|
288
|
+
| `agent_s` | `agentic_general_question`, `agentic_metric_skill` |
|
|
289
|
+
| `judge_s` | `agentic_general_question` only — `agentic_metric_skill` compares MAQL by string, it has no LLM judge |
|
|
290
|
+
| `simulated_user_s` | `agentic_metric_skill` only — `agentic_general_question` is single-turn, it has no simulated user |
|
|
291
|
+
| `langfuse_s` | every agentic kind, but only on the `gd-eval` path and only when Langfuse credentials are present |
|
|
292
|
+
|
|
293
|
+
The other six agentic kinds report `0.0` for the first three. Trace linking itself happens whenever
|
|
294
|
+
`LANGFUSE_PUBLIC_KEY`/`LANGFUSE_SECRET_KEY` are exported, with or without `--langfuse`, because each
|
|
295
|
+
`evaluate_agentic_*` falls back to `try_make_langfuse_client()`. But its *duration* is measured by the CLI
|
|
296
|
+
runner rather than by `evaluate_agentic_*`, so a direct library caller sees `langfuse_s: 0.0` even though its
|
|
297
|
+
linking ran. Pass `TAVERN_E2E_SKIP_TRACE_LINK=1` to opt out of linking altogether.
|
|
298
|
+
|
|
193
299
|
---
|
|
194
300
|
|
|
195
301
|
## `gd-eval models`
|
|
@@ -114,6 +114,7 @@ gd-eval run \
|
|
|
114
114
|
|---|---|
|
|
115
115
|
| `--dataset PATH` | Flat folder of JSON files — one question per file. |
|
|
116
116
|
| `--langfuse-dataset NAME` | Pull items by name from a Langfuse dataset. Requires `LANGFUSE_PUBLIC_KEY`, `LANGFUSE_SECRET_KEY`, `LANGFUSE_HOST`. |
|
|
117
|
+
| `--kind TEST_KIND` | Fallback `test_kind` for dataset items that do not embed one. Defaults to `visualization`; use e.g. `agentic_metric_skill` for multi-turn agentic evaluation. Items that declare their own `test_kind` ignore this. |
|
|
117
118
|
|
|
118
119
|
#### Model selection
|
|
119
120
|
|
|
@@ -126,21 +127,62 @@ gd-eval run \
|
|
|
126
127
|
| Flag | Default | Description |
|
|
127
128
|
|---|---|---|
|
|
128
129
|
| `--runs K` | `2` | Independent runs per item (pass@K). An item passes if any run passes. |
|
|
129
|
-
| `--concurrency K` | `1` | Number of items evaluated concurrently. `1` = sequential (default). Increase to load-test the agent under simultaneous requests
|
|
130
|
+
| `--concurrency K` | `1` | Number of items evaluated concurrently. `1` = sequential (default). Increase to load-test the agent under simultaneous requests — see *Concurrency and workspace safety* below. |
|
|
131
|
+
| `--judge-model MODEL` | `gpt-4o` | Model used for LLM-as-judge scoring — `agentic_general_question`, `agentic_guardrail`, `general_question`, `guardrail` and `dashboard_summary`. Also settable via `GD_EVAL_JUDGE_MODEL`. Two things to weigh before changing it: the gpt-5 family rejects `temperature=0`, so verdicts stop being reproducible (the run warns when this happens); and choosing the same model the agent runs means the judge grades its own family's output. |
|
|
130
132
|
| `--reasoning-effort LEVEL` | server default | `LOW`, `MEDIUM` or `HIGH`, sent as `options.reasoningEffort` on every chat message. Requires the `enableGenAiReasoningEffort` feature flag on the target organization — without it the server ignores the value. Applies to chat items only; `dashboard_summary` items go through the summary endpoint, which has no such option. |
|
|
131
133
|
|
|
134
|
+
**Concurrency and workspace safety.** Agentic kinds that create workspace objects
|
|
135
|
+
(`agentic_metric_skill`, `agentic_alert_skill`, `agentic_conversation`, `agentic_kda_skill`) always run one at a
|
|
136
|
+
time whatever `--concurrency` says — a metric or alert created and dropped mid-run would otherwise be visible to
|
|
137
|
+
another item reading the same catalog. **That protection is for the agentic kinds only:** the single-turn
|
|
138
|
+
`metric_skill` and `alert_skill` kinds are still fanned out and the agent performs the same server-side writes on
|
|
139
|
+
that path, so avoid raising `--concurrency` on a dataset of those against a shared workspace. Progress output
|
|
140
|
+
interleaves when K > 1, and per-item latencies rise, so they stop being clean single-request measurements.
|
|
141
|
+
|
|
132
142
|
#### Output
|
|
133
143
|
|
|
134
144
|
| Flag | Description |
|
|
135
145
|
|---|---|
|
|
136
146
|
| `--json PATH` | Write a JSON report to this path. Always uses the nested `{models, runs, comparison}` shape even for a single model. |
|
|
137
147
|
| `--quiet` | Suppress per-item progress. Per-model result tables and the comparison summary are still printed. |
|
|
148
|
+
| `--preserve-failed` | Keep failed conversations on the server instead of deleting them, so they can be inspected afterwards. Applies to the single-turn chat path; agentic kinds manage their own conversation lifecycle. |
|
|
149
|
+
| `--timers` | Print per-turn `[timer]` diagnostics — GoodData response, judge, and simulated-user seconds as they happen. Off by default: an 18-item `--runs 2` run emits ~72 lines and buries the progress output. The same measurements are always in the JSON report's `latency_breakdown_s`, so this only adds a live view. Also settable via `GD_EVAL_TIMERS=1`. |
|
|
138
150
|
|
|
139
151
|
#### Langfuse sink
|
|
140
152
|
|
|
141
153
|
| Flag | Description |
|
|
142
154
|
|---|---|
|
|
143
|
-
| `--langfuse` | Log scores and traces to Langfuse after each item. Requires `--langfuse-dataset`.
|
|
155
|
+
| `--langfuse` | Log scores and traces to Langfuse after each item. Requires `--langfuse-dataset`. Names each experiment run `{dataset_name}_{timestamp}_{model}`, suffixed `_effort-{level}` when `--reasoning-effort` is set (so runs differing only by effort stay separate) and `_run{N}` per run when `--runs` > 1 — e.g. `general_question_2026-09-02-11-13_gpt-5.2_run0`. Requires `LANGFUSE_PUBLIC_KEY`, `LANGFUSE_SECRET_KEY`, `LANGFUSE_HOST`. |
|
|
156
|
+
|
|
157
|
+
Set `TAVERN_E2E_SKIP_TRACE_LINK=1` to skip trace lookup entirely (scores are then orphaned; the run says so).
|
|
158
|
+
|
|
159
|
+
**A local `--dataset` cannot be attached to a Langfuse run.** `--langfuse` is refused alongside `--dataset`
|
|
160
|
+
because a local folder's item ids are not Langfuse dataset item ids. But trace linking does not depend on that
|
|
161
|
+
flag — each `evaluate_agentic_*` builds its own client whenever `LANGFUSE_PUBLIC_KEY`/`LANGFUSE_SECRET_KEY` are
|
|
162
|
+
exported — so a local run still finds its traces and writes its scores onto them, and only the per-run grouping
|
|
163
|
+
fails, with one `404 from dataset-run-items` reported per run. The run warns about this before it starts. Use
|
|
164
|
+
`--langfuse-dataset` when you want runs that are comparable across models, or `TAVERN_E2E_SKIP_TRACE_LINK=1` to
|
|
165
|
+
skip linking altogether.
|
|
166
|
+
|
|
167
|
+
**When trace linking happens.** Finding a gen-ai trace means polling until Langfuse has ingested it, which is
|
|
168
|
+
lag measured in seconds to minutes. That work produces no verdict — the pass/fail is already decided — so it
|
|
169
|
+
does not run inline per item. Every item's Langfuse block is queued and the whole batch runs *after* the agent
|
|
170
|
+
phase, draining before any report is written. Two consequences worth knowing:
|
|
171
|
+
|
|
172
|
+
- **No item's `latency_s` includes trace linking.** Its cost is reported separately as
|
|
173
|
+
`latency_breakdown_s.langfuse_s`, and the run prints
|
|
174
|
+
`[langfuse] trace linking finished in Xs for N item(s); slowest Ys`. If `slowest` approaches the **120s**
|
|
175
|
+
batched retry budget, links are timing out and scores are being orphaned — look for
|
|
176
|
+
`[langfuse] WARNING: no trace found for conversation ...`.
|
|
177
|
+
- **The budget depends on who is waiting.** 120s is affordable only because the batch blocks nobody. A direct
|
|
178
|
+
library caller (`evaluate_agentic_*` without a `submit_trace_link`) polls inline, on its own critical path, and
|
|
179
|
+
gets **35s** instead — the same cost as before batching existed, so no inline caller pays for a budget raised
|
|
180
|
+
on the CLI's behalf. Either way a trace that is already ingested costs nothing: the loop looks before it sleeps.
|
|
181
|
+
Scores are always final before the command exits — the run blocks on the batch. Interrupting with Ctrl-C drops
|
|
182
|
+
whatever is still queued rather than making you wait it out: both the queued trace links and, under
|
|
183
|
+
`--concurrency`, the items that have not started. The handful of items already in flight still have to finish —
|
|
184
|
+
worker threads are joined at exit and an in-progress agent call cannot be cancelled — so expect to wait up to one
|
|
185
|
+
`--concurrency`-wide wave, not the rest of the dataset.
|
|
144
186
|
|
|
145
187
|
### JSON report shape
|
|
146
188
|
|
|
@@ -162,6 +204,70 @@ The JSON report always uses the nested multi-model shape:
|
|
|
162
204
|
|
|
163
205
|
Winner is selected by **pass rate → quality score → latency** (lower latency wins all-equal ties).
|
|
164
206
|
|
|
207
|
+
Each item reports **how many of its runs passed**, not only whether one did:
|
|
208
|
+
|
|
209
|
+
```json
|
|
210
|
+
"runs": 5, "runs_passed": 4, "pass_at_k": true, "pass_power_k": false
|
|
211
|
+
```
|
|
212
|
+
|
|
213
|
+
`pass_at_k` is "did any run pass" and is what `passed` counts. `runs_passed` is the fact that separates a
|
|
214
|
+
reliable item from a coin-flip — without it a 5/5 item and a 1/5 item are identical in every field, because
|
|
215
|
+
`quality_score` is derived from the best run alone. `pass_power_k` is true only when every run passed, and the
|
|
216
|
+
run summary carries `passed_all_runs` beside `passed`; a large gap between the two means the model is
|
|
217
|
+
inconsistent rather than wrong. The console shows `4/5 runs passed` in `Notes` for a non-unanimous pass and
|
|
218
|
+
stays quiet for a unanimous one, and its summary line reads `3/4 passed, 1 on every run`.
|
|
219
|
+
|
|
220
|
+
`runs` is what the item actually ran, which is not always the requested `--runs`: `agentic_conversation` takes
|
|
221
|
+
no K and drives its fixture exactly once.
|
|
222
|
+
|
|
223
|
+
Each item additionally carries a per-phase breakdown:
|
|
224
|
+
|
|
225
|
+
```json
|
|
226
|
+
"latency_breakdown_s": {
|
|
227
|
+
"agent_s": 4.02, // GoodData's own response time — the system under test
|
|
228
|
+
"judge_s": 1.31, // LLM-as-judge scoring, post-hoc
|
|
229
|
+
"simulated_user_s": 0.0, // our simulated user composing the next turn (multi-turn kinds)
|
|
230
|
+
"langfuse_s": 5.70 // trace lookup + score writing, off the critical path
|
|
231
|
+
}
|
|
232
|
+
```
|
|
233
|
+
|
|
234
|
+
Every item reports `runs_ungraded` beside `runs_passed`: runs the agent answered but the LLM judge returned
|
|
235
|
+
nothing readable for. Agentic items also list them as `unscored_runs` / `judge_errors` in their `detail`, and a
|
|
236
|
+
`dashboard_summary` item carries `ungraded_criteria`. Such a run — or, for `dashboard_summary`, such a
|
|
237
|
+
criterion — is excluded from pass@K and from the quality score rather than counted as a failure: scoring it 0
|
|
238
|
+
would be indistinguishable from the judge genuinely failing the answer, which is the confusion
|
|
239
|
+
`JudgeResponseError` exists to end. `pass@K` still holds on the runs that *were* graded, so an item can pass with
|
|
240
|
+
`runs_ungraded` set; `pass^K` cannot, because a run nobody graded leaves "all K passed" unverified. For
|
|
241
|
+
`dashboard_summary` an ungraded `must_include` / `must_not_include` criterion likewise cannot carry a pass —
|
|
242
|
+
"the judge could not tell" is not evidence the fact is present — while an ungraded `rubric` line only narrows
|
|
243
|
+
the quality score. When *no* run could be graded (for `dashboard_summary`: no gating criterion on any run) the
|
|
244
|
+
item errors instead of reporting failures. A non-zero count means pass@K was computed over fewer runs than
|
|
245
|
+
`--runs` asked for, so treat the result as weaker evidence and check the judge (`GD_EVAL_JUDGE_DIAGNOSTICS=1`,
|
|
246
|
+
or raise `JUDGE_MAX_COMPLETION_TOKENS` if the cause is `finish_reason=length`). The console says so in `Notes`
|
|
247
|
+
(`1 run(s) ungraded`, `2 criterion(s) ungraded`).
|
|
248
|
+
|
|
249
|
+
`agent_s` + `judge_s` + `simulated_user_s` are the instrumented parts of the item's `latency_s`; they do not add
|
|
250
|
+
up to it exactly, because `latency_s` is wall-clock around the whole item and also covers the conversation
|
|
251
|
+
create/delete round trips, SDK construction and any cleanup. `langfuse_s` sits **beside** `latency_s`, never
|
|
252
|
+
inside it, because trace linking runs outside every item's critical path (see above) — summing all four would
|
|
253
|
+
re-inflate exactly what that design removes.
|
|
254
|
+
|
|
255
|
+
A phase that a kind does not have reports `0.0` rather than an invented number, so read the zeroes as "not
|
|
256
|
+
applicable here", not "instant". Today:
|
|
257
|
+
|
|
258
|
+
| Field | Populated by |
|
|
259
|
+
|---|---|
|
|
260
|
+
| `agent_s` | `agentic_general_question`, `agentic_metric_skill` |
|
|
261
|
+
| `judge_s` | `agentic_general_question` only — `agentic_metric_skill` compares MAQL by string, it has no LLM judge |
|
|
262
|
+
| `simulated_user_s` | `agentic_metric_skill` only — `agentic_general_question` is single-turn, it has no simulated user |
|
|
263
|
+
| `langfuse_s` | every agentic kind, but only on the `gd-eval` path and only when Langfuse credentials are present |
|
|
264
|
+
|
|
265
|
+
The other six agentic kinds report `0.0` for the first three. Trace linking itself happens whenever
|
|
266
|
+
`LANGFUSE_PUBLIC_KEY`/`LANGFUSE_SECRET_KEY` are exported, with or without `--langfuse`, because each
|
|
267
|
+
`evaluate_agentic_*` falls back to `try_make_langfuse_client()`. But its *duration* is measured by the CLI
|
|
268
|
+
runner rather than by `evaluate_agentic_*`, so a direct library caller sees `langfuse_s: 0.0` even though its
|
|
269
|
+
linking ran. Pass `TAVERN_E2E_SKIP_TRACE_LINK=1` to opt out of linking altogether.
|
|
270
|
+
|
|
165
271
|
---
|
|
166
272
|
|
|
167
273
|
## `gd-eval models`
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
# (C) 2026 GoodData Corporation
|
|
2
2
|
[project]
|
|
3
3
|
name = "gooddata-eval"
|
|
4
|
-
version = "1.74.
|
|
4
|
+
version = "1.74.1.dev1"
|
|
5
5
|
description = "Evaluate the GoodData AI agent against your own questions and models."
|
|
6
6
|
readme = "README.md"
|
|
7
7
|
license = "MIT"
|
|
@@ -11,7 +11,7 @@ authors = [
|
|
|
11
11
|
keywords = ["gooddata", "ai", "evaluation", "llm", "analytics", "cli"]
|
|
12
12
|
requires-python = ">=3.10"
|
|
13
13
|
dependencies = [
|
|
14
|
-
"gooddata-sdk~=1.74.
|
|
14
|
+
"gooddata-sdk~=1.74.1.dev1",
|
|
15
15
|
"httpx>=0.27,<1.0",
|
|
16
16
|
"orjson>=3.9.15,<4.0.0",
|
|
17
17
|
"pydantic>=2.6,<3.0",
|
|
@@ -30,7 +30,7 @@ classifiers = [
|
|
|
30
30
|
]
|
|
31
31
|
|
|
32
32
|
[project.optional-dependencies]
|
|
33
|
-
llm-judge = ["openai>=1.
|
|
33
|
+
llm-judge = ["openai>=1.45,<2.0"]
|
|
34
34
|
|
|
35
35
|
[project.scripts]
|
|
36
36
|
gd-eval = "gooddata_eval.cli.main:main"
|
|
@@ -3,10 +3,13 @@
|
|
|
3
3
|
|
|
4
4
|
from __future__ import annotations
|
|
5
5
|
|
|
6
|
+
import sys
|
|
6
7
|
import time
|
|
8
|
+
from concurrent.futures import ThreadPoolExecutor, as_completed
|
|
7
9
|
from typing import Any, TypedDict
|
|
8
10
|
|
|
9
11
|
from gooddata_eval.core.agentic._langfuse import make_langfuse_client
|
|
12
|
+
from gooddata_eval.core.agentic._trace_linker import BackgroundTraceLinker, SubmitTraceLink, run_trace_link_inline
|
|
10
13
|
from gooddata_eval.core.agentic.alert_skill import evaluate_agentic_alert_skill
|
|
11
14
|
from gooddata_eval.core.agentic.conversation import ConversationFixture, evaluate_agentic_conversation
|
|
12
15
|
from gooddata_eval.core.agentic.general_question import evaluate_agentic_general_question
|
|
@@ -27,6 +30,7 @@ class _LfKw(TypedDict, total=False):
|
|
|
27
30
|
run_timestamp: str
|
|
28
31
|
model_version_override: str | None
|
|
29
32
|
reasoning_effort: ReasoningEffort | None
|
|
33
|
+
submit_trace_link: SubmitTraceLink
|
|
30
34
|
|
|
31
35
|
|
|
32
36
|
AGENTIC_TEST_KINDS = frozenset(
|
|
@@ -44,6 +48,43 @@ AGENTIC_TEST_KINDS = frozenset(
|
|
|
44
48
|
)
|
|
45
49
|
|
|
46
50
|
|
|
51
|
+
# Kinds cleared to run several at a time. An EXPLICIT allowlist, not a subtraction: nothing
|
|
52
|
+
# in this package can prove a kind is read-only, because the mutation happens server-side in
|
|
53
|
+
# whichever tools the agent decides to call. So each entry here is a reviewed judgement, and
|
|
54
|
+
# anything absent -- including a kind added later -- runs serially. Slow is a recoverable
|
|
55
|
+
# mistake; two runs sharing a workspace mid-mutation corrupts eval results silently and
|
|
56
|
+
# reads like a model regression.
|
|
57
|
+
#
|
|
58
|
+
# agentic_general_question, agentic_guardrail answer questions only, no tool writes
|
|
59
|
+
# agentic_search search_objects, read-only by definition
|
|
60
|
+
# vis_agentic, agentic_visualization visualizations come back as AAC proposals
|
|
61
|
+
# in the chat response; nothing is persisted
|
|
62
|
+
# and neither module has cleanup code
|
|
63
|
+
PARALLEL_SAFE_TEST_KINDS = frozenset(
|
|
64
|
+
{
|
|
65
|
+
"agentic_general_question",
|
|
66
|
+
"agentic_guardrail",
|
|
67
|
+
"agentic_search",
|
|
68
|
+
"vis_agentic",
|
|
69
|
+
"agentic_visualization",
|
|
70
|
+
}
|
|
71
|
+
)
|
|
72
|
+
|
|
73
|
+
# Everything else. metric_skill and alert_skill demonstrably create workspace objects (they
|
|
74
|
+
# carry delete_entity_metrics / delete_entity_automations cleanup) and metric_skill._delete_metric
|
|
75
|
+
# records that a leaked metric gets reused by a later test. agentic_conversation drives the
|
|
76
|
+
# metric skill. agentic_kda_skill is here on suspicion rather than proof: it triggers
|
|
77
|
+
# create_key_driver_analysis with no cleanup, and while the evaluator only ever reads that
|
|
78
|
+
# call's ARGUMENTS -- never a created object id -- whether the platform persists anything is
|
|
79
|
+
# unverified. Move it to the allowlist once someone confirms it does not.
|
|
80
|
+
WORKSPACE_MUTATING_TEST_KINDS = frozenset(AGENTIC_TEST_KINDS) - PARALLEL_SAFE_TEST_KINDS
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
def runs_in_parallel(test_kind: str) -> bool:
|
|
84
|
+
"""True only for kinds explicitly cleared for concurrent execution."""
|
|
85
|
+
return test_kind in PARALLEL_SAFE_TEST_KINDS
|
|
86
|
+
|
|
87
|
+
|
|
47
88
|
def _parse_visualization_expected(expected_output: Any) -> list[CreatedVisualization]:
|
|
48
89
|
"""Parse expected_output into a list of CreatedVisualization candidates.
|
|
49
90
|
|
|
@@ -86,6 +127,7 @@ def _dispatch_agentic(
|
|
|
86
127
|
model_version_override: str | None,
|
|
87
128
|
reasoning_effort: ReasoningEffort | None = None,
|
|
88
129
|
agent_id: str | None = None,
|
|
130
|
+
submit_trace_link: SubmitTraceLink = run_trace_link_inline,
|
|
89
131
|
) -> AgenticEvalOutcome:
|
|
90
132
|
"""Call the appropriate evaluate_agentic_* function for the item's test_kind.
|
|
91
133
|
|
|
@@ -102,6 +144,7 @@ def _dispatch_agentic(
|
|
|
102
144
|
"run_timestamp": run_ts,
|
|
103
145
|
"model_version_override": model_version_override,
|
|
104
146
|
"reasoning_effort": reasoning_effort,
|
|
147
|
+
"submit_trace_link": submit_trace_link,
|
|
105
148
|
}
|
|
106
149
|
|
|
107
150
|
if kind in ("vis_agentic", "agentic_visualization"):
|
|
@@ -160,6 +203,7 @@ def _dispatch_agentic(
|
|
|
160
203
|
expected_output=eo if isinstance(eo, str) else str(eo),
|
|
161
204
|
k=k,
|
|
162
205
|
agent_id=agent_id,
|
|
206
|
+
user_context=item.user_context,
|
|
163
207
|
**lf_kw,
|
|
164
208
|
)
|
|
165
209
|
elif kind == "agentic_guardrail":
|
|
@@ -198,6 +242,40 @@ def _dispatch_agentic(
|
|
|
198
242
|
raise ValueError(f"Unknown agentic test kind: {kind!r}")
|
|
199
243
|
|
|
200
244
|
|
|
245
|
+
def _apply_run_counts(item_report: ItemReport, source: Any) -> None:
|
|
246
|
+
"""Copy how many runs passed, and how many actually ran, onto the item report.
|
|
247
|
+
|
|
248
|
+
Kinds that report neither keep the requested K and a 0 count, which reads as "not
|
|
249
|
+
instrumented" rather than "nothing passed" because ``pass_power_k`` is only consulted
|
|
250
|
+
for an item that already passed.
|
|
251
|
+
"""
|
|
252
|
+
runs_passed = getattr(source, "runs_passed", None)
|
|
253
|
+
if runs_passed is not None:
|
|
254
|
+
item_report.runs_passed = runs_passed
|
|
255
|
+
effective = getattr(source, "runs_effective", None)
|
|
256
|
+
if effective:
|
|
257
|
+
# Only when the kind knows better than K -- agentic_conversation runs once.
|
|
258
|
+
item_report.runs_effective = effective
|
|
259
|
+
# The agentic kinds record their unscored runs in the detail; the report field is the
|
|
260
|
+
# one place every kind's count is read from.
|
|
261
|
+
unscored = (getattr(source, "detail", None) or {}).get("unscored_runs")
|
|
262
|
+
if isinstance(unscored, int):
|
|
263
|
+
item_report.runs_ungraded = unscored
|
|
264
|
+
|
|
265
|
+
|
|
266
|
+
def _apply_timings(item_report: ItemReport, timings: Any) -> None:
|
|
267
|
+
"""Copy an outcome's phase breakdown onto the item report, if the kind recorded one.
|
|
268
|
+
|
|
269
|
+
Kinds with no phase instrumentation pass None and keep their 0.0 defaults rather than
|
|
270
|
+
reporting invented numbers.
|
|
271
|
+
"""
|
|
272
|
+
if timings is None:
|
|
273
|
+
return
|
|
274
|
+
item_report.agent_latency_s = timings.agent_s
|
|
275
|
+
item_report.judge_latency_s = timings.judge_s
|
|
276
|
+
item_report.simulated_user_latency_s = timings.simulated_user_s
|
|
277
|
+
|
|
278
|
+
|
|
201
279
|
def run_agentic_items(
|
|
202
280
|
items: list[DatasetItem],
|
|
203
281
|
host: str,
|
|
@@ -212,14 +290,29 @@ def run_agentic_items(
|
|
|
212
290
|
on_item_start: Any = None,
|
|
213
291
|
on_item_done: Any = None,
|
|
214
292
|
agent_id: str | None = None,
|
|
293
|
+
concurrency: int = 1,
|
|
215
294
|
) -> EvalReport:
|
|
216
|
-
"""Run agentic items through evaluate_agentic_* and return an EvalReport.
|
|
295
|
+
"""Run agentic items through evaluate_agentic_* and return an EvalReport.
|
|
296
|
+
|
|
297
|
+
``concurrency`` > 1 runs PARALLEL_SAFE_TEST_KINDS items simultaneously.
|
|
298
|
+
WORKSPACE_MUTATING_TEST_KINDS items always run one at a time, and in a separate phase
|
|
299
|
+
from the parallel ones -- a metric being created and dropped mid-run would otherwise be
|
|
300
|
+
visible to a catalog-reading item running alongside it.
|
|
301
|
+
|
|
302
|
+
Results are collected in dataset order regardless of completion order.
|
|
303
|
+
"""
|
|
217
304
|
langfuse = make_langfuse_client() if use_langfuse else None
|
|
218
305
|
|
|
219
306
|
report = EvalReport(model=model_version)
|
|
220
307
|
total = len(items)
|
|
308
|
+
# Trace linking runs here rather than inside each evaluate_agentic_*, so an item's
|
|
309
|
+
# Langfuse poll overlaps the NEXT item's agent call instead of extending its own
|
|
310
|
+
# latency. Drained below before this function returns, so every score is written
|
|
311
|
+
# before the caller renders a report or decides an exit code.
|
|
312
|
+
linker = BackgroundTraceLinker()
|
|
313
|
+
_t0 = time.perf_counter()
|
|
221
314
|
|
|
222
|
-
|
|
315
|
+
def _process_item(index: int, item: DatasetItem) -> ItemReport:
|
|
223
316
|
if on_item_start is not None:
|
|
224
317
|
try:
|
|
225
318
|
on_item_start(index, total, item)
|
|
@@ -235,7 +328,17 @@ def run_agentic_items(
|
|
|
235
328
|
t0 = time.perf_counter()
|
|
236
329
|
try:
|
|
237
330
|
outcome = _dispatch_agentic(
|
|
238
|
-
item,
|
|
331
|
+
item,
|
|
332
|
+
host,
|
|
333
|
+
token,
|
|
334
|
+
workspace_id,
|
|
335
|
+
k,
|
|
336
|
+
langfuse,
|
|
337
|
+
run_ts,
|
|
338
|
+
model_version,
|
|
339
|
+
reasoning_effort,
|
|
340
|
+
agent_id,
|
|
341
|
+
submit_trace_link=linker.submit,
|
|
239
342
|
)
|
|
240
343
|
if isinstance(outcome, AgenticEvalOutcome):
|
|
241
344
|
reasoning_steps = outcome.reasoning_steps
|
|
@@ -250,6 +353,8 @@ def run_agentic_items(
|
|
|
250
353
|
item_report.conversation_id = conversation_id
|
|
251
354
|
item_report.response_id = response_id
|
|
252
355
|
item_report.best_detail = detail or {}
|
|
356
|
+
_apply_timings(item_report, getattr(outcome, "timings", None))
|
|
357
|
+
_apply_run_counts(item_report, outcome)
|
|
253
358
|
except AssertionError as exc:
|
|
254
359
|
item_report.pass_at_k = False
|
|
255
360
|
item_report.runs = k
|
|
@@ -257,10 +362,17 @@ def run_agentic_items(
|
|
|
257
362
|
item_report.conversation_id = getattr(exc, "conversation_id", None)
|
|
258
363
|
item_report.response_id = getattr(exc, "response_id", None)
|
|
259
364
|
item_report.best_detail = getattr(exc, "detail", None) or {}
|
|
365
|
+
_apply_timings(item_report, getattr(exc, "timings", None))
|
|
366
|
+
_apply_run_counts(item_report, exc)
|
|
260
367
|
print(f"[agentic] {item.id} FAIL: {exc}", flush=True)
|
|
261
368
|
except Exception as exc:
|
|
262
369
|
item_report.error = f"{type(exc).__name__}: {exc}"
|
|
263
370
|
item_report.runs = 0
|
|
371
|
+
# An item that errored still measured whatever it got through, and those are
|
|
372
|
+
# the most useful numbers on the report -- an item unevaluable because its
|
|
373
|
+
# judge broke should not also report the agent as costing 0s. Kinds that
|
|
374
|
+
# attach no timings to the exception keep their 0.0 defaults.
|
|
375
|
+
_apply_timings(item_report, getattr(exc, "timings", None))
|
|
264
376
|
finally:
|
|
265
377
|
item_report.latency_s = time.perf_counter() - t0
|
|
266
378
|
|
|
@@ -270,7 +382,79 @@ def run_agentic_items(
|
|
|
270
382
|
except Exception:
|
|
271
383
|
pass
|
|
272
384
|
|
|
273
|
-
|
|
385
|
+
return item_report
|
|
386
|
+
|
|
387
|
+
concurrency = max(1, concurrency)
|
|
388
|
+
indexed = list(enumerate(items, start=1))
|
|
389
|
+
if concurrency > 1:
|
|
390
|
+
parallel = [(i, it) for i, it in indexed if runs_in_parallel(it.test_kind)]
|
|
391
|
+
serial = [(i, it) for i, it in indexed if not runs_in_parallel(it.test_kind)]
|
|
392
|
+
if not parallel and serial:
|
|
393
|
+
blocked = ", ".join(sorted({it.test_kind for _, it in serial}))
|
|
394
|
+
print(
|
|
395
|
+
f"warning: --concurrency {concurrency} has no effect here; every item is a "
|
|
396
|
+
f"workspace-mutating kind ({blocked}) and those always run one at a time.",
|
|
397
|
+
file=sys.stderr,
|
|
398
|
+
)
|
|
399
|
+
else:
|
|
400
|
+
parallel, serial = [], indexed
|
|
401
|
+
|
|
402
|
+
results: dict[int, ItemReport] = {}
|
|
403
|
+
try:
|
|
404
|
+
# Two phases, never interleaved: a mutating item creating and dropping a metric
|
|
405
|
+
# mid-run would otherwise be visible to a catalog-reading item beside it.
|
|
406
|
+
if parallel:
|
|
407
|
+
# NOT a `with` block. ThreadPoolExecutor.__exit__ is shutdown(wait=True) with
|
|
408
|
+
# cancel_futures left False, so an interrupt raised in this thread while it
|
|
409
|
+
# waits on as_completed runs every QUEUED item to completion before the
|
|
410
|
+
# KeyboardInterrupt is honoured. cancel_futures drops whatever has not started; the handful already in flight
|
|
411
|
+
# cannot be cancelled (the interpreter joins those worker threads at exit
|
|
412
|
+
# regardless), so this bounds the wait at one wave rather than the dataset.
|
|
413
|
+
pool = ThreadPoolExecutor(max_workers=concurrency, thread_name_prefix="agentic")
|
|
414
|
+
try:
|
|
415
|
+
futures = {pool.submit(_process_item, i, it): i for i, it in parallel}
|
|
416
|
+
for future in as_completed(futures):
|
|
417
|
+
results[futures[future]] = future.result()
|
|
418
|
+
except BaseException:
|
|
419
|
+
pool.shutdown(wait=False, cancel_futures=True)
|
|
420
|
+
raise
|
|
421
|
+
else:
|
|
422
|
+
pool.shutdown(wait=True)
|
|
423
|
+
for i, it in serial:
|
|
424
|
+
results[i] = _process_item(i, it)
|
|
425
|
+
report.items.extend(results[i] for i in sorted(results))
|
|
426
|
+
except BaseException:
|
|
427
|
+
# Ctrl-C, or anything else escaping the loop: drop the queued polls rather than
|
|
428
|
+
# make the user sit through them (see BackgroundTraceLinker.abandon).
|
|
429
|
+
linker.abandon()
|
|
430
|
+
raise
|
|
431
|
+
|
|
432
|
+
# Blocks until every deferred trace link has finished: "async" here means the poll
|
|
433
|
+
# overlaps other items' work, never that the command finishes before scores are final.
|
|
434
|
+
if linker.pending:
|
|
435
|
+
# Said before the wait, not after. The batch runs once the last item is done and
|
|
436
|
+
# can take tens of seconds waiting on Langfuse ingestion; without this the
|
|
437
|
+
# terminal sits silent right after the final item and looks hung.
|
|
438
|
+
print(
|
|
439
|
+
f"[langfuse] linking traces for {linker.pending} item(s); waiting on Langfuse ingestion...",
|
|
440
|
+
flush=True,
|
|
441
|
+
)
|
|
442
|
+
_link_t0 = time.perf_counter()
|
|
443
|
+
linker.drain()
|
|
444
|
+
_link_elapsed = time.perf_counter() - _link_t0
|
|
445
|
+
for finished in report.items:
|
|
446
|
+
finished.langfuse_latency_s = linker.durations.get(finished.id, 0.0)
|
|
447
|
+
if linker.durations:
|
|
448
|
+
# Surfaces whether the retry budget is binding. A slowest close to
|
|
449
|
+
# _langfuse._LINK_BUDGET_SEC means links are timing out and scores are being
|
|
450
|
+
# orphaned -- read it together with any "no trace found for conversation" warnings.
|
|
451
|
+
slowest = max(linker.durations.values())
|
|
452
|
+
print(
|
|
453
|
+
f"[langfuse] trace linking finished in {_link_elapsed:.1f}s "
|
|
454
|
+
f"for {len(linker.durations)} item(s); slowest {slowest:.1f}s",
|
|
455
|
+
flush=True,
|
|
456
|
+
)
|
|
457
|
+
report.wall_clock_s = time.perf_counter() - _t0
|
|
274
458
|
|
|
275
459
|
if langfuse is not None:
|
|
276
460
|
try:
|