gooddata-eval 1.74.1.dev2__tar.gz → 1.74.1.dev3__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/AGENTS.md +4 -4
- {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/PKG-INFO +58 -14
- {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/README.md +56 -12
- {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/pyproject.toml +2 -2
- {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/cli/agentic_runner.py +32 -2
- {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/cli/main.py +62 -11
- gooddata_eval-1.74.1.dev3/src/gooddata_eval/core/agentic/_gate.py +70 -0
- {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/agentic/_langfuse.py +199 -240
- {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/agentic/_trace_linker.py +24 -5
- {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/agentic/alert_skill.py +16 -3
- {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/agentic/conversation.py +10 -1
- {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/agentic/general_question.py +18 -3
- {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/agentic/guardrail.py +20 -3
- {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/agentic/kda_skill.py +16 -3
- {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/agentic/metric_skill.py +21 -3
- {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/agentic/search_tool.py +18 -3
- {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/agentic/visualization.py +26 -9
- {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/config.py +21 -0
- {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/dataset/langfuse_source.py +4 -14
- gooddata_eval-1.74.1.dev3/src/gooddata_eval/core/langfuse/_env.py +39 -0
- gooddata_eval-1.74.1.dev3/src/gooddata_eval/core/langfuse/client.py +205 -0
- gooddata_eval-1.74.1.dev3/src/gooddata_eval/core/langfuse/experiment.py +156 -0
- gooddata_eval-1.74.1.dev3/src/gooddata_eval/core/langfuse/observations.py +125 -0
- gooddata_eval-1.74.1.dev3/src/gooddata_eval/core/langfuse/otlp.py +164 -0
- gooddata_eval-1.74.1.dev3/src/gooddata_eval/core/langfuse/sink.py +184 -0
- {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/reporting/console.py +12 -2
- {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/reporting/json_report.py +7 -1
- {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/runner.py +20 -3
- gooddata_eval-1.74.1.dev3/tests/_fake_langfuse.py +273 -0
- gooddata_eval-1.74.1.dev3/tests/conftest.py +22 -0
- gooddata_eval-1.74.1.dev3/tests/test_agentic_gate.py +381 -0
- {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/tests/test_agentic_kda_skill.py +5 -7
- {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/tests/test_agentic_langfuse_trace.py +76 -86
- gooddata_eval-1.74.1.dev3/tests/test_agentic_observe_experiment.py +325 -0
- {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/tests/test_cli.py +2 -2
- gooddata_eval-1.74.1.dev3/tests/test_fake_langfuse.py +140 -0
- gooddata_eval-1.74.1.dev3/tests/test_langfuse_client.py +379 -0
- gooddata_eval-1.74.1.dev3/tests/test_langfuse_e2e_fake_server.py +385 -0
- gooddata_eval-1.74.1.dev3/tests/test_langfuse_env.py +70 -0
- gooddata_eval-1.74.1.dev3/tests/test_langfuse_experiment.py +206 -0
- gooddata_eval-1.74.1.dev3/tests/test_langfuse_observations.py +264 -0
- gooddata_eval-1.74.1.dev3/tests/test_langfuse_otlp.py +197 -0
- gooddata_eval-1.74.1.dev3/tests/test_langfuse_sink.py +251 -0
- {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/tests/test_langfuse_source.py +19 -1
- {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/tests/test_trace_linker.py +16 -2
- gooddata_eval-1.74.1.dev2/src/gooddata_eval/core/langfuse/sink.py +0 -197
- gooddata_eval-1.74.1.dev2/tests/conftest.py +0 -9
- gooddata_eval-1.74.1.dev2/tests/test_langfuse_sink.py +0 -165
- {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/.gitignore +0 -0
- {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/CLAUDE.md +0 -0
- {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/LICENSE.txt +0 -0
- {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/Makefile +0 -0
- {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/__init__.py +0 -0
- {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/_version.py +0 -0
- {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/cli/__init__.py +0 -0
- {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/__init__.py +0 -0
- {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/_output.py +0 -0
- {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/agentic/__init__.py +0 -0
- {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/agentic/_catalog.py +0 -0
- {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/chat/__init__.py +0 -0
- {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/chat/render.py +0 -0
- {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/chat/sse_client.py +0 -0
- {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/connection.py +0 -0
- {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/dataset/__init__.py +0 -0
- {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/dataset/local.py +0 -0
- {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/evaluators/__init__.py +0 -0
- {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/evaluators/_deep_subset.py +0 -0
- {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/evaluators/_llm_judge.py +0 -0
- {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/evaluators/_maql.py +0 -0
- {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/evaluators/_text_utils.py +0 -0
- {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/evaluators/alert_skill.py +0 -0
- {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/evaluators/base.py +0 -0
- {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/evaluators/general_question.py +0 -0
- {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/evaluators/guardrail.py +0 -0
- {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/evaluators/metric_skill.py +0 -0
- {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/evaluators/search_tool.py +0 -0
- {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/evaluators/summary.py +0 -0
- {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/evaluators/visualization.py +0 -0
- {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/langfuse/__init__.py +0 -0
- {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/models.py +0 -0
- {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/reporting/__init__.py +0 -0
- {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/scoring.py +0 -0
- {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/summary/__init__.py +0 -0
- {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/summary/http_client.py +0 -0
- {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/timing.py +0 -0
- {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/workspace.py +0 -0
- {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/tests/__init__.py +0 -0
- {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/tests/fixtures/sample_dataset/metric_skill_create.json +0 -0
- {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/tests/fixtures/sample_dataset/visualization_revenue.json +0 -0
- {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/tests/fixtures/sse_visualization_stream.txt +0 -0
- {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/tests/test_agentic_alert_skill.py +0 -0
- {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/tests/test_agentic_conversation.py +0 -0
- {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/tests/test_agentic_general_question.py +0 -0
- {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/tests/test_agentic_guardrail.py +0 -0
- {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/tests/test_agentic_metric_skill.py +0 -0
- {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/tests/test_agentic_run_context.py +0 -0
- {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/tests/test_agentic_runner.py +0 -0
- {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/tests/test_agentic_search_tool.py +0 -0
- {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/tests/test_agentic_visualization.py +0 -0
- {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/tests/test_alert_skill_evaluator.py +0 -0
- {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/tests/test_chat_render.py +0 -0
- {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/tests/test_connection.py +0 -0
- {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/tests/test_deep_subset.py +0 -0
- {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/tests/test_llm_judge.py +0 -0
- {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/tests/test_local_loader.py +0 -0
- {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/tests/test_maql_normalize.py +0 -0
- {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/tests/test_metric_skill_evaluator.py +0 -0
- {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/tests/test_models.py +0 -0
- {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/tests/test_reporting.py +0 -0
- {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/tests/test_runner.py +0 -0
- {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/tests/test_scoring.py +0 -0
- {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/tests/test_search_tool_evaluator.py +0 -0
- {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/tests/test_sse_client.py +0 -0
- {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/tests/test_summary_client.py +0 -0
- {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/tests/test_summary_evaluator.py +0 -0
- {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/tests/test_text_evaluators.py +0 -0
- {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/tests/test_timing.py +0 -0
- {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/tests/test_visualization_evaluator.py +0 -0
- {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/tests/test_workspace.py +0 -0
- {gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/tox.ini +0 -0
|
@@ -4,15 +4,15 @@
|
|
|
4
4
|
`gdc-nas`) through a dataset of natural-language questions and scores what comes back,
|
|
5
5
|
including side-by-side comparison across models. Each dataset item is a JSON envelope
|
|
6
6
|
loaded from a local folder or pulled from a Langfuse dataset. Results are aggregated into
|
|
7
|
-
pass@K / pass^K reports and optionally pushed to Langfuse as scored traces tied to
|
|
8
|
-
|
|
7
|
+
pass@K / pass^K reports and optionally pushed to Langfuse as scored traces tied to an
|
|
8
|
+
experiment. The newest and most actively developed package in the repo.
|
|
9
9
|
|
|
10
10
|
## Owns
|
|
11
11
|
|
|
12
12
|
- The `gd-eval` CLI (`gd-eval run`, `gd-eval models`)
|
|
13
13
|
- Dataset loading and the evaluation run loop
|
|
14
14
|
- Per-capability evaluators and their scoring
|
|
15
|
-
- Result reporting, and pushing
|
|
15
|
+
- Result reporting, and pushing experiments, scores and trace links to Langfuse
|
|
16
16
|
|
|
17
17
|
## Does NOT Own
|
|
18
18
|
|
|
@@ -29,7 +29,7 @@ dataset run. The newest and most actively developed package in the repo.
|
|
|
29
29
|
| `core/summary/` | HTTP client for the dedicated dashboard-summary endpoint — a single-shot chat backend, not reporting |
|
|
30
30
|
| `core/dataset/` | dataset format and loading |
|
|
31
31
|
| `core/evaluators/` | single-shot evaluators and their registry |
|
|
32
|
-
| `core/langfuse/` | `
|
|
32
|
+
| `core/langfuse/` | the whole Langfuse v4 client: `_env` (base URL + credentials), `otlp` (OTLP/JSON encoding), `experiment` (root-span construction, score targets), `observations` (trace reads), `client` (httpx calls), `sink` (single-shot results as experiments) |
|
|
33
33
|
| `core/reporting/` | console and JSON output rendering |
|
|
34
34
|
| `core/scoring.py`, `core/runner.py` | scoring and orchestration |
|
|
35
35
|
| `core/models.py` | `DatasetItem`, `ChatResult`, `ItemReport` and friends |
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: gooddata-eval
|
|
3
|
-
Version: 1.74.1.
|
|
3
|
+
Version: 1.74.1.dev3
|
|
4
4
|
Summary: Evaluate the GoodData AI agent against your own questions and models.
|
|
5
5
|
Project-URL: Source, https://github.com/gooddata/gooddata-python-sdk
|
|
6
6
|
Author-email: GoodData <support@gooddata.com>
|
|
@@ -17,7 +17,7 @@ Classifier: Topic :: Scientific/Engineering
|
|
|
17
17
|
Classifier: Topic :: Software Development
|
|
18
18
|
Classifier: Typing :: Typed
|
|
19
19
|
Requires-Python: >=3.10
|
|
20
|
-
Requires-Dist: gooddata-sdk~=1.74.1.
|
|
20
|
+
Requires-Dist: gooddata-sdk~=1.74.1.dev3
|
|
21
21
|
Requires-Dist: httpx<1.0,>=0.27
|
|
22
22
|
Requires-Dist: orjson<4.0.0,>=3.9.15
|
|
23
23
|
Requires-Dist: pydantic<3.0,>=2.6
|
|
@@ -141,7 +141,7 @@ gd-eval run \
|
|
|
141
141
|
| Flag | Description |
|
|
142
142
|
|---|---|
|
|
143
143
|
| `--dataset PATH` | Flat folder of JSON files — one question per file. |
|
|
144
|
-
| `--langfuse-dataset NAME` | Pull items by name from a Langfuse dataset. Requires `LANGFUSE_PUBLIC_KEY`, `LANGFUSE_SECRET_KEY
|
|
144
|
+
| `--langfuse-dataset NAME` | Pull items by name from a Langfuse dataset. Requires `LANGFUSE_PUBLIC_KEY`, `LANGFUSE_SECRET_KEY` and `LANGFUSE_BASE_URL` (or the legacy `LANGFUSE_HOST`). |
|
|
145
145
|
| `--kind TEST_KIND` | Fallback `test_kind` for dataset items that do not embed one. Defaults to `visualization`; use e.g. `agentic_metric_skill` for multi-turn agentic evaluation. Items that declare their own `test_kind` ignore this. |
|
|
146
146
|
|
|
147
147
|
#### Model selection
|
|
@@ -154,7 +154,8 @@ gd-eval run \
|
|
|
154
154
|
|
|
155
155
|
| Flag | Default | Description |
|
|
156
156
|
|---|---|---|
|
|
157
|
-
| `--runs K` | `2` | Independent runs per item
|
|
157
|
+
| `--runs K` | `2` | Independent runs per item. |
|
|
158
|
+
| `--gate` | `any` | Which verdict decides an item: `any` = pass@K (a run passing is enough), `power` = pass^K (every run must pass, so the verdict measures stability). Identical at `--runs 1`. Kinds that repeat K runs only — `power` is refused when the dataset also has non-agentic items or `agentic_conversation`, which are always decided on pass@K. |
|
|
158
159
|
| `--concurrency K` | `1` | Number of items evaluated concurrently. `1` = sequential (default). Increase to load-test the agent under simultaneous requests — see *Concurrency and workspace safety* below. |
|
|
159
160
|
| `--judge-model MODEL` | `gpt-4o` | Model used for LLM-as-judge scoring — `agentic_general_question`, `agentic_guardrail`, `general_question`, `guardrail` and `dashboard_summary`. Also settable via `GD_EVAL_JUDGE_MODEL`. Two things to weigh before changing it: the gpt-5 family rejects `temperature=0`, so verdicts stop being reproducible (the run warns when this happens); and choosing the same model the agent runs means the judge grades its own family's output. |
|
|
160
161
|
| `--reasoning-effort LEVEL` | server default | `LOW`, `MEDIUM` or `HIGH`, sent as `options.reasoningEffort` on every chat message. Requires the `enableGenAiReasoningEffort` feature flag on the target organization — without it the server ignores the value. Applies to chat items only; `dashboard_summary` items go through the summary endpoint, which has no such option. |
|
|
@@ -180,15 +181,39 @@ interleaves when K > 1, and per-item latencies rise, so they stop being clean si
|
|
|
180
181
|
|
|
181
182
|
| Flag | Description |
|
|
182
183
|
|---|---|
|
|
183
|
-
| `--langfuse` | Log scores and traces to Langfuse after each item. Requires `--langfuse-dataset`. Names each experiment run `{dataset_name}_{timestamp}_{model}`, suffixed `_effort-{level}` when `--reasoning-effort` is set (so runs differing only by effort stay separate) and `_run{N}` per run when `--runs` > 1 — e.g. `general_question_2026-09-02-11-13_gpt-5.2_run0`. Requires `LANGFUSE_PUBLIC_KEY`, `LANGFUSE_SECRET_KEY
|
|
184
|
+
| `--langfuse` | Log scores and traces to Langfuse after each item. Requires `--langfuse-dataset`. Names each experiment run `{dataset_name}_{timestamp}_{model}`, suffixed `_effort-{level}` when `--reasoning-effort` is set (so runs differing only by effort stay separate) and `_run{N}` per run when `--runs` > 1 — e.g. `general_question_2026-09-02-11-13_gpt-5.2_run0`. Requires `LANGFUSE_PUBLIC_KEY`, `LANGFUSE_SECRET_KEY` and `LANGFUSE_BASE_URL` (or the legacy `LANGFUSE_HOST`). |
|
|
184
185
|
|
|
185
|
-
|
|
186
|
+
##### Langfuse v4
|
|
186
187
|
|
|
187
|
-
|
|
188
|
-
|
|
189
|
-
|
|
190
|
-
|
|
191
|
-
|
|
188
|
+
One run is one Langfuse **experiment**. Each evaluated item becomes its own trace whose root span carries the
|
|
189
|
+
experiment and dataset-item attributes (`langfuse.experiment.name`, `langfuse.experiment.dataset.id`,
|
|
190
|
+
`langfuse.experiment.item.id`), and the four scores attach to that root observation. gd-eval speaks to Langfuse
|
|
191
|
+
over four REST endpoints and uses no Langfuse SDK, so it runs on every Python version the package supports:
|
|
192
|
+
|
|
193
|
+
| Endpoint | Used for |
|
|
194
|
+
|---|---|
|
|
195
|
+
| `POST /api/public/otel/v1/traces` | exporting the experiment root span as OTLP/HTTP JSON |
|
|
196
|
+
| `POST /api/public/scores` | one score per write, on a trace or on a single observation inside it |
|
|
197
|
+
| `GET /api/public/v2/observations` | finding the agent's gen-ai trace for a conversation |
|
|
198
|
+
| `GET /api/public/dataset-items` | loading `--langfuse-dataset` items, and resolving an item's dataset id |
|
|
199
|
+
|
|
200
|
+
Two consequences of v4's immutable observations. gd-eval sets `version` only on its own experiment span, never on
|
|
201
|
+
the agent's gen-ai trace — filter on the gd-eval experiment's `langfuse.version` to compare models. And on the
|
|
202
|
+
agentic kinds the latency in `value_score` is the gen-ai trace's root generation latency, read from the
|
|
203
|
+
observations endpoint; the single-shot `--langfuse` sink keeps using the item's own measured average latency.
|
|
204
|
+
|
|
205
|
+
Langfuse Cloud drops v3 on **2026-11-16**; a self-hosted Langfuse must be on v4 for any of this to work.
|
|
206
|
+
|
|
207
|
+
Set `TAVERN_E2E_SKIP_TRACE_LINK=1` to turn the whole **agentic** Langfuse write path off — no trace lookup, no
|
|
208
|
+
span export and no scores for `agentic_*` items. The run says so once. It does not reach the `--langfuse` sink,
|
|
209
|
+
which still writes a span and four scores for every single-shot item; drop `--langfuse` to silence that too.
|
|
210
|
+
|
|
211
|
+
**A local `--dataset` cannot be attached to a Langfuse experiment.** `--langfuse` is refused alongside
|
|
212
|
+
`--dataset` because a local folder's item ids are not Langfuse dataset item ids. But trace linking does not
|
|
213
|
+
depend on that flag — each `evaluate_agentic_*` builds its own client whenever
|
|
214
|
+
`LANGFUSE_PUBLIC_KEY`/`LANGFUSE_SECRET_KEY` are exported — so a local run still finds its traces and writes its
|
|
215
|
+
scores onto them, and only the per-run grouping fails: the dataset-item lookup 404s and the run reports the item
|
|
216
|
+
as one that does not exist in Langfuse, once. The run also warns about this before it starts. Use
|
|
192
217
|
`--langfuse-dataset` when you want runs that are comparable across models, or `TAVERN_E2E_SKIP_TRACE_LINK=1` to
|
|
193
218
|
skip linking altogether.
|
|
194
219
|
|
|
@@ -235,10 +260,12 @@ Winner is selected by **pass rate → quality score → latency** (lower latency
|
|
|
235
260
|
Each item reports **how many of its runs passed**, not only whether one did:
|
|
236
261
|
|
|
237
262
|
```json
|
|
238
|
-
"runs": 5, "runs_passed": 4, "pass_at_k": true, "pass_power_k": false
|
|
263
|
+
"runs": 5, "runs_passed": 4, "pass_at_k": true, "pass_power_k": false, "gate_passed": true
|
|
239
264
|
```
|
|
240
265
|
|
|
241
|
-
`pass_at_k` is "did any run pass"
|
|
266
|
+
`pass_at_k` is "did any run pass" — always literal, whatever the gate. `gate_passed` is the verdict the item
|
|
267
|
+
was decided on and is what `passed` counts; the two differ only under `--gate power`, where an item that
|
|
268
|
+
passed 4 of 5 runs is `"pass_at_k": true, "gate_passed": false`. `runs_passed` is the fact that separates a
|
|
242
269
|
reliable item from a coin-flip — without it a 5/5 item and a 1/5 item are identical in every field, because
|
|
243
270
|
`quality_score` is derived from the best run alone. `pass_power_k` is true only when every run passed, and the
|
|
244
271
|
run summary carries `passed_all_runs` beside `passed`; a large gap between the two means the model is
|
|
@@ -248,6 +275,14 @@ stays quiet for a unanimous one, and its summary line reads `3/4 passed, 1 on ev
|
|
|
248
275
|
`runs` is what the item actually ran, which is not always the requested `--runs`: `agentic_conversation` takes
|
|
249
276
|
no K and drives its fixture exactly once.
|
|
250
277
|
|
|
278
|
+
Which of the two decides pass/fail is `--gate`: `any` (default) gates on `pass_at_k`, `power` gates on
|
|
279
|
+
`pass_power_k`. The run records it as a top-level `gate`, and a failure under `power` says so —
|
|
280
|
+
`Gate pass^3 failed: 2/3 runs passed — unstable, not a clean failure` — because the message body describes
|
|
281
|
+
the best run, which under pass^K can be a run that passed, and the console repeats the count in `Notes` for
|
|
282
|
+
the same reason. When some runs were ungraded the note says so instead of calling the remainder unstable.
|
|
283
|
+
`agentic_conversation` has no K gate, so `--gate power` is refused for a dataset containing one rather than
|
|
284
|
+
labelling a report `power` that only part of the dataset was decided under.
|
|
285
|
+
|
|
251
286
|
Each item additionally carries a per-phase breakdown:
|
|
252
287
|
|
|
253
288
|
```json
|
|
@@ -408,10 +443,19 @@ Without `[llm-judge]`, those items are **skipped**.
|
|
|
408
443
|
|
|
409
444
|
## Scores (in JSON report and Langfuse)
|
|
410
445
|
|
|
446
|
+
In Langfuse every score is written to the experiment run's root observation — `traceId` plus `observationId` of
|
|
447
|
+
the item's own root span. On the agentic path each score is mirrored onto the agent's gen-ai trace as well
|
|
448
|
+
(`traceId` only), so a score survives even when one of the two traces is missing.
|
|
449
|
+
|
|
411
450
|
| Score | Description |
|
|
412
451
|
|---|---|
|
|
413
|
-
| `pass_at_k` | 1 if any of the K runs passed strict checks, else 0. |
|
|
452
|
+
| `pass_at_k` | 1 if **any** of the K runs passed strict checks, else 0. |
|
|
453
|
+
| `pass_power_k` | 1 only if **every** one of the K runs passed. Agentic kinds only. |
|
|
454
|
+
| `gate_passed` | The verdict that decided the item: `pass_at_k` under `--gate any`, `pass_power_k` under `--gate power`. Agentic kinds only. |
|
|
414
455
|
| `quality_score` | Fraction of strict check flags that are `True` (0.0–1.0). Shown in CLI as a percentage. |
|
|
415
456
|
| `value_score` | Weighted blend: 0.6 × quality + 0.2 × speed (speed = max(0, 1 − latency/60s)). |
|
|
416
457
|
| `latency_s` | Average per-run latency in seconds. |
|
|
417
458
|
| `provider_type` | Model vendor + gateway label (e.g. `ANTHROPIC`, `BEDROCK/ANTHROPIC`, `AZURE/OPENAI`). Stored in Langfuse trace metadata and tags. |
|
|
459
|
+
|
|
460
|
+
Score names carry no K; K and the gate are on the dataset-run metadata as `eval_k` and `eval_gate`.
|
|
461
|
+
`agentic_visualization` also still writes its historic `pass_at_{K}` / `pass_power_{K}` pair.
|
|
@@ -113,7 +113,7 @@ gd-eval run \
|
|
|
113
113
|
| Flag | Description |
|
|
114
114
|
|---|---|
|
|
115
115
|
| `--dataset PATH` | Flat folder of JSON files — one question per file. |
|
|
116
|
-
| `--langfuse-dataset NAME` | Pull items by name from a Langfuse dataset. Requires `LANGFUSE_PUBLIC_KEY`, `LANGFUSE_SECRET_KEY
|
|
116
|
+
| `--langfuse-dataset NAME` | Pull items by name from a Langfuse dataset. Requires `LANGFUSE_PUBLIC_KEY`, `LANGFUSE_SECRET_KEY` and `LANGFUSE_BASE_URL` (or the legacy `LANGFUSE_HOST`). |
|
|
117
117
|
| `--kind TEST_KIND` | Fallback `test_kind` for dataset items that do not embed one. Defaults to `visualization`; use e.g. `agentic_metric_skill` for multi-turn agentic evaluation. Items that declare their own `test_kind` ignore this. |
|
|
118
118
|
|
|
119
119
|
#### Model selection
|
|
@@ -126,7 +126,8 @@ gd-eval run \
|
|
|
126
126
|
|
|
127
127
|
| Flag | Default | Description |
|
|
128
128
|
|---|---|---|
|
|
129
|
-
| `--runs K` | `2` | Independent runs per item
|
|
129
|
+
| `--runs K` | `2` | Independent runs per item. |
|
|
130
|
+
| `--gate` | `any` | Which verdict decides an item: `any` = pass@K (a run passing is enough), `power` = pass^K (every run must pass, so the verdict measures stability). Identical at `--runs 1`. Kinds that repeat K runs only — `power` is refused when the dataset also has non-agentic items or `agentic_conversation`, which are always decided on pass@K. |
|
|
130
131
|
| `--concurrency K` | `1` | Number of items evaluated concurrently. `1` = sequential (default). Increase to load-test the agent under simultaneous requests — see *Concurrency and workspace safety* below. |
|
|
131
132
|
| `--judge-model MODEL` | `gpt-4o` | Model used for LLM-as-judge scoring — `agentic_general_question`, `agentic_guardrail`, `general_question`, `guardrail` and `dashboard_summary`. Also settable via `GD_EVAL_JUDGE_MODEL`. Two things to weigh before changing it: the gpt-5 family rejects `temperature=0`, so verdicts stop being reproducible (the run warns when this happens); and choosing the same model the agent runs means the judge grades its own family's output. |
|
|
132
133
|
| `--reasoning-effort LEVEL` | server default | `LOW`, `MEDIUM` or `HIGH`, sent as `options.reasoningEffort` on every chat message. Requires the `enableGenAiReasoningEffort` feature flag on the target organization — without it the server ignores the value. Applies to chat items only; `dashboard_summary` items go through the summary endpoint, which has no such option. |
|
|
@@ -152,15 +153,39 @@ interleaves when K > 1, and per-item latencies rise, so they stop being clean si
|
|
|
152
153
|
|
|
153
154
|
| Flag | Description |
|
|
154
155
|
|---|---|
|
|
155
|
-
| `--langfuse` | Log scores and traces to Langfuse after each item. Requires `--langfuse-dataset`. Names each experiment run `{dataset_name}_{timestamp}_{model}`, suffixed `_effort-{level}` when `--reasoning-effort` is set (so runs differing only by effort stay separate) and `_run{N}` per run when `--runs` > 1 — e.g. `general_question_2026-09-02-11-13_gpt-5.2_run0`. Requires `LANGFUSE_PUBLIC_KEY`, `LANGFUSE_SECRET_KEY
|
|
156
|
+
| `--langfuse` | Log scores and traces to Langfuse after each item. Requires `--langfuse-dataset`. Names each experiment run `{dataset_name}_{timestamp}_{model}`, suffixed `_effort-{level}` when `--reasoning-effort` is set (so runs differing only by effort stay separate) and `_run{N}` per run when `--runs` > 1 — e.g. `general_question_2026-09-02-11-13_gpt-5.2_run0`. Requires `LANGFUSE_PUBLIC_KEY`, `LANGFUSE_SECRET_KEY` and `LANGFUSE_BASE_URL` (or the legacy `LANGFUSE_HOST`). |
|
|
156
157
|
|
|
157
|
-
|
|
158
|
+
##### Langfuse v4
|
|
158
159
|
|
|
159
|
-
|
|
160
|
-
|
|
161
|
-
|
|
162
|
-
|
|
163
|
-
|
|
160
|
+
One run is one Langfuse **experiment**. Each evaluated item becomes its own trace whose root span carries the
|
|
161
|
+
experiment and dataset-item attributes (`langfuse.experiment.name`, `langfuse.experiment.dataset.id`,
|
|
162
|
+
`langfuse.experiment.item.id`), and the four scores attach to that root observation. gd-eval speaks to Langfuse
|
|
163
|
+
over four REST endpoints and uses no Langfuse SDK, so it runs on every Python version the package supports:
|
|
164
|
+
|
|
165
|
+
| Endpoint | Used for |
|
|
166
|
+
|---|---|
|
|
167
|
+
| `POST /api/public/otel/v1/traces` | exporting the experiment root span as OTLP/HTTP JSON |
|
|
168
|
+
| `POST /api/public/scores` | one score per write, on a trace or on a single observation inside it |
|
|
169
|
+
| `GET /api/public/v2/observations` | finding the agent's gen-ai trace for a conversation |
|
|
170
|
+
| `GET /api/public/dataset-items` | loading `--langfuse-dataset` items, and resolving an item's dataset id |
|
|
171
|
+
|
|
172
|
+
Two consequences of v4's immutable observations. gd-eval sets `version` only on its own experiment span, never on
|
|
173
|
+
the agent's gen-ai trace — filter on the gd-eval experiment's `langfuse.version` to compare models. And on the
|
|
174
|
+
agentic kinds the latency in `value_score` is the gen-ai trace's root generation latency, read from the
|
|
175
|
+
observations endpoint; the single-shot `--langfuse` sink keeps using the item's own measured average latency.
|
|
176
|
+
|
|
177
|
+
Langfuse Cloud drops v3 on **2026-11-16**; a self-hosted Langfuse must be on v4 for any of this to work.
|
|
178
|
+
|
|
179
|
+
Set `TAVERN_E2E_SKIP_TRACE_LINK=1` to turn the whole **agentic** Langfuse write path off — no trace lookup, no
|
|
180
|
+
span export and no scores for `agentic_*` items. The run says so once. It does not reach the `--langfuse` sink,
|
|
181
|
+
which still writes a span and four scores for every single-shot item; drop `--langfuse` to silence that too.
|
|
182
|
+
|
|
183
|
+
**A local `--dataset` cannot be attached to a Langfuse experiment.** `--langfuse` is refused alongside
|
|
184
|
+
`--dataset` because a local folder's item ids are not Langfuse dataset item ids. But trace linking does not
|
|
185
|
+
depend on that flag — each `evaluate_agentic_*` builds its own client whenever
|
|
186
|
+
`LANGFUSE_PUBLIC_KEY`/`LANGFUSE_SECRET_KEY` are exported — so a local run still finds its traces and writes its
|
|
187
|
+
scores onto them, and only the per-run grouping fails: the dataset-item lookup 404s and the run reports the item
|
|
188
|
+
as one that does not exist in Langfuse, once. The run also warns about this before it starts. Use
|
|
164
189
|
`--langfuse-dataset` when you want runs that are comparable across models, or `TAVERN_E2E_SKIP_TRACE_LINK=1` to
|
|
165
190
|
skip linking altogether.
|
|
166
191
|
|
|
@@ -207,10 +232,12 @@ Winner is selected by **pass rate → quality score → latency** (lower latency
|
|
|
207
232
|
Each item reports **how many of its runs passed**, not only whether one did:
|
|
208
233
|
|
|
209
234
|
```json
|
|
210
|
-
"runs": 5, "runs_passed": 4, "pass_at_k": true, "pass_power_k": false
|
|
235
|
+
"runs": 5, "runs_passed": 4, "pass_at_k": true, "pass_power_k": false, "gate_passed": true
|
|
211
236
|
```
|
|
212
237
|
|
|
213
|
-
`pass_at_k` is "did any run pass"
|
|
238
|
+
`pass_at_k` is "did any run pass" — always literal, whatever the gate. `gate_passed` is the verdict the item
|
|
239
|
+
was decided on and is what `passed` counts; the two differ only under `--gate power`, where an item that
|
|
240
|
+
passed 4 of 5 runs is `"pass_at_k": true, "gate_passed": false`. `runs_passed` is the fact that separates a
|
|
214
241
|
reliable item from a coin-flip — without it a 5/5 item and a 1/5 item are identical in every field, because
|
|
215
242
|
`quality_score` is derived from the best run alone. `pass_power_k` is true only when every run passed, and the
|
|
216
243
|
run summary carries `passed_all_runs` beside `passed`; a large gap between the two means the model is
|
|
@@ -220,6 +247,14 @@ stays quiet for a unanimous one, and its summary line reads `3/4 passed, 1 on ev
|
|
|
220
247
|
`runs` is what the item actually ran, which is not always the requested `--runs`: `agentic_conversation` takes
|
|
221
248
|
no K and drives its fixture exactly once.
|
|
222
249
|
|
|
250
|
+
Which of the two decides pass/fail is `--gate`: `any` (default) gates on `pass_at_k`, `power` gates on
|
|
251
|
+
`pass_power_k`. The run records it as a top-level `gate`, and a failure under `power` says so —
|
|
252
|
+
`Gate pass^3 failed: 2/3 runs passed — unstable, not a clean failure` — because the message body describes
|
|
253
|
+
the best run, which under pass^K can be a run that passed, and the console repeats the count in `Notes` for
|
|
254
|
+
the same reason. When some runs were ungraded the note says so instead of calling the remainder unstable.
|
|
255
|
+
`agentic_conversation` has no K gate, so `--gate power` is refused for a dataset containing one rather than
|
|
256
|
+
labelling a report `power` that only part of the dataset was decided under.
|
|
257
|
+
|
|
223
258
|
Each item additionally carries a per-phase breakdown:
|
|
224
259
|
|
|
225
260
|
```json
|
|
@@ -380,10 +415,19 @@ Without `[llm-judge]`, those items are **skipped**.
|
|
|
380
415
|
|
|
381
416
|
## Scores (in JSON report and Langfuse)
|
|
382
417
|
|
|
418
|
+
In Langfuse every score is written to the experiment run's root observation — `traceId` plus `observationId` of
|
|
419
|
+
the item's own root span. On the agentic path each score is mirrored onto the agent's gen-ai trace as well
|
|
420
|
+
(`traceId` only), so a score survives even when one of the two traces is missing.
|
|
421
|
+
|
|
383
422
|
| Score | Description |
|
|
384
423
|
|---|---|
|
|
385
|
-
| `pass_at_k` | 1 if any of the K runs passed strict checks, else 0. |
|
|
424
|
+
| `pass_at_k` | 1 if **any** of the K runs passed strict checks, else 0. |
|
|
425
|
+
| `pass_power_k` | 1 only if **every** one of the K runs passed. Agentic kinds only. |
|
|
426
|
+
| `gate_passed` | The verdict that decided the item: `pass_at_k` under `--gate any`, `pass_power_k` under `--gate power`. Agentic kinds only. |
|
|
386
427
|
| `quality_score` | Fraction of strict check flags that are `True` (0.0–1.0). Shown in CLI as a percentage. |
|
|
387
428
|
| `value_score` | Weighted blend: 0.6 × quality + 0.2 × speed (speed = max(0, 1 − latency/60s)). |
|
|
388
429
|
| `latency_s` | Average per-run latency in seconds. |
|
|
389
430
|
| `provider_type` | Model vendor + gateway label (e.g. `ANTHROPIC`, `BEDROCK/ANTHROPIC`, `AZURE/OPENAI`). Stored in Langfuse trace metadata and tags. |
|
|
431
|
+
|
|
432
|
+
Score names carry no K; K and the gate are on the dataset-run metadata as `eval_k` and `eval_gate`.
|
|
433
|
+
`agentic_visualization` also still writes its historic `pass_at_{K}` / `pass_power_{K}` pair.
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
# (C) 2026 GoodData Corporation
|
|
2
2
|
[project]
|
|
3
3
|
name = "gooddata-eval"
|
|
4
|
-
version = "1.74.1.
|
|
4
|
+
version = "1.74.1.dev3"
|
|
5
5
|
description = "Evaluate the GoodData AI agent against your own questions and models."
|
|
6
6
|
readme = "README.md"
|
|
7
7
|
license = "MIT"
|
|
@@ -11,7 +11,7 @@ authors = [
|
|
|
11
11
|
keywords = ["gooddata", "ai", "evaluation", "llm", "analytics", "cli"]
|
|
12
12
|
requires-python = ">=3.10"
|
|
13
13
|
dependencies = [
|
|
14
|
-
"gooddata-sdk~=1.74.1.
|
|
14
|
+
"gooddata-sdk~=1.74.1.dev3",
|
|
15
15
|
"httpx>=0.27,<1.0",
|
|
16
16
|
"orjson>=3.9.15,<4.0.0",
|
|
17
17
|
"pydantic>=2.6,<3.0",
|
{gooddata_eval-1.74.1.dev2 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/cli/agentic_runner.py
RENAMED
|
@@ -8,6 +8,7 @@ import time
|
|
|
8
8
|
from concurrent.futures import ThreadPoolExecutor, as_completed
|
|
9
9
|
from typing import Any, TypedDict
|
|
10
10
|
|
|
11
|
+
from gooddata_eval.core.agentic._gate import DEFAULT_GATE, EvalGate, normalize_gate
|
|
11
12
|
from gooddata_eval.core.agentic._langfuse import make_langfuse_client
|
|
12
13
|
from gooddata_eval.core.agentic._trace_linker import BackgroundTraceLinker, SubmitTraceLink, run_trace_link_inline
|
|
13
14
|
from gooddata_eval.core.agentic.alert_skill import evaluate_agentic_alert_skill
|
|
@@ -48,6 +49,13 @@ AGENTIC_TEST_KINDS = frozenset(
|
|
|
48
49
|
)
|
|
49
50
|
|
|
50
51
|
|
|
52
|
+
# Agentic kinds that no gate applies to: they drive their fixture exactly once, so there is
|
|
53
|
+
# no K to take pass@K or pass^K over. Named here rather than inline in _dispatch_agentic so
|
|
54
|
+
# the CLI can refuse --gate power for a dataset containing one instead of labelling the whole
|
|
55
|
+
# report `power` when part of it was never gated.
|
|
56
|
+
UNGATED_AGENTIC_TEST_KINDS = frozenset({"agentic_conversation"})
|
|
57
|
+
|
|
58
|
+
|
|
51
59
|
# Kinds cleared to run several at a time. An EXPLICIT allowlist, not a subtraction: nothing
|
|
52
60
|
# in this package can prove a kind is read-only, because the mutation happens server-side in
|
|
53
61
|
# whichever tools the agent decides to call. So each entry here is a reviewed judgement, and
|
|
@@ -128,9 +136,13 @@ def _dispatch_agentic(
|
|
|
128
136
|
reasoning_effort: ReasoningEffort | None = None,
|
|
129
137
|
agent_id: str | None = None,
|
|
130
138
|
submit_trace_link: SubmitTraceLink = run_trace_link_inline,
|
|
139
|
+
gate: EvalGate = DEFAULT_GATE,
|
|
131
140
|
) -> AgenticEvalOutcome:
|
|
132
141
|
"""Call the appropriate evaluate_agentic_* function for the item's test_kind.
|
|
133
142
|
|
|
143
|
+
`gate` reaches every kind except those in UNGATED_AGENTIC_TEST_KINDS, which have no K
|
|
144
|
+
to gate over; the CLI refuses --gate power for a dataset containing one.
|
|
145
|
+
|
|
134
146
|
Every evaluate_agentic_* function returns an AgenticEvalOutcome (reasoning_steps,
|
|
135
147
|
conversation_id, response_id, detail) on success and attaches the same four attributes
|
|
136
148
|
to its raised *AssertionError on failure -- no kind is exempt.
|
|
@@ -155,6 +167,7 @@ def _dispatch_agentic(
|
|
|
155
167
|
question=item.question,
|
|
156
168
|
expected_outputs=_parse_visualization_expected(eo),
|
|
157
169
|
k=k,
|
|
170
|
+
gate=gate,
|
|
158
171
|
agent_id=agent_id,
|
|
159
172
|
**lf_kw,
|
|
160
173
|
)
|
|
@@ -166,6 +179,7 @@ def _dispatch_agentic(
|
|
|
166
179
|
question=item.question,
|
|
167
180
|
expected_output=eo if isinstance(eo, (dict, list)) else {},
|
|
168
181
|
k=k,
|
|
182
|
+
gate=gate,
|
|
169
183
|
agent_id=agent_id,
|
|
170
184
|
**lf_kw,
|
|
171
185
|
)
|
|
@@ -177,6 +191,7 @@ def _dispatch_agentic(
|
|
|
177
191
|
question=item.question,
|
|
178
192
|
expected_output=eo if isinstance(eo, dict) else {},
|
|
179
193
|
k=k,
|
|
194
|
+
gate=gate,
|
|
180
195
|
agent_id=agent_id,
|
|
181
196
|
**lf_kw,
|
|
182
197
|
)
|
|
@@ -191,6 +206,7 @@ def _dispatch_agentic(
|
|
|
191
206
|
question=item.question,
|
|
192
207
|
expected_tool_call=expected_args,
|
|
193
208
|
k=k,
|
|
209
|
+
gate=gate,
|
|
194
210
|
agent_id=agent_id,
|
|
195
211
|
**lf_kw,
|
|
196
212
|
)
|
|
@@ -202,6 +218,7 @@ def _dispatch_agentic(
|
|
|
202
218
|
question=item.question,
|
|
203
219
|
expected_output=eo if isinstance(eo, str) else str(eo),
|
|
204
220
|
k=k,
|
|
221
|
+
gate=gate,
|
|
205
222
|
agent_id=agent_id,
|
|
206
223
|
user_context=item.user_context,
|
|
207
224
|
**lf_kw,
|
|
@@ -214,6 +231,7 @@ def _dispatch_agentic(
|
|
|
214
231
|
question=item.question,
|
|
215
232
|
expected_output=eo if isinstance(eo, str) else str(eo),
|
|
216
233
|
k=k,
|
|
234
|
+
gate=gate,
|
|
217
235
|
agent_id=agent_id,
|
|
218
236
|
**lf_kw,
|
|
219
237
|
)
|
|
@@ -225,6 +243,7 @@ def _dispatch_agentic(
|
|
|
225
243
|
question=item.question,
|
|
226
244
|
expected_output=eo if isinstance(eo, dict) else {},
|
|
227
245
|
k=k,
|
|
246
|
+
gate=gate,
|
|
228
247
|
agent_id=agent_id,
|
|
229
248
|
**lf_kw,
|
|
230
249
|
)
|
|
@@ -291,6 +310,7 @@ def run_agentic_items(
|
|
|
291
310
|
on_item_done: Any = None,
|
|
292
311
|
agent_id: str | None = None,
|
|
293
312
|
concurrency: int = 1,
|
|
313
|
+
gate: EvalGate = DEFAULT_GATE,
|
|
294
314
|
) -> EvalReport:
|
|
295
315
|
"""Run agentic items through evaluate_agentic_* and return an EvalReport.
|
|
296
316
|
|
|
@@ -303,7 +323,7 @@ def run_agentic_items(
|
|
|
303
323
|
"""
|
|
304
324
|
langfuse = make_langfuse_client() if use_langfuse else None
|
|
305
325
|
|
|
306
|
-
report = EvalReport(model=model_version)
|
|
326
|
+
report = EvalReport(model=model_version, gate=normalize_gate(gate))
|
|
307
327
|
total = len(items)
|
|
308
328
|
# Trace linking runs here rather than inside each evaluate_agentic_*, so an item's
|
|
309
329
|
# Langfuse poll overlaps the NEXT item's agent call instead of extending its own
|
|
@@ -325,6 +345,9 @@ def run_agentic_items(
|
|
|
325
345
|
test_kind=item.test_kind,
|
|
326
346
|
question=item.question,
|
|
327
347
|
)
|
|
348
|
+
# None, not False, for the kinds _dispatch_agentic passes no gate to: ItemReport.passed
|
|
349
|
+
# then falls back to pass_at_k and gate_passed keeps meaning "a gate ran".
|
|
350
|
+
gated = item.test_kind not in UNGATED_AGENTIC_TEST_KINDS
|
|
328
351
|
t0 = time.perf_counter()
|
|
329
352
|
try:
|
|
330
353
|
outcome = _dispatch_agentic(
|
|
@@ -339,6 +362,7 @@ def run_agentic_items(
|
|
|
339
362
|
reasoning_effort,
|
|
340
363
|
agent_id,
|
|
341
364
|
submit_trace_link=linker.submit,
|
|
365
|
+
gate=gate,
|
|
342
366
|
)
|
|
343
367
|
if isinstance(outcome, AgenticEvalOutcome):
|
|
344
368
|
reasoning_steps = outcome.reasoning_steps
|
|
@@ -347,6 +371,8 @@ def run_agentic_items(
|
|
|
347
371
|
detail = outcome.detail
|
|
348
372
|
else:
|
|
349
373
|
reasoning_steps, conversation_id, response_id, detail = outcome, None, None, {}
|
|
374
|
+
item_report.gate_passed = True if gated else None
|
|
375
|
+
# Whichever gate decided the item, clearing it means at least one run passed.
|
|
350
376
|
item_report.pass_at_k = True
|
|
351
377
|
item_report.runs = k
|
|
352
378
|
item_report.reasoning_steps = reasoning_steps or []
|
|
@@ -356,7 +382,7 @@ def run_agentic_items(
|
|
|
356
382
|
_apply_timings(item_report, getattr(outcome, "timings", None))
|
|
357
383
|
_apply_run_counts(item_report, outcome)
|
|
358
384
|
except AssertionError as exc:
|
|
359
|
-
item_report.
|
|
385
|
+
item_report.gate_passed = False if gated else None
|
|
360
386
|
item_report.runs = k
|
|
361
387
|
item_report.reasoning_steps = getattr(exc, "reasoning_steps", None) or []
|
|
362
388
|
item_report.conversation_id = getattr(exc, "conversation_id", None)
|
|
@@ -364,6 +390,10 @@ def run_agentic_items(
|
|
|
364
390
|
item_report.best_detail = getattr(exc, "detail", None) or {}
|
|
365
391
|
_apply_timings(item_report, getattr(exc, "timings", None))
|
|
366
392
|
_apply_run_counts(item_report, exc)
|
|
393
|
+
# Read off the counts, not off the gate: pass^K fails items where runs did pass,
|
|
394
|
+
# and reporting those as pass_at_k False would contradict the Langfuse score of
|
|
395
|
+
# the same name. Kinds that report no count read as 0, i.e. a clean failure.
|
|
396
|
+
item_report.pass_at_k = item_report.runs_passed > 0
|
|
367
397
|
print(f"[agentic] {item.id} FAIL: {exc}", flush=True)
|
|
368
398
|
except Exception as exc:
|
|
369
399
|
item_report.error = f"{type(exc).__name__}: {exc}"
|
|
@@ -14,11 +14,20 @@ from gooddata_api_client.exceptions import ApiException
|
|
|
14
14
|
from rich.console import Console
|
|
15
15
|
from rich.table import Table
|
|
16
16
|
|
|
17
|
-
from gooddata_eval.cli.agentic_runner import AGENTIC_TEST_KINDS, run_agentic_items
|
|
17
|
+
from gooddata_eval.cli.agentic_runner import AGENTIC_TEST_KINDS, UNGATED_AGENTIC_TEST_KINDS, run_agentic_items
|
|
18
18
|
from gooddata_eval.core.chat.sse_client import ChatClient
|
|
19
|
-
from gooddata_eval.core.config import
|
|
19
|
+
from gooddata_eval.core.config import (
|
|
20
|
+
DEFAULT_GATE,
|
|
21
|
+
DEFAULT_JUDGE_MODEL,
|
|
22
|
+
JUDGE_MODEL_ENV_VAR,
|
|
23
|
+
EvalGate,
|
|
24
|
+
ReasoningEffort,
|
|
25
|
+
RunConfig,
|
|
26
|
+
normalize_gate,
|
|
27
|
+
)
|
|
20
28
|
from gooddata_eval.core.connection import ConnectionError_, resolve_connection
|
|
21
29
|
from gooddata_eval.core.dataset.local import load_local_dataset
|
|
30
|
+
from gooddata_eval.core.evaluators import supported_test_kinds
|
|
22
31
|
from gooddata_eval.core.langfuse.sink import LangfuseSink
|
|
23
32
|
from gooddata_eval.core.models import ChatResult, DatasetItem
|
|
24
33
|
from gooddata_eval.core.reporting.console import render_comparison, render_console
|
|
@@ -91,7 +100,15 @@ def _build_parser() -> argparse.ArgumentParser:
|
|
|
91
100
|
"Default: workspace's current active model."
|
|
92
101
|
),
|
|
93
102
|
)
|
|
94
|
-
run.add_argument("--runs", type=int, default=2, help="Independent runs per item
|
|
103
|
+
run.add_argument("--runs", type=int, default=2, help="Independent runs per item. Default 2.")
|
|
104
|
+
run.add_argument(
|
|
105
|
+
"--gate",
|
|
106
|
+
choices=get_args(EvalGate),
|
|
107
|
+
default=DEFAULT_GATE,
|
|
108
|
+
help="Which verdict decides an item: 'any' = pass@K (a run passing is enough, the "
|
|
109
|
+
"default and historic behaviour), 'power' = pass^K (every run must pass, so the verdict "
|
|
110
|
+
"measures stability). Identical at --runs 1. Agentic kinds only.",
|
|
111
|
+
)
|
|
95
112
|
run.add_argument(
|
|
96
113
|
"--concurrency",
|
|
97
114
|
type=int,
|
|
@@ -180,16 +197,46 @@ def _apply_timer_flag(enabled: bool) -> None:
|
|
|
180
197
|
os.environ[TIMERS_ENV_VAR] = "1"
|
|
181
198
|
|
|
182
199
|
|
|
200
|
+
def _reject_power_gate_on_ungated_items(config: RunConfig, items: list) -> None:
|
|
201
|
+
"""Refuse a pass^K request the run cannot honour for every item.
|
|
202
|
+
|
|
203
|
+
Two kinds of item are never gated: everything on the non-agentic path, because
|
|
204
|
+
`run_items` has no gate and always decides on pass@K, and agentic_conversation, which
|
|
205
|
+
drives its fixture once whatever --runs says and so has no K to gate over. Running a
|
|
206
|
+
mixed dataset anyway would decide part of it under each rule and label the whole report
|
|
207
|
+
`power`. test_kind is resolved per item, so a dataset does not have to be homogeneous.
|
|
208
|
+
|
|
209
|
+
Kinds no evaluator supports are not counted: those items are skipped rather than
|
|
210
|
+
decided, so refusing on them would make --gate power fail where --gate any runs.
|
|
211
|
+
"""
|
|
212
|
+
if normalize_gate(config.gate) != "power":
|
|
213
|
+
return
|
|
214
|
+
supported = supported_test_kinds()
|
|
215
|
+
ungated = [
|
|
216
|
+
i
|
|
217
|
+
for i in items
|
|
218
|
+
if i.test_kind in UNGATED_AGENTIC_TEST_KINDS
|
|
219
|
+
or (i.test_kind not in AGENTIC_TEST_KINDS and i.test_kind in supported)
|
|
220
|
+
]
|
|
221
|
+
if not ungated:
|
|
222
|
+
return
|
|
223
|
+
kinds = sorted({i.test_kind for i in ungated})
|
|
224
|
+
raise ValueError(
|
|
225
|
+
f"--gate power applies to kinds that repeat K runs, but this dataset has {len(ungated)} "
|
|
226
|
+
f"item(s) of kind {kinds}, which are always decided on pass@K. Run them separately, or "
|
|
227
|
+
f"use --gate any."
|
|
228
|
+
)
|
|
229
|
+
|
|
230
|
+
|
|
183
231
|
def _warn_if_local_dataset_cannot_link(config: RunConfig, agentic_items: list) -> None:
|
|
184
|
-
"""Say up front that
|
|
232
|
+
"""Say up front that experiment assembly will fail, rather than after the run.
|
|
185
233
|
|
|
186
234
|
--langfuse is refused outright with a local dataset because local item ids cannot be
|
|
187
235
|
linked. But every evaluate_agentic_* falls back to try_make_langfuse_client() when the
|
|
188
|
-
caller passes none, so with LANGFUSE_* exported the linking runs anyway and
|
|
189
|
-
|
|
190
|
-
|
|
191
|
-
|
|
192
|
-
instead of disabling it.
|
|
236
|
+
caller passes none, so with LANGFUSE_* exported the linking runs anyway and every
|
|
237
|
+
dataset-item lookup 404s -- arriving in a block at the very end of the run, long after
|
|
238
|
+
the flag that would have prevented it could be changed. The fallback is deliberate
|
|
239
|
+
(direct library and tavern callers rely on it), so this warns instead of disabling it.
|
|
193
240
|
"""
|
|
194
241
|
from gooddata_eval.core.agentic._langfuse import SKIP_ENV_VAR, langfuse_credentials_present # noqa: PLC0415
|
|
195
242
|
from gooddata_eval.core.config import env_flag # noqa: PLC0415
|
|
@@ -201,7 +248,7 @@ def _warn_if_local_dataset_cannot_link(config: RunConfig, agentic_items: list) -
|
|
|
201
248
|
print(
|
|
202
249
|
f"warning: --dataset is a local folder, so its item ids are not Langfuse dataset item ids. "
|
|
203
250
|
f"Traces will be found and scored, but the per-run grouping that makes models comparable "
|
|
204
|
-
f"cannot be created and each conversation will report
|
|
251
|
+
f"cannot be created and each conversation will report that its item does not exist in Langfuse. "
|
|
205
252
|
f"Use --langfuse-dataset for comparable runs, or set {SKIP_ENV_VAR}=1 to skip trace linking.",
|
|
206
253
|
file=sys.stderr,
|
|
207
254
|
)
|
|
@@ -253,7 +300,7 @@ def _make_progress_callbacks(console: Console):
|
|
|
253
300
|
tag = "[yellow]SKIP[/yellow]"
|
|
254
301
|
elif report.error:
|
|
255
302
|
tag = "[red]ERR [/red]"
|
|
256
|
-
elif report.
|
|
303
|
+
elif report.passed:
|
|
257
304
|
tag = "[green]PASS[/green]"
|
|
258
305
|
else:
|
|
259
306
|
tag = "[red]FAIL[/red]"
|
|
@@ -343,6 +390,7 @@ def _run(config: RunConfig) -> int:
|
|
|
343
390
|
items = _load_dataset(config)
|
|
344
391
|
agentic_items = [i for i in items if i.test_kind in AGENTIC_TEST_KINDS]
|
|
345
392
|
non_agentic_items = [i for i in items if i.test_kind not in AGENTIC_TEST_KINDS]
|
|
393
|
+
_reject_power_gate_on_ungated_items(config, items)
|
|
346
394
|
_warn_if_local_dataset_cannot_link(config, agentic_items)
|
|
347
395
|
models = config.models or []
|
|
348
396
|
run_ts = datetime.now(timezone.utc).strftime("%Y-%m-%d-%H-%M")
|
|
@@ -417,6 +465,7 @@ def _run(config: RunConfig) -> int:
|
|
|
417
465
|
token=config.token,
|
|
418
466
|
workspace_id=config.workspace_id,
|
|
419
467
|
k=config.runs,
|
|
468
|
+
gate=config.gate,
|
|
420
469
|
model_version=resolved.model_id,
|
|
421
470
|
reasoning_effort=config.reasoning_effort,
|
|
422
471
|
use_langfuse=config.log_to_langfuse,
|
|
@@ -466,6 +515,7 @@ def _run(config: RunConfig) -> int:
|
|
|
466
515
|
provider_name=resolved.provider_name or resolved.provider_id,
|
|
467
516
|
provider_type=resolved.provider_type,
|
|
468
517
|
workspace_id=config.workspace_id,
|
|
518
|
+
gate=config.gate,
|
|
469
519
|
)
|
|
470
520
|
if agentic_report is not None:
|
|
471
521
|
report.items.extend(agentic_report.items)
|
|
@@ -530,6 +580,7 @@ def main(argv: list[str] | None = None) -> int:
|
|
|
530
580
|
kind=args.kind,
|
|
531
581
|
preserve_failed=args.preserve_failed,
|
|
532
582
|
reasoning_effort=args.reasoning_effort,
|
|
583
|
+
gate=normalize_gate(args.gate),
|
|
533
584
|
agent_id=args.agent_id or os.environ.get("GD_EVAL_AGENT_ID"),
|
|
534
585
|
)
|
|
535
586
|
return _run(config)
|