gooddata-eval 1.74.0__tar.gz → 1.74.1.dev2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- gooddata_eval-1.74.1.dev2/AGENTS.md +114 -0
- gooddata_eval-1.74.1.dev2/CLAUDE.md +1 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/PKG-INFO +111 -5
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/README.md +108 -2
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/pyproject.toml +5 -5
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/cli/agentic_runner.py +188 -4
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/cli/main.py +70 -2
- gooddata_eval-1.74.1.dev2/src/gooddata_eval/core/_output.py +18 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/agentic/_catalog.py +41 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/agentic/_langfuse.py +185 -29
- gooddata_eval-1.74.1.dev2/src/gooddata_eval/core/agentic/_trace_linker.py +302 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/agentic/alert_skill.py +263 -110
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/agentic/conversation.py +199 -106
- gooddata_eval-1.74.1.dev2/src/gooddata_eval/core/agentic/general_question.py +344 -0
- gooddata_eval-1.74.1.dev2/src/gooddata_eval/core/agentic/guardrail.py +306 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/agentic/kda_skill.py +128 -93
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/agentic/metric_skill.py +123 -147
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/agentic/search_tool.py +78 -63
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/agentic/visualization.py +94 -96
- gooddata_eval-1.74.1.dev2/src/gooddata_eval/core/chat/render.py +47 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/chat/sse_client.py +42 -4
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/config.py +25 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/dataset/langfuse_source.py +53 -22
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/evaluators/__init__.py +13 -5
- gooddata_eval-1.74.1.dev2/src/gooddata_eval/core/evaluators/_llm_judge.py +311 -0
- gooddata_eval-1.74.1.dev2/src/gooddata_eval/core/evaluators/_maql.py +103 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/evaluators/_text_utils.py +3 -4
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/evaluators/alert_skill.py +11 -2
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/evaluators/general_question.py +19 -13
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/evaluators/guardrail.py +20 -17
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/evaluators/metric_skill.py +16 -4
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/evaluators/summary.py +56 -13
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/models.py +80 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/reporting/console.py +34 -10
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/reporting/json_report.py +26 -2
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/runner.py +68 -2
- gooddata_eval-1.74.1.dev2/src/gooddata_eval/core/timing.py +74 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/tests/test_agentic_alert_skill.py +334 -59
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/tests/test_agentic_conversation.py +277 -1
- gooddata_eval-1.74.1.dev2/tests/test_agentic_general_question.py +730 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/tests/test_agentic_guardrail.py +108 -82
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/tests/test_agentic_kda_skill.py +199 -208
- gooddata_eval-1.74.1.dev2/tests/test_agentic_langfuse_trace.py +552 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/tests/test_agentic_metric_skill.py +181 -101
- gooddata_eval-1.74.1.dev2/tests/test_agentic_runner.py +768 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/tests/test_agentic_search_tool.py +42 -0
- gooddata_eval-1.74.1.dev2/tests/test_chat_render.py +118 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/tests/test_cli.py +188 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/tests/test_connection.py +5 -0
- gooddata_eval-1.74.1.dev2/tests/test_langfuse_source.py +198 -0
- gooddata_eval-1.74.1.dev2/tests/test_llm_judge.py +616 -0
- gooddata_eval-1.74.1.dev2/tests/test_maql_normalize.py +106 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/tests/test_models.py +27 -0
- gooddata_eval-1.74.1.dev2/tests/test_reporting.py +444 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/tests/test_runner.py +60 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/tests/test_sse_client.py +54 -0
- gooddata_eval-1.74.1.dev2/tests/test_summary_evaluator.py +192 -0
- gooddata_eval-1.74.1.dev2/tests/test_text_evaluators.py +132 -0
- gooddata_eval-1.74.1.dev2/tests/test_timing.py +93 -0
- gooddata_eval-1.74.1.dev2/tests/test_trace_linker.py +568 -0
- gooddata_eval-1.74.0/src/gooddata_eval/core/agentic/general_question.py +0 -264
- gooddata_eval-1.74.0/src/gooddata_eval/core/agentic/guardrail.py +0 -268
- gooddata_eval-1.74.0/src/gooddata_eval/core/evaluators/_llm_judge.py +0 -66
- gooddata_eval-1.74.0/tests/test_agentic_general_question.py +0 -210
- gooddata_eval-1.74.0/tests/test_agentic_langfuse_trace.py +0 -26
- gooddata_eval-1.74.0/tests/test_agentic_runner.py +0 -220
- gooddata_eval-1.74.0/tests/test_langfuse_source.py +0 -105
- gooddata_eval-1.74.0/tests/test_llm_judge.py +0 -45
- gooddata_eval-1.74.0/tests/test_reporting.py +0 -194
- gooddata_eval-1.74.0/tests/test_summary_evaluator.py +0 -87
- gooddata_eval-1.74.0/tests/test_text_evaluators.py +0 -72
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/.gitignore +0 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/LICENSE.txt +0 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/Makefile +0 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/__init__.py +0 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/_version.py +0 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/cli/__init__.py +0 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/__init__.py +0 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/agentic/__init__.py +0 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/chat/__init__.py +0 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/connection.py +0 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/dataset/__init__.py +0 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/dataset/local.py +0 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/evaluators/_deep_subset.py +0 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/evaluators/base.py +0 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/evaluators/search_tool.py +0 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/evaluators/visualization.py +0 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/langfuse/__init__.py +0 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/langfuse/sink.py +0 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/reporting/__init__.py +0 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/scoring.py +0 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/summary/__init__.py +0 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/summary/http_client.py +0 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/workspace.py +0 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/tests/__init__.py +0 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/tests/conftest.py +0 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/tests/fixtures/sample_dataset/metric_skill_create.json +0 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/tests/fixtures/sample_dataset/visualization_revenue.json +0 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/tests/fixtures/sse_visualization_stream.txt +0 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/tests/test_agentic_run_context.py +0 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/tests/test_agentic_visualization.py +0 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/tests/test_alert_skill_evaluator.py +0 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/tests/test_deep_subset.py +0 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/tests/test_langfuse_sink.py +0 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/tests/test_local_loader.py +0 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/tests/test_metric_skill_evaluator.py +0 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/tests/test_scoring.py +0 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/tests/test_search_tool_evaluator.py +0 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/tests/test_summary_client.py +0 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/tests/test_visualization_evaluator.py +0 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/tests/test_workspace.py +0 -0
- {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/tox.ini +0 -0
|
@@ -0,0 +1,114 @@
|
|
|
1
|
+
# gooddata-eval
|
|
2
|
+
|
|
3
|
+
`gd-eval` — a CLI and library that drives the GoodData AI agent (a separate service, in
|
|
4
|
+
`gdc-nas`) through a dataset of natural-language questions and scores what comes back,
|
|
5
|
+
including side-by-side comparison across models. Each dataset item is a JSON envelope
|
|
6
|
+
loaded from a local folder or pulled from a Langfuse dataset. Results are aggregated into
|
|
7
|
+
pass@K / pass^K reports and optionally pushed to Langfuse as scored traces tied to a
|
|
8
|
+
dataset run. The newest and most actively developed package in the repo.
|
|
9
|
+
|
|
10
|
+
## Owns
|
|
11
|
+
|
|
12
|
+
- The `gd-eval` CLI (`gd-eval run`, `gd-eval models`)
|
|
13
|
+
- Dataset loading and the evaluation run loop
|
|
14
|
+
- Per-capability evaluators and their scoring
|
|
15
|
+
- Result reporting, and pushing runs, scores and trace links to Langfuse
|
|
16
|
+
|
|
17
|
+
## Does NOT Own
|
|
18
|
+
|
|
19
|
+
- The agent under evaluation — that lives in `gdc-nas` (gen-ai)
|
|
20
|
+
- Platform access → `gooddata-sdk`
|
|
21
|
+
|
|
22
|
+
## Architecture
|
|
23
|
+
|
|
24
|
+
| Path | Role |
|
|
25
|
+
|---|---|
|
|
26
|
+
| `cli/` | argument parsing, and `agentic_runner` — the agentic dispatch and concurrency phases |
|
|
27
|
+
| `core/agentic/` | multi-turn agentic evaluation per capability, **plus** all Langfuse trace polling and linking (`_langfuse.py`, `_trace_linker.py`) |
|
|
28
|
+
| `core/chat/` | SSE client for the agent's streaming chat endpoint |
|
|
29
|
+
| `core/summary/` | HTTP client for the dedicated dashboard-summary endpoint — a single-shot chat backend, not reporting |
|
|
30
|
+
| `core/dataset/` | dataset format and loading |
|
|
31
|
+
| `core/evaluators/` | single-shot evaluators and their registry |
|
|
32
|
+
| `core/langfuse/` | `sink.py` only — pushes single-turn scores and dataset-run items |
|
|
33
|
+
| `core/reporting/` | console and JSON output rendering |
|
|
34
|
+
| `core/scoring.py`, `core/runner.py` | scoring and orchestration |
|
|
35
|
+
| `core/models.py` | `DatasetItem`, `ChatResult`, `ItemReport` and friends |
|
|
36
|
+
|
|
37
|
+
**Depends on**: `gooddata-sdk`, `httpx`, `pydantic`, `orjson`, `rich`. The LLM-judge
|
|
38
|
+
evaluator is an optional extra (`llm-judge`, pulling `openai>=1.45,<2.0`); every `openai`
|
|
39
|
+
import site is guarded or deferred so the base install stays usable without it — keep it
|
|
40
|
+
that way.
|
|
41
|
+
|
|
42
|
+
### Two evaluation paths that share almost nothing
|
|
43
|
+
|
|
44
|
+
- **Single-shot** kinds send one chat turn and are scored by an `Evaluator` (a Protocol:
|
|
45
|
+
a `test_kind` attribute plus `evaluate(item, chat_result) -> ItemEvaluation`) looked up
|
|
46
|
+
from a registry in `core/evaluators/__init__.py`.
|
|
47
|
+
- **Agentic** kinds (`agentic_*`, `vis_agentic`) drive a full multi-turn conversation over
|
|
48
|
+
the SSE endpoint and are dispatched by an explicit `if`/`elif` chain in
|
|
49
|
+
`cli/agentic_runner.py`.
|
|
50
|
+
|
|
51
|
+
Many capabilities exist in **both** forms — visualization, metric skill, alert skill,
|
|
52
|
+
search, general question and guardrail each have a single-turn and a multi-turn
|
|
53
|
+
implementation, sometimes under different `test_kind` strings (`search_tool` vs
|
|
54
|
+
`agentic_search`). These are parallel implementations, not layers.
|
|
55
|
+
|
|
56
|
+
### Dataset items
|
|
57
|
+
|
|
58
|
+
`DatasetItem` is the envelope: `id`, `dataset_name`, `test_kind`, `question`, and
|
|
59
|
+
`expected_output: Any`. `expected_output` is deliberately untyped — each evaluator parses
|
|
60
|
+
its own shape. `test_kind` on the item is what labels the result, not the evaluator class,
|
|
61
|
+
which is why `knowledge_question` can reuse `GeneralQuestionEvaluator` verbatim.
|
|
62
|
+
`dashboard_summary` items additionally need `summary_input`.
|
|
63
|
+
|
|
64
|
+
## Gotchas
|
|
65
|
+
|
|
66
|
+
**Adding an evaluator is a registry change, not a naming convention.** Single-shot kinds go
|
|
67
|
+
into `_EAGER_EVALUATORS`, or `_LAZY_EVALUATOR_MODULES` plus `_LAZY_EVALUATOR_CLASSES`, in
|
|
68
|
+
`core/evaluators/__init__.py`. Agentic kinds need the string added to `AGENTIC_TEST_KINDS`
|
|
69
|
+
and a new branch in `_dispatch_agentic`. Test file naming follows the capability, but
|
|
70
|
+
naming a test file correctly registers nothing.
|
|
71
|
+
|
|
72
|
+
**Parallel-safety is a reviewed allowlist, and getting it wrong corrupts results.**
|
|
73
|
+
`WORKSPACE_MUTATING_TEST_KINDS` is computed as `AGENTIC_TEST_KINDS - PARALLEL_SAFE_TEST_KINDS`,
|
|
74
|
+
so a newly added kind defaults to workspace-mutating and runs serially in its own phase.
|
|
75
|
+
That default is correct: agent tool calls create real server-side objects (metrics, alerts).
|
|
76
|
+
Adding a kind to `PARALLEL_SAFE_TEST_KINDS` is a deliberate assertion that it is read-only,
|
|
77
|
+
which nothing in the package can prove for you.
|
|
78
|
+
|
|
79
|
+
**The SSE client's retry predicate is load-bearing.** `core/chat/sse_client.py` retries
|
|
80
|
+
429/502/503/504 and `httpx.RemoteProtocolError` (a mid-stream disconnect) with exponential
|
|
81
|
+
backoff, and treats a `METADATA_SYNC_IN_PROGRESS` payload as transient. The
|
|
82
|
+
`RemoteProtocolError` case was added after it was confirmed live to contaminate a small
|
|
83
|
+
percentage of visualization runs with a hard fail and no retry. Narrowing that predicate
|
|
84
|
+
reintroduces the problem.
|
|
85
|
+
|
|
86
|
+
**Langfuse trace linking is deliberately off the item critical path.** Polling for trace
|
|
87
|
+
ingestion has no pass/fail signal and inflates or misattributes per-item latency, so
|
|
88
|
+
`BackgroundTraceLinker` defers it and is drained before the report renders
|
|
89
|
+
(`run_trace_link_inline` is the synchronous alternative). Do not "fix" a slow item by
|
|
90
|
+
making trace scoring synchronous again.
|
|
91
|
+
|
|
92
|
+
**Scoring weights do not sum to 1.** `quality_score` is the fraction of boolean-valued keys
|
|
93
|
+
in `best_detail` that are true, falling back to `pass_at_k` when there are none (text
|
|
94
|
+
evaluators). `value_score` is `0.6 * quality + 0.2 * speed` — the 0.8 total is what the
|
|
95
|
+
code does; treat it as intentional unless you have checked with the owner.
|
|
96
|
+
|
|
97
|
+
### Fixture shapes
|
|
98
|
+
|
|
99
|
+
Group-by / attribute expectations in the alert-skill fixtures are written in AAC shape,
|
|
100
|
+
while the tool arguments the agent emits are AFM-shaped. Never deep-compare those two
|
|
101
|
+
directly — convert, or compare field by field. This applies specifically to the
|
|
102
|
+
attribute/group-by fields: `Filters` in the same fixtures is AFM-shaped on both sides and
|
|
103
|
+
is correctly deep-compared as-is. The attribute comparison itself lands with the
|
|
104
|
+
alert group-by work currently on `jt/gdai-2175-eval-alert-attributes`, so on `master` this
|
|
105
|
+
is guidance for the incoming code rather than a description of what is already there.
|
|
106
|
+
|
|
107
|
+
## Testing
|
|
108
|
+
|
|
109
|
+
Plain pytest under `tests/`, no cassettes — the agent is stubbed with
|
|
110
|
+
`unittest.mock`. Tests are named per capability (`test_agentic_*.py`), which is the
|
|
111
|
+
convention to follow when adding one.
|
|
112
|
+
|
|
113
|
+
`ty` is configured here with `allowed-unresolved-imports` for `openai.**` and
|
|
114
|
+
`gooddata_api_client.**`; do not widen that list to paper over a real typing problem.
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
@AGENTS.md
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: gooddata-eval
|
|
3
|
-
Version: 1.74.
|
|
3
|
+
Version: 1.74.1.dev2
|
|
4
4
|
Summary: Evaluate the GoodData AI agent against your own questions and models.
|
|
5
5
|
Project-URL: Source, https://github.com/gooddata/gooddata-python-sdk
|
|
6
6
|
Author-email: GoodData <support@gooddata.com>
|
|
@@ -17,13 +17,13 @@ Classifier: Topic :: Scientific/Engineering
|
|
|
17
17
|
Classifier: Topic :: Software Development
|
|
18
18
|
Classifier: Typing :: Typed
|
|
19
19
|
Requires-Python: >=3.10
|
|
20
|
-
Requires-Dist: gooddata-sdk~=1.74.
|
|
20
|
+
Requires-Dist: gooddata-sdk~=1.74.1.dev2
|
|
21
21
|
Requires-Dist: httpx<1.0,>=0.27
|
|
22
22
|
Requires-Dist: orjson<4.0.0,>=3.9.15
|
|
23
23
|
Requires-Dist: pydantic<3.0,>=2.6
|
|
24
24
|
Requires-Dist: rich<15.0,>=13.0
|
|
25
25
|
Provides-Extra: llm-judge
|
|
26
|
-
Requires-Dist: openai<2.0,>=1.
|
|
26
|
+
Requires-Dist: openai<2.0,>=1.45; extra == 'llm-judge'
|
|
27
27
|
Description-Content-Type: text/markdown
|
|
28
28
|
|
|
29
29
|
# gooddata-eval
|
|
@@ -142,6 +142,7 @@ gd-eval run \
|
|
|
142
142
|
|---|---|
|
|
143
143
|
| `--dataset PATH` | Flat folder of JSON files — one question per file. |
|
|
144
144
|
| `--langfuse-dataset NAME` | Pull items by name from a Langfuse dataset. Requires `LANGFUSE_PUBLIC_KEY`, `LANGFUSE_SECRET_KEY`, `LANGFUSE_HOST`. |
|
|
145
|
+
| `--kind TEST_KIND` | Fallback `test_kind` for dataset items that do not embed one. Defaults to `visualization`; use e.g. `agentic_metric_skill` for multi-turn agentic evaluation. Items that declare their own `test_kind` ignore this. |
|
|
145
146
|
|
|
146
147
|
#### Model selection
|
|
147
148
|
|
|
@@ -154,21 +155,62 @@ gd-eval run \
|
|
|
154
155
|
| Flag | Default | Description |
|
|
155
156
|
|---|---|---|
|
|
156
157
|
| `--runs K` | `2` | Independent runs per item (pass@K). An item passes if any run passes. |
|
|
157
|
-
| `--concurrency K` | `1` | Number of items evaluated concurrently. `1` = sequential (default). Increase to load-test the agent under simultaneous requests
|
|
158
|
+
| `--concurrency K` | `1` | Number of items evaluated concurrently. `1` = sequential (default). Increase to load-test the agent under simultaneous requests — see *Concurrency and workspace safety* below. |
|
|
159
|
+
| `--judge-model MODEL` | `gpt-4o` | Model used for LLM-as-judge scoring — `agentic_general_question`, `agentic_guardrail`, `general_question`, `guardrail` and `dashboard_summary`. Also settable via `GD_EVAL_JUDGE_MODEL`. Two things to weigh before changing it: the gpt-5 family rejects `temperature=0`, so verdicts stop being reproducible (the run warns when this happens); and choosing the same model the agent runs means the judge grades its own family's output. |
|
|
158
160
|
| `--reasoning-effort LEVEL` | server default | `LOW`, `MEDIUM` or `HIGH`, sent as `options.reasoningEffort` on every chat message. Requires the `enableGenAiReasoningEffort` feature flag on the target organization — without it the server ignores the value. Applies to chat items only; `dashboard_summary` items go through the summary endpoint, which has no such option. |
|
|
159
161
|
|
|
162
|
+
**Concurrency and workspace safety.** Agentic kinds that create workspace objects
|
|
163
|
+
(`agentic_metric_skill`, `agentic_alert_skill`, `agentic_conversation`, `agentic_kda_skill`) always run one at a
|
|
164
|
+
time whatever `--concurrency` says — a metric or alert created and dropped mid-run would otherwise be visible to
|
|
165
|
+
another item reading the same catalog. **That protection is for the agentic kinds only:** the single-turn
|
|
166
|
+
`metric_skill` and `alert_skill` kinds are still fanned out and the agent performs the same server-side writes on
|
|
167
|
+
that path, so avoid raising `--concurrency` on a dataset of those against a shared workspace. Progress output
|
|
168
|
+
interleaves when K > 1, and per-item latencies rise, so they stop being clean single-request measurements.
|
|
169
|
+
|
|
160
170
|
#### Output
|
|
161
171
|
|
|
162
172
|
| Flag | Description |
|
|
163
173
|
|---|---|
|
|
164
174
|
| `--json PATH` | Write a JSON report to this path. Always uses the nested `{models, runs, comparison}` shape even for a single model. |
|
|
165
175
|
| `--quiet` | Suppress per-item progress. Per-model result tables and the comparison summary are still printed. |
|
|
176
|
+
| `--preserve-failed` | Keep failed conversations on the server instead of deleting them, so they can be inspected afterwards. Applies to the single-turn chat path; agentic kinds manage their own conversation lifecycle. |
|
|
177
|
+
| `--timers` | Print per-turn `[timer]` diagnostics — GoodData response, judge, and simulated-user seconds as they happen. Off by default: an 18-item `--runs 2` run emits ~72 lines and buries the progress output. The same measurements are always in the JSON report's `latency_breakdown_s`, so this only adds a live view. Also settable via `GD_EVAL_TIMERS=1`. |
|
|
166
178
|
|
|
167
179
|
#### Langfuse sink
|
|
168
180
|
|
|
169
181
|
| Flag | Description |
|
|
170
182
|
|---|---|
|
|
171
|
-
| `--langfuse` | Log scores and traces to Langfuse after each item. Requires `--langfuse-dataset`.
|
|
183
|
+
| `--langfuse` | Log scores and traces to Langfuse after each item. Requires `--langfuse-dataset`. Names each experiment run `{dataset_name}_{timestamp}_{model}`, suffixed `_effort-{level}` when `--reasoning-effort` is set (so runs differing only by effort stay separate) and `_run{N}` per run when `--runs` > 1 — e.g. `general_question_2026-09-02-11-13_gpt-5.2_run0`. Requires `LANGFUSE_PUBLIC_KEY`, `LANGFUSE_SECRET_KEY`, `LANGFUSE_HOST`. |
|
|
184
|
+
|
|
185
|
+
Set `TAVERN_E2E_SKIP_TRACE_LINK=1` to skip trace lookup entirely (scores are then orphaned; the run says so).
|
|
186
|
+
|
|
187
|
+
**A local `--dataset` cannot be attached to a Langfuse run.** `--langfuse` is refused alongside `--dataset`
|
|
188
|
+
because a local folder's item ids are not Langfuse dataset item ids. But trace linking does not depend on that
|
|
189
|
+
flag — each `evaluate_agentic_*` builds its own client whenever `LANGFUSE_PUBLIC_KEY`/`LANGFUSE_SECRET_KEY` are
|
|
190
|
+
exported — so a local run still finds its traces and writes its scores onto them, and only the per-run grouping
|
|
191
|
+
fails, with one `404 from dataset-run-items` reported per run. The run warns about this before it starts. Use
|
|
192
|
+
`--langfuse-dataset` when you want runs that are comparable across models, or `TAVERN_E2E_SKIP_TRACE_LINK=1` to
|
|
193
|
+
skip linking altogether.
|
|
194
|
+
|
|
195
|
+
**When trace linking happens.** Finding a gen-ai trace means polling until Langfuse has ingested it, which is
|
|
196
|
+
lag measured in seconds to minutes. That work produces no verdict — the pass/fail is already decided — so it
|
|
197
|
+
does not run inline per item. Every item's Langfuse block is queued and the whole batch runs *after* the agent
|
|
198
|
+
phase, draining before any report is written. Two consequences worth knowing:
|
|
199
|
+
|
|
200
|
+
- **No item's `latency_s` includes trace linking.** Its cost is reported separately as
|
|
201
|
+
`latency_breakdown_s.langfuse_s`, and the run prints
|
|
202
|
+
`[langfuse] trace linking finished in Xs for N item(s); slowest Ys`. If `slowest` approaches the **120s**
|
|
203
|
+
batched retry budget, links are timing out and scores are being orphaned — look for
|
|
204
|
+
`[langfuse] WARNING: no trace found for conversation ...`.
|
|
205
|
+
- **The budget depends on who is waiting.** 120s is affordable only because the batch blocks nobody. A direct
|
|
206
|
+
library caller (`evaluate_agentic_*` without a `submit_trace_link`) polls inline, on its own critical path, and
|
|
207
|
+
gets **35s** instead — the same cost as before batching existed, so no inline caller pays for a budget raised
|
|
208
|
+
on the CLI's behalf. Either way a trace that is already ingested costs nothing: the loop looks before it sleeps.
|
|
209
|
+
Scores are always final before the command exits — the run blocks on the batch. Interrupting with Ctrl-C drops
|
|
210
|
+
whatever is still queued rather than making you wait it out: both the queued trace links and, under
|
|
211
|
+
`--concurrency`, the items that have not started. The handful of items already in flight still have to finish —
|
|
212
|
+
worker threads are joined at exit and an in-progress agent call cannot be cancelled — so expect to wait up to one
|
|
213
|
+
`--concurrency`-wide wave, not the rest of the dataset.
|
|
172
214
|
|
|
173
215
|
### JSON report shape
|
|
174
216
|
|
|
@@ -190,6 +232,70 @@ The JSON report always uses the nested multi-model shape:
|
|
|
190
232
|
|
|
191
233
|
Winner is selected by **pass rate → quality score → latency** (lower latency wins all-equal ties).
|
|
192
234
|
|
|
235
|
+
Each item reports **how many of its runs passed**, not only whether one did:
|
|
236
|
+
|
|
237
|
+
```json
|
|
238
|
+
"runs": 5, "runs_passed": 4, "pass_at_k": true, "pass_power_k": false
|
|
239
|
+
```
|
|
240
|
+
|
|
241
|
+
`pass_at_k` is "did any run pass" and is what `passed` counts. `runs_passed` is the fact that separates a
|
|
242
|
+
reliable item from a coin-flip — without it a 5/5 item and a 1/5 item are identical in every field, because
|
|
243
|
+
`quality_score` is derived from the best run alone. `pass_power_k` is true only when every run passed, and the
|
|
244
|
+
run summary carries `passed_all_runs` beside `passed`; a large gap between the two means the model is
|
|
245
|
+
inconsistent rather than wrong. The console shows `4/5 runs passed` in `Notes` for a non-unanimous pass and
|
|
246
|
+
stays quiet for a unanimous one, and its summary line reads `3/4 passed, 1 on every run`.
|
|
247
|
+
|
|
248
|
+
`runs` is what the item actually ran, which is not always the requested `--runs`: `agentic_conversation` takes
|
|
249
|
+
no K and drives its fixture exactly once.
|
|
250
|
+
|
|
251
|
+
Each item additionally carries a per-phase breakdown:
|
|
252
|
+
|
|
253
|
+
```json
|
|
254
|
+
"latency_breakdown_s": {
|
|
255
|
+
"agent_s": 4.02, // GoodData's own response time — the system under test
|
|
256
|
+
"judge_s": 1.31, // LLM-as-judge scoring, post-hoc
|
|
257
|
+
"simulated_user_s": 0.0, // our simulated user composing the next turn (multi-turn kinds)
|
|
258
|
+
"langfuse_s": 5.70 // trace lookup + score writing, off the critical path
|
|
259
|
+
}
|
|
260
|
+
```
|
|
261
|
+
|
|
262
|
+
Every item reports `runs_ungraded` beside `runs_passed`: runs the agent answered but the LLM judge returned
|
|
263
|
+
nothing readable for. Agentic items also list them as `unscored_runs` / `judge_errors` in their `detail`, and a
|
|
264
|
+
`dashboard_summary` item carries `ungraded_criteria`. Such a run — or, for `dashboard_summary`, such a
|
|
265
|
+
criterion — is excluded from pass@K and from the quality score rather than counted as a failure: scoring it 0
|
|
266
|
+
would be indistinguishable from the judge genuinely failing the answer, which is the confusion
|
|
267
|
+
`JudgeResponseError` exists to end. `pass@K` still holds on the runs that *were* graded, so an item can pass with
|
|
268
|
+
`runs_ungraded` set; `pass^K` cannot, because a run nobody graded leaves "all K passed" unverified. For
|
|
269
|
+
`dashboard_summary` an ungraded `must_include` / `must_not_include` criterion likewise cannot carry a pass —
|
|
270
|
+
"the judge could not tell" is not evidence the fact is present — while an ungraded `rubric` line only narrows
|
|
271
|
+
the quality score. When *no* run could be graded (for `dashboard_summary`: no gating criterion on any run) the
|
|
272
|
+
item errors instead of reporting failures. A non-zero count means pass@K was computed over fewer runs than
|
|
273
|
+
`--runs` asked for, so treat the result as weaker evidence and check the judge (`GD_EVAL_JUDGE_DIAGNOSTICS=1`,
|
|
274
|
+
or raise `JUDGE_MAX_COMPLETION_TOKENS` if the cause is `finish_reason=length`). The console says so in `Notes`
|
|
275
|
+
(`1 run(s) ungraded`, `2 criterion(s) ungraded`).
|
|
276
|
+
|
|
277
|
+
`agent_s` + `judge_s` + `simulated_user_s` are the instrumented parts of the item's `latency_s`; they do not add
|
|
278
|
+
up to it exactly, because `latency_s` is wall-clock around the whole item and also covers the conversation
|
|
279
|
+
create/delete round trips, SDK construction and any cleanup. `langfuse_s` sits **beside** `latency_s`, never
|
|
280
|
+
inside it, because trace linking runs outside every item's critical path (see above) — summing all four would
|
|
281
|
+
re-inflate exactly what that design removes.
|
|
282
|
+
|
|
283
|
+
A phase that a kind does not have reports `0.0` rather than an invented number, so read the zeroes as "not
|
|
284
|
+
applicable here", not "instant". Today:
|
|
285
|
+
|
|
286
|
+
| Field | Populated by |
|
|
287
|
+
|---|---|
|
|
288
|
+
| `agent_s` | `agentic_general_question`, `agentic_metric_skill` |
|
|
289
|
+
| `judge_s` | `agentic_general_question` only — `agentic_metric_skill` compares MAQL by string, it has no LLM judge |
|
|
290
|
+
| `simulated_user_s` | `agentic_metric_skill` only — `agentic_general_question` is single-turn, it has no simulated user |
|
|
291
|
+
| `langfuse_s` | every agentic kind, but only on the `gd-eval` path and only when Langfuse credentials are present |
|
|
292
|
+
|
|
293
|
+
The other six agentic kinds report `0.0` for the first three. Trace linking itself happens whenever
|
|
294
|
+
`LANGFUSE_PUBLIC_KEY`/`LANGFUSE_SECRET_KEY` are exported, with or without `--langfuse`, because each
|
|
295
|
+
`evaluate_agentic_*` falls back to `try_make_langfuse_client()`. But its *duration* is measured by the CLI
|
|
296
|
+
runner rather than by `evaluate_agentic_*`, so a direct library caller sees `langfuse_s: 0.0` even though its
|
|
297
|
+
linking ran. Pass `TAVERN_E2E_SKIP_TRACE_LINK=1` to opt out of linking altogether.
|
|
298
|
+
|
|
193
299
|
---
|
|
194
300
|
|
|
195
301
|
## `gd-eval models`
|
|
@@ -114,6 +114,7 @@ gd-eval run \
|
|
|
114
114
|
|---|---|
|
|
115
115
|
| `--dataset PATH` | Flat folder of JSON files — one question per file. |
|
|
116
116
|
| `--langfuse-dataset NAME` | Pull items by name from a Langfuse dataset. Requires `LANGFUSE_PUBLIC_KEY`, `LANGFUSE_SECRET_KEY`, `LANGFUSE_HOST`. |
|
|
117
|
+
| `--kind TEST_KIND` | Fallback `test_kind` for dataset items that do not embed one. Defaults to `visualization`; use e.g. `agentic_metric_skill` for multi-turn agentic evaluation. Items that declare their own `test_kind` ignore this. |
|
|
117
118
|
|
|
118
119
|
#### Model selection
|
|
119
120
|
|
|
@@ -126,21 +127,62 @@ gd-eval run \
|
|
|
126
127
|
| Flag | Default | Description |
|
|
127
128
|
|---|---|---|
|
|
128
129
|
| `--runs K` | `2` | Independent runs per item (pass@K). An item passes if any run passes. |
|
|
129
|
-
| `--concurrency K` | `1` | Number of items evaluated concurrently. `1` = sequential (default). Increase to load-test the agent under simultaneous requests
|
|
130
|
+
| `--concurrency K` | `1` | Number of items evaluated concurrently. `1` = sequential (default). Increase to load-test the agent under simultaneous requests — see *Concurrency and workspace safety* below. |
|
|
131
|
+
| `--judge-model MODEL` | `gpt-4o` | Model used for LLM-as-judge scoring — `agentic_general_question`, `agentic_guardrail`, `general_question`, `guardrail` and `dashboard_summary`. Also settable via `GD_EVAL_JUDGE_MODEL`. Two things to weigh before changing it: the gpt-5 family rejects `temperature=0`, so verdicts stop being reproducible (the run warns when this happens); and choosing the same model the agent runs means the judge grades its own family's output. |
|
|
130
132
|
| `--reasoning-effort LEVEL` | server default | `LOW`, `MEDIUM` or `HIGH`, sent as `options.reasoningEffort` on every chat message. Requires the `enableGenAiReasoningEffort` feature flag on the target organization — without it the server ignores the value. Applies to chat items only; `dashboard_summary` items go through the summary endpoint, which has no such option. |
|
|
131
133
|
|
|
134
|
+
**Concurrency and workspace safety.** Agentic kinds that create workspace objects
|
|
135
|
+
(`agentic_metric_skill`, `agentic_alert_skill`, `agentic_conversation`, `agentic_kda_skill`) always run one at a
|
|
136
|
+
time whatever `--concurrency` says — a metric or alert created and dropped mid-run would otherwise be visible to
|
|
137
|
+
another item reading the same catalog. **That protection is for the agentic kinds only:** the single-turn
|
|
138
|
+
`metric_skill` and `alert_skill` kinds are still fanned out and the agent performs the same server-side writes on
|
|
139
|
+
that path, so avoid raising `--concurrency` on a dataset of those against a shared workspace. Progress output
|
|
140
|
+
interleaves when K > 1, and per-item latencies rise, so they stop being clean single-request measurements.
|
|
141
|
+
|
|
132
142
|
#### Output
|
|
133
143
|
|
|
134
144
|
| Flag | Description |
|
|
135
145
|
|---|---|
|
|
136
146
|
| `--json PATH` | Write a JSON report to this path. Always uses the nested `{models, runs, comparison}` shape even for a single model. |
|
|
137
147
|
| `--quiet` | Suppress per-item progress. Per-model result tables and the comparison summary are still printed. |
|
|
148
|
+
| `--preserve-failed` | Keep failed conversations on the server instead of deleting them, so they can be inspected afterwards. Applies to the single-turn chat path; agentic kinds manage their own conversation lifecycle. |
|
|
149
|
+
| `--timers` | Print per-turn `[timer]` diagnostics — GoodData response, judge, and simulated-user seconds as they happen. Off by default: an 18-item `--runs 2` run emits ~72 lines and buries the progress output. The same measurements are always in the JSON report's `latency_breakdown_s`, so this only adds a live view. Also settable via `GD_EVAL_TIMERS=1`. |
|
|
138
150
|
|
|
139
151
|
#### Langfuse sink
|
|
140
152
|
|
|
141
153
|
| Flag | Description |
|
|
142
154
|
|---|---|
|
|
143
|
-
| `--langfuse` | Log scores and traces to Langfuse after each item. Requires `--langfuse-dataset`.
|
|
155
|
+
| `--langfuse` | Log scores and traces to Langfuse after each item. Requires `--langfuse-dataset`. Names each experiment run `{dataset_name}_{timestamp}_{model}`, suffixed `_effort-{level}` when `--reasoning-effort` is set (so runs differing only by effort stay separate) and `_run{N}` per run when `--runs` > 1 — e.g. `general_question_2026-09-02-11-13_gpt-5.2_run0`. Requires `LANGFUSE_PUBLIC_KEY`, `LANGFUSE_SECRET_KEY`, `LANGFUSE_HOST`. |
|
|
156
|
+
|
|
157
|
+
Set `TAVERN_E2E_SKIP_TRACE_LINK=1` to skip trace lookup entirely (scores are then orphaned; the run says so).
|
|
158
|
+
|
|
159
|
+
**A local `--dataset` cannot be attached to a Langfuse run.** `--langfuse` is refused alongside `--dataset`
|
|
160
|
+
because a local folder's item ids are not Langfuse dataset item ids. But trace linking does not depend on that
|
|
161
|
+
flag — each `evaluate_agentic_*` builds its own client whenever `LANGFUSE_PUBLIC_KEY`/`LANGFUSE_SECRET_KEY` are
|
|
162
|
+
exported — so a local run still finds its traces and writes its scores onto them, and only the per-run grouping
|
|
163
|
+
fails, with one `404 from dataset-run-items` reported per run. The run warns about this before it starts. Use
|
|
164
|
+
`--langfuse-dataset` when you want runs that are comparable across models, or `TAVERN_E2E_SKIP_TRACE_LINK=1` to
|
|
165
|
+
skip linking altogether.
|
|
166
|
+
|
|
167
|
+
**When trace linking happens.** Finding a gen-ai trace means polling until Langfuse has ingested it, which is
|
|
168
|
+
lag measured in seconds to minutes. That work produces no verdict — the pass/fail is already decided — so it
|
|
169
|
+
does not run inline per item. Every item's Langfuse block is queued and the whole batch runs *after* the agent
|
|
170
|
+
phase, draining before any report is written. Two consequences worth knowing:
|
|
171
|
+
|
|
172
|
+
- **No item's `latency_s` includes trace linking.** Its cost is reported separately as
|
|
173
|
+
`latency_breakdown_s.langfuse_s`, and the run prints
|
|
174
|
+
`[langfuse] trace linking finished in Xs for N item(s); slowest Ys`. If `slowest` approaches the **120s**
|
|
175
|
+
batched retry budget, links are timing out and scores are being orphaned — look for
|
|
176
|
+
`[langfuse] WARNING: no trace found for conversation ...`.
|
|
177
|
+
- **The budget depends on who is waiting.** 120s is affordable only because the batch blocks nobody. A direct
|
|
178
|
+
library caller (`evaluate_agentic_*` without a `submit_trace_link`) polls inline, on its own critical path, and
|
|
179
|
+
gets **35s** instead — the same cost as before batching existed, so no inline caller pays for a budget raised
|
|
180
|
+
on the CLI's behalf. Either way a trace that is already ingested costs nothing: the loop looks before it sleeps.
|
|
181
|
+
Scores are always final before the command exits — the run blocks on the batch. Interrupting with Ctrl-C drops
|
|
182
|
+
whatever is still queued rather than making you wait it out: both the queued trace links and, under
|
|
183
|
+
`--concurrency`, the items that have not started. The handful of items already in flight still have to finish —
|
|
184
|
+
worker threads are joined at exit and an in-progress agent call cannot be cancelled — so expect to wait up to one
|
|
185
|
+
`--concurrency`-wide wave, not the rest of the dataset.
|
|
144
186
|
|
|
145
187
|
### JSON report shape
|
|
146
188
|
|
|
@@ -162,6 +204,70 @@ The JSON report always uses the nested multi-model shape:
|
|
|
162
204
|
|
|
163
205
|
Winner is selected by **pass rate → quality score → latency** (lower latency wins all-equal ties).
|
|
164
206
|
|
|
207
|
+
Each item reports **how many of its runs passed**, not only whether one did:
|
|
208
|
+
|
|
209
|
+
```json
|
|
210
|
+
"runs": 5, "runs_passed": 4, "pass_at_k": true, "pass_power_k": false
|
|
211
|
+
```
|
|
212
|
+
|
|
213
|
+
`pass_at_k` is "did any run pass" and is what `passed` counts. `runs_passed` is the fact that separates a
|
|
214
|
+
reliable item from a coin-flip — without it a 5/5 item and a 1/5 item are identical in every field, because
|
|
215
|
+
`quality_score` is derived from the best run alone. `pass_power_k` is true only when every run passed, and the
|
|
216
|
+
run summary carries `passed_all_runs` beside `passed`; a large gap between the two means the model is
|
|
217
|
+
inconsistent rather than wrong. The console shows `4/5 runs passed` in `Notes` for a non-unanimous pass and
|
|
218
|
+
stays quiet for a unanimous one, and its summary line reads `3/4 passed, 1 on every run`.
|
|
219
|
+
|
|
220
|
+
`runs` is what the item actually ran, which is not always the requested `--runs`: `agentic_conversation` takes
|
|
221
|
+
no K and drives its fixture exactly once.
|
|
222
|
+
|
|
223
|
+
Each item additionally carries a per-phase breakdown:
|
|
224
|
+
|
|
225
|
+
```json
|
|
226
|
+
"latency_breakdown_s": {
|
|
227
|
+
"agent_s": 4.02, // GoodData's own response time — the system under test
|
|
228
|
+
"judge_s": 1.31, // LLM-as-judge scoring, post-hoc
|
|
229
|
+
"simulated_user_s": 0.0, // our simulated user composing the next turn (multi-turn kinds)
|
|
230
|
+
"langfuse_s": 5.70 // trace lookup + score writing, off the critical path
|
|
231
|
+
}
|
|
232
|
+
```
|
|
233
|
+
|
|
234
|
+
Every item reports `runs_ungraded` beside `runs_passed`: runs the agent answered but the LLM judge returned
|
|
235
|
+
nothing readable for. Agentic items also list them as `unscored_runs` / `judge_errors` in their `detail`, and a
|
|
236
|
+
`dashboard_summary` item carries `ungraded_criteria`. Such a run — or, for `dashboard_summary`, such a
|
|
237
|
+
criterion — is excluded from pass@K and from the quality score rather than counted as a failure: scoring it 0
|
|
238
|
+
would be indistinguishable from the judge genuinely failing the answer, which is the confusion
|
|
239
|
+
`JudgeResponseError` exists to end. `pass@K` still holds on the runs that *were* graded, so an item can pass with
|
|
240
|
+
`runs_ungraded` set; `pass^K` cannot, because a run nobody graded leaves "all K passed" unverified. For
|
|
241
|
+
`dashboard_summary` an ungraded `must_include` / `must_not_include` criterion likewise cannot carry a pass —
|
|
242
|
+
"the judge could not tell" is not evidence the fact is present — while an ungraded `rubric` line only narrows
|
|
243
|
+
the quality score. When *no* run could be graded (for `dashboard_summary`: no gating criterion on any run) the
|
|
244
|
+
item errors instead of reporting failures. A non-zero count means pass@K was computed over fewer runs than
|
|
245
|
+
`--runs` asked for, so treat the result as weaker evidence and check the judge (`GD_EVAL_JUDGE_DIAGNOSTICS=1`,
|
|
246
|
+
or raise `JUDGE_MAX_COMPLETION_TOKENS` if the cause is `finish_reason=length`). The console says so in `Notes`
|
|
247
|
+
(`1 run(s) ungraded`, `2 criterion(s) ungraded`).
|
|
248
|
+
|
|
249
|
+
`agent_s` + `judge_s` + `simulated_user_s` are the instrumented parts of the item's `latency_s`; they do not add
|
|
250
|
+
up to it exactly, because `latency_s` is wall-clock around the whole item and also covers the conversation
|
|
251
|
+
create/delete round trips, SDK construction and any cleanup. `langfuse_s` sits **beside** `latency_s`, never
|
|
252
|
+
inside it, because trace linking runs outside every item's critical path (see above) — summing all four would
|
|
253
|
+
re-inflate exactly what that design removes.
|
|
254
|
+
|
|
255
|
+
A phase that a kind does not have reports `0.0` rather than an invented number, so read the zeroes as "not
|
|
256
|
+
applicable here", not "instant". Today:
|
|
257
|
+
|
|
258
|
+
| Field | Populated by |
|
|
259
|
+
|---|---|
|
|
260
|
+
| `agent_s` | `agentic_general_question`, `agentic_metric_skill` |
|
|
261
|
+
| `judge_s` | `agentic_general_question` only — `agentic_metric_skill` compares MAQL by string, it has no LLM judge |
|
|
262
|
+
| `simulated_user_s` | `agentic_metric_skill` only — `agentic_general_question` is single-turn, it has no simulated user |
|
|
263
|
+
| `langfuse_s` | every agentic kind, but only on the `gd-eval` path and only when Langfuse credentials are present |
|
|
264
|
+
|
|
265
|
+
The other six agentic kinds report `0.0` for the first three. Trace linking itself happens whenever
|
|
266
|
+
`LANGFUSE_PUBLIC_KEY`/`LANGFUSE_SECRET_KEY` are exported, with or without `--langfuse`, because each
|
|
267
|
+
`evaluate_agentic_*` falls back to `try_make_langfuse_client()`. But its *duration* is measured by the CLI
|
|
268
|
+
runner rather than by `evaluate_agentic_*`, so a direct library caller sees `langfuse_s: 0.0` even though its
|
|
269
|
+
linking ran. Pass `TAVERN_E2E_SKIP_TRACE_LINK=1` to opt out of linking altogether.
|
|
270
|
+
|
|
165
271
|
---
|
|
166
272
|
|
|
167
273
|
## `gd-eval models`
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
# (C) 2026 GoodData Corporation
|
|
2
2
|
[project]
|
|
3
3
|
name = "gooddata-eval"
|
|
4
|
-
version = "1.74.
|
|
4
|
+
version = "1.74.1.dev2"
|
|
5
5
|
description = "Evaluate the GoodData AI agent against your own questions and models."
|
|
6
6
|
readme = "README.md"
|
|
7
7
|
license = "MIT"
|
|
@@ -11,7 +11,7 @@ authors = [
|
|
|
11
11
|
keywords = ["gooddata", "ai", "evaluation", "llm", "analytics", "cli"]
|
|
12
12
|
requires-python = ">=3.10"
|
|
13
13
|
dependencies = [
|
|
14
|
-
"gooddata-sdk~=1.74.
|
|
14
|
+
"gooddata-sdk~=1.74.1.dev2",
|
|
15
15
|
"httpx>=0.27,<1.0",
|
|
16
16
|
"orjson>=3.9.15,<4.0.0",
|
|
17
17
|
"pydantic>=2.6,<3.0",
|
|
@@ -30,7 +30,7 @@ classifiers = [
|
|
|
30
30
|
]
|
|
31
31
|
|
|
32
32
|
[project.optional-dependencies]
|
|
33
|
-
llm-judge = ["openai>=1.
|
|
33
|
+
llm-judge = ["openai>=1.45,<2.0"]
|
|
34
34
|
|
|
35
35
|
[project.scripts]
|
|
36
36
|
gd-eval = "gooddata_eval.cli.main:main"
|
|
@@ -43,8 +43,8 @@ dev = [
|
|
|
43
43
|
"pytest>=8.3.5",
|
|
44
44
|
]
|
|
45
45
|
test = [
|
|
46
|
-
"pytest~=
|
|
47
|
-
"pytest-cov~=
|
|
46
|
+
"pytest~=9.1.1",
|
|
47
|
+
"pytest-cov~=7.1.0",
|
|
48
48
|
"pytest-json-report==1.5.0",
|
|
49
49
|
"pytest-mock>=3.14.0",
|
|
50
50
|
]
|