gooddata-eval 1.74.1.dev1__tar.gz → 1.74.1.dev3__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (120) hide show
  1. gooddata_eval-1.74.1.dev3/AGENTS.md +114 -0
  2. gooddata_eval-1.74.1.dev3/CLAUDE.md +1 -0
  3. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev3}/PKG-INFO +58 -14
  4. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev3}/README.md +56 -12
  5. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev3}/pyproject.toml +4 -4
  6. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/cli/agentic_runner.py +32 -2
  7. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/cli/main.py +62 -11
  8. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/agentic/_catalog.py +41 -0
  9. gooddata_eval-1.74.1.dev3/src/gooddata_eval/core/agentic/_gate.py +70 -0
  10. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/agentic/_langfuse.py +199 -240
  11. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/agentic/_trace_linker.py +24 -5
  12. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/agentic/alert_skill.py +191 -26
  13. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/agentic/conversation.py +129 -40
  14. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/agentic/general_question.py +32 -5
  15. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/agentic/guardrail.py +22 -4
  16. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/agentic/kda_skill.py +50 -5
  17. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/agentic/metric_skill.py +37 -81
  18. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/agentic/search_tool.py +32 -4
  19. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/agentic/visualization.py +37 -24
  20. gooddata_eval-1.74.1.dev3/src/gooddata_eval/core/chat/render.py +47 -0
  21. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/chat/sse_client.py +34 -2
  22. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/config.py +21 -0
  23. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/dataset/langfuse_source.py +4 -14
  24. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/evaluators/__init__.py +13 -5
  25. gooddata_eval-1.74.1.dev3/src/gooddata_eval/core/evaluators/_maql.py +103 -0
  26. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/evaluators/_text_utils.py +3 -4
  27. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/evaluators/alert_skill.py +11 -2
  28. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/evaluators/general_question.py +4 -0
  29. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/evaluators/metric_skill.py +16 -4
  30. gooddata_eval-1.74.1.dev3/src/gooddata_eval/core/langfuse/_env.py +39 -0
  31. gooddata_eval-1.74.1.dev3/src/gooddata_eval/core/langfuse/client.py +205 -0
  32. gooddata_eval-1.74.1.dev3/src/gooddata_eval/core/langfuse/experiment.py +156 -0
  33. gooddata_eval-1.74.1.dev3/src/gooddata_eval/core/langfuse/observations.py +125 -0
  34. gooddata_eval-1.74.1.dev3/src/gooddata_eval/core/langfuse/otlp.py +164 -0
  35. gooddata_eval-1.74.1.dev3/src/gooddata_eval/core/langfuse/sink.py +184 -0
  36. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/models.py +42 -0
  37. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/reporting/console.py +12 -2
  38. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/reporting/json_report.py +7 -1
  39. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/runner.py +20 -3
  40. gooddata_eval-1.74.1.dev3/tests/_fake_langfuse.py +273 -0
  41. gooddata_eval-1.74.1.dev3/tests/conftest.py +22 -0
  42. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev3}/tests/test_agentic_alert_skill.py +285 -0
  43. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev3}/tests/test_agentic_conversation.py +277 -1
  44. gooddata_eval-1.74.1.dev3/tests/test_agentic_gate.py +381 -0
  45. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev3}/tests/test_agentic_general_question.py +41 -0
  46. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev3}/tests/test_agentic_kda_skill.py +84 -7
  47. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev3}/tests/test_agentic_langfuse_trace.py +76 -86
  48. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev3}/tests/test_agentic_metric_skill.py +3 -47
  49. gooddata_eval-1.74.1.dev3/tests/test_agentic_observe_experiment.py +325 -0
  50. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev3}/tests/test_agentic_search_tool.py +42 -0
  51. gooddata_eval-1.74.1.dev3/tests/test_chat_render.py +118 -0
  52. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev3}/tests/test_cli.py +2 -2
  53. gooddata_eval-1.74.1.dev3/tests/test_fake_langfuse.py +140 -0
  54. gooddata_eval-1.74.1.dev3/tests/test_langfuse_client.py +379 -0
  55. gooddata_eval-1.74.1.dev3/tests/test_langfuse_e2e_fake_server.py +385 -0
  56. gooddata_eval-1.74.1.dev3/tests/test_langfuse_env.py +70 -0
  57. gooddata_eval-1.74.1.dev3/tests/test_langfuse_experiment.py +206 -0
  58. gooddata_eval-1.74.1.dev3/tests/test_langfuse_observations.py +264 -0
  59. gooddata_eval-1.74.1.dev3/tests/test_langfuse_otlp.py +197 -0
  60. gooddata_eval-1.74.1.dev3/tests/test_langfuse_sink.py +251 -0
  61. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev3}/tests/test_langfuse_source.py +19 -1
  62. gooddata_eval-1.74.1.dev3/tests/test_maql_normalize.py +106 -0
  63. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev3}/tests/test_runner.py +1 -0
  64. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev3}/tests/test_text_evaluators.py +28 -0
  65. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev3}/tests/test_trace_linker.py +16 -2
  66. gooddata_eval-1.74.1.dev1/src/gooddata_eval/core/langfuse/sink.py +0 -197
  67. gooddata_eval-1.74.1.dev1/tests/conftest.py +0 -9
  68. gooddata_eval-1.74.1.dev1/tests/test_langfuse_sink.py +0 -165
  69. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev3}/.gitignore +0 -0
  70. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev3}/LICENSE.txt +0 -0
  71. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev3}/Makefile +0 -0
  72. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/__init__.py +0 -0
  73. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/_version.py +0 -0
  74. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/cli/__init__.py +0 -0
  75. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/__init__.py +0 -0
  76. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/_output.py +0 -0
  77. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/agentic/__init__.py +0 -0
  78. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/chat/__init__.py +0 -0
  79. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/connection.py +0 -0
  80. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/dataset/__init__.py +0 -0
  81. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/dataset/local.py +0 -0
  82. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/evaluators/_deep_subset.py +0 -0
  83. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/evaluators/_llm_judge.py +0 -0
  84. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/evaluators/base.py +0 -0
  85. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/evaluators/guardrail.py +0 -0
  86. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/evaluators/search_tool.py +0 -0
  87. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/evaluators/summary.py +0 -0
  88. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/evaluators/visualization.py +0 -0
  89. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/langfuse/__init__.py +0 -0
  90. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/reporting/__init__.py +0 -0
  91. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/scoring.py +0 -0
  92. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/summary/__init__.py +0 -0
  93. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/summary/http_client.py +0 -0
  94. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/timing.py +0 -0
  95. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev3}/src/gooddata_eval/core/workspace.py +0 -0
  96. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev3}/tests/__init__.py +0 -0
  97. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev3}/tests/fixtures/sample_dataset/metric_skill_create.json +0 -0
  98. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev3}/tests/fixtures/sample_dataset/visualization_revenue.json +0 -0
  99. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev3}/tests/fixtures/sse_visualization_stream.txt +0 -0
  100. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev3}/tests/test_agentic_guardrail.py +0 -0
  101. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev3}/tests/test_agentic_run_context.py +0 -0
  102. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev3}/tests/test_agentic_runner.py +0 -0
  103. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev3}/tests/test_agentic_visualization.py +0 -0
  104. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev3}/tests/test_alert_skill_evaluator.py +0 -0
  105. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev3}/tests/test_connection.py +0 -0
  106. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev3}/tests/test_deep_subset.py +0 -0
  107. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev3}/tests/test_llm_judge.py +0 -0
  108. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev3}/tests/test_local_loader.py +0 -0
  109. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev3}/tests/test_metric_skill_evaluator.py +0 -0
  110. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev3}/tests/test_models.py +0 -0
  111. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev3}/tests/test_reporting.py +0 -0
  112. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev3}/tests/test_scoring.py +0 -0
  113. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev3}/tests/test_search_tool_evaluator.py +0 -0
  114. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev3}/tests/test_sse_client.py +0 -0
  115. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev3}/tests/test_summary_client.py +0 -0
  116. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev3}/tests/test_summary_evaluator.py +0 -0
  117. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev3}/tests/test_timing.py +0 -0
  118. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev3}/tests/test_visualization_evaluator.py +0 -0
  119. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev3}/tests/test_workspace.py +0 -0
  120. {gooddata_eval-1.74.1.dev1 → gooddata_eval-1.74.1.dev3}/tox.ini +0 -0
@@ -0,0 +1,114 @@
1
+ # gooddata-eval
2
+
3
+ `gd-eval` — a CLI and library that drives the GoodData AI agent (a separate service, in
4
+ `gdc-nas`) through a dataset of natural-language questions and scores what comes back,
5
+ including side-by-side comparison across models. Each dataset item is a JSON envelope
6
+ loaded from a local folder or pulled from a Langfuse dataset. Results are aggregated into
7
+ pass@K / pass^K reports and optionally pushed to Langfuse as scored traces tied to an
8
+ experiment. The newest and most actively developed package in the repo.
9
+
10
+ ## Owns
11
+
12
+ - The `gd-eval` CLI (`gd-eval run`, `gd-eval models`)
13
+ - Dataset loading and the evaluation run loop
14
+ - Per-capability evaluators and their scoring
15
+ - Result reporting, and pushing experiments, scores and trace links to Langfuse
16
+
17
+ ## Does NOT Own
18
+
19
+ - The agent under evaluation — that lives in `gdc-nas` (gen-ai)
20
+ - Platform access → `gooddata-sdk`
21
+
22
+ ## Architecture
23
+
24
+ | Path | Role |
25
+ |---|---|
26
+ | `cli/` | argument parsing, and `agentic_runner` — the agentic dispatch and concurrency phases |
27
+ | `core/agentic/` | multi-turn agentic evaluation per capability, **plus** all Langfuse trace polling and linking (`_langfuse.py`, `_trace_linker.py`) |
28
+ | `core/chat/` | SSE client for the agent's streaming chat endpoint |
29
+ | `core/summary/` | HTTP client for the dedicated dashboard-summary endpoint — a single-shot chat backend, not reporting |
30
+ | `core/dataset/` | dataset format and loading |
31
+ | `core/evaluators/` | single-shot evaluators and their registry |
32
+ | `core/langfuse/` | the whole Langfuse v4 client: `_env` (base URL + credentials), `otlp` (OTLP/JSON encoding), `experiment` (root-span construction, score targets), `observations` (trace reads), `client` (httpx calls), `sink` (single-shot results as experiments) |
33
+ | `core/reporting/` | console and JSON output rendering |
34
+ | `core/scoring.py`, `core/runner.py` | scoring and orchestration |
35
+ | `core/models.py` | `DatasetItem`, `ChatResult`, `ItemReport` and friends |
36
+
37
+ **Depends on**: `gooddata-sdk`, `httpx`, `pydantic`, `orjson`, `rich`. The LLM-judge
38
+ evaluator is an optional extra (`llm-judge`, pulling `openai>=1.45,<2.0`); every `openai`
39
+ import site is guarded or deferred so the base install stays usable without it — keep it
40
+ that way.
41
+
42
+ ### Two evaluation paths that share almost nothing
43
+
44
+ - **Single-shot** kinds send one chat turn and are scored by an `Evaluator` (a Protocol:
45
+ a `test_kind` attribute plus `evaluate(item, chat_result) -> ItemEvaluation`) looked up
46
+ from a registry in `core/evaluators/__init__.py`.
47
+ - **Agentic** kinds (`agentic_*`, `vis_agentic`) drive a full multi-turn conversation over
48
+ the SSE endpoint and are dispatched by an explicit `if`/`elif` chain in
49
+ `cli/agentic_runner.py`.
50
+
51
+ Many capabilities exist in **both** forms — visualization, metric skill, alert skill,
52
+ search, general question and guardrail each have a single-turn and a multi-turn
53
+ implementation, sometimes under different `test_kind` strings (`search_tool` vs
54
+ `agentic_search`). These are parallel implementations, not layers.
55
+
56
+ ### Dataset items
57
+
58
+ `DatasetItem` is the envelope: `id`, `dataset_name`, `test_kind`, `question`, and
59
+ `expected_output: Any`. `expected_output` is deliberately untyped — each evaluator parses
60
+ its own shape. `test_kind` on the item is what labels the result, not the evaluator class,
61
+ which is why `knowledge_question` can reuse `GeneralQuestionEvaluator` verbatim.
62
+ `dashboard_summary` items additionally need `summary_input`.
63
+
64
+ ## Gotchas
65
+
66
+ **Adding an evaluator is a registry change, not a naming convention.** Single-shot kinds go
67
+ into `_EAGER_EVALUATORS`, or `_LAZY_EVALUATOR_MODULES` plus `_LAZY_EVALUATOR_CLASSES`, in
68
+ `core/evaluators/__init__.py`. Agentic kinds need the string added to `AGENTIC_TEST_KINDS`
69
+ and a new branch in `_dispatch_agentic`. Test file naming follows the capability, but
70
+ naming a test file correctly registers nothing.
71
+
72
+ **Parallel-safety is a reviewed allowlist, and getting it wrong corrupts results.**
73
+ `WORKSPACE_MUTATING_TEST_KINDS` is computed as `AGENTIC_TEST_KINDS - PARALLEL_SAFE_TEST_KINDS`,
74
+ so a newly added kind defaults to workspace-mutating and runs serially in its own phase.
75
+ That default is correct: agent tool calls create real server-side objects (metrics, alerts).
76
+ Adding a kind to `PARALLEL_SAFE_TEST_KINDS` is a deliberate assertion that it is read-only,
77
+ which nothing in the package can prove for you.
78
+
79
+ **The SSE client's retry predicate is load-bearing.** `core/chat/sse_client.py` retries
80
+ 429/502/503/504 and `httpx.RemoteProtocolError` (a mid-stream disconnect) with exponential
81
+ backoff, and treats a `METADATA_SYNC_IN_PROGRESS` payload as transient. The
82
+ `RemoteProtocolError` case was added after it was confirmed live to contaminate a small
83
+ percentage of visualization runs with a hard fail and no retry. Narrowing that predicate
84
+ reintroduces the problem.
85
+
86
+ **Langfuse trace linking is deliberately off the item critical path.** Polling for trace
87
+ ingestion has no pass/fail signal and inflates or misattributes per-item latency, so
88
+ `BackgroundTraceLinker` defers it and is drained before the report renders
89
+ (`run_trace_link_inline` is the synchronous alternative). Do not "fix" a slow item by
90
+ making trace scoring synchronous again.
91
+
92
+ **Scoring weights do not sum to 1.** `quality_score` is the fraction of boolean-valued keys
93
+ in `best_detail` that are true, falling back to `pass_at_k` when there are none (text
94
+ evaluators). `value_score` is `0.6 * quality + 0.2 * speed` — the 0.8 total is what the
95
+ code does; treat it as intentional unless you have checked with the owner.
96
+
97
+ ### Fixture shapes
98
+
99
+ Group-by / attribute expectations in the alert-skill fixtures are written in AAC shape,
100
+ while the tool arguments the agent emits are AFM-shaped. Never deep-compare those two
101
+ directly — convert, or compare field by field. This applies specifically to the
102
+ attribute/group-by fields: `Filters` in the same fixtures is AFM-shaped on both sides and
103
+ is correctly deep-compared as-is. The attribute comparison itself lands with the
104
+ alert group-by work currently on `jt/gdai-2175-eval-alert-attributes`, so on `master` this
105
+ is guidance for the incoming code rather than a description of what is already there.
106
+
107
+ ## Testing
108
+
109
+ Plain pytest under `tests/`, no cassettes — the agent is stubbed with
110
+ `unittest.mock`. Tests are named per capability (`test_agentic_*.py`), which is the
111
+ convention to follow when adding one.
112
+
113
+ `ty` is configured here with `allowed-unresolved-imports` for `openai.**` and
114
+ `gooddata_api_client.**`; do not widen that list to paper over a real typing problem.
@@ -0,0 +1 @@
1
+ @AGENTS.md
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: gooddata-eval
3
- Version: 1.74.1.dev1
3
+ Version: 1.74.1.dev3
4
4
  Summary: Evaluate the GoodData AI agent against your own questions and models.
5
5
  Project-URL: Source, https://github.com/gooddata/gooddata-python-sdk
6
6
  Author-email: GoodData <support@gooddata.com>
@@ -17,7 +17,7 @@ Classifier: Topic :: Scientific/Engineering
17
17
  Classifier: Topic :: Software Development
18
18
  Classifier: Typing :: Typed
19
19
  Requires-Python: >=3.10
20
- Requires-Dist: gooddata-sdk~=1.74.1.dev1
20
+ Requires-Dist: gooddata-sdk~=1.74.1.dev3
21
21
  Requires-Dist: httpx<1.0,>=0.27
22
22
  Requires-Dist: orjson<4.0.0,>=3.9.15
23
23
  Requires-Dist: pydantic<3.0,>=2.6
@@ -141,7 +141,7 @@ gd-eval run \
141
141
  | Flag | Description |
142
142
  |---|---|
143
143
  | `--dataset PATH` | Flat folder of JSON files — one question per file. |
144
- | `--langfuse-dataset NAME` | Pull items by name from a Langfuse dataset. Requires `LANGFUSE_PUBLIC_KEY`, `LANGFUSE_SECRET_KEY`, `LANGFUSE_HOST`. |
144
+ | `--langfuse-dataset NAME` | Pull items by name from a Langfuse dataset. Requires `LANGFUSE_PUBLIC_KEY`, `LANGFUSE_SECRET_KEY` and `LANGFUSE_BASE_URL` (or the legacy `LANGFUSE_HOST`). |
145
145
  | `--kind TEST_KIND` | Fallback `test_kind` for dataset items that do not embed one. Defaults to `visualization`; use e.g. `agentic_metric_skill` for multi-turn agentic evaluation. Items that declare their own `test_kind` ignore this. |
146
146
 
147
147
  #### Model selection
@@ -154,7 +154,8 @@ gd-eval run \
154
154
 
155
155
  | Flag | Default | Description |
156
156
  |---|---|---|
157
- | `--runs K` | `2` | Independent runs per item (pass@K). An item passes if any run passes. |
157
+ | `--runs K` | `2` | Independent runs per item. |
158
+ | `--gate` | `any` | Which verdict decides an item: `any` = pass@K (a run passing is enough), `power` = pass^K (every run must pass, so the verdict measures stability). Identical at `--runs 1`. Kinds that repeat K runs only — `power` is refused when the dataset also has non-agentic items or `agentic_conversation`, which are always decided on pass@K. |
158
159
  | `--concurrency K` | `1` | Number of items evaluated concurrently. `1` = sequential (default). Increase to load-test the agent under simultaneous requests — see *Concurrency and workspace safety* below. |
159
160
  | `--judge-model MODEL` | `gpt-4o` | Model used for LLM-as-judge scoring — `agentic_general_question`, `agentic_guardrail`, `general_question`, `guardrail` and `dashboard_summary`. Also settable via `GD_EVAL_JUDGE_MODEL`. Two things to weigh before changing it: the gpt-5 family rejects `temperature=0`, so verdicts stop being reproducible (the run warns when this happens); and choosing the same model the agent runs means the judge grades its own family's output. |
160
161
  | `--reasoning-effort LEVEL` | server default | `LOW`, `MEDIUM` or `HIGH`, sent as `options.reasoningEffort` on every chat message. Requires the `enableGenAiReasoningEffort` feature flag on the target organization — without it the server ignores the value. Applies to chat items only; `dashboard_summary` items go through the summary endpoint, which has no such option. |
@@ -180,15 +181,39 @@ interleaves when K > 1, and per-item latencies rise, so they stop being clean si
180
181
 
181
182
  | Flag | Description |
182
183
  |---|---|
183
- | `--langfuse` | Log scores and traces to Langfuse after each item. Requires `--langfuse-dataset`. Names each experiment run `{dataset_name}_{timestamp}_{model}`, suffixed `_effort-{level}` when `--reasoning-effort` is set (so runs differing only by effort stay separate) and `_run{N}` per run when `--runs` > 1 — e.g. `general_question_2026-09-02-11-13_gpt-5.2_run0`. Requires `LANGFUSE_PUBLIC_KEY`, `LANGFUSE_SECRET_KEY`, `LANGFUSE_HOST`. |
184
+ | `--langfuse` | Log scores and traces to Langfuse after each item. Requires `--langfuse-dataset`. Names each experiment run `{dataset_name}_{timestamp}_{model}`, suffixed `_effort-{level}` when `--reasoning-effort` is set (so runs differing only by effort stay separate) and `_run{N}` per run when `--runs` > 1 — e.g. `general_question_2026-09-02-11-13_gpt-5.2_run0`. Requires `LANGFUSE_PUBLIC_KEY`, `LANGFUSE_SECRET_KEY` and `LANGFUSE_BASE_URL` (or the legacy `LANGFUSE_HOST`). |
184
185
 
185
- Set `TAVERN_E2E_SKIP_TRACE_LINK=1` to skip trace lookup entirely (scores are then orphaned; the run says so).
186
+ ##### Langfuse v4
186
187
 
187
- **A local `--dataset` cannot be attached to a Langfuse run.** `--langfuse` is refused alongside `--dataset`
188
- because a local folder's item ids are not Langfuse dataset item ids. But trace linking does not depend on that
189
- flag — each `evaluate_agentic_*` builds its own client whenever `LANGFUSE_PUBLIC_KEY`/`LANGFUSE_SECRET_KEY` are
190
- exported — so a local run still finds its traces and writes its scores onto them, and only the per-run grouping
191
- fails, with one `404 from dataset-run-items` reported per run. The run warns about this before it starts. Use
188
+ One run is one Langfuse **experiment**. Each evaluated item becomes its own trace whose root span carries the
189
+ experiment and dataset-item attributes (`langfuse.experiment.name`, `langfuse.experiment.dataset.id`,
190
+ `langfuse.experiment.item.id`), and the four scores attach to that root observation. gd-eval speaks to Langfuse
191
+ over four REST endpoints and uses no Langfuse SDK, so it runs on every Python version the package supports:
192
+
193
+ | Endpoint | Used for |
194
+ |---|---|
195
+ | `POST /api/public/otel/v1/traces` | exporting the experiment root span as OTLP/HTTP JSON |
196
+ | `POST /api/public/scores` | one score per write, on a trace or on a single observation inside it |
197
+ | `GET /api/public/v2/observations` | finding the agent's gen-ai trace for a conversation |
198
+ | `GET /api/public/dataset-items` | loading `--langfuse-dataset` items, and resolving an item's dataset id |
199
+
200
+ Two consequences of v4's immutable observations. gd-eval sets `version` only on its own experiment span, never on
201
+ the agent's gen-ai trace — filter on the gd-eval experiment's `langfuse.version` to compare models. And on the
202
+ agentic kinds the latency in `value_score` is the gen-ai trace's root generation latency, read from the
203
+ observations endpoint; the single-shot `--langfuse` sink keeps using the item's own measured average latency.
204
+
205
+ Langfuse Cloud drops v3 on **2026-11-16**; a self-hosted Langfuse must be on v4 for any of this to work.
206
+
207
+ Set `TAVERN_E2E_SKIP_TRACE_LINK=1` to turn the whole **agentic** Langfuse write path off — no trace lookup, no
208
+ span export and no scores for `agentic_*` items. The run says so once. It does not reach the `--langfuse` sink,
209
+ which still writes a span and four scores for every single-shot item; drop `--langfuse` to silence that too.
210
+
211
+ **A local `--dataset` cannot be attached to a Langfuse experiment.** `--langfuse` is refused alongside
212
+ `--dataset` because a local folder's item ids are not Langfuse dataset item ids. But trace linking does not
213
+ depend on that flag — each `evaluate_agentic_*` builds its own client whenever
214
+ `LANGFUSE_PUBLIC_KEY`/`LANGFUSE_SECRET_KEY` are exported — so a local run still finds its traces and writes its
215
+ scores onto them, and only the per-run grouping fails: the dataset-item lookup 404s and the run reports the item
216
+ as one that does not exist in Langfuse, once. The run also warns about this before it starts. Use
192
217
  `--langfuse-dataset` when you want runs that are comparable across models, or `TAVERN_E2E_SKIP_TRACE_LINK=1` to
193
218
  skip linking altogether.
194
219
 
@@ -235,10 +260,12 @@ Winner is selected by **pass rate → quality score → latency** (lower latency
235
260
  Each item reports **how many of its runs passed**, not only whether one did:
236
261
 
237
262
  ```json
238
- "runs": 5, "runs_passed": 4, "pass_at_k": true, "pass_power_k": false
263
+ "runs": 5, "runs_passed": 4, "pass_at_k": true, "pass_power_k": false, "gate_passed": true
239
264
  ```
240
265
 
241
- `pass_at_k` is "did any run pass" and is what `passed` counts. `runs_passed` is the fact that separates a
266
+ `pass_at_k` is "did any run pass" — always literal, whatever the gate. `gate_passed` is the verdict the item
267
+ was decided on and is what `passed` counts; the two differ only under `--gate power`, where an item that
268
+ passed 4 of 5 runs is `"pass_at_k": true, "gate_passed": false`. `runs_passed` is the fact that separates a
242
269
  reliable item from a coin-flip — without it a 5/5 item and a 1/5 item are identical in every field, because
243
270
  `quality_score` is derived from the best run alone. `pass_power_k` is true only when every run passed, and the
244
271
  run summary carries `passed_all_runs` beside `passed`; a large gap between the two means the model is
@@ -248,6 +275,14 @@ stays quiet for a unanimous one, and its summary line reads `3/4 passed, 1 on ev
248
275
  `runs` is what the item actually ran, which is not always the requested `--runs`: `agentic_conversation` takes
249
276
  no K and drives its fixture exactly once.
250
277
 
278
+ Which of the two decides pass/fail is `--gate`: `any` (default) gates on `pass_at_k`, `power` gates on
279
+ `pass_power_k`. The run records it as a top-level `gate`, and a failure under `power` says so —
280
+ `Gate pass^3 failed: 2/3 runs passed — unstable, not a clean failure` — because the message body describes
281
+ the best run, which under pass^K can be a run that passed, and the console repeats the count in `Notes` for
282
+ the same reason. When some runs were ungraded the note says so instead of calling the remainder unstable.
283
+ `agentic_conversation` has no K gate, so `--gate power` is refused for a dataset containing one rather than
284
+ labelling a report `power` that only part of the dataset was decided under.
285
+
251
286
  Each item additionally carries a per-phase breakdown:
252
287
 
253
288
  ```json
@@ -408,10 +443,19 @@ Without `[llm-judge]`, those items are **skipped**.
408
443
 
409
444
  ## Scores (in JSON report and Langfuse)
410
445
 
446
+ In Langfuse every score is written to the experiment run's root observation — `traceId` plus `observationId` of
447
+ the item's own root span. On the agentic path each score is mirrored onto the agent's gen-ai trace as well
448
+ (`traceId` only), so a score survives even when one of the two traces is missing.
449
+
411
450
  | Score | Description |
412
451
  |---|---|
413
- | `pass_at_k` | 1 if any of the K runs passed strict checks, else 0. |
452
+ | `pass_at_k` | 1 if **any** of the K runs passed strict checks, else 0. |
453
+ | `pass_power_k` | 1 only if **every** one of the K runs passed. Agentic kinds only. |
454
+ | `gate_passed` | The verdict that decided the item: `pass_at_k` under `--gate any`, `pass_power_k` under `--gate power`. Agentic kinds only. |
414
455
  | `quality_score` | Fraction of strict check flags that are `True` (0.0–1.0). Shown in CLI as a percentage. |
415
456
  | `value_score` | Weighted blend: 0.6 × quality + 0.2 × speed (speed = max(0, 1 − latency/60s)). |
416
457
  | `latency_s` | Average per-run latency in seconds. |
417
458
  | `provider_type` | Model vendor + gateway label (e.g. `ANTHROPIC`, `BEDROCK/ANTHROPIC`, `AZURE/OPENAI`). Stored in Langfuse trace metadata and tags. |
459
+
460
+ Score names carry no K; K and the gate are on the dataset-run metadata as `eval_k` and `eval_gate`.
461
+ `agentic_visualization` also still writes its historic `pass_at_{K}` / `pass_power_{K}` pair.
@@ -113,7 +113,7 @@ gd-eval run \
113
113
  | Flag | Description |
114
114
  |---|---|
115
115
  | `--dataset PATH` | Flat folder of JSON files — one question per file. |
116
- | `--langfuse-dataset NAME` | Pull items by name from a Langfuse dataset. Requires `LANGFUSE_PUBLIC_KEY`, `LANGFUSE_SECRET_KEY`, `LANGFUSE_HOST`. |
116
+ | `--langfuse-dataset NAME` | Pull items by name from a Langfuse dataset. Requires `LANGFUSE_PUBLIC_KEY`, `LANGFUSE_SECRET_KEY` and `LANGFUSE_BASE_URL` (or the legacy `LANGFUSE_HOST`). |
117
117
  | `--kind TEST_KIND` | Fallback `test_kind` for dataset items that do not embed one. Defaults to `visualization`; use e.g. `agentic_metric_skill` for multi-turn agentic evaluation. Items that declare their own `test_kind` ignore this. |
118
118
 
119
119
  #### Model selection
@@ -126,7 +126,8 @@ gd-eval run \
126
126
 
127
127
  | Flag | Default | Description |
128
128
  |---|---|---|
129
- | `--runs K` | `2` | Independent runs per item (pass@K). An item passes if any run passes. |
129
+ | `--runs K` | `2` | Independent runs per item. |
130
+ | `--gate` | `any` | Which verdict decides an item: `any` = pass@K (a run passing is enough), `power` = pass^K (every run must pass, so the verdict measures stability). Identical at `--runs 1`. Kinds that repeat K runs only — `power` is refused when the dataset also has non-agentic items or `agentic_conversation`, which are always decided on pass@K. |
130
131
  | `--concurrency K` | `1` | Number of items evaluated concurrently. `1` = sequential (default). Increase to load-test the agent under simultaneous requests — see *Concurrency and workspace safety* below. |
131
132
  | `--judge-model MODEL` | `gpt-4o` | Model used for LLM-as-judge scoring — `agentic_general_question`, `agentic_guardrail`, `general_question`, `guardrail` and `dashboard_summary`. Also settable via `GD_EVAL_JUDGE_MODEL`. Two things to weigh before changing it: the gpt-5 family rejects `temperature=0`, so verdicts stop being reproducible (the run warns when this happens); and choosing the same model the agent runs means the judge grades its own family's output. |
132
133
  | `--reasoning-effort LEVEL` | server default | `LOW`, `MEDIUM` or `HIGH`, sent as `options.reasoningEffort` on every chat message. Requires the `enableGenAiReasoningEffort` feature flag on the target organization — without it the server ignores the value. Applies to chat items only; `dashboard_summary` items go through the summary endpoint, which has no such option. |
@@ -152,15 +153,39 @@ interleaves when K > 1, and per-item latencies rise, so they stop being clean si
152
153
 
153
154
  | Flag | Description |
154
155
  |---|---|
155
- | `--langfuse` | Log scores and traces to Langfuse after each item. Requires `--langfuse-dataset`. Names each experiment run `{dataset_name}_{timestamp}_{model}`, suffixed `_effort-{level}` when `--reasoning-effort` is set (so runs differing only by effort stay separate) and `_run{N}` per run when `--runs` > 1 — e.g. `general_question_2026-09-02-11-13_gpt-5.2_run0`. Requires `LANGFUSE_PUBLIC_KEY`, `LANGFUSE_SECRET_KEY`, `LANGFUSE_HOST`. |
156
+ | `--langfuse` | Log scores and traces to Langfuse after each item. Requires `--langfuse-dataset`. Names each experiment run `{dataset_name}_{timestamp}_{model}`, suffixed `_effort-{level}` when `--reasoning-effort` is set (so runs differing only by effort stay separate) and `_run{N}` per run when `--runs` > 1 — e.g. `general_question_2026-09-02-11-13_gpt-5.2_run0`. Requires `LANGFUSE_PUBLIC_KEY`, `LANGFUSE_SECRET_KEY` and `LANGFUSE_BASE_URL` (or the legacy `LANGFUSE_HOST`). |
156
157
 
157
- Set `TAVERN_E2E_SKIP_TRACE_LINK=1` to skip trace lookup entirely (scores are then orphaned; the run says so).
158
+ ##### Langfuse v4
158
159
 
159
- **A local `--dataset` cannot be attached to a Langfuse run.** `--langfuse` is refused alongside `--dataset`
160
- because a local folder's item ids are not Langfuse dataset item ids. But trace linking does not depend on that
161
- flag — each `evaluate_agentic_*` builds its own client whenever `LANGFUSE_PUBLIC_KEY`/`LANGFUSE_SECRET_KEY` are
162
- exported — so a local run still finds its traces and writes its scores onto them, and only the per-run grouping
163
- fails, with one `404 from dataset-run-items` reported per run. The run warns about this before it starts. Use
160
+ One run is one Langfuse **experiment**. Each evaluated item becomes its own trace whose root span carries the
161
+ experiment and dataset-item attributes (`langfuse.experiment.name`, `langfuse.experiment.dataset.id`,
162
+ `langfuse.experiment.item.id`), and the four scores attach to that root observation. gd-eval speaks to Langfuse
163
+ over four REST endpoints and uses no Langfuse SDK, so it runs on every Python version the package supports:
164
+
165
+ | Endpoint | Used for |
166
+ |---|---|
167
+ | `POST /api/public/otel/v1/traces` | exporting the experiment root span as OTLP/HTTP JSON |
168
+ | `POST /api/public/scores` | one score per write, on a trace or on a single observation inside it |
169
+ | `GET /api/public/v2/observations` | finding the agent's gen-ai trace for a conversation |
170
+ | `GET /api/public/dataset-items` | loading `--langfuse-dataset` items, and resolving an item's dataset id |
171
+
172
+ Two consequences of v4's immutable observations. gd-eval sets `version` only on its own experiment span, never on
173
+ the agent's gen-ai trace — filter on the gd-eval experiment's `langfuse.version` to compare models. And on the
174
+ agentic kinds the latency in `value_score` is the gen-ai trace's root generation latency, read from the
175
+ observations endpoint; the single-shot `--langfuse` sink keeps using the item's own measured average latency.
176
+
177
+ Langfuse Cloud drops v3 on **2026-11-16**; a self-hosted Langfuse must be on v4 for any of this to work.
178
+
179
+ Set `TAVERN_E2E_SKIP_TRACE_LINK=1` to turn the whole **agentic** Langfuse write path off — no trace lookup, no
180
+ span export and no scores for `agentic_*` items. The run says so once. It does not reach the `--langfuse` sink,
181
+ which still writes a span and four scores for every single-shot item; drop `--langfuse` to silence that too.
182
+
183
+ **A local `--dataset` cannot be attached to a Langfuse experiment.** `--langfuse` is refused alongside
184
+ `--dataset` because a local folder's item ids are not Langfuse dataset item ids. But trace linking does not
185
+ depend on that flag — each `evaluate_agentic_*` builds its own client whenever
186
+ `LANGFUSE_PUBLIC_KEY`/`LANGFUSE_SECRET_KEY` are exported — so a local run still finds its traces and writes its
187
+ scores onto them, and only the per-run grouping fails: the dataset-item lookup 404s and the run reports the item
188
+ as one that does not exist in Langfuse, once. The run also warns about this before it starts. Use
164
189
  `--langfuse-dataset` when you want runs that are comparable across models, or `TAVERN_E2E_SKIP_TRACE_LINK=1` to
165
190
  skip linking altogether.
166
191
 
@@ -207,10 +232,12 @@ Winner is selected by **pass rate → quality score → latency** (lower latency
207
232
  Each item reports **how many of its runs passed**, not only whether one did:
208
233
 
209
234
  ```json
210
- "runs": 5, "runs_passed": 4, "pass_at_k": true, "pass_power_k": false
235
+ "runs": 5, "runs_passed": 4, "pass_at_k": true, "pass_power_k": false, "gate_passed": true
211
236
  ```
212
237
 
213
- `pass_at_k` is "did any run pass" and is what `passed` counts. `runs_passed` is the fact that separates a
238
+ `pass_at_k` is "did any run pass" — always literal, whatever the gate. `gate_passed` is the verdict the item
239
+ was decided on and is what `passed` counts; the two differ only under `--gate power`, where an item that
240
+ passed 4 of 5 runs is `"pass_at_k": true, "gate_passed": false`. `runs_passed` is the fact that separates a
214
241
  reliable item from a coin-flip — without it a 5/5 item and a 1/5 item are identical in every field, because
215
242
  `quality_score` is derived from the best run alone. `pass_power_k` is true only when every run passed, and the
216
243
  run summary carries `passed_all_runs` beside `passed`; a large gap between the two means the model is
@@ -220,6 +247,14 @@ stays quiet for a unanimous one, and its summary line reads `3/4 passed, 1 on ev
220
247
  `runs` is what the item actually ran, which is not always the requested `--runs`: `agentic_conversation` takes
221
248
  no K and drives its fixture exactly once.
222
249
 
250
+ Which of the two decides pass/fail is `--gate`: `any` (default) gates on `pass_at_k`, `power` gates on
251
+ `pass_power_k`. The run records it as a top-level `gate`, and a failure under `power` says so —
252
+ `Gate pass^3 failed: 2/3 runs passed — unstable, not a clean failure` — because the message body describes
253
+ the best run, which under pass^K can be a run that passed, and the console repeats the count in `Notes` for
254
+ the same reason. When some runs were ungraded the note says so instead of calling the remainder unstable.
255
+ `agentic_conversation` has no K gate, so `--gate power` is refused for a dataset containing one rather than
256
+ labelling a report `power` that only part of the dataset was decided under.
257
+
223
258
  Each item additionally carries a per-phase breakdown:
224
259
 
225
260
  ```json
@@ -380,10 +415,19 @@ Without `[llm-judge]`, those items are **skipped**.
380
415
 
381
416
  ## Scores (in JSON report and Langfuse)
382
417
 
418
+ In Langfuse every score is written to the experiment run's root observation — `traceId` plus `observationId` of
419
+ the item's own root span. On the agentic path each score is mirrored onto the agent's gen-ai trace as well
420
+ (`traceId` only), so a score survives even when one of the two traces is missing.
421
+
383
422
  | Score | Description |
384
423
  |---|---|
385
- | `pass_at_k` | 1 if any of the K runs passed strict checks, else 0. |
424
+ | `pass_at_k` | 1 if **any** of the K runs passed strict checks, else 0. |
425
+ | `pass_power_k` | 1 only if **every** one of the K runs passed. Agentic kinds only. |
426
+ | `gate_passed` | The verdict that decided the item: `pass_at_k` under `--gate any`, `pass_power_k` under `--gate power`. Agentic kinds only. |
386
427
  | `quality_score` | Fraction of strict check flags that are `True` (0.0–1.0). Shown in CLI as a percentage. |
387
428
  | `value_score` | Weighted blend: 0.6 × quality + 0.2 × speed (speed = max(0, 1 − latency/60s)). |
388
429
  | `latency_s` | Average per-run latency in seconds. |
389
430
  | `provider_type` | Model vendor + gateway label (e.g. `ANTHROPIC`, `BEDROCK/ANTHROPIC`, `AZURE/OPENAI`). Stored in Langfuse trace metadata and tags. |
431
+
432
+ Score names carry no K; K and the gate are on the dataset-run metadata as `eval_k` and `eval_gate`.
433
+ `agentic_visualization` also still writes its historic `pass_at_{K}` / `pass_power_{K}` pair.
@@ -1,7 +1,7 @@
1
1
  # (C) 2026 GoodData Corporation
2
2
  [project]
3
3
  name = "gooddata-eval"
4
- version = "1.74.1.dev1"
4
+ version = "1.74.1.dev3"
5
5
  description = "Evaluate the GoodData AI agent against your own questions and models."
6
6
  readme = "README.md"
7
7
  license = "MIT"
@@ -11,7 +11,7 @@ authors = [
11
11
  keywords = ["gooddata", "ai", "evaluation", "llm", "analytics", "cli"]
12
12
  requires-python = ">=3.10"
13
13
  dependencies = [
14
- "gooddata-sdk~=1.74.1.dev1",
14
+ "gooddata-sdk~=1.74.1.dev3",
15
15
  "httpx>=0.27,<1.0",
16
16
  "orjson>=3.9.15,<4.0.0",
17
17
  "pydantic>=2.6,<3.0",
@@ -43,8 +43,8 @@ dev = [
43
43
  "pytest>=8.3.5",
44
44
  ]
45
45
  test = [
46
- "pytest~=8.3.4",
47
- "pytest-cov~=6.0.0",
46
+ "pytest~=9.1.1",
47
+ "pytest-cov~=7.1.0",
48
48
  "pytest-json-report==1.5.0",
49
49
  "pytest-mock>=3.14.0",
50
50
  ]
@@ -8,6 +8,7 @@ import time
8
8
  from concurrent.futures import ThreadPoolExecutor, as_completed
9
9
  from typing import Any, TypedDict
10
10
 
11
+ from gooddata_eval.core.agentic._gate import DEFAULT_GATE, EvalGate, normalize_gate
11
12
  from gooddata_eval.core.agentic._langfuse import make_langfuse_client
12
13
  from gooddata_eval.core.agentic._trace_linker import BackgroundTraceLinker, SubmitTraceLink, run_trace_link_inline
13
14
  from gooddata_eval.core.agentic.alert_skill import evaluate_agentic_alert_skill
@@ -48,6 +49,13 @@ AGENTIC_TEST_KINDS = frozenset(
48
49
  )
49
50
 
50
51
 
52
+ # Agentic kinds that no gate applies to: they drive their fixture exactly once, so there is
53
+ # no K to take pass@K or pass^K over. Named here rather than inline in _dispatch_agentic so
54
+ # the CLI can refuse --gate power for a dataset containing one instead of labelling the whole
55
+ # report `power` when part of it was never gated.
56
+ UNGATED_AGENTIC_TEST_KINDS = frozenset({"agentic_conversation"})
57
+
58
+
51
59
  # Kinds cleared to run several at a time. An EXPLICIT allowlist, not a subtraction: nothing
52
60
  # in this package can prove a kind is read-only, because the mutation happens server-side in
53
61
  # whichever tools the agent decides to call. So each entry here is a reviewed judgement, and
@@ -128,9 +136,13 @@ def _dispatch_agentic(
128
136
  reasoning_effort: ReasoningEffort | None = None,
129
137
  agent_id: str | None = None,
130
138
  submit_trace_link: SubmitTraceLink = run_trace_link_inline,
139
+ gate: EvalGate = DEFAULT_GATE,
131
140
  ) -> AgenticEvalOutcome:
132
141
  """Call the appropriate evaluate_agentic_* function for the item's test_kind.
133
142
 
143
+ `gate` reaches every kind except those in UNGATED_AGENTIC_TEST_KINDS, which have no K
144
+ to gate over; the CLI refuses --gate power for a dataset containing one.
145
+
134
146
  Every evaluate_agentic_* function returns an AgenticEvalOutcome (reasoning_steps,
135
147
  conversation_id, response_id, detail) on success and attaches the same four attributes
136
148
  to its raised *AssertionError on failure -- no kind is exempt.
@@ -155,6 +167,7 @@ def _dispatch_agentic(
155
167
  question=item.question,
156
168
  expected_outputs=_parse_visualization_expected(eo),
157
169
  k=k,
170
+ gate=gate,
158
171
  agent_id=agent_id,
159
172
  **lf_kw,
160
173
  )
@@ -166,6 +179,7 @@ def _dispatch_agentic(
166
179
  question=item.question,
167
180
  expected_output=eo if isinstance(eo, (dict, list)) else {},
168
181
  k=k,
182
+ gate=gate,
169
183
  agent_id=agent_id,
170
184
  **lf_kw,
171
185
  )
@@ -177,6 +191,7 @@ def _dispatch_agentic(
177
191
  question=item.question,
178
192
  expected_output=eo if isinstance(eo, dict) else {},
179
193
  k=k,
194
+ gate=gate,
180
195
  agent_id=agent_id,
181
196
  **lf_kw,
182
197
  )
@@ -191,6 +206,7 @@ def _dispatch_agentic(
191
206
  question=item.question,
192
207
  expected_tool_call=expected_args,
193
208
  k=k,
209
+ gate=gate,
194
210
  agent_id=agent_id,
195
211
  **lf_kw,
196
212
  )
@@ -202,6 +218,7 @@ def _dispatch_agentic(
202
218
  question=item.question,
203
219
  expected_output=eo if isinstance(eo, str) else str(eo),
204
220
  k=k,
221
+ gate=gate,
205
222
  agent_id=agent_id,
206
223
  user_context=item.user_context,
207
224
  **lf_kw,
@@ -214,6 +231,7 @@ def _dispatch_agentic(
214
231
  question=item.question,
215
232
  expected_output=eo if isinstance(eo, str) else str(eo),
216
233
  k=k,
234
+ gate=gate,
217
235
  agent_id=agent_id,
218
236
  **lf_kw,
219
237
  )
@@ -225,6 +243,7 @@ def _dispatch_agentic(
225
243
  question=item.question,
226
244
  expected_output=eo if isinstance(eo, dict) else {},
227
245
  k=k,
246
+ gate=gate,
228
247
  agent_id=agent_id,
229
248
  **lf_kw,
230
249
  )
@@ -291,6 +310,7 @@ def run_agentic_items(
291
310
  on_item_done: Any = None,
292
311
  agent_id: str | None = None,
293
312
  concurrency: int = 1,
313
+ gate: EvalGate = DEFAULT_GATE,
294
314
  ) -> EvalReport:
295
315
  """Run agentic items through evaluate_agentic_* and return an EvalReport.
296
316
 
@@ -303,7 +323,7 @@ def run_agentic_items(
303
323
  """
304
324
  langfuse = make_langfuse_client() if use_langfuse else None
305
325
 
306
- report = EvalReport(model=model_version)
326
+ report = EvalReport(model=model_version, gate=normalize_gate(gate))
307
327
  total = len(items)
308
328
  # Trace linking runs here rather than inside each evaluate_agentic_*, so an item's
309
329
  # Langfuse poll overlaps the NEXT item's agent call instead of extending its own
@@ -325,6 +345,9 @@ def run_agentic_items(
325
345
  test_kind=item.test_kind,
326
346
  question=item.question,
327
347
  )
348
+ # None, not False, for the kinds _dispatch_agentic passes no gate to: ItemReport.passed
349
+ # then falls back to pass_at_k and gate_passed keeps meaning "a gate ran".
350
+ gated = item.test_kind not in UNGATED_AGENTIC_TEST_KINDS
328
351
  t0 = time.perf_counter()
329
352
  try:
330
353
  outcome = _dispatch_agentic(
@@ -339,6 +362,7 @@ def run_agentic_items(
339
362
  reasoning_effort,
340
363
  agent_id,
341
364
  submit_trace_link=linker.submit,
365
+ gate=gate,
342
366
  )
343
367
  if isinstance(outcome, AgenticEvalOutcome):
344
368
  reasoning_steps = outcome.reasoning_steps
@@ -347,6 +371,8 @@ def run_agentic_items(
347
371
  detail = outcome.detail
348
372
  else:
349
373
  reasoning_steps, conversation_id, response_id, detail = outcome, None, None, {}
374
+ item_report.gate_passed = True if gated else None
375
+ # Whichever gate decided the item, clearing it means at least one run passed.
350
376
  item_report.pass_at_k = True
351
377
  item_report.runs = k
352
378
  item_report.reasoning_steps = reasoning_steps or []
@@ -356,7 +382,7 @@ def run_agentic_items(
356
382
  _apply_timings(item_report, getattr(outcome, "timings", None))
357
383
  _apply_run_counts(item_report, outcome)
358
384
  except AssertionError as exc:
359
- item_report.pass_at_k = False
385
+ item_report.gate_passed = False if gated else None
360
386
  item_report.runs = k
361
387
  item_report.reasoning_steps = getattr(exc, "reasoning_steps", None) or []
362
388
  item_report.conversation_id = getattr(exc, "conversation_id", None)
@@ -364,6 +390,10 @@ def run_agentic_items(
364
390
  item_report.best_detail = getattr(exc, "detail", None) or {}
365
391
  _apply_timings(item_report, getattr(exc, "timings", None))
366
392
  _apply_run_counts(item_report, exc)
393
+ # Read off the counts, not off the gate: pass^K fails items where runs did pass,
394
+ # and reporting those as pass_at_k False would contradict the Langfuse score of
395
+ # the same name. Kinds that report no count read as 0, i.e. a clean failure.
396
+ item_report.pass_at_k = item_report.runs_passed > 0
367
397
  print(f"[agentic] {item.id} FAIL: {exc}", flush=True)
368
398
  except Exception as exc:
369
399
  item_report.error = f"{type(exc).__name__}: {exc}"