gooddata-eval 1.74.0__tar.gz → 1.74.1.dev2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (112) hide show
  1. gooddata_eval-1.74.1.dev2/AGENTS.md +114 -0
  2. gooddata_eval-1.74.1.dev2/CLAUDE.md +1 -0
  3. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/PKG-INFO +111 -5
  4. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/README.md +108 -2
  5. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/pyproject.toml +5 -5
  6. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/cli/agentic_runner.py +188 -4
  7. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/cli/main.py +70 -2
  8. gooddata_eval-1.74.1.dev2/src/gooddata_eval/core/_output.py +18 -0
  9. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/agentic/_catalog.py +41 -0
  10. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/agentic/_langfuse.py +185 -29
  11. gooddata_eval-1.74.1.dev2/src/gooddata_eval/core/agentic/_trace_linker.py +302 -0
  12. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/agentic/alert_skill.py +263 -110
  13. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/agentic/conversation.py +199 -106
  14. gooddata_eval-1.74.1.dev2/src/gooddata_eval/core/agentic/general_question.py +344 -0
  15. gooddata_eval-1.74.1.dev2/src/gooddata_eval/core/agentic/guardrail.py +306 -0
  16. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/agentic/kda_skill.py +128 -93
  17. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/agentic/metric_skill.py +123 -147
  18. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/agentic/search_tool.py +78 -63
  19. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/agentic/visualization.py +94 -96
  20. gooddata_eval-1.74.1.dev2/src/gooddata_eval/core/chat/render.py +47 -0
  21. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/chat/sse_client.py +42 -4
  22. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/config.py +25 -0
  23. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/dataset/langfuse_source.py +53 -22
  24. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/evaluators/__init__.py +13 -5
  25. gooddata_eval-1.74.1.dev2/src/gooddata_eval/core/evaluators/_llm_judge.py +311 -0
  26. gooddata_eval-1.74.1.dev2/src/gooddata_eval/core/evaluators/_maql.py +103 -0
  27. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/evaluators/_text_utils.py +3 -4
  28. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/evaluators/alert_skill.py +11 -2
  29. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/evaluators/general_question.py +19 -13
  30. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/evaluators/guardrail.py +20 -17
  31. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/evaluators/metric_skill.py +16 -4
  32. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/evaluators/summary.py +56 -13
  33. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/models.py +80 -0
  34. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/reporting/console.py +34 -10
  35. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/reporting/json_report.py +26 -2
  36. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/runner.py +68 -2
  37. gooddata_eval-1.74.1.dev2/src/gooddata_eval/core/timing.py +74 -0
  38. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/tests/test_agentic_alert_skill.py +334 -59
  39. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/tests/test_agentic_conversation.py +277 -1
  40. gooddata_eval-1.74.1.dev2/tests/test_agentic_general_question.py +730 -0
  41. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/tests/test_agentic_guardrail.py +108 -82
  42. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/tests/test_agentic_kda_skill.py +199 -208
  43. gooddata_eval-1.74.1.dev2/tests/test_agentic_langfuse_trace.py +552 -0
  44. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/tests/test_agentic_metric_skill.py +181 -101
  45. gooddata_eval-1.74.1.dev2/tests/test_agentic_runner.py +768 -0
  46. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/tests/test_agentic_search_tool.py +42 -0
  47. gooddata_eval-1.74.1.dev2/tests/test_chat_render.py +118 -0
  48. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/tests/test_cli.py +188 -0
  49. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/tests/test_connection.py +5 -0
  50. gooddata_eval-1.74.1.dev2/tests/test_langfuse_source.py +198 -0
  51. gooddata_eval-1.74.1.dev2/tests/test_llm_judge.py +616 -0
  52. gooddata_eval-1.74.1.dev2/tests/test_maql_normalize.py +106 -0
  53. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/tests/test_models.py +27 -0
  54. gooddata_eval-1.74.1.dev2/tests/test_reporting.py +444 -0
  55. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/tests/test_runner.py +60 -0
  56. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/tests/test_sse_client.py +54 -0
  57. gooddata_eval-1.74.1.dev2/tests/test_summary_evaluator.py +192 -0
  58. gooddata_eval-1.74.1.dev2/tests/test_text_evaluators.py +132 -0
  59. gooddata_eval-1.74.1.dev2/tests/test_timing.py +93 -0
  60. gooddata_eval-1.74.1.dev2/tests/test_trace_linker.py +568 -0
  61. gooddata_eval-1.74.0/src/gooddata_eval/core/agentic/general_question.py +0 -264
  62. gooddata_eval-1.74.0/src/gooddata_eval/core/agentic/guardrail.py +0 -268
  63. gooddata_eval-1.74.0/src/gooddata_eval/core/evaluators/_llm_judge.py +0 -66
  64. gooddata_eval-1.74.0/tests/test_agentic_general_question.py +0 -210
  65. gooddata_eval-1.74.0/tests/test_agentic_langfuse_trace.py +0 -26
  66. gooddata_eval-1.74.0/tests/test_agentic_runner.py +0 -220
  67. gooddata_eval-1.74.0/tests/test_langfuse_source.py +0 -105
  68. gooddata_eval-1.74.0/tests/test_llm_judge.py +0 -45
  69. gooddata_eval-1.74.0/tests/test_reporting.py +0 -194
  70. gooddata_eval-1.74.0/tests/test_summary_evaluator.py +0 -87
  71. gooddata_eval-1.74.0/tests/test_text_evaluators.py +0 -72
  72. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/.gitignore +0 -0
  73. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/LICENSE.txt +0 -0
  74. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/Makefile +0 -0
  75. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/__init__.py +0 -0
  76. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/_version.py +0 -0
  77. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/cli/__init__.py +0 -0
  78. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/__init__.py +0 -0
  79. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/agentic/__init__.py +0 -0
  80. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/chat/__init__.py +0 -0
  81. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/connection.py +0 -0
  82. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/dataset/__init__.py +0 -0
  83. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/dataset/local.py +0 -0
  84. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/evaluators/_deep_subset.py +0 -0
  85. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/evaluators/base.py +0 -0
  86. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/evaluators/search_tool.py +0 -0
  87. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/evaluators/visualization.py +0 -0
  88. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/langfuse/__init__.py +0 -0
  89. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/langfuse/sink.py +0 -0
  90. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/reporting/__init__.py +0 -0
  91. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/scoring.py +0 -0
  92. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/summary/__init__.py +0 -0
  93. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/summary/http_client.py +0 -0
  94. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/src/gooddata_eval/core/workspace.py +0 -0
  95. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/tests/__init__.py +0 -0
  96. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/tests/conftest.py +0 -0
  97. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/tests/fixtures/sample_dataset/metric_skill_create.json +0 -0
  98. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/tests/fixtures/sample_dataset/visualization_revenue.json +0 -0
  99. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/tests/fixtures/sse_visualization_stream.txt +0 -0
  100. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/tests/test_agentic_run_context.py +0 -0
  101. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/tests/test_agentic_visualization.py +0 -0
  102. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/tests/test_alert_skill_evaluator.py +0 -0
  103. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/tests/test_deep_subset.py +0 -0
  104. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/tests/test_langfuse_sink.py +0 -0
  105. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/tests/test_local_loader.py +0 -0
  106. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/tests/test_metric_skill_evaluator.py +0 -0
  107. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/tests/test_scoring.py +0 -0
  108. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/tests/test_search_tool_evaluator.py +0 -0
  109. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/tests/test_summary_client.py +0 -0
  110. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/tests/test_visualization_evaluator.py +0 -0
  111. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/tests/test_workspace.py +0 -0
  112. {gooddata_eval-1.74.0 → gooddata_eval-1.74.1.dev2}/tox.ini +0 -0
@@ -0,0 +1,114 @@
1
+ # gooddata-eval
2
+
3
+ `gd-eval` — a CLI and library that drives the GoodData AI agent (a separate service, in
4
+ `gdc-nas`) through a dataset of natural-language questions and scores what comes back,
5
+ including side-by-side comparison across models. Each dataset item is a JSON envelope
6
+ loaded from a local folder or pulled from a Langfuse dataset. Results are aggregated into
7
+ pass@K / pass^K reports and optionally pushed to Langfuse as scored traces tied to a
8
+ dataset run. The newest and most actively developed package in the repo.
9
+
10
+ ## Owns
11
+
12
+ - The `gd-eval` CLI (`gd-eval run`, `gd-eval models`)
13
+ - Dataset loading and the evaluation run loop
14
+ - Per-capability evaluators and their scoring
15
+ - Result reporting, and pushing runs, scores and trace links to Langfuse
16
+
17
+ ## Does NOT Own
18
+
19
+ - The agent under evaluation — that lives in `gdc-nas` (gen-ai)
20
+ - Platform access → `gooddata-sdk`
21
+
22
+ ## Architecture
23
+
24
+ | Path | Role |
25
+ |---|---|
26
+ | `cli/` | argument parsing, and `agentic_runner` — the agentic dispatch and concurrency phases |
27
+ | `core/agentic/` | multi-turn agentic evaluation per capability, **plus** all Langfuse trace polling and linking (`_langfuse.py`, `_trace_linker.py`) |
28
+ | `core/chat/` | SSE client for the agent's streaming chat endpoint |
29
+ | `core/summary/` | HTTP client for the dedicated dashboard-summary endpoint — a single-shot chat backend, not reporting |
30
+ | `core/dataset/` | dataset format and loading |
31
+ | `core/evaluators/` | single-shot evaluators and their registry |
32
+ | `core/langfuse/` | `sink.py` only — pushes single-turn scores and dataset-run items |
33
+ | `core/reporting/` | console and JSON output rendering |
34
+ | `core/scoring.py`, `core/runner.py` | scoring and orchestration |
35
+ | `core/models.py` | `DatasetItem`, `ChatResult`, `ItemReport` and friends |
36
+
37
+ **Depends on**: `gooddata-sdk`, `httpx`, `pydantic`, `orjson`, `rich`. The LLM-judge
38
+ evaluator is an optional extra (`llm-judge`, pulling `openai>=1.45,<2.0`); every `openai`
39
+ import site is guarded or deferred so the base install stays usable without it — keep it
40
+ that way.
41
+
42
+ ### Two evaluation paths that share almost nothing
43
+
44
+ - **Single-shot** kinds send one chat turn and are scored by an `Evaluator` (a Protocol:
45
+ a `test_kind` attribute plus `evaluate(item, chat_result) -> ItemEvaluation`) looked up
46
+ from a registry in `core/evaluators/__init__.py`.
47
+ - **Agentic** kinds (`agentic_*`, `vis_agentic`) drive a full multi-turn conversation over
48
+ the SSE endpoint and are dispatched by an explicit `if`/`elif` chain in
49
+ `cli/agentic_runner.py`.
50
+
51
+ Many capabilities exist in **both** forms — visualization, metric skill, alert skill,
52
+ search, general question and guardrail each have a single-turn and a multi-turn
53
+ implementation, sometimes under different `test_kind` strings (`search_tool` vs
54
+ `agentic_search`). These are parallel implementations, not layers.
55
+
56
+ ### Dataset items
57
+
58
+ `DatasetItem` is the envelope: `id`, `dataset_name`, `test_kind`, `question`, and
59
+ `expected_output: Any`. `expected_output` is deliberately untyped — each evaluator parses
60
+ its own shape. `test_kind` on the item is what labels the result, not the evaluator class,
61
+ which is why `knowledge_question` can reuse `GeneralQuestionEvaluator` verbatim.
62
+ `dashboard_summary` items additionally need `summary_input`.
63
+
64
+ ## Gotchas
65
+
66
+ **Adding an evaluator is a registry change, not a naming convention.** Single-shot kinds go
67
+ into `_EAGER_EVALUATORS`, or `_LAZY_EVALUATOR_MODULES` plus `_LAZY_EVALUATOR_CLASSES`, in
68
+ `core/evaluators/__init__.py`. Agentic kinds need the string added to `AGENTIC_TEST_KINDS`
69
+ and a new branch in `_dispatch_agentic`. Test file naming follows the capability, but
70
+ naming a test file correctly registers nothing.
71
+
72
+ **Parallel-safety is a reviewed allowlist, and getting it wrong corrupts results.**
73
+ `WORKSPACE_MUTATING_TEST_KINDS` is computed as `AGENTIC_TEST_KINDS - PARALLEL_SAFE_TEST_KINDS`,
74
+ so a newly added kind defaults to workspace-mutating and runs serially in its own phase.
75
+ That default is correct: agent tool calls create real server-side objects (metrics, alerts).
76
+ Adding a kind to `PARALLEL_SAFE_TEST_KINDS` is a deliberate assertion that it is read-only,
77
+ which nothing in the package can prove for you.
78
+
79
+ **The SSE client's retry predicate is load-bearing.** `core/chat/sse_client.py` retries
80
+ 429/502/503/504 and `httpx.RemoteProtocolError` (a mid-stream disconnect) with exponential
81
+ backoff, and treats a `METADATA_SYNC_IN_PROGRESS` payload as transient. The
82
+ `RemoteProtocolError` case was added after it was confirmed live to contaminate a small
83
+ percentage of visualization runs with a hard fail and no retry. Narrowing that predicate
84
+ reintroduces the problem.
85
+
86
+ **Langfuse trace linking is deliberately off the item critical path.** Polling for trace
87
+ ingestion has no pass/fail signal and inflates or misattributes per-item latency, so
88
+ `BackgroundTraceLinker` defers it and is drained before the report renders
89
+ (`run_trace_link_inline` is the synchronous alternative). Do not "fix" a slow item by
90
+ making trace scoring synchronous again.
91
+
92
+ **Scoring weights do not sum to 1.** `quality_score` is the fraction of boolean-valued keys
93
+ in `best_detail` that are true, falling back to `pass_at_k` when there are none (text
94
+ evaluators). `value_score` is `0.6 * quality + 0.2 * speed` — the 0.8 total is what the
95
+ code does; treat it as intentional unless you have checked with the owner.
96
+
97
+ ### Fixture shapes
98
+
99
+ Group-by / attribute expectations in the alert-skill fixtures are written in AAC shape,
100
+ while the tool arguments the agent emits are AFM-shaped. Never deep-compare those two
101
+ directly — convert, or compare field by field. This applies specifically to the
102
+ attribute/group-by fields: `Filters` in the same fixtures is AFM-shaped on both sides and
103
+ is correctly deep-compared as-is. The attribute comparison itself lands with the
104
+ alert group-by work currently on `jt/gdai-2175-eval-alert-attributes`, so on `master` this
105
+ is guidance for the incoming code rather than a description of what is already there.
106
+
107
+ ## Testing
108
+
109
+ Plain pytest under `tests/`, no cassettes — the agent is stubbed with
110
+ `unittest.mock`. Tests are named per capability (`test_agentic_*.py`), which is the
111
+ convention to follow when adding one.
112
+
113
+ `ty` is configured here with `allowed-unresolved-imports` for `openai.**` and
114
+ `gooddata_api_client.**`; do not widen that list to paper over a real typing problem.
@@ -0,0 +1 @@
1
+ @AGENTS.md
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: gooddata-eval
3
- Version: 1.74.0
3
+ Version: 1.74.1.dev2
4
4
  Summary: Evaluate the GoodData AI agent against your own questions and models.
5
5
  Project-URL: Source, https://github.com/gooddata/gooddata-python-sdk
6
6
  Author-email: GoodData <support@gooddata.com>
@@ -17,13 +17,13 @@ Classifier: Topic :: Scientific/Engineering
17
17
  Classifier: Topic :: Software Development
18
18
  Classifier: Typing :: Typed
19
19
  Requires-Python: >=3.10
20
- Requires-Dist: gooddata-sdk~=1.74.0
20
+ Requires-Dist: gooddata-sdk~=1.74.1.dev2
21
21
  Requires-Dist: httpx<1.0,>=0.27
22
22
  Requires-Dist: orjson<4.0.0,>=3.9.15
23
23
  Requires-Dist: pydantic<3.0,>=2.6
24
24
  Requires-Dist: rich<15.0,>=13.0
25
25
  Provides-Extra: llm-judge
26
- Requires-Dist: openai<2.0,>=1.40; extra == 'llm-judge'
26
+ Requires-Dist: openai<2.0,>=1.45; extra == 'llm-judge'
27
27
  Description-Content-Type: text/markdown
28
28
 
29
29
  # gooddata-eval
@@ -142,6 +142,7 @@ gd-eval run \
142
142
  |---|---|
143
143
  | `--dataset PATH` | Flat folder of JSON files — one question per file. |
144
144
  | `--langfuse-dataset NAME` | Pull items by name from a Langfuse dataset. Requires `LANGFUSE_PUBLIC_KEY`, `LANGFUSE_SECRET_KEY`, `LANGFUSE_HOST`. |
145
+ | `--kind TEST_KIND` | Fallback `test_kind` for dataset items that do not embed one. Defaults to `visualization`; use e.g. `agentic_metric_skill` for multi-turn agentic evaluation. Items that declare their own `test_kind` ignore this. |
145
146
 
146
147
  #### Model selection
147
148
 
@@ -154,21 +155,62 @@ gd-eval run \
154
155
  | Flag | Default | Description |
155
156
  |---|---|---|
156
157
  | `--runs K` | `2` | Independent runs per item (pass@K). An item passes if any run passes. |
157
- | `--concurrency K` | `1` | Number of items evaluated concurrently. `1` = sequential (default). Increase to load-test the agent under simultaneous requests. Progress output interleaves when K > 1. |
158
+ | `--concurrency K` | `1` | Number of items evaluated concurrently. `1` = sequential (default). Increase to load-test the agent under simultaneous requests — see *Concurrency and workspace safety* below. |
159
+ | `--judge-model MODEL` | `gpt-4o` | Model used for LLM-as-judge scoring — `agentic_general_question`, `agentic_guardrail`, `general_question`, `guardrail` and `dashboard_summary`. Also settable via `GD_EVAL_JUDGE_MODEL`. Two things to weigh before changing it: the gpt-5 family rejects `temperature=0`, so verdicts stop being reproducible (the run warns when this happens); and choosing the same model the agent runs means the judge grades its own family's output. |
158
160
  | `--reasoning-effort LEVEL` | server default | `LOW`, `MEDIUM` or `HIGH`, sent as `options.reasoningEffort` on every chat message. Requires the `enableGenAiReasoningEffort` feature flag on the target organization — without it the server ignores the value. Applies to chat items only; `dashboard_summary` items go through the summary endpoint, which has no such option. |
159
161
 
162
+ **Concurrency and workspace safety.** Agentic kinds that create workspace objects
163
+ (`agentic_metric_skill`, `agentic_alert_skill`, `agentic_conversation`, `agentic_kda_skill`) always run one at a
164
+ time whatever `--concurrency` says — a metric or alert created and dropped mid-run would otherwise be visible to
165
+ another item reading the same catalog. **That protection is for the agentic kinds only:** the single-turn
166
+ `metric_skill` and `alert_skill` kinds are still fanned out and the agent performs the same server-side writes on
167
+ that path, so avoid raising `--concurrency` on a dataset of those against a shared workspace. Progress output
168
+ interleaves when K > 1, and per-item latencies rise, so they stop being clean single-request measurements.
169
+
160
170
  #### Output
161
171
 
162
172
  | Flag | Description |
163
173
  |---|---|
164
174
  | `--json PATH` | Write a JSON report to this path. Always uses the nested `{models, runs, comparison}` shape even for a single model. |
165
175
  | `--quiet` | Suppress per-item progress. Per-model result tables and the comparison summary are still printed. |
176
+ | `--preserve-failed` | Keep failed conversations on the server instead of deleting them, so they can be inspected afterwards. Applies to the single-turn chat path; agentic kinds manage their own conversation lifecycle. |
177
+ | `--timers` | Print per-turn `[timer]` diagnostics — GoodData response, judge, and simulated-user seconds as they happen. Off by default: an 18-item `--runs 2` run emits ~72 lines and buries the progress output. The same measurements are always in the JSON report's `latency_breakdown_s`, so this only adds a live view. Also settable via `GD_EVAL_TIMERS=1`. |
166
178
 
167
179
  #### Langfuse sink
168
180
 
169
181
  | Flag | Description |
170
182
  |---|---|
171
- | `--langfuse` | Log scores and traces to Langfuse after each item. Requires `--langfuse-dataset`. Creates one named experiment run per model (`gd-eval-{timestamp}-{model}`, suffixed `-effort-{level}` when `--reasoning-effort` is set so runs differing only by effort stay separate). Requires `LANGFUSE_PUBLIC_KEY`, `LANGFUSE_SECRET_KEY`, `LANGFUSE_HOST`. |
183
+ | `--langfuse` | Log scores and traces to Langfuse after each item. Requires `--langfuse-dataset`. Names each experiment run `{dataset_name}_{timestamp}_{model}`, suffixed `_effort-{level}` when `--reasoning-effort` is set (so runs differing only by effort stay separate) and `_run{N}` per run when `--runs` > 1 — e.g. `general_question_2026-09-02-11-13_gpt-5.2_run0`. Requires `LANGFUSE_PUBLIC_KEY`, `LANGFUSE_SECRET_KEY`, `LANGFUSE_HOST`. |
184
+
185
+ Set `TAVERN_E2E_SKIP_TRACE_LINK=1` to skip trace lookup entirely (scores are then orphaned; the run says so).
186
+
187
+ **A local `--dataset` cannot be attached to a Langfuse run.** `--langfuse` is refused alongside `--dataset`
188
+ because a local folder's item ids are not Langfuse dataset item ids. But trace linking does not depend on that
189
+ flag — each `evaluate_agentic_*` builds its own client whenever `LANGFUSE_PUBLIC_KEY`/`LANGFUSE_SECRET_KEY` are
190
+ exported — so a local run still finds its traces and writes its scores onto them, and only the per-run grouping
191
+ fails, with one `404 from dataset-run-items` reported per run. The run warns about this before it starts. Use
192
+ `--langfuse-dataset` when you want runs that are comparable across models, or `TAVERN_E2E_SKIP_TRACE_LINK=1` to
193
+ skip linking altogether.
194
+
195
+ **When trace linking happens.** Finding a gen-ai trace means polling until Langfuse has ingested it, which is
196
+ lag measured in seconds to minutes. That work produces no verdict — the pass/fail is already decided — so it
197
+ does not run inline per item. Every item's Langfuse block is queued and the whole batch runs *after* the agent
198
+ phase, draining before any report is written. Two consequences worth knowing:
199
+
200
+ - **No item's `latency_s` includes trace linking.** Its cost is reported separately as
201
+ `latency_breakdown_s.langfuse_s`, and the run prints
202
+ `[langfuse] trace linking finished in Xs for N item(s); slowest Ys`. If `slowest` approaches the **120s**
203
+ batched retry budget, links are timing out and scores are being orphaned — look for
204
+ `[langfuse] WARNING: no trace found for conversation ...`.
205
+ - **The budget depends on who is waiting.** 120s is affordable only because the batch blocks nobody. A direct
206
+ library caller (`evaluate_agentic_*` without a `submit_trace_link`) polls inline, on its own critical path, and
207
+ gets **35s** instead — the same cost as before batching existed, so no inline caller pays for a budget raised
208
+ on the CLI's behalf. Either way a trace that is already ingested costs nothing: the loop looks before it sleeps.
209
+ Scores are always final before the command exits — the run blocks on the batch. Interrupting with Ctrl-C drops
210
+ whatever is still queued rather than making you wait it out: both the queued trace links and, under
211
+ `--concurrency`, the items that have not started. The handful of items already in flight still have to finish —
212
+ worker threads are joined at exit and an in-progress agent call cannot be cancelled — so expect to wait up to one
213
+ `--concurrency`-wide wave, not the rest of the dataset.
172
214
 
173
215
  ### JSON report shape
174
216
 
@@ -190,6 +232,70 @@ The JSON report always uses the nested multi-model shape:
190
232
 
191
233
  Winner is selected by **pass rate → quality score → latency** (lower latency wins all-equal ties).
192
234
 
235
+ Each item reports **how many of its runs passed**, not only whether one did:
236
+
237
+ ```json
238
+ "runs": 5, "runs_passed": 4, "pass_at_k": true, "pass_power_k": false
239
+ ```
240
+
241
+ `pass_at_k` is "did any run pass" and is what `passed` counts. `runs_passed` is the fact that separates a
242
+ reliable item from a coin-flip — without it a 5/5 item and a 1/5 item are identical in every field, because
243
+ `quality_score` is derived from the best run alone. `pass_power_k` is true only when every run passed, and the
244
+ run summary carries `passed_all_runs` beside `passed`; a large gap between the two means the model is
245
+ inconsistent rather than wrong. The console shows `4/5 runs passed` in `Notes` for a non-unanimous pass and
246
+ stays quiet for a unanimous one, and its summary line reads `3/4 passed, 1 on every run`.
247
+
248
+ `runs` is what the item actually ran, which is not always the requested `--runs`: `agentic_conversation` takes
249
+ no K and drives its fixture exactly once.
250
+
251
+ Each item additionally carries a per-phase breakdown:
252
+
253
+ ```json
254
+ "latency_breakdown_s": {
255
+ "agent_s": 4.02, // GoodData's own response time — the system under test
256
+ "judge_s": 1.31, // LLM-as-judge scoring, post-hoc
257
+ "simulated_user_s": 0.0, // our simulated user composing the next turn (multi-turn kinds)
258
+ "langfuse_s": 5.70 // trace lookup + score writing, off the critical path
259
+ }
260
+ ```
261
+
262
+ Every item reports `runs_ungraded` beside `runs_passed`: runs the agent answered but the LLM judge returned
263
+ nothing readable for. Agentic items also list them as `unscored_runs` / `judge_errors` in their `detail`, and a
264
+ `dashboard_summary` item carries `ungraded_criteria`. Such a run — or, for `dashboard_summary`, such a
265
+ criterion — is excluded from pass@K and from the quality score rather than counted as a failure: scoring it 0
266
+ would be indistinguishable from the judge genuinely failing the answer, which is the confusion
267
+ `JudgeResponseError` exists to end. `pass@K` still holds on the runs that *were* graded, so an item can pass with
268
+ `runs_ungraded` set; `pass^K` cannot, because a run nobody graded leaves "all K passed" unverified. For
269
+ `dashboard_summary` an ungraded `must_include` / `must_not_include` criterion likewise cannot carry a pass —
270
+ "the judge could not tell" is not evidence the fact is present — while an ungraded `rubric` line only narrows
271
+ the quality score. When *no* run could be graded (for `dashboard_summary`: no gating criterion on any run) the
272
+ item errors instead of reporting failures. A non-zero count means pass@K was computed over fewer runs than
273
+ `--runs` asked for, so treat the result as weaker evidence and check the judge (`GD_EVAL_JUDGE_DIAGNOSTICS=1`,
274
+ or raise `JUDGE_MAX_COMPLETION_TOKENS` if the cause is `finish_reason=length`). The console says so in `Notes`
275
+ (`1 run(s) ungraded`, `2 criterion(s) ungraded`).
276
+
277
+ `agent_s` + `judge_s` + `simulated_user_s` are the instrumented parts of the item's `latency_s`; they do not add
278
+ up to it exactly, because `latency_s` is wall-clock around the whole item and also covers the conversation
279
+ create/delete round trips, SDK construction and any cleanup. `langfuse_s` sits **beside** `latency_s`, never
280
+ inside it, because trace linking runs outside every item's critical path (see above) — summing all four would
281
+ re-inflate exactly what that design removes.
282
+
283
+ A phase that a kind does not have reports `0.0` rather than an invented number, so read the zeroes as "not
284
+ applicable here", not "instant". Today:
285
+
286
+ | Field | Populated by |
287
+ |---|---|
288
+ | `agent_s` | `agentic_general_question`, `agentic_metric_skill` |
289
+ | `judge_s` | `agentic_general_question` only — `agentic_metric_skill` compares MAQL by string, it has no LLM judge |
290
+ | `simulated_user_s` | `agentic_metric_skill` only — `agentic_general_question` is single-turn, it has no simulated user |
291
+ | `langfuse_s` | every agentic kind, but only on the `gd-eval` path and only when Langfuse credentials are present |
292
+
293
+ The other six agentic kinds report `0.0` for the first three. Trace linking itself happens whenever
294
+ `LANGFUSE_PUBLIC_KEY`/`LANGFUSE_SECRET_KEY` are exported, with or without `--langfuse`, because each
295
+ `evaluate_agentic_*` falls back to `try_make_langfuse_client()`. But its *duration* is measured by the CLI
296
+ runner rather than by `evaluate_agentic_*`, so a direct library caller sees `langfuse_s: 0.0` even though its
297
+ linking ran. Pass `TAVERN_E2E_SKIP_TRACE_LINK=1` to opt out of linking altogether.
298
+
193
299
  ---
194
300
 
195
301
  ## `gd-eval models`
@@ -114,6 +114,7 @@ gd-eval run \
114
114
  |---|---|
115
115
  | `--dataset PATH` | Flat folder of JSON files — one question per file. |
116
116
  | `--langfuse-dataset NAME` | Pull items by name from a Langfuse dataset. Requires `LANGFUSE_PUBLIC_KEY`, `LANGFUSE_SECRET_KEY`, `LANGFUSE_HOST`. |
117
+ | `--kind TEST_KIND` | Fallback `test_kind` for dataset items that do not embed one. Defaults to `visualization`; use e.g. `agentic_metric_skill` for multi-turn agentic evaluation. Items that declare their own `test_kind` ignore this. |
117
118
 
118
119
  #### Model selection
119
120
 
@@ -126,21 +127,62 @@ gd-eval run \
126
127
  | Flag | Default | Description |
127
128
  |---|---|---|
128
129
  | `--runs K` | `2` | Independent runs per item (pass@K). An item passes if any run passes. |
129
- | `--concurrency K` | `1` | Number of items evaluated concurrently. `1` = sequential (default). Increase to load-test the agent under simultaneous requests. Progress output interleaves when K > 1. |
130
+ | `--concurrency K` | `1` | Number of items evaluated concurrently. `1` = sequential (default). Increase to load-test the agent under simultaneous requests — see *Concurrency and workspace safety* below. |
131
+ | `--judge-model MODEL` | `gpt-4o` | Model used for LLM-as-judge scoring — `agentic_general_question`, `agentic_guardrail`, `general_question`, `guardrail` and `dashboard_summary`. Also settable via `GD_EVAL_JUDGE_MODEL`. Two things to weigh before changing it: the gpt-5 family rejects `temperature=0`, so verdicts stop being reproducible (the run warns when this happens); and choosing the same model the agent runs means the judge grades its own family's output. |
130
132
  | `--reasoning-effort LEVEL` | server default | `LOW`, `MEDIUM` or `HIGH`, sent as `options.reasoningEffort` on every chat message. Requires the `enableGenAiReasoningEffort` feature flag on the target organization — without it the server ignores the value. Applies to chat items only; `dashboard_summary` items go through the summary endpoint, which has no such option. |
131
133
 
134
+ **Concurrency and workspace safety.** Agentic kinds that create workspace objects
135
+ (`agentic_metric_skill`, `agentic_alert_skill`, `agentic_conversation`, `agentic_kda_skill`) always run one at a
136
+ time whatever `--concurrency` says — a metric or alert created and dropped mid-run would otherwise be visible to
137
+ another item reading the same catalog. **That protection is for the agentic kinds only:** the single-turn
138
+ `metric_skill` and `alert_skill` kinds are still fanned out and the agent performs the same server-side writes on
139
+ that path, so avoid raising `--concurrency` on a dataset of those against a shared workspace. Progress output
140
+ interleaves when K > 1, and per-item latencies rise, so they stop being clean single-request measurements.
141
+
132
142
  #### Output
133
143
 
134
144
  | Flag | Description |
135
145
  |---|---|
136
146
  | `--json PATH` | Write a JSON report to this path. Always uses the nested `{models, runs, comparison}` shape even for a single model. |
137
147
  | `--quiet` | Suppress per-item progress. Per-model result tables and the comparison summary are still printed. |
148
+ | `--preserve-failed` | Keep failed conversations on the server instead of deleting them, so they can be inspected afterwards. Applies to the single-turn chat path; agentic kinds manage their own conversation lifecycle. |
149
+ | `--timers` | Print per-turn `[timer]` diagnostics — GoodData response, judge, and simulated-user seconds as they happen. Off by default: an 18-item `--runs 2` run emits ~72 lines and buries the progress output. The same measurements are always in the JSON report's `latency_breakdown_s`, so this only adds a live view. Also settable via `GD_EVAL_TIMERS=1`. |
138
150
 
139
151
  #### Langfuse sink
140
152
 
141
153
  | Flag | Description |
142
154
  |---|---|
143
- | `--langfuse` | Log scores and traces to Langfuse after each item. Requires `--langfuse-dataset`. Creates one named experiment run per model (`gd-eval-{timestamp}-{model}`, suffixed `-effort-{level}` when `--reasoning-effort` is set so runs differing only by effort stay separate). Requires `LANGFUSE_PUBLIC_KEY`, `LANGFUSE_SECRET_KEY`, `LANGFUSE_HOST`. |
155
+ | `--langfuse` | Log scores and traces to Langfuse after each item. Requires `--langfuse-dataset`. Names each experiment run `{dataset_name}_{timestamp}_{model}`, suffixed `_effort-{level}` when `--reasoning-effort` is set (so runs differing only by effort stay separate) and `_run{N}` per run when `--runs` > 1 — e.g. `general_question_2026-09-02-11-13_gpt-5.2_run0`. Requires `LANGFUSE_PUBLIC_KEY`, `LANGFUSE_SECRET_KEY`, `LANGFUSE_HOST`. |
156
+
157
+ Set `TAVERN_E2E_SKIP_TRACE_LINK=1` to skip trace lookup entirely (scores are then orphaned; the run says so).
158
+
159
+ **A local `--dataset` cannot be attached to a Langfuse run.** `--langfuse` is refused alongside `--dataset`
160
+ because a local folder's item ids are not Langfuse dataset item ids. But trace linking does not depend on that
161
+ flag — each `evaluate_agentic_*` builds its own client whenever `LANGFUSE_PUBLIC_KEY`/`LANGFUSE_SECRET_KEY` are
162
+ exported — so a local run still finds its traces and writes its scores onto them, and only the per-run grouping
163
+ fails, with one `404 from dataset-run-items` reported per run. The run warns about this before it starts. Use
164
+ `--langfuse-dataset` when you want runs that are comparable across models, or `TAVERN_E2E_SKIP_TRACE_LINK=1` to
165
+ skip linking altogether.
166
+
167
+ **When trace linking happens.** Finding a gen-ai trace means polling until Langfuse has ingested it, which is
168
+ lag measured in seconds to minutes. That work produces no verdict — the pass/fail is already decided — so it
169
+ does not run inline per item. Every item's Langfuse block is queued and the whole batch runs *after* the agent
170
+ phase, draining before any report is written. Two consequences worth knowing:
171
+
172
+ - **No item's `latency_s` includes trace linking.** Its cost is reported separately as
173
+ `latency_breakdown_s.langfuse_s`, and the run prints
174
+ `[langfuse] trace linking finished in Xs for N item(s); slowest Ys`. If `slowest` approaches the **120s**
175
+ batched retry budget, links are timing out and scores are being orphaned — look for
176
+ `[langfuse] WARNING: no trace found for conversation ...`.
177
+ - **The budget depends on who is waiting.** 120s is affordable only because the batch blocks nobody. A direct
178
+ library caller (`evaluate_agentic_*` without a `submit_trace_link`) polls inline, on its own critical path, and
179
+ gets **35s** instead — the same cost as before batching existed, so no inline caller pays for a budget raised
180
+ on the CLI's behalf. Either way a trace that is already ingested costs nothing: the loop looks before it sleeps.
181
+ Scores are always final before the command exits — the run blocks on the batch. Interrupting with Ctrl-C drops
182
+ whatever is still queued rather than making you wait it out: both the queued trace links and, under
183
+ `--concurrency`, the items that have not started. The handful of items already in flight still have to finish —
184
+ worker threads are joined at exit and an in-progress agent call cannot be cancelled — so expect to wait up to one
185
+ `--concurrency`-wide wave, not the rest of the dataset.
144
186
 
145
187
  ### JSON report shape
146
188
 
@@ -162,6 +204,70 @@ The JSON report always uses the nested multi-model shape:
162
204
 
163
205
  Winner is selected by **pass rate → quality score → latency** (lower latency wins all-equal ties).
164
206
 
207
+ Each item reports **how many of its runs passed**, not only whether one did:
208
+
209
+ ```json
210
+ "runs": 5, "runs_passed": 4, "pass_at_k": true, "pass_power_k": false
211
+ ```
212
+
213
+ `pass_at_k` is "did any run pass" and is what `passed` counts. `runs_passed` is the fact that separates a
214
+ reliable item from a coin-flip — without it a 5/5 item and a 1/5 item are identical in every field, because
215
+ `quality_score` is derived from the best run alone. `pass_power_k` is true only when every run passed, and the
216
+ run summary carries `passed_all_runs` beside `passed`; a large gap between the two means the model is
217
+ inconsistent rather than wrong. The console shows `4/5 runs passed` in `Notes` for a non-unanimous pass and
218
+ stays quiet for a unanimous one, and its summary line reads `3/4 passed, 1 on every run`.
219
+
220
+ `runs` is what the item actually ran, which is not always the requested `--runs`: `agentic_conversation` takes
221
+ no K and drives its fixture exactly once.
222
+
223
+ Each item additionally carries a per-phase breakdown:
224
+
225
+ ```json
226
+ "latency_breakdown_s": {
227
+ "agent_s": 4.02, // GoodData's own response time — the system under test
228
+ "judge_s": 1.31, // LLM-as-judge scoring, post-hoc
229
+ "simulated_user_s": 0.0, // our simulated user composing the next turn (multi-turn kinds)
230
+ "langfuse_s": 5.70 // trace lookup + score writing, off the critical path
231
+ }
232
+ ```
233
+
234
+ Every item reports `runs_ungraded` beside `runs_passed`: runs the agent answered but the LLM judge returned
235
+ nothing readable for. Agentic items also list them as `unscored_runs` / `judge_errors` in their `detail`, and a
236
+ `dashboard_summary` item carries `ungraded_criteria`. Such a run — or, for `dashboard_summary`, such a
237
+ criterion — is excluded from pass@K and from the quality score rather than counted as a failure: scoring it 0
238
+ would be indistinguishable from the judge genuinely failing the answer, which is the confusion
239
+ `JudgeResponseError` exists to end. `pass@K` still holds on the runs that *were* graded, so an item can pass with
240
+ `runs_ungraded` set; `pass^K` cannot, because a run nobody graded leaves "all K passed" unverified. For
241
+ `dashboard_summary` an ungraded `must_include` / `must_not_include` criterion likewise cannot carry a pass —
242
+ "the judge could not tell" is not evidence the fact is present — while an ungraded `rubric` line only narrows
243
+ the quality score. When *no* run could be graded (for `dashboard_summary`: no gating criterion on any run) the
244
+ item errors instead of reporting failures. A non-zero count means pass@K was computed over fewer runs than
245
+ `--runs` asked for, so treat the result as weaker evidence and check the judge (`GD_EVAL_JUDGE_DIAGNOSTICS=1`,
246
+ or raise `JUDGE_MAX_COMPLETION_TOKENS` if the cause is `finish_reason=length`). The console says so in `Notes`
247
+ (`1 run(s) ungraded`, `2 criterion(s) ungraded`).
248
+
249
+ `agent_s` + `judge_s` + `simulated_user_s` are the instrumented parts of the item's `latency_s`; they do not add
250
+ up to it exactly, because `latency_s` is wall-clock around the whole item and also covers the conversation
251
+ create/delete round trips, SDK construction and any cleanup. `langfuse_s` sits **beside** `latency_s`, never
252
+ inside it, because trace linking runs outside every item's critical path (see above) — summing all four would
253
+ re-inflate exactly what that design removes.
254
+
255
+ A phase that a kind does not have reports `0.0` rather than an invented number, so read the zeroes as "not
256
+ applicable here", not "instant". Today:
257
+
258
+ | Field | Populated by |
259
+ |---|---|
260
+ | `agent_s` | `agentic_general_question`, `agentic_metric_skill` |
261
+ | `judge_s` | `agentic_general_question` only — `agentic_metric_skill` compares MAQL by string, it has no LLM judge |
262
+ | `simulated_user_s` | `agentic_metric_skill` only — `agentic_general_question` is single-turn, it has no simulated user |
263
+ | `langfuse_s` | every agentic kind, but only on the `gd-eval` path and only when Langfuse credentials are present |
264
+
265
+ The other six agentic kinds report `0.0` for the first three. Trace linking itself happens whenever
266
+ `LANGFUSE_PUBLIC_KEY`/`LANGFUSE_SECRET_KEY` are exported, with or without `--langfuse`, because each
267
+ `evaluate_agentic_*` falls back to `try_make_langfuse_client()`. But its *duration* is measured by the CLI
268
+ runner rather than by `evaluate_agentic_*`, so a direct library caller sees `langfuse_s: 0.0` even though its
269
+ linking ran. Pass `TAVERN_E2E_SKIP_TRACE_LINK=1` to opt out of linking altogether.
270
+
165
271
  ---
166
272
 
167
273
  ## `gd-eval models`
@@ -1,7 +1,7 @@
1
1
  # (C) 2026 GoodData Corporation
2
2
  [project]
3
3
  name = "gooddata-eval"
4
- version = "1.74.0"
4
+ version = "1.74.1.dev2"
5
5
  description = "Evaluate the GoodData AI agent against your own questions and models."
6
6
  readme = "README.md"
7
7
  license = "MIT"
@@ -11,7 +11,7 @@ authors = [
11
11
  keywords = ["gooddata", "ai", "evaluation", "llm", "analytics", "cli"]
12
12
  requires-python = ">=3.10"
13
13
  dependencies = [
14
- "gooddata-sdk~=1.74.0",
14
+ "gooddata-sdk~=1.74.1.dev2",
15
15
  "httpx>=0.27,<1.0",
16
16
  "orjson>=3.9.15,<4.0.0",
17
17
  "pydantic>=2.6,<3.0",
@@ -30,7 +30,7 @@ classifiers = [
30
30
  ]
31
31
 
32
32
  [project.optional-dependencies]
33
- llm-judge = ["openai>=1.40,<2.0"]
33
+ llm-judge = ["openai>=1.45,<2.0"]
34
34
 
35
35
  [project.scripts]
36
36
  gd-eval = "gooddata_eval.cli.main:main"
@@ -43,8 +43,8 @@ dev = [
43
43
  "pytest>=8.3.5",
44
44
  ]
45
45
  test = [
46
- "pytest~=8.3.4",
47
- "pytest-cov~=6.0.0",
46
+ "pytest~=9.1.1",
47
+ "pytest-cov~=7.1.0",
48
48
  "pytest-json-report==1.5.0",
49
49
  "pytest-mock>=3.14.0",
50
50
  ]