gooddata-eval 1.74.1.dev2__py3-none-any.whl → 1.74.1.dev4__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (29) hide show
  1. gooddata_eval/cli/agentic_runner.py +32 -2
  2. gooddata_eval/cli/main.py +62 -11
  3. gooddata_eval/core/agentic/_gate.py +70 -0
  4. gooddata_eval/core/agentic/_langfuse.py +199 -240
  5. gooddata_eval/core/agentic/_trace_linker.py +24 -5
  6. gooddata_eval/core/agentic/alert_skill.py +16 -3
  7. gooddata_eval/core/agentic/conversation.py +10 -1
  8. gooddata_eval/core/agentic/general_question.py +18 -3
  9. gooddata_eval/core/agentic/guardrail.py +20 -3
  10. gooddata_eval/core/agentic/kda_skill.py +16 -3
  11. gooddata_eval/core/agentic/metric_skill.py +21 -3
  12. gooddata_eval/core/agentic/search_tool.py +18 -3
  13. gooddata_eval/core/agentic/visualization.py +26 -9
  14. gooddata_eval/core/config.py +21 -0
  15. gooddata_eval/core/dataset/langfuse_source.py +4 -14
  16. gooddata_eval/core/langfuse/_env.py +39 -0
  17. gooddata_eval/core/langfuse/client.py +205 -0
  18. gooddata_eval/core/langfuse/experiment.py +156 -0
  19. gooddata_eval/core/langfuse/observations.py +125 -0
  20. gooddata_eval/core/langfuse/otlp.py +164 -0
  21. gooddata_eval/core/langfuse/sink.py +105 -118
  22. gooddata_eval/core/reporting/console.py +12 -2
  23. gooddata_eval/core/reporting/json_report.py +7 -1
  24. gooddata_eval/core/runner.py +20 -3
  25. {gooddata_eval-1.74.1.dev2.dist-info → gooddata_eval-1.74.1.dev4.dist-info}/METADATA +58 -14
  26. {gooddata_eval-1.74.1.dev2.dist-info → gooddata_eval-1.74.1.dev4.dist-info}/RECORD +29 -23
  27. {gooddata_eval-1.74.1.dev2.dist-info → gooddata_eval-1.74.1.dev4.dist-info}/WHEEL +0 -0
  28. {gooddata_eval-1.74.1.dev2.dist-info → gooddata_eval-1.74.1.dev4.dist-info}/entry_points.txt +0 -0
  29. {gooddata_eval-1.74.1.dev2.dist-info → gooddata_eval-1.74.1.dev4.dist-info}/licenses/LICENSE.txt +0 -0
@@ -8,6 +8,7 @@ import time
8
8
  from concurrent.futures import ThreadPoolExecutor, as_completed
9
9
  from typing import Any, TypedDict
10
10
 
11
+ from gooddata_eval.core.agentic._gate import DEFAULT_GATE, EvalGate, normalize_gate
11
12
  from gooddata_eval.core.agentic._langfuse import make_langfuse_client
12
13
  from gooddata_eval.core.agentic._trace_linker import BackgroundTraceLinker, SubmitTraceLink, run_trace_link_inline
13
14
  from gooddata_eval.core.agentic.alert_skill import evaluate_agentic_alert_skill
@@ -48,6 +49,13 @@ AGENTIC_TEST_KINDS = frozenset(
48
49
  )
49
50
 
50
51
 
52
+ # Agentic kinds that no gate applies to: they drive their fixture exactly once, so there is
53
+ # no K to take pass@K or pass^K over. Named here rather than inline in _dispatch_agentic so
54
+ # the CLI can refuse --gate power for a dataset containing one instead of labelling the whole
55
+ # report `power` when part of it was never gated.
56
+ UNGATED_AGENTIC_TEST_KINDS = frozenset({"agentic_conversation"})
57
+
58
+
51
59
  # Kinds cleared to run several at a time. An EXPLICIT allowlist, not a subtraction: nothing
52
60
  # in this package can prove a kind is read-only, because the mutation happens server-side in
53
61
  # whichever tools the agent decides to call. So each entry here is a reviewed judgement, and
@@ -128,9 +136,13 @@ def _dispatch_agentic(
128
136
  reasoning_effort: ReasoningEffort | None = None,
129
137
  agent_id: str | None = None,
130
138
  submit_trace_link: SubmitTraceLink = run_trace_link_inline,
139
+ gate: EvalGate = DEFAULT_GATE,
131
140
  ) -> AgenticEvalOutcome:
132
141
  """Call the appropriate evaluate_agentic_* function for the item's test_kind.
133
142
 
143
+ `gate` reaches every kind except those in UNGATED_AGENTIC_TEST_KINDS, which have no K
144
+ to gate over; the CLI refuses --gate power for a dataset containing one.
145
+
134
146
  Every evaluate_agentic_* function returns an AgenticEvalOutcome (reasoning_steps,
135
147
  conversation_id, response_id, detail) on success and attaches the same four attributes
136
148
  to its raised *AssertionError on failure -- no kind is exempt.
@@ -155,6 +167,7 @@ def _dispatch_agentic(
155
167
  question=item.question,
156
168
  expected_outputs=_parse_visualization_expected(eo),
157
169
  k=k,
170
+ gate=gate,
158
171
  agent_id=agent_id,
159
172
  **lf_kw,
160
173
  )
@@ -166,6 +179,7 @@ def _dispatch_agentic(
166
179
  question=item.question,
167
180
  expected_output=eo if isinstance(eo, (dict, list)) else {},
168
181
  k=k,
182
+ gate=gate,
169
183
  agent_id=agent_id,
170
184
  **lf_kw,
171
185
  )
@@ -177,6 +191,7 @@ def _dispatch_agentic(
177
191
  question=item.question,
178
192
  expected_output=eo if isinstance(eo, dict) else {},
179
193
  k=k,
194
+ gate=gate,
180
195
  agent_id=agent_id,
181
196
  **lf_kw,
182
197
  )
@@ -191,6 +206,7 @@ def _dispatch_agentic(
191
206
  question=item.question,
192
207
  expected_tool_call=expected_args,
193
208
  k=k,
209
+ gate=gate,
194
210
  agent_id=agent_id,
195
211
  **lf_kw,
196
212
  )
@@ -202,6 +218,7 @@ def _dispatch_agentic(
202
218
  question=item.question,
203
219
  expected_output=eo if isinstance(eo, str) else str(eo),
204
220
  k=k,
221
+ gate=gate,
205
222
  agent_id=agent_id,
206
223
  user_context=item.user_context,
207
224
  **lf_kw,
@@ -214,6 +231,7 @@ def _dispatch_agentic(
214
231
  question=item.question,
215
232
  expected_output=eo if isinstance(eo, str) else str(eo),
216
233
  k=k,
234
+ gate=gate,
217
235
  agent_id=agent_id,
218
236
  **lf_kw,
219
237
  )
@@ -225,6 +243,7 @@ def _dispatch_agentic(
225
243
  question=item.question,
226
244
  expected_output=eo if isinstance(eo, dict) else {},
227
245
  k=k,
246
+ gate=gate,
228
247
  agent_id=agent_id,
229
248
  **lf_kw,
230
249
  )
@@ -291,6 +310,7 @@ def run_agentic_items(
291
310
  on_item_done: Any = None,
292
311
  agent_id: str | None = None,
293
312
  concurrency: int = 1,
313
+ gate: EvalGate = DEFAULT_GATE,
294
314
  ) -> EvalReport:
295
315
  """Run agentic items through evaluate_agentic_* and return an EvalReport.
296
316
 
@@ -303,7 +323,7 @@ def run_agentic_items(
303
323
  """
304
324
  langfuse = make_langfuse_client() if use_langfuse else None
305
325
 
306
- report = EvalReport(model=model_version)
326
+ report = EvalReport(model=model_version, gate=normalize_gate(gate))
307
327
  total = len(items)
308
328
  # Trace linking runs here rather than inside each evaluate_agentic_*, so an item's
309
329
  # Langfuse poll overlaps the NEXT item's agent call instead of extending its own
@@ -325,6 +345,9 @@ def run_agentic_items(
325
345
  test_kind=item.test_kind,
326
346
  question=item.question,
327
347
  )
348
+ # None, not False, for the kinds _dispatch_agentic passes no gate to: ItemReport.passed
349
+ # then falls back to pass_at_k and gate_passed keeps meaning "a gate ran".
350
+ gated = item.test_kind not in UNGATED_AGENTIC_TEST_KINDS
328
351
  t0 = time.perf_counter()
329
352
  try:
330
353
  outcome = _dispatch_agentic(
@@ -339,6 +362,7 @@ def run_agentic_items(
339
362
  reasoning_effort,
340
363
  agent_id,
341
364
  submit_trace_link=linker.submit,
365
+ gate=gate,
342
366
  )
343
367
  if isinstance(outcome, AgenticEvalOutcome):
344
368
  reasoning_steps = outcome.reasoning_steps
@@ -347,6 +371,8 @@ def run_agentic_items(
347
371
  detail = outcome.detail
348
372
  else:
349
373
  reasoning_steps, conversation_id, response_id, detail = outcome, None, None, {}
374
+ item_report.gate_passed = True if gated else None
375
+ # Whichever gate decided the item, clearing it means at least one run passed.
350
376
  item_report.pass_at_k = True
351
377
  item_report.runs = k
352
378
  item_report.reasoning_steps = reasoning_steps or []
@@ -356,7 +382,7 @@ def run_agentic_items(
356
382
  _apply_timings(item_report, getattr(outcome, "timings", None))
357
383
  _apply_run_counts(item_report, outcome)
358
384
  except AssertionError as exc:
359
- item_report.pass_at_k = False
385
+ item_report.gate_passed = False if gated else None
360
386
  item_report.runs = k
361
387
  item_report.reasoning_steps = getattr(exc, "reasoning_steps", None) or []
362
388
  item_report.conversation_id = getattr(exc, "conversation_id", None)
@@ -364,6 +390,10 @@ def run_agentic_items(
364
390
  item_report.best_detail = getattr(exc, "detail", None) or {}
365
391
  _apply_timings(item_report, getattr(exc, "timings", None))
366
392
  _apply_run_counts(item_report, exc)
393
+ # Read off the counts, not off the gate: pass^K fails items where runs did pass,
394
+ # and reporting those as pass_at_k False would contradict the Langfuse score of
395
+ # the same name. Kinds that report no count read as 0, i.e. a clean failure.
396
+ item_report.pass_at_k = item_report.runs_passed > 0
367
397
  print(f"[agentic] {item.id} FAIL: {exc}", flush=True)
368
398
  except Exception as exc:
369
399
  item_report.error = f"{type(exc).__name__}: {exc}"
gooddata_eval/cli/main.py CHANGED
@@ -14,11 +14,20 @@ from gooddata_api_client.exceptions import ApiException
14
14
  from rich.console import Console
15
15
  from rich.table import Table
16
16
 
17
- from gooddata_eval.cli.agentic_runner import AGENTIC_TEST_KINDS, run_agentic_items
17
+ from gooddata_eval.cli.agentic_runner import AGENTIC_TEST_KINDS, UNGATED_AGENTIC_TEST_KINDS, run_agentic_items
18
18
  from gooddata_eval.core.chat.sse_client import ChatClient
19
- from gooddata_eval.core.config import DEFAULT_JUDGE_MODEL, JUDGE_MODEL_ENV_VAR, ReasoningEffort, RunConfig
19
+ from gooddata_eval.core.config import (
20
+ DEFAULT_GATE,
21
+ DEFAULT_JUDGE_MODEL,
22
+ JUDGE_MODEL_ENV_VAR,
23
+ EvalGate,
24
+ ReasoningEffort,
25
+ RunConfig,
26
+ normalize_gate,
27
+ )
20
28
  from gooddata_eval.core.connection import ConnectionError_, resolve_connection
21
29
  from gooddata_eval.core.dataset.local import load_local_dataset
30
+ from gooddata_eval.core.evaluators import supported_test_kinds
22
31
  from gooddata_eval.core.langfuse.sink import LangfuseSink
23
32
  from gooddata_eval.core.models import ChatResult, DatasetItem
24
33
  from gooddata_eval.core.reporting.console import render_comparison, render_console
@@ -91,7 +100,15 @@ def _build_parser() -> argparse.ArgumentParser:
91
100
  "Default: workspace's current active model."
92
101
  ),
93
102
  )
94
- run.add_argument("--runs", type=int, default=2, help="Independent runs per item (pass@K). Default 2.")
103
+ run.add_argument("--runs", type=int, default=2, help="Independent runs per item. Default 2.")
104
+ run.add_argument(
105
+ "--gate",
106
+ choices=get_args(EvalGate),
107
+ default=DEFAULT_GATE,
108
+ help="Which verdict decides an item: 'any' = pass@K (a run passing is enough, the "
109
+ "default and historic behaviour), 'power' = pass^K (every run must pass, so the verdict "
110
+ "measures stability). Identical at --runs 1. Agentic kinds only.",
111
+ )
95
112
  run.add_argument(
96
113
  "--concurrency",
97
114
  type=int,
@@ -180,16 +197,46 @@ def _apply_timer_flag(enabled: bool) -> None:
180
197
  os.environ[TIMERS_ENV_VAR] = "1"
181
198
 
182
199
 
200
+ def _reject_power_gate_on_ungated_items(config: RunConfig, items: list) -> None:
201
+ """Refuse a pass^K request the run cannot honour for every item.
202
+
203
+ Two kinds of item are never gated: everything on the non-agentic path, because
204
+ `run_items` has no gate and always decides on pass@K, and agentic_conversation, which
205
+ drives its fixture once whatever --runs says and so has no K to gate over. Running a
206
+ mixed dataset anyway would decide part of it under each rule and label the whole report
207
+ `power`. test_kind is resolved per item, so a dataset does not have to be homogeneous.
208
+
209
+ Kinds no evaluator supports are not counted: those items are skipped rather than
210
+ decided, so refusing on them would make --gate power fail where --gate any runs.
211
+ """
212
+ if normalize_gate(config.gate) != "power":
213
+ return
214
+ supported = supported_test_kinds()
215
+ ungated = [
216
+ i
217
+ for i in items
218
+ if i.test_kind in UNGATED_AGENTIC_TEST_KINDS
219
+ or (i.test_kind not in AGENTIC_TEST_KINDS and i.test_kind in supported)
220
+ ]
221
+ if not ungated:
222
+ return
223
+ kinds = sorted({i.test_kind for i in ungated})
224
+ raise ValueError(
225
+ f"--gate power applies to kinds that repeat K runs, but this dataset has {len(ungated)} "
226
+ f"item(s) of kind {kinds}, which are always decided on pass@K. Run them separately, or "
227
+ f"use --gate any."
228
+ )
229
+
230
+
183
231
  def _warn_if_local_dataset_cannot_link(config: RunConfig, agentic_items: list) -> None:
184
- """Say up front that dataset-run assembly will fail, rather than after the run.
232
+ """Say up front that experiment assembly will fail, rather than after the run.
185
233
 
186
234
  --langfuse is refused outright with a local dataset because local item ids cannot be
187
235
  linked. But every evaluate_agentic_* falls back to try_make_langfuse_client() when the
188
- caller passes none, so with LANGFUSE_* exported the linking runs anyway and each
189
- conversation earns a 404 from dataset-run-items -- arriving in a block at the very end
190
- of the run, long after the flag that would have prevented it could be changed. The
191
- fallback is deliberate (direct library and tavern callers rely on it), so this warns
192
- instead of disabling it.
236
+ caller passes none, so with LANGFUSE_* exported the linking runs anyway and every
237
+ dataset-item lookup 404s -- arriving in a block at the very end of the run, long after
238
+ the flag that would have prevented it could be changed. The fallback is deliberate
239
+ (direct library and tavern callers rely on it), so this warns instead of disabling it.
193
240
  """
194
241
  from gooddata_eval.core.agentic._langfuse import SKIP_ENV_VAR, langfuse_credentials_present # noqa: PLC0415
195
242
  from gooddata_eval.core.config import env_flag # noqa: PLC0415
@@ -201,7 +248,7 @@ def _warn_if_local_dataset_cannot_link(config: RunConfig, agentic_items: list) -
201
248
  print(
202
249
  f"warning: --dataset is a local folder, so its item ids are not Langfuse dataset item ids. "
203
250
  f"Traces will be found and scored, but the per-run grouping that makes models comparable "
204
- f"cannot be created and each conversation will report a 404 from dataset-run-items. "
251
+ f"cannot be created and each conversation will report that its item does not exist in Langfuse. "
205
252
  f"Use --langfuse-dataset for comparable runs, or set {SKIP_ENV_VAR}=1 to skip trace linking.",
206
253
  file=sys.stderr,
207
254
  )
@@ -253,7 +300,7 @@ def _make_progress_callbacks(console: Console):
253
300
  tag = "[yellow]SKIP[/yellow]"
254
301
  elif report.error:
255
302
  tag = "[red]ERR [/red]"
256
- elif report.pass_at_k:
303
+ elif report.passed:
257
304
  tag = "[green]PASS[/green]"
258
305
  else:
259
306
  tag = "[red]FAIL[/red]"
@@ -343,6 +390,7 @@ def _run(config: RunConfig) -> int:
343
390
  items = _load_dataset(config)
344
391
  agentic_items = [i for i in items if i.test_kind in AGENTIC_TEST_KINDS]
345
392
  non_agentic_items = [i for i in items if i.test_kind not in AGENTIC_TEST_KINDS]
393
+ _reject_power_gate_on_ungated_items(config, items)
346
394
  _warn_if_local_dataset_cannot_link(config, agentic_items)
347
395
  models = config.models or []
348
396
  run_ts = datetime.now(timezone.utc).strftime("%Y-%m-%d-%H-%M")
@@ -417,6 +465,7 @@ def _run(config: RunConfig) -> int:
417
465
  token=config.token,
418
466
  workspace_id=config.workspace_id,
419
467
  k=config.runs,
468
+ gate=config.gate,
420
469
  model_version=resolved.model_id,
421
470
  reasoning_effort=config.reasoning_effort,
422
471
  use_langfuse=config.log_to_langfuse,
@@ -466,6 +515,7 @@ def _run(config: RunConfig) -> int:
466
515
  provider_name=resolved.provider_name or resolved.provider_id,
467
516
  provider_type=resolved.provider_type,
468
517
  workspace_id=config.workspace_id,
518
+ gate=config.gate,
469
519
  )
470
520
  if agentic_report is not None:
471
521
  report.items.extend(agentic_report.items)
@@ -530,6 +580,7 @@ def main(argv: list[str] | None = None) -> int:
530
580
  kind=args.kind,
531
581
  preserve_failed=args.preserve_failed,
532
582
  reasoning_effort=args.reasoning_effort,
583
+ gate=normalize_gate(args.gate),
533
584
  agent_id=args.agent_id or os.environ.get("GD_EVAL_AGENT_ID"),
534
585
  )
535
586
  return _run(config)
@@ -0,0 +1,70 @@
1
+ # (C) 2026 GoodData Corporation
2
+ """Which of pass@K / pass^K decides an item, and how both reach Langfuse."""
3
+
4
+ from __future__ import annotations
5
+
6
+ from typing import Any
7
+
8
+ from gooddata_eval.core.config import DEFAULT_GATE, EvalGate, normalize_gate
9
+
10
+ __all__ = [
11
+ "DEFAULT_GATE",
12
+ "EvalGate",
13
+ "gate_failure_note",
14
+ "gate_label",
15
+ "gate_passed",
16
+ "log_gate_scores",
17
+ "normalize_gate",
18
+ "stamp_gate_metadata",
19
+ ]
20
+
21
+
22
+ def gate_passed(gate: str | None, *, pass_at_k: bool, pass_power_k: bool) -> bool:
23
+ """Whether the item passes under ``gate``."""
24
+ return pass_power_k if normalize_gate(gate) == "power" else pass_at_k
25
+
26
+
27
+ def gate_label(gate: str | None, k: int) -> str:
28
+ """Short name for the gate, e.g. ``pass^3`` or ``pass@2``."""
29
+ return f"pass^{k}" if normalize_gate(gate) == "power" else f"pass@{k}"
30
+
31
+
32
+ def gate_failure_note(gate: str | None, runs_passed: int, runs_total: int, runs_ungraded: int = 0) -> str:
33
+ """Gate and how many runs met it, for the assertion message.
34
+
35
+ Needed because the message body describes the BEST run, which under pass^K can be a run
36
+ that passed — so the reported detail on its own looks like a pass.
37
+
38
+ An ungraded run counts in runs_total but can never count in runs_passed, so the remainder
39
+ is not evidence of instability — saying so would blame the agent for a judge outage.
40
+ """
41
+ label = gate_label(gate, runs_total)
42
+ note = f"Gate {label} failed: {runs_passed}/{runs_total} runs passed"
43
+ if runs_ungraded:
44
+ return f"{note}, {runs_ungraded} ungraded."
45
+ if normalize_gate(gate) == "power" and runs_passed:
46
+ return f"{note} — unstable, not a clean failure."
47
+ return f"{note}."
48
+
49
+
50
+ def log_gate_scores(ctx: Any, trace_id: Any, *, gate: str | None, pass_at_k: bool, pass_power_k: bool) -> None:
51
+ """Log both candidate verdicts and the one that decided.
52
+
53
+ The names carry no K on purpose: `pass_at_2` becomes `pass_at_3` the moment K changes,
54
+ splitting every Langfuse view built on the old name.
55
+ """
56
+ ctx.score(trace_id, name="pass_at_k", value=pass_at_k, data_type="BOOLEAN")
57
+ ctx.score(trace_id, name="pass_power_k", value=pass_power_k, data_type="BOOLEAN")
58
+ ctx.score(
59
+ trace_id,
60
+ name="gate_passed",
61
+ value=gate_passed(gate, pass_at_k=pass_at_k, pass_power_k=pass_power_k),
62
+ data_type="BOOLEAN",
63
+ )
64
+
65
+
66
+ def stamp_gate_metadata(metadata: dict, *, k: int, gate: str | None) -> dict:
67
+ """Record K and the gate on the dataset-run metadata (mutates and returns it)."""
68
+ metadata["eval_k"] = k
69
+ metadata["eval_gate"] = normalize_gate(gate)
70
+ return metadata