gooddata-eval 1.74.1.dev2__py3-none-any.whl → 1.74.1.dev4__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- gooddata_eval/cli/agentic_runner.py +32 -2
- gooddata_eval/cli/main.py +62 -11
- gooddata_eval/core/agentic/_gate.py +70 -0
- gooddata_eval/core/agentic/_langfuse.py +199 -240
- gooddata_eval/core/agentic/_trace_linker.py +24 -5
- gooddata_eval/core/agentic/alert_skill.py +16 -3
- gooddata_eval/core/agentic/conversation.py +10 -1
- gooddata_eval/core/agentic/general_question.py +18 -3
- gooddata_eval/core/agentic/guardrail.py +20 -3
- gooddata_eval/core/agentic/kda_skill.py +16 -3
- gooddata_eval/core/agentic/metric_skill.py +21 -3
- gooddata_eval/core/agentic/search_tool.py +18 -3
- gooddata_eval/core/agentic/visualization.py +26 -9
- gooddata_eval/core/config.py +21 -0
- gooddata_eval/core/dataset/langfuse_source.py +4 -14
- gooddata_eval/core/langfuse/_env.py +39 -0
- gooddata_eval/core/langfuse/client.py +205 -0
- gooddata_eval/core/langfuse/experiment.py +156 -0
- gooddata_eval/core/langfuse/observations.py +125 -0
- gooddata_eval/core/langfuse/otlp.py +164 -0
- gooddata_eval/core/langfuse/sink.py +105 -118
- gooddata_eval/core/reporting/console.py +12 -2
- gooddata_eval/core/reporting/json_report.py +7 -1
- gooddata_eval/core/runner.py +20 -3
- {gooddata_eval-1.74.1.dev2.dist-info → gooddata_eval-1.74.1.dev4.dist-info}/METADATA +58 -14
- {gooddata_eval-1.74.1.dev2.dist-info → gooddata_eval-1.74.1.dev4.dist-info}/RECORD +29 -23
- {gooddata_eval-1.74.1.dev2.dist-info → gooddata_eval-1.74.1.dev4.dist-info}/WHEEL +0 -0
- {gooddata_eval-1.74.1.dev2.dist-info → gooddata_eval-1.74.1.dev4.dist-info}/entry_points.txt +0 -0
- {gooddata_eval-1.74.1.dev2.dist-info → gooddata_eval-1.74.1.dev4.dist-info}/licenses/LICENSE.txt +0 -0
|
@@ -8,6 +8,7 @@ import time
|
|
|
8
8
|
from concurrent.futures import ThreadPoolExecutor, as_completed
|
|
9
9
|
from typing import Any, TypedDict
|
|
10
10
|
|
|
11
|
+
from gooddata_eval.core.agentic._gate import DEFAULT_GATE, EvalGate, normalize_gate
|
|
11
12
|
from gooddata_eval.core.agentic._langfuse import make_langfuse_client
|
|
12
13
|
from gooddata_eval.core.agentic._trace_linker import BackgroundTraceLinker, SubmitTraceLink, run_trace_link_inline
|
|
13
14
|
from gooddata_eval.core.agentic.alert_skill import evaluate_agentic_alert_skill
|
|
@@ -48,6 +49,13 @@ AGENTIC_TEST_KINDS = frozenset(
|
|
|
48
49
|
)
|
|
49
50
|
|
|
50
51
|
|
|
52
|
+
# Agentic kinds that no gate applies to: they drive their fixture exactly once, so there is
|
|
53
|
+
# no K to take pass@K or pass^K over. Named here rather than inline in _dispatch_agentic so
|
|
54
|
+
# the CLI can refuse --gate power for a dataset containing one instead of labelling the whole
|
|
55
|
+
# report `power` when part of it was never gated.
|
|
56
|
+
UNGATED_AGENTIC_TEST_KINDS = frozenset({"agentic_conversation"})
|
|
57
|
+
|
|
58
|
+
|
|
51
59
|
# Kinds cleared to run several at a time. An EXPLICIT allowlist, not a subtraction: nothing
|
|
52
60
|
# in this package can prove a kind is read-only, because the mutation happens server-side in
|
|
53
61
|
# whichever tools the agent decides to call. So each entry here is a reviewed judgement, and
|
|
@@ -128,9 +136,13 @@ def _dispatch_agentic(
|
|
|
128
136
|
reasoning_effort: ReasoningEffort | None = None,
|
|
129
137
|
agent_id: str | None = None,
|
|
130
138
|
submit_trace_link: SubmitTraceLink = run_trace_link_inline,
|
|
139
|
+
gate: EvalGate = DEFAULT_GATE,
|
|
131
140
|
) -> AgenticEvalOutcome:
|
|
132
141
|
"""Call the appropriate evaluate_agentic_* function for the item's test_kind.
|
|
133
142
|
|
|
143
|
+
`gate` reaches every kind except those in UNGATED_AGENTIC_TEST_KINDS, which have no K
|
|
144
|
+
to gate over; the CLI refuses --gate power for a dataset containing one.
|
|
145
|
+
|
|
134
146
|
Every evaluate_agentic_* function returns an AgenticEvalOutcome (reasoning_steps,
|
|
135
147
|
conversation_id, response_id, detail) on success and attaches the same four attributes
|
|
136
148
|
to its raised *AssertionError on failure -- no kind is exempt.
|
|
@@ -155,6 +167,7 @@ def _dispatch_agentic(
|
|
|
155
167
|
question=item.question,
|
|
156
168
|
expected_outputs=_parse_visualization_expected(eo),
|
|
157
169
|
k=k,
|
|
170
|
+
gate=gate,
|
|
158
171
|
agent_id=agent_id,
|
|
159
172
|
**lf_kw,
|
|
160
173
|
)
|
|
@@ -166,6 +179,7 @@ def _dispatch_agentic(
|
|
|
166
179
|
question=item.question,
|
|
167
180
|
expected_output=eo if isinstance(eo, (dict, list)) else {},
|
|
168
181
|
k=k,
|
|
182
|
+
gate=gate,
|
|
169
183
|
agent_id=agent_id,
|
|
170
184
|
**lf_kw,
|
|
171
185
|
)
|
|
@@ -177,6 +191,7 @@ def _dispatch_agentic(
|
|
|
177
191
|
question=item.question,
|
|
178
192
|
expected_output=eo if isinstance(eo, dict) else {},
|
|
179
193
|
k=k,
|
|
194
|
+
gate=gate,
|
|
180
195
|
agent_id=agent_id,
|
|
181
196
|
**lf_kw,
|
|
182
197
|
)
|
|
@@ -191,6 +206,7 @@ def _dispatch_agentic(
|
|
|
191
206
|
question=item.question,
|
|
192
207
|
expected_tool_call=expected_args,
|
|
193
208
|
k=k,
|
|
209
|
+
gate=gate,
|
|
194
210
|
agent_id=agent_id,
|
|
195
211
|
**lf_kw,
|
|
196
212
|
)
|
|
@@ -202,6 +218,7 @@ def _dispatch_agentic(
|
|
|
202
218
|
question=item.question,
|
|
203
219
|
expected_output=eo if isinstance(eo, str) else str(eo),
|
|
204
220
|
k=k,
|
|
221
|
+
gate=gate,
|
|
205
222
|
agent_id=agent_id,
|
|
206
223
|
user_context=item.user_context,
|
|
207
224
|
**lf_kw,
|
|
@@ -214,6 +231,7 @@ def _dispatch_agentic(
|
|
|
214
231
|
question=item.question,
|
|
215
232
|
expected_output=eo if isinstance(eo, str) else str(eo),
|
|
216
233
|
k=k,
|
|
234
|
+
gate=gate,
|
|
217
235
|
agent_id=agent_id,
|
|
218
236
|
**lf_kw,
|
|
219
237
|
)
|
|
@@ -225,6 +243,7 @@ def _dispatch_agentic(
|
|
|
225
243
|
question=item.question,
|
|
226
244
|
expected_output=eo if isinstance(eo, dict) else {},
|
|
227
245
|
k=k,
|
|
246
|
+
gate=gate,
|
|
228
247
|
agent_id=agent_id,
|
|
229
248
|
**lf_kw,
|
|
230
249
|
)
|
|
@@ -291,6 +310,7 @@ def run_agentic_items(
|
|
|
291
310
|
on_item_done: Any = None,
|
|
292
311
|
agent_id: str | None = None,
|
|
293
312
|
concurrency: int = 1,
|
|
313
|
+
gate: EvalGate = DEFAULT_GATE,
|
|
294
314
|
) -> EvalReport:
|
|
295
315
|
"""Run agentic items through evaluate_agentic_* and return an EvalReport.
|
|
296
316
|
|
|
@@ -303,7 +323,7 @@ def run_agentic_items(
|
|
|
303
323
|
"""
|
|
304
324
|
langfuse = make_langfuse_client() if use_langfuse else None
|
|
305
325
|
|
|
306
|
-
report = EvalReport(model=model_version)
|
|
326
|
+
report = EvalReport(model=model_version, gate=normalize_gate(gate))
|
|
307
327
|
total = len(items)
|
|
308
328
|
# Trace linking runs here rather than inside each evaluate_agentic_*, so an item's
|
|
309
329
|
# Langfuse poll overlaps the NEXT item's agent call instead of extending its own
|
|
@@ -325,6 +345,9 @@ def run_agentic_items(
|
|
|
325
345
|
test_kind=item.test_kind,
|
|
326
346
|
question=item.question,
|
|
327
347
|
)
|
|
348
|
+
# None, not False, for the kinds _dispatch_agentic passes no gate to: ItemReport.passed
|
|
349
|
+
# then falls back to pass_at_k and gate_passed keeps meaning "a gate ran".
|
|
350
|
+
gated = item.test_kind not in UNGATED_AGENTIC_TEST_KINDS
|
|
328
351
|
t0 = time.perf_counter()
|
|
329
352
|
try:
|
|
330
353
|
outcome = _dispatch_agentic(
|
|
@@ -339,6 +362,7 @@ def run_agentic_items(
|
|
|
339
362
|
reasoning_effort,
|
|
340
363
|
agent_id,
|
|
341
364
|
submit_trace_link=linker.submit,
|
|
365
|
+
gate=gate,
|
|
342
366
|
)
|
|
343
367
|
if isinstance(outcome, AgenticEvalOutcome):
|
|
344
368
|
reasoning_steps = outcome.reasoning_steps
|
|
@@ -347,6 +371,8 @@ def run_agentic_items(
|
|
|
347
371
|
detail = outcome.detail
|
|
348
372
|
else:
|
|
349
373
|
reasoning_steps, conversation_id, response_id, detail = outcome, None, None, {}
|
|
374
|
+
item_report.gate_passed = True if gated else None
|
|
375
|
+
# Whichever gate decided the item, clearing it means at least one run passed.
|
|
350
376
|
item_report.pass_at_k = True
|
|
351
377
|
item_report.runs = k
|
|
352
378
|
item_report.reasoning_steps = reasoning_steps or []
|
|
@@ -356,7 +382,7 @@ def run_agentic_items(
|
|
|
356
382
|
_apply_timings(item_report, getattr(outcome, "timings", None))
|
|
357
383
|
_apply_run_counts(item_report, outcome)
|
|
358
384
|
except AssertionError as exc:
|
|
359
|
-
item_report.
|
|
385
|
+
item_report.gate_passed = False if gated else None
|
|
360
386
|
item_report.runs = k
|
|
361
387
|
item_report.reasoning_steps = getattr(exc, "reasoning_steps", None) or []
|
|
362
388
|
item_report.conversation_id = getattr(exc, "conversation_id", None)
|
|
@@ -364,6 +390,10 @@ def run_agentic_items(
|
|
|
364
390
|
item_report.best_detail = getattr(exc, "detail", None) or {}
|
|
365
391
|
_apply_timings(item_report, getattr(exc, "timings", None))
|
|
366
392
|
_apply_run_counts(item_report, exc)
|
|
393
|
+
# Read off the counts, not off the gate: pass^K fails items where runs did pass,
|
|
394
|
+
# and reporting those as pass_at_k False would contradict the Langfuse score of
|
|
395
|
+
# the same name. Kinds that report no count read as 0, i.e. a clean failure.
|
|
396
|
+
item_report.pass_at_k = item_report.runs_passed > 0
|
|
367
397
|
print(f"[agentic] {item.id} FAIL: {exc}", flush=True)
|
|
368
398
|
except Exception as exc:
|
|
369
399
|
item_report.error = f"{type(exc).__name__}: {exc}"
|
gooddata_eval/cli/main.py
CHANGED
|
@@ -14,11 +14,20 @@ from gooddata_api_client.exceptions import ApiException
|
|
|
14
14
|
from rich.console import Console
|
|
15
15
|
from rich.table import Table
|
|
16
16
|
|
|
17
|
-
from gooddata_eval.cli.agentic_runner import AGENTIC_TEST_KINDS, run_agentic_items
|
|
17
|
+
from gooddata_eval.cli.agentic_runner import AGENTIC_TEST_KINDS, UNGATED_AGENTIC_TEST_KINDS, run_agentic_items
|
|
18
18
|
from gooddata_eval.core.chat.sse_client import ChatClient
|
|
19
|
-
from gooddata_eval.core.config import
|
|
19
|
+
from gooddata_eval.core.config import (
|
|
20
|
+
DEFAULT_GATE,
|
|
21
|
+
DEFAULT_JUDGE_MODEL,
|
|
22
|
+
JUDGE_MODEL_ENV_VAR,
|
|
23
|
+
EvalGate,
|
|
24
|
+
ReasoningEffort,
|
|
25
|
+
RunConfig,
|
|
26
|
+
normalize_gate,
|
|
27
|
+
)
|
|
20
28
|
from gooddata_eval.core.connection import ConnectionError_, resolve_connection
|
|
21
29
|
from gooddata_eval.core.dataset.local import load_local_dataset
|
|
30
|
+
from gooddata_eval.core.evaluators import supported_test_kinds
|
|
22
31
|
from gooddata_eval.core.langfuse.sink import LangfuseSink
|
|
23
32
|
from gooddata_eval.core.models import ChatResult, DatasetItem
|
|
24
33
|
from gooddata_eval.core.reporting.console import render_comparison, render_console
|
|
@@ -91,7 +100,15 @@ def _build_parser() -> argparse.ArgumentParser:
|
|
|
91
100
|
"Default: workspace's current active model."
|
|
92
101
|
),
|
|
93
102
|
)
|
|
94
|
-
run.add_argument("--runs", type=int, default=2, help="Independent runs per item
|
|
103
|
+
run.add_argument("--runs", type=int, default=2, help="Independent runs per item. Default 2.")
|
|
104
|
+
run.add_argument(
|
|
105
|
+
"--gate",
|
|
106
|
+
choices=get_args(EvalGate),
|
|
107
|
+
default=DEFAULT_GATE,
|
|
108
|
+
help="Which verdict decides an item: 'any' = pass@K (a run passing is enough, the "
|
|
109
|
+
"default and historic behaviour), 'power' = pass^K (every run must pass, so the verdict "
|
|
110
|
+
"measures stability). Identical at --runs 1. Agentic kinds only.",
|
|
111
|
+
)
|
|
95
112
|
run.add_argument(
|
|
96
113
|
"--concurrency",
|
|
97
114
|
type=int,
|
|
@@ -180,16 +197,46 @@ def _apply_timer_flag(enabled: bool) -> None:
|
|
|
180
197
|
os.environ[TIMERS_ENV_VAR] = "1"
|
|
181
198
|
|
|
182
199
|
|
|
200
|
+
def _reject_power_gate_on_ungated_items(config: RunConfig, items: list) -> None:
|
|
201
|
+
"""Refuse a pass^K request the run cannot honour for every item.
|
|
202
|
+
|
|
203
|
+
Two kinds of item are never gated: everything on the non-agentic path, because
|
|
204
|
+
`run_items` has no gate and always decides on pass@K, and agentic_conversation, which
|
|
205
|
+
drives its fixture once whatever --runs says and so has no K to gate over. Running a
|
|
206
|
+
mixed dataset anyway would decide part of it under each rule and label the whole report
|
|
207
|
+
`power`. test_kind is resolved per item, so a dataset does not have to be homogeneous.
|
|
208
|
+
|
|
209
|
+
Kinds no evaluator supports are not counted: those items are skipped rather than
|
|
210
|
+
decided, so refusing on them would make --gate power fail where --gate any runs.
|
|
211
|
+
"""
|
|
212
|
+
if normalize_gate(config.gate) != "power":
|
|
213
|
+
return
|
|
214
|
+
supported = supported_test_kinds()
|
|
215
|
+
ungated = [
|
|
216
|
+
i
|
|
217
|
+
for i in items
|
|
218
|
+
if i.test_kind in UNGATED_AGENTIC_TEST_KINDS
|
|
219
|
+
or (i.test_kind not in AGENTIC_TEST_KINDS and i.test_kind in supported)
|
|
220
|
+
]
|
|
221
|
+
if not ungated:
|
|
222
|
+
return
|
|
223
|
+
kinds = sorted({i.test_kind for i in ungated})
|
|
224
|
+
raise ValueError(
|
|
225
|
+
f"--gate power applies to kinds that repeat K runs, but this dataset has {len(ungated)} "
|
|
226
|
+
f"item(s) of kind {kinds}, which are always decided on pass@K. Run them separately, or "
|
|
227
|
+
f"use --gate any."
|
|
228
|
+
)
|
|
229
|
+
|
|
230
|
+
|
|
183
231
|
def _warn_if_local_dataset_cannot_link(config: RunConfig, agentic_items: list) -> None:
|
|
184
|
-
"""Say up front that
|
|
232
|
+
"""Say up front that experiment assembly will fail, rather than after the run.
|
|
185
233
|
|
|
186
234
|
--langfuse is refused outright with a local dataset because local item ids cannot be
|
|
187
235
|
linked. But every evaluate_agentic_* falls back to try_make_langfuse_client() when the
|
|
188
|
-
caller passes none, so with LANGFUSE_* exported the linking runs anyway and
|
|
189
|
-
|
|
190
|
-
|
|
191
|
-
|
|
192
|
-
instead of disabling it.
|
|
236
|
+
caller passes none, so with LANGFUSE_* exported the linking runs anyway and every
|
|
237
|
+
dataset-item lookup 404s -- arriving in a block at the very end of the run, long after
|
|
238
|
+
the flag that would have prevented it could be changed. The fallback is deliberate
|
|
239
|
+
(direct library and tavern callers rely on it), so this warns instead of disabling it.
|
|
193
240
|
"""
|
|
194
241
|
from gooddata_eval.core.agentic._langfuse import SKIP_ENV_VAR, langfuse_credentials_present # noqa: PLC0415
|
|
195
242
|
from gooddata_eval.core.config import env_flag # noqa: PLC0415
|
|
@@ -201,7 +248,7 @@ def _warn_if_local_dataset_cannot_link(config: RunConfig, agentic_items: list) -
|
|
|
201
248
|
print(
|
|
202
249
|
f"warning: --dataset is a local folder, so its item ids are not Langfuse dataset item ids. "
|
|
203
250
|
f"Traces will be found and scored, but the per-run grouping that makes models comparable "
|
|
204
|
-
f"cannot be created and each conversation will report
|
|
251
|
+
f"cannot be created and each conversation will report that its item does not exist in Langfuse. "
|
|
205
252
|
f"Use --langfuse-dataset for comparable runs, or set {SKIP_ENV_VAR}=1 to skip trace linking.",
|
|
206
253
|
file=sys.stderr,
|
|
207
254
|
)
|
|
@@ -253,7 +300,7 @@ def _make_progress_callbacks(console: Console):
|
|
|
253
300
|
tag = "[yellow]SKIP[/yellow]"
|
|
254
301
|
elif report.error:
|
|
255
302
|
tag = "[red]ERR [/red]"
|
|
256
|
-
elif report.
|
|
303
|
+
elif report.passed:
|
|
257
304
|
tag = "[green]PASS[/green]"
|
|
258
305
|
else:
|
|
259
306
|
tag = "[red]FAIL[/red]"
|
|
@@ -343,6 +390,7 @@ def _run(config: RunConfig) -> int:
|
|
|
343
390
|
items = _load_dataset(config)
|
|
344
391
|
agentic_items = [i for i in items if i.test_kind in AGENTIC_TEST_KINDS]
|
|
345
392
|
non_agentic_items = [i for i in items if i.test_kind not in AGENTIC_TEST_KINDS]
|
|
393
|
+
_reject_power_gate_on_ungated_items(config, items)
|
|
346
394
|
_warn_if_local_dataset_cannot_link(config, agentic_items)
|
|
347
395
|
models = config.models or []
|
|
348
396
|
run_ts = datetime.now(timezone.utc).strftime("%Y-%m-%d-%H-%M")
|
|
@@ -417,6 +465,7 @@ def _run(config: RunConfig) -> int:
|
|
|
417
465
|
token=config.token,
|
|
418
466
|
workspace_id=config.workspace_id,
|
|
419
467
|
k=config.runs,
|
|
468
|
+
gate=config.gate,
|
|
420
469
|
model_version=resolved.model_id,
|
|
421
470
|
reasoning_effort=config.reasoning_effort,
|
|
422
471
|
use_langfuse=config.log_to_langfuse,
|
|
@@ -466,6 +515,7 @@ def _run(config: RunConfig) -> int:
|
|
|
466
515
|
provider_name=resolved.provider_name or resolved.provider_id,
|
|
467
516
|
provider_type=resolved.provider_type,
|
|
468
517
|
workspace_id=config.workspace_id,
|
|
518
|
+
gate=config.gate,
|
|
469
519
|
)
|
|
470
520
|
if agentic_report is not None:
|
|
471
521
|
report.items.extend(agentic_report.items)
|
|
@@ -530,6 +580,7 @@ def main(argv: list[str] | None = None) -> int:
|
|
|
530
580
|
kind=args.kind,
|
|
531
581
|
preserve_failed=args.preserve_failed,
|
|
532
582
|
reasoning_effort=args.reasoning_effort,
|
|
583
|
+
gate=normalize_gate(args.gate),
|
|
533
584
|
agent_id=args.agent_id or os.environ.get("GD_EVAL_AGENT_ID"),
|
|
534
585
|
)
|
|
535
586
|
return _run(config)
|
|
@@ -0,0 +1,70 @@
|
|
|
1
|
+
# (C) 2026 GoodData Corporation
|
|
2
|
+
"""Which of pass@K / pass^K decides an item, and how both reach Langfuse."""
|
|
3
|
+
|
|
4
|
+
from __future__ import annotations
|
|
5
|
+
|
|
6
|
+
from typing import Any
|
|
7
|
+
|
|
8
|
+
from gooddata_eval.core.config import DEFAULT_GATE, EvalGate, normalize_gate
|
|
9
|
+
|
|
10
|
+
__all__ = [
|
|
11
|
+
"DEFAULT_GATE",
|
|
12
|
+
"EvalGate",
|
|
13
|
+
"gate_failure_note",
|
|
14
|
+
"gate_label",
|
|
15
|
+
"gate_passed",
|
|
16
|
+
"log_gate_scores",
|
|
17
|
+
"normalize_gate",
|
|
18
|
+
"stamp_gate_metadata",
|
|
19
|
+
]
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def gate_passed(gate: str | None, *, pass_at_k: bool, pass_power_k: bool) -> bool:
|
|
23
|
+
"""Whether the item passes under ``gate``."""
|
|
24
|
+
return pass_power_k if normalize_gate(gate) == "power" else pass_at_k
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def gate_label(gate: str | None, k: int) -> str:
|
|
28
|
+
"""Short name for the gate, e.g. ``pass^3`` or ``pass@2``."""
|
|
29
|
+
return f"pass^{k}" if normalize_gate(gate) == "power" else f"pass@{k}"
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def gate_failure_note(gate: str | None, runs_passed: int, runs_total: int, runs_ungraded: int = 0) -> str:
|
|
33
|
+
"""Gate and how many runs met it, for the assertion message.
|
|
34
|
+
|
|
35
|
+
Needed because the message body describes the BEST run, which under pass^K can be a run
|
|
36
|
+
that passed — so the reported detail on its own looks like a pass.
|
|
37
|
+
|
|
38
|
+
An ungraded run counts in runs_total but can never count in runs_passed, so the remainder
|
|
39
|
+
is not evidence of instability — saying so would blame the agent for a judge outage.
|
|
40
|
+
"""
|
|
41
|
+
label = gate_label(gate, runs_total)
|
|
42
|
+
note = f"Gate {label} failed: {runs_passed}/{runs_total} runs passed"
|
|
43
|
+
if runs_ungraded:
|
|
44
|
+
return f"{note}, {runs_ungraded} ungraded."
|
|
45
|
+
if normalize_gate(gate) == "power" and runs_passed:
|
|
46
|
+
return f"{note} — unstable, not a clean failure."
|
|
47
|
+
return f"{note}."
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def log_gate_scores(ctx: Any, trace_id: Any, *, gate: str | None, pass_at_k: bool, pass_power_k: bool) -> None:
|
|
51
|
+
"""Log both candidate verdicts and the one that decided.
|
|
52
|
+
|
|
53
|
+
The names carry no K on purpose: `pass_at_2` becomes `pass_at_3` the moment K changes,
|
|
54
|
+
splitting every Langfuse view built on the old name.
|
|
55
|
+
"""
|
|
56
|
+
ctx.score(trace_id, name="pass_at_k", value=pass_at_k, data_type="BOOLEAN")
|
|
57
|
+
ctx.score(trace_id, name="pass_power_k", value=pass_power_k, data_type="BOOLEAN")
|
|
58
|
+
ctx.score(
|
|
59
|
+
trace_id,
|
|
60
|
+
name="gate_passed",
|
|
61
|
+
value=gate_passed(gate, pass_at_k=pass_at_k, pass_power_k=pass_power_k),
|
|
62
|
+
data_type="BOOLEAN",
|
|
63
|
+
)
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
def stamp_gate_metadata(metadata: dict, *, k: int, gate: str | None) -> dict:
|
|
67
|
+
"""Record K and the gate on the dataset-run metadata (mutates and returns it)."""
|
|
68
|
+
metadata["eval_k"] = k
|
|
69
|
+
metadata["eval_gate"] = normalize_gate(gate)
|
|
70
|
+
return metadata
|