gooddata-eval 1.75.1.dev2__py3-none-any.whl → 1.75.1.dev3__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- gooddata_eval/core/agentic/dashboard_skill.py +534 -124
- {gooddata_eval-1.75.1.dev2.dist-info → gooddata_eval-1.75.1.dev3.dist-info}/METADATA +3 -2
- {gooddata_eval-1.75.1.dev2.dist-info → gooddata_eval-1.75.1.dev3.dist-info}/RECORD +6 -6
- {gooddata_eval-1.75.1.dev2.dist-info → gooddata_eval-1.75.1.dev3.dist-info}/WHEEL +0 -0
- {gooddata_eval-1.75.1.dev2.dist-info → gooddata_eval-1.75.1.dev3.dist-info}/entry_points.txt +0 -0
- {gooddata_eval-1.75.1.dev2.dist-info → gooddata_eval-1.75.1.dev3.dist-info}/licenses/LICENSE.txt +0 -0
|
@@ -7,6 +7,8 @@ import time
|
|
|
7
7
|
from dataclasses import dataclass, field
|
|
8
8
|
from typing import Any
|
|
9
9
|
|
|
10
|
+
import jsonpatch
|
|
11
|
+
|
|
10
12
|
from gooddata_eval.core.agentic._gate import (
|
|
11
13
|
DEFAULT_GATE,
|
|
12
14
|
EvalGate,
|
|
@@ -41,11 +43,14 @@ from gooddata_eval.core.timing import PhaseTimings, log_timer, sum_timings
|
|
|
41
43
|
_DEFAULT_K = 1
|
|
42
44
|
# Matches kda_skill and visualization. The reply this loop sends is built from the expectation
|
|
43
45
|
# and is byte-identical every turn, so the extra rounds are not there to say anything new --
|
|
44
|
-
# they are slack for a turn that
|
|
45
|
-
#
|
|
46
|
-
#
|
|
47
|
-
#
|
|
48
|
-
#
|
|
46
|
+
# they are slack for a turn that answered without drafting, which can happen for reasons that
|
|
47
|
+
# are not the model's. A turn that comes back with nothing at all does not consume the slack:
|
|
48
|
+
# the loop ends there rather than replying into silence. Fewer than metric_skill's seven
|
|
49
|
+
# because those rounds do carry new content, its reply being generated per turn by an LLM.
|
|
50
|
+
#
|
|
51
|
+
# Only creation ever reaches the later rounds. An edit stops after the first turn, since the
|
|
52
|
+
# creation reply names charts and a date range and answers nothing a rename or a resize could
|
|
53
|
+
# have asked.
|
|
49
54
|
#
|
|
50
55
|
# The cost of the slack is that four turns at ChatClient's 300s read timeout exceed the 720s
|
|
51
56
|
# per-test timeout gdc-nas derives for a k=1 dataset, so a run that stalls on every turn is cut
|
|
@@ -55,8 +60,15 @@ _DEFAULT_K = 1
|
|
|
55
60
|
_DEFAULT_MAX_ITERATIONS = 4
|
|
56
61
|
|
|
57
62
|
_DRAFT_TOOL = "draft_dashboard"
|
|
63
|
+
_PATCH_TOOL = "patch_dashboard"
|
|
58
64
|
_SET_SKILLS_TOOL = "set_skills"
|
|
59
65
|
_BUILDER_SKILL = "dashboard_builder"
|
|
66
|
+
_EDITOR_SKILL = "dashboard_editor"
|
|
67
|
+
# The expectation's `type` is the only switch between creating and editing: it selects the
|
|
68
|
+
# response part to read, the tool that must have succeeded, and the skill that must have been
|
|
69
|
+
# activated. There is no "either one" case -- an edit fixture that routes to the builder is a
|
|
70
|
+
# failure, not an alternative route to the same answer.
|
|
71
|
+
_PATCH_TYPE = "dashboardPatch"
|
|
60
72
|
|
|
61
73
|
|
|
62
74
|
def _norm(text: str) -> str:
|
|
@@ -64,14 +76,14 @@ def _norm(text: str) -> str:
|
|
|
64
76
|
return " ".join(text.split()).casefold()
|
|
65
77
|
|
|
66
78
|
|
|
67
|
-
def
|
|
68
|
-
"""Result payload of the ``
|
|
79
|
+
def _extract_tool_result(tool_call_events: list[ToolCallEvent], tool_name: str) -> dict | None:
|
|
80
|
+
"""Result payload of the ``tool_name`` call that produced the response.
|
|
69
81
|
|
|
70
82
|
Takes the most recent *successful* call: when the agent retries after a rejected
|
|
71
83
|
draft, the earlier failed attempt must not shadow the one that worked.
|
|
72
84
|
"""
|
|
73
85
|
for tc in reversed(tool_call_events):
|
|
74
|
-
if tc.function_name !=
|
|
86
|
+
if tc.function_name != tool_name or not tc.result:
|
|
75
87
|
continue
|
|
76
88
|
result_data = tc.parsed_result()
|
|
77
89
|
if not isinstance(result_data, dict):
|
|
@@ -83,14 +95,22 @@ def _extract_draft_result(tool_call_events: list[ToolCallEvent]) -> dict | None:
|
|
|
83
95
|
return None
|
|
84
96
|
|
|
85
97
|
|
|
86
|
-
def
|
|
87
|
-
"""Whether
|
|
98
|
+
def _skill_activated(tool_call_events: list[ToolCallEvent], skill: str) -> bool:
|
|
99
|
+
"""Whether the run's skill routing landed on ``skill``.
|
|
88
100
|
|
|
89
|
-
|
|
90
|
-
|
|
91
|
-
|
|
92
|
-
model that simply
|
|
101
|
+
Gated, because the expectation's type names one skill and only one: a creation fixture
|
|
102
|
+
answered by the editor, or an edit fixture answered by the builder, is a routing failure
|
|
103
|
+
even when the dashboard that comes back looks plausible. It also names the failure -- a
|
|
104
|
+
mis-route reads as a mis-route instead of as a model that simply produced nothing.
|
|
105
|
+
|
|
106
|
+
A run that never routed passes. Absence of a ``set_skills`` call is absence of evidence,
|
|
107
|
+
not evidence of a mis-route, and there are two ways to reach it that say nothing about the
|
|
108
|
+
model: gen-ai omits the call for an agent whose skills are pinned
|
|
109
|
+
(``include_set_skills=not resolved_pinned_skills``), and a run continuing an existing
|
|
110
|
+
conversation through ``initial_conversation_id`` can have routed in a turn this run never
|
|
111
|
+
saw. Failing those would fail the case for the harness's configuration.
|
|
93
112
|
"""
|
|
113
|
+
routed = False
|
|
94
114
|
for tc in tool_call_events:
|
|
95
115
|
if tc.function_name != _SET_SKILLS_TOOL or not tc.result:
|
|
96
116
|
continue
|
|
@@ -99,9 +119,27 @@ def _builder_skill_activated(tool_call_events: list[ToolCallEvent]) -> bool:
|
|
|
99
119
|
continue
|
|
100
120
|
payload = result_data.get("data", result_data)
|
|
101
121
|
skills = payload.get("skills_to_activate") if isinstance(payload, dict) else None
|
|
102
|
-
if isinstance(skills, list)
|
|
122
|
+
if not isinstance(skills, list):
|
|
123
|
+
continue
|
|
124
|
+
routed = True
|
|
125
|
+
if skill in skills:
|
|
103
126
|
return True
|
|
104
|
-
return
|
|
127
|
+
return not routed
|
|
128
|
+
|
|
129
|
+
|
|
130
|
+
def _is_edit(expected_output: dict) -> bool:
|
|
131
|
+
"""Whether the expectation describes an edit rather than a fresh dashboard."""
|
|
132
|
+
return _expected_type(expected_output) == _PATCH_TYPE
|
|
133
|
+
|
|
134
|
+
|
|
135
|
+
def _required_skill(expected_output: dict) -> str:
|
|
136
|
+
"""The skill the run must have activated, decided by the expectation's type."""
|
|
137
|
+
return _EDITOR_SKILL if _is_edit(expected_output) else _BUILDER_SKILL
|
|
138
|
+
|
|
139
|
+
|
|
140
|
+
def _producing_tool(expected_output: dict) -> str:
|
|
141
|
+
"""The tool whose successful call the response must come from."""
|
|
142
|
+
return _PATCH_TOOL if _is_edit(expected_output) else _DRAFT_TOOL
|
|
105
143
|
|
|
106
144
|
|
|
107
145
|
def _expected_type(expected_output: dict) -> str:
|
|
@@ -114,7 +152,7 @@ def _extract_dashboard_part(chat_result: ChatResult, part_type: str) -> dict | N
|
|
|
114
152
|
|
|
115
153
|
That type is in the SSE client's known-part set but has no dedicated accumulator, so it
|
|
116
154
|
arrives in ``unhandled_parts`` verbatim and is read back by type. Taken from the end for
|
|
117
|
-
the same reason ``
|
|
155
|
+
the same reason ``_extract_tool_result`` does: a turn that drafts and then refines must
|
|
118
156
|
be read as the state it left behind, not the one it passed through.
|
|
119
157
|
"""
|
|
120
158
|
for part in reversed(chat_result.unhandled_parts):
|
|
@@ -123,64 +161,113 @@ def _extract_dashboard_part(chat_result: ChatResult, part_type: str) -> dict | N
|
|
|
123
161
|
return None
|
|
124
162
|
|
|
125
163
|
|
|
164
|
+
def _sections_of(dashboard: dict) -> list[dict]:
|
|
165
|
+
"""The dashboard's sections, whichever shape the document uses.
|
|
166
|
+
|
|
167
|
+
A freshly drafted dashboard is version 3 and nests its sections under ``tabs``. A saved
|
|
168
|
+
dashboard relayed for editing can be version 2, which carries ``sections`` at the root and
|
|
169
|
+
no ``tabs`` at all. ``version`` is not the discriminator -- the convertor computes it from
|
|
170
|
+
the declarative input and separately flattens a single untitled tab into root sections, so
|
|
171
|
+
a document can say ``version: "3"`` and still have no tabs. Read the presence of ``tabs``.
|
|
172
|
+
"""
|
|
173
|
+
tabs = dashboard.get("tabs")
|
|
174
|
+
if tabs:
|
|
175
|
+
return [section for tab in tabs for section in tab.get("sections") or []]
|
|
176
|
+
return dashboard.get("sections") or []
|
|
177
|
+
|
|
178
|
+
|
|
126
179
|
def _widgets_of(dashboard: dict) -> list[dict]:
|
|
127
|
-
"""Every widget of the
|
|
180
|
+
"""Every widget of the dashboard, flattened across sections.
|
|
128
181
|
|
|
129
182
|
Charts are matched dashboard-wide on purpose: the expectation names charts, not a layout,
|
|
130
|
-
so which tab one lands on is the agent's call. The date filter is deliberately
|
|
131
|
-
opposite -- ``_check_date_range`` requires *every* tab to match. That is defensive
|
|
132
|
-
than observed: ``draft_dashboard`` copies
|
|
133
|
-
draft cannot currently disagree with itself across tabs, and the
|
|
134
|
-
notice if that
|
|
135
|
-
|
|
136
|
-
Only ``tabs`` is read. That is sound for a draft, which ``draft_dashboard`` always builds
|
|
137
|
-
tabbed, but an AAC v2 document carries its layout in root ``sections`` instead, so an
|
|
138
|
-
editing dataset cannot reuse this as it stands.
|
|
183
|
+
so which section or tab one lands on is the agent's call. The date filter is deliberately
|
|
184
|
+
the opposite -- ``_check_date_range`` requires *every* tab to match. That one is defensive
|
|
185
|
+
rather than observed: ``draft_dashboard`` copies a single filter onto every tab
|
|
186
|
+
unconditionally, so a draft cannot currently disagree with itself across tabs, and the
|
|
187
|
+
per-tab check is there to notice if that stops being true.
|
|
139
188
|
"""
|
|
140
|
-
return [
|
|
141
|
-
|
|
142
|
-
|
|
143
|
-
|
|
144
|
-
|
|
145
|
-
|
|
189
|
+
return [widget for section in _sections_of(dashboard) for widget in section.get("widgets") or []]
|
|
190
|
+
|
|
191
|
+
|
|
192
|
+
def _apply_patch(base: dict, operations: list[dict]) -> tuple[dict | None, str | None]:
|
|
193
|
+
"""Apply RFC 6902 ``operations`` to a copy of ``base``; return ``(result, failure)``.
|
|
194
|
+
|
|
195
|
+
The agent emits ``test`` guards ahead of its real operations, pinning the widget it is
|
|
196
|
+
about to touch. Those are applied rather than skipped: a failing guard means the patch was
|
|
197
|
+
written against a document that is not the one it shipped, which is a failure of the edit
|
|
198
|
+
even though every later operation might apply cleanly. Scoring the base instead of the
|
|
199
|
+
result would pass a case whose change never happened.
|
|
200
|
+
|
|
201
|
+
A patch carrying no operations is that same failure in its purest form. ``jsonpatch``
|
|
202
|
+
applies an empty list happily and hands back the document untouched, so the run would go on
|
|
203
|
+
to score the saved dashboard and pass every case the saved dashboard already satisfied.
|
|
204
|
+
"""
|
|
205
|
+
if not operations:
|
|
206
|
+
return None, "the patch carries no operations, so it proposes the dashboard it started from"
|
|
207
|
+
try:
|
|
208
|
+
# in_place=False is what makes the copy -- JsonPatch.apply deepcopies for us, so
|
|
209
|
+
# copying here as well would leave a second one behind for no reason.
|
|
210
|
+
return jsonpatch.JsonPatch(operations).apply(base, in_place=False), None
|
|
211
|
+
except Exception as exc: # jsonpatch raises several unrelated types for a bad patch
|
|
212
|
+
return None, f"the patch does not apply to the dashboard it shipped with: {exc}"
|
|
146
213
|
|
|
147
214
|
|
|
148
215
|
def _check_visualizations(widgets: list[dict], new_ids: set[str], expected: list[dict]) -> tuple[list[str], list[str]]:
|
|
149
|
-
"""Match every expected chart against the
|
|
216
|
+
"""Match every expected chart against the dashboard; extra widgets are allowed.
|
|
150
217
|
|
|
151
218
|
Returns ``(failures, title_notes)``.
|
|
152
219
|
|
|
153
|
-
An entry with an ``id`` is
|
|
154
|
-
|
|
155
|
-
|
|
156
|
-
|
|
220
|
+
An entry with an ``id`` is matched on that id. ``id: null`` marks a chart the agent had to
|
|
221
|
+
author, whose real id is generated at run time, so there the title *is* the match -- taken
|
|
222
|
+
among the authored charts only, since matching against every chart would let an existing
|
|
223
|
+
one of the same title satisfy it.
|
|
224
|
+
|
|
225
|
+
Title mismatches come back separately from the failures rather than mixed into them, so
|
|
226
|
+
the caller can gate on them or merely record them without the two decisions drifting: on
|
|
227
|
+
creation the title is ``title_override or fallback_title`` and the override is the model's
|
|
228
|
+
own choice, while on an edit the expectation states the title the change must leave behind
|
|
229
|
+
and a rename case that does not gate on it asserts nothing at all. Either way the
|
|
230
|
+
mismatches are returned, which is what keeps the reported score honest.
|
|
157
231
|
|
|
158
|
-
``
|
|
159
|
-
|
|
160
|
-
authored charts only, without which an existing chart of the same title would satisfy it.
|
|
232
|
+
``columns`` is checked whenever the expectation carries it. A newly placed widget defaults
|
|
233
|
+
to half width, so a resize case is only meaningful against the number.
|
|
161
234
|
"""
|
|
162
235
|
failures: list[str] = []
|
|
163
|
-
|
|
236
|
+
title_mismatches: list[str] = []
|
|
164
237
|
for exp in expected:
|
|
165
238
|
exp_id = exp.get("id")
|
|
166
239
|
exp_title = str(exp.get("title") or "")
|
|
240
|
+
exp_columns = exp.get("columns")
|
|
241
|
+
|
|
167
242
|
if exp_id is not None:
|
|
168
243
|
matches = [w for w in widgets if w.get("visualization") == exp_id]
|
|
169
244
|
if not matches:
|
|
170
245
|
failures.append(f"missing existing chart id={exp_id!r} title={exp_title!r}")
|
|
171
|
-
|
|
246
|
+
continue
|
|
247
|
+
titled = [w for w in matches if _norm(str(w.get("title") or "")) == _norm(exp_title)]
|
|
248
|
+
if not titled:
|
|
172
249
|
# Every title the id appears under: the same chart can be placed twice, and
|
|
173
250
|
# naming only the first would hide the rest.
|
|
174
251
|
titles = ", ".join(repr(w.get("title")) for w in matches)
|
|
175
|
-
|
|
176
|
-
|
|
177
|
-
|
|
178
|
-
|
|
179
|
-
|
|
180
|
-
|
|
181
|
-
|
|
182
|
-
|
|
183
|
-
|
|
252
|
+
title_mismatches.append(f"chart {exp_id!r} is titled {titles}, expected {exp_title!r}")
|
|
253
|
+
candidates = titled or matches
|
|
254
|
+
else:
|
|
255
|
+
candidates = [
|
|
256
|
+
w
|
|
257
|
+
for w in widgets
|
|
258
|
+
if w.get("visualization") in new_ids and _norm(str(w.get("title") or "")) == _norm(exp_title)
|
|
259
|
+
]
|
|
260
|
+
if not candidates:
|
|
261
|
+
failures.append(
|
|
262
|
+
f"missing authored chart title={exp_title!r}"
|
|
263
|
+
+ ("" if new_ids else " (the response listed no authored charts)")
|
|
264
|
+
)
|
|
265
|
+
continue
|
|
266
|
+
|
|
267
|
+
if exp_columns is not None and not any(w.get("columns") == exp_columns for w in candidates):
|
|
268
|
+
widths = ", ".join(repr(w.get("columns")) for w in candidates)
|
|
269
|
+
failures.append(f"chart {exp_title!r} is {widths} column(s) wide, expected {exp_columns}")
|
|
270
|
+
return failures, title_mismatches
|
|
184
271
|
|
|
185
272
|
|
|
186
273
|
def _check_references(widgets: list[dict], known_ids: set[str]) -> list[str]:
|
|
@@ -199,34 +286,148 @@ def _check_references(widgets: list[dict], known_ids: set[str]) -> list[str]:
|
|
|
199
286
|
return notes
|
|
200
287
|
|
|
201
288
|
|
|
289
|
+
def _filter_entries(dashboard: dict) -> list[tuple[str, dict]]:
|
|
290
|
+
"""``(label, filter)`` for every filter the dashboard carries, whichever shape it uses.
|
|
291
|
+
|
|
292
|
+
A drafted dashboard hangs one filter map off each tab, keyed by role (``date``). A saved
|
|
293
|
+
dashboard relayed for editing carries a single map at the root, keyed by the filter's own
|
|
294
|
+
local identifier, so the role has to be read from each entry's ``type`` instead. The label
|
|
295
|
+
is only there to say *where* a failure was found.
|
|
296
|
+
"""
|
|
297
|
+
tabs = dashboard.get("tabs")
|
|
298
|
+
if tabs:
|
|
299
|
+
return [
|
|
300
|
+
(f"tab {tab.get('id')!r}", value)
|
|
301
|
+
for tab in tabs
|
|
302
|
+
for value in (tab.get("filters") or {}).values()
|
|
303
|
+
if isinstance(value, dict)
|
|
304
|
+
]
|
|
305
|
+
return [
|
|
306
|
+
(f"filter {key!r}", value) for key, value in (dashboard.get("filters") or {}).items() if isinstance(value, dict)
|
|
307
|
+
]
|
|
308
|
+
|
|
309
|
+
|
|
310
|
+
def _filters_of_type(dashboard: dict, filter_type: str) -> list[tuple[str, dict]]:
|
|
311
|
+
return [(label, f) for label, f in _filter_entries(dashboard) if f.get("type") == filter_type]
|
|
312
|
+
|
|
313
|
+
|
|
314
|
+
def _date_filter_mismatch(label: str, date_filter: dict, expected: dict | None) -> list[str]:
|
|
315
|
+
"""Why ``date_filter`` is not the expected range, or nothing.
|
|
316
|
+
|
|
317
|
+
``expected is None`` means all time, which the document spells as a date filter with
|
|
318
|
+
neither bound — absent keys, not null ones.
|
|
319
|
+
"""
|
|
320
|
+
if expected is None:
|
|
321
|
+
if "from" in date_filter or "to" in date_filter:
|
|
322
|
+
return [f"{label}: expected all time, got from={date_filter.get('from')!r} to={date_filter.get('to')!r}"]
|
|
323
|
+
return []
|
|
324
|
+
return [
|
|
325
|
+
f"{label}: date {key} expected {expected.get(key)!r}, got {date_filter.get(key)!r}"
|
|
326
|
+
for key in ("granularity", "from", "to")
|
|
327
|
+
if date_filter.get(key) != expected.get(key)
|
|
328
|
+
]
|
|
329
|
+
|
|
330
|
+
|
|
202
331
|
def _check_date_range(dashboard: dict, expected: dict | None) -> list[str]:
|
|
203
|
-
"""
|
|
332
|
+
"""Check the dashboard's date range against the expectation.
|
|
333
|
+
|
|
334
|
+
What counts as "the" date filter differs by shape, so the two are checked differently.
|
|
335
|
+
|
|
336
|
+
A drafted dashboard carries one per tab, and the draft tool writes the same one onto every
|
|
337
|
+
tab, so every tab must have it and every one must match. A tab left without a date filter is
|
|
338
|
+
a failure: the expectation describes what the whole dashboard shows.
|
|
339
|
+
|
|
340
|
+
A saved dashboard relayed for editing carries a flat map that can legitimately hold more
|
|
341
|
+
than one date filter — a dashboard-wide one plus a dataset-scoped one — and nothing in the
|
|
342
|
+
document says which is which. Requiring every entry to match would fail a dashboard for
|
|
343
|
+
carrying a filter the fixture never described, so one matching entry satisfies the check and
|
|
344
|
+
the rest are left to the notes.
|
|
345
|
+
"""
|
|
346
|
+
date_filters = _filters_of_type(dashboard, "date_filter")
|
|
347
|
+
if not date_filters:
|
|
348
|
+
return ["the dashboard carries no date filter to check the expected range against"]
|
|
349
|
+
|
|
350
|
+
tabs = dashboard.get("tabs")
|
|
351
|
+
if tabs:
|
|
352
|
+
failures: list[str] = [
|
|
353
|
+
f"tab {tab.get('id')!r} carries no date filter"
|
|
354
|
+
for tab in tabs
|
|
355
|
+
if not any(
|
|
356
|
+
f.get("type") == "date_filter" for f in (tab.get("filters") or {}).values() if isinstance(f, dict)
|
|
357
|
+
)
|
|
358
|
+
]
|
|
359
|
+
for label, date_filter in date_filters:
|
|
360
|
+
failures.extend(_date_filter_mismatch(label, date_filter, expected))
|
|
361
|
+
return failures
|
|
362
|
+
|
|
363
|
+
mismatches = [_date_filter_mismatch(label, date_filter, expected) for label, date_filter in date_filters]
|
|
364
|
+
if any(not m for m in mismatches):
|
|
365
|
+
return []
|
|
366
|
+
return [reason for m in mismatches for reason in m]
|
|
204
367
|
|
|
205
|
-
|
|
206
|
-
|
|
368
|
+
|
|
369
|
+
def _selection_of(attribute_filter: dict) -> tuple[str, list] | None:
|
|
370
|
+
"""The elements an attribute filter restricts to, or ``None`` when it restricts nothing.
|
|
371
|
+
|
|
372
|
+
An unrestricted filter is spelled as the absence of a selection, and an empty list means
|
|
373
|
+
the same thing, so both read as "all".
|
|
207
374
|
"""
|
|
208
|
-
|
|
209
|
-
|
|
210
|
-
|
|
375
|
+
state = attribute_filter.get("state") or {}
|
|
376
|
+
for kind in ("include", "exclude"):
|
|
377
|
+
values = state.get(kind)
|
|
378
|
+
if values:
|
|
379
|
+
return kind, list(values)
|
|
380
|
+
return None
|
|
381
|
+
|
|
382
|
+
|
|
383
|
+
def _describe_selection(selection: tuple[str, list] | None) -> str:
|
|
384
|
+
return "all" if selection is None else f"{selection[0]} {selection[1]!r}"
|
|
385
|
+
|
|
386
|
+
|
|
387
|
+
def _check_filters(dashboard: dict, expected: list[dict]) -> list[str]:
|
|
388
|
+
"""Every expected attribute filter must be on the dashboard, with the stated selection.
|
|
389
|
+
|
|
390
|
+
This exists because an edit can silently drop or widen the filters the user already had:
|
|
391
|
+
the change asked for was a rename, and the filters coming back untouched is part of what
|
|
392
|
+
"untouched" means. Filters the dashboard carries beyond the expected ones are left alone,
|
|
393
|
+
the same way extra widgets are.
|
|
394
|
+
"""
|
|
395
|
+
actual: dict[str | None, list[dict]] = {}
|
|
396
|
+
for _label, attribute_filter in _filters_of_type(dashboard, "attribute_filter"):
|
|
397
|
+
actual.setdefault(attribute_filter.get("using"), []).append(attribute_filter)
|
|
398
|
+
|
|
211
399
|
failures: list[str] = []
|
|
212
|
-
for
|
|
213
|
-
|
|
214
|
-
|
|
215
|
-
if
|
|
216
|
-
|
|
217
|
-
failures.append(
|
|
218
|
-
f"tab {tab_id!r}: expected all time, got from={date_filter.get('from')!r} "
|
|
219
|
-
f"to={date_filter.get('to')!r}"
|
|
220
|
-
)
|
|
400
|
+
for exp in expected:
|
|
401
|
+
using = exp.get("using")
|
|
402
|
+
matches = actual.get(using) or []
|
|
403
|
+
if not matches:
|
|
404
|
+
failures.append(f"the dashboard carries no attribute filter on {using!r}")
|
|
221
405
|
continue
|
|
222
|
-
|
|
223
|
-
|
|
224
|
-
|
|
225
|
-
|
|
226
|
-
)
|
|
406
|
+
wanted = _expected_selection(exp)
|
|
407
|
+
if not any(_selection_of(f) == wanted for f in matches):
|
|
408
|
+
found = ", ".join(_describe_selection(_selection_of(f)) for f in matches)
|
|
409
|
+
failures.append(f"filter on {using!r} selects {found}, expected {_describe_selection(wanted)}")
|
|
227
410
|
return failures
|
|
228
411
|
|
|
229
412
|
|
|
413
|
+
def _expected_selection(expected_filter: dict) -> tuple[str, list] | None:
|
|
414
|
+
"""Read a fixture's filter entry into the same shape ``_selection_of`` produces.
|
|
415
|
+
|
|
416
|
+
Raises:
|
|
417
|
+
ValueError: the entry states no selection the evaluator knows how to check.
|
|
418
|
+
Deliberately loud -- a typo here would otherwise assert nothing and read green.
|
|
419
|
+
"""
|
|
420
|
+
for kind in ("include", "exclude"):
|
|
421
|
+
if kind in expected_filter:
|
|
422
|
+
return kind, list(expected_filter[kind] or [])
|
|
423
|
+
if expected_filter.get("selection") == "all":
|
|
424
|
+
return None
|
|
425
|
+
raise ValueError(
|
|
426
|
+
f"filter expectation for {expected_filter.get('using')!r} needs 'selection': 'all', "
|
|
427
|
+
f"'include' or 'exclude', got {expected_filter!r}"
|
|
428
|
+
)
|
|
429
|
+
|
|
430
|
+
|
|
230
431
|
def _date_range_text(date_range: dict | None) -> str:
|
|
231
432
|
"""Render a date range the way a user would say it, for the simulated reply.
|
|
232
433
|
|
|
@@ -290,6 +491,26 @@ def _min_new_visualizations(expected_output: dict) -> int:
|
|
|
290
491
|
return value
|
|
291
492
|
|
|
292
493
|
|
|
494
|
+
def _has_filters(expected_output: dict) -> bool:
|
|
495
|
+
"""Whether the expectation says anything about the attribute filters.
|
|
496
|
+
|
|
497
|
+
Absent means the case does not look at them, exactly as for ``date_range``. Only the
|
|
498
|
+
cases that are about preserving filters carry the key, so the others stay unaffected.
|
|
499
|
+
"""
|
|
500
|
+
return "filters" in expected_output
|
|
501
|
+
|
|
502
|
+
|
|
503
|
+
def _has_date_range(expected_output: dict) -> bool:
|
|
504
|
+
"""Whether the expectation says anything about the date filter at all.
|
|
505
|
+
|
|
506
|
+
Absence and ``null`` are different answers and must not be collapsed: ``null`` means the
|
|
507
|
+
dashboard has to be on all time, while a missing key means the case does not look at the
|
|
508
|
+
filter. Edit cases omit it because a dashboard the user has not opened carries no filter
|
|
509
|
+
context, so ``patch_dashboard`` refuses filter changes outright.
|
|
510
|
+
"""
|
|
511
|
+
return "date_range" in expected_output
|
|
512
|
+
|
|
513
|
+
|
|
293
514
|
def _validate_expectation(expected_output: dict) -> None:
|
|
294
515
|
"""Reject a fixture the run could not score meaningfully, before the first API call.
|
|
295
516
|
|
|
@@ -302,19 +523,67 @@ def _validate_expectation(expected_output: dict) -> None:
|
|
|
302
523
|
"""
|
|
303
524
|
if not expected_output.get("visualizations"):
|
|
304
525
|
raise ValueError("expected_output lists no visualizations; every chart check would pass vacuously")
|
|
305
|
-
|
|
526
|
+
if _has_date_range(expected_output):
|
|
527
|
+
date_range = expected_output.get("date_range")
|
|
528
|
+
# Shape first, and on both kinds of case. Only a creation case goes on to check that
|
|
529
|
+
# the reply can phrase the period -- an edit records whatever range the saved dashboard
|
|
530
|
+
# carries, which `_check_date_range` compares by number rather than reading aloud, so
|
|
531
|
+
# demanding a phrasable one there would cap what the preserved-filter cases can cover.
|
|
532
|
+
# Without the shape check a string slips through to scoring and dies there instead,
|
|
533
|
+
# after the run has already been spent.
|
|
534
|
+
if date_range is not None and not isinstance(date_range, dict):
|
|
535
|
+
raise ValueError(f"date_range must be an object or null, got {date_range!r}")
|
|
536
|
+
if not _is_edit(expected_output):
|
|
537
|
+
_date_range_text(date_range)
|
|
538
|
+
|
|
539
|
+
filters = expected_output.get("filters")
|
|
540
|
+
if filters is not None and not isinstance(filters, list):
|
|
541
|
+
raise ValueError(f"filters must be a list of filter expectations, got {filters!r}")
|
|
542
|
+
for entry in filters or []:
|
|
543
|
+
if not isinstance(entry, dict):
|
|
544
|
+
raise ValueError(f"a filter expectation must be an object, got {entry!r}")
|
|
545
|
+
if not entry.get("using"):
|
|
546
|
+
raise ValueError(f"a filter expectation needs the label it filters on, got {entry!r}")
|
|
547
|
+
_expected_selection(entry)
|
|
548
|
+
saved_id = expected_output.get("saved_dashboard_id")
|
|
549
|
+
if _is_edit(expected_output):
|
|
550
|
+
if not isinstance(saved_id, str) or not saved_id:
|
|
551
|
+
raise ValueError(f"an edit expectation needs the edited dashboard's saved_dashboard_id, got {saved_id!r}")
|
|
552
|
+
elif saved_id is not None:
|
|
553
|
+
raise ValueError(f"a creation expectation must not name a saved_dashboard_id, got {saved_id!r}")
|
|
306
554
|
_min_new_visualizations(expected_output)
|
|
307
555
|
|
|
308
556
|
|
|
557
|
+
@dataclass(frozen=True)
|
|
558
|
+
class _Applies:
|
|
559
|
+
"""Which of the conditional checks the case applies.
|
|
560
|
+
|
|
561
|
+
They are published only when they do. A check that could not fail is not evidence, and
|
|
562
|
+
publishing it as passed lifts `quality_score` -- the fraction of true booleans in the
|
|
563
|
+
detail -- above what the run earned, which would make a creation case read better than the
|
|
564
|
+
same case scored before editing existed.
|
|
565
|
+
"""
|
|
566
|
+
|
|
567
|
+
patch: bool
|
|
568
|
+
date: bool
|
|
569
|
+
filters: bool
|
|
570
|
+
|
|
571
|
+
|
|
309
572
|
@dataclass
|
|
310
573
|
class DashboardEvaluation:
|
|
311
574
|
"""Per-run outcome of the dashboard-skill checks.
|
|
312
575
|
|
|
313
|
-
``strict_checks`` is what the run is scored on
|
|
314
|
-
|
|
315
|
-
|
|
316
|
-
|
|
317
|
-
|
|
576
|
+
``strict_checks`` is what the run is scored on. ``skill_activated`` is in it because the
|
|
577
|
+
expectation's type names one skill and only one: a creation fixture answered by the editor,
|
|
578
|
+
or an edit fixture answered by the builder, is a routing failure even if the dashboard that
|
|
579
|
+
came back looks plausible.
|
|
580
|
+
|
|
581
|
+
``references_carried`` stays out of the gate: gen-ai rejects a response naming an
|
|
582
|
+
unresolvable chart before it can succeed, so a widget missing from the references means
|
|
583
|
+
reference building degraded on the way out -- a platform fault, not the model's.
|
|
584
|
+
``titles_matched`` is also out for creation, where the title is the model's own wording;
|
|
585
|
+
on an edit the expectation states the title the change must leave behind, so there it is
|
|
586
|
+
scored through ``charts_matched`` instead.
|
|
318
587
|
"""
|
|
319
588
|
|
|
320
589
|
drafted: bool
|
|
@@ -322,7 +591,11 @@ class DashboardEvaluation:
|
|
|
322
591
|
charts_matched: bool
|
|
323
592
|
date_range_correct: bool
|
|
324
593
|
new_visualizations_met: bool
|
|
594
|
+
applies: _Applies
|
|
595
|
+
filters_correct: bool = True
|
|
325
596
|
skill_activated: bool = False
|
|
597
|
+
saved_dashboard_id_correct: bool = True
|
|
598
|
+
patch_applies: bool = True
|
|
326
599
|
references_carried: bool = True
|
|
327
600
|
titles_matched: bool = True
|
|
328
601
|
failures: list[str] = field(default_factory=list)
|
|
@@ -334,26 +607,42 @@ class DashboardEvaluation:
|
|
|
334
607
|
|
|
335
608
|
@property
|
|
336
609
|
def strict_checks(self) -> dict[str, bool]:
|
|
337
|
-
|
|
610
|
+
checks = {
|
|
611
|
+
# Kept as-is through the editing work: every Langfuse view, saved filter and
|
|
612
|
+
# combo-report field list already refers to it, and a rename would break them for
|
|
613
|
+
# a word. It means "the tool that had to succeed did", patch or draft.
|
|
338
614
|
"dashboard_drafted": self.drafted,
|
|
339
615
|
"dashboard_part_present": self.part_present,
|
|
616
|
+
"dashboard_skill_activated": self.skill_activated,
|
|
617
|
+
"saved_dashboard_id_correct": self.saved_dashboard_id_correct,
|
|
340
618
|
"charts_matched": self.charts_matched,
|
|
341
|
-
"date_range_correct": self.date_range_correct,
|
|
342
619
|
"new_visualizations_met": self.new_visualizations_met,
|
|
343
620
|
}
|
|
621
|
+
if self.applies.patch:
|
|
622
|
+
checks["patch_applies"] = self.patch_applies
|
|
623
|
+
if self.applies.date:
|
|
624
|
+
checks["date_range_correct"] = self.date_range_correct
|
|
625
|
+
if self.applies.filters:
|
|
626
|
+
# Prefixed: alert_skill already publishes a `filters_correct` score, and the combo
|
|
627
|
+
# report resolves a trace's skill by which score names it carries.
|
|
628
|
+
checks["dashboard_filters_correct"] = self.filters_correct
|
|
629
|
+
return checks
|
|
344
630
|
|
|
345
631
|
@property
|
|
346
632
|
def diagnostics(self) -> dict[str, bool]:
|
|
347
633
|
"""Observed but not scored — see the class docstring.
|
|
348
634
|
|
|
349
|
-
``titles_matched``
|
|
350
|
-
|
|
351
|
-
|
|
352
|
-
|
|
353
|
-
|
|
635
|
+
``titles_matched`` reports the titles on both kinds of case, but only creation leaves
|
|
636
|
+
it out of the gate -- an edit scores the same mismatches through ``charts_matched``,
|
|
637
|
+
because there the expectation states the title the change must leave behind. Keeping
|
|
638
|
+
the score in both cases is what makes the creation decision reviewable later.
|
|
639
|
+
|
|
640
|
+
Both chart-level observations are omitted whenever no document was scored -- no part
|
|
641
|
+
read, or a patch that would not apply. Reporting them as true would claim references
|
|
642
|
+
and titles were fine on a run that never produced anything to look at.
|
|
354
643
|
"""
|
|
355
|
-
observed = {
|
|
356
|
-
if self.part_present:
|
|
644
|
+
observed: dict[str, bool] = {}
|
|
645
|
+
if self.part_present and self.patch_applies:
|
|
357
646
|
observed["references_carried"] = self.references_carried
|
|
358
647
|
observed["titles_matched"] = self.titles_matched
|
|
359
648
|
return observed
|
|
@@ -361,12 +650,13 @@ class DashboardEvaluation:
|
|
|
361
650
|
|
|
362
651
|
@dataclass
|
|
363
652
|
class DashboardRunResult:
|
|
364
|
-
"""Outcome of one K-run conversation
|
|
653
|
+
"""Outcome of one K-run conversation, creating or editing."""
|
|
365
654
|
|
|
366
655
|
conversation_id: str
|
|
367
656
|
evaluation: DashboardEvaluation
|
|
368
|
-
|
|
657
|
+
tool_result: dict | None = None
|
|
369
658
|
dashboard_part: dict | None = None
|
|
659
|
+
patch_part: dict | None = None
|
|
370
660
|
total_turns: int = 0
|
|
371
661
|
total_steps: int = 0
|
|
372
662
|
reasoning_steps: list[str] = field(default_factory=list)
|
|
@@ -378,7 +668,7 @@ class DashboardRunResult:
|
|
|
378
668
|
|
|
379
669
|
@dataclass
|
|
380
670
|
class AgenticDashboardSummary:
|
|
381
|
-
"""Aggregated outcome of K runs
|
|
671
|
+
"""Aggregated outcome of K runs, creating or editing."""
|
|
382
672
|
|
|
383
673
|
run_results: list[DashboardRunResult]
|
|
384
674
|
pass_at_k: bool
|
|
@@ -386,30 +676,43 @@ class AgenticDashboardSummary:
|
|
|
386
676
|
best: DashboardRunResult
|
|
387
677
|
|
|
388
678
|
|
|
389
|
-
def
|
|
390
|
-
|
|
679
|
+
def evaluate_dashboard_response(
|
|
680
|
+
tool_result: dict | None,
|
|
391
681
|
dashboard_part: dict | None,
|
|
392
682
|
expected_output: dict,
|
|
393
683
|
skill_activated: bool,
|
|
684
|
+
patch_part: dict | None = None,
|
|
394
685
|
) -> DashboardEvaluation:
|
|
395
|
-
"""Score one
|
|
686
|
+
"""Score one dashboard response against its expectation.
|
|
687
|
+
|
|
688
|
+
Creation reads the drafted dashboard straight off the ``dashboard`` part. Editing reads
|
|
689
|
+
two parts -- the saved dashboard as it stood, and the patch against it -- and scores the
|
|
690
|
+
document that results from applying one to the other, never the base.
|
|
396
691
|
|
|
397
|
-
Pure
|
|
692
|
+
Pure: no network, no conversation state, so the whole assertion surface is unit-testable
|
|
398
693
|
without an agent.
|
|
399
694
|
"""
|
|
400
|
-
|
|
695
|
+
is_edit = _is_edit(expected_output)
|
|
696
|
+
applies = _Applies(patch=is_edit, date=_has_date_range(expected_output), filters=_has_filters(expected_output))
|
|
697
|
+
tool = _producing_tool(expected_output)
|
|
698
|
+
|
|
699
|
+
if tool_result is None:
|
|
401
700
|
return DashboardEvaluation(
|
|
402
701
|
drafted=False,
|
|
403
702
|
part_present=False,
|
|
404
703
|
charts_matched=False,
|
|
405
704
|
date_range_correct=False,
|
|
705
|
+
filters_correct=False,
|
|
406
706
|
new_visualizations_met=False,
|
|
707
|
+
applies=applies,
|
|
407
708
|
skill_activated=skill_activated,
|
|
408
|
-
|
|
709
|
+
saved_dashboard_id_correct=False,
|
|
710
|
+
patch_applies=False,
|
|
711
|
+
failures=[f"the agent never produced a successful {tool} call"],
|
|
409
712
|
)
|
|
410
713
|
|
|
411
714
|
min_new = _min_new_visualizations(expected_output)
|
|
412
|
-
actual_new =
|
|
715
|
+
actual_new = tool_result.get("new_visualization_count")
|
|
413
716
|
if isinstance(actual_new, int):
|
|
414
717
|
new_met = actual_new >= min_new
|
|
415
718
|
# Naming the shortfall, never "expected at least 0" -- see the envelope branch below.
|
|
@@ -419,42 +722,116 @@ def evaluate_dashboard_draft(
|
|
|
419
722
|
# "expected at least 0 authored chart(s), tool reported None" reads as a broken test
|
|
420
723
|
# and would send whoever is on nightly duty looking at the model instead.
|
|
421
724
|
new_met = False
|
|
422
|
-
new_failures = [f"the
|
|
725
|
+
new_failures = [f"the {tool} result carries no new_visualization_count (got {actual_new!r})"]
|
|
423
726
|
|
|
727
|
+
missing_parts: list[str] = []
|
|
424
728
|
if dashboard_part is None:
|
|
729
|
+
missing_parts.append("dashboard")
|
|
730
|
+
if is_edit and patch_part is None:
|
|
731
|
+
missing_parts.append(_PATCH_TYPE)
|
|
732
|
+
if missing_parts:
|
|
425
733
|
return DashboardEvaluation(
|
|
426
734
|
drafted=True,
|
|
427
735
|
part_present=False,
|
|
428
736
|
charts_matched=False,
|
|
429
737
|
date_range_correct=False,
|
|
738
|
+
filters_correct=False,
|
|
430
739
|
new_visualizations_met=new_met,
|
|
740
|
+
applies=applies,
|
|
431
741
|
skill_activated=skill_activated,
|
|
742
|
+
saved_dashboard_id_correct=False,
|
|
743
|
+
patch_applies=False,
|
|
432
744
|
failures=[
|
|
433
|
-
f"the response carries no {
|
|
745
|
+
f"the response carries no {' and no '.join(repr(p) for p in missing_parts)} part",
|
|
434
746
|
*new_failures,
|
|
435
747
|
],
|
|
436
748
|
)
|
|
437
749
|
|
|
438
|
-
|
|
439
|
-
|
|
440
|
-
|
|
441
|
-
|
|
442
|
-
|
|
750
|
+
assert dashboard_part is not None # noqa: S101 narrowed by missing_parts above
|
|
751
|
+
base = dashboard_part.get("dashboard") or {}
|
|
752
|
+
base_references = dashboard_part.get("references") or {}
|
|
753
|
+
expected_saved_id = expected_output.get("saved_dashboard_id")
|
|
754
|
+
actual_saved_id = dashboard_part.get("saved_dashboard_id")
|
|
755
|
+
|
|
756
|
+
saved_failures: list[str] = []
|
|
757
|
+
patch_failures: list[str] = []
|
|
758
|
+
|
|
759
|
+
if is_edit:
|
|
760
|
+
patch = (patch_part or {}).get("patch") or {}
|
|
761
|
+
patch_references = patch.get("references") or {}
|
|
762
|
+
# The patch names the dashboard it addresses; the part names the one it shipped with.
|
|
763
|
+
# Both have to be the dashboard the fixture asked to edit, or the change landed
|
|
764
|
+
# somewhere nobody looked.
|
|
765
|
+
for label, value in (("the response", actual_saved_id), ("the patch", patch.get("dashboard_id"))):
|
|
766
|
+
if value != expected_saved_id:
|
|
767
|
+
saved_failures.append(f"{label} addresses dashboard {value!r}, expected {expected_saved_id!r}")
|
|
768
|
+
document, patch_error = _apply_patch(base, patch.get("operations") or [])
|
|
769
|
+
if patch_error is not None:
|
|
770
|
+
patch_failures.append(patch_error)
|
|
771
|
+
document = None
|
|
772
|
+
new_ids = {v.get("id") for v in patch_references.get("new_visualizations") or [] if v.get("id")}
|
|
773
|
+
known_ids = (
|
|
774
|
+
{v.get("id") for v in base_references.get("visualizations") or [] if v.get("id")}
|
|
775
|
+
| {v.get("id") for v in patch_references.get("visualizations") or [] if v.get("id")}
|
|
776
|
+
| new_ids
|
|
777
|
+
)
|
|
778
|
+
else:
|
|
779
|
+
if actual_saved_id is not None:
|
|
780
|
+
saved_failures.append(f"a new draft must not be saved yet, but it reports id {actual_saved_id!r}")
|
|
781
|
+
document = base
|
|
782
|
+
new_ids = {v.get("id") for v in base_references.get("new_visualizations") or [] if v.get("id")}
|
|
783
|
+
known_ids = {v.get("id") for v in base_references.get("visualizations") or [] if v.get("id")} | new_ids
|
|
443
784
|
|
|
444
|
-
|
|
785
|
+
if document is None:
|
|
786
|
+
return DashboardEvaluation(
|
|
787
|
+
drafted=True,
|
|
788
|
+
part_present=True,
|
|
789
|
+
charts_matched=False,
|
|
790
|
+
date_range_correct=False,
|
|
791
|
+
filters_correct=False,
|
|
792
|
+
new_visualizations_met=new_met,
|
|
793
|
+
applies=applies,
|
|
794
|
+
skill_activated=skill_activated,
|
|
795
|
+
saved_dashboard_id_correct=not saved_failures,
|
|
796
|
+
patch_applies=False,
|
|
797
|
+
failures=[*patch_failures, *saved_failures, *new_failures],
|
|
798
|
+
)
|
|
799
|
+
|
|
800
|
+
widgets = _widgets_of(document)
|
|
801
|
+
chart_failures, title_mismatches = _check_visualizations(
|
|
802
|
+
widgets, new_ids, expected_output.get("visualizations") or []
|
|
803
|
+
)
|
|
804
|
+
# On an edit the title is part of what was asked for, so it fails the case; on creation it
|
|
805
|
+
# is only recorded. Either way `titles_matched` reads the mismatches themselves, never the
|
|
806
|
+
# list they were routed into -- a score that says "titles fine" beside a failed rename is
|
|
807
|
+
# worse than no score at all.
|
|
808
|
+
if is_edit:
|
|
809
|
+
chart_failures = [*chart_failures, *title_mismatches]
|
|
810
|
+
title_notes: list[str] = []
|
|
811
|
+
else:
|
|
812
|
+
title_notes = title_mismatches
|
|
445
813
|
reference_notes = _check_references(widgets, known_ids)
|
|
446
|
-
date_failures =
|
|
814
|
+
date_failures = (
|
|
815
|
+
_check_date_range(document, expected_output.get("date_range")) if _has_date_range(expected_output) else []
|
|
816
|
+
)
|
|
817
|
+
filter_failures = (
|
|
818
|
+
_check_filters(document, expected_output.get("filters") or []) if _has_filters(expected_output) else []
|
|
819
|
+
)
|
|
447
820
|
|
|
448
821
|
return DashboardEvaluation(
|
|
449
822
|
drafted=True,
|
|
450
823
|
part_present=True,
|
|
451
824
|
charts_matched=not chart_failures,
|
|
452
825
|
date_range_correct=not date_failures,
|
|
826
|
+
filters_correct=not filter_failures,
|
|
453
827
|
new_visualizations_met=new_met,
|
|
828
|
+
applies=applies,
|
|
454
829
|
skill_activated=skill_activated,
|
|
830
|
+
saved_dashboard_id_correct=not saved_failures,
|
|
831
|
+
patch_applies=True,
|
|
455
832
|
references_carried=not reference_notes,
|
|
456
|
-
titles_matched=not
|
|
457
|
-
failures=[*chart_failures, *date_failures, *new_failures],
|
|
833
|
+
titles_matched=not title_mismatches,
|
|
834
|
+
failures=[*chart_failures, *saved_failures, *date_failures, *filter_failures, *new_failures],
|
|
458
835
|
notes=[*title_notes, *reference_notes],
|
|
459
836
|
)
|
|
460
837
|
|
|
@@ -466,14 +843,16 @@ def _execute_single_dashboard_run(
|
|
|
466
843
|
expected_output: dict,
|
|
467
844
|
max_iterations: int,
|
|
468
845
|
) -> DashboardRunResult:
|
|
469
|
-
"""Drive one conversation until the agent
|
|
846
|
+
"""Drive one conversation until the agent produces a dashboard, then evaluate it.
|
|
470
847
|
|
|
471
848
|
Nothing is cleaned up on the way out by design: gen-ai keeps the draft and any authored
|
|
472
849
|
chart in conversation state and persists neither until a user saves from the UI.
|
|
473
850
|
"""
|
|
474
|
-
|
|
475
|
-
|
|
851
|
+
tool = _producing_tool(expected_output)
|
|
852
|
+
is_edit = _is_edit(expected_output)
|
|
853
|
+
tool_result: dict | None = None
|
|
476
854
|
dashboard_part: dict | None = None
|
|
855
|
+
patch_part: dict | None = None
|
|
477
856
|
turns = 0
|
|
478
857
|
steps = 0
|
|
479
858
|
current_question = question
|
|
@@ -504,14 +883,26 @@ def _execute_single_dashboard_run(
|
|
|
504
883
|
all_reasoning_step_events.extend(chat_result.reasoning_step_events or [])
|
|
505
884
|
steps += chat_result.reasoning_step_count
|
|
506
885
|
|
|
507
|
-
candidate =
|
|
886
|
+
candidate = _extract_tool_result(chat_result.tool_call_events or [], tool)
|
|
508
887
|
if candidate is not None:
|
|
509
888
|
log_timer(
|
|
510
889
|
f"[timer] dashboard_skill {conversation_id} GoodData turn {turns} complete after "
|
|
511
|
-
f"{agent_elapsed:.2f}s;
|
|
890
|
+
f"{agent_elapsed:.2f}s; {tool} result received"
|
|
512
891
|
)
|
|
513
|
-
|
|
514
|
-
|
|
892
|
+
tool_result = candidate
|
|
893
|
+
# An edit relays two parts: the dashboard as it stood, and the patch against it.
|
|
894
|
+
# Both are read from the turn that produced the patch -- the base is what the patch
|
|
895
|
+
# was written for, so pairing it with any other turn's would score a document the
|
|
896
|
+
# agent never proposed.
|
|
897
|
+
# A turn may carry several patches. They are alternatives rebased on the same base
|
|
898
|
+
# rather than steps to compose, and the client resolves the last one as the proposal
|
|
899
|
+
# that holds (gdc-ui's applyDashboardPatch; gdc-nas
|
|
900
|
+
# dashboard_edit_assertion.verify_last_proposal_holds does the same). Taking the last
|
|
901
|
+
# of each part type is therefore what the user would end up looking at, and the
|
|
902
|
+
# earlier patches are alternatives that lost, not changes that went missing.
|
|
903
|
+
dashboard_part = _extract_dashboard_part(chat_result, "dashboard")
|
|
904
|
+
if is_edit:
|
|
905
|
+
patch_part = _extract_dashboard_part(chat_result, _PATCH_TYPE)
|
|
515
906
|
break
|
|
516
907
|
|
|
517
908
|
response_text = (chat_result.text_response or "").strip() or render_answer_text(chat_result)
|
|
@@ -523,18 +914,26 @@ def _execute_single_dashboard_run(
|
|
|
523
914
|
f"[timer] dashboard_skill {conversation_id} GoodData turn {turns} complete after "
|
|
524
915
|
f"{agent_elapsed:.2f}s; answering with the expected charts and date range"
|
|
525
916
|
)
|
|
917
|
+
if is_edit:
|
|
918
|
+
# The creation reply names charts and a date range, which answers nothing a rename
|
|
919
|
+
# or a resize could have asked. Rather than send something the question did not
|
|
920
|
+
# ask for, stop: the failure is the signal that an edit case needs a reply of its
|
|
921
|
+
# own, and inventing one here would hide which cases actually need it.
|
|
922
|
+
break
|
|
526
923
|
current_question = build_simulated_reply(expected_output)
|
|
527
924
|
|
|
528
925
|
return DashboardRunResult(
|
|
529
926
|
conversation_id=conversation_id,
|
|
530
|
-
evaluation=
|
|
531
|
-
|
|
927
|
+
evaluation=evaluate_dashboard_response(
|
|
928
|
+
tool_result,
|
|
532
929
|
dashboard_part,
|
|
533
930
|
expected_output,
|
|
534
|
-
|
|
931
|
+
_skill_activated(all_tool_call_events, _required_skill(expected_output)),
|
|
932
|
+
patch_part=patch_part,
|
|
535
933
|
),
|
|
536
|
-
|
|
934
|
+
tool_result=tool_result,
|
|
537
935
|
dashboard_part=dashboard_part,
|
|
936
|
+
patch_part=patch_part,
|
|
538
937
|
total_turns=turns,
|
|
539
938
|
total_steps=steps,
|
|
540
939
|
reasoning_steps=reasoning_steps,
|
|
@@ -711,11 +1110,22 @@ def evaluate_agentic_dashboard_skill(
|
|
|
711
1110
|
|
|
712
1111
|
if not gate_passed(gate, pass_at_k=summary.pass_at_k, pass_power_k=summary.pass_power_k):
|
|
713
1112
|
gate_note = gate_failure_note(gate, runs_passed, runs_effective)
|
|
714
|
-
|
|
715
|
-
|
|
716
|
-
|
|
717
|
-
|
|
718
|
-
|
|
1113
|
+
# Two failures wear the same missing skill, and the advice differs. If the other
|
|
1114
|
+
# dashboard skill was activated, the flag that registers both is plainly on and the
|
|
1115
|
+
# model simply routed the wrong way. If neither was, the flag is off or the agent has
|
|
1116
|
+
# its skills pinned. Sending a reader after the flag in the first case walks them past
|
|
1117
|
+
# the real failure, which is why the run's own tool calls decide which line they get.
|
|
1118
|
+
required_skill = _required_skill(expected_output)
|
|
1119
|
+
other_skill = _BUILDER_SKILL if required_skill == _EDITOR_SKILL else _EDITOR_SKILL
|
|
1120
|
+
if best.evaluation.skill_activated:
|
|
1121
|
+
skill_note = ""
|
|
1122
|
+
elif _skill_activated(best.tool_call_events, other_skill):
|
|
1123
|
+
skill_note = f" It activated {other_skill} instead of {required_skill}: a routing failure, not the flag."
|
|
1124
|
+
else:
|
|
1125
|
+
skill_note = (
|
|
1126
|
+
f" No set_skills call activated {required_skill} or {other_skill};"
|
|
1127
|
+
" one feature flag registers both, so check that before the model."
|
|
1128
|
+
)
|
|
719
1129
|
notes = "; ".join(best.evaluation.notes)
|
|
720
1130
|
exc = DashboardSkillAssertionError(
|
|
721
1131
|
f"Dashboard skill assertion failed. {gate_note}{skill_note} "
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: gooddata-eval
|
|
3
|
-
Version: 1.75.1.
|
|
3
|
+
Version: 1.75.1.dev3
|
|
4
4
|
Summary: Evaluate the GoodData AI agent against your own questions and models.
|
|
5
5
|
Project-URL: Source, https://github.com/gooddata/gooddata-python-sdk
|
|
6
6
|
Author-email: GoodData <support@gooddata.com>
|
|
@@ -17,8 +17,9 @@ Classifier: Topic :: Scientific/Engineering
|
|
|
17
17
|
Classifier: Topic :: Software Development
|
|
18
18
|
Classifier: Typing :: Typed
|
|
19
19
|
Requires-Python: >=3.10
|
|
20
|
-
Requires-Dist: gooddata-sdk~=1.75.1.
|
|
20
|
+
Requires-Dist: gooddata-sdk~=1.75.1.dev3
|
|
21
21
|
Requires-Dist: httpx<1.0,>=0.27
|
|
22
|
+
Requires-Dist: jsonpatch<2.0,>=1.33
|
|
22
23
|
Requires-Dist: orjson<4.0.0,>=3.9.15
|
|
23
24
|
Requires-Dist: pydantic<3.0,>=2.6
|
|
24
25
|
Requires-Dist: rich<15.0,>=13.0
|
|
@@ -20,7 +20,7 @@ gooddata_eval/core/agentic/_langfuse.py,sha256=7V1PFQbWpRnFz0a3wklcI-ROYjGC1wGQv
|
|
|
20
20
|
gooddata_eval/core/agentic/_trace_linker.py,sha256=iXMTxIXUsF5WNSen7v_K8JJ2j1_37lh-2VAGqQxxmkI,13538
|
|
21
21
|
gooddata_eval/core/agentic/alert_skill.py,sha256=47V6LVQemAL0OmPeY9Yr1-lszkPu5bFkCgYgHSJJI_8,41712
|
|
22
22
|
gooddata_eval/core/agentic/conversation.py,sha256=Y-SsGOth-TzXFlOKsElCigT05ZH6wNqIOOz1IUkzUo4,30044
|
|
23
|
-
gooddata_eval/core/agentic/dashboard_skill.py,sha256=
|
|
23
|
+
gooddata_eval/core/agentic/dashboard_skill.py,sha256=4lu7v6VeIR_h9FUFBrpD57upA0Rd2Kox1ZP_X-7jQaQ,51854
|
|
24
24
|
gooddata_eval/core/agentic/general_question.py,sha256=Jp3FaS4Tvmfo0dx9iceiZ3Fr1QzJKDl7fVZZDetS22I,14684
|
|
25
25
|
gooddata_eval/core/agentic/guardrail.py,sha256=rYpnsun9mLOSYucWfg1dbdrpbzQcz2V3eLwgII8dD0w,13145
|
|
26
26
|
gooddata_eval/core/agentic/kda_skill.py,sha256=P_pR0iC12b_hIwnLzBmH0IBEjsO_CxZrlfkmu10cVWE,22721
|
|
@@ -63,8 +63,8 @@ gooddata_eval/core/reporting/json_report.py,sha256=q3Jirc_Zhvzg5lkG_kM6JdnViNLe6
|
|
|
63
63
|
gooddata_eval/core/reporting/report_template.html,sha256=vLkl95OZsm3dMdijv5zda-15wYPE5UvTs4uKeWnjCiI,22332
|
|
64
64
|
gooddata_eval/core/summary/__init__.py,sha256=U54lANKm34yjqm7dZL9KkGouPbwAaQWC9wiAmb5B5g4,32
|
|
65
65
|
gooddata_eval/core/summary/http_client.py,sha256=dPArJ6zoCu1W9xmYXFA1WMDNsSOkhn3gaCpYgCUp3gk,2174
|
|
66
|
-
gooddata_eval-1.75.1.
|
|
67
|
-
gooddata_eval-1.75.1.
|
|
68
|
-
gooddata_eval-1.75.1.
|
|
69
|
-
gooddata_eval-1.75.1.
|
|
70
|
-
gooddata_eval-1.75.1.
|
|
66
|
+
gooddata_eval-1.75.1.dev3.dist-info/METADATA,sha256=hm6riuge5SGz7wMiNFSuWNZk39Qcx3ph7w8NDaMddGo,35069
|
|
67
|
+
gooddata_eval-1.75.1.dev3.dist-info/WHEEL,sha256=W3fkpkm7-wf9vBI5Z-7s0eWkeM-spu78I8Neb98DeEg,87
|
|
68
|
+
gooddata_eval-1.75.1.dev3.dist-info/entry_points.txt,sha256=28nFp5Viknx4haPYgzK9QHlwPFfC6rPF6RmjC2ht8EI,56
|
|
69
|
+
gooddata_eval-1.75.1.dev3.dist-info/licenses/LICENSE.txt,sha256=LVfVlC9maU3K9lgMwdZnClQQsOIbOekdcdVKr9qzJtk,256836
|
|
70
|
+
gooddata_eval-1.75.1.dev3.dist-info/RECORD,,
|
|
File without changes
|
{gooddata_eval-1.75.1.dev2.dist-info → gooddata_eval-1.75.1.dev3.dist-info}/entry_points.txt
RENAMED
|
File without changes
|
{gooddata_eval-1.75.1.dev2.dist-info → gooddata_eval-1.75.1.dev3.dist-info}/licenses/LICENSE.txt
RENAMED
|
File without changes
|