gooddata-eval 1.75.1.dev2__py3-none-any.whl → 1.75.1.dev3__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -7,6 +7,8 @@ import time
7
7
  from dataclasses import dataclass, field
8
8
  from typing import Any
9
9
 
10
+ import jsonpatch
11
+
10
12
  from gooddata_eval.core.agentic._gate import (
11
13
  DEFAULT_GATE,
12
14
  EvalGate,
@@ -41,11 +43,14 @@ from gooddata_eval.core.timing import PhaseTimings, log_timer, sum_timings
41
43
  _DEFAULT_K = 1
42
44
  # Matches kda_skill and visualization. The reply this loop sends is built from the expectation
43
45
  # and is byte-identical every turn, so the extra rounds are not there to say anything new --
44
- # they are slack for a turn that produced nothing. A turn can come back empty or answer without
45
- # drafting for reasons that are not the model's, and with a tight cap that one wasted turn
46
- # spends the only reply the case had; the run then fails as "no draft" when nothing was wrong
47
- # with the agent. Fewer than metric_skill's seven because those rounds do carry new content:
48
- # its reply is generated per turn by an LLM.
46
+ # they are slack for a turn that answered without drafting, which can happen for reasons that
47
+ # are not the model's. A turn that comes back with nothing at all does not consume the slack:
48
+ # the loop ends there rather than replying into silence. Fewer than metric_skill's seven
49
+ # because those rounds do carry new content, its reply being generated per turn by an LLM.
50
+ #
51
+ # Only creation ever reaches the later rounds. An edit stops after the first turn, since the
52
+ # creation reply names charts and a date range and answers nothing a rename or a resize could
53
+ # have asked.
49
54
  #
50
55
  # The cost of the slack is that four turns at ChatClient's 300s read timeout exceed the 720s
51
56
  # per-test timeout gdc-nas derives for a k=1 dataset, so a run that stalls on every turn is cut
@@ -55,8 +60,15 @@ _DEFAULT_K = 1
55
60
  _DEFAULT_MAX_ITERATIONS = 4
56
61
 
57
62
  _DRAFT_TOOL = "draft_dashboard"
63
+ _PATCH_TOOL = "patch_dashboard"
58
64
  _SET_SKILLS_TOOL = "set_skills"
59
65
  _BUILDER_SKILL = "dashboard_builder"
66
+ _EDITOR_SKILL = "dashboard_editor"
67
+ # The expectation's `type` is the only switch between creating and editing: it selects the
68
+ # response part to read, the tool that must have succeeded, and the skill that must have been
69
+ # activated. There is no "either one" case -- an edit fixture that routes to the builder is a
70
+ # failure, not an alternative route to the same answer.
71
+ _PATCH_TYPE = "dashboardPatch"
60
72
 
61
73
 
62
74
  def _norm(text: str) -> str:
@@ -64,14 +76,14 @@ def _norm(text: str) -> str:
64
76
  return " ".join(text.split()).casefold()
65
77
 
66
78
 
67
- def _extract_draft_result(tool_call_events: list[ToolCallEvent]) -> dict | None:
68
- """Result payload of the ``draft_dashboard`` call that produced a draft.
79
+ def _extract_tool_result(tool_call_events: list[ToolCallEvent], tool_name: str) -> dict | None:
80
+ """Result payload of the ``tool_name`` call that produced the response.
69
81
 
70
82
  Takes the most recent *successful* call: when the agent retries after a rejected
71
83
  draft, the earlier failed attempt must not shadow the one that worked.
72
84
  """
73
85
  for tc in reversed(tool_call_events):
74
- if tc.function_name != _DRAFT_TOOL or not tc.result:
86
+ if tc.function_name != tool_name or not tc.result:
75
87
  continue
76
88
  result_data = tc.parsed_result()
77
89
  if not isinstance(result_data, dict):
@@ -83,14 +95,22 @@ def _extract_draft_result(tool_call_events: list[ToolCallEvent]) -> dict | None:
83
95
  return None
84
96
 
85
97
 
86
- def _builder_skill_activated(tool_call_events: list[ToolCallEvent]) -> bool:
87
- """Whether any ``set_skills`` result activated the dashboard-builder skill.
98
+ def _skill_activated(tool_call_events: list[ToolCallEvent], skill: str) -> bool:
99
+ """Whether the run's skill routing landed on ``skill``.
88
100
 
89
- Diagnostic only, deliberately outside ``strict_pass``: a mis-route already fails on the
90
- missing draft, and gating on this would turn any future route that skips ``set_skills``
91
- into a false failure. Reported so such a failure reads as a mis-route rather than as a
92
- model that simply did not draft.
101
+ Gated, because the expectation's type names one skill and only one: a creation fixture
102
+ answered by the editor, or an edit fixture answered by the builder, is a routing failure
103
+ even when the dashboard that comes back looks plausible. It also names the failure -- a
104
+ mis-route reads as a mis-route instead of as a model that simply produced nothing.
105
+
106
+ A run that never routed passes. Absence of a ``set_skills`` call is absence of evidence,
107
+ not evidence of a mis-route, and there are two ways to reach it that say nothing about the
108
+ model: gen-ai omits the call for an agent whose skills are pinned
109
+ (``include_set_skills=not resolved_pinned_skills``), and a run continuing an existing
110
+ conversation through ``initial_conversation_id`` can have routed in a turn this run never
111
+ saw. Failing those would fail the case for the harness's configuration.
93
112
  """
113
+ routed = False
94
114
  for tc in tool_call_events:
95
115
  if tc.function_name != _SET_SKILLS_TOOL or not tc.result:
96
116
  continue
@@ -99,9 +119,27 @@ def _builder_skill_activated(tool_call_events: list[ToolCallEvent]) -> bool:
99
119
  continue
100
120
  payload = result_data.get("data", result_data)
101
121
  skills = payload.get("skills_to_activate") if isinstance(payload, dict) else None
102
- if isinstance(skills, list) and _BUILDER_SKILL in skills:
122
+ if not isinstance(skills, list):
123
+ continue
124
+ routed = True
125
+ if skill in skills:
103
126
  return True
104
- return False
127
+ return not routed
128
+
129
+
130
+ def _is_edit(expected_output: dict) -> bool:
131
+ """Whether the expectation describes an edit rather than a fresh dashboard."""
132
+ return _expected_type(expected_output) == _PATCH_TYPE
133
+
134
+
135
+ def _required_skill(expected_output: dict) -> str:
136
+ """The skill the run must have activated, decided by the expectation's type."""
137
+ return _EDITOR_SKILL if _is_edit(expected_output) else _BUILDER_SKILL
138
+
139
+
140
+ def _producing_tool(expected_output: dict) -> str:
141
+ """The tool whose successful call the response must come from."""
142
+ return _PATCH_TOOL if _is_edit(expected_output) else _DRAFT_TOOL
105
143
 
106
144
 
107
145
  def _expected_type(expected_output: dict) -> str:
@@ -114,7 +152,7 @@ def _extract_dashboard_part(chat_result: ChatResult, part_type: str) -> dict | N
114
152
 
115
153
  That type is in the SSE client's known-part set but has no dedicated accumulator, so it
116
154
  arrives in ``unhandled_parts`` verbatim and is read back by type. Taken from the end for
117
- the same reason ``_extract_draft_result`` does: a turn that drafts and then refines must
155
+ the same reason ``_extract_tool_result`` does: a turn that drafts and then refines must
118
156
  be read as the state it left behind, not the one it passed through.
119
157
  """
120
158
  for part in reversed(chat_result.unhandled_parts):
@@ -123,64 +161,113 @@ def _extract_dashboard_part(chat_result: ChatResult, part_type: str) -> dict | N
123
161
  return None
124
162
 
125
163
 
164
+ def _sections_of(dashboard: dict) -> list[dict]:
165
+ """The dashboard's sections, whichever shape the document uses.
166
+
167
+ A freshly drafted dashboard is version 3 and nests its sections under ``tabs``. A saved
168
+ dashboard relayed for editing can be version 2, which carries ``sections`` at the root and
169
+ no ``tabs`` at all. ``version`` is not the discriminator -- the convertor computes it from
170
+ the declarative input and separately flattens a single untitled tab into root sections, so
171
+ a document can say ``version: "3"`` and still have no tabs. Read the presence of ``tabs``.
172
+ """
173
+ tabs = dashboard.get("tabs")
174
+ if tabs:
175
+ return [section for tab in tabs for section in tab.get("sections") or []]
176
+ return dashboard.get("sections") or []
177
+
178
+
126
179
  def _widgets_of(dashboard: dict) -> list[dict]:
127
- """Every widget of the draft, flattened across tabs and sections.
180
+ """Every widget of the dashboard, flattened across sections.
128
181
 
129
182
  Charts are matched dashboard-wide on purpose: the expectation names charts, not a layout,
130
- so which tab one lands on is the agent's call. The date filter is deliberately the
131
- opposite -- ``_check_date_range`` requires *every* tab to match. That is defensive rather
132
- than observed: ``draft_dashboard`` copies one filter onto every tab unconditionally, so a
133
- draft cannot currently disagree with itself across tabs, and the per-tab check is there to
134
- notice if that ever stops being true.
135
-
136
- Only ``tabs`` is read. That is sound for a draft, which ``draft_dashboard`` always builds
137
- tabbed, but an AAC v2 document carries its layout in root ``sections`` instead, so an
138
- editing dataset cannot reuse this as it stands.
183
+ so which section or tab one lands on is the agent's call. The date filter is deliberately
184
+ the opposite -- ``_check_date_range`` requires *every* tab to match. That one is defensive
185
+ rather than observed: ``draft_dashboard`` copies a single filter onto every tab
186
+ unconditionally, so a draft cannot currently disagree with itself across tabs, and the
187
+ per-tab check is there to notice if that stops being true.
139
188
  """
140
- return [
141
- widget
142
- for tab in dashboard.get("tabs") or []
143
- for section in tab.get("sections") or []
144
- for widget in section.get("widgets") or []
145
- ]
189
+ return [widget for section in _sections_of(dashboard) for widget in section.get("widgets") or []]
190
+
191
+
192
+ def _apply_patch(base: dict, operations: list[dict]) -> tuple[dict | None, str | None]:
193
+ """Apply RFC 6902 ``operations`` to a copy of ``base``; return ``(result, failure)``.
194
+
195
+ The agent emits ``test`` guards ahead of its real operations, pinning the widget it is
196
+ about to touch. Those are applied rather than skipped: a failing guard means the patch was
197
+ written against a document that is not the one it shipped, which is a failure of the edit
198
+ even though every later operation might apply cleanly. Scoring the base instead of the
199
+ result would pass a case whose change never happened.
200
+
201
+ A patch carrying no operations is that same failure in its purest form. ``jsonpatch``
202
+ applies an empty list happily and hands back the document untouched, so the run would go on
203
+ to score the saved dashboard and pass every case the saved dashboard already satisfied.
204
+ """
205
+ if not operations:
206
+ return None, "the patch carries no operations, so it proposes the dashboard it started from"
207
+ try:
208
+ # in_place=False is what makes the copy -- JsonPatch.apply deepcopies for us, so
209
+ # copying here as well would leave a second one behind for no reason.
210
+ return jsonpatch.JsonPatch(operations).apply(base, in_place=False), None
211
+ except Exception as exc: # jsonpatch raises several unrelated types for a bad patch
212
+ return None, f"the patch does not apply to the dashboard it shipped with: {exc}"
146
213
 
147
214
 
148
215
  def _check_visualizations(widgets: list[dict], new_ids: set[str], expected: list[dict]) -> tuple[list[str], list[str]]:
149
- """Match every expected chart against the draft; extra charts the agent added are allowed.
216
+ """Match every expected chart against the dashboard; extra widgets are allowed.
150
217
 
151
218
  Returns ``(failures, title_notes)``.
152
219
 
153
- An entry with an ``id`` is an existing chart and is matched on that id alone: the id is
154
- the chart's identity, while the widget title is ``title_override or fallback_title`` and
155
- the override is the model's own choice, so requiring it adds flake without signal. A
156
- title that differs is still reported, as a note rather than a failure.
220
+ An entry with an ``id`` is matched on that id. ``id: null`` marks a chart the agent had to
221
+ author, whose real id is generated at run time, so there the title *is* the match -- taken
222
+ among the authored charts only, since matching against every chart would let an existing
223
+ one of the same title satisfy it.
224
+
225
+ Title mismatches come back separately from the failures rather than mixed into them, so
226
+ the caller can gate on them or merely record them without the two decisions drifting: on
227
+ creation the title is ``title_override or fallback_title`` and the override is the model's
228
+ own choice, while on an edit the expectation states the title the change must leave behind
229
+ and a rename case that does not gate on it asserts nothing at all. Either way the
230
+ mismatches are returned, which is what keeps the reported score honest.
157
231
 
158
- ``id: null`` marks a chart the agent had to author, whose real id is generated at run
159
- time. It has no identity to match on, so the title *is* the match — taken among the
160
- authored charts only, without which an existing chart of the same title would satisfy it.
232
+ ``columns`` is checked whenever the expectation carries it. A newly placed widget defaults
233
+ to half width, so a resize case is only meaningful against the number.
161
234
  """
162
235
  failures: list[str] = []
163
- title_notes: list[str] = []
236
+ title_mismatches: list[str] = []
164
237
  for exp in expected:
165
238
  exp_id = exp.get("id")
166
239
  exp_title = str(exp.get("title") or "")
240
+ exp_columns = exp.get("columns")
241
+
167
242
  if exp_id is not None:
168
243
  matches = [w for w in widgets if w.get("visualization") == exp_id]
169
244
  if not matches:
170
245
  failures.append(f"missing existing chart id={exp_id!r} title={exp_title!r}")
171
- elif not any(_norm(str(w.get("title") or "")) == _norm(exp_title) for w in matches):
246
+ continue
247
+ titled = [w for w in matches if _norm(str(w.get("title") or "")) == _norm(exp_title)]
248
+ if not titled:
172
249
  # Every title the id appears under: the same chart can be placed twice, and
173
250
  # naming only the first would hide the rest.
174
251
  titles = ", ".join(repr(w.get("title")) for w in matches)
175
- title_notes.append(f"chart {exp_id!r} is titled {titles}, expected {exp_title!r}")
176
- elif not any(
177
- w.get("visualization") in new_ids and _norm(str(w.get("title") or "")) == _norm(exp_title) for w in widgets
178
- ):
179
- failures.append(
180
- f"missing authored chart title={exp_title!r}"
181
- + ("" if new_ids else " (the response listed no authored charts)")
182
- )
183
- return failures, title_notes
252
+ title_mismatches.append(f"chart {exp_id!r} is titled {titles}, expected {exp_title!r}")
253
+ candidates = titled or matches
254
+ else:
255
+ candidates = [
256
+ w
257
+ for w in widgets
258
+ if w.get("visualization") in new_ids and _norm(str(w.get("title") or "")) == _norm(exp_title)
259
+ ]
260
+ if not candidates:
261
+ failures.append(
262
+ f"missing authored chart title={exp_title!r}"
263
+ + ("" if new_ids else " (the response listed no authored charts)")
264
+ )
265
+ continue
266
+
267
+ if exp_columns is not None and not any(w.get("columns") == exp_columns for w in candidates):
268
+ widths = ", ".join(repr(w.get("columns")) for w in candidates)
269
+ failures.append(f"chart {exp_title!r} is {widths} column(s) wide, expected {exp_columns}")
270
+ return failures, title_mismatches
184
271
 
185
272
 
186
273
  def _check_references(widgets: list[dict], known_ids: set[str]) -> list[str]:
@@ -199,34 +286,148 @@ def _check_references(widgets: list[dict], known_ids: set[str]) -> list[str]:
199
286
  return notes
200
287
 
201
288
 
289
+ def _filter_entries(dashboard: dict) -> list[tuple[str, dict]]:
290
+ """``(label, filter)`` for every filter the dashboard carries, whichever shape it uses.
291
+
292
+ A drafted dashboard hangs one filter map off each tab, keyed by role (``date``). A saved
293
+ dashboard relayed for editing carries a single map at the root, keyed by the filter's own
294
+ local identifier, so the role has to be read from each entry's ``type`` instead. The label
295
+ is only there to say *where* a failure was found.
296
+ """
297
+ tabs = dashboard.get("tabs")
298
+ if tabs:
299
+ return [
300
+ (f"tab {tab.get('id')!r}", value)
301
+ for tab in tabs
302
+ for value in (tab.get("filters") or {}).values()
303
+ if isinstance(value, dict)
304
+ ]
305
+ return [
306
+ (f"filter {key!r}", value) for key, value in (dashboard.get("filters") or {}).items() if isinstance(value, dict)
307
+ ]
308
+
309
+
310
+ def _filters_of_type(dashboard: dict, filter_type: str) -> list[tuple[str, dict]]:
311
+ return [(label, f) for label, f in _filter_entries(dashboard) if f.get("type") == filter_type]
312
+
313
+
314
+ def _date_filter_mismatch(label: str, date_filter: dict, expected: dict | None) -> list[str]:
315
+ """Why ``date_filter`` is not the expected range, or nothing.
316
+
317
+ ``expected is None`` means all time, which the document spells as a date filter with
318
+ neither bound — absent keys, not null ones.
319
+ """
320
+ if expected is None:
321
+ if "from" in date_filter or "to" in date_filter:
322
+ return [f"{label}: expected all time, got from={date_filter.get('from')!r} to={date_filter.get('to')!r}"]
323
+ return []
324
+ return [
325
+ f"{label}: date {key} expected {expected.get(key)!r}, got {date_filter.get(key)!r}"
326
+ for key in ("granularity", "from", "to")
327
+ if date_filter.get(key) != expected.get(key)
328
+ ]
329
+
330
+
202
331
  def _check_date_range(dashboard: dict, expected: dict | None) -> list[str]:
203
- """The date filter of every tab must match the expectation.
332
+ """Check the dashboard's date range against the expectation.
333
+
334
+ What counts as "the" date filter differs by shape, so the two are checked differently.
335
+
336
+ A drafted dashboard carries one per tab, and the draft tool writes the same one onto every
337
+ tab, so every tab must have it and every one must match. A tab left without a date filter is
338
+ a failure: the expectation describes what the whole dashboard shows.
339
+
340
+ A saved dashboard relayed for editing carries a flat map that can legitimately hold more
341
+ than one date filter — a dashboard-wide one plus a dataset-scoped one — and nothing in the
342
+ document says which is which. Requiring every entry to match would fail a dashboard for
343
+ carrying a filter the fixture never described, so one matching entry satisfies the check and
344
+ the rest are left to the notes.
345
+ """
346
+ date_filters = _filters_of_type(dashboard, "date_filter")
347
+ if not date_filters:
348
+ return ["the dashboard carries no date filter to check the expected range against"]
349
+
350
+ tabs = dashboard.get("tabs")
351
+ if tabs:
352
+ failures: list[str] = [
353
+ f"tab {tab.get('id')!r} carries no date filter"
354
+ for tab in tabs
355
+ if not any(
356
+ f.get("type") == "date_filter" for f in (tab.get("filters") or {}).values() if isinstance(f, dict)
357
+ )
358
+ ]
359
+ for label, date_filter in date_filters:
360
+ failures.extend(_date_filter_mismatch(label, date_filter, expected))
361
+ return failures
362
+
363
+ mismatches = [_date_filter_mismatch(label, date_filter, expected) for label, date_filter in date_filters]
364
+ if any(not m for m in mismatches):
365
+ return []
366
+ return [reason for m in mismatches for reason in m]
204
367
 
205
- A draft always carries a date filter, so ``expected is None`` (all time) means the tab
206
- filter has neither bound — absent keys, not null ones.
368
+
369
+ def _selection_of(attribute_filter: dict) -> tuple[str, list] | None:
370
+ """The elements an attribute filter restricts to, or ``None`` when it restricts nothing.
371
+
372
+ An unrestricted filter is spelled as the absence of a selection, and an empty list means
373
+ the same thing, so both read as "all".
207
374
  """
208
- tabs = dashboard.get("tabs") or []
209
- if not tabs:
210
- return ["draft has no tabs to read a date filter from"]
375
+ state = attribute_filter.get("state") or {}
376
+ for kind in ("include", "exclude"):
377
+ values = state.get(kind)
378
+ if values:
379
+ return kind, list(values)
380
+ return None
381
+
382
+
383
+ def _describe_selection(selection: tuple[str, list] | None) -> str:
384
+ return "all" if selection is None else f"{selection[0]} {selection[1]!r}"
385
+
386
+
387
+ def _check_filters(dashboard: dict, expected: list[dict]) -> list[str]:
388
+ """Every expected attribute filter must be on the dashboard, with the stated selection.
389
+
390
+ This exists because an edit can silently drop or widen the filters the user already had:
391
+ the change asked for was a rename, and the filters coming back untouched is part of what
392
+ "untouched" means. Filters the dashboard carries beyond the expected ones are left alone,
393
+ the same way extra widgets are.
394
+ """
395
+ actual: dict[str | None, list[dict]] = {}
396
+ for _label, attribute_filter in _filters_of_type(dashboard, "attribute_filter"):
397
+ actual.setdefault(attribute_filter.get("using"), []).append(attribute_filter)
398
+
211
399
  failures: list[str] = []
212
- for tab in tabs:
213
- tab_id = tab.get("id")
214
- date_filter = (tab.get("filters") or {}).get("date") or {}
215
- if expected is None:
216
- if "from" in date_filter or "to" in date_filter:
217
- failures.append(
218
- f"tab {tab_id!r}: expected all time, got from={date_filter.get('from')!r} "
219
- f"to={date_filter.get('to')!r}"
220
- )
400
+ for exp in expected:
401
+ using = exp.get("using")
402
+ matches = actual.get(using) or []
403
+ if not matches:
404
+ failures.append(f"the dashboard carries no attribute filter on {using!r}")
221
405
  continue
222
- failures.extend(
223
- f"tab {tab_id!r}: date {key} expected {expected.get(key)!r}, got {date_filter.get(key)!r}"
224
- for key in ("granularity", "from", "to")
225
- if date_filter.get(key) != expected.get(key)
226
- )
406
+ wanted = _expected_selection(exp)
407
+ if not any(_selection_of(f) == wanted for f in matches):
408
+ found = ", ".join(_describe_selection(_selection_of(f)) for f in matches)
409
+ failures.append(f"filter on {using!r} selects {found}, expected {_describe_selection(wanted)}")
227
410
  return failures
228
411
 
229
412
 
413
+ def _expected_selection(expected_filter: dict) -> tuple[str, list] | None:
414
+ """Read a fixture's filter entry into the same shape ``_selection_of`` produces.
415
+
416
+ Raises:
417
+ ValueError: the entry states no selection the evaluator knows how to check.
418
+ Deliberately loud -- a typo here would otherwise assert nothing and read green.
419
+ """
420
+ for kind in ("include", "exclude"):
421
+ if kind in expected_filter:
422
+ return kind, list(expected_filter[kind] or [])
423
+ if expected_filter.get("selection") == "all":
424
+ return None
425
+ raise ValueError(
426
+ f"filter expectation for {expected_filter.get('using')!r} needs 'selection': 'all', "
427
+ f"'include' or 'exclude', got {expected_filter!r}"
428
+ )
429
+
430
+
230
431
  def _date_range_text(date_range: dict | None) -> str:
231
432
  """Render a date range the way a user would say it, for the simulated reply.
232
433
 
@@ -290,6 +491,26 @@ def _min_new_visualizations(expected_output: dict) -> int:
290
491
  return value
291
492
 
292
493
 
494
+ def _has_filters(expected_output: dict) -> bool:
495
+ """Whether the expectation says anything about the attribute filters.
496
+
497
+ Absent means the case does not look at them, exactly as for ``date_range``. Only the
498
+ cases that are about preserving filters carry the key, so the others stay unaffected.
499
+ """
500
+ return "filters" in expected_output
501
+
502
+
503
+ def _has_date_range(expected_output: dict) -> bool:
504
+ """Whether the expectation says anything about the date filter at all.
505
+
506
+ Absence and ``null`` are different answers and must not be collapsed: ``null`` means the
507
+ dashboard has to be on all time, while a missing key means the case does not look at the
508
+ filter. Edit cases omit it because a dashboard the user has not opened carries no filter
509
+ context, so ``patch_dashboard`` refuses filter changes outright.
510
+ """
511
+ return "date_range" in expected_output
512
+
513
+
293
514
  def _validate_expectation(expected_output: dict) -> None:
294
515
  """Reject a fixture the run could not score meaningfully, before the first API call.
295
516
 
@@ -302,19 +523,67 @@ def _validate_expectation(expected_output: dict) -> None:
302
523
  """
303
524
  if not expected_output.get("visualizations"):
304
525
  raise ValueError("expected_output lists no visualizations; every chart check would pass vacuously")
305
- _date_range_text(expected_output.get("date_range"))
526
+ if _has_date_range(expected_output):
527
+ date_range = expected_output.get("date_range")
528
+ # Shape first, and on both kinds of case. Only a creation case goes on to check that
529
+ # the reply can phrase the period -- an edit records whatever range the saved dashboard
530
+ # carries, which `_check_date_range` compares by number rather than reading aloud, so
531
+ # demanding a phrasable one there would cap what the preserved-filter cases can cover.
532
+ # Without the shape check a string slips through to scoring and dies there instead,
533
+ # after the run has already been spent.
534
+ if date_range is not None and not isinstance(date_range, dict):
535
+ raise ValueError(f"date_range must be an object or null, got {date_range!r}")
536
+ if not _is_edit(expected_output):
537
+ _date_range_text(date_range)
538
+
539
+ filters = expected_output.get("filters")
540
+ if filters is not None and not isinstance(filters, list):
541
+ raise ValueError(f"filters must be a list of filter expectations, got {filters!r}")
542
+ for entry in filters or []:
543
+ if not isinstance(entry, dict):
544
+ raise ValueError(f"a filter expectation must be an object, got {entry!r}")
545
+ if not entry.get("using"):
546
+ raise ValueError(f"a filter expectation needs the label it filters on, got {entry!r}")
547
+ _expected_selection(entry)
548
+ saved_id = expected_output.get("saved_dashboard_id")
549
+ if _is_edit(expected_output):
550
+ if not isinstance(saved_id, str) or not saved_id:
551
+ raise ValueError(f"an edit expectation needs the edited dashboard's saved_dashboard_id, got {saved_id!r}")
552
+ elif saved_id is not None:
553
+ raise ValueError(f"a creation expectation must not name a saved_dashboard_id, got {saved_id!r}")
306
554
  _min_new_visualizations(expected_output)
307
555
 
308
556
 
557
+ @dataclass(frozen=True)
558
+ class _Applies:
559
+ """Which of the conditional checks the case applies.
560
+
561
+ They are published only when they do. A check that could not fail is not evidence, and
562
+ publishing it as passed lifts `quality_score` -- the fraction of true booleans in the
563
+ detail -- above what the run earned, which would make a creation case read better than the
564
+ same case scored before editing existed.
565
+ """
566
+
567
+ patch: bool
568
+ date: bool
569
+ filters: bool
570
+
571
+
309
572
  @dataclass
310
573
  class DashboardEvaluation:
311
574
  """Per-run outcome of the dashboard-skill checks.
312
575
 
313
- ``strict_checks`` is what the run is scored on, and every entry in it is something the
314
- agent decided. ``skill_activated`` and ``references_carried`` are reported beside it but
315
- never gate: the first can be false on a route that pins skills instead of calling
316
- ``set_skills``, and the second reflects gen-ai's reference building rather than the
317
- agent's choices. ``notes`` carries the same kind of observation for titles.
576
+ ``strict_checks`` is what the run is scored on. ``skill_activated`` is in it because the
577
+ expectation's type names one skill and only one: a creation fixture answered by the editor,
578
+ or an edit fixture answered by the builder, is a routing failure even if the dashboard that
579
+ came back looks plausible.
580
+
581
+ ``references_carried`` stays out of the gate: gen-ai rejects a response naming an
582
+ unresolvable chart before it can succeed, so a widget missing from the references means
583
+ reference building degraded on the way out -- a platform fault, not the model's.
584
+ ``titles_matched`` is also out for creation, where the title is the model's own wording;
585
+ on an edit the expectation states the title the change must leave behind, so there it is
586
+ scored through ``charts_matched`` instead.
318
587
  """
319
588
 
320
589
  drafted: bool
@@ -322,7 +591,11 @@ class DashboardEvaluation:
322
591
  charts_matched: bool
323
592
  date_range_correct: bool
324
593
  new_visualizations_met: bool
594
+ applies: _Applies
595
+ filters_correct: bool = True
325
596
  skill_activated: bool = False
597
+ saved_dashboard_id_correct: bool = True
598
+ patch_applies: bool = True
326
599
  references_carried: bool = True
327
600
  titles_matched: bool = True
328
601
  failures: list[str] = field(default_factory=list)
@@ -334,26 +607,42 @@ class DashboardEvaluation:
334
607
 
335
608
  @property
336
609
  def strict_checks(self) -> dict[str, bool]:
337
- return {
610
+ checks = {
611
+ # Kept as-is through the editing work: every Langfuse view, saved filter and
612
+ # combo-report field list already refers to it, and a rename would break them for
613
+ # a word. It means "the tool that had to succeed did", patch or draft.
338
614
  "dashboard_drafted": self.drafted,
339
615
  "dashboard_part_present": self.part_present,
616
+ "dashboard_skill_activated": self.skill_activated,
617
+ "saved_dashboard_id_correct": self.saved_dashboard_id_correct,
340
618
  "charts_matched": self.charts_matched,
341
- "date_range_correct": self.date_range_correct,
342
619
  "new_visualizations_met": self.new_visualizations_met,
343
620
  }
621
+ if self.applies.patch:
622
+ checks["patch_applies"] = self.patch_applies
623
+ if self.applies.date:
624
+ checks["date_range_correct"] = self.date_range_correct
625
+ if self.applies.filters:
626
+ # Prefixed: alert_skill already publishes a `filters_correct` score, and the combo
627
+ # report resolves a trace's skill by which score names it carries.
628
+ checks["dashboard_filters_correct"] = self.filters_correct
629
+ return checks
344
630
 
345
631
  @property
346
632
  def diagnostics(self) -> dict[str, bool]:
347
633
  """Observed but not scored — see the class docstring.
348
634
 
349
- ``titles_matched`` is here rather than in the gate so the decision to stop failing on
350
- a model-chosen title stays reviewable: without a score, nothing would record how often
351
- it happens. The two chart-level observations are omitted when no dashboard part was
352
- read, because reporting them as true would claim references and titles were fine on a
353
- run that produced neither.
635
+ ``titles_matched`` reports the titles on both kinds of case, but only creation leaves
636
+ it out of the gate -- an edit scores the same mismatches through ``charts_matched``,
637
+ because there the expectation states the title the change must leave behind. Keeping
638
+ the score in both cases is what makes the creation decision reviewable later.
639
+
640
+ Both chart-level observations are omitted whenever no document was scored -- no part
641
+ read, or a patch that would not apply. Reporting them as true would claim references
642
+ and titles were fine on a run that never produced anything to look at.
354
643
  """
355
- observed = {"dashboard_skill_activated": self.skill_activated}
356
- if self.part_present:
644
+ observed: dict[str, bool] = {}
645
+ if self.part_present and self.patch_applies:
357
646
  observed["references_carried"] = self.references_carried
358
647
  observed["titles_matched"] = self.titles_matched
359
648
  return observed
@@ -361,12 +650,13 @@ class DashboardEvaluation:
361
650
 
362
651
  @dataclass
363
652
  class DashboardRunResult:
364
- """Outcome of one K-run conversation for dashboard creation."""
653
+ """Outcome of one K-run conversation, creating or editing."""
365
654
 
366
655
  conversation_id: str
367
656
  evaluation: DashboardEvaluation
368
- draft_result: dict | None = None
657
+ tool_result: dict | None = None
369
658
  dashboard_part: dict | None = None
659
+ patch_part: dict | None = None
370
660
  total_turns: int = 0
371
661
  total_steps: int = 0
372
662
  reasoning_steps: list[str] = field(default_factory=list)
@@ -378,7 +668,7 @@ class DashboardRunResult:
378
668
 
379
669
  @dataclass
380
670
  class AgenticDashboardSummary:
381
- """Aggregated outcome of K runs for dashboard creation."""
671
+ """Aggregated outcome of K runs, creating or editing."""
382
672
 
383
673
  run_results: list[DashboardRunResult]
384
674
  pass_at_k: bool
@@ -386,30 +676,43 @@ class AgenticDashboardSummary:
386
676
  best: DashboardRunResult
387
677
 
388
678
 
389
- def evaluate_dashboard_draft(
390
- draft_result: dict | None,
679
+ def evaluate_dashboard_response(
680
+ tool_result: dict | None,
391
681
  dashboard_part: dict | None,
392
682
  expected_output: dict,
393
683
  skill_activated: bool,
684
+ patch_part: dict | None = None,
394
685
  ) -> DashboardEvaluation:
395
- """Score one drafted dashboard against its expectation.
686
+ """Score one dashboard response against its expectation.
687
+
688
+ Creation reads the drafted dashboard straight off the ``dashboard`` part. Editing reads
689
+ two parts -- the saved dashboard as it stood, and the patch against it -- and scores the
690
+ document that results from applying one to the other, never the base.
396
691
 
397
- Pure — no network, no conversation state — so the assertion logic is unit-testable
692
+ Pure: no network, no conversation state, so the whole assertion surface is unit-testable
398
693
  without an agent.
399
694
  """
400
- if draft_result is None:
695
+ is_edit = _is_edit(expected_output)
696
+ applies = _Applies(patch=is_edit, date=_has_date_range(expected_output), filters=_has_filters(expected_output))
697
+ tool = _producing_tool(expected_output)
698
+
699
+ if tool_result is None:
401
700
  return DashboardEvaluation(
402
701
  drafted=False,
403
702
  part_present=False,
404
703
  charts_matched=False,
405
704
  date_range_correct=False,
705
+ filters_correct=False,
406
706
  new_visualizations_met=False,
707
+ applies=applies,
407
708
  skill_activated=skill_activated,
408
- failures=["the agent never produced a successful draft_dashboard call"],
709
+ saved_dashboard_id_correct=False,
710
+ patch_applies=False,
711
+ failures=[f"the agent never produced a successful {tool} call"],
409
712
  )
410
713
 
411
714
  min_new = _min_new_visualizations(expected_output)
412
- actual_new = draft_result.get("new_visualization_count")
715
+ actual_new = tool_result.get("new_visualization_count")
413
716
  if isinstance(actual_new, int):
414
717
  new_met = actual_new >= min_new
415
718
  # Naming the shortfall, never "expected at least 0" -- see the envelope branch below.
@@ -419,42 +722,116 @@ def evaluate_dashboard_draft(
419
722
  # "expected at least 0 authored chart(s), tool reported None" reads as a broken test
420
723
  # and would send whoever is on nightly duty looking at the model instead.
421
724
  new_met = False
422
- new_failures = [f"the draft result carries no new_visualization_count (got {actual_new!r})"]
725
+ new_failures = [f"the {tool} result carries no new_visualization_count (got {actual_new!r})"]
423
726
 
727
+ missing_parts: list[str] = []
424
728
  if dashboard_part is None:
729
+ missing_parts.append("dashboard")
730
+ if is_edit and patch_part is None:
731
+ missing_parts.append(_PATCH_TYPE)
732
+ if missing_parts:
425
733
  return DashboardEvaluation(
426
734
  drafted=True,
427
735
  part_present=False,
428
736
  charts_matched=False,
429
737
  date_range_correct=False,
738
+ filters_correct=False,
430
739
  new_visualizations_met=new_met,
740
+ applies=applies,
431
741
  skill_activated=skill_activated,
742
+ saved_dashboard_id_correct=False,
743
+ patch_applies=False,
432
744
  failures=[
433
- f"the response carries no {_expected_type(expected_output)!r} part",
745
+ f"the response carries no {' and no '.join(repr(p) for p in missing_parts)} part",
434
746
  *new_failures,
435
747
  ],
436
748
  )
437
749
 
438
- dashboard = dashboard_part.get("dashboard") or {}
439
- references = dashboard_part.get("references") or {}
440
- new_ids = {v.get("id") for v in references.get("new_visualizations") or [] if v.get("id")}
441
- known_ids = {v.get("id") for v in references.get("visualizations") or [] if v.get("id")} | new_ids
442
- widgets = _widgets_of(dashboard)
750
+ assert dashboard_part is not None # noqa: S101 narrowed by missing_parts above
751
+ base = dashboard_part.get("dashboard") or {}
752
+ base_references = dashboard_part.get("references") or {}
753
+ expected_saved_id = expected_output.get("saved_dashboard_id")
754
+ actual_saved_id = dashboard_part.get("saved_dashboard_id")
755
+
756
+ saved_failures: list[str] = []
757
+ patch_failures: list[str] = []
758
+
759
+ if is_edit:
760
+ patch = (patch_part or {}).get("patch") or {}
761
+ patch_references = patch.get("references") or {}
762
+ # The patch names the dashboard it addresses; the part names the one it shipped with.
763
+ # Both have to be the dashboard the fixture asked to edit, or the change landed
764
+ # somewhere nobody looked.
765
+ for label, value in (("the response", actual_saved_id), ("the patch", patch.get("dashboard_id"))):
766
+ if value != expected_saved_id:
767
+ saved_failures.append(f"{label} addresses dashboard {value!r}, expected {expected_saved_id!r}")
768
+ document, patch_error = _apply_patch(base, patch.get("operations") or [])
769
+ if patch_error is not None:
770
+ patch_failures.append(patch_error)
771
+ document = None
772
+ new_ids = {v.get("id") for v in patch_references.get("new_visualizations") or [] if v.get("id")}
773
+ known_ids = (
774
+ {v.get("id") for v in base_references.get("visualizations") or [] if v.get("id")}
775
+ | {v.get("id") for v in patch_references.get("visualizations") or [] if v.get("id")}
776
+ | new_ids
777
+ )
778
+ else:
779
+ if actual_saved_id is not None:
780
+ saved_failures.append(f"a new draft must not be saved yet, but it reports id {actual_saved_id!r}")
781
+ document = base
782
+ new_ids = {v.get("id") for v in base_references.get("new_visualizations") or [] if v.get("id")}
783
+ known_ids = {v.get("id") for v in base_references.get("visualizations") or [] if v.get("id")} | new_ids
443
784
 
444
- chart_failures, title_notes = _check_visualizations(widgets, new_ids, expected_output.get("visualizations") or [])
785
+ if document is None:
786
+ return DashboardEvaluation(
787
+ drafted=True,
788
+ part_present=True,
789
+ charts_matched=False,
790
+ date_range_correct=False,
791
+ filters_correct=False,
792
+ new_visualizations_met=new_met,
793
+ applies=applies,
794
+ skill_activated=skill_activated,
795
+ saved_dashboard_id_correct=not saved_failures,
796
+ patch_applies=False,
797
+ failures=[*patch_failures, *saved_failures, *new_failures],
798
+ )
799
+
800
+ widgets = _widgets_of(document)
801
+ chart_failures, title_mismatches = _check_visualizations(
802
+ widgets, new_ids, expected_output.get("visualizations") or []
803
+ )
804
+ # On an edit the title is part of what was asked for, so it fails the case; on creation it
805
+ # is only recorded. Either way `titles_matched` reads the mismatches themselves, never the
806
+ # list they were routed into -- a score that says "titles fine" beside a failed rename is
807
+ # worse than no score at all.
808
+ if is_edit:
809
+ chart_failures = [*chart_failures, *title_mismatches]
810
+ title_notes: list[str] = []
811
+ else:
812
+ title_notes = title_mismatches
445
813
  reference_notes = _check_references(widgets, known_ids)
446
- date_failures = _check_date_range(dashboard, expected_output.get("date_range"))
814
+ date_failures = (
815
+ _check_date_range(document, expected_output.get("date_range")) if _has_date_range(expected_output) else []
816
+ )
817
+ filter_failures = (
818
+ _check_filters(document, expected_output.get("filters") or []) if _has_filters(expected_output) else []
819
+ )
447
820
 
448
821
  return DashboardEvaluation(
449
822
  drafted=True,
450
823
  part_present=True,
451
824
  charts_matched=not chart_failures,
452
825
  date_range_correct=not date_failures,
826
+ filters_correct=not filter_failures,
453
827
  new_visualizations_met=new_met,
828
+ applies=applies,
454
829
  skill_activated=skill_activated,
830
+ saved_dashboard_id_correct=not saved_failures,
831
+ patch_applies=True,
455
832
  references_carried=not reference_notes,
456
- titles_matched=not title_notes,
457
- failures=[*chart_failures, *date_failures, *new_failures],
833
+ titles_matched=not title_mismatches,
834
+ failures=[*chart_failures, *saved_failures, *date_failures, *filter_failures, *new_failures],
458
835
  notes=[*title_notes, *reference_notes],
459
836
  )
460
837
 
@@ -466,14 +843,16 @@ def _execute_single_dashboard_run(
466
843
  expected_output: dict,
467
844
  max_iterations: int,
468
845
  ) -> DashboardRunResult:
469
- """Drive one conversation until the agent drafts a dashboard, then evaluate it.
846
+ """Drive one conversation until the agent produces a dashboard, then evaluate it.
470
847
 
471
848
  Nothing is cleaned up on the way out by design: gen-ai keeps the draft and any authored
472
849
  chart in conversation state and persists neither until a user saves from the UI.
473
850
  """
474
- expected_type = str(expected_output.get("type") or "dashboard")
475
- draft_result: dict | None = None
851
+ tool = _producing_tool(expected_output)
852
+ is_edit = _is_edit(expected_output)
853
+ tool_result: dict | None = None
476
854
  dashboard_part: dict | None = None
855
+ patch_part: dict | None = None
477
856
  turns = 0
478
857
  steps = 0
479
858
  current_question = question
@@ -504,14 +883,26 @@ def _execute_single_dashboard_run(
504
883
  all_reasoning_step_events.extend(chat_result.reasoning_step_events or [])
505
884
  steps += chat_result.reasoning_step_count
506
885
 
507
- candidate = _extract_draft_result(chat_result.tool_call_events or [])
886
+ candidate = _extract_tool_result(chat_result.tool_call_events or [], tool)
508
887
  if candidate is not None:
509
888
  log_timer(
510
889
  f"[timer] dashboard_skill {conversation_id} GoodData turn {turns} complete after "
511
- f"{agent_elapsed:.2f}s; draft received"
890
+ f"{agent_elapsed:.2f}s; {tool} result received"
512
891
  )
513
- draft_result = candidate
514
- dashboard_part = _extract_dashboard_part(chat_result, expected_type)
892
+ tool_result = candidate
893
+ # An edit relays two parts: the dashboard as it stood, and the patch against it.
894
+ # Both are read from the turn that produced the patch -- the base is what the patch
895
+ # was written for, so pairing it with any other turn's would score a document the
896
+ # agent never proposed.
897
+ # A turn may carry several patches. They are alternatives rebased on the same base
898
+ # rather than steps to compose, and the client resolves the last one as the proposal
899
+ # that holds (gdc-ui's applyDashboardPatch; gdc-nas
900
+ # dashboard_edit_assertion.verify_last_proposal_holds does the same). Taking the last
901
+ # of each part type is therefore what the user would end up looking at, and the
902
+ # earlier patches are alternatives that lost, not changes that went missing.
903
+ dashboard_part = _extract_dashboard_part(chat_result, "dashboard")
904
+ if is_edit:
905
+ patch_part = _extract_dashboard_part(chat_result, _PATCH_TYPE)
515
906
  break
516
907
 
517
908
  response_text = (chat_result.text_response or "").strip() or render_answer_text(chat_result)
@@ -523,18 +914,26 @@ def _execute_single_dashboard_run(
523
914
  f"[timer] dashboard_skill {conversation_id} GoodData turn {turns} complete after "
524
915
  f"{agent_elapsed:.2f}s; answering with the expected charts and date range"
525
916
  )
917
+ if is_edit:
918
+ # The creation reply names charts and a date range, which answers nothing a rename
919
+ # or a resize could have asked. Rather than send something the question did not
920
+ # ask for, stop: the failure is the signal that an edit case needs a reply of its
921
+ # own, and inventing one here would hide which cases actually need it.
922
+ break
526
923
  current_question = build_simulated_reply(expected_output)
527
924
 
528
925
  return DashboardRunResult(
529
926
  conversation_id=conversation_id,
530
- evaluation=evaluate_dashboard_draft(
531
- draft_result,
927
+ evaluation=evaluate_dashboard_response(
928
+ tool_result,
532
929
  dashboard_part,
533
930
  expected_output,
534
- _builder_skill_activated(all_tool_call_events),
931
+ _skill_activated(all_tool_call_events, _required_skill(expected_output)),
932
+ patch_part=patch_part,
535
933
  ),
536
- draft_result=draft_result,
934
+ tool_result=tool_result,
537
935
  dashboard_part=dashboard_part,
936
+ patch_part=patch_part,
538
937
  total_turns=turns,
539
938
  total_steps=steps,
540
939
  reasoning_steps=reasoning_steps,
@@ -711,11 +1110,22 @@ def evaluate_agentic_dashboard_skill(
711
1110
 
712
1111
  if not gate_passed(gate, pass_at_k=summary.pass_at_k, pass_power_k=summary.pass_power_k):
713
1112
  gate_note = gate_failure_note(gate, runs_passed, runs_effective)
714
- skill_note = (
715
- ""
716
- if best.evaluation.skill_activated or best.evaluation.drafted
717
- else " No set_skills call activated dashboard_builder, so check the feature flag before the model."
718
- )
1113
+ # Two failures wear the same missing skill, and the advice differs. If the other
1114
+ # dashboard skill was activated, the flag that registers both is plainly on and the
1115
+ # model simply routed the wrong way. If neither was, the flag is off or the agent has
1116
+ # its skills pinned. Sending a reader after the flag in the first case walks them past
1117
+ # the real failure, which is why the run's own tool calls decide which line they get.
1118
+ required_skill = _required_skill(expected_output)
1119
+ other_skill = _BUILDER_SKILL if required_skill == _EDITOR_SKILL else _EDITOR_SKILL
1120
+ if best.evaluation.skill_activated:
1121
+ skill_note = ""
1122
+ elif _skill_activated(best.tool_call_events, other_skill):
1123
+ skill_note = f" It activated {other_skill} instead of {required_skill}: a routing failure, not the flag."
1124
+ else:
1125
+ skill_note = (
1126
+ f" No set_skills call activated {required_skill} or {other_skill};"
1127
+ " one feature flag registers both, so check that before the model."
1128
+ )
719
1129
  notes = "; ".join(best.evaluation.notes)
720
1130
  exc = DashboardSkillAssertionError(
721
1131
  f"Dashboard skill assertion failed. {gate_note}{skill_note} "
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: gooddata-eval
3
- Version: 1.75.1.dev2
3
+ Version: 1.75.1.dev3
4
4
  Summary: Evaluate the GoodData AI agent against your own questions and models.
5
5
  Project-URL: Source, https://github.com/gooddata/gooddata-python-sdk
6
6
  Author-email: GoodData <support@gooddata.com>
@@ -17,8 +17,9 @@ Classifier: Topic :: Scientific/Engineering
17
17
  Classifier: Topic :: Software Development
18
18
  Classifier: Typing :: Typed
19
19
  Requires-Python: >=3.10
20
- Requires-Dist: gooddata-sdk~=1.75.1.dev2
20
+ Requires-Dist: gooddata-sdk~=1.75.1.dev3
21
21
  Requires-Dist: httpx<1.0,>=0.27
22
+ Requires-Dist: jsonpatch<2.0,>=1.33
22
23
  Requires-Dist: orjson<4.0.0,>=3.9.15
23
24
  Requires-Dist: pydantic<3.0,>=2.6
24
25
  Requires-Dist: rich<15.0,>=13.0
@@ -20,7 +20,7 @@ gooddata_eval/core/agentic/_langfuse.py,sha256=7V1PFQbWpRnFz0a3wklcI-ROYjGC1wGQv
20
20
  gooddata_eval/core/agentic/_trace_linker.py,sha256=iXMTxIXUsF5WNSen7v_K8JJ2j1_37lh-2VAGqQxxmkI,13538
21
21
  gooddata_eval/core/agentic/alert_skill.py,sha256=47V6LVQemAL0OmPeY9Yr1-lszkPu5bFkCgYgHSJJI_8,41712
22
22
  gooddata_eval/core/agentic/conversation.py,sha256=Y-SsGOth-TzXFlOKsElCigT05ZH6wNqIOOz1IUkzUo4,30044
23
- gooddata_eval/core/agentic/dashboard_skill.py,sha256=bzdGUsZ7PwqoTZHNYq_82Fj75aAkHBAXp8jIUAa2XsI,30905
23
+ gooddata_eval/core/agentic/dashboard_skill.py,sha256=4lu7v6VeIR_h9FUFBrpD57upA0Rd2Kox1ZP_X-7jQaQ,51854
24
24
  gooddata_eval/core/agentic/general_question.py,sha256=Jp3FaS4Tvmfo0dx9iceiZ3Fr1QzJKDl7fVZZDetS22I,14684
25
25
  gooddata_eval/core/agentic/guardrail.py,sha256=rYpnsun9mLOSYucWfg1dbdrpbzQcz2V3eLwgII8dD0w,13145
26
26
  gooddata_eval/core/agentic/kda_skill.py,sha256=P_pR0iC12b_hIwnLzBmH0IBEjsO_CxZrlfkmu10cVWE,22721
@@ -63,8 +63,8 @@ gooddata_eval/core/reporting/json_report.py,sha256=q3Jirc_Zhvzg5lkG_kM6JdnViNLe6
63
63
  gooddata_eval/core/reporting/report_template.html,sha256=vLkl95OZsm3dMdijv5zda-15wYPE5UvTs4uKeWnjCiI,22332
64
64
  gooddata_eval/core/summary/__init__.py,sha256=U54lANKm34yjqm7dZL9KkGouPbwAaQWC9wiAmb5B5g4,32
65
65
  gooddata_eval/core/summary/http_client.py,sha256=dPArJ6zoCu1W9xmYXFA1WMDNsSOkhn3gaCpYgCUp3gk,2174
66
- gooddata_eval-1.75.1.dev2.dist-info/METADATA,sha256=r44ORmpL9SsDmr7ceeoLp2wjIgL4Oqv4TPBnKszyS18,35033
67
- gooddata_eval-1.75.1.dev2.dist-info/WHEEL,sha256=W3fkpkm7-wf9vBI5Z-7s0eWkeM-spu78I8Neb98DeEg,87
68
- gooddata_eval-1.75.1.dev2.dist-info/entry_points.txt,sha256=28nFp5Viknx4haPYgzK9QHlwPFfC6rPF6RmjC2ht8EI,56
69
- gooddata_eval-1.75.1.dev2.dist-info/licenses/LICENSE.txt,sha256=LVfVlC9maU3K9lgMwdZnClQQsOIbOekdcdVKr9qzJtk,256836
70
- gooddata_eval-1.75.1.dev2.dist-info/RECORD,,
66
+ gooddata_eval-1.75.1.dev3.dist-info/METADATA,sha256=hm6riuge5SGz7wMiNFSuWNZk39Qcx3ph7w8NDaMddGo,35069
67
+ gooddata_eval-1.75.1.dev3.dist-info/WHEEL,sha256=W3fkpkm7-wf9vBI5Z-7s0eWkeM-spu78I8Neb98DeEg,87
68
+ gooddata_eval-1.75.1.dev3.dist-info/entry_points.txt,sha256=28nFp5Viknx4haPYgzK9QHlwPFfC6rPF6RmjC2ht8EI,56
69
+ gooddata_eval-1.75.1.dev3.dist-info/licenses/LICENSE.txt,sha256=LVfVlC9maU3K9lgMwdZnClQQsOIbOekdcdVKr9qzJtk,256836
70
+ gooddata_eval-1.75.1.dev3.dist-info/RECORD,,