gooddata-eval 1.74.1.dev1__py3-none-any.whl → 1.74.1.dev2__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -2,6 +2,41 @@
2
2
  from __future__ import annotations
3
3
 
4
4
  from dataclasses import dataclass, field
5
+ from enum import Enum
6
+
7
+
8
+ class AnomalyDetectionGranularity(str, Enum):
9
+ """Detection intervals an anomaly alert accepts.
10
+
11
+ Mirrors gen-ai's enum of the same name; `StrEnum` is unavailable on the 3.10 floor, so
12
+ the `str` mixin carries the comparison against the raw tool argument.
13
+ """
14
+
15
+ HOUR = "HOUR"
16
+ DAY = "DAY"
17
+ WEEK = "WEEK"
18
+ MONTH = "MONTH"
19
+ QUARTER = "QUARTER"
20
+ YEAR = "YEAR"
21
+
22
+ @classmethod
23
+ def parse(cls, value: object) -> AnomalyDetectionGranularity | None:
24
+ """Coerce a fixture value, or None when the fixture states none.
25
+
26
+ Raises ValueError on an unknown interval: fixtures are hand-written, and a typo has
27
+ to fail before the run spends an API call rather than score the item against an
28
+ interval the product cannot produce.
29
+ """
30
+ if value is None:
31
+ return None
32
+ candidate = str(value).strip().upper()
33
+ if not candidate:
34
+ return None
35
+ try:
36
+ return cls(candidate)
37
+ except ValueError:
38
+ expected = ", ".join(member.value for member in cls)
39
+ raise ValueError(f"Invalid granularity {value!r}; expected one of {expected}.") from None
5
40
 
6
41
 
7
42
  @dataclass
@@ -28,6 +63,10 @@ class CatalogMetricAlert:
28
63
  """List of recipient email addresses."""
29
64
  filters: list | str | None = None
30
65
  """Attribute filters applied to the alert condition."""
66
+ attributes: list | None = None
67
+ """Expected group-by attributes; ``None`` means the fixture states no expectation."""
68
+ granularity: AnomalyDetectionGranularity | None = None
69
+ """Detection interval for an ANOMALY alert (DAY/WEEK/MONTH/...). Not a date filter."""
31
70
 
32
71
  @classmethod
33
72
  def from_dict(cls, d: dict) -> CatalogMetricAlert:
@@ -46,4 +85,6 @@ class CatalogMetricAlert:
46
85
  metric_id=d.get("metric_id"),
47
86
  recipients=recipients,
48
87
  filters=d.get("filters"),
88
+ attributes=d.get("attributes"),
89
+ granularity=AnomalyDetectionGranularity.parse(d.get("granularity")),
49
90
  )
@@ -11,7 +11,7 @@ from typing import Any
11
11
 
12
12
  from gooddata_sdk import GoodDataSdk
13
13
 
14
- from gooddata_eval.core.agentic._catalog import CatalogMetricAlert
14
+ from gooddata_eval.core.agentic._catalog import AnomalyDetectionGranularity, CatalogMetricAlert
15
15
  from gooddata_eval.core.agentic._trace_linker import (
16
16
  RunIdentity,
17
17
  RunTraceContext,
@@ -21,6 +21,7 @@ from gooddata_eval.core.agentic._trace_linker import (
21
21
  submit_trace_scoring,
22
22
  utc_now,
23
23
  )
24
+ from gooddata_eval.core.chat.render import render_answer_text
24
25
  from gooddata_eval.core.chat.sse_client import ChatClient
25
26
  from gooddata_eval.core.config import ReasoningEffort
26
27
  from gooddata_eval.core.models import (
@@ -29,6 +30,7 @@ from gooddata_eval.core.models import (
29
30
  ReasoningStepEvent,
30
31
  ToolCallEvent,
31
32
  build_latency_breakdown,
33
+ shift_and_index_events,
32
34
  )
33
35
 
34
36
  try:
@@ -126,6 +128,94 @@ def _check_filters(expected: CatalogMetricAlert, actual_args: dict) -> bool:
126
128
  return _deep_subset(exp_filters, act_filters)
127
129
 
128
130
 
131
+ def _attribute_label_ids(items: list, *, side: str) -> list[str]:
132
+ """Canonicalise group-by entries to bare label ids, whatever spelling they arrive in.
133
+
134
+ The two sides of the comparison speak different vocabularies for the same grouping.
135
+ Fixtures author the AAC tool-input form, ``{"using": "label/x"}``; ``create_metric_alert``
136
+ receives the resolved AFM form, ``{"localIdentifier": "a0", "label": {"identifier":
137
+ {"id": "x", "type": "label"}}}``, forwarded verbatim from ``prepare_metric_alert_proposal``.
138
+ Identity is therefore the only thing they can be compared on.
139
+
140
+ A shape not listed here, or a URI prefix other than ``label/``, raises: ``label/x`` and
141
+ ``attribute/x`` are different objects, and an unknown spelling must fail loudly rather
142
+ than quietly compare unequal.
143
+ """
144
+ if not isinstance(items, list):
145
+ raise ValueError(f"Unrecognised {side} group-by attributes, expected a list: {items!r}")
146
+ ids: list[str] = []
147
+ for item in items:
148
+ raw: object = None
149
+ if isinstance(item, str):
150
+ raw = item
151
+ elif isinstance(item, dict):
152
+ label = item.get("label")
153
+ identifier = item.get("identifier")
154
+ if isinstance(item.get("using"), str):
155
+ raw = item["using"]
156
+ elif isinstance(label, dict) and isinstance(label.get("identifier"), dict):
157
+ raw = label["identifier"].get("id")
158
+ elif isinstance(identifier, dict):
159
+ raw = identifier.get("id")
160
+ if not isinstance(raw, str) or not raw:
161
+ raise ValueError(f"Unrecognised {side} group-by attribute entry: {item!r}")
162
+ prefix, slash, rest = raw.partition("/")
163
+ if not slash:
164
+ ids.append(raw)
165
+ elif prefix == "label" and rest:
166
+ ids.append(rest)
167
+ else:
168
+ raise ValueError(f"Unrecognised {side} group-by attribute reference: {raw!r}")
169
+ return ids
170
+
171
+
172
+ def _check_attributes(expected: CatalogMetricAlert, actual_args: dict) -> bool:
173
+ """Compare group-by identity only.
174
+
175
+ Per-entry properties — ``showAllValues``, the converter-assigned ``localIdentifier`` —
176
+ are deliberately not asserted, and the comparison is a multiset so entry order does not
177
+ matter.
178
+ """
179
+ exp_attributes = expected.attributes
180
+ if exp_attributes is None:
181
+ return True
182
+ act_attributes = actual_args.get("attributes")
183
+ if act_attributes is None:
184
+ # Arguments are raw `json.loads` output, where an unset nullable argument arrives as
185
+ # null rather than absent. Both spellings of "no grouping" have to land on [], which
186
+ # is why this is not `actual_args.get("attributes", [])`.
187
+ act_attributes = []
188
+ elif not isinstance(act_attributes, list):
189
+ # An argument that is not a list of groupings is the agent answering wrongly, so it
190
+ # scores False. Raising instead would make the runner record an ERROR, and errored
191
+ # items are excluded from the failure count — a malformed answer must not rank above
192
+ # a merely wrong one. An unreadable *entry* still raises, in `_attribute_label_ids`:
193
+ # entries are typed at the tool boundary, so the plausible cause there is the wire
194
+ # format moving, which has to be unmissable.
195
+ return False
196
+ if not exp_attributes:
197
+ return not act_attributes
198
+ exp_ids = sorted(_attribute_label_ids(exp_attributes, side="expected"))
199
+ act_ids = sorted(_attribute_label_ids(act_attributes, side="actual"))
200
+ return exp_ids == act_ids
201
+
202
+
203
+ def _check_granularity(expected: CatalogMetricAlert, actual_args: dict) -> bool:
204
+ """Compare the ANOMALY detection interval when the fixture states one.
205
+
206
+ ``None`` means unasserted, mirroring ``attributes``: only the ANOMALY items carry a
207
+ ``Granularity``, and every other item must stay unaffected. The expectation is already
208
+ canonical by the time it lands here; the tool argument is a raw string, so only that
209
+ side needs folding.
210
+ """
211
+ if expected.granularity is None:
212
+ return True
213
+ actual = actual_args.get("granularity")
214
+ if not actual:
215
+ return False
216
+ return str(actual).strip().upper() == expected.granularity.value
217
+
218
+
129
219
  def _check_metric(expected: CatalogMetricAlert, actual_args: dict) -> bool:
130
220
  if not expected.metric_id:
131
221
  return True
@@ -257,20 +347,38 @@ def generate_simulated_alert_response(
257
347
  )
258
348
  elif filters == []:
259
349
  filters_rule = (
260
- "5. Your alert must have NO filters and NO date/time window — it evaluates over all time. "
261
- "If the agent asks which time period each check should cover, or offers a choice such as "
262
- "'last Day / Week / Month', do NOT pick one: reply that you want no date filter at all, "
263
- "all time. Never invent a period, a granularity or an 'evaluate each run on a X basis' "
264
- "instruction the goal did not ask for.\n"
350
+ "5. Your alert must have NO filters and NO date/time window on the metric — it evaluates "
351
+ "over all time. If the agent asks which time period each check should cover, or offers a "
352
+ "choice such as 'last Day / Week / Month', do NOT pick one: reply that you want no date "
353
+ "filter at all, all time.\n"
265
354
  )
266
355
  else:
267
356
  filters_rule = (
268
357
  "5. Ask only for the filters your original request implies — do not invent an evaluation "
269
- "period, granularity or date window that was not requested. If the agent offers a choice "
358
+ "period or date window that was not requested. If the agent offers a choice "
270
359
  "such as 'last Day / Week / Month' that your request never mentioned, say you do not want "
271
360
  "a date window.\n"
272
361
  )
273
362
 
363
+ if operator == "ANOMALY":
364
+ # The fallback keeps the conversation alive when the fixture names no interval -- an
365
+ # anomaly alert cannot be created without one. It is deliberately NOT mirrored into
366
+ # `expected.granularity`: `_check_granularity` asserts only what the fixture stated,
367
+ # and scoring an item against an interval it never asked for is the defect this rule
368
+ # exists to undo.
369
+ granularity = (expected.granularity or AnomalyDetectionGranularity.DAY).value
370
+ anomaly_rule = (
371
+ "7. This is an ANOMALY alert. Anomaly detection REQUIRES a time granularity, and that "
372
+ f"granularity is NOT a date filter. State it in your first reply and repeat it whenever "
373
+ f"asked: use {granularity} granularity. Rule 5 constrains filters on the metric only — it "
374
+ "never applies to this detection interval, so never refuse to give one.\n"
375
+ )
376
+ else:
377
+ anomaly_rule = (
378
+ "7. Do not invent an evaluation period, a granularity or an 'evaluate each run on a X "
379
+ "basis' instruction your goal never asked for.\n"
380
+ )
381
+
274
382
  original_request = f'Your original request to the agent was: "{question}"\n' if question else ""
275
383
 
276
384
  system_prompt = (
@@ -294,8 +402,7 @@ def generate_simulated_alert_response(
294
402
  " Do not wait for the agent to ask — state it alongside the metric and condition answers.\n"
295
403
  + filters_rule
296
404
  + f"6. Proactively state how often you want to be alerted in your first reply: {trigger_request}. "
297
- " Repeat it if the agent proposes a different cadence.\n"
298
- "Reply concisely and directly."
405
+ " Repeat it if the agent proposes a different cadence.\n" + anomaly_rule + "Reply concisely and directly."
299
406
  )
300
407
 
301
408
  messages: list = [{"role": "system", "content": system_prompt}]
@@ -334,6 +441,8 @@ class AlertEvaluation:
334
441
  filters_correct: bool
335
442
  metric_correct: bool
336
443
  recipients_correct: bool
444
+ attributes_correct: bool = True
445
+ granularity_correct: bool = True
337
446
 
338
447
  @property
339
448
  def strict_pass(self) -> bool:
@@ -346,6 +455,8 @@ class AlertEvaluation:
346
455
  self.filters_correct,
347
456
  self.metric_correct,
348
457
  self.recipients_correct,
458
+ self.attributes_correct,
459
+ self.granularity_correct,
349
460
  ]
350
461
  )
351
462
 
@@ -409,6 +520,35 @@ def _normalize_expected_filters(expected: dict) -> list | str | None:
409
520
  return None
410
521
 
411
522
 
523
+ _NO_GROUPING_MARKERS = ("none", "no grouping")
524
+
525
+
526
+ def _normalize_expected_attributes(expected: dict) -> list | None:
527
+ """
528
+ * ``Attributes`` list -> that list (exact expectation)
529
+ * "None" / "no grouping" -> ``[]`` (stated: no group-by; extras fail)
530
+ * absent, or other prose -> ``None`` (unstated; grouping not asserted)
531
+
532
+ A date narrows an alert as a group-by as well as a filter, and a group-by makes it fire
533
+ per period value instead of on the latest one — so ``[]`` has to be expressible separately
534
+ from "absent", exactly as it is for ``filters``.
535
+
536
+ The simulated user is told nothing about groupings, so a non-empty expectation requires the
537
+ item's own question to request that grouping; ``[]`` needs no such support, because the
538
+ simulated user does not invent a grouping and the check verifies it did not.
539
+ """
540
+ attributes = _case_insensitive_get(expected, "attributes")
541
+ if isinstance(attributes, list):
542
+ # Validated here so a malformed fixture fails before the run spends an API call.
543
+ _attribute_label_ids(attributes, side="expected")
544
+ return attributes
545
+ if attributes is None:
546
+ return None
547
+ if isinstance(attributes, str):
548
+ return [] if any(kw in attributes.lower() for kw in _NO_GROUPING_MARKERS) else None
549
+ raise ValueError(f"Attributes expectation must be a list or a display string, got {type(attributes).__name__}")
550
+
551
+
412
552
  def _normalize_expected_output(expected: dict) -> CatalogMetricAlert:
413
553
  """Parse expected_output dict into CatalogMetricAlert, accepting display-format or internal-format keys."""
414
554
  operator = _case_insensitive_get(expected, "operator") or "GREATER_THAN"
@@ -433,6 +573,11 @@ def _normalize_expected_output(expected: dict) -> CatalogMetricAlert:
433
573
  recipients = list(raw_recip)
434
574
 
435
575
  filters = _normalize_expected_filters(expected)
576
+ attributes = _normalize_expected_attributes(expected)
577
+
578
+ granularity = AnomalyDetectionGranularity.parse(
579
+ _case_insensitive_get(expected, "granularity", "detection granularity")
580
+ )
436
581
 
437
582
  return CatalogMetricAlert(
438
583
  operator=operator,
@@ -443,6 +588,8 @@ def _normalize_expected_output(expected: dict) -> CatalogMetricAlert:
443
588
  metric_id=metric_id,
444
589
  recipients=recipients,
445
590
  filters=filters,
591
+ attributes=attributes,
592
+ granularity=granularity,
446
593
  )
447
594
 
448
595
 
@@ -526,21 +673,14 @@ def run_agentic_alert_skill(
526
673
  chat_result = client.send_message(conv_id, current_question)
527
674
  reasoning_steps.extend(chat_result.reasoning_steps or [])
528
675
  response_id = chat_result.response_id or response_id
529
- for tc in chat_result.tool_call_events or []:
530
- if tc.call_ts is not None:
531
- tc.call_ts += turn_offset
532
- if tc.result_ts is not None:
533
- tc.result_ts += turn_offset
534
- if tc.index is not None:
535
- tc.index += tool_index_offset
536
- for rs in chat_result.reasoning_step_events or []:
537
- rs.ts += turn_offset
538
- rs.index += reasoning_index_offset
676
+ turn_offset, tool_index_offset, reasoning_index_offset = shift_and_index_events(
677
+ chat_result,
678
+ turn_offset=turn_offset,
679
+ tool_index_offset=tool_index_offset,
680
+ reasoning_index_offset=reasoning_index_offset,
681
+ )
539
682
  all_tool_call_events.extend(chat_result.tool_call_events or [])
540
683
  all_reasoning_step_events.extend(chat_result.reasoning_step_events or [])
541
- tool_index_offset += len(chat_result.tool_call_events or [])
542
- reasoning_index_offset += len(chat_result.reasoning_step_events or [])
543
- turn_offset += chat_result.turn_wall_clock_sec or 0.0
544
684
  alert_id, actual_args, tool_called = _extract_alert_call(chat_result.tool_call_events or [])
545
685
  if tool_called:
546
686
  alert_id_to_delete = alert_id
@@ -548,6 +688,8 @@ def run_agentic_alert_skill(
548
688
  response_text = (chat_result.text_response or "").strip()
549
689
  if not response_text and chat_result.alert_proposals:
550
690
  response_text = render_alert_proposal(chat_result.alert_proposals[-1])
691
+ if not response_text:
692
+ response_text = render_answer_text(chat_result)
551
693
  # Stop if agent gave a completely empty response (stuck)
552
694
  if not response_text and not chat_result.tool_call_events:
553
695
  break
@@ -570,6 +712,8 @@ def run_agentic_alert_skill(
570
712
  filters_correct=tool_called and _check_filters(expected, actual_args),
571
713
  metric_correct=tool_called and _check_metric(expected, actual_args),
572
714
  recipients_correct=tool_called and _check_recipients(expected, actual_args, sdk=sdk),
715
+ attributes_correct=tool_called and _check_attributes(expected, actual_args),
716
+ granularity_correct=tool_called and _check_granularity(expected, actual_args),
573
717
  )
574
718
  return AlertRunResult(
575
719
  conversation_id=conv_id,
@@ -615,6 +759,8 @@ def run_agentic_alert_skill(
615
759
  r.eval.filters_correct,
616
760
  r.eval.metric_correct,
617
761
  r.eval.recipients_correct,
762
+ r.eval.attributes_correct,
763
+ r.eval.granularity_correct,
618
764
  ]
619
765
  ),
620
766
  )
@@ -689,6 +835,8 @@ def evaluate_agentic_alert_skill(
689
835
  "filters_correct": ev.filters_correct,
690
836
  "metric_correct": ev.metric_correct,
691
837
  "recipients_correct": ev.recipients_correct,
838
+ "attributes_correct": ev.attributes_correct,
839
+ "granularity_correct": ev.granularity_correct,
692
840
  }
693
841
  with ctx.observe(pt, run_idx) as tid:
694
842
  for score_name, value in strict_checks.items():
@@ -735,6 +883,8 @@ def evaluate_agentic_alert_skill(
735
883
  "filters_correct": ev.filters_correct,
736
884
  "metric_correct": ev.metric_correct,
737
885
  "recipients_correct": ev.recipients_correct,
886
+ "attributes_correct": ev.attributes_correct,
887
+ "granularity_correct": ev.granularity_correct,
738
888
  "actual_alert_arguments": best.actual_alert_arguments,
739
889
  "latency_breakdown": build_latency_breakdown(best.tool_call_events, best.reasoning_step_events),
740
890
  }
@@ -745,7 +895,9 @@ def evaluate_agentic_alert_skill(
745
895
  f"alert_created={ev.alert_created}, operator_correct={ev.operator_correct}, "
746
896
  f"threshold_correct={ev.threshold_correct}, trigger_correct={ev.trigger_correct}, "
747
897
  f"filters_correct={ev.filters_correct}, metric_correct={ev.metric_correct}, "
748
- f"recipients_correct={ev.recipients_correct}. "
898
+ f"recipients_correct={ev.recipients_correct}, "
899
+ f"attributes_correct={ev.attributes_correct}, "
900
+ f"granularity_correct={ev.granularity_correct}. "
749
901
  f"Actual args: {best.actual_alert_arguments}"
750
902
  )
751
903
  exc.reasoning_steps = best.reasoning_steps
@@ -6,10 +6,10 @@ from __future__ import annotations
6
6
  import json
7
7
  import re
8
8
  from dataclasses import dataclass, field
9
- from typing import Literal
9
+ from typing import ClassVar, Literal
10
10
 
11
11
  from gooddata_sdk import GoodDataSdk
12
- from pydantic import BaseModel
12
+ from pydantic import BaseModel, Field
13
13
 
14
14
  from gooddata_eval.core.agentic._trace_linker import (
15
15
  RunIdentity,
@@ -22,6 +22,7 @@ from gooddata_eval.core.agentic._trace_linker import (
22
22
  )
23
23
  from gooddata_eval.core.agentic.alert_skill import render_alert_proposal
24
24
  from gooddata_eval.core.agentic.metric_skill import _delete_metric, _extract_created_metric_ids, _extract_metric_result
25
+ from gooddata_eval.core.chat.render import render_answer_text
25
26
  from gooddata_eval.core.chat.sse_client import ChatClient
26
27
  from gooddata_eval.core.config import ReasoningEffort
27
28
  from gooddata_eval.core.models import (
@@ -31,6 +32,7 @@ from gooddata_eval.core.models import (
31
32
  ReasoningStepEvent,
32
33
  ToolCallEvent,
33
34
  build_latency_breakdown,
35
+ shift_and_index_events,
34
36
  )
35
37
  from gooddata_eval.core.scoring import (
36
38
  check_filters,
@@ -65,7 +67,21 @@ class ConversationFixture(BaseModel):
65
67
 
66
68
 
67
69
  class TurnResult(BaseModel):
68
- """Evaluation result for a single conversation turn."""
70
+ """Evaluation result for a single conversation turn.
71
+
72
+ The two skill fields measure DIFFERENT SCOPES, so a turn can legitimately report
73
+ ``skill_routing=True`` with an empty ``activated_skills``:
74
+
75
+ - ``activated_skills`` -- what THIS turn's own ``set_skills`` call declared. Empty
76
+ whenever the agent reused an already-active skill without re-declaring it.
77
+ - ``active_skills`` -- what was actually active DURING this turn: the last declared
78
+ set, carried over on turns that declare nothing. This is the set ``skill_routing``
79
+ is judged against, so a report never has to infer it.
80
+ - ``skill_routing`` -- whether ``expected_skill`` appears in ``active_skills``.
81
+
82
+ ``skill_routing=True`` with ``activated_skills=[]`` is the reused-skill case, not a
83
+ scoring bug -- ``active_skills`` shows where the credit came from.
84
+ """
69
85
 
70
86
  turn_id: str
71
87
  expected_skill: str
@@ -73,6 +89,8 @@ class TurnResult(BaseModel):
73
89
  output_present: bool
74
90
  no_error: bool
75
91
  activated_skills: list[str]
92
+ # Sorted for stable output: the source is a set, whose iteration order is not.
93
+ active_skills: list[str] = Field(default_factory=list)
76
94
  clarification_turns_used: int = 0
77
95
  output_correct: bool | None = None
78
96
 
@@ -80,6 +98,26 @@ class TurnResult(BaseModel):
80
98
  def skill_success(self) -> bool:
81
99
  return self.skill_routing and self.output_present and self.no_error
82
100
 
101
+ # Reported per turn in detail["turns"]. A name listed here that no longer exists on the
102
+ # model raises rather than silently emitting a stale key, which a hand-written dict
103
+ # literal of the same fields would not -- and model_dump deep-copies activated_skills,
104
+ # so a caller mutating the returned dict cannot reach back into this TurnResult.
105
+ _DETAIL_FIELDS: ClassVar[set[str]] = {
106
+ "turn_id",
107
+ "expected_skill",
108
+ "skill_routing",
109
+ "output_present",
110
+ "output_correct",
111
+ "activated_skills",
112
+ # What skill_routing was judged against -- without it, a turn showing
113
+ # skill_routing=True and activated_skills=[] looks like a scoring bug.
114
+ "active_skills",
115
+ }
116
+
117
+ def detail(self) -> dict:
118
+ """The subset of this result reported in detail["turns"] for one conversation turn."""
119
+ return self.model_dump(include=self._DETAIL_FIELDS)
120
+
83
121
 
84
122
  def _resolve_refs(
85
123
  expected_output: dict | None,
@@ -117,15 +155,42 @@ def _resolve_refs(
117
155
  return json.loads(resolved_raw)
118
156
 
119
157
 
120
- def _activated_skills(tool_call_events: list[ToolCallEvent]) -> list[str]:
121
- """Collect all skill names passed to set_skills across all tool call events."""
122
- skills: list[str] = []
158
+ def _set_skills_declarations(tool_call_events: list[ToolCallEvent]) -> list[list[str]]:
159
+ """Every set_skills declaration in these events, in call order.
160
+
161
+ `skill_names` is the key the tool declares; `skills` is a legacy spelling kept as a
162
+ fallback. A call carrying neither is treated as declaring an empty list, which is what
163
+ the platform would do with one.
164
+ """
165
+ declarations: list[list[str]] = []
123
166
  for tc in tool_call_events:
124
167
  if tc.function_name != "set_skills":
125
168
  continue
126
169
  args = tc.parsed_arguments() or {}
127
- skills.extend(args.get("skill_names") or args.get("skills") or [])
128
- return list(set(skills))
170
+ names = args.get("skill_names")
171
+ if names is None:
172
+ names = args.get("skills")
173
+ declarations.append(list(names or []))
174
+ return declarations
175
+
176
+
177
+ def _final_skill_declaration(tool_call_events: list[ToolCallEvent]) -> list[str] | None:
178
+ """The skill list from the LAST set_skills call, or None when there was no call.
179
+
180
+ set_skills replaces the active set, so when a turn issues several calls -- which it can,
181
+ since these events span every clarification sub-turn within one logical turn -- only the
182
+ final one describes the resulting state. Merging them would credit a skill that an
183
+ earlier call declared and a later one dropped.
184
+
185
+ An empty list is a real declaration: it deactivates everything. That has to stay
186
+ distinguishable from ``None`` ("no call at all"), which leaves the previous turn's set
187
+ untouched -- hence the Optional rather than just an empty list for both.
188
+
189
+ This answers "what is active NOW". For "was this skill ever exercised" -- what
190
+ full_skill_coverage asks -- use every declaration, not just the last one.
191
+ """
192
+ declarations = _set_skills_declarations(tool_call_events)
193
+ return declarations[-1] if declarations else None
129
194
 
130
195
 
131
196
  def _check_output_present(turn: TurnDefinition, chat_result: ChatResult) -> bool:
@@ -150,7 +215,7 @@ def _check_output_correct(turn: TurnDefinition, chat_result: ChatResult) -> bool
150
215
 
151
216
  Returns None when expected_output is absent (presence check only).
152
217
  """
153
- from gooddata_eval.core.agentic.metric_skill import _normalize_maql # noqa: PLC0415
218
+ from gooddata_eval.core.evaluators._maql import normalize_maql # noqa: PLC0415
154
219
 
155
220
  otype = turn.expected_output_type
156
221
  expected = turn.expected_output
@@ -192,7 +257,7 @@ def _check_output_correct(turn: TurnDefinition, chat_result: ChatResult) -> bool
192
257
  metric_result = _extract_metric_result(chat_result.tool_call_events or [])
193
258
  if not metric_result:
194
259
  return False
195
- return _normalize_maql(metric_result.get("maql", "")) == _normalize_maql(expected.get("maql", ""))
260
+ return normalize_maql(metric_result.get("maql", "")) == normalize_maql(expected.get("maql", ""))
196
261
 
197
262
  return None
198
263
 
@@ -310,6 +375,20 @@ def run_agentic_conversation(
310
375
  response_id: str | None = None
311
376
  conversation_tool_call_events: list[ToolCallEvent] = []
312
377
  conversation_reasoning_step_events: list[ReasoningStepEvent] = []
378
+ # The skills active right now, mirroring the platform's own state machine: set_skills
379
+ # REPLACES the active set rather than adding to it -- verified against the gen-ai
380
+ # service's skill registry, and stated in the tool's own description. So a turn that
381
+ # issues no set_skills call inherits the previous turn's set unchanged, while a turn
382
+ # that does issue one drops whatever it left out. Tracking this as a running UNION
383
+ # would credit a skill a later call had already switched off.
384
+ active_skills: set[str] = set()
385
+ # Every skill declared at any point, for full_skill_coverage. This asks a DIFFERENT
386
+ # question from active_skills -- "did the conversation ever exercise this skill" rather
387
+ # than "is it active now" -- so replace semantics do not apply: a skill switched on and
388
+ # later switched off was still exercised. Deriving coverage from the per-turn final
389
+ # declaration instead would drop any skill a turn declared and then replaced within
390
+ # itself (across its clarification sub-turns), a false FAIL on a genuine activation.
391
+ ever_declared_skills: set[str] = set()
313
392
  # Every send_message() call (across every logical turn AND every clarification
314
393
  # sub-turn within it) restarts call_ts/ts near 0 -- these run across the whole
315
394
  # conversation, not reset per logical turn, so every one of those calls shifts them.
@@ -337,6 +416,10 @@ def run_agentic_conversation(
337
416
  output_present=False,
338
417
  no_error=False,
339
418
  activated_skills=[],
419
+ # The turn never ran, so it declared nothing -- but a set carried over
420
+ # from an earlier turn is still active, and reporting [] here would
421
+ # read as "nothing was active", which is a different claim.
422
+ active_skills=sorted(active_skills),
340
423
  output_correct=False,
341
424
  )
342
425
  )
@@ -351,22 +434,15 @@ def run_agentic_conversation(
351
434
  for _iter in range(max_clarification_turns + 1):
352
435
  chat_result = client.send_message(conversation_id, current_message)
353
436
  final_result = chat_result
354
- for tc in chat_result.tool_call_events or []:
355
- if tc.call_ts is not None:
356
- tc.call_ts += turn_offset
357
- if tc.result_ts is not None:
358
- tc.result_ts += turn_offset
359
- if tc.index is not None:
360
- tc.index += tool_index_offset
361
- for rs in chat_result.reasoning_step_events or []:
362
- rs.ts += turn_offset
363
- rs.index += reasoning_index_offset
437
+ turn_offset, tool_index_offset, reasoning_index_offset = shift_and_index_events(
438
+ chat_result,
439
+ turn_offset=turn_offset,
440
+ tool_index_offset=tool_index_offset,
441
+ reasoning_index_offset=reasoning_index_offset,
442
+ )
364
443
  all_tool_calls.extend(chat_result.tool_call_events or [])
365
444
  conversation_tool_call_events.extend(chat_result.tool_call_events or [])
366
445
  conversation_reasoning_step_events.extend(chat_result.reasoning_step_events or [])
367
- tool_index_offset += len(chat_result.tool_call_events or [])
368
- reasoning_index_offset += len(chat_result.reasoning_step_events or [])
369
- turn_offset += chat_result.turn_wall_clock_sec or 0.0
370
446
  reasoning_steps.extend(chat_result.reasoning_steps or [])
371
447
  response_id = chat_result.response_id or response_id
372
448
 
@@ -376,6 +452,8 @@ def run_agentic_conversation(
376
452
  response_text = (chat_result.text_response or "").strip()
377
453
  if not response_text and chat_result.alert_proposals:
378
454
  response_text = render_alert_proposal(chat_result.alert_proposals[-1])
455
+ if not response_text:
456
+ response_text = render_answer_text(chat_result)
379
457
  if not response_text and not chat_result.tool_call_events:
380
458
  break
381
459
  if clarification_turns >= max_clarification_turns:
@@ -384,8 +462,17 @@ def run_agentic_conversation(
384
462
  total_clarification_turns += 1
385
463
  current_message = _get_sim_user_response(response_text, resolved_turn, resolved_expected)
386
464
 
387
- activated = _activated_skills(all_tool_calls)
388
- skill_routing = turn.expected_skill in activated if activated else False
465
+ # `declared` is what THIS turn's final set_skills call asked for (None when it
466
+ # made no call); `active_skills` is what is actually active during the turn. No
467
+ # call carries the previous set over; a call replaces it outright, including
468
+ # when it declares an empty list. See active_skills' declaration above.
469
+ declarations = _set_skills_declarations(all_tool_calls)
470
+ for declaration in declarations:
471
+ ever_declared_skills.update(declaration)
472
+ declared = declarations[-1] if declarations else None
473
+ if declared is not None:
474
+ active_skills = set(declared)
475
+ skill_routing = turn.expected_skill in active_skills
389
476
  output_present = _check_output_present(resolved_turn, final_result) if final_result else False
390
477
  output_correct = (
391
478
  _check_output_correct(resolved_turn, final_result) if (final_result and output_present) else None
@@ -409,7 +496,8 @@ def run_agentic_conversation(
409
496
  skill_routing=skill_routing,
410
497
  output_present=output_present,
411
498
  no_error=True, # SDK raises on errors; reaching here means no critical error.
412
- activated_skills=activated,
499
+ activated_skills=declared or [],
500
+ active_skills=sorted(active_skills),
413
501
  clarification_turns_used=clarification_turns,
414
502
  output_correct=output_correct,
415
503
  )
@@ -422,8 +510,10 @@ def run_agentic_conversation(
422
510
  _delete_metric(sdk, workspace_id, metric_id)
423
511
  client.close()
424
512
 
425
- activated_all = {skill for tr in turn_results for skill in tr.activated_skills}
426
- full_skill_coverage = set(fixture.expected_skills).issubset(activated_all)
513
+ # Not derived from TurnResult.activated_skills: that field carries only each turn's FINAL
514
+ # declaration, so a skill replaced within its own turn is absent from it despite having
515
+ # been activated. See ever_declared_skills' declaration above.
516
+ full_skill_coverage = set(fixture.expected_skills).issubset(ever_declared_skills)
427
517
  conversation_success = all(tr.skill_success for tr in turn_results)
428
518
 
429
519
  return ConversationResult(
@@ -443,17 +533,7 @@ def _conversation_detail(result: ConversationResult) -> dict:
443
533
  return {
444
534
  "full_skill_coverage": result.full_skill_coverage,
445
535
  "total_clarification_turns": result.total_clarification_turns,
446
- "turns": [
447
- {
448
- "turn_id": tr.turn_id,
449
- "expected_skill": tr.expected_skill,
450
- "skill_routing": tr.skill_routing,
451
- "output_present": tr.output_present,
452
- "output_correct": tr.output_correct,
453
- "activated_skills": tr.activated_skills,
454
- }
455
- for tr in result.turn_results
456
- ],
536
+ "turns": [tr.detail() for tr in result.turn_results],
457
537
  "latency_breakdown": build_latency_breakdown(result.tool_call_events, result.reasoning_step_events),
458
538
  }
459
539