gooddata-eval 1.74.1.dev1__py3-none-any.whl → 1.74.1.dev2__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- gooddata_eval/core/agentic/_catalog.py +41 -0
- gooddata_eval/core/agentic/alert_skill.py +175 -23
- gooddata_eval/core/agentic/conversation.py +119 -39
- gooddata_eval/core/agentic/general_question.py +14 -2
- gooddata_eval/core/agentic/guardrail.py +2 -1
- gooddata_eval/core/agentic/kda_skill.py +34 -2
- gooddata_eval/core/agentic/metric_skill.py +16 -78
- gooddata_eval/core/agentic/search_tool.py +14 -1
- gooddata_eval/core/agentic/visualization.py +11 -15
- gooddata_eval/core/chat/render.py +47 -0
- gooddata_eval/core/chat/sse_client.py +34 -2
- gooddata_eval/core/evaluators/__init__.py +13 -5
- gooddata_eval/core/evaluators/_maql.py +103 -0
- gooddata_eval/core/evaluators/_text_utils.py +3 -4
- gooddata_eval/core/evaluators/alert_skill.py +11 -2
- gooddata_eval/core/evaluators/general_question.py +4 -0
- gooddata_eval/core/evaluators/metric_skill.py +16 -4
- gooddata_eval/core/models.py +42 -0
- {gooddata_eval-1.74.1.dev1.dist-info → gooddata_eval-1.74.1.dev2.dist-info}/METADATA +2 -2
- {gooddata_eval-1.74.1.dev1.dist-info → gooddata_eval-1.74.1.dev2.dist-info}/RECORD +23 -21
- {gooddata_eval-1.74.1.dev1.dist-info → gooddata_eval-1.74.1.dev2.dist-info}/WHEEL +0 -0
- {gooddata_eval-1.74.1.dev1.dist-info → gooddata_eval-1.74.1.dev2.dist-info}/entry_points.txt +0 -0
- {gooddata_eval-1.74.1.dev1.dist-info → gooddata_eval-1.74.1.dev2.dist-info}/licenses/LICENSE.txt +0 -0
|
@@ -2,6 +2,41 @@
|
|
|
2
2
|
from __future__ import annotations
|
|
3
3
|
|
|
4
4
|
from dataclasses import dataclass, field
|
|
5
|
+
from enum import Enum
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
class AnomalyDetectionGranularity(str, Enum):
|
|
9
|
+
"""Detection intervals an anomaly alert accepts.
|
|
10
|
+
|
|
11
|
+
Mirrors gen-ai's enum of the same name; `StrEnum` is unavailable on the 3.10 floor, so
|
|
12
|
+
the `str` mixin carries the comparison against the raw tool argument.
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
HOUR = "HOUR"
|
|
16
|
+
DAY = "DAY"
|
|
17
|
+
WEEK = "WEEK"
|
|
18
|
+
MONTH = "MONTH"
|
|
19
|
+
QUARTER = "QUARTER"
|
|
20
|
+
YEAR = "YEAR"
|
|
21
|
+
|
|
22
|
+
@classmethod
|
|
23
|
+
def parse(cls, value: object) -> AnomalyDetectionGranularity | None:
|
|
24
|
+
"""Coerce a fixture value, or None when the fixture states none.
|
|
25
|
+
|
|
26
|
+
Raises ValueError on an unknown interval: fixtures are hand-written, and a typo has
|
|
27
|
+
to fail before the run spends an API call rather than score the item against an
|
|
28
|
+
interval the product cannot produce.
|
|
29
|
+
"""
|
|
30
|
+
if value is None:
|
|
31
|
+
return None
|
|
32
|
+
candidate = str(value).strip().upper()
|
|
33
|
+
if not candidate:
|
|
34
|
+
return None
|
|
35
|
+
try:
|
|
36
|
+
return cls(candidate)
|
|
37
|
+
except ValueError:
|
|
38
|
+
expected = ", ".join(member.value for member in cls)
|
|
39
|
+
raise ValueError(f"Invalid granularity {value!r}; expected one of {expected}.") from None
|
|
5
40
|
|
|
6
41
|
|
|
7
42
|
@dataclass
|
|
@@ -28,6 +63,10 @@ class CatalogMetricAlert:
|
|
|
28
63
|
"""List of recipient email addresses."""
|
|
29
64
|
filters: list | str | None = None
|
|
30
65
|
"""Attribute filters applied to the alert condition."""
|
|
66
|
+
attributes: list | None = None
|
|
67
|
+
"""Expected group-by attributes; ``None`` means the fixture states no expectation."""
|
|
68
|
+
granularity: AnomalyDetectionGranularity | None = None
|
|
69
|
+
"""Detection interval for an ANOMALY alert (DAY/WEEK/MONTH/...). Not a date filter."""
|
|
31
70
|
|
|
32
71
|
@classmethod
|
|
33
72
|
def from_dict(cls, d: dict) -> CatalogMetricAlert:
|
|
@@ -46,4 +85,6 @@ class CatalogMetricAlert:
|
|
|
46
85
|
metric_id=d.get("metric_id"),
|
|
47
86
|
recipients=recipients,
|
|
48
87
|
filters=d.get("filters"),
|
|
88
|
+
attributes=d.get("attributes"),
|
|
89
|
+
granularity=AnomalyDetectionGranularity.parse(d.get("granularity")),
|
|
49
90
|
)
|
|
@@ -11,7 +11,7 @@ from typing import Any
|
|
|
11
11
|
|
|
12
12
|
from gooddata_sdk import GoodDataSdk
|
|
13
13
|
|
|
14
|
-
from gooddata_eval.core.agentic._catalog import CatalogMetricAlert
|
|
14
|
+
from gooddata_eval.core.agentic._catalog import AnomalyDetectionGranularity, CatalogMetricAlert
|
|
15
15
|
from gooddata_eval.core.agentic._trace_linker import (
|
|
16
16
|
RunIdentity,
|
|
17
17
|
RunTraceContext,
|
|
@@ -21,6 +21,7 @@ from gooddata_eval.core.agentic._trace_linker import (
|
|
|
21
21
|
submit_trace_scoring,
|
|
22
22
|
utc_now,
|
|
23
23
|
)
|
|
24
|
+
from gooddata_eval.core.chat.render import render_answer_text
|
|
24
25
|
from gooddata_eval.core.chat.sse_client import ChatClient
|
|
25
26
|
from gooddata_eval.core.config import ReasoningEffort
|
|
26
27
|
from gooddata_eval.core.models import (
|
|
@@ -29,6 +30,7 @@ from gooddata_eval.core.models import (
|
|
|
29
30
|
ReasoningStepEvent,
|
|
30
31
|
ToolCallEvent,
|
|
31
32
|
build_latency_breakdown,
|
|
33
|
+
shift_and_index_events,
|
|
32
34
|
)
|
|
33
35
|
|
|
34
36
|
try:
|
|
@@ -126,6 +128,94 @@ def _check_filters(expected: CatalogMetricAlert, actual_args: dict) -> bool:
|
|
|
126
128
|
return _deep_subset(exp_filters, act_filters)
|
|
127
129
|
|
|
128
130
|
|
|
131
|
+
def _attribute_label_ids(items: list, *, side: str) -> list[str]:
|
|
132
|
+
"""Canonicalise group-by entries to bare label ids, whatever spelling they arrive in.
|
|
133
|
+
|
|
134
|
+
The two sides of the comparison speak different vocabularies for the same grouping.
|
|
135
|
+
Fixtures author the AAC tool-input form, ``{"using": "label/x"}``; ``create_metric_alert``
|
|
136
|
+
receives the resolved AFM form, ``{"localIdentifier": "a0", "label": {"identifier":
|
|
137
|
+
{"id": "x", "type": "label"}}}``, forwarded verbatim from ``prepare_metric_alert_proposal``.
|
|
138
|
+
Identity is therefore the only thing they can be compared on.
|
|
139
|
+
|
|
140
|
+
A shape not listed here, or a URI prefix other than ``label/``, raises: ``label/x`` and
|
|
141
|
+
``attribute/x`` are different objects, and an unknown spelling must fail loudly rather
|
|
142
|
+
than quietly compare unequal.
|
|
143
|
+
"""
|
|
144
|
+
if not isinstance(items, list):
|
|
145
|
+
raise ValueError(f"Unrecognised {side} group-by attributes, expected a list: {items!r}")
|
|
146
|
+
ids: list[str] = []
|
|
147
|
+
for item in items:
|
|
148
|
+
raw: object = None
|
|
149
|
+
if isinstance(item, str):
|
|
150
|
+
raw = item
|
|
151
|
+
elif isinstance(item, dict):
|
|
152
|
+
label = item.get("label")
|
|
153
|
+
identifier = item.get("identifier")
|
|
154
|
+
if isinstance(item.get("using"), str):
|
|
155
|
+
raw = item["using"]
|
|
156
|
+
elif isinstance(label, dict) and isinstance(label.get("identifier"), dict):
|
|
157
|
+
raw = label["identifier"].get("id")
|
|
158
|
+
elif isinstance(identifier, dict):
|
|
159
|
+
raw = identifier.get("id")
|
|
160
|
+
if not isinstance(raw, str) or not raw:
|
|
161
|
+
raise ValueError(f"Unrecognised {side} group-by attribute entry: {item!r}")
|
|
162
|
+
prefix, slash, rest = raw.partition("/")
|
|
163
|
+
if not slash:
|
|
164
|
+
ids.append(raw)
|
|
165
|
+
elif prefix == "label" and rest:
|
|
166
|
+
ids.append(rest)
|
|
167
|
+
else:
|
|
168
|
+
raise ValueError(f"Unrecognised {side} group-by attribute reference: {raw!r}")
|
|
169
|
+
return ids
|
|
170
|
+
|
|
171
|
+
|
|
172
|
+
def _check_attributes(expected: CatalogMetricAlert, actual_args: dict) -> bool:
|
|
173
|
+
"""Compare group-by identity only.
|
|
174
|
+
|
|
175
|
+
Per-entry properties — ``showAllValues``, the converter-assigned ``localIdentifier`` —
|
|
176
|
+
are deliberately not asserted, and the comparison is a multiset so entry order does not
|
|
177
|
+
matter.
|
|
178
|
+
"""
|
|
179
|
+
exp_attributes = expected.attributes
|
|
180
|
+
if exp_attributes is None:
|
|
181
|
+
return True
|
|
182
|
+
act_attributes = actual_args.get("attributes")
|
|
183
|
+
if act_attributes is None:
|
|
184
|
+
# Arguments are raw `json.loads` output, where an unset nullable argument arrives as
|
|
185
|
+
# null rather than absent. Both spellings of "no grouping" have to land on [], which
|
|
186
|
+
# is why this is not `actual_args.get("attributes", [])`.
|
|
187
|
+
act_attributes = []
|
|
188
|
+
elif not isinstance(act_attributes, list):
|
|
189
|
+
# An argument that is not a list of groupings is the agent answering wrongly, so it
|
|
190
|
+
# scores False. Raising instead would make the runner record an ERROR, and errored
|
|
191
|
+
# items are excluded from the failure count — a malformed answer must not rank above
|
|
192
|
+
# a merely wrong one. An unreadable *entry* still raises, in `_attribute_label_ids`:
|
|
193
|
+
# entries are typed at the tool boundary, so the plausible cause there is the wire
|
|
194
|
+
# format moving, which has to be unmissable.
|
|
195
|
+
return False
|
|
196
|
+
if not exp_attributes:
|
|
197
|
+
return not act_attributes
|
|
198
|
+
exp_ids = sorted(_attribute_label_ids(exp_attributes, side="expected"))
|
|
199
|
+
act_ids = sorted(_attribute_label_ids(act_attributes, side="actual"))
|
|
200
|
+
return exp_ids == act_ids
|
|
201
|
+
|
|
202
|
+
|
|
203
|
+
def _check_granularity(expected: CatalogMetricAlert, actual_args: dict) -> bool:
|
|
204
|
+
"""Compare the ANOMALY detection interval when the fixture states one.
|
|
205
|
+
|
|
206
|
+
``None`` means unasserted, mirroring ``attributes``: only the ANOMALY items carry a
|
|
207
|
+
``Granularity``, and every other item must stay unaffected. The expectation is already
|
|
208
|
+
canonical by the time it lands here; the tool argument is a raw string, so only that
|
|
209
|
+
side needs folding.
|
|
210
|
+
"""
|
|
211
|
+
if expected.granularity is None:
|
|
212
|
+
return True
|
|
213
|
+
actual = actual_args.get("granularity")
|
|
214
|
+
if not actual:
|
|
215
|
+
return False
|
|
216
|
+
return str(actual).strip().upper() == expected.granularity.value
|
|
217
|
+
|
|
218
|
+
|
|
129
219
|
def _check_metric(expected: CatalogMetricAlert, actual_args: dict) -> bool:
|
|
130
220
|
if not expected.metric_id:
|
|
131
221
|
return True
|
|
@@ -257,20 +347,38 @@ def generate_simulated_alert_response(
|
|
|
257
347
|
)
|
|
258
348
|
elif filters == []:
|
|
259
349
|
filters_rule = (
|
|
260
|
-
"5. Your alert must have NO filters and NO date/time window — it evaluates
|
|
261
|
-
"If the agent asks which time period each check should cover, or offers a
|
|
262
|
-
"'last Day / Week / Month', do NOT pick one: reply that you want no date
|
|
263
|
-
"
|
|
264
|
-
"instruction the goal did not ask for.\n"
|
|
350
|
+
"5. Your alert must have NO filters and NO date/time window on the metric — it evaluates "
|
|
351
|
+
"over all time. If the agent asks which time period each check should cover, or offers a "
|
|
352
|
+
"choice such as 'last Day / Week / Month', do NOT pick one: reply that you want no date "
|
|
353
|
+
"filter at all, all time.\n"
|
|
265
354
|
)
|
|
266
355
|
else:
|
|
267
356
|
filters_rule = (
|
|
268
357
|
"5. Ask only for the filters your original request implies — do not invent an evaluation "
|
|
269
|
-
"period
|
|
358
|
+
"period or date window that was not requested. If the agent offers a choice "
|
|
270
359
|
"such as 'last Day / Week / Month' that your request never mentioned, say you do not want "
|
|
271
360
|
"a date window.\n"
|
|
272
361
|
)
|
|
273
362
|
|
|
363
|
+
if operator == "ANOMALY":
|
|
364
|
+
# The fallback keeps the conversation alive when the fixture names no interval -- an
|
|
365
|
+
# anomaly alert cannot be created without one. It is deliberately NOT mirrored into
|
|
366
|
+
# `expected.granularity`: `_check_granularity` asserts only what the fixture stated,
|
|
367
|
+
# and scoring an item against an interval it never asked for is the defect this rule
|
|
368
|
+
# exists to undo.
|
|
369
|
+
granularity = (expected.granularity or AnomalyDetectionGranularity.DAY).value
|
|
370
|
+
anomaly_rule = (
|
|
371
|
+
"7. This is an ANOMALY alert. Anomaly detection REQUIRES a time granularity, and that "
|
|
372
|
+
f"granularity is NOT a date filter. State it in your first reply and repeat it whenever "
|
|
373
|
+
f"asked: use {granularity} granularity. Rule 5 constrains filters on the metric only — it "
|
|
374
|
+
"never applies to this detection interval, so never refuse to give one.\n"
|
|
375
|
+
)
|
|
376
|
+
else:
|
|
377
|
+
anomaly_rule = (
|
|
378
|
+
"7. Do not invent an evaluation period, a granularity or an 'evaluate each run on a X "
|
|
379
|
+
"basis' instruction your goal never asked for.\n"
|
|
380
|
+
)
|
|
381
|
+
|
|
274
382
|
original_request = f'Your original request to the agent was: "{question}"\n' if question else ""
|
|
275
383
|
|
|
276
384
|
system_prompt = (
|
|
@@ -294,8 +402,7 @@ def generate_simulated_alert_response(
|
|
|
294
402
|
" Do not wait for the agent to ask — state it alongside the metric and condition answers.\n"
|
|
295
403
|
+ filters_rule
|
|
296
404
|
+ f"6. Proactively state how often you want to be alerted in your first reply: {trigger_request}. "
|
|
297
|
-
" Repeat it if the agent proposes a different cadence.\n"
|
|
298
|
-
"Reply concisely and directly."
|
|
405
|
+
" Repeat it if the agent proposes a different cadence.\n" + anomaly_rule + "Reply concisely and directly."
|
|
299
406
|
)
|
|
300
407
|
|
|
301
408
|
messages: list = [{"role": "system", "content": system_prompt}]
|
|
@@ -334,6 +441,8 @@ class AlertEvaluation:
|
|
|
334
441
|
filters_correct: bool
|
|
335
442
|
metric_correct: bool
|
|
336
443
|
recipients_correct: bool
|
|
444
|
+
attributes_correct: bool = True
|
|
445
|
+
granularity_correct: bool = True
|
|
337
446
|
|
|
338
447
|
@property
|
|
339
448
|
def strict_pass(self) -> bool:
|
|
@@ -346,6 +455,8 @@ class AlertEvaluation:
|
|
|
346
455
|
self.filters_correct,
|
|
347
456
|
self.metric_correct,
|
|
348
457
|
self.recipients_correct,
|
|
458
|
+
self.attributes_correct,
|
|
459
|
+
self.granularity_correct,
|
|
349
460
|
]
|
|
350
461
|
)
|
|
351
462
|
|
|
@@ -409,6 +520,35 @@ def _normalize_expected_filters(expected: dict) -> list | str | None:
|
|
|
409
520
|
return None
|
|
410
521
|
|
|
411
522
|
|
|
523
|
+
_NO_GROUPING_MARKERS = ("none", "no grouping")
|
|
524
|
+
|
|
525
|
+
|
|
526
|
+
def _normalize_expected_attributes(expected: dict) -> list | None:
|
|
527
|
+
"""
|
|
528
|
+
* ``Attributes`` list -> that list (exact expectation)
|
|
529
|
+
* "None" / "no grouping" -> ``[]`` (stated: no group-by; extras fail)
|
|
530
|
+
* absent, or other prose -> ``None`` (unstated; grouping not asserted)
|
|
531
|
+
|
|
532
|
+
A date narrows an alert as a group-by as well as a filter, and a group-by makes it fire
|
|
533
|
+
per period value instead of on the latest one — so ``[]`` has to be expressible separately
|
|
534
|
+
from "absent", exactly as it is for ``filters``.
|
|
535
|
+
|
|
536
|
+
The simulated user is told nothing about groupings, so a non-empty expectation requires the
|
|
537
|
+
item's own question to request that grouping; ``[]`` needs no such support, because the
|
|
538
|
+
simulated user does not invent a grouping and the check verifies it did not.
|
|
539
|
+
"""
|
|
540
|
+
attributes = _case_insensitive_get(expected, "attributes")
|
|
541
|
+
if isinstance(attributes, list):
|
|
542
|
+
# Validated here so a malformed fixture fails before the run spends an API call.
|
|
543
|
+
_attribute_label_ids(attributes, side="expected")
|
|
544
|
+
return attributes
|
|
545
|
+
if attributes is None:
|
|
546
|
+
return None
|
|
547
|
+
if isinstance(attributes, str):
|
|
548
|
+
return [] if any(kw in attributes.lower() for kw in _NO_GROUPING_MARKERS) else None
|
|
549
|
+
raise ValueError(f"Attributes expectation must be a list or a display string, got {type(attributes).__name__}")
|
|
550
|
+
|
|
551
|
+
|
|
412
552
|
def _normalize_expected_output(expected: dict) -> CatalogMetricAlert:
|
|
413
553
|
"""Parse expected_output dict into CatalogMetricAlert, accepting display-format or internal-format keys."""
|
|
414
554
|
operator = _case_insensitive_get(expected, "operator") or "GREATER_THAN"
|
|
@@ -433,6 +573,11 @@ def _normalize_expected_output(expected: dict) -> CatalogMetricAlert:
|
|
|
433
573
|
recipients = list(raw_recip)
|
|
434
574
|
|
|
435
575
|
filters = _normalize_expected_filters(expected)
|
|
576
|
+
attributes = _normalize_expected_attributes(expected)
|
|
577
|
+
|
|
578
|
+
granularity = AnomalyDetectionGranularity.parse(
|
|
579
|
+
_case_insensitive_get(expected, "granularity", "detection granularity")
|
|
580
|
+
)
|
|
436
581
|
|
|
437
582
|
return CatalogMetricAlert(
|
|
438
583
|
operator=operator,
|
|
@@ -443,6 +588,8 @@ def _normalize_expected_output(expected: dict) -> CatalogMetricAlert:
|
|
|
443
588
|
metric_id=metric_id,
|
|
444
589
|
recipients=recipients,
|
|
445
590
|
filters=filters,
|
|
591
|
+
attributes=attributes,
|
|
592
|
+
granularity=granularity,
|
|
446
593
|
)
|
|
447
594
|
|
|
448
595
|
|
|
@@ -526,21 +673,14 @@ def run_agentic_alert_skill(
|
|
|
526
673
|
chat_result = client.send_message(conv_id, current_question)
|
|
527
674
|
reasoning_steps.extend(chat_result.reasoning_steps or [])
|
|
528
675
|
response_id = chat_result.response_id or response_id
|
|
529
|
-
|
|
530
|
-
|
|
531
|
-
|
|
532
|
-
|
|
533
|
-
|
|
534
|
-
|
|
535
|
-
tc.index += tool_index_offset
|
|
536
|
-
for rs in chat_result.reasoning_step_events or []:
|
|
537
|
-
rs.ts += turn_offset
|
|
538
|
-
rs.index += reasoning_index_offset
|
|
676
|
+
turn_offset, tool_index_offset, reasoning_index_offset = shift_and_index_events(
|
|
677
|
+
chat_result,
|
|
678
|
+
turn_offset=turn_offset,
|
|
679
|
+
tool_index_offset=tool_index_offset,
|
|
680
|
+
reasoning_index_offset=reasoning_index_offset,
|
|
681
|
+
)
|
|
539
682
|
all_tool_call_events.extend(chat_result.tool_call_events or [])
|
|
540
683
|
all_reasoning_step_events.extend(chat_result.reasoning_step_events or [])
|
|
541
|
-
tool_index_offset += len(chat_result.tool_call_events or [])
|
|
542
|
-
reasoning_index_offset += len(chat_result.reasoning_step_events or [])
|
|
543
|
-
turn_offset += chat_result.turn_wall_clock_sec or 0.0
|
|
544
684
|
alert_id, actual_args, tool_called = _extract_alert_call(chat_result.tool_call_events or [])
|
|
545
685
|
if tool_called:
|
|
546
686
|
alert_id_to_delete = alert_id
|
|
@@ -548,6 +688,8 @@ def run_agentic_alert_skill(
|
|
|
548
688
|
response_text = (chat_result.text_response or "").strip()
|
|
549
689
|
if not response_text and chat_result.alert_proposals:
|
|
550
690
|
response_text = render_alert_proposal(chat_result.alert_proposals[-1])
|
|
691
|
+
if not response_text:
|
|
692
|
+
response_text = render_answer_text(chat_result)
|
|
551
693
|
# Stop if agent gave a completely empty response (stuck)
|
|
552
694
|
if not response_text and not chat_result.tool_call_events:
|
|
553
695
|
break
|
|
@@ -570,6 +712,8 @@ def run_agentic_alert_skill(
|
|
|
570
712
|
filters_correct=tool_called and _check_filters(expected, actual_args),
|
|
571
713
|
metric_correct=tool_called and _check_metric(expected, actual_args),
|
|
572
714
|
recipients_correct=tool_called and _check_recipients(expected, actual_args, sdk=sdk),
|
|
715
|
+
attributes_correct=tool_called and _check_attributes(expected, actual_args),
|
|
716
|
+
granularity_correct=tool_called and _check_granularity(expected, actual_args),
|
|
573
717
|
)
|
|
574
718
|
return AlertRunResult(
|
|
575
719
|
conversation_id=conv_id,
|
|
@@ -615,6 +759,8 @@ def run_agentic_alert_skill(
|
|
|
615
759
|
r.eval.filters_correct,
|
|
616
760
|
r.eval.metric_correct,
|
|
617
761
|
r.eval.recipients_correct,
|
|
762
|
+
r.eval.attributes_correct,
|
|
763
|
+
r.eval.granularity_correct,
|
|
618
764
|
]
|
|
619
765
|
),
|
|
620
766
|
)
|
|
@@ -689,6 +835,8 @@ def evaluate_agentic_alert_skill(
|
|
|
689
835
|
"filters_correct": ev.filters_correct,
|
|
690
836
|
"metric_correct": ev.metric_correct,
|
|
691
837
|
"recipients_correct": ev.recipients_correct,
|
|
838
|
+
"attributes_correct": ev.attributes_correct,
|
|
839
|
+
"granularity_correct": ev.granularity_correct,
|
|
692
840
|
}
|
|
693
841
|
with ctx.observe(pt, run_idx) as tid:
|
|
694
842
|
for score_name, value in strict_checks.items():
|
|
@@ -735,6 +883,8 @@ def evaluate_agentic_alert_skill(
|
|
|
735
883
|
"filters_correct": ev.filters_correct,
|
|
736
884
|
"metric_correct": ev.metric_correct,
|
|
737
885
|
"recipients_correct": ev.recipients_correct,
|
|
886
|
+
"attributes_correct": ev.attributes_correct,
|
|
887
|
+
"granularity_correct": ev.granularity_correct,
|
|
738
888
|
"actual_alert_arguments": best.actual_alert_arguments,
|
|
739
889
|
"latency_breakdown": build_latency_breakdown(best.tool_call_events, best.reasoning_step_events),
|
|
740
890
|
}
|
|
@@ -745,7 +895,9 @@ def evaluate_agentic_alert_skill(
|
|
|
745
895
|
f"alert_created={ev.alert_created}, operator_correct={ev.operator_correct}, "
|
|
746
896
|
f"threshold_correct={ev.threshold_correct}, trigger_correct={ev.trigger_correct}, "
|
|
747
897
|
f"filters_correct={ev.filters_correct}, metric_correct={ev.metric_correct}, "
|
|
748
|
-
f"recipients_correct={ev.recipients_correct}
|
|
898
|
+
f"recipients_correct={ev.recipients_correct}, "
|
|
899
|
+
f"attributes_correct={ev.attributes_correct}, "
|
|
900
|
+
f"granularity_correct={ev.granularity_correct}. "
|
|
749
901
|
f"Actual args: {best.actual_alert_arguments}"
|
|
750
902
|
)
|
|
751
903
|
exc.reasoning_steps = best.reasoning_steps
|
|
@@ -6,10 +6,10 @@ from __future__ import annotations
|
|
|
6
6
|
import json
|
|
7
7
|
import re
|
|
8
8
|
from dataclasses import dataclass, field
|
|
9
|
-
from typing import Literal
|
|
9
|
+
from typing import ClassVar, Literal
|
|
10
10
|
|
|
11
11
|
from gooddata_sdk import GoodDataSdk
|
|
12
|
-
from pydantic import BaseModel
|
|
12
|
+
from pydantic import BaseModel, Field
|
|
13
13
|
|
|
14
14
|
from gooddata_eval.core.agentic._trace_linker import (
|
|
15
15
|
RunIdentity,
|
|
@@ -22,6 +22,7 @@ from gooddata_eval.core.agentic._trace_linker import (
|
|
|
22
22
|
)
|
|
23
23
|
from gooddata_eval.core.agentic.alert_skill import render_alert_proposal
|
|
24
24
|
from gooddata_eval.core.agentic.metric_skill import _delete_metric, _extract_created_metric_ids, _extract_metric_result
|
|
25
|
+
from gooddata_eval.core.chat.render import render_answer_text
|
|
25
26
|
from gooddata_eval.core.chat.sse_client import ChatClient
|
|
26
27
|
from gooddata_eval.core.config import ReasoningEffort
|
|
27
28
|
from gooddata_eval.core.models import (
|
|
@@ -31,6 +32,7 @@ from gooddata_eval.core.models import (
|
|
|
31
32
|
ReasoningStepEvent,
|
|
32
33
|
ToolCallEvent,
|
|
33
34
|
build_latency_breakdown,
|
|
35
|
+
shift_and_index_events,
|
|
34
36
|
)
|
|
35
37
|
from gooddata_eval.core.scoring import (
|
|
36
38
|
check_filters,
|
|
@@ -65,7 +67,21 @@ class ConversationFixture(BaseModel):
|
|
|
65
67
|
|
|
66
68
|
|
|
67
69
|
class TurnResult(BaseModel):
|
|
68
|
-
"""Evaluation result for a single conversation turn.
|
|
70
|
+
"""Evaluation result for a single conversation turn.
|
|
71
|
+
|
|
72
|
+
The two skill fields measure DIFFERENT SCOPES, so a turn can legitimately report
|
|
73
|
+
``skill_routing=True`` with an empty ``activated_skills``:
|
|
74
|
+
|
|
75
|
+
- ``activated_skills`` -- what THIS turn's own ``set_skills`` call declared. Empty
|
|
76
|
+
whenever the agent reused an already-active skill without re-declaring it.
|
|
77
|
+
- ``active_skills`` -- what was actually active DURING this turn: the last declared
|
|
78
|
+
set, carried over on turns that declare nothing. This is the set ``skill_routing``
|
|
79
|
+
is judged against, so a report never has to infer it.
|
|
80
|
+
- ``skill_routing`` -- whether ``expected_skill`` appears in ``active_skills``.
|
|
81
|
+
|
|
82
|
+
``skill_routing=True`` with ``activated_skills=[]`` is the reused-skill case, not a
|
|
83
|
+
scoring bug -- ``active_skills`` shows where the credit came from.
|
|
84
|
+
"""
|
|
69
85
|
|
|
70
86
|
turn_id: str
|
|
71
87
|
expected_skill: str
|
|
@@ -73,6 +89,8 @@ class TurnResult(BaseModel):
|
|
|
73
89
|
output_present: bool
|
|
74
90
|
no_error: bool
|
|
75
91
|
activated_skills: list[str]
|
|
92
|
+
# Sorted for stable output: the source is a set, whose iteration order is not.
|
|
93
|
+
active_skills: list[str] = Field(default_factory=list)
|
|
76
94
|
clarification_turns_used: int = 0
|
|
77
95
|
output_correct: bool | None = None
|
|
78
96
|
|
|
@@ -80,6 +98,26 @@ class TurnResult(BaseModel):
|
|
|
80
98
|
def skill_success(self) -> bool:
|
|
81
99
|
return self.skill_routing and self.output_present and self.no_error
|
|
82
100
|
|
|
101
|
+
# Reported per turn in detail["turns"]. A name listed here that no longer exists on the
|
|
102
|
+
# model raises rather than silently emitting a stale key, which a hand-written dict
|
|
103
|
+
# literal of the same fields would not -- and model_dump deep-copies activated_skills,
|
|
104
|
+
# so a caller mutating the returned dict cannot reach back into this TurnResult.
|
|
105
|
+
_DETAIL_FIELDS: ClassVar[set[str]] = {
|
|
106
|
+
"turn_id",
|
|
107
|
+
"expected_skill",
|
|
108
|
+
"skill_routing",
|
|
109
|
+
"output_present",
|
|
110
|
+
"output_correct",
|
|
111
|
+
"activated_skills",
|
|
112
|
+
# What skill_routing was judged against -- without it, a turn showing
|
|
113
|
+
# skill_routing=True and activated_skills=[] looks like a scoring bug.
|
|
114
|
+
"active_skills",
|
|
115
|
+
}
|
|
116
|
+
|
|
117
|
+
def detail(self) -> dict:
|
|
118
|
+
"""The subset of this result reported in detail["turns"] for one conversation turn."""
|
|
119
|
+
return self.model_dump(include=self._DETAIL_FIELDS)
|
|
120
|
+
|
|
83
121
|
|
|
84
122
|
def _resolve_refs(
|
|
85
123
|
expected_output: dict | None,
|
|
@@ -117,15 +155,42 @@ def _resolve_refs(
|
|
|
117
155
|
return json.loads(resolved_raw)
|
|
118
156
|
|
|
119
157
|
|
|
120
|
-
def
|
|
121
|
-
"""
|
|
122
|
-
|
|
158
|
+
def _set_skills_declarations(tool_call_events: list[ToolCallEvent]) -> list[list[str]]:
|
|
159
|
+
"""Every set_skills declaration in these events, in call order.
|
|
160
|
+
|
|
161
|
+
`skill_names` is the key the tool declares; `skills` is a legacy spelling kept as a
|
|
162
|
+
fallback. A call carrying neither is treated as declaring an empty list, which is what
|
|
163
|
+
the platform would do with one.
|
|
164
|
+
"""
|
|
165
|
+
declarations: list[list[str]] = []
|
|
123
166
|
for tc in tool_call_events:
|
|
124
167
|
if tc.function_name != "set_skills":
|
|
125
168
|
continue
|
|
126
169
|
args = tc.parsed_arguments() or {}
|
|
127
|
-
|
|
128
|
-
|
|
170
|
+
names = args.get("skill_names")
|
|
171
|
+
if names is None:
|
|
172
|
+
names = args.get("skills")
|
|
173
|
+
declarations.append(list(names or []))
|
|
174
|
+
return declarations
|
|
175
|
+
|
|
176
|
+
|
|
177
|
+
def _final_skill_declaration(tool_call_events: list[ToolCallEvent]) -> list[str] | None:
|
|
178
|
+
"""The skill list from the LAST set_skills call, or None when there was no call.
|
|
179
|
+
|
|
180
|
+
set_skills replaces the active set, so when a turn issues several calls -- which it can,
|
|
181
|
+
since these events span every clarification sub-turn within one logical turn -- only the
|
|
182
|
+
final one describes the resulting state. Merging them would credit a skill that an
|
|
183
|
+
earlier call declared and a later one dropped.
|
|
184
|
+
|
|
185
|
+
An empty list is a real declaration: it deactivates everything. That has to stay
|
|
186
|
+
distinguishable from ``None`` ("no call at all"), which leaves the previous turn's set
|
|
187
|
+
untouched -- hence the Optional rather than just an empty list for both.
|
|
188
|
+
|
|
189
|
+
This answers "what is active NOW". For "was this skill ever exercised" -- what
|
|
190
|
+
full_skill_coverage asks -- use every declaration, not just the last one.
|
|
191
|
+
"""
|
|
192
|
+
declarations = _set_skills_declarations(tool_call_events)
|
|
193
|
+
return declarations[-1] if declarations else None
|
|
129
194
|
|
|
130
195
|
|
|
131
196
|
def _check_output_present(turn: TurnDefinition, chat_result: ChatResult) -> bool:
|
|
@@ -150,7 +215,7 @@ def _check_output_correct(turn: TurnDefinition, chat_result: ChatResult) -> bool
|
|
|
150
215
|
|
|
151
216
|
Returns None when expected_output is absent (presence check only).
|
|
152
217
|
"""
|
|
153
|
-
from gooddata_eval.core.
|
|
218
|
+
from gooddata_eval.core.evaluators._maql import normalize_maql # noqa: PLC0415
|
|
154
219
|
|
|
155
220
|
otype = turn.expected_output_type
|
|
156
221
|
expected = turn.expected_output
|
|
@@ -192,7 +257,7 @@ def _check_output_correct(turn: TurnDefinition, chat_result: ChatResult) -> bool
|
|
|
192
257
|
metric_result = _extract_metric_result(chat_result.tool_call_events or [])
|
|
193
258
|
if not metric_result:
|
|
194
259
|
return False
|
|
195
|
-
return
|
|
260
|
+
return normalize_maql(metric_result.get("maql", "")) == normalize_maql(expected.get("maql", ""))
|
|
196
261
|
|
|
197
262
|
return None
|
|
198
263
|
|
|
@@ -310,6 +375,20 @@ def run_agentic_conversation(
|
|
|
310
375
|
response_id: str | None = None
|
|
311
376
|
conversation_tool_call_events: list[ToolCallEvent] = []
|
|
312
377
|
conversation_reasoning_step_events: list[ReasoningStepEvent] = []
|
|
378
|
+
# The skills active right now, mirroring the platform's own state machine: set_skills
|
|
379
|
+
# REPLACES the active set rather than adding to it -- verified against the gen-ai
|
|
380
|
+
# service's skill registry, and stated in the tool's own description. So a turn that
|
|
381
|
+
# issues no set_skills call inherits the previous turn's set unchanged, while a turn
|
|
382
|
+
# that does issue one drops whatever it left out. Tracking this as a running UNION
|
|
383
|
+
# would credit a skill a later call had already switched off.
|
|
384
|
+
active_skills: set[str] = set()
|
|
385
|
+
# Every skill declared at any point, for full_skill_coverage. This asks a DIFFERENT
|
|
386
|
+
# question from active_skills -- "did the conversation ever exercise this skill" rather
|
|
387
|
+
# than "is it active now" -- so replace semantics do not apply: a skill switched on and
|
|
388
|
+
# later switched off was still exercised. Deriving coverage from the per-turn final
|
|
389
|
+
# declaration instead would drop any skill a turn declared and then replaced within
|
|
390
|
+
# itself (across its clarification sub-turns), a false FAIL on a genuine activation.
|
|
391
|
+
ever_declared_skills: set[str] = set()
|
|
313
392
|
# Every send_message() call (across every logical turn AND every clarification
|
|
314
393
|
# sub-turn within it) restarts call_ts/ts near 0 -- these run across the whole
|
|
315
394
|
# conversation, not reset per logical turn, so every one of those calls shifts them.
|
|
@@ -337,6 +416,10 @@ def run_agentic_conversation(
|
|
|
337
416
|
output_present=False,
|
|
338
417
|
no_error=False,
|
|
339
418
|
activated_skills=[],
|
|
419
|
+
# The turn never ran, so it declared nothing -- but a set carried over
|
|
420
|
+
# from an earlier turn is still active, and reporting [] here would
|
|
421
|
+
# read as "nothing was active", which is a different claim.
|
|
422
|
+
active_skills=sorted(active_skills),
|
|
340
423
|
output_correct=False,
|
|
341
424
|
)
|
|
342
425
|
)
|
|
@@ -351,22 +434,15 @@ def run_agentic_conversation(
|
|
|
351
434
|
for _iter in range(max_clarification_turns + 1):
|
|
352
435
|
chat_result = client.send_message(conversation_id, current_message)
|
|
353
436
|
final_result = chat_result
|
|
354
|
-
|
|
355
|
-
|
|
356
|
-
|
|
357
|
-
|
|
358
|
-
|
|
359
|
-
|
|
360
|
-
tc.index += tool_index_offset
|
|
361
|
-
for rs in chat_result.reasoning_step_events or []:
|
|
362
|
-
rs.ts += turn_offset
|
|
363
|
-
rs.index += reasoning_index_offset
|
|
437
|
+
turn_offset, tool_index_offset, reasoning_index_offset = shift_and_index_events(
|
|
438
|
+
chat_result,
|
|
439
|
+
turn_offset=turn_offset,
|
|
440
|
+
tool_index_offset=tool_index_offset,
|
|
441
|
+
reasoning_index_offset=reasoning_index_offset,
|
|
442
|
+
)
|
|
364
443
|
all_tool_calls.extend(chat_result.tool_call_events or [])
|
|
365
444
|
conversation_tool_call_events.extend(chat_result.tool_call_events or [])
|
|
366
445
|
conversation_reasoning_step_events.extend(chat_result.reasoning_step_events or [])
|
|
367
|
-
tool_index_offset += len(chat_result.tool_call_events or [])
|
|
368
|
-
reasoning_index_offset += len(chat_result.reasoning_step_events or [])
|
|
369
|
-
turn_offset += chat_result.turn_wall_clock_sec or 0.0
|
|
370
446
|
reasoning_steps.extend(chat_result.reasoning_steps or [])
|
|
371
447
|
response_id = chat_result.response_id or response_id
|
|
372
448
|
|
|
@@ -376,6 +452,8 @@ def run_agentic_conversation(
|
|
|
376
452
|
response_text = (chat_result.text_response or "").strip()
|
|
377
453
|
if not response_text and chat_result.alert_proposals:
|
|
378
454
|
response_text = render_alert_proposal(chat_result.alert_proposals[-1])
|
|
455
|
+
if not response_text:
|
|
456
|
+
response_text = render_answer_text(chat_result)
|
|
379
457
|
if not response_text and not chat_result.tool_call_events:
|
|
380
458
|
break
|
|
381
459
|
if clarification_turns >= max_clarification_turns:
|
|
@@ -384,8 +462,17 @@ def run_agentic_conversation(
|
|
|
384
462
|
total_clarification_turns += 1
|
|
385
463
|
current_message = _get_sim_user_response(response_text, resolved_turn, resolved_expected)
|
|
386
464
|
|
|
387
|
-
|
|
388
|
-
|
|
465
|
+
# `declared` is what THIS turn's final set_skills call asked for (None when it
|
|
466
|
+
# made no call); `active_skills` is what is actually active during the turn. No
|
|
467
|
+
# call carries the previous set over; a call replaces it outright, including
|
|
468
|
+
# when it declares an empty list. See active_skills' declaration above.
|
|
469
|
+
declarations = _set_skills_declarations(all_tool_calls)
|
|
470
|
+
for declaration in declarations:
|
|
471
|
+
ever_declared_skills.update(declaration)
|
|
472
|
+
declared = declarations[-1] if declarations else None
|
|
473
|
+
if declared is not None:
|
|
474
|
+
active_skills = set(declared)
|
|
475
|
+
skill_routing = turn.expected_skill in active_skills
|
|
389
476
|
output_present = _check_output_present(resolved_turn, final_result) if final_result else False
|
|
390
477
|
output_correct = (
|
|
391
478
|
_check_output_correct(resolved_turn, final_result) if (final_result and output_present) else None
|
|
@@ -409,7 +496,8 @@ def run_agentic_conversation(
|
|
|
409
496
|
skill_routing=skill_routing,
|
|
410
497
|
output_present=output_present,
|
|
411
498
|
no_error=True, # SDK raises on errors; reaching here means no critical error.
|
|
412
|
-
activated_skills=
|
|
499
|
+
activated_skills=declared or [],
|
|
500
|
+
active_skills=sorted(active_skills),
|
|
413
501
|
clarification_turns_used=clarification_turns,
|
|
414
502
|
output_correct=output_correct,
|
|
415
503
|
)
|
|
@@ -422,8 +510,10 @@ def run_agentic_conversation(
|
|
|
422
510
|
_delete_metric(sdk, workspace_id, metric_id)
|
|
423
511
|
client.close()
|
|
424
512
|
|
|
425
|
-
|
|
426
|
-
|
|
513
|
+
# Not derived from TurnResult.activated_skills: that field carries only each turn's FINAL
|
|
514
|
+
# declaration, so a skill replaced within its own turn is absent from it despite having
|
|
515
|
+
# been activated. See ever_declared_skills' declaration above.
|
|
516
|
+
full_skill_coverage = set(fixture.expected_skills).issubset(ever_declared_skills)
|
|
427
517
|
conversation_success = all(tr.skill_success for tr in turn_results)
|
|
428
518
|
|
|
429
519
|
return ConversationResult(
|
|
@@ -443,17 +533,7 @@ def _conversation_detail(result: ConversationResult) -> dict:
|
|
|
443
533
|
return {
|
|
444
534
|
"full_skill_coverage": result.full_skill_coverage,
|
|
445
535
|
"total_clarification_turns": result.total_clarification_turns,
|
|
446
|
-
"turns": [
|
|
447
|
-
{
|
|
448
|
-
"turn_id": tr.turn_id,
|
|
449
|
-
"expected_skill": tr.expected_skill,
|
|
450
|
-
"skill_routing": tr.skill_routing,
|
|
451
|
-
"output_present": tr.output_present,
|
|
452
|
-
"output_correct": tr.output_correct,
|
|
453
|
-
"activated_skills": tr.activated_skills,
|
|
454
|
-
}
|
|
455
|
-
for tr in result.turn_results
|
|
456
|
-
],
|
|
536
|
+
"turns": [tr.detail() for tr in result.turn_results],
|
|
457
537
|
"latency_breakdown": build_latency_breakdown(result.tool_call_events, result.reasoning_step_events),
|
|
458
538
|
}
|
|
459
539
|
|