gooddata-eval 1.75.1.dev3__tar.gz → 1.75.1.dev4__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/PKG-INFO +2 -2
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/pyproject.toml +2 -2
- gooddata_eval-1.75.1.dev4/src/gooddata_eval/core/agentic/_conversation_context.py +295 -0
- gooddata_eval-1.75.1.dev4/src/gooddata_eval/core/agentic/conversation.py +1171 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/src/gooddata_eval/core/agentic/metric_skill.py +5 -2
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/src/gooddata_eval/core/chat/sse_client.py +19 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/tests/test_agentic_conversation.py +815 -1
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/tests/test_agentic_metric_skill.py +17 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/tests/test_sse_client.py +31 -1
- gooddata_eval-1.75.1.dev3/src/gooddata_eval/core/agentic/conversation.py +0 -699
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/.gitignore +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/AGENTS.md +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/CLAUDE.md +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/LICENSE.txt +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/Makefile +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/README.md +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/scripts/verify_guardrail_refusal_criteria.py +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/src/gooddata_eval/__init__.py +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/src/gooddata_eval/_version.py +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/src/gooddata_eval/cli/__init__.py +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/src/gooddata_eval/cli/agentic_runner.py +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/src/gooddata_eval/cli/main.py +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/src/gooddata_eval/core/__init__.py +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/src/gooddata_eval/core/_output.py +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/src/gooddata_eval/core/agentic/__init__.py +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/src/gooddata_eval/core/agentic/_catalog.py +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/src/gooddata_eval/core/agentic/_gate.py +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/src/gooddata_eval/core/agentic/_langfuse.py +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/src/gooddata_eval/core/agentic/_trace_linker.py +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/src/gooddata_eval/core/agentic/alert_skill.py +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/src/gooddata_eval/core/agentic/dashboard_skill.py +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/src/gooddata_eval/core/agentic/general_question.py +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/src/gooddata_eval/core/agentic/guardrail.py +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/src/gooddata_eval/core/agentic/kda_skill.py +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/src/gooddata_eval/core/agentic/search_tool.py +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/src/gooddata_eval/core/agentic/visualization.py +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/src/gooddata_eval/core/agentic/what_if.py +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/src/gooddata_eval/core/chat/__init__.py +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/src/gooddata_eval/core/chat/render.py +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/src/gooddata_eval/core/config.py +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/src/gooddata_eval/core/connection.py +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/src/gooddata_eval/core/dataset/__init__.py +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/src/gooddata_eval/core/dataset/from_insights.py +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/src/gooddata_eval/core/dataset/langfuse_source.py +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/src/gooddata_eval/core/dataset/local.py +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/src/gooddata_eval/core/evaluators/__init__.py +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/src/gooddata_eval/core/evaluators/_deep_subset.py +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/src/gooddata_eval/core/evaluators/_guardrail_criteria.py +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/src/gooddata_eval/core/evaluators/_llm_judge.py +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/src/gooddata_eval/core/evaluators/_maql.py +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/src/gooddata_eval/core/evaluators/_text_utils.py +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/src/gooddata_eval/core/evaluators/alert_skill.py +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/src/gooddata_eval/core/evaluators/base.py +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/src/gooddata_eval/core/evaluators/general_question.py +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/src/gooddata_eval/core/evaluators/guardrail.py +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/src/gooddata_eval/core/evaluators/metric_skill.py +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/src/gooddata_eval/core/evaluators/search_tool.py +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/src/gooddata_eval/core/evaluators/summary.py +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/src/gooddata_eval/core/evaluators/visualization.py +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/src/gooddata_eval/core/granularity.py +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/src/gooddata_eval/core/langfuse/__init__.py +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/src/gooddata_eval/core/langfuse/_env.py +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/src/gooddata_eval/core/langfuse/client.py +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/src/gooddata_eval/core/langfuse/experiment.py +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/src/gooddata_eval/core/langfuse/observations.py +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/src/gooddata_eval/core/langfuse/otlp.py +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/src/gooddata_eval/core/langfuse/sink.py +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/src/gooddata_eval/core/models.py +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/src/gooddata_eval/core/reporting/__init__.py +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/src/gooddata_eval/core/reporting/console.py +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/src/gooddata_eval/core/reporting/html_report.py +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/src/gooddata_eval/core/reporting/json_report.py +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/src/gooddata_eval/core/reporting/report_template.html +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/src/gooddata_eval/core/runner.py +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/src/gooddata_eval/core/scoring.py +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/src/gooddata_eval/core/summary/__init__.py +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/src/gooddata_eval/core/summary/http_client.py +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/src/gooddata_eval/core/timing.py +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/src/gooddata_eval/core/workspace.py +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/tests/__init__.py +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/tests/_fake_langfuse.py +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/tests/conftest.py +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/tests/fixtures/sample_dataset/metric_skill_create.json +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/tests/fixtures/sample_dataset/visualization_revenue.json +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/tests/fixtures/sse_visualization_stream.txt +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/tests/test_agentic_alert_skill.py +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/tests/test_agentic_dashboard_skill.py +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/tests/test_agentic_gate.py +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/tests/test_agentic_general_question.py +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/tests/test_agentic_guardrail.py +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/tests/test_agentic_kda_skill.py +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/tests/test_agentic_langfuse_trace.py +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/tests/test_agentic_observe_experiment.py +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/tests/test_agentic_run_context.py +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/tests/test_agentic_runner.py +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/tests/test_agentic_search_tool.py +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/tests/test_agentic_visualization.py +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/tests/test_agentic_what_if.py +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/tests/test_alert_skill_evaluator.py +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/tests/test_chat_render.py +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/tests/test_cli.py +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/tests/test_connection.py +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/tests/test_deep_subset.py +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/tests/test_fake_langfuse.py +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/tests/test_from_insights.py +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/tests/test_guardrail_criteria.py +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/tests/test_html_report.py +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/tests/test_langfuse_client.py +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/tests/test_langfuse_e2e_fake_server.py +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/tests/test_langfuse_env.py +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/tests/test_langfuse_experiment.py +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/tests/test_langfuse_observations.py +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/tests/test_langfuse_otlp.py +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/tests/test_langfuse_sink.py +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/tests/test_langfuse_source.py +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/tests/test_llm_judge.py +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/tests/test_local_loader.py +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/tests/test_maql_normalize.py +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/tests/test_metric_skill_evaluator.py +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/tests/test_models.py +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/tests/test_reporting.py +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/tests/test_runner.py +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/tests/test_scoring.py +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/tests/test_search_tool_evaluator.py +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/tests/test_summary_client.py +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/tests/test_summary_evaluator.py +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/tests/test_text_evaluators.py +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/tests/test_timing.py +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/tests/test_trace_linker.py +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/tests/test_visualization_evaluator.py +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/tests/test_workspace.py +0 -0
- {gooddata_eval-1.75.1.dev3 → gooddata_eval-1.75.1.dev4}/tox.ini +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: gooddata-eval
|
|
3
|
-
Version: 1.75.1.
|
|
3
|
+
Version: 1.75.1.dev4
|
|
4
4
|
Summary: Evaluate the GoodData AI agent against your own questions and models.
|
|
5
5
|
Project-URL: Source, https://github.com/gooddata/gooddata-python-sdk
|
|
6
6
|
Author-email: GoodData <support@gooddata.com>
|
|
@@ -17,7 +17,7 @@ Classifier: Topic :: Scientific/Engineering
|
|
|
17
17
|
Classifier: Topic :: Software Development
|
|
18
18
|
Classifier: Typing :: Typed
|
|
19
19
|
Requires-Python: >=3.10
|
|
20
|
-
Requires-Dist: gooddata-sdk~=1.75.1.
|
|
20
|
+
Requires-Dist: gooddata-sdk~=1.75.1.dev4
|
|
21
21
|
Requires-Dist: httpx<1.0,>=0.27
|
|
22
22
|
Requires-Dist: jsonpatch<2.0,>=1.33
|
|
23
23
|
Requires-Dist: orjson<4.0.0,>=3.9.15
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
# (C) 2026 GoodData Corporation
|
|
2
2
|
[project]
|
|
3
3
|
name = "gooddata-eval"
|
|
4
|
-
version = "1.75.1.
|
|
4
|
+
version = "1.75.1.dev4"
|
|
5
5
|
description = "Evaluate the GoodData AI agent against your own questions and models."
|
|
6
6
|
readme = "README.md"
|
|
7
7
|
license = "MIT"
|
|
@@ -11,7 +11,7 @@ authors = [
|
|
|
11
11
|
keywords = ["gooddata", "ai", "evaluation", "llm", "analytics", "cli"]
|
|
12
12
|
requires-python = ">=3.10"
|
|
13
13
|
dependencies = [
|
|
14
|
-
"gooddata-sdk~=1.75.1.
|
|
14
|
+
"gooddata-sdk~=1.75.1.dev4",
|
|
15
15
|
"httpx>=0.27,<1.0",
|
|
16
16
|
"jsonpatch>=1.33,<2.0",
|
|
17
17
|
"orjson>=3.9.15,<4.0.0",
|
|
@@ -0,0 +1,295 @@
|
|
|
1
|
+
# (C) 2026 GoodData Corporation. All rights reserved.
|
|
2
|
+
"""What the conversation evaluator needs to judge context, not just output presence.
|
|
3
|
+
|
|
4
|
+
Three questions get asked of every assistant reply that did not produce the turn's
|
|
5
|
+
expected output:
|
|
6
|
+
|
|
7
|
+
1. What kind of reply is it -- a confirmation request ("Should I create this metric?"),
|
|
8
|
+
a question to the user, or a turn that ended without doing anything (a stall)?
|
|
9
|
+
2. For a question: did the assistant ask for something the conversation had already
|
|
10
|
+
established? That is the lost-context signal, graded by a binary LLM judge.
|
|
11
|
+
3. What does the simulated user say back? In ``context`` mode it never sees the turn's
|
|
12
|
+
expected output -- only the fixture's set answers and the conversation itself -- so a
|
|
13
|
+
lost context can no longer be repaired silently by a user who knows the answer key.
|
|
14
|
+
|
|
15
|
+
Only information that is IN the conversation counts. A question about something the
|
|
16
|
+
assistant could have looked up in the workspace, but that nobody mentioned, is graded as
|
|
17
|
+
legitimate: whether the assistant should have looked it up is a separate check.
|
|
18
|
+
"""
|
|
19
|
+
|
|
20
|
+
from __future__ import annotations
|
|
21
|
+
|
|
22
|
+
import hashlib
|
|
23
|
+
import json
|
|
24
|
+
import os
|
|
25
|
+
import re
|
|
26
|
+
from dataclasses import dataclass
|
|
27
|
+
from typing import Literal
|
|
28
|
+
|
|
29
|
+
from gooddata_eval.core.config import judge_model
|
|
30
|
+
from gooddata_eval.core.evaluators._llm_judge import JudgeResponseError, LLMJudge, _message_content
|
|
31
|
+
from gooddata_eval.core.models import ChatResult
|
|
32
|
+
|
|
33
|
+
ReplyKind = Literal["confirmation", "question", "no_action"]
|
|
34
|
+
|
|
35
|
+
NUDGE_MESSAGE = "Please go ahead and do it."
|
|
36
|
+
CONFIRMATION_REPLY = "Yes, go ahead."
|
|
37
|
+
# Asked to pick from a list (which dashboard to bind an alert to), a real user picks one.
|
|
38
|
+
NO_ANSWER_REPLY = "Use your best judgement and go ahead. If you need me to pick from a list, take the first option."
|
|
39
|
+
|
|
40
|
+
# Part of the metric and alert skills' designed flow, not a clarification.
|
|
41
|
+
_CONFIRMATION_RE = re.compile(
|
|
42
|
+
r"\b(?:should|shall|may|can)\s+i\s+(?:go\s+ahead|proceed|create|save|update|apply|add|set\s+(?:it|this)\s+up)"
|
|
43
|
+
r"|\bdo\s+you\s+want\s+me\s+to\s+(?:go\s+ahead|proceed|create|save|update|apply)"
|
|
44
|
+
r"|\bwould\s+you\s+like\s+me\s+to\s+(?:go\s+ahead|proceed|create|save|update|apply)"
|
|
45
|
+
r"|\bplease\s+confirm\b|\bconfirm\s+(?:if|whether|that)\b",
|
|
46
|
+
re.IGNORECASE,
|
|
47
|
+
)
|
|
48
|
+
|
|
49
|
+
# Checked after _CONFIRMATION_RE, so "please confirm" stays a confirmation.
|
|
50
|
+
_REQUEST_RE = re.compile(
|
|
51
|
+
r"\bplease\s+(?:select|choose|pick|specify|provide|tell\s+me|let\s+me\s+know|clarify|share)\b"
|
|
52
|
+
r"|\breply\s+with\b|\b(?:select|choose|pick)\s+(?:one|a|an|the)\b",
|
|
53
|
+
re.IGNORECASE,
|
|
54
|
+
)
|
|
55
|
+
|
|
56
|
+
_MESSAGE_CHARS = 800
|
|
57
|
+
_TRANSCRIPT_MESSAGES = 40
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def has_structured_question(chat_result: ChatResult) -> bool:
|
|
61
|
+
"""gen-ai's own signal that the reply is a question: a ``clarifyingQuestions`` part."""
|
|
62
|
+
return any(part.get("type") == "clarifyingQuestions" for part in chat_result.unhandled_parts or [])
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
def classify_reply(chat_result: ChatResult, rendered_text: str) -> ReplyKind:
|
|
66
|
+
"""Kind of a reply that did not produce the expected output.
|
|
67
|
+
|
|
68
|
+
The structured ``clarifyingQuestions`` part is authoritative when present. Without it
|
|
69
|
+
the text decides: a confirmation phrase or an alert proposal is a confirmation, a
|
|
70
|
+
question mark is a question, and anything else -- including an empty reply -- is a turn
|
|
71
|
+
that stopped without acting.
|
|
72
|
+
"""
|
|
73
|
+
if has_structured_question(chat_result):
|
|
74
|
+
return "question"
|
|
75
|
+
if chat_result.alert_proposals:
|
|
76
|
+
return "confirmation"
|
|
77
|
+
text = (rendered_text or "").strip()
|
|
78
|
+
if not text:
|
|
79
|
+
return "no_action"
|
|
80
|
+
if _CONFIRMATION_RE.search(text):
|
|
81
|
+
return "confirmation"
|
|
82
|
+
if "?" in text or _REQUEST_RE.search(text):
|
|
83
|
+
return "question"
|
|
84
|
+
return "no_action"
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
@dataclass
|
|
88
|
+
class TranscriptEntry:
|
|
89
|
+
role: Literal["user", "assistant"]
|
|
90
|
+
text: str
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
def clip(text: str, limit: int = _MESSAGE_CHARS) -> str:
|
|
94
|
+
text = " ".join((text or "").split())
|
|
95
|
+
return text if len(text) <= limit else text[: limit - 1] + "…"
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
def summarize_visualizations(chat_result: ChatResult) -> str:
|
|
99
|
+
"""One line per chart the reply showed: what it plotted and how it was filtered.
|
|
100
|
+
|
|
101
|
+
Charts carry most of what a conversation establishes ("the Apparel filter", "by
|
|
102
|
+
quarter"), and they are invisible in the text part. The judge needs them to know what
|
|
103
|
+
was already settled.
|
|
104
|
+
"""
|
|
105
|
+
vizzes = chat_result.created_visualizations
|
|
106
|
+
objects = getattr(vizzes, "objects", None) or []
|
|
107
|
+
lines = []
|
|
108
|
+
for viz in objects:
|
|
109
|
+
fields = [getattr(f, "using", f) for f in (viz.query.fields or {}).values()]
|
|
110
|
+
filters = [
|
|
111
|
+
{k: v for k, v in f.items() if k in ("type", "using", "state", "from", "to", "top", "bottom")}
|
|
112
|
+
for f in (viz.query.filter_by or {}).values()
|
|
113
|
+
if isinstance(f, dict)
|
|
114
|
+
]
|
|
115
|
+
lines.append(f"[chart {viz.type or ''}: fields={fields} filters={filters}]")
|
|
116
|
+
return " ".join(lines)
|
|
117
|
+
|
|
118
|
+
|
|
119
|
+
def render_transcript(entries: list[TranscriptEntry]) -> str:
|
|
120
|
+
recent = entries[-_TRANSCRIPT_MESSAGES:]
|
|
121
|
+
return "\n".join(f"{e.role.upper()}: {clip(e.text)}" for e in recent)
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
_JUDGE_SYSTEM = """\
|
|
125
|
+
You grade one question that an analytics assistant asked its user. You get:
|
|
126
|
+
- CONVERSATION BEFORE: everything said before the user's latest message (may be empty),
|
|
127
|
+
- LATEST MESSAGE: what the user just asked,
|
|
128
|
+
- QUESTION: the assistant's reply, in which it asks the user something instead of doing it.
|
|
129
|
+
|
|
130
|
+
Decide whether the information the QUESTION asks for already appears in CONVERSATION BEFORE:
|
|
131
|
+
stated by the user, or determined by an assistant answer -- for example which metric was meant,
|
|
132
|
+
which filters apply, which item came out weakest or biggest, what a chart showed, which object
|
|
133
|
+
was created.
|
|
134
|
+
|
|
135
|
+
Rules:
|
|
136
|
+
- The LATEST MESSAGE does not count. A reference in it ("it", "that second one", "those
|
|
137
|
+
brands") only points at something that must appear in CONVERSATION BEFORE.
|
|
138
|
+
- If CONVERSATION BEFORE is empty, the answer is false.
|
|
139
|
+
- Information that exists in the workspace but was never mentioned in the conversation does not
|
|
140
|
+
count.
|
|
141
|
+
- Do not judge whether asking was useful, necessary or well phrased.
|
|
142
|
+
|
|
143
|
+
Examples:
|
|
144
|
+
- BEFORE: user "Net sales by country." / assistant "Here is net sales by country." LATEST: "Add
|
|
145
|
+
order count to it." QUESTION: "Which chart should I add order count to?" -> true (the chart
|
|
146
|
+
just made).
|
|
147
|
+
- BEFORE: user "Top 10 products by revenue." / assistant "Here are the top 10 products."
|
|
148
|
+
LATEST: "Remove the limit." QUESTION: "Which limit do you mean?" -> true (the top 10).
|
|
149
|
+
- BEFORE: (empty) LATEST: "Show revenue by month." QUESTION: "Which revenue metric: gross or
|
|
150
|
+
net?" -> false (nothing earlier says which).
|
|
151
|
+
|
|
152
|
+
Return a JSON object with exactly two keys:
|
|
153
|
+
"already_in_conversation": true or false
|
|
154
|
+
"reasoning": one sentence naming the information and where it appears, or that it does not
|
|
155
|
+
"""
|
|
156
|
+
|
|
157
|
+
_JUDGE_USER = "CONVERSATION BEFORE:\n{history}\n\nLATEST MESSAGE:\n{latest}\n\nQUESTION:\n{question}"
|
|
158
|
+
|
|
159
|
+
|
|
160
|
+
def judge_prompt(history: str, latest: str, question: str) -> str:
|
|
161
|
+
"""The user prompt the clarification judge reads, also recorded for auditing."""
|
|
162
|
+
return _JUDGE_USER.format(history=history or "(empty)", latest=latest, question=question)
|
|
163
|
+
|
|
164
|
+
|
|
165
|
+
@dataclass(frozen=True)
|
|
166
|
+
class ClarificationVerdict:
|
|
167
|
+
lost_context: bool | None # None when the judge could not be run or gave no verdict
|
|
168
|
+
reasoning: str
|
|
169
|
+
error: str | None = None
|
|
170
|
+
|
|
171
|
+
|
|
172
|
+
class _ClarificationLLM(LLMJudge):
|
|
173
|
+
"""``LLMJudge``'s client, temperature handling and retry, with this judge's own prompt and
|
|
174
|
+
verdict key. The key names the question asked, so a 0/1 cannot be read the wrong way round."""
|
|
175
|
+
|
|
176
|
+
def __init__(self, model: str | None = None) -> None:
|
|
177
|
+
super().__init__([], model=model)
|
|
178
|
+
self._system_prompt = _JUDGE_SYSTEM
|
|
179
|
+
|
|
180
|
+
def already_in_conversation(self, prompt: str) -> tuple[bool, str]:
|
|
181
|
+
messages = [{"role": "system", "content": self._system_prompt}, {"role": "user", "content": prompt}]
|
|
182
|
+
raw = None
|
|
183
|
+
for _attempt in range(2):
|
|
184
|
+
raw = _message_content(self._create_completion(messages))
|
|
185
|
+
if isinstance(raw, str) and raw.strip():
|
|
186
|
+
break
|
|
187
|
+
if not (isinstance(raw, str) and raw.strip()):
|
|
188
|
+
raise JudgeResponseError(f"clarification judge {self.model!r} returned an empty body twice")
|
|
189
|
+
try:
|
|
190
|
+
data = json.loads(raw)
|
|
191
|
+
except json.JSONDecodeError as exc:
|
|
192
|
+
raise JudgeResponseError(f"clarification judge returned unparseable JSON: {raw!r}") from exc
|
|
193
|
+
value = data.get("already_in_conversation") if isinstance(data, dict) else None
|
|
194
|
+
if not isinstance(value, bool):
|
|
195
|
+
raise JudgeResponseError(f"clarification judge returned no boolean verdict: {raw!r}")
|
|
196
|
+
return value, str(data.get("reasoning", ""))
|
|
197
|
+
|
|
198
|
+
|
|
199
|
+
class ClarificationJudge:
|
|
200
|
+
"""Binary judge: did the assistant ask for something the conversation already established?
|
|
201
|
+
|
|
202
|
+
Temperature 0 and a JSON verdict, on ``LLMJudge``'s client. Verdicts are cached per prompt,
|
|
203
|
+
so a conversation that repeats the same question does not pay for it twice.
|
|
204
|
+
"""
|
|
205
|
+
|
|
206
|
+
def __init__(self, model: str | None = None) -> None:
|
|
207
|
+
self._model = model
|
|
208
|
+
self._llm: _ClarificationLLM | None = None
|
|
209
|
+
self._cache: dict[str, ClarificationVerdict] = {}
|
|
210
|
+
self._warned = False
|
|
211
|
+
|
|
212
|
+
@property
|
|
213
|
+
def model_name(self) -> str:
|
|
214
|
+
"""The judge model, without building a client (which needs an API key)."""
|
|
215
|
+
return self._llm.model if self._llm is not None else (self._model or judge_model())
|
|
216
|
+
|
|
217
|
+
def judge(self, history: str, latest: str, question: str) -> ClarificationVerdict:
|
|
218
|
+
prompt = judge_prompt(history, latest, question)
|
|
219
|
+
key = hashlib.sha256(prompt.encode()).hexdigest()
|
|
220
|
+
if key in self._cache:
|
|
221
|
+
return self._cache[key]
|
|
222
|
+
try:
|
|
223
|
+
if self._llm is None:
|
|
224
|
+
self._llm = _ClarificationLLM(self._model)
|
|
225
|
+
try:
|
|
226
|
+
found, reasoning = self._llm.already_in_conversation(prompt)
|
|
227
|
+
verdict = ClarificationVerdict(lost_context=found, reasoning=reasoning)
|
|
228
|
+
except JudgeResponseError as exc:
|
|
229
|
+
verdict = ClarificationVerdict(lost_context=None, reasoning="", error=str(exc))
|
|
230
|
+
except Exception as exc: # noqa: BLE001 -- a grading fault must not fail the agent's conversation
|
|
231
|
+
if not self._warned:
|
|
232
|
+
print(f"[JUDGE] clarifications left unjudged: {exc}")
|
|
233
|
+
self._warned = True
|
|
234
|
+
verdict = ClarificationVerdict(lost_context=None, reasoning="", error=str(exc))
|
|
235
|
+
self._cache[key] = verdict
|
|
236
|
+
return verdict
|
|
237
|
+
|
|
238
|
+
|
|
239
|
+
_SIM_USER_MODEL = "gpt-4o"
|
|
240
|
+
|
|
241
|
+
|
|
242
|
+
def _chat(prompt_system: str, prompt_user: str) -> str | None:
|
|
243
|
+
try:
|
|
244
|
+
from openai import OpenAI # noqa: PLC0415
|
|
245
|
+
except ImportError:
|
|
246
|
+
return None
|
|
247
|
+
api_key = os.environ.get("OPENAI_API_KEY")
|
|
248
|
+
if not api_key:
|
|
249
|
+
return None
|
|
250
|
+
try:
|
|
251
|
+
response = OpenAI(api_key=api_key).chat.completions.create(
|
|
252
|
+
model=_SIM_USER_MODEL,
|
|
253
|
+
messages=[{"role": "system", "content": prompt_system}, {"role": "user", "content": prompt_user}],
|
|
254
|
+
temperature=0,
|
|
255
|
+
max_tokens=200,
|
|
256
|
+
)
|
|
257
|
+
except Exception as exc:
|
|
258
|
+
print(f"[SIM-USER] reply failed: {exc}")
|
|
259
|
+
return None
|
|
260
|
+
content = response.choices[0].message.content if response.choices else None
|
|
261
|
+
return content.strip() if content else None
|
|
262
|
+
|
|
263
|
+
|
|
264
|
+
def reply_from_facts(transcript: str, question: str, facts: list[str]) -> str:
|
|
265
|
+
"""A legitimate question the fixture has no unused set answer for.
|
|
266
|
+
|
|
267
|
+
The reply may use only the fixture's facts for this turn -- never the expected output --
|
|
268
|
+
and falls back to "use your best judgement" when the facts do not cover the question.
|
|
269
|
+
"""
|
|
270
|
+
if not facts:
|
|
271
|
+
return NO_ANSWER_REPLY
|
|
272
|
+
reply = _chat(
|
|
273
|
+
"You play a business user talking to an analytics assistant. Answer its question briefly, "
|
|
274
|
+
"using ONLY the facts listed below. Never invent requirements. If the facts do not answer "
|
|
275
|
+
f"the question, reply exactly: {NO_ANSWER_REPLY}",
|
|
276
|
+
f"Conversation so far:\n{transcript}\n\nThe assistant now asks:\n{question}\n\n"
|
|
277
|
+
"Facts you know:\n" + "\n".join(f"- {f}" for f in facts),
|
|
278
|
+
)
|
|
279
|
+
return reply or NO_ANSWER_REPLY
|
|
280
|
+
|
|
281
|
+
|
|
282
|
+
def reply_restating(transcript: str, question: str) -> str:
|
|
283
|
+
"""The assistant asked for something the conversation already established.
|
|
284
|
+
|
|
285
|
+
The turn is already failed; restating keeps the rest of the conversation measurable,
|
|
286
|
+
which is what a real user would do too.
|
|
287
|
+
"""
|
|
288
|
+
reply = _chat(
|
|
289
|
+
"You play a business user talking to an analytics assistant. The assistant just asked for "
|
|
290
|
+
"something that was already said or found earlier in the conversation. Reply briefly, "
|
|
291
|
+
"starting with 'As I said earlier,' and repeat only that information, taken from the "
|
|
292
|
+
"conversation. Do not add anything new.",
|
|
293
|
+
f"Conversation so far:\n{transcript}\n\nThe assistant now asks:\n{question}",
|
|
294
|
+
)
|
|
295
|
+
return reply or "As I said earlier in this conversation — please use what we already established."
|