actionagent 1.7.2 → 1.8.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/app/assets/builds/action_agent.css +1 -1
- data/app/assets/builds/action_agent.js +59 -54
- data/app/controllers/action_agent/api/agents_controller.rb +19 -3
- data/app/controllers/action_agent/api/code_sessions_controller.rb +156 -0
- data/app/controllers/action_agent/api/evaluations_controller.rb +27 -130
- data/app/controllers/action_agent/api/github_connections_controller.rb +147 -0
- data/app/controllers/action_agent/api/mcp_controller.rb +38 -5
- data/app/controllers/action_agent/api/mcp_servers_controller.rb +10 -1
- data/app/controllers/action_agent/api/provider_keys_controller.rb +54 -3
- data/app/controllers/action_agent/api/provider_models_controller.rb +10 -6
- data/app/controllers/action_agent/api/sandboxes_controller.rb +97 -6
- data/app/controllers/concerns/action_agent/api/evaluation_run_starting.rb +93 -0
- data/app/controllers/concerns/action_agent/api/mcp_dashboard_tools.rb +507 -0
- data/app/controllers/concerns/action_agent/api/run_sandbox.rb +65 -0
- data/app/jobs/action_agent/code_session_job.rb +166 -0
- data/app/jobs/action_agent/sandbox_cleanup_job.rb +80 -11
- data/app/jobs/action_agent/sandbox_provision_job.rb +122 -14
- data/app/jobs/action_agent/sandbox_run_job.rb +10 -3
- data/app/models/action_agent/agent.rb +16 -6
- data/app/models/action_agent/agent_run.rb +20 -1
- data/app/models/action_agent/code_session.rb +141 -0
- data/app/models/action_agent/evaluation_run.rb +23 -1
- data/app/models/action_agent/github_connection.rb +75 -0
- data/app/models/action_agent/provider_key.rb +142 -10
- data/app/models/action_agent/sandbox_session.rb +193 -17
- data/app/serializers/action_agent/evaluation_serializer.rb +118 -0
- data/app/serializers/action_agent/telemetry_trace_serializer.rb +10 -2
- data/app/services/action_agent/agent_execution_service.rb +4 -2
- data/app/services/action_agent/agent_tool_roster.rb +20 -9
- data/app/services/action_agent/claude_code_auth.rb +86 -0
- data/app/services/action_agent/dashboard_assistant_service.rb +47 -5
- data/app/services/action_agent/evaluation_tool_resolver.rb +18 -0
- data/app/services/action_agent/github_client.rb +111 -0
- data/app/services/action_agent/local_sandbox_backend.rb +1689 -0
- data/app/services/action_agent/local_sandbox_databases.rb +257 -0
- data/app/services/action_agent/mcp_client.rb +5 -1
- data/app/services/action_agent/mcp_tool_dispatcher.rb +137 -17
- data/app/services/action_agent/mock_sandbox_backend.rb +39 -0
- data/app/services/action_agent/ollama_host_probe.rb +75 -0
- data/app/services/action_agent/payload_bounds.rb +36 -0
- data/app/services/action_agent/sandbox_manifest.rb +67 -0
- data/app/services/action_agent/sandbox_orchestrator.rb +69 -14
- data/app/services/action_agent/scenario_evaluation_runner.rb +50 -4
- data/app/services/action_agent/secret_scrubber.rb +37 -0
- data/app/services/action_agent/tool_discovery.rb +19 -5
- data/config/routes.rb +22 -3
- data/lib/action_agent/engine.rb +1 -0
- data/lib/action_agent/version.rb +1 -1
- data/lib/action_agent.rb +147 -3
- data/lib/generators/action_agent/install_generator.rb +30 -3
- data/lib/generators/action_agent/templates/action_agent.rb.erb +44 -0
- data/lib/generators/action_agent/templates/add_provider_key_api_key.rb.erb +25 -0
- data/lib/generators/action_agent/templates/create_active_agent_code_sessions.rb.erb +59 -0
- data/lib/generators/action_agent/templates/create_active_agent_dashboard_tables.rb.erb +2 -0
- data/lib/generators/action_agent/templates/create_active_agent_github_connections.rb.erb +61 -0
- data/lib/tasks/claude_code.rake +16 -0
- data/lib/tasks/sandbox.rake +26 -0
- metadata +23 -1
|
@@ -0,0 +1,507 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module ActionAgent
|
|
4
|
+
module Api
|
|
5
|
+
# The dashboard's own evaluation and telemetry tools, served by the MCP
|
|
6
|
+
# facade (Api::MCPController) so a client's coding harness can edit an
|
|
7
|
+
# agent in its own checkout, run the agent's evaluations, read the runs'
|
|
8
|
+
# fix items and failing traces, and iterate. The harness brings its own
|
|
9
|
+
# model; the dashboard only answers these calls.
|
|
10
|
+
#
|
|
11
|
+
# Each tool reads under the key's owner the way the dashboard's JSON API
|
|
12
|
+
# reads under the signed-in owner: evaluations of the agents the owner
|
|
13
|
+
# can reach (Api::EvaluationsController), traces of the owner's tenant
|
|
14
|
+
# (Api::TraceReportsController). A record outside that scope reads as
|
|
15
|
+
# nonexistent.
|
|
16
|
+
#
|
|
17
|
+
# Names are a noun family followed by a verb (`evaluations_list`,
|
|
18
|
+
# `traces_get`). Host schema tools are always `find_`, `count_` or `get_`
|
|
19
|
+
# plus a model name, and agent tools are `run_<slug>`, so no host model or
|
|
20
|
+
# agent slug can produce one of these names.
|
|
21
|
+
#
|
|
22
|
+
# A call the client can correct (an unknown id, a sandbox it cannot use)
|
|
23
|
+
# comes back as a tool result with isError. The execution switch and the
|
|
24
|
+
# execution quota answer as JSON-RPC errors, as they do for `run_<slug>`.
|
|
25
|
+
module MCPDashboardTools
|
|
26
|
+
extend ActiveSupport::Concern
|
|
27
|
+
|
|
28
|
+
include EvaluationRunStarting
|
|
29
|
+
|
|
30
|
+
NAMES = %w[
|
|
31
|
+
evaluations_list evaluations_get evaluations_run evaluation_runs_get evaluation_runs_compare
|
|
32
|
+
traces_search traces_get
|
|
33
|
+
].freeze
|
|
34
|
+
|
|
35
|
+
LIST_LIMIT = 20
|
|
36
|
+
MAX_LIST_LIMIT = 50
|
|
37
|
+
SCENARIO_LIMIT = 200
|
|
38
|
+
RUN_HISTORY_LIMIT = 10
|
|
39
|
+
RESULT_LIMIT = 50
|
|
40
|
+
MAX_RESULT_LIMIT = 200
|
|
41
|
+
TRACE_WINDOW_MINUTES = 60 * 24
|
|
42
|
+
MAX_TRACE_WINDOW_MINUTES = TraceReportsController::MAX_WINDOW_MINUTES
|
|
43
|
+
TRACE_LIMIT = 20
|
|
44
|
+
MAX_TRACE_LIMIT = 100
|
|
45
|
+
MAX_TRACE_SPANS = 100
|
|
46
|
+
# Strings in a result row or a span, cut past this many characters.
|
|
47
|
+
MAX_STRING = 1_000
|
|
48
|
+
# Owned credentials read for scrubbing, per kind.
|
|
49
|
+
SECRET_LOOKUP_LIMIT = 100
|
|
50
|
+
|
|
51
|
+
SELECTION_PROPERTIES = {
|
|
52
|
+
scenario_ids: { type: "array", items: { type: "integer" }, description: "Replay only these scenarios, by id" },
|
|
53
|
+
keys: { type: "array", items: { type: "string" }, description: "Replay only these scenarios, by key" },
|
|
54
|
+
group: { type: "string", description: "Replay only this scenario group" },
|
|
55
|
+
models: { type: "array", items: { type: "string" }, description: "Candidate models to compare; the agent's own model when omitted" },
|
|
56
|
+
sandbox_id: {
|
|
57
|
+
type: "string",
|
|
58
|
+
description: "A ready checkout sandbox's session id: every replay reaches that checkout's app tools, " \
|
|
59
|
+
"without the agent being edited"
|
|
60
|
+
}
|
|
61
|
+
}.freeze
|
|
62
|
+
|
|
63
|
+
DEFINITIONS = [
|
|
64
|
+
{
|
|
65
|
+
name: "evaluations_list",
|
|
66
|
+
description: "List the evaluations of this key's agents, newest first, each with its latest run's status and " \
|
|
67
|
+
"score. Filter to one agent by slug or id.",
|
|
68
|
+
inputSchema: {
|
|
69
|
+
type: "object",
|
|
70
|
+
properties: {
|
|
71
|
+
agent: { type: "string", description: "An agent's slug or id" },
|
|
72
|
+
limit: { type: "integer", description: "At most this many (default #{LIST_LIMIT}, max #{MAX_LIST_LIMIT})" }
|
|
73
|
+
}
|
|
74
|
+
}
|
|
75
|
+
},
|
|
76
|
+
{
|
|
77
|
+
name: "evaluations_get",
|
|
78
|
+
description: "One evaluation: its criteria, its scenarios (for a scenario suite) and its #{RUN_HISTORY_LIMIT} " \
|
|
79
|
+
"most recent runs.",
|
|
80
|
+
inputSchema: {
|
|
81
|
+
type: "object",
|
|
82
|
+
properties: { evaluation_id: { type: "integer", description: "The evaluation's id" } },
|
|
83
|
+
required: [ "evaluation_id" ]
|
|
84
|
+
}
|
|
85
|
+
},
|
|
86
|
+
{
|
|
87
|
+
name: "evaluations_run",
|
|
88
|
+
description: "Start a run of an evaluation. A scenario suite replays each scenario through the agent in the " \
|
|
89
|
+
"background: the run comes back pending, so poll evaluation_runs_get with its id. A sampling " \
|
|
90
|
+
"evaluation scores recorded generations and finishes before this returns.",
|
|
91
|
+
inputSchema: {
|
|
92
|
+
type: "object",
|
|
93
|
+
properties: { evaluation_id: { type: "integer", description: "The evaluation's id" } }.merge(SELECTION_PROPERTIES),
|
|
94
|
+
required: [ "evaluation_id" ]
|
|
95
|
+
}
|
|
96
|
+
},
|
|
97
|
+
{
|
|
98
|
+
name: "evaluation_runs_get",
|
|
99
|
+
description: "One evaluation run: status, scores, usage, its fix items (what to change, and where) and its " \
|
|
100
|
+
"per-scenario, per-model results, each naming its telemetry trace (for traces_get) when one " \
|
|
101
|
+
"was recorded. Defaults to the evaluation's latest run.",
|
|
102
|
+
inputSchema: {
|
|
103
|
+
type: "object",
|
|
104
|
+
properties: {
|
|
105
|
+
evaluation_id: { type: "integer", description: "The evaluation's id" },
|
|
106
|
+
run_id: { type: "integer", description: "The run's id; the latest run when omitted" },
|
|
107
|
+
failed_only: { type: "boolean", description: "Only results that did not pass" },
|
|
108
|
+
limit: { type: "integer", description: "At most this many results (default #{RESULT_LIMIT}, max #{MAX_RESULT_LIMIT})" }
|
|
109
|
+
},
|
|
110
|
+
required: [ "evaluation_id" ]
|
|
111
|
+
}
|
|
112
|
+
},
|
|
113
|
+
{
|
|
114
|
+
name: "evaluation_runs_compare",
|
|
115
|
+
description: "Compare two runs of one evaluation, scenario by scenario and model by model: which results " \
|
|
116
|
+
"were fixed, which regressed, which still fail. Defaults to the latest run against the one before it.",
|
|
117
|
+
inputSchema: {
|
|
118
|
+
type: "object",
|
|
119
|
+
properties: {
|
|
120
|
+
evaluation_id: { type: "integer", description: "The evaluation's id" },
|
|
121
|
+
base_run_id: { type: "integer", description: "The earlier run's id" },
|
|
122
|
+
head_run_id: { type: "integer", description: "The later run's id" }
|
|
123
|
+
},
|
|
124
|
+
required: [ "evaluation_id" ]
|
|
125
|
+
}
|
|
126
|
+
},
|
|
127
|
+
{
|
|
128
|
+
name: "traces_search",
|
|
129
|
+
description: "Search telemetry traces, newest first. Returns summary rows (no spans); read one in full with " \
|
|
130
|
+
"traces_get.",
|
|
131
|
+
inputSchema: {
|
|
132
|
+
type: "object",
|
|
133
|
+
properties: {
|
|
134
|
+
agent: { type: "string", description: "An agent class name (as traces report it) or a dashboard agent's slug" },
|
|
135
|
+
status: { type: "string", enum: %w[error ok], description: "Only failed traces, or only successful ones" },
|
|
136
|
+
service: { type: "string", description: "The reporting service's name" },
|
|
137
|
+
since_minutes: {
|
|
138
|
+
type: "integer",
|
|
139
|
+
description: "How far back to look (default #{TRACE_WINDOW_MINUTES}, max #{MAX_TRACE_WINDOW_MINUTES})"
|
|
140
|
+
},
|
|
141
|
+
min_tokens: { type: "integer", description: "Only traces that used at least this many input + output tokens" },
|
|
142
|
+
min_duration_ms: { type: "integer", description: "Only traces that took at least this long" },
|
|
143
|
+
limit: { type: "integer", description: "At most this many (default #{TRACE_LIMIT}, max #{MAX_TRACE_LIMIT})" }
|
|
144
|
+
}
|
|
145
|
+
}
|
|
146
|
+
},
|
|
147
|
+
{
|
|
148
|
+
name: "traces_get",
|
|
149
|
+
description: "One trace in full: its spans, tool calls with their arguments and results, tokens, estimated " \
|
|
150
|
+
"cost and errors. Values longer than #{MAX_STRING} characters are cut and marked.",
|
|
151
|
+
inputSchema: {
|
|
152
|
+
type: "object",
|
|
153
|
+
properties: { trace_id: { type: "string", description: "The trace's id, its OpenTelemetry trace id, or that id's first 8 characters" } },
|
|
154
|
+
required: [ "trace_id" ]
|
|
155
|
+
}
|
|
156
|
+
}
|
|
157
|
+
].freeze
|
|
158
|
+
|
|
159
|
+
# Raised by a tool for a call the client can correct. Answered as a tool
|
|
160
|
+
# result with isError.
|
|
161
|
+
class ToolError < StandardError; end
|
|
162
|
+
|
|
163
|
+
private
|
|
164
|
+
|
|
165
|
+
def dashboard_tools_list
|
|
166
|
+
ActionAgent.mcp_dashboard_tools? ? DEFINITIONS.map(&:dup) : []
|
|
167
|
+
end
|
|
168
|
+
|
|
169
|
+
def dashboard_tool?(name)
|
|
170
|
+
ActionAgent.mcp_dashboard_tools? && NAMES.include?(name)
|
|
171
|
+
end
|
|
172
|
+
|
|
173
|
+
def dashboard_tool_call(name)
|
|
174
|
+
dashboard_tool_result(dispatch_dashboard_tool(name))
|
|
175
|
+
rescue ToolError, RunSandbox::Refused => e
|
|
176
|
+
message = SecretScrubber.scrub(e.message, dashboard_tool_secrets)
|
|
177
|
+
{ content: [ { type: "text", text: message } ], structuredContent: { error: message }, isError: true }
|
|
178
|
+
end
|
|
179
|
+
|
|
180
|
+
def dispatch_dashboard_tool(name)
|
|
181
|
+
case name
|
|
182
|
+
when "evaluations_list" then evaluations_list_tool
|
|
183
|
+
when "evaluations_get" then evaluations_get_tool
|
|
184
|
+
when "evaluations_run" then evaluations_run_tool
|
|
185
|
+
when "evaluation_runs_get" then evaluation_runs_get_tool
|
|
186
|
+
when "evaluation_runs_compare" then evaluation_runs_compare_tool
|
|
187
|
+
when "traces_search" then traces_search_tool
|
|
188
|
+
when "traces_get" then traces_get_tool
|
|
189
|
+
end
|
|
190
|
+
end
|
|
191
|
+
|
|
192
|
+
# The payload as JSON-compatible data with every credential the owner
|
|
193
|
+
# holds masked.
|
|
194
|
+
def dashboard_tool_result(payload)
|
|
195
|
+
payload = SecretScrubber.scrub(payload.as_json, dashboard_tool_secrets)
|
|
196
|
+
{ content: [ { type: "text", text: payload.to_json } ], structuredContent: payload }
|
|
197
|
+
end
|
|
198
|
+
|
|
199
|
+
def evaluations_list_tool
|
|
200
|
+
scope = dashboard_evaluations
|
|
201
|
+
if (agent = tool_argument(:agent).presence)
|
|
202
|
+
scope = scope.where(agent: resolve_tool_agent!(agent))
|
|
203
|
+
end
|
|
204
|
+
limit = integer_argument(:limit, default: LIST_LIMIT, min: 1, max: MAX_LIST_LIMIT)
|
|
205
|
+
evaluations = scope.includes(:agent, :evaluation_runs, :scenarios).recent.limit(limit)
|
|
206
|
+
|
|
207
|
+
{
|
|
208
|
+
evaluations: evaluations.map do |evaluation|
|
|
209
|
+
summary = EvaluationSerializer.summary(evaluation)
|
|
210
|
+
latest = EvaluationSerializer.recent_runs(evaluation, 1).first
|
|
211
|
+
summary.slice(:id, :name, :agent, :judge_kind, :scenario_suite, :scenario_count, :scenario_groups, :created_at, :run_count)
|
|
212
|
+
.merge(latest_run: latest && EvaluationSerializer.run_summary(latest, number: summary[:run_count]))
|
|
213
|
+
end
|
|
214
|
+
}
|
|
215
|
+
end
|
|
216
|
+
|
|
217
|
+
def evaluations_get_tool
|
|
218
|
+
evaluation = find_tool_evaluation!
|
|
219
|
+
run_count = evaluation.evaluation_runs.count
|
|
220
|
+
runs = evaluation.evaluation_runs.recent.limit(RUN_HISTORY_LIMIT).to_a
|
|
221
|
+
scenarios = evaluation.scenarios.ordered.limit(SCENARIO_LIMIT + 1).map(&:as_json_summary)
|
|
222
|
+
|
|
223
|
+
{
|
|
224
|
+
evaluation: EvaluationSerializer.summary(evaluation),
|
|
225
|
+
scenarios: PayloadBounds.bound(scenarios.first(SCENARIO_LIMIT), max_string: MAX_STRING, max_items: SCENARIO_LIMIT),
|
|
226
|
+
scenarios_truncated: scenarios.size > SCENARIO_LIMIT,
|
|
227
|
+
runs: runs.each_with_index.map { |run, index| EvaluationSerializer.run_summary(run, number: run_count - index) }
|
|
228
|
+
}
|
|
229
|
+
end
|
|
230
|
+
|
|
231
|
+
# Checked in the order Api::EvaluationsController#run checks a request:
|
|
232
|
+
# the execution switch, an executable agent, the execution quota, then
|
|
233
|
+
# the sandbox. Only a scenario suite executes the agent, so a sampling
|
|
234
|
+
# run passes the first three. Each replay counts one execution as it
|
|
235
|
+
# runs (ScenarioEvaluationRunner), so starting the run records none.
|
|
236
|
+
def evaluations_run_tool
|
|
237
|
+
evaluation = find_tool_evaluation!
|
|
238
|
+
if evaluation.scenario_suite?
|
|
239
|
+
raise MCPController::McpError.new("Agent execution is disabled on this dashboard") unless ActionAgent.execution_enabled?
|
|
240
|
+
if unexecutable_scenario_run?(evaluation)
|
|
241
|
+
raise ToolError, "Observed agents are read-only — duplicate this agent to create an executable copy"
|
|
242
|
+
end
|
|
243
|
+
if (denial = ActionAgent.quota_denial(current_owner, :execution)).present?
|
|
244
|
+
raise MCPController::McpError.new(denial.is_a?(Hash) ? denial[:message] || denial["message"] : denial)
|
|
245
|
+
end
|
|
246
|
+
end
|
|
247
|
+
|
|
248
|
+
selection = evaluation_run_selection(tool_arguments)
|
|
249
|
+
if (sandbox = evaluation_run_sandbox(evaluation, tool_argument(:sandbox_id).presence))
|
|
250
|
+
selection[:sandbox_id] = sandbox.session_id
|
|
251
|
+
end
|
|
252
|
+
run = start_evaluation_run(evaluation, selection)
|
|
253
|
+
|
|
254
|
+
{
|
|
255
|
+
evaluation: { id: evaluation.id, name: evaluation.name },
|
|
256
|
+
run: EvaluationSerializer.run_summary(run, number: EvaluationSerializer.run_number(evaluation, run))
|
|
257
|
+
.merge(selection: run.selection, error_message: run.error_message),
|
|
258
|
+
background: evaluation.scenario_suite?
|
|
259
|
+
}
|
|
260
|
+
end
|
|
261
|
+
|
|
262
|
+
def evaluation_runs_get_tool
|
|
263
|
+
evaluation = find_tool_evaluation!
|
|
264
|
+
run = find_tool_run!(evaluation, tool_argument(:run_id), label: "run_id")
|
|
265
|
+
limit = integer_argument(:limit, default: RESULT_LIMIT, min: 1, max: MAX_RESULT_LIMIT)
|
|
266
|
+
|
|
267
|
+
results = run.scenario_results.includes(:scenario, :agent_run).joins(:scenario)
|
|
268
|
+
.order(EvaluationScenario.arel_table[:position], EvaluationScenario.arel_table[:id], :model)
|
|
269
|
+
results = results.where.not(status: :passed) if boolean_argument(:failed_only)
|
|
270
|
+
total = results.count
|
|
271
|
+
page = results.limit(limit).to_a
|
|
272
|
+
recorded = recorded_trace_ids(page.filter_map { |result| result.agent_run&.trace_id })
|
|
273
|
+
rows = page.map { |result| result_row(result, recorded) }
|
|
274
|
+
|
|
275
|
+
{
|
|
276
|
+
evaluation: { id: evaluation.id, name: evaluation.name, agent: evaluation.agent.slug },
|
|
277
|
+
run: EvaluationSerializer.run(run, number: EvaluationSerializer.run_number(evaluation, run)),
|
|
278
|
+
fix_items: PayloadBounds.bound(EvaluationSerializer.fix_items(run, links: dashboard_links(run)), max_string: MAX_STRING),
|
|
279
|
+
results: rows.map { |row| PayloadBounds.bound(row, max_string: MAX_STRING, max_items: 20) },
|
|
280
|
+
results_total: total,
|
|
281
|
+
results_omitted: [ total - rows.size, 0 ].max
|
|
282
|
+
}
|
|
283
|
+
end
|
|
284
|
+
|
|
285
|
+
def evaluation_runs_compare_tool
|
|
286
|
+
evaluation = find_tool_evaluation!
|
|
287
|
+
base_id = tool_argument(:base_run_id)
|
|
288
|
+
head_id = tool_argument(:head_run_id)
|
|
289
|
+
head = find_tool_run!(evaluation, head_id, label: "head_run_id")
|
|
290
|
+
base =
|
|
291
|
+
if base_id.present?
|
|
292
|
+
find_tool_run!(evaluation, base_id, label: "base_run_id")
|
|
293
|
+
else
|
|
294
|
+
evaluation.evaluation_runs.where("created_at < ? OR (created_at = ? AND id < ?)", head.created_at, head.created_at, head.id)
|
|
295
|
+
.recent.first or raise ToolError, "Run #{head.id} is the evaluation's first run; there is nothing to compare it with"
|
|
296
|
+
end
|
|
297
|
+
|
|
298
|
+
base_rows = comparison_rows(base)
|
|
299
|
+
head_rows = comparison_rows(head)
|
|
300
|
+
changes = (base_rows.keys | head_rows.keys).filter_map do |key|
|
|
301
|
+
before = base_rows[key]
|
|
302
|
+
after = head_rows[key]
|
|
303
|
+
change = comparison_change(before, after)
|
|
304
|
+
next if change == "unchanged_pass"
|
|
305
|
+
|
|
306
|
+
{ scenario_key: key[0], model: key[1], change: change, before: before&.status, after: after&.status,
|
|
307
|
+
fault: after&.fault, result_id: after&.id }
|
|
308
|
+
end
|
|
309
|
+
|
|
310
|
+
{
|
|
311
|
+
evaluation: { id: evaluation.id, name: evaluation.name },
|
|
312
|
+
base_run: EvaluationSerializer.run_summary(base, number: EvaluationSerializer.run_number(evaluation, base)),
|
|
313
|
+
head_run: EvaluationSerializer.run_summary(head, number: EvaluationSerializer.run_number(evaluation, head)),
|
|
314
|
+
counts: changes.group_by { |change| change[:change] }.transform_values(&:size),
|
|
315
|
+
changes: PayloadBounds.bound(changes, max_items: MAX_RESULT_LIMIT)
|
|
316
|
+
}
|
|
317
|
+
end
|
|
318
|
+
|
|
319
|
+
def traces_search_tool
|
|
320
|
+
window = integer_argument(:since_minutes, default: TRACE_WINDOW_MINUTES, min: 1, max: MAX_TRACE_WINDOW_MINUTES)
|
|
321
|
+
limit = integer_argument(:limit, default: TRACE_LIMIT, min: 1, max: MAX_TRACE_LIMIT)
|
|
322
|
+
model = ActionAgent.trace_model
|
|
323
|
+
|
|
324
|
+
scope = owned_traces.for_date_range(window.minutes.ago, Time.current)
|
|
325
|
+
if (agent = tool_argument(:agent).presence)
|
|
326
|
+
by_slug = owner_agents.find_by(slug: agent.to_s)
|
|
327
|
+
scope = by_slug ? scope.where(agent_id: by_slug.id).or(scope.for_agent(agent.to_s)) : scope.for_agent(agent.to_s)
|
|
328
|
+
end
|
|
329
|
+
scope = scope.for_service(tool_argument(:service).to_s) if tool_argument(:service).present?
|
|
330
|
+
case tool_argument(:status).to_s.downcase
|
|
331
|
+
when "error" then scope = scope.with_errors
|
|
332
|
+
when "ok" then scope = scope.where.not(status: model::STATUS_ERROR)
|
|
333
|
+
end
|
|
334
|
+
if (min_tokens = integer_argument(:min_tokens, default: nil))
|
|
335
|
+
table = model.quoted_table_name
|
|
336
|
+
scope = scope.where(
|
|
337
|
+
"COALESCE(#{table}.total_input_tokens, 0) + COALESCE(#{table}.total_output_tokens, 0) >= ?", min_tokens
|
|
338
|
+
)
|
|
339
|
+
end
|
|
340
|
+
if (min_duration = integer_argument(:min_duration_ms, default: nil))
|
|
341
|
+
scope = scope.where(model.arel_table[:total_duration_ms].gteq(min_duration))
|
|
342
|
+
end
|
|
343
|
+
|
|
344
|
+
traces = scope.recent.limit(limit + 1).to_a
|
|
345
|
+
|
|
346
|
+
{
|
|
347
|
+
traces: traces.first(limit).map { |trace| PayloadBounds.bound(TelemetryTraceSerializer.row(trace), max_string: MAX_STRING) },
|
|
348
|
+
truncated: traces.size > limit,
|
|
349
|
+
window_minutes: window
|
|
350
|
+
}
|
|
351
|
+
end
|
|
352
|
+
|
|
353
|
+
# Looked up as Api::TraceReportsController#show looks a trace up: the
|
|
354
|
+
# OpenTelemetry trace id, then a prefix of it, then the record id.
|
|
355
|
+
def traces_get_tool
|
|
356
|
+
id = tool_argument(:trace_id).to_s
|
|
357
|
+
raise MCPController::McpError.new("Missing required argument: trace_id", MCPController::JSONRPC_INVALID_PARAMS) if id.blank?
|
|
358
|
+
|
|
359
|
+
scope = owned_traces
|
|
360
|
+
trace = scope.find_by(trace_id: id) ||
|
|
361
|
+
scope.where("trace_id LIKE ?", "#{ActionAgent.trace_model.sanitize_sql_like(id)}%").first ||
|
|
362
|
+
(id.match?(/\A\d+\z/) && scope.find_by(id: id)) or
|
|
363
|
+
raise ToolError, "No trace #{id.truncate(64)} was found"
|
|
364
|
+
|
|
365
|
+
detail = TelemetryTraceSerializer.detail(trace)
|
|
366
|
+
spans = detail.delete(:spans)
|
|
367
|
+
failed_spans = spans.select { |span| span[:error] }.map do |span|
|
|
368
|
+
span.slice(:span_id, :name, :type).merge(message: span.dig(:attributes, "error.message"))
|
|
369
|
+
end
|
|
370
|
+
|
|
371
|
+
payload = detail.merge(
|
|
372
|
+
tool_calls: trace.tool_usage,
|
|
373
|
+
failed_spans: failed_spans,
|
|
374
|
+
spans: spans.first(MAX_TRACE_SPANS),
|
|
375
|
+
spans_total: spans.size,
|
|
376
|
+
spans_omitted: [ spans.size - MAX_TRACE_SPANS, 0 ].max
|
|
377
|
+
)
|
|
378
|
+
PayloadBounds.bound(payload, max_string: MAX_STRING, max_items: MAX_TRACE_SPANS)
|
|
379
|
+
end
|
|
380
|
+
|
|
381
|
+
# `trace_id` names the result's telemetry trace for traces_get, and is
|
|
382
|
+
# nil when telemetry recorded none for the replay.
|
|
383
|
+
def result_row(result, recorded_trace_ids)
|
|
384
|
+
{
|
|
385
|
+
id: result.id,
|
|
386
|
+
scenario_key: result.evaluated_scenario["key"],
|
|
387
|
+
group: result.evaluated_scenario["group"],
|
|
388
|
+
prompt: result.evaluated_scenario["prompt"],
|
|
389
|
+
model: result.model,
|
|
390
|
+
status: result.status,
|
|
391
|
+
score: result.score,
|
|
392
|
+
fault: result.fault,
|
|
393
|
+
recommendation: result.recommendation,
|
|
394
|
+
diagnosis: result.evaluation_diagnosis,
|
|
395
|
+
output: result.output,
|
|
396
|
+
tool_calls: result.tool_calls,
|
|
397
|
+
error_message: result.error_message,
|
|
398
|
+
duration_ms: result.duration_ms,
|
|
399
|
+
input_tokens: result.input_tokens,
|
|
400
|
+
output_tokens: result.output_tokens,
|
|
401
|
+
cost: result.cost&.to_f,
|
|
402
|
+
agent_run_id: result.agent_run_id,
|
|
403
|
+
trace_id: recorded_trace_ids.include?(result.agent_run&.trace_id) ? result.agent_run.trace_id : nil
|
|
404
|
+
}
|
|
405
|
+
end
|
|
406
|
+
|
|
407
|
+
# The subset of +trace_ids+ the owner's telemetry holds.
|
|
408
|
+
def recorded_trace_ids(trace_ids)
|
|
409
|
+
return Set.new if trace_ids.empty?
|
|
410
|
+
|
|
411
|
+
owned_traces.where(trace_id: trace_ids).pluck(:trace_id).to_set
|
|
412
|
+
end
|
|
413
|
+
|
|
414
|
+
# The run's results keyed by [scenario key, model].
|
|
415
|
+
def comparison_rows(run)
|
|
416
|
+
run.scenario_results.includes(:scenario).index_by { |result| [ result.evaluated_scenario["key"], result.model ] }
|
|
417
|
+
end
|
|
418
|
+
|
|
419
|
+
# One of: removed, added_pass, added_fail, unchanged_pass, regressed,
|
|
420
|
+
# fixed, still_failing.
|
|
421
|
+
def comparison_change(before, after)
|
|
422
|
+
return "removed" if after.nil?
|
|
423
|
+
return after.passed? ? "added_pass" : "added_fail" if before.nil?
|
|
424
|
+
return after.passed? ? "unchanged_pass" : "regressed" if before.passed?
|
|
425
|
+
|
|
426
|
+
after.passed? ? "fixed" : "still_failing"
|
|
427
|
+
end
|
|
428
|
+
|
|
429
|
+
# Fix item links at the facade's absolute mount, since a client has no
|
|
430
|
+
# dashboard page to resolve a relative path against.
|
|
431
|
+
def dashboard_links(run)
|
|
432
|
+
run.report_links(mount: "#{request.base_url}#{request.script_name}")
|
|
433
|
+
end
|
|
434
|
+
|
|
435
|
+
def dashboard_evaluations
|
|
436
|
+
Evaluation.joins(:agent).where(agent: owner_agents)
|
|
437
|
+
end
|
|
438
|
+
|
|
439
|
+
def find_tool_evaluation!
|
|
440
|
+
id = tool_argument(:evaluation_id)
|
|
441
|
+
raise MCPController::McpError.new("Missing required argument: evaluation_id", MCPController::JSONRPC_INVALID_PARAMS) if id.blank?
|
|
442
|
+
|
|
443
|
+
dashboard_evaluations.find_by(id: id.to_s) or raise ToolError, "No evaluation #{id.to_s.truncate(32)} was found"
|
|
444
|
+
end
|
|
445
|
+
|
|
446
|
+
# The evaluation's run +id+ names, or its latest run when +id+ is blank.
|
|
447
|
+
def find_tool_run!(evaluation, id, label:)
|
|
448
|
+
if id.blank?
|
|
449
|
+
evaluation.evaluation_runs.recent.first or raise ToolError, "Evaluation #{evaluation.id} has not been run yet"
|
|
450
|
+
else
|
|
451
|
+
evaluation.evaluation_runs.find_by(id: id.to_s) or
|
|
452
|
+
raise ToolError, "No run #{id.to_s.truncate(32)} of evaluation #{evaluation.id} was found (#{label})"
|
|
453
|
+
end
|
|
454
|
+
end
|
|
455
|
+
|
|
456
|
+
def resolve_tool_agent!(value)
|
|
457
|
+
agents = owner_agents
|
|
458
|
+
agents.find_by(slug: value.to_s) || (value.to_s.match?(/\A\d+\z/) && agents.find_by(id: value.to_s)) or
|
|
459
|
+
raise ToolError, "No agent #{value.to_s.truncate(64)} was found"
|
|
460
|
+
end
|
|
461
|
+
|
|
462
|
+
def tool_arguments
|
|
463
|
+
@tool_arguments ||= begin
|
|
464
|
+
arguments = params.dig(:params, :arguments)
|
|
465
|
+
arguments.is_a?(ActionController::Parameters) ? arguments : ActionController::Parameters.new({})
|
|
466
|
+
end
|
|
467
|
+
end
|
|
468
|
+
|
|
469
|
+
def tool_argument(name)
|
|
470
|
+
tool_arguments[name]
|
|
471
|
+
end
|
|
472
|
+
|
|
473
|
+
# An integer argument clamped into [min, max], or +default+ when absent
|
|
474
|
+
# or not a number.
|
|
475
|
+
def integer_argument(name, default:, min: nil, max: nil)
|
|
476
|
+
raw = tool_argument(name)
|
|
477
|
+
value = raw.is_a?(Integer) || raw.to_s.match?(/\A-?\d+\z/) ? raw.to_i : default
|
|
478
|
+
return value if value.nil?
|
|
479
|
+
|
|
480
|
+
value = [ value, min ].max if min
|
|
481
|
+
value = [ value, max ].min if max
|
|
482
|
+
value
|
|
483
|
+
end
|
|
484
|
+
|
|
485
|
+
def boolean_argument(name)
|
|
486
|
+
ActiveModel::Type::Boolean.new.cast(tool_argument(name)) == true
|
|
487
|
+
end
|
|
488
|
+
|
|
489
|
+
# Every credential the owner holds, masked out of each tool's output:
|
|
490
|
+
# this key's token, provider keys, the GitHub token, and each checkout
|
|
491
|
+
# sandbox's runtime token. None of them belongs in these payloads; this
|
|
492
|
+
# keeps one that leaked into a recorded output or a trace from being
|
|
493
|
+
# handed on. A failed lookup masks nothing rather than failing the call.
|
|
494
|
+
def dashboard_tool_secrets
|
|
495
|
+
@dashboard_tool_secrets ||= [
|
|
496
|
+
@api_key&.token,
|
|
497
|
+
*owned(ProviderKey).limit(SECRET_LOOKUP_LIMIT).pluck(:credential, :api_key).flatten,
|
|
498
|
+
*owned(GithubConnection).limit(SECRET_LOOKUP_LIMIT).pluck(:access_token),
|
|
499
|
+
*owned(SandboxSession).where.not(runtime_mcp_token: nil).order(id: :desc).limit(SECRET_LOOKUP_LIMIT).pluck(:runtime_mcp_token)
|
|
500
|
+
].compact
|
|
501
|
+
rescue StandardError => e
|
|
502
|
+
Rails.logger.warn("[ActionAgent] MCP secret lookup failed: #{e.class}: #{e.message}")
|
|
503
|
+
@dashboard_tool_secrets = [ @api_key&.token ].compact
|
|
504
|
+
end
|
|
505
|
+
end
|
|
506
|
+
end
|
|
507
|
+
end
|
|
@@ -0,0 +1,65 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module ActionAgent
|
|
4
|
+
module Api
|
|
5
|
+
# A run against a checkout sandbox the agent does not name: `sandbox_id`
|
|
6
|
+
# on an evaluation run or a runner run makes that one run reach the
|
|
7
|
+
# sandbox's app runtime, as if the agent listed "sandbox:<session_id>" in
|
|
8
|
+
# mcp_servers, without saving it there. It is how a change being tried in
|
|
9
|
+
# a checkout is evaluated before anyone edits the agent.
|
|
10
|
+
#
|
|
11
|
+
# The sandbox has to be the caller's (found the way CodeSessionsController
|
|
12
|
+
# finds one: owned, and in the current account), a checkout, live, and
|
|
13
|
+
# owned by the agent's owner as well, since the dispatcher resolves the
|
|
14
|
+
# runtime among that owner's sessions only (see
|
|
15
|
+
# SandboxSession.runtime_server_entry). Anything else is refused with a
|
|
16
|
+
# 422 saying which. Only the key and the checkout it names are passed on
|
|
17
|
+
# and recorded; the runtime's token stays on the session.
|
|
18
|
+
module RunSandbox
|
|
19
|
+
extend ActiveSupport::Concern
|
|
20
|
+
|
|
21
|
+
class Refused < StandardError; end
|
|
22
|
+
|
|
23
|
+
included do
|
|
24
|
+
rescue_from Refused do |error|
|
|
25
|
+
render json: { error: error.message, code: "sandbox_refused" }, status: :unprocessable_entity
|
|
26
|
+
end
|
|
27
|
+
end
|
|
28
|
+
|
|
29
|
+
private
|
|
30
|
+
|
|
31
|
+
# The caller's live checkout sandbox that +sandbox_id+ names for a run
|
|
32
|
+
# of +agent+, or nil when none was asked for. Raises Refused.
|
|
33
|
+
#
|
|
34
|
+
# @return [SandboxSession, nil]
|
|
35
|
+
def run_sandbox_for(agent, sandbox_id)
|
|
36
|
+
return nil if sandbox_id.blank?
|
|
37
|
+
raise Refused, "sandbox_id must be a sandbox's session id" unless sandbox_id.is_a?(String)
|
|
38
|
+
|
|
39
|
+
scope = owned(SandboxSession)
|
|
40
|
+
scope = scope.where(account_id: current_account.id) if current_account
|
|
41
|
+
# Unknown and someone else's read the same: whether a session id
|
|
42
|
+
# exists is not the caller's to learn.
|
|
43
|
+
sandbox = scope.find_by(session_id: sandbox_id) or
|
|
44
|
+
raise Refused, "No sandbox #{sandbox_id.truncate(64)} of yours was found"
|
|
45
|
+
label = "Sandbox #{sandbox.session_id}"
|
|
46
|
+
|
|
47
|
+
unless sandbox.app_runtime?
|
|
48
|
+
raise Refused, "#{label} is a #{sandbox.sandbox_type} sandbox; only a checkout (app_runtime) sandbox " \
|
|
49
|
+
"serves an app's tools"
|
|
50
|
+
end
|
|
51
|
+
# Ready (or running a task), and serving: a sandbox still booting
|
|
52
|
+
# can already carry an endpoint that is not answering yet.
|
|
53
|
+
if !(sandbox.ready? || sandbox.running?) || sandbox.runtime_server_entry.nil?
|
|
54
|
+
state = sandbox.ready? && !sandbox.active? ? "expired" : sandbox.status
|
|
55
|
+
raise Refused, "#{label} is #{state}; run against a checkout sandbox once it is ready"
|
|
56
|
+
end
|
|
57
|
+
unless SandboxSession.runtime_server_entry(sandbox.runtime_server_key, owner: agent.owner)
|
|
58
|
+
raise Refused, "#{label} does not belong to the owner of #{agent.name}, so this agent's runs cannot reach it"
|
|
59
|
+
end
|
|
60
|
+
|
|
61
|
+
sandbox
|
|
62
|
+
end
|
|
63
|
+
end
|
|
64
|
+
end
|
|
65
|
+
end
|