actionagent 1.7.2 → 1.8.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (59) hide show
  1. checksums.yaml +4 -4
  2. data/app/assets/builds/action_agent.css +1 -1
  3. data/app/assets/builds/action_agent.js +59 -54
  4. data/app/controllers/action_agent/api/agents_controller.rb +19 -3
  5. data/app/controllers/action_agent/api/code_sessions_controller.rb +156 -0
  6. data/app/controllers/action_agent/api/evaluations_controller.rb +27 -130
  7. data/app/controllers/action_agent/api/github_connections_controller.rb +147 -0
  8. data/app/controllers/action_agent/api/mcp_controller.rb +38 -5
  9. data/app/controllers/action_agent/api/mcp_servers_controller.rb +10 -1
  10. data/app/controllers/action_agent/api/provider_keys_controller.rb +54 -3
  11. data/app/controllers/action_agent/api/provider_models_controller.rb +10 -6
  12. data/app/controllers/action_agent/api/sandboxes_controller.rb +97 -6
  13. data/app/controllers/concerns/action_agent/api/evaluation_run_starting.rb +93 -0
  14. data/app/controllers/concerns/action_agent/api/mcp_dashboard_tools.rb +507 -0
  15. data/app/controllers/concerns/action_agent/api/run_sandbox.rb +65 -0
  16. data/app/jobs/action_agent/code_session_job.rb +166 -0
  17. data/app/jobs/action_agent/sandbox_cleanup_job.rb +80 -11
  18. data/app/jobs/action_agent/sandbox_provision_job.rb +122 -14
  19. data/app/jobs/action_agent/sandbox_run_job.rb +10 -3
  20. data/app/models/action_agent/agent.rb +16 -6
  21. data/app/models/action_agent/agent_run.rb +20 -1
  22. data/app/models/action_agent/code_session.rb +141 -0
  23. data/app/models/action_agent/evaluation_run.rb +23 -1
  24. data/app/models/action_agent/github_connection.rb +75 -0
  25. data/app/models/action_agent/provider_key.rb +142 -10
  26. data/app/models/action_agent/sandbox_session.rb +193 -17
  27. data/app/serializers/action_agent/evaluation_serializer.rb +118 -0
  28. data/app/serializers/action_agent/telemetry_trace_serializer.rb +10 -2
  29. data/app/services/action_agent/agent_execution_service.rb +4 -2
  30. data/app/services/action_agent/agent_tool_roster.rb +20 -9
  31. data/app/services/action_agent/claude_code_auth.rb +86 -0
  32. data/app/services/action_agent/dashboard_assistant_service.rb +47 -5
  33. data/app/services/action_agent/evaluation_tool_resolver.rb +18 -0
  34. data/app/services/action_agent/github_client.rb +111 -0
  35. data/app/services/action_agent/local_sandbox_backend.rb +1689 -0
  36. data/app/services/action_agent/local_sandbox_databases.rb +257 -0
  37. data/app/services/action_agent/mcp_client.rb +5 -1
  38. data/app/services/action_agent/mcp_tool_dispatcher.rb +137 -17
  39. data/app/services/action_agent/mock_sandbox_backend.rb +39 -0
  40. data/app/services/action_agent/ollama_host_probe.rb +75 -0
  41. data/app/services/action_agent/payload_bounds.rb +36 -0
  42. data/app/services/action_agent/sandbox_manifest.rb +67 -0
  43. data/app/services/action_agent/sandbox_orchestrator.rb +69 -14
  44. data/app/services/action_agent/scenario_evaluation_runner.rb +50 -4
  45. data/app/services/action_agent/secret_scrubber.rb +37 -0
  46. data/app/services/action_agent/tool_discovery.rb +19 -5
  47. data/config/routes.rb +22 -3
  48. data/lib/action_agent/engine.rb +1 -0
  49. data/lib/action_agent/version.rb +1 -1
  50. data/lib/action_agent.rb +147 -3
  51. data/lib/generators/action_agent/install_generator.rb +30 -3
  52. data/lib/generators/action_agent/templates/action_agent.rb.erb +44 -0
  53. data/lib/generators/action_agent/templates/add_provider_key_api_key.rb.erb +25 -0
  54. data/lib/generators/action_agent/templates/create_active_agent_code_sessions.rb.erb +59 -0
  55. data/lib/generators/action_agent/templates/create_active_agent_dashboard_tables.rb.erb +2 -0
  56. data/lib/generators/action_agent/templates/create_active_agent_github_connections.rb.erb +61 -0
  57. data/lib/tasks/claude_code.rake +16 -0
  58. data/lib/tasks/sandbox.rake +26 -0
  59. metadata +23 -1
@@ -0,0 +1,507 @@
1
+ # frozen_string_literal: true
2
+
3
+ module ActionAgent
4
+ module Api
5
+ # The dashboard's own evaluation and telemetry tools, served by the MCP
6
+ # facade (Api::MCPController) so a client's coding harness can edit an
7
+ # agent in its own checkout, run the agent's evaluations, read the runs'
8
+ # fix items and failing traces, and iterate. The harness brings its own
9
+ # model; the dashboard only answers these calls.
10
+ #
11
+ # Each tool reads under the key's owner the way the dashboard's JSON API
12
+ # reads under the signed-in owner: evaluations of the agents the owner
13
+ # can reach (Api::EvaluationsController), traces of the owner's tenant
14
+ # (Api::TraceReportsController). A record outside that scope reads as
15
+ # nonexistent.
16
+ #
17
+ # Names are a noun family followed by a verb (`evaluations_list`,
18
+ # `traces_get`). Host schema tools are always `find_`, `count_` or `get_`
19
+ # plus a model name, and agent tools are `run_<slug>`, so no host model or
20
+ # agent slug can produce one of these names.
21
+ #
22
+ # A call the client can correct (an unknown id, a sandbox it cannot use)
23
+ # comes back as a tool result with isError. The execution switch and the
24
+ # execution quota answer as JSON-RPC errors, as they do for `run_<slug>`.
25
+ module MCPDashboardTools
26
+ extend ActiveSupport::Concern
27
+
28
+ include EvaluationRunStarting
29
+
30
+ NAMES = %w[
31
+ evaluations_list evaluations_get evaluations_run evaluation_runs_get evaluation_runs_compare
32
+ traces_search traces_get
33
+ ].freeze
34
+
35
+ LIST_LIMIT = 20
36
+ MAX_LIST_LIMIT = 50
37
+ SCENARIO_LIMIT = 200
38
+ RUN_HISTORY_LIMIT = 10
39
+ RESULT_LIMIT = 50
40
+ MAX_RESULT_LIMIT = 200
41
+ TRACE_WINDOW_MINUTES = 60 * 24
42
+ MAX_TRACE_WINDOW_MINUTES = TraceReportsController::MAX_WINDOW_MINUTES
43
+ TRACE_LIMIT = 20
44
+ MAX_TRACE_LIMIT = 100
45
+ MAX_TRACE_SPANS = 100
46
+ # Strings in a result row or a span, cut past this many characters.
47
+ MAX_STRING = 1_000
48
+ # Owned credentials read for scrubbing, per kind.
49
+ SECRET_LOOKUP_LIMIT = 100
50
+
51
+ SELECTION_PROPERTIES = {
52
+ scenario_ids: { type: "array", items: { type: "integer" }, description: "Replay only these scenarios, by id" },
53
+ keys: { type: "array", items: { type: "string" }, description: "Replay only these scenarios, by key" },
54
+ group: { type: "string", description: "Replay only this scenario group" },
55
+ models: { type: "array", items: { type: "string" }, description: "Candidate models to compare; the agent's own model when omitted" },
56
+ sandbox_id: {
57
+ type: "string",
58
+ description: "A ready checkout sandbox's session id: every replay reaches that checkout's app tools, " \
59
+ "without the agent being edited"
60
+ }
61
+ }.freeze
62
+
63
+ DEFINITIONS = [
64
+ {
65
+ name: "evaluations_list",
66
+ description: "List the evaluations of this key's agents, newest first, each with its latest run's status and " \
67
+ "score. Filter to one agent by slug or id.",
68
+ inputSchema: {
69
+ type: "object",
70
+ properties: {
71
+ agent: { type: "string", description: "An agent's slug or id" },
72
+ limit: { type: "integer", description: "At most this many (default #{LIST_LIMIT}, max #{MAX_LIST_LIMIT})" }
73
+ }
74
+ }
75
+ },
76
+ {
77
+ name: "evaluations_get",
78
+ description: "One evaluation: its criteria, its scenarios (for a scenario suite) and its #{RUN_HISTORY_LIMIT} " \
79
+ "most recent runs.",
80
+ inputSchema: {
81
+ type: "object",
82
+ properties: { evaluation_id: { type: "integer", description: "The evaluation's id" } },
83
+ required: [ "evaluation_id" ]
84
+ }
85
+ },
86
+ {
87
+ name: "evaluations_run",
88
+ description: "Start a run of an evaluation. A scenario suite replays each scenario through the agent in the " \
89
+ "background: the run comes back pending, so poll evaluation_runs_get with its id. A sampling " \
90
+ "evaluation scores recorded generations and finishes before this returns.",
91
+ inputSchema: {
92
+ type: "object",
93
+ properties: { evaluation_id: { type: "integer", description: "The evaluation's id" } }.merge(SELECTION_PROPERTIES),
94
+ required: [ "evaluation_id" ]
95
+ }
96
+ },
97
+ {
98
+ name: "evaluation_runs_get",
99
+ description: "One evaluation run: status, scores, usage, its fix items (what to change, and where) and its " \
100
+ "per-scenario, per-model results, each naming its telemetry trace (for traces_get) when one " \
101
+ "was recorded. Defaults to the evaluation's latest run.",
102
+ inputSchema: {
103
+ type: "object",
104
+ properties: {
105
+ evaluation_id: { type: "integer", description: "The evaluation's id" },
106
+ run_id: { type: "integer", description: "The run's id; the latest run when omitted" },
107
+ failed_only: { type: "boolean", description: "Only results that did not pass" },
108
+ limit: { type: "integer", description: "At most this many results (default #{RESULT_LIMIT}, max #{MAX_RESULT_LIMIT})" }
109
+ },
110
+ required: [ "evaluation_id" ]
111
+ }
112
+ },
113
+ {
114
+ name: "evaluation_runs_compare",
115
+ description: "Compare two runs of one evaluation, scenario by scenario and model by model: which results " \
116
+ "were fixed, which regressed, which still fail. Defaults to the latest run against the one before it.",
117
+ inputSchema: {
118
+ type: "object",
119
+ properties: {
120
+ evaluation_id: { type: "integer", description: "The evaluation's id" },
121
+ base_run_id: { type: "integer", description: "The earlier run's id" },
122
+ head_run_id: { type: "integer", description: "The later run's id" }
123
+ },
124
+ required: [ "evaluation_id" ]
125
+ }
126
+ },
127
+ {
128
+ name: "traces_search",
129
+ description: "Search telemetry traces, newest first. Returns summary rows (no spans); read one in full with " \
130
+ "traces_get.",
131
+ inputSchema: {
132
+ type: "object",
133
+ properties: {
134
+ agent: { type: "string", description: "An agent class name (as traces report it) or a dashboard agent's slug" },
135
+ status: { type: "string", enum: %w[error ok], description: "Only failed traces, or only successful ones" },
136
+ service: { type: "string", description: "The reporting service's name" },
137
+ since_minutes: {
138
+ type: "integer",
139
+ description: "How far back to look (default #{TRACE_WINDOW_MINUTES}, max #{MAX_TRACE_WINDOW_MINUTES})"
140
+ },
141
+ min_tokens: { type: "integer", description: "Only traces that used at least this many input + output tokens" },
142
+ min_duration_ms: { type: "integer", description: "Only traces that took at least this long" },
143
+ limit: { type: "integer", description: "At most this many (default #{TRACE_LIMIT}, max #{MAX_TRACE_LIMIT})" }
144
+ }
145
+ }
146
+ },
147
+ {
148
+ name: "traces_get",
149
+ description: "One trace in full: its spans, tool calls with their arguments and results, tokens, estimated " \
150
+ "cost and errors. Values longer than #{MAX_STRING} characters are cut and marked.",
151
+ inputSchema: {
152
+ type: "object",
153
+ properties: { trace_id: { type: "string", description: "The trace's id, its OpenTelemetry trace id, or that id's first 8 characters" } },
154
+ required: [ "trace_id" ]
155
+ }
156
+ }
157
+ ].freeze
158
+
159
+ # Raised by a tool for a call the client can correct. Answered as a tool
160
+ # result with isError.
161
+ class ToolError < StandardError; end
162
+
163
+ private
164
+
165
+ def dashboard_tools_list
166
+ ActionAgent.mcp_dashboard_tools? ? DEFINITIONS.map(&:dup) : []
167
+ end
168
+
169
+ def dashboard_tool?(name)
170
+ ActionAgent.mcp_dashboard_tools? && NAMES.include?(name)
171
+ end
172
+
173
+ def dashboard_tool_call(name)
174
+ dashboard_tool_result(dispatch_dashboard_tool(name))
175
+ rescue ToolError, RunSandbox::Refused => e
176
+ message = SecretScrubber.scrub(e.message, dashboard_tool_secrets)
177
+ { content: [ { type: "text", text: message } ], structuredContent: { error: message }, isError: true }
178
+ end
179
+
180
+ def dispatch_dashboard_tool(name)
181
+ case name
182
+ when "evaluations_list" then evaluations_list_tool
183
+ when "evaluations_get" then evaluations_get_tool
184
+ when "evaluations_run" then evaluations_run_tool
185
+ when "evaluation_runs_get" then evaluation_runs_get_tool
186
+ when "evaluation_runs_compare" then evaluation_runs_compare_tool
187
+ when "traces_search" then traces_search_tool
188
+ when "traces_get" then traces_get_tool
189
+ end
190
+ end
191
+
192
+ # The payload as JSON-compatible data with every credential the owner
193
+ # holds masked.
194
+ def dashboard_tool_result(payload)
195
+ payload = SecretScrubber.scrub(payload.as_json, dashboard_tool_secrets)
196
+ { content: [ { type: "text", text: payload.to_json } ], structuredContent: payload }
197
+ end
198
+
199
+ def evaluations_list_tool
200
+ scope = dashboard_evaluations
201
+ if (agent = tool_argument(:agent).presence)
202
+ scope = scope.where(agent: resolve_tool_agent!(agent))
203
+ end
204
+ limit = integer_argument(:limit, default: LIST_LIMIT, min: 1, max: MAX_LIST_LIMIT)
205
+ evaluations = scope.includes(:agent, :evaluation_runs, :scenarios).recent.limit(limit)
206
+
207
+ {
208
+ evaluations: evaluations.map do |evaluation|
209
+ summary = EvaluationSerializer.summary(evaluation)
210
+ latest = EvaluationSerializer.recent_runs(evaluation, 1).first
211
+ summary.slice(:id, :name, :agent, :judge_kind, :scenario_suite, :scenario_count, :scenario_groups, :created_at, :run_count)
212
+ .merge(latest_run: latest && EvaluationSerializer.run_summary(latest, number: summary[:run_count]))
213
+ end
214
+ }
215
+ end
216
+
217
+ def evaluations_get_tool
218
+ evaluation = find_tool_evaluation!
219
+ run_count = evaluation.evaluation_runs.count
220
+ runs = evaluation.evaluation_runs.recent.limit(RUN_HISTORY_LIMIT).to_a
221
+ scenarios = evaluation.scenarios.ordered.limit(SCENARIO_LIMIT + 1).map(&:as_json_summary)
222
+
223
+ {
224
+ evaluation: EvaluationSerializer.summary(evaluation),
225
+ scenarios: PayloadBounds.bound(scenarios.first(SCENARIO_LIMIT), max_string: MAX_STRING, max_items: SCENARIO_LIMIT),
226
+ scenarios_truncated: scenarios.size > SCENARIO_LIMIT,
227
+ runs: runs.each_with_index.map { |run, index| EvaluationSerializer.run_summary(run, number: run_count - index) }
228
+ }
229
+ end
230
+
231
+ # Checked in the order Api::EvaluationsController#run checks a request:
232
+ # the execution switch, an executable agent, the execution quota, then
233
+ # the sandbox. Only a scenario suite executes the agent, so a sampling
234
+ # run passes the first three. Each replay counts one execution as it
235
+ # runs (ScenarioEvaluationRunner), so starting the run records none.
236
+ def evaluations_run_tool
237
+ evaluation = find_tool_evaluation!
238
+ if evaluation.scenario_suite?
239
+ raise MCPController::McpError.new("Agent execution is disabled on this dashboard") unless ActionAgent.execution_enabled?
240
+ if unexecutable_scenario_run?(evaluation)
241
+ raise ToolError, "Observed agents are read-only — duplicate this agent to create an executable copy"
242
+ end
243
+ if (denial = ActionAgent.quota_denial(current_owner, :execution)).present?
244
+ raise MCPController::McpError.new(denial.is_a?(Hash) ? denial[:message] || denial["message"] : denial)
245
+ end
246
+ end
247
+
248
+ selection = evaluation_run_selection(tool_arguments)
249
+ if (sandbox = evaluation_run_sandbox(evaluation, tool_argument(:sandbox_id).presence))
250
+ selection[:sandbox_id] = sandbox.session_id
251
+ end
252
+ run = start_evaluation_run(evaluation, selection)
253
+
254
+ {
255
+ evaluation: { id: evaluation.id, name: evaluation.name },
256
+ run: EvaluationSerializer.run_summary(run, number: EvaluationSerializer.run_number(evaluation, run))
257
+ .merge(selection: run.selection, error_message: run.error_message),
258
+ background: evaluation.scenario_suite?
259
+ }
260
+ end
261
+
262
+ def evaluation_runs_get_tool
263
+ evaluation = find_tool_evaluation!
264
+ run = find_tool_run!(evaluation, tool_argument(:run_id), label: "run_id")
265
+ limit = integer_argument(:limit, default: RESULT_LIMIT, min: 1, max: MAX_RESULT_LIMIT)
266
+
267
+ results = run.scenario_results.includes(:scenario, :agent_run).joins(:scenario)
268
+ .order(EvaluationScenario.arel_table[:position], EvaluationScenario.arel_table[:id], :model)
269
+ results = results.where.not(status: :passed) if boolean_argument(:failed_only)
270
+ total = results.count
271
+ page = results.limit(limit).to_a
272
+ recorded = recorded_trace_ids(page.filter_map { |result| result.agent_run&.trace_id })
273
+ rows = page.map { |result| result_row(result, recorded) }
274
+
275
+ {
276
+ evaluation: { id: evaluation.id, name: evaluation.name, agent: evaluation.agent.slug },
277
+ run: EvaluationSerializer.run(run, number: EvaluationSerializer.run_number(evaluation, run)),
278
+ fix_items: PayloadBounds.bound(EvaluationSerializer.fix_items(run, links: dashboard_links(run)), max_string: MAX_STRING),
279
+ results: rows.map { |row| PayloadBounds.bound(row, max_string: MAX_STRING, max_items: 20) },
280
+ results_total: total,
281
+ results_omitted: [ total - rows.size, 0 ].max
282
+ }
283
+ end
284
+
285
+ def evaluation_runs_compare_tool
286
+ evaluation = find_tool_evaluation!
287
+ base_id = tool_argument(:base_run_id)
288
+ head_id = tool_argument(:head_run_id)
289
+ head = find_tool_run!(evaluation, head_id, label: "head_run_id")
290
+ base =
291
+ if base_id.present?
292
+ find_tool_run!(evaluation, base_id, label: "base_run_id")
293
+ else
294
+ evaluation.evaluation_runs.where("created_at < ? OR (created_at = ? AND id < ?)", head.created_at, head.created_at, head.id)
295
+ .recent.first or raise ToolError, "Run #{head.id} is the evaluation's first run; there is nothing to compare it with"
296
+ end
297
+
298
+ base_rows = comparison_rows(base)
299
+ head_rows = comparison_rows(head)
300
+ changes = (base_rows.keys | head_rows.keys).filter_map do |key|
301
+ before = base_rows[key]
302
+ after = head_rows[key]
303
+ change = comparison_change(before, after)
304
+ next if change == "unchanged_pass"
305
+
306
+ { scenario_key: key[0], model: key[1], change: change, before: before&.status, after: after&.status,
307
+ fault: after&.fault, result_id: after&.id }
308
+ end
309
+
310
+ {
311
+ evaluation: { id: evaluation.id, name: evaluation.name },
312
+ base_run: EvaluationSerializer.run_summary(base, number: EvaluationSerializer.run_number(evaluation, base)),
313
+ head_run: EvaluationSerializer.run_summary(head, number: EvaluationSerializer.run_number(evaluation, head)),
314
+ counts: changes.group_by { |change| change[:change] }.transform_values(&:size),
315
+ changes: PayloadBounds.bound(changes, max_items: MAX_RESULT_LIMIT)
316
+ }
317
+ end
318
+
319
+ def traces_search_tool
320
+ window = integer_argument(:since_minutes, default: TRACE_WINDOW_MINUTES, min: 1, max: MAX_TRACE_WINDOW_MINUTES)
321
+ limit = integer_argument(:limit, default: TRACE_LIMIT, min: 1, max: MAX_TRACE_LIMIT)
322
+ model = ActionAgent.trace_model
323
+
324
+ scope = owned_traces.for_date_range(window.minutes.ago, Time.current)
325
+ if (agent = tool_argument(:agent).presence)
326
+ by_slug = owner_agents.find_by(slug: agent.to_s)
327
+ scope = by_slug ? scope.where(agent_id: by_slug.id).or(scope.for_agent(agent.to_s)) : scope.for_agent(agent.to_s)
328
+ end
329
+ scope = scope.for_service(tool_argument(:service).to_s) if tool_argument(:service).present?
330
+ case tool_argument(:status).to_s.downcase
331
+ when "error" then scope = scope.with_errors
332
+ when "ok" then scope = scope.where.not(status: model::STATUS_ERROR)
333
+ end
334
+ if (min_tokens = integer_argument(:min_tokens, default: nil))
335
+ table = model.quoted_table_name
336
+ scope = scope.where(
337
+ "COALESCE(#{table}.total_input_tokens, 0) + COALESCE(#{table}.total_output_tokens, 0) >= ?", min_tokens
338
+ )
339
+ end
340
+ if (min_duration = integer_argument(:min_duration_ms, default: nil))
341
+ scope = scope.where(model.arel_table[:total_duration_ms].gteq(min_duration))
342
+ end
343
+
344
+ traces = scope.recent.limit(limit + 1).to_a
345
+
346
+ {
347
+ traces: traces.first(limit).map { |trace| PayloadBounds.bound(TelemetryTraceSerializer.row(trace), max_string: MAX_STRING) },
348
+ truncated: traces.size > limit,
349
+ window_minutes: window
350
+ }
351
+ end
352
+
353
+ # Looked up as Api::TraceReportsController#show looks a trace up: the
354
+ # OpenTelemetry trace id, then a prefix of it, then the record id.
355
+ def traces_get_tool
356
+ id = tool_argument(:trace_id).to_s
357
+ raise MCPController::McpError.new("Missing required argument: trace_id", MCPController::JSONRPC_INVALID_PARAMS) if id.blank?
358
+
359
+ scope = owned_traces
360
+ trace = scope.find_by(trace_id: id) ||
361
+ scope.where("trace_id LIKE ?", "#{ActionAgent.trace_model.sanitize_sql_like(id)}%").first ||
362
+ (id.match?(/\A\d+\z/) && scope.find_by(id: id)) or
363
+ raise ToolError, "No trace #{id.truncate(64)} was found"
364
+
365
+ detail = TelemetryTraceSerializer.detail(trace)
366
+ spans = detail.delete(:spans)
367
+ failed_spans = spans.select { |span| span[:error] }.map do |span|
368
+ span.slice(:span_id, :name, :type).merge(message: span.dig(:attributes, "error.message"))
369
+ end
370
+
371
+ payload = detail.merge(
372
+ tool_calls: trace.tool_usage,
373
+ failed_spans: failed_spans,
374
+ spans: spans.first(MAX_TRACE_SPANS),
375
+ spans_total: spans.size,
376
+ spans_omitted: [ spans.size - MAX_TRACE_SPANS, 0 ].max
377
+ )
378
+ PayloadBounds.bound(payload, max_string: MAX_STRING, max_items: MAX_TRACE_SPANS)
379
+ end
380
+
381
+ # `trace_id` names the result's telemetry trace for traces_get, and is
382
+ # nil when telemetry recorded none for the replay.
383
+ def result_row(result, recorded_trace_ids)
384
+ {
385
+ id: result.id,
386
+ scenario_key: result.evaluated_scenario["key"],
387
+ group: result.evaluated_scenario["group"],
388
+ prompt: result.evaluated_scenario["prompt"],
389
+ model: result.model,
390
+ status: result.status,
391
+ score: result.score,
392
+ fault: result.fault,
393
+ recommendation: result.recommendation,
394
+ diagnosis: result.evaluation_diagnosis,
395
+ output: result.output,
396
+ tool_calls: result.tool_calls,
397
+ error_message: result.error_message,
398
+ duration_ms: result.duration_ms,
399
+ input_tokens: result.input_tokens,
400
+ output_tokens: result.output_tokens,
401
+ cost: result.cost&.to_f,
402
+ agent_run_id: result.agent_run_id,
403
+ trace_id: recorded_trace_ids.include?(result.agent_run&.trace_id) ? result.agent_run.trace_id : nil
404
+ }
405
+ end
406
+
407
+ # The subset of +trace_ids+ the owner's telemetry holds.
408
+ def recorded_trace_ids(trace_ids)
409
+ return Set.new if trace_ids.empty?
410
+
411
+ owned_traces.where(trace_id: trace_ids).pluck(:trace_id).to_set
412
+ end
413
+
414
+ # The run's results keyed by [scenario key, model].
415
+ def comparison_rows(run)
416
+ run.scenario_results.includes(:scenario).index_by { |result| [ result.evaluated_scenario["key"], result.model ] }
417
+ end
418
+
419
+ # One of: removed, added_pass, added_fail, unchanged_pass, regressed,
420
+ # fixed, still_failing.
421
+ def comparison_change(before, after)
422
+ return "removed" if after.nil?
423
+ return after.passed? ? "added_pass" : "added_fail" if before.nil?
424
+ return after.passed? ? "unchanged_pass" : "regressed" if before.passed?
425
+
426
+ after.passed? ? "fixed" : "still_failing"
427
+ end
428
+
429
+ # Fix item links at the facade's absolute mount, since a client has no
430
+ # dashboard page to resolve a relative path against.
431
+ def dashboard_links(run)
432
+ run.report_links(mount: "#{request.base_url}#{request.script_name}")
433
+ end
434
+
435
+ def dashboard_evaluations
436
+ Evaluation.joins(:agent).where(agent: owner_agents)
437
+ end
438
+
439
+ def find_tool_evaluation!
440
+ id = tool_argument(:evaluation_id)
441
+ raise MCPController::McpError.new("Missing required argument: evaluation_id", MCPController::JSONRPC_INVALID_PARAMS) if id.blank?
442
+
443
+ dashboard_evaluations.find_by(id: id.to_s) or raise ToolError, "No evaluation #{id.to_s.truncate(32)} was found"
444
+ end
445
+
446
+ # The evaluation's run +id+ names, or its latest run when +id+ is blank.
447
+ def find_tool_run!(evaluation, id, label:)
448
+ if id.blank?
449
+ evaluation.evaluation_runs.recent.first or raise ToolError, "Evaluation #{evaluation.id} has not been run yet"
450
+ else
451
+ evaluation.evaluation_runs.find_by(id: id.to_s) or
452
+ raise ToolError, "No run #{id.to_s.truncate(32)} of evaluation #{evaluation.id} was found (#{label})"
453
+ end
454
+ end
455
+
456
+ def resolve_tool_agent!(value)
457
+ agents = owner_agents
458
+ agents.find_by(slug: value.to_s) || (value.to_s.match?(/\A\d+\z/) && agents.find_by(id: value.to_s)) or
459
+ raise ToolError, "No agent #{value.to_s.truncate(64)} was found"
460
+ end
461
+
462
+ def tool_arguments
463
+ @tool_arguments ||= begin
464
+ arguments = params.dig(:params, :arguments)
465
+ arguments.is_a?(ActionController::Parameters) ? arguments : ActionController::Parameters.new({})
466
+ end
467
+ end
468
+
469
+ def tool_argument(name)
470
+ tool_arguments[name]
471
+ end
472
+
473
+ # An integer argument clamped into [min, max], or +default+ when absent
474
+ # or not a number.
475
+ def integer_argument(name, default:, min: nil, max: nil)
476
+ raw = tool_argument(name)
477
+ value = raw.is_a?(Integer) || raw.to_s.match?(/\A-?\d+\z/) ? raw.to_i : default
478
+ return value if value.nil?
479
+
480
+ value = [ value, min ].max if min
481
+ value = [ value, max ].min if max
482
+ value
483
+ end
484
+
485
+ def boolean_argument(name)
486
+ ActiveModel::Type::Boolean.new.cast(tool_argument(name)) == true
487
+ end
488
+
489
+ # Every credential the owner holds, masked out of each tool's output:
490
+ # this key's token, provider keys, the GitHub token, and each checkout
491
+ # sandbox's runtime token. None of them belongs in these payloads; this
492
+ # keeps one that leaked into a recorded output or a trace from being
493
+ # handed on. A failed lookup masks nothing rather than failing the call.
494
+ def dashboard_tool_secrets
495
+ @dashboard_tool_secrets ||= [
496
+ @api_key&.token,
497
+ *owned(ProviderKey).limit(SECRET_LOOKUP_LIMIT).pluck(:credential, :api_key).flatten,
498
+ *owned(GithubConnection).limit(SECRET_LOOKUP_LIMIT).pluck(:access_token),
499
+ *owned(SandboxSession).where.not(runtime_mcp_token: nil).order(id: :desc).limit(SECRET_LOOKUP_LIMIT).pluck(:runtime_mcp_token)
500
+ ].compact
501
+ rescue StandardError => e
502
+ Rails.logger.warn("[ActionAgent] MCP secret lookup failed: #{e.class}: #{e.message}")
503
+ @dashboard_tool_secrets = [ @api_key&.token ].compact
504
+ end
505
+ end
506
+ end
507
+ end
@@ -0,0 +1,65 @@
1
+ # frozen_string_literal: true
2
+
3
+ module ActionAgent
4
+ module Api
5
+ # A run against a checkout sandbox the agent does not name: `sandbox_id`
6
+ # on an evaluation run or a runner run makes that one run reach the
7
+ # sandbox's app runtime, as if the agent listed "sandbox:<session_id>" in
8
+ # mcp_servers, without saving it there. It is how a change being tried in
9
+ # a checkout is evaluated before anyone edits the agent.
10
+ #
11
+ # The sandbox has to be the caller's (found the way CodeSessionsController
12
+ # finds one: owned, and in the current account), a checkout, live, and
13
+ # owned by the agent's owner as well, since the dispatcher resolves the
14
+ # runtime among that owner's sessions only (see
15
+ # SandboxSession.runtime_server_entry). Anything else is refused with a
16
+ # 422 saying which. Only the key and the checkout it names are passed on
17
+ # and recorded; the runtime's token stays on the session.
18
+ module RunSandbox
19
+ extend ActiveSupport::Concern
20
+
21
+ class Refused < StandardError; end
22
+
23
+ included do
24
+ rescue_from Refused do |error|
25
+ render json: { error: error.message, code: "sandbox_refused" }, status: :unprocessable_entity
26
+ end
27
+ end
28
+
29
+ private
30
+
31
+ # The caller's live checkout sandbox that +sandbox_id+ names for a run
32
+ # of +agent+, or nil when none was asked for. Raises Refused.
33
+ #
34
+ # @return [SandboxSession, nil]
35
+ def run_sandbox_for(agent, sandbox_id)
36
+ return nil if sandbox_id.blank?
37
+ raise Refused, "sandbox_id must be a sandbox's session id" unless sandbox_id.is_a?(String)
38
+
39
+ scope = owned(SandboxSession)
40
+ scope = scope.where(account_id: current_account.id) if current_account
41
+ # Unknown and someone else's read the same: whether a session id
42
+ # exists is not the caller's to learn.
43
+ sandbox = scope.find_by(session_id: sandbox_id) or
44
+ raise Refused, "No sandbox #{sandbox_id.truncate(64)} of yours was found"
45
+ label = "Sandbox #{sandbox.session_id}"
46
+
47
+ unless sandbox.app_runtime?
48
+ raise Refused, "#{label} is a #{sandbox.sandbox_type} sandbox; only a checkout (app_runtime) sandbox " \
49
+ "serves an app's tools"
50
+ end
51
+ # Ready (or running a task), and serving: a sandbox still booting
52
+ # can already carry an endpoint that is not answering yet.
53
+ if !(sandbox.ready? || sandbox.running?) || sandbox.runtime_server_entry.nil?
54
+ state = sandbox.ready? && !sandbox.active? ? "expired" : sandbox.status
55
+ raise Refused, "#{label} is #{state}; run against a checkout sandbox once it is ready"
56
+ end
57
+ unless SandboxSession.runtime_server_entry(sandbox.runtime_server_key, owner: agent.owner)
58
+ raise Refused, "#{label} does not belong to the owner of #{agent.name}, so this agent's runs cannot reach it"
59
+ end
60
+
61
+ sandbox
62
+ end
63
+ end
64
+ end
65
+ end