actionagent 1.7.2 → 1.8.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/app/assets/builds/action_agent.css +1 -1
- data/app/assets/builds/action_agent.js +59 -54
- data/app/controllers/action_agent/api/agents_controller.rb +19 -3
- data/app/controllers/action_agent/api/code_sessions_controller.rb +156 -0
- data/app/controllers/action_agent/api/evaluations_controller.rb +27 -130
- data/app/controllers/action_agent/api/github_connections_controller.rb +147 -0
- data/app/controllers/action_agent/api/mcp_controller.rb +38 -5
- data/app/controllers/action_agent/api/mcp_servers_controller.rb +10 -1
- data/app/controllers/action_agent/api/provider_keys_controller.rb +54 -3
- data/app/controllers/action_agent/api/provider_models_controller.rb +10 -6
- data/app/controllers/action_agent/api/sandboxes_controller.rb +97 -6
- data/app/controllers/concerns/action_agent/api/evaluation_run_starting.rb +93 -0
- data/app/controllers/concerns/action_agent/api/mcp_dashboard_tools.rb +507 -0
- data/app/controllers/concerns/action_agent/api/run_sandbox.rb +65 -0
- data/app/jobs/action_agent/code_session_job.rb +166 -0
- data/app/jobs/action_agent/sandbox_cleanup_job.rb +80 -11
- data/app/jobs/action_agent/sandbox_provision_job.rb +122 -14
- data/app/jobs/action_agent/sandbox_run_job.rb +10 -3
- data/app/models/action_agent/agent.rb +16 -6
- data/app/models/action_agent/agent_run.rb +20 -1
- data/app/models/action_agent/code_session.rb +141 -0
- data/app/models/action_agent/evaluation_run.rb +23 -1
- data/app/models/action_agent/github_connection.rb +75 -0
- data/app/models/action_agent/provider_key.rb +142 -10
- data/app/models/action_agent/sandbox_session.rb +193 -17
- data/app/serializers/action_agent/evaluation_serializer.rb +118 -0
- data/app/serializers/action_agent/telemetry_trace_serializer.rb +10 -2
- data/app/services/action_agent/agent_execution_service.rb +4 -2
- data/app/services/action_agent/agent_tool_roster.rb +20 -9
- data/app/services/action_agent/claude_code_auth.rb +86 -0
- data/app/services/action_agent/dashboard_assistant_service.rb +47 -5
- data/app/services/action_agent/evaluation_tool_resolver.rb +18 -0
- data/app/services/action_agent/github_client.rb +111 -0
- data/app/services/action_agent/local_sandbox_backend.rb +1689 -0
- data/app/services/action_agent/local_sandbox_databases.rb +257 -0
- data/app/services/action_agent/mcp_client.rb +5 -1
- data/app/services/action_agent/mcp_tool_dispatcher.rb +137 -17
- data/app/services/action_agent/mock_sandbox_backend.rb +39 -0
- data/app/services/action_agent/ollama_host_probe.rb +75 -0
- data/app/services/action_agent/payload_bounds.rb +36 -0
- data/app/services/action_agent/sandbox_manifest.rb +67 -0
- data/app/services/action_agent/sandbox_orchestrator.rb +69 -14
- data/app/services/action_agent/scenario_evaluation_runner.rb +50 -4
- data/app/services/action_agent/secret_scrubber.rb +37 -0
- data/app/services/action_agent/tool_discovery.rb +19 -5
- data/config/routes.rb +22 -3
- data/lib/action_agent/engine.rb +1 -0
- data/lib/action_agent/version.rb +1 -1
- data/lib/action_agent.rb +147 -3
- data/lib/generators/action_agent/install_generator.rb +30 -3
- data/lib/generators/action_agent/templates/action_agent.rb.erb +44 -0
- data/lib/generators/action_agent/templates/add_provider_key_api_key.rb.erb +25 -0
- data/lib/generators/action_agent/templates/create_active_agent_code_sessions.rb.erb +59 -0
- data/lib/generators/action_agent/templates/create_active_agent_dashboard_tables.rb.erb +2 -0
- data/lib/generators/action_agent/templates/create_active_agent_github_connections.rb.erb +61 -0
- data/lib/tasks/claude_code.rake +16 -0
- data/lib/tasks/sandbox.rake +26 -0
- metadata +23 -1
|
@@ -4,6 +4,7 @@ module ActionAgent
|
|
|
4
4
|
module Api
|
|
5
5
|
class AgentsController < BaseController
|
|
6
6
|
include AgentSerialization
|
|
7
|
+
include RunSandbox
|
|
7
8
|
|
|
8
9
|
# Ranking for the agent cards. Every dimension except "recent" reads the
|
|
9
10
|
# scorecard, which is computed in Ruby over both execution sources, so
|
|
@@ -25,7 +26,7 @@ module ActionAgent
|
|
|
25
26
|
# same reason as the rest and one more: a keyword splat wins over the
|
|
26
27
|
# arguments before it, so a client sending params[params][actor] would
|
|
27
28
|
# otherwise name the caller its own run is authorized as.
|
|
28
|
-
RESERVED_EXECUTION_KEYS = [ :attachments, :action, :actor, :current_user ].freeze
|
|
29
|
+
RESERVED_EXECUTION_KEYS = [ :attachments, :action, :actor, :current_user, :runtime_sandbox ].freeze
|
|
29
30
|
|
|
30
31
|
before_action :set_agent, only: [
|
|
31
32
|
:show, :update, :destroy, :versions, :runs, :execute, :test, :restore, :duplicate, :export, :analytics,
|
|
@@ -169,13 +170,16 @@ module ActionAgent
|
|
|
169
170
|
#
|
|
170
171
|
# JSON as before, or multipart from the runner's composer: the new
|
|
171
172
|
# user message, its files (attachments[]) and the conversation to
|
|
172
|
-
# continue (params[context_id], or a top-level context_id).
|
|
173
|
+
# continue (params[context_id], or a top-level context_id). A
|
|
174
|
+
# `sandbox_id` runs it against that checkout sandbox's app runtime
|
|
175
|
+
# too, without editing the agent (see RunSandbox); so does /test.
|
|
173
176
|
def execute
|
|
174
177
|
run = @agent.execute(
|
|
175
178
|
execution_prompt,
|
|
176
179
|
action: params[:action_name],
|
|
177
180
|
attachments: uploaded_attachments,
|
|
178
181
|
actor: agent_actor,
|
|
182
|
+
runtime_sandbox: run_sandbox_for(@agent, params[:sandbox_id])&.runtime_server_key,
|
|
179
183
|
**execution_params
|
|
180
184
|
)
|
|
181
185
|
record_execution_usage
|
|
@@ -190,6 +194,7 @@ module ActionAgent
|
|
|
190
194
|
action: params[:action_name],
|
|
191
195
|
attachments: uploaded_attachments,
|
|
192
196
|
actor: agent_actor,
|
|
197
|
+
runtime_sandbox: run_sandbox_for(@agent, params[:sandbox_id])&.runtime_server_key,
|
|
193
198
|
**execution_params
|
|
194
199
|
)
|
|
195
200
|
record_execution_usage
|
|
@@ -268,11 +273,22 @@ module ActionAgent
|
|
|
268
273
|
# What the Tools tab edits: the MCP services this agent can be given
|
|
269
274
|
# and the tools it can be offered, each with the calls, errors and
|
|
270
275
|
# latency recorded for it in the window.
|
|
276
|
+
#
|
|
277
|
+
# The checkout runtimes offered are the agent owner's — the only ones
|
|
278
|
+
# MCPToolDispatcher reaches for this agent — and of those, only the
|
|
279
|
+
# ones the caller may see as well. Intersected by id rather than with
|
|
280
|
+
# merge: both relations constrain the same owner column, and merge
|
|
281
|
+
# lets the second condition replace the first instead of adding to it.
|
|
271
282
|
def tool_roster
|
|
283
|
+
runtimes = SandboxSession.runtime_server_listings(
|
|
284
|
+
SandboxSession.for_owner(@agent.owner).where(id: owned(SandboxSession).select(:id))
|
|
285
|
+
)
|
|
286
|
+
|
|
272
287
|
render json: AgentToolRoster.new(
|
|
273
288
|
agent: @agent,
|
|
274
289
|
traces: owned_traces,
|
|
275
|
-
hours: params.fetch(:hours, ToolDiscovery::DEFAULT_WINDOW_HOURS).to_i
|
|
290
|
+
hours: params.fetch(:hours, ToolDiscovery::DEFAULT_WINDOW_HOURS).to_i,
|
|
291
|
+
runtimes: runtimes
|
|
276
292
|
).as_json
|
|
277
293
|
end
|
|
278
294
|
|
|
@@ -0,0 +1,156 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module ActionAgent
|
|
4
|
+
module Api
|
|
5
|
+
# Headless Claude Code sessions inside one of the caller's checkout
|
|
6
|
+
# sandboxes (Settings -> Integrations): start one with a prompt, poll its
|
|
7
|
+
# transcript, cancel it.
|
|
8
|
+
#
|
|
9
|
+
# A session runs on the owner's Anthropic API key (or, with
|
|
10
|
+
# ActionAgent.claude_code_auth = :local_login, this machine's own Claude
|
|
11
|
+
# Code login; see ClaudeCodeAuth) and edits the checkout, so starting one
|
|
12
|
+
# is execution — gated, and counted against the host app's quota, like
|
|
13
|
+
# running an agent.
|
|
14
|
+
class CodeSessionsController < BaseController
|
|
15
|
+
before_action :require_owner!
|
|
16
|
+
before_action :require_execution_enabled!, only: [ :create ]
|
|
17
|
+
before_action :set_sandbox
|
|
18
|
+
before_action :set_code_session, only: [ :show, :cancel ]
|
|
19
|
+
|
|
20
|
+
# A model name as Claude Code's --model takes it ("sonnet",
|
|
21
|
+
# "claude-sonnet-4-5", "claude-sonnet-4-5[1m]"). It becomes an argument
|
|
22
|
+
# to the CLI, so nothing that could read as another option gets there.
|
|
23
|
+
MODEL_NAME = /\A[A-Za-z0-9][A-Za-z0-9._:\[\]-]{0,99}\z/
|
|
24
|
+
|
|
25
|
+
# GET /api/sandboxes/:sandbox_id/code_sessions
|
|
26
|
+
def index
|
|
27
|
+
render json: { code_sessions: @sandbox.code_sessions.recent.limit(20).map(&:summary) }
|
|
28
|
+
end
|
|
29
|
+
|
|
30
|
+
# POST /api/sandboxes/:sandbox_id/code_sessions
|
|
31
|
+
def create
|
|
32
|
+
if (refusal = refusal_for(@sandbox))
|
|
33
|
+
return render json: { error: refusal }, status: :unprocessable_entity
|
|
34
|
+
end
|
|
35
|
+
|
|
36
|
+
if (current = busy_session)
|
|
37
|
+
return render json: busy_body(current), status: :conflict
|
|
38
|
+
end
|
|
39
|
+
|
|
40
|
+
code_session = CodeSession.new(
|
|
41
|
+
sandbox_session: @sandbox,
|
|
42
|
+
prompt: string_param(:prompt),
|
|
43
|
+
model: string_param(:model).presence,
|
|
44
|
+
# Owned like the sandbox it runs in.
|
|
45
|
+
user_id: @sandbox.try(:user_id),
|
|
46
|
+
account_id: @sandbox.try(:account_id)
|
|
47
|
+
)
|
|
48
|
+
if (error = invalid_request(code_session))
|
|
49
|
+
return render json: { error: error }, status: :unprocessable_entity
|
|
50
|
+
end
|
|
51
|
+
|
|
52
|
+
enforce_execution_quota!
|
|
53
|
+
return if performed?
|
|
54
|
+
|
|
55
|
+
# Checked again under the sandbox's row lock, so two requests racing
|
|
56
|
+
# past the check above cannot both start a session in one checkout.
|
|
57
|
+
current = nil
|
|
58
|
+
@sandbox.with_lock do
|
|
59
|
+
current = busy_session
|
|
60
|
+
code_session.save! unless current
|
|
61
|
+
end
|
|
62
|
+
return render json: busy_body(current), status: :conflict if current
|
|
63
|
+
|
|
64
|
+
record_execution_usage
|
|
65
|
+
CodeSessionJob.perform_later(code_session.id)
|
|
66
|
+
|
|
67
|
+
render json: { code_session: code_session.summary }, status: :created
|
|
68
|
+
end
|
|
69
|
+
|
|
70
|
+
# GET /api/sandboxes/:sandbox_id/code_sessions/:id?after=N
|
|
71
|
+
# The transcript from event N onward, for incremental polling.
|
|
72
|
+
def show
|
|
73
|
+
render json: { code_session: @code_session.details(after: integer_param(:after, default: 0)) }
|
|
74
|
+
end
|
|
75
|
+
|
|
76
|
+
# POST /api/sandboxes/:sandbox_id/code_sessions/:id/cancel
|
|
77
|
+
def cancel
|
|
78
|
+
was_running = false
|
|
79
|
+
@code_session.with_lock do
|
|
80
|
+
next unless @code_session.queued? || @code_session.running?
|
|
81
|
+
|
|
82
|
+
was_running = @code_session.running?
|
|
83
|
+
# A queued session is settled at once: nothing will ever run it. A
|
|
84
|
+
# running one is settled by CodeSessionJob once its Claude Code has
|
|
85
|
+
# stopped and its diff was taken (see CodeSession#diff_pending?).
|
|
86
|
+
@code_session.update!(status: :cancelled, finished_at: was_running ? nil : Time.current)
|
|
87
|
+
end
|
|
88
|
+
|
|
89
|
+
# A queued session never started: CodeSessionJob skips it. A running
|
|
90
|
+
# one is stopped by its backend; CodeSessionJob then finds it
|
|
91
|
+
# cancelled and leaves it so.
|
|
92
|
+
stop(@code_session) if was_running
|
|
93
|
+
|
|
94
|
+
render json: { code_session: @code_session.summary }
|
|
95
|
+
end
|
|
96
|
+
|
|
97
|
+
private
|
|
98
|
+
|
|
99
|
+
# A checkout runs on its account's GitHub token and Claude Code
|
|
100
|
+
# credential, so in a multi-tenant install it must belong to the
|
|
101
|
+
# caller's current account too — not only to the caller, who may have
|
|
102
|
+
# left that account or switched to another.
|
|
103
|
+
def set_sandbox
|
|
104
|
+
scope = owned(SandboxSession)
|
|
105
|
+
scope = scope.where(account_id: current_account.id) if current_account
|
|
106
|
+
@sandbox = scope.find_by!(session_id: params[:sandbox_id])
|
|
107
|
+
end
|
|
108
|
+
|
|
109
|
+
def set_code_session
|
|
110
|
+
@code_session = @sandbox.code_sessions.find(params[:id])
|
|
111
|
+
end
|
|
112
|
+
|
|
113
|
+
# Why +sandbox+ cannot take a Claude Code session now, or nil.
|
|
114
|
+
def refusal_for(sandbox)
|
|
115
|
+
return "Claude Code sessions run only in a checkout (app_runtime) sandbox" unless sandbox.app_runtime?
|
|
116
|
+
return "The sandbox has expired; start a new one" if sandbox.ready? && !sandbox.active?
|
|
117
|
+
return "The sandbox is #{sandbox.status}; wait until it is ready" unless sandbox.ready?
|
|
118
|
+
|
|
119
|
+
orchestrator = SandboxOrchestrator.new
|
|
120
|
+
unless orchestrator.supports?(:code_session)
|
|
121
|
+
return "The #{orchestrator.backend_name} sandbox backend cannot run Claude Code sessions"
|
|
122
|
+
end
|
|
123
|
+
|
|
124
|
+
ClaudeCodeAuth.backend_refusal(orchestrator) || ClaudeCodeAuth.credential_refusal(sandbox)
|
|
125
|
+
end
|
|
126
|
+
|
|
127
|
+
# The session of this sandbox that is queued or running, if any: one
|
|
128
|
+
# checkout, one Claude Code at a time.
|
|
129
|
+
def busy_session
|
|
130
|
+
@sandbox.code_sessions.where(status: [ :queued, :running ]).first
|
|
131
|
+
end
|
|
132
|
+
|
|
133
|
+
def busy_body(current)
|
|
134
|
+
{ error: "A Claude Code session is already #{current.status} in this sandbox", code_session: current.summary }
|
|
135
|
+
end
|
|
136
|
+
|
|
137
|
+
def invalid_request(code_session)
|
|
138
|
+
return "model is not a Claude Code model name" if code_session.model && !code_session.model.match?(MODEL_NAME)
|
|
139
|
+
|
|
140
|
+
code_session.errors.full_messages.to_sentence unless code_session.valid?
|
|
141
|
+
end
|
|
142
|
+
|
|
143
|
+
# A prompt or model sent as an array or object is not one.
|
|
144
|
+
def string_param(name)
|
|
145
|
+
value = params[name]
|
|
146
|
+
value.is_a?(String) ? value : nil
|
|
147
|
+
end
|
|
148
|
+
|
|
149
|
+
def stop(code_session)
|
|
150
|
+
SandboxOrchestrator.new.cancel_code_session(@sandbox, code_session)
|
|
151
|
+
rescue StandardError => e
|
|
152
|
+
Rails.logger.warn("Failed to stop Claude Code session #{code_session.id}: #{e.message}")
|
|
153
|
+
end
|
|
154
|
+
end
|
|
155
|
+
end
|
|
156
|
+
end
|
|
@@ -10,6 +10,8 @@ module ActionAgent
|
|
|
10
10
|
# than sampling recorded generations, and can be narrowed to a group, to
|
|
11
11
|
# specific scenarios, or to specific models.
|
|
12
12
|
class EvaluationsController < BaseController
|
|
13
|
+
include EvaluationRunStarting
|
|
14
|
+
|
|
13
15
|
rescue_from ActiveAgent::Evals::ScenarioParser::ParseError do |error|
|
|
14
16
|
render json: { errors: [ error.message ] }, status: :unprocessable_entity
|
|
15
17
|
end
|
|
@@ -107,7 +109,7 @@ module ActionAgent
|
|
|
107
109
|
end
|
|
108
110
|
|
|
109
111
|
if evaluation.save
|
|
110
|
-
|
|
112
|
+
start_evaluation_run(evaluation, selection_params) if run_requested?
|
|
111
113
|
render json: { evaluation: serialize(evaluation.reload) }, status: :created
|
|
112
114
|
else
|
|
113
115
|
render json: { errors: evaluation.errors.full_messages }, status: :unprocessable_entity
|
|
@@ -116,10 +118,17 @@ module ActionAgent
|
|
|
116
118
|
|
|
117
119
|
# POST /api/evaluations/:id/run
|
|
118
120
|
# A scenario suite accepts a selection: scenario_ids[], keys[], group,
|
|
119
|
-
# models[] (or a comma-separated `models` string)
|
|
121
|
+
# models[] (or a comma-separated `models` string), and `sandbox_id`: a
|
|
122
|
+
# checkout sandbox of the caller's whose app runtime every replay of
|
|
123
|
+
# this run reaches as well, without the agent being edited (see
|
|
124
|
+
# RunSandbox). The run records which one it used.
|
|
120
125
|
def run
|
|
121
126
|
evaluation = current_evaluation
|
|
122
|
-
|
|
127
|
+
selection = selection_params
|
|
128
|
+
if (sandbox = evaluation_run_sandbox(evaluation, requested_sandbox_id))
|
|
129
|
+
selection[:sandbox_id] = sandbox.session_id
|
|
130
|
+
end
|
|
131
|
+
run = start_evaluation_run(evaluation, selection)
|
|
123
132
|
evaluation.reload
|
|
124
133
|
|
|
125
134
|
render json: {
|
|
@@ -216,30 +225,6 @@ module ActionAgent
|
|
|
216
225
|
|
|
217
226
|
private
|
|
218
227
|
|
|
219
|
-
# A scenario suite replays through the provider once per scenario and
|
|
220
|
-
# model, so it runs in the background; a generation-sampling evaluation
|
|
221
|
-
# scores recorded data and finishes inline.
|
|
222
|
-
def start_run(evaluation, selection)
|
|
223
|
-
return evaluation.run_later!(**selection) if evaluation.scenario_suite?
|
|
224
|
-
|
|
225
|
-
# EvaluationRunnerService marks the run failed with the error message
|
|
226
|
-
# and then re-raises. Letting that escape returned an HTML 500 for a
|
|
227
|
-
# request that had already persisted the evaluation and its failed
|
|
228
|
-
# run: the client saw a JSON parse error, the form stayed open, and a
|
|
229
|
-
# resubmit failed on the now-taken name. The failure is on the run
|
|
230
|
-
# record, which is what the response carries.
|
|
231
|
-
evaluation.run!
|
|
232
|
-
rescue StandardError => e
|
|
233
|
-
Rails.logger.warn(
|
|
234
|
-
"[ActionAgent] evaluation #{evaluation.id} run failed: #{e.class}: #{e.message}"
|
|
235
|
-
)
|
|
236
|
-
# The service records the failure before re-raising; a failure that
|
|
237
|
-
# predates the run record (creating it, say) is recorded here so the
|
|
238
|
-
# response always carries one.
|
|
239
|
-
evaluation.evaluation_runs.recent.first ||
|
|
240
|
-
evaluation.evaluation_runs.create!(status: :failed, error_message: e.message, completed_at: Time.current)
|
|
241
|
-
end
|
|
242
|
-
|
|
243
228
|
# Returns whose credentials the index's model picker fields describe.
|
|
244
229
|
# Agent runs and their judge use the evaluated agent's owner's
|
|
245
230
|
# credentials, so a list scoped to one agent reads that agent's owner,
|
|
@@ -292,10 +277,7 @@ module ActionAgent
|
|
|
292
277
|
def require_executable_scenario_agent!
|
|
293
278
|
agent = action_name == "create" ? requested_agent : current_evaluation.agent
|
|
294
279
|
return unless agent.observed?
|
|
295
|
-
if action_name == "run"
|
|
296
|
-
adapter = ActionAgent.scenario_evaluation_adapter_resolver&.call(current_evaluation)
|
|
297
|
-
return if adapter.respond_to?(:call)
|
|
298
|
-
end
|
|
280
|
+
return if action_name == "run" && !unexecutable_scenario_run?(current_evaluation)
|
|
299
281
|
|
|
300
282
|
render json: {
|
|
301
283
|
error: "Observed agents are read-only — duplicate this agent to create an executable copy"
|
|
@@ -310,19 +292,18 @@ module ActionAgent
|
|
|
310
292
|
params.require(:scenario).permit(:prompt, :group, :notes, :enabled, :key, expectations: {})
|
|
311
293
|
end
|
|
312
294
|
|
|
313
|
-
#
|
|
314
|
-
#
|
|
295
|
+
# The sandbox id a run was asked to use, from the selection or the top
|
|
296
|
+
# level of the request.
|
|
297
|
+
def requested_sandbox_id
|
|
298
|
+
selection_source[:sandbox_id].presence || params[:sandbox_id].presence
|
|
299
|
+
end
|
|
300
|
+
|
|
315
301
|
def selection_params
|
|
316
|
-
|
|
317
|
-
|
|
318
|
-
models = models.to_s.split(",") unless models.is_a?(Array)
|
|
302
|
+
evaluation_run_selection(selection_source)
|
|
303
|
+
end
|
|
319
304
|
|
|
320
|
-
|
|
321
|
-
|
|
322
|
-
keys: Array(source[:keys]).map(&:to_s).reject(&:blank?),
|
|
323
|
-
group: source[:group].to_s.presence,
|
|
324
|
-
models: models.map(&:to_s).map(&:strip).reject(&:blank?)
|
|
325
|
-
}.compact_blank
|
|
305
|
+
def selection_source
|
|
306
|
+
params[:evaluation].is_a?(ActionController::Parameters) && params[:evaluation].key?(:selection) ? params[:evaluation][:selection] : params
|
|
326
307
|
end
|
|
327
308
|
|
|
328
309
|
# Scenarios from text, a YAML/JSON suite, or a list of objects. Production
|
|
@@ -373,97 +354,13 @@ module ActionAgent
|
|
|
373
354
|
models.map(&:to_s).map(&:strip).reject(&:blank?)
|
|
374
355
|
end
|
|
375
356
|
|
|
376
|
-
def serialize(evaluation)
|
|
377
|
-
# size reads the preloaded association on index and COUNTs elsewhere.
|
|
378
|
-
run_count = evaluation.evaluation_runs.size
|
|
379
|
-
latest, previous = recent_runs(evaluation, 2)
|
|
380
|
-
|
|
381
|
-
{
|
|
382
|
-
id: evaluation.id,
|
|
383
|
-
name: evaluation.name,
|
|
384
|
-
agent: { id: evaluation.agent.id, name: evaluation.agent.name, slug: evaluation.agent.slug },
|
|
385
|
-
judge_kind: evaluation.judge_kind,
|
|
386
|
-
judge_model: evaluation.judge_model,
|
|
387
|
-
criteria: evaluation.criteria,
|
|
388
|
-
compare_models: evaluation.compare_models,
|
|
389
|
-
config: evaluation.config,
|
|
390
|
-
sample_size: evaluation.sample_size,
|
|
391
|
-
scenario_suite: evaluation.scenario_suite?,
|
|
392
|
-
scenario_count: evaluation.scenarios.size,
|
|
393
|
-
scenario_groups: evaluation.scenario_suite? ? evaluation.scenario_groups : [],
|
|
394
|
-
created_at: evaluation.created_at.iso8601,
|
|
395
|
-
run_count: run_count,
|
|
396
|
-
latest_run: latest ? serialize_run(latest, number: run_count) : nil,
|
|
397
|
-
# Just enough of the run before it for the list to show movement
|
|
398
|
-
# ("+3 passed vs #2") without a request per evaluation.
|
|
399
|
-
previous_run: previous ? serialize_run_summary(previous, number: run_count - 1) : nil
|
|
400
|
-
}
|
|
401
|
-
end
|
|
402
|
-
|
|
403
|
-
# Newest first. Sorts the preloaded association when index loaded it
|
|
404
|
-
# rather than issuing one ORDER BY query per evaluation.
|
|
405
|
-
def recent_runs(evaluation, limit)
|
|
406
|
-
runs = evaluation.evaluation_runs
|
|
407
|
-
if runs.loaded?
|
|
408
|
-
runs.sort_by { |run| [ run.created_at, run.id ] }.reverse.first(limit)
|
|
409
|
-
else
|
|
410
|
-
runs.recent.limit(limit).to_a
|
|
411
|
-
end
|
|
412
|
-
end
|
|
413
|
-
|
|
414
|
-
# A run's position in its evaluation's history, oldest = 1.
|
|
415
|
-
def run_number(evaluation, run)
|
|
416
|
-
evaluation.evaluation_runs.where("created_at < ? OR (created_at = ? AND id <= ?)", run.created_at, run.created_at, run.id).count
|
|
417
|
-
end
|
|
357
|
+
def serialize(evaluation) = EvaluationSerializer.evaluation(evaluation)
|
|
418
358
|
|
|
419
|
-
|
|
420
|
-
# 1, so the dashboard can say "Run #3" and "vs #2".
|
|
421
|
-
def serialize_run(run, number: nil)
|
|
422
|
-
serialize_run_summary(run, number: number).merge(
|
|
423
|
-
scores: run.scores,
|
|
424
|
-
selection: run.selection,
|
|
425
|
-
models: run.models,
|
|
426
|
-
usage: run.usage,
|
|
427
|
-
error_message: run.error_message
|
|
428
|
-
)
|
|
429
|
-
end
|
|
359
|
+
def serialize_run(run, number: nil) = EvaluationSerializer.run(run, number: number)
|
|
430
360
|
|
|
431
|
-
def
|
|
432
|
-
{
|
|
433
|
-
id: run.id,
|
|
434
|
-
number: number,
|
|
435
|
-
status: run.status,
|
|
436
|
-
average_score: safe_average_score(run),
|
|
437
|
-
samples_evaluated: run.samples_evaluated,
|
|
438
|
-
samples_passed: run.samples_passed,
|
|
439
|
-
completed_at: run.completed_at&.iso8601,
|
|
440
|
-
created_at: run.created_at.iso8601
|
|
441
|
-
}
|
|
442
|
-
end
|
|
361
|
+
def run_number(evaluation, run) = EvaluationSerializer.run_number(evaluation, run)
|
|
443
362
|
|
|
444
|
-
|
|
445
|
-
# which older runs recorded in earlier shapes; a run they cannot be
|
|
446
|
-
# built for still serves its results rather than 500-ing the panel.
|
|
447
|
-
def safe_fix_items(run)
|
|
448
|
-
run.fix_items
|
|
449
|
-
rescue StandardError => e
|
|
450
|
-
Rails.logger.warn(
|
|
451
|
-
"[ActionAgent] evaluation run #{run.id} fix_items failed: #{e.class}: #{e.message}"
|
|
452
|
-
)
|
|
453
|
-
[]
|
|
454
|
-
end
|
|
455
|
-
|
|
456
|
-
# index serializes the latest run of every listed evaluation, so an
|
|
457
|
-
# unaverageable scores payload used to 500 the entire Evaluations page
|
|
458
|
-
# instead of degrading that one run's headline number.
|
|
459
|
-
def safe_average_score(run)
|
|
460
|
-
run.average_score
|
|
461
|
-
rescue StandardError => e
|
|
462
|
-
Rails.logger.warn(
|
|
463
|
-
"[ActionAgent] evaluation run #{run.id} average_score failed: #{e.class}: #{e.message}"
|
|
464
|
-
)
|
|
465
|
-
nil
|
|
466
|
-
end
|
|
363
|
+
def safe_fix_items(run) = EvaluationSerializer.fix_items(run)
|
|
467
364
|
end
|
|
468
365
|
end
|
|
469
366
|
end
|
|
@@ -0,0 +1,147 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module ActionAgent
|
|
4
|
+
module Api
|
|
5
|
+
# The owner's GitHub connection (Settings -> Integrations): the OAuth
|
|
6
|
+
# web flow that creates it, the repositories it makes available, and
|
|
7
|
+
# disconnecting it. The access token is write-only; responses carry the
|
|
8
|
+
# GitHub login and the chosen repositories, never the token.
|
|
9
|
+
#
|
|
10
|
+
# #connect and #callback are browser navigations rather than fetches, so
|
|
11
|
+
# they answer with redirects back into the dashboard's Settings view.
|
|
12
|
+
class GithubConnectionsController < BaseController
|
|
13
|
+
STATE_SESSION_KEY = :action_agent_github_oauth_state
|
|
14
|
+
|
|
15
|
+
before_action :require_owner!
|
|
16
|
+
before_action :require_github_oauth!, only: [ :connect, :callback ]
|
|
17
|
+
before_action :set_connection, only: [ :repositories, :update, :destroy ]
|
|
18
|
+
|
|
19
|
+
# Later handlers win, so the subclass is registered last.
|
|
20
|
+
rescue_from GithubClient::Error, with: :github_unavailable
|
|
21
|
+
rescue_from GithubClient::Unauthorized, with: :github_unauthorized
|
|
22
|
+
|
|
23
|
+
# GET /api/github_connection
|
|
24
|
+
def show
|
|
25
|
+
connection = owned(GithubConnection).first
|
|
26
|
+
|
|
27
|
+
render json: {
|
|
28
|
+
configured: ActionAgent.github_oauth_configured?,
|
|
29
|
+
connected: connection.present?,
|
|
30
|
+
connection: connection&.as_summary
|
|
31
|
+
}
|
|
32
|
+
end
|
|
33
|
+
|
|
34
|
+
# GET /api/github_connection/repositories — what the token reaches,
|
|
35
|
+
# each marked with whether the workspace has it selected.
|
|
36
|
+
def repositories
|
|
37
|
+
selected = @connection.repository_names.map(&:downcase).to_set
|
|
38
|
+
|
|
39
|
+
render json: {
|
|
40
|
+
repositories: @connection.client.repositories.map do |repo|
|
|
41
|
+
repo.merge("selected" => selected.include?(repo["full_name"].downcase))
|
|
42
|
+
end
|
|
43
|
+
}
|
|
44
|
+
end
|
|
45
|
+
|
|
46
|
+
# PATCH /api/github_connection { repositories: ["owner/name", ...] }
|
|
47
|
+
#
|
|
48
|
+
# Only names GitHub lists for this token are kept: the selection is
|
|
49
|
+
# what a checkout sandbox trusts, so it is never taken from the client.
|
|
50
|
+
def update
|
|
51
|
+
# Not params.require: an empty list (clear the selection) is valid.
|
|
52
|
+
requested = params[:repositories]
|
|
53
|
+
unless requested.is_a?(Array) && requested.all? { |name| name.is_a?(String) }
|
|
54
|
+
return render json: { error: "repositories must be a list of owner/name strings" }, status: :bad_request
|
|
55
|
+
end
|
|
56
|
+
|
|
57
|
+
available = @connection.client.repositories.index_by { |repo| repo["full_name"].downcase }
|
|
58
|
+
unknown = requested.reject { |name| available.key?(name.downcase) }
|
|
59
|
+
if unknown.any?
|
|
60
|
+
return render json: { error: "Not reachable with this GitHub connection: #{unknown.join(', ')}" },
|
|
61
|
+
status: :unprocessable_entity
|
|
62
|
+
end
|
|
63
|
+
|
|
64
|
+
@connection.update!(repositories: requested.map { |name| available.fetch(name.downcase) }.uniq { |repo| repo["id"] })
|
|
65
|
+
render json: { connected: true, connection: @connection.as_summary }
|
|
66
|
+
end
|
|
67
|
+
|
|
68
|
+
# DELETE /api/github_connection
|
|
69
|
+
def destroy
|
|
70
|
+
@connection.destroy!
|
|
71
|
+
head :no_content
|
|
72
|
+
end
|
|
73
|
+
|
|
74
|
+
# GET /api/github_connection/connect — starts the OAuth web flow.
|
|
75
|
+
def connect
|
|
76
|
+
state = SecureRandom.urlsafe_base64(32)
|
|
77
|
+
session[STATE_SESSION_KEY] = state
|
|
78
|
+
|
|
79
|
+
redirect_to GithubClient.authorize_url(redirect_uri: callback_url, state: state), allow_other_host: true
|
|
80
|
+
end
|
|
81
|
+
|
|
82
|
+
# GET /api/github_connection/callback?code=...&state=...
|
|
83
|
+
def callback
|
|
84
|
+
# Single use: read and cleared before anything else can fail.
|
|
85
|
+
expected = session.delete(STATE_SESSION_KEY)
|
|
86
|
+
return redirect_to_settings(github: "denied") if params[:error].present?
|
|
87
|
+
|
|
88
|
+
unless expected.present? && params[:state].is_a?(String) &&
|
|
89
|
+
ActiveSupport::SecurityUtils.secure_compare(expected, params[:state])
|
|
90
|
+
return redirect_to_settings(github: "invalid_state")
|
|
91
|
+
end
|
|
92
|
+
return redirect_to_settings(github: "missing_code") unless params[:code].is_a?(String) && params[:code].present?
|
|
93
|
+
|
|
94
|
+
grant = GithubClient.exchange_code(code: params[:code], redirect_uri: callback_url)
|
|
95
|
+
user = GithubClient.new(grant[:access_token]).user
|
|
96
|
+
|
|
97
|
+
# Built through the owner scope, like a provider key, so the new
|
|
98
|
+
# record carries whichever owner column this install uses.
|
|
99
|
+
connection = owned(GithubConnection).first || owned(GithubConnection).new
|
|
100
|
+
# A different GitHub account starts with nothing selected: the old
|
|
101
|
+
# selection was checked against the other account's access.
|
|
102
|
+
connection.repositories = [] if connection.persisted? && connection.github_user_id != user["id"]
|
|
103
|
+
connection.update!(
|
|
104
|
+
access_token: grant[:access_token],
|
|
105
|
+
scopes: grant[:scope],
|
|
106
|
+
github_user_id: user["id"],
|
|
107
|
+
login: user["login"],
|
|
108
|
+
avatar_url: user["avatar_url"]
|
|
109
|
+
)
|
|
110
|
+
|
|
111
|
+
redirect_to_settings(github: "connected")
|
|
112
|
+
rescue GithubClient::Error => e
|
|
113
|
+
Rails.logger.warn("[ActionAgent] GitHub OAuth callback failed: #{e.message}")
|
|
114
|
+
redirect_to_settings(github: "error")
|
|
115
|
+
end
|
|
116
|
+
|
|
117
|
+
private
|
|
118
|
+
|
|
119
|
+
def set_connection
|
|
120
|
+
@connection = owned(GithubConnection).first!
|
|
121
|
+
end
|
|
122
|
+
|
|
123
|
+
def require_github_oauth!
|
|
124
|
+
return if ActionAgent.github_oauth_configured?
|
|
125
|
+
|
|
126
|
+
redirect_to_settings(github: "not_configured")
|
|
127
|
+
end
|
|
128
|
+
|
|
129
|
+
def callback_url
|
|
130
|
+
"#{request.base_url}#{request.script_name}/api/github_connection/callback"
|
|
131
|
+
end
|
|
132
|
+
|
|
133
|
+
def redirect_to_settings(**query)
|
|
134
|
+
redirect_to "#{request.script_name}/settings?#{{ tab: 'integrations' }.merge(query).to_query}"
|
|
135
|
+
end
|
|
136
|
+
|
|
137
|
+
def github_unauthorized
|
|
138
|
+
render json: { error: "GitHub rejected the stored token. Reconnect GitHub.", reconnect_required: true },
|
|
139
|
+
status: :unprocessable_entity
|
|
140
|
+
end
|
|
141
|
+
|
|
142
|
+
def github_unavailable(exception)
|
|
143
|
+
render json: { error: exception.message }, status: :bad_gateway
|
|
144
|
+
end
|
|
145
|
+
end
|
|
146
|
+
end
|
|
147
|
+
end
|
|
@@ -11,7 +11,9 @@ module ActionAgent
|
|
|
11
11
|
# schema tools (ActiveAgent::SchemaTools, the classes the dashboard
|
|
12
12
|
# discovers) is callable directly — find_<records>, count_<records>,
|
|
13
13
|
# get_<record> — as this key's caller, so a client reads the host's
|
|
14
|
-
# records under the same scope an agent run would (#439).
|
|
14
|
+
# records under the same scope an agent run would (#439). The
|
|
15
|
+
# dashboard's own evaluation and telemetry tools (MCPDashboardTools)
|
|
16
|
+
# are offered beside them.
|
|
15
17
|
# - resources/list & resources/read: each agent is an agent://<slug>
|
|
16
18
|
# resource whose content is its live scorecard (config + stats + memory
|
|
17
19
|
# summary from the solid_agent datasets).
|
|
@@ -30,6 +32,8 @@ module ActionAgent
|
|
|
30
32
|
skip_forgery_protection
|
|
31
33
|
before_action :authenticate_api_key!, except: [ :unsupported ]
|
|
32
34
|
|
|
35
|
+
include MCPDashboardTools
|
|
36
|
+
|
|
33
37
|
PROTOCOL_VERSION = "2025-03-26"
|
|
34
38
|
JSONRPC_METHOD_NOT_FOUND = -32601
|
|
35
39
|
JSONRPC_INVALID_PARAMS = -32602
|
|
@@ -105,6 +109,21 @@ module ActionAgent
|
|
|
105
109
|
@owner = api_key.owner
|
|
106
110
|
end
|
|
107
111
|
|
|
112
|
+
# The key's owner stands in for the signed-in owner, so the dashboard's
|
|
113
|
+
# own scopes (owned, owner_agents, owned_traces, RunSandbox) read what
|
|
114
|
+
# the JSON API would read for that owner.
|
|
115
|
+
def current_owner
|
|
116
|
+
@owner
|
|
117
|
+
end
|
|
118
|
+
|
|
119
|
+
# The key's user when it records one. Otherwise the owner, unless the
|
|
120
|
+
# owner is an account: a tenant is never a user.
|
|
121
|
+
def current_user
|
|
122
|
+
return @api_key.user if @api_key.respond_to?(:user) && @api_key.user
|
|
123
|
+
|
|
124
|
+
ActionAgent.multi_tenant? ? nil : @owner
|
|
125
|
+
end
|
|
126
|
+
|
|
108
127
|
# The agents this key can reach. A key belongs to whoever owns it, and
|
|
109
128
|
# a single-user install has no owner, so the key reaches every agent
|
|
110
129
|
# the dashboard holds.
|
|
@@ -149,12 +168,23 @@ module ActionAgent
|
|
|
149
168
|
protocolVersion: PROTOCOL_VERSION,
|
|
150
169
|
capabilities: { tools: {}, resources: {} },
|
|
151
170
|
serverInfo: { name: "activeagents", version: "1.0" },
|
|
152
|
-
instructions:
|
|
153
|
-
"application's records directly, as the caller this key authenticates. Each agent://<slug> " \
|
|
154
|
-
"resource returns the agent's live scorecard."
|
|
171
|
+
instructions: initialize_instructions
|
|
155
172
|
}
|
|
156
173
|
end
|
|
157
174
|
|
|
175
|
+
def initialize_instructions
|
|
176
|
+
text = "Each run_<slug> tool runs one of this account's agents; each find_, count_ and get_ tool reads the " \
|
|
177
|
+
"host application's records directly, as the caller this key authenticates. Each agent://<slug> " \
|
|
178
|
+
"resource returns the agent's live scorecard."
|
|
179
|
+
return text unless ActionAgent.mcp_dashboard_tools?
|
|
180
|
+
|
|
181
|
+
"#{text} The evaluations_, evaluation_runs_ and traces_ tools work on this account's evaluations and " \
|
|
182
|
+
"telemetry: edit the agent in your own checkout, start a run with evaluations_run (pass sandbox_id to run " \
|
|
183
|
+
"against a checkout sandbox), poll evaluation_runs_get for its status, results and fix items, compare runs " \
|
|
184
|
+
"with evaluation_runs_compare, and read a failing result's trace with traces_get (traces_search finds " \
|
|
185
|
+
"recent failures)."
|
|
186
|
+
end
|
|
187
|
+
|
|
158
188
|
MESSAGE_INPUT_SCHEMA = {
|
|
159
189
|
type: "object",
|
|
160
190
|
properties: {
|
|
@@ -182,7 +212,7 @@ module ActionAgent
|
|
|
182
212
|
agent_tools
|
|
183
213
|
end
|
|
184
214
|
|
|
185
|
-
{ tools: tools + schema_tools_list }
|
|
215
|
+
{ tools: dashboard_tools_list + tools + schema_tools_list }
|
|
186
216
|
end
|
|
187
217
|
|
|
188
218
|
# The host's schema tools, offered as the same tool definitions an
|
|
@@ -206,6 +236,9 @@ module ActionAgent
|
|
|
206
236
|
|
|
207
237
|
def tools_call
|
|
208
238
|
name = params.dig(:params, :name).to_s
|
|
239
|
+
# The three families' names are disjoint (see MCPDashboardTools), so
|
|
240
|
+
# this order never decides between two tools.
|
|
241
|
+
return dashboard_tool_call(name) if dashboard_tool?(name)
|
|
209
242
|
return schema_tool_call(name) if ActionAgent.mcp_schema_tools? && ActionAgent.schema_tool_class_for(name)
|
|
210
243
|
|
|
211
244
|
slug, action = name.delete_prefix("run_").split("__", 2)
|