actionagent 1.7.2 → 1.8.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/app/assets/builds/action_agent.css +1 -1
- data/app/assets/builds/action_agent.js +59 -54
- data/app/controllers/action_agent/api/agents_controller.rb +19 -3
- data/app/controllers/action_agent/api/code_sessions_controller.rb +156 -0
- data/app/controllers/action_agent/api/evaluations_controller.rb +27 -130
- data/app/controllers/action_agent/api/github_connections_controller.rb +147 -0
- data/app/controllers/action_agent/api/mcp_controller.rb +38 -5
- data/app/controllers/action_agent/api/mcp_servers_controller.rb +10 -1
- data/app/controllers/action_agent/api/provider_keys_controller.rb +54 -3
- data/app/controllers/action_agent/api/provider_models_controller.rb +10 -6
- data/app/controllers/action_agent/api/sandboxes_controller.rb +97 -6
- data/app/controllers/concerns/action_agent/api/evaluation_run_starting.rb +93 -0
- data/app/controllers/concerns/action_agent/api/mcp_dashboard_tools.rb +507 -0
- data/app/controllers/concerns/action_agent/api/run_sandbox.rb +65 -0
- data/app/jobs/action_agent/code_session_job.rb +166 -0
- data/app/jobs/action_agent/sandbox_cleanup_job.rb +80 -11
- data/app/jobs/action_agent/sandbox_provision_job.rb +122 -14
- data/app/jobs/action_agent/sandbox_run_job.rb +10 -3
- data/app/models/action_agent/agent.rb +16 -6
- data/app/models/action_agent/agent_run.rb +20 -1
- data/app/models/action_agent/code_session.rb +141 -0
- data/app/models/action_agent/evaluation_run.rb +23 -1
- data/app/models/action_agent/github_connection.rb +75 -0
- data/app/models/action_agent/provider_key.rb +142 -10
- data/app/models/action_agent/sandbox_session.rb +193 -17
- data/app/serializers/action_agent/evaluation_serializer.rb +118 -0
- data/app/serializers/action_agent/telemetry_trace_serializer.rb +10 -2
- data/app/services/action_agent/agent_execution_service.rb +4 -2
- data/app/services/action_agent/agent_tool_roster.rb +20 -9
- data/app/services/action_agent/claude_code_auth.rb +86 -0
- data/app/services/action_agent/dashboard_assistant_service.rb +47 -5
- data/app/services/action_agent/evaluation_tool_resolver.rb +18 -0
- data/app/services/action_agent/github_client.rb +111 -0
- data/app/services/action_agent/local_sandbox_backend.rb +1689 -0
- data/app/services/action_agent/local_sandbox_databases.rb +257 -0
- data/app/services/action_agent/mcp_client.rb +5 -1
- data/app/services/action_agent/mcp_tool_dispatcher.rb +137 -17
- data/app/services/action_agent/mock_sandbox_backend.rb +39 -0
- data/app/services/action_agent/ollama_host_probe.rb +75 -0
- data/app/services/action_agent/payload_bounds.rb +36 -0
- data/app/services/action_agent/sandbox_manifest.rb +67 -0
- data/app/services/action_agent/sandbox_orchestrator.rb +69 -14
- data/app/services/action_agent/scenario_evaluation_runner.rb +50 -4
- data/app/services/action_agent/secret_scrubber.rb +37 -0
- data/app/services/action_agent/tool_discovery.rb +19 -5
- data/config/routes.rb +22 -3
- data/lib/action_agent/engine.rb +1 -0
- data/lib/action_agent/version.rb +1 -1
- data/lib/action_agent.rb +147 -3
- data/lib/generators/action_agent/install_generator.rb +30 -3
- data/lib/generators/action_agent/templates/action_agent.rb.erb +44 -0
- data/lib/generators/action_agent/templates/add_provider_key_api_key.rb.erb +25 -0
- data/lib/generators/action_agent/templates/create_active_agent_code_sessions.rb.erb +59 -0
- data/lib/generators/action_agent/templates/create_active_agent_dashboard_tables.rb.erb +2 -0
- data/lib/generators/action_agent/templates/create_active_agent_github_connections.rb.erb +61 -0
- data/lib/tasks/claude_code.rake +16 -0
- data/lib/tasks/sandbox.rake +26 -0
- metadata +23 -1
|
@@ -6,6 +6,7 @@ module ActionAgent
|
|
|
6
6
|
owned_by :user, :account
|
|
7
7
|
|
|
8
8
|
belongs_to :agent_template, optional: true
|
|
9
|
+
has_many :code_sessions, dependent: :destroy
|
|
9
10
|
|
|
10
11
|
# Session statuses
|
|
11
12
|
enum :status, {
|
|
@@ -18,8 +19,14 @@ module ActionAgent
|
|
|
18
19
|
failed: 6
|
|
19
20
|
}
|
|
20
21
|
|
|
21
|
-
# Sandbox types
|
|
22
|
-
|
|
22
|
+
# Sandbox types. +app_runtime+ boots a checkout of one of the owner's
|
|
23
|
+
# GitHub repositories (see GithubConnection) and exposes that app's own
|
|
24
|
+
# runtime, so agents and evaluations can use its tools.
|
|
25
|
+
SANDBOX_TYPES = %w[playwright_mcp terminal research app_runtime].freeze
|
|
26
|
+
|
|
27
|
+
# MCP server keys naming a checkout sandbox's app runtime, as an agent's
|
|
28
|
+
# mcp_servers lists them: "sandbox:<session_id>".
|
|
29
|
+
RUNTIME_SERVER_PREFIX = "sandbox:"
|
|
23
30
|
|
|
24
31
|
# Free tier limits
|
|
25
32
|
FREE_TIER_LIMITS = {
|
|
@@ -29,9 +36,21 @@ module ActionAgent
|
|
|
29
36
|
session_duration_minutes: 15
|
|
30
37
|
}.freeze
|
|
31
38
|
|
|
39
|
+
# How long a checkout sandbox lives. Booting one (clone, bundle install,
|
|
40
|
+
# db:prepare) can take minutes, and it is worked in for a while after —
|
|
41
|
+
# Claude Code sessions, agents using its tools — so the free tier's 15
|
|
42
|
+
# minutes would expire it about as soon as it was ready.
|
|
43
|
+
APP_RUNTIME_SESSION_DURATION = 2.hours
|
|
44
|
+
|
|
45
|
+
encrypts :runtime_mcp_token if ActionAgent.encrypt_credentials
|
|
46
|
+
|
|
32
47
|
# Validations
|
|
33
48
|
validates :session_id, presence: true, uniqueness: true
|
|
34
49
|
validates :sandbox_type, inclusion: { in: SANDBOX_TYPES }
|
|
50
|
+
validates :repository, presence: true, if: :app_runtime?
|
|
51
|
+
validates :repository_ref, length: { maximum: 255 }, format: { without: /\A-|\s|\.\./, message: "is not a valid git ref" },
|
|
52
|
+
allow_blank: true
|
|
53
|
+
validate :repository_available, on: :create, if: :app_runtime?
|
|
35
54
|
|
|
36
55
|
# Callbacks
|
|
37
56
|
before_validation :generate_session_id, on: :create
|
|
@@ -51,6 +70,100 @@ module ActionAgent
|
|
|
51
70
|
Array(mcp_servers).filter_map { |key| MCPCatalog.find(key) }
|
|
52
71
|
end
|
|
53
72
|
|
|
73
|
+
def self.runtime_server_key?(key)
|
|
74
|
+
key.to_s.start_with?(RUNTIME_SERVER_PREFIX)
|
|
75
|
+
end
|
|
76
|
+
|
|
77
|
+
# The MCP catalog entry for a checkout sandbox's app runtime, looked up
|
|
78
|
+
# among +owner+'s sessions only. Nil unless the session is live and its
|
|
79
|
+
# backend reported an endpoint.
|
|
80
|
+
#
|
|
81
|
+
# @return [Hash, nil]
|
|
82
|
+
def self.runtime_server_entry(key, owner:)
|
|
83
|
+
return nil unless runtime_server_key?(key)
|
|
84
|
+
|
|
85
|
+
session = for_owner(owner).find_by(session_id: key.to_s.delete_prefix(RUNTIME_SERVER_PREFIX))
|
|
86
|
+
session&.runtime_server_entry
|
|
87
|
+
end
|
|
88
|
+
|
|
89
|
+
# The live runtimes in +scope+ (a relation already scoped to an owner) as
|
|
90
|
+
# MCP server listings: the catalog entry shape MCPCatalog serves, without
|
|
91
|
+
# the bearer token runtime_server_entry carries for the dispatcher.
|
|
92
|
+
#
|
|
93
|
+
# @return [Array<Hash>]
|
|
94
|
+
def self.runtime_server_listings(scope)
|
|
95
|
+
scope.active.by_type("app_runtime")
|
|
96
|
+
.where.not(runtime_mcp_url: [ nil, "" ])
|
|
97
|
+
.where("expires_at > ?", Time.current)
|
|
98
|
+
.recent.limit(20)
|
|
99
|
+
.filter_map(&:runtime_server_listing)
|
|
100
|
+
end
|
|
101
|
+
|
|
102
|
+
def app_runtime?
|
|
103
|
+
sandbox_type == "app_runtime"
|
|
104
|
+
end
|
|
105
|
+
|
|
106
|
+
def runtime_server_key
|
|
107
|
+
"#{RUNTIME_SERVER_PREFIX}#{session_id}"
|
|
108
|
+
end
|
|
109
|
+
|
|
110
|
+
# This session's app runtime as an MCP catalog entry — the shape
|
|
111
|
+
# MCPToolDispatcher reaches servers through.
|
|
112
|
+
def runtime_server_entry
|
|
113
|
+
return nil unless app_runtime? && active? && runtime_mcp_url.present?
|
|
114
|
+
|
|
115
|
+
{
|
|
116
|
+
key: runtime_server_key,
|
|
117
|
+
name: "#{repository}@#{repository_ref} (sandbox)",
|
|
118
|
+
description: "App runtime booted from a checkout of #{repository}",
|
|
119
|
+
transport: "streamable_http",
|
|
120
|
+
url: runtime_mcp_url,
|
|
121
|
+
headers: runtime_mcp_token.present? ? { "Authorization" => "Bearer #{runtime_mcp_token}" } : {}
|
|
122
|
+
}
|
|
123
|
+
end
|
|
124
|
+
|
|
125
|
+
# This runtime as a token-free MCP server listing (see
|
|
126
|
+
# runtime_server_listings), or nil when it is not live.
|
|
127
|
+
def runtime_server_listing
|
|
128
|
+
entry = runtime_server_entry or return nil
|
|
129
|
+
|
|
130
|
+
entry.except(:headers).merge(
|
|
131
|
+
command: nil, package: nil, categories: [ "runtime" ], docs_url: nil,
|
|
132
|
+
sandbox: false, sandbox_type: "app_runtime", first_party: false,
|
|
133
|
+
requires_credentials: [], tools: [], runtime: true
|
|
134
|
+
)
|
|
135
|
+
end
|
|
136
|
+
|
|
137
|
+
# What a sandbox backend clones for an app_runtime session: repository,
|
|
138
|
+
# ref, clone URL and the credentials to fetch it. Nil for any other
|
|
139
|
+
# sandbox type. Carries the owner's GitHub token, so it goes to the
|
|
140
|
+
# backend and never into a response.
|
|
141
|
+
#
|
|
142
|
+
# @return [Hash, nil]
|
|
143
|
+
def checkout_spec
|
|
144
|
+
return nil unless app_runtime?
|
|
145
|
+
|
|
146
|
+
github_connection&.checkout_spec(repository, ref: repository_ref)
|
|
147
|
+
end
|
|
148
|
+
|
|
149
|
+
# The GitHub connection whose selection this session's checkout comes
|
|
150
|
+
# from.
|
|
151
|
+
def github_connection
|
|
152
|
+
owners_record(GithubConnection)
|
|
153
|
+
end
|
|
154
|
+
|
|
155
|
+
# Environment the backend passes into an app_runtime checkout, so the
|
|
156
|
+
# booted app can run Claude Code sessions against it with the owner's
|
|
157
|
+
# connected credential (Settings -> Integrations). Empty when none is
|
|
158
|
+
# connected. Secret — for the backend, never a response.
|
|
159
|
+
#
|
|
160
|
+
# @return [Hash{String => String}]
|
|
161
|
+
def runtime_environment
|
|
162
|
+
return {} unless app_runtime?
|
|
163
|
+
|
|
164
|
+
owners_record(ProviderKey.where(provider: "claude_code"))&.runtime_environment || {}
|
|
165
|
+
end
|
|
166
|
+
|
|
54
167
|
# Check if session is still valid
|
|
55
168
|
def active?
|
|
56
169
|
!expired? && !failed? && !completed? && expires_at > Time.current
|
|
@@ -95,28 +208,42 @@ module ActionAgent
|
|
|
95
208
|
|
|
96
209
|
update!(status: :provisioning)
|
|
97
210
|
|
|
98
|
-
#
|
|
99
|
-
|
|
211
|
+
# A checkout always boots in the background: cloning and setting up a
|
|
212
|
+
# real app takes minutes, and a request must not wait on it, in
|
|
213
|
+
# development either. The client polls the session until it is ready
|
|
214
|
+
# (or failed). The other types are simulated in development and test,
|
|
215
|
+
# synchronously for immediate feedback.
|
|
216
|
+
if !app_runtime? && (Rails.env.development? || Rails.env.test?)
|
|
100
217
|
SandboxProvisionJob.perform_now(id)
|
|
101
218
|
else
|
|
102
219
|
SandboxProvisionJob.perform_later(id)
|
|
103
220
|
end
|
|
104
221
|
end
|
|
105
222
|
|
|
106
|
-
# Mark as ready with Cloud Run URL
|
|
107
|
-
|
|
108
|
-
|
|
109
|
-
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
)
|
|
223
|
+
# Mark as ready with Cloud Run URL. A checkout sandbox's backend also
|
|
224
|
+
# reports the app runtime's MCP endpoint and the token it expects.
|
|
225
|
+
def mark_ready!(cloud_run_url:, cloud_run_job_id: nil, runtime_mcp_url: nil, runtime_mcp_token: nil)
|
|
226
|
+
attributes = { status: :ready, cloud_run_url: cloud_run_url, cloud_run_job_id: cloud_run_job_id }
|
|
227
|
+
attributes[:runtime_mcp_url] = runtime_mcp_url if runtime_mcp_url
|
|
228
|
+
attributes[:runtime_mcp_token] = runtime_mcp_token if runtime_mcp_token
|
|
229
|
+
update!(attributes)
|
|
113
230
|
end
|
|
114
231
|
|
|
115
|
-
# Expire the session
|
|
232
|
+
# Expire the session. Its runtime stops being reachable at once — the
|
|
233
|
+
# endpoint and its token are cleared, so no agent is handed a runtime
|
|
234
|
+
# that is going away — and the backend's resource is released by
|
|
235
|
+
# SandboxCleanupJob, which keeps the handle until that succeeds.
|
|
236
|
+
#
|
|
237
|
+
# Under the row lock, which reloads the row first: callers (DELETE, the
|
|
238
|
+
# reaper) loaded this copy earlier, and SandboxProvisionJob may have
|
|
239
|
+
# marked it ready since. Acting on the stale copy would neither clear the
|
|
240
|
+
# endpoint it recorded (nil -> nil writes nothing) nor see the handle to
|
|
241
|
+
# terminate, leaving the booted sandbox running.
|
|
116
242
|
def expire!
|
|
117
|
-
update!(status: :expired)
|
|
118
|
-
#
|
|
119
|
-
|
|
243
|
+
with_lock { update!(status: :expired, runtime_mcp_url: nil, runtime_mcp_token: nil) }
|
|
244
|
+
# A checkout with no handle may still have a boot behind it (its job
|
|
245
|
+
# died mid-boot, say); the cleanup job asks the backend for it.
|
|
246
|
+
SandboxCleanupJob.perform_later(id) if cloud_run_job_id.present? || app_runtime?
|
|
120
247
|
end
|
|
121
248
|
|
|
122
249
|
# Summary for API responses
|
|
@@ -132,10 +259,30 @@ module ActionAgent
|
|
|
132
259
|
expires_at: expires_at&.iso8601,
|
|
133
260
|
created_at: created_at.iso8601,
|
|
134
261
|
cloud_run_url: cloud_run_url,
|
|
135
|
-
mcp_servers: Array(mcp_servers)
|
|
262
|
+
mcp_servers: Array(mcp_servers),
|
|
263
|
+
repository: repository,
|
|
264
|
+
repository_ref: repository_ref,
|
|
265
|
+
# The key an agent adds to its mcp_servers to use this runtime's
|
|
266
|
+
# tools; nil until the backend has reported the endpoint.
|
|
267
|
+
runtime_server_key: runtime_mcp_url.present? ? runtime_server_key : nil,
|
|
268
|
+
# Why provisioning failed. Scrubbed of the session's secrets when
|
|
269
|
+
# SandboxProvisionJob stored it.
|
|
270
|
+
error_message: error_summary
|
|
136
271
|
}
|
|
137
272
|
end
|
|
138
273
|
|
|
274
|
+
# A failed boot's message is a one-line reason followed by the tail of
|
|
275
|
+
# the failing step's log, and the log's last lines usually hold the
|
|
276
|
+
# actual error — so a long message keeps its head and its end.
|
|
277
|
+
ERROR_SUMMARY_HEAD = 300
|
|
278
|
+
ERROR_SUMMARY_TAIL = 1_700
|
|
279
|
+
|
|
280
|
+
def error_summary
|
|
281
|
+
return error_message if error_message.nil? || error_message.length <= ERROR_SUMMARY_HEAD + ERROR_SUMMARY_TAIL
|
|
282
|
+
|
|
283
|
+
"#{error_message[0, ERROR_SUMMARY_HEAD]}\n…\n#{error_message[-ERROR_SUMMARY_TAIL..]}"
|
|
284
|
+
end
|
|
285
|
+
|
|
139
286
|
# Detailed info including runs
|
|
140
287
|
def details
|
|
141
288
|
summary.merge(
|
|
@@ -147,12 +294,41 @@ module ActionAgent
|
|
|
147
294
|
|
|
148
295
|
private
|
|
149
296
|
|
|
297
|
+
# The owner's record in +scope+, found through that model's own owner
|
|
298
|
+
# column. GitHub connections and provider keys are account-owned before
|
|
299
|
+
# user-owned, the opposite of a session, so the session's #owner is not
|
|
300
|
+
# necessarily theirs.
|
|
301
|
+
def owners_record(scope)
|
|
302
|
+
scope = scope.all
|
|
303
|
+
case scope.klass.owner_association
|
|
304
|
+
when :account then account_id && scope.find_by(account_id: account_id)
|
|
305
|
+
when :user then user_id && scope.find_by(user_id: user_id)
|
|
306
|
+
else scope.first
|
|
307
|
+
end
|
|
308
|
+
end
|
|
309
|
+
|
|
310
|
+
def repository_available
|
|
311
|
+
return if repository.blank?
|
|
312
|
+
|
|
313
|
+
connection = github_connection
|
|
314
|
+
if connection.nil?
|
|
315
|
+
errors.add(:repository, "needs a GitHub connection (Settings -> Integrations)")
|
|
316
|
+
elsif (repo = connection.repository(repository))
|
|
317
|
+
# Canonical spelling, and the default branch unless a ref was asked for.
|
|
318
|
+
self.repository = repo["full_name"]
|
|
319
|
+
self.repository_ref = repository_ref.presence || repo["default_branch"]
|
|
320
|
+
else
|
|
321
|
+
errors.add(:repository, "is not one of the repositories selected in Settings -> Integrations")
|
|
322
|
+
end
|
|
323
|
+
end
|
|
324
|
+
|
|
150
325
|
def generate_session_id
|
|
151
326
|
self.session_id ||= SecureRandom.uuid
|
|
152
327
|
end
|
|
153
328
|
|
|
154
329
|
def set_expiration
|
|
155
|
-
|
|
330
|
+
duration = app_runtime? ? APP_RUNTIME_SESSION_DURATION : FREE_TIER_LIMITS[:session_duration_minutes].minutes
|
|
331
|
+
self.expires_at ||= duration.from_now
|
|
156
332
|
self.max_runs ||= FREE_TIER_LIMITS[:max_runs]
|
|
157
333
|
self.timeout_seconds ||= FREE_TIER_LIMITS[:timeout_seconds]
|
|
158
334
|
end
|
|
@@ -0,0 +1,118 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module ActionAgent
|
|
4
|
+
# Used to render evaluations and their runs for the dashboard's JSON API
|
|
5
|
+
# (Api::EvaluationsController) and its MCP facade (Api::MCPController), so
|
|
6
|
+
# both describe an evaluation the same way.
|
|
7
|
+
#
|
|
8
|
+
# A run's `number` is its position in its evaluation's history, oldest = 1,
|
|
9
|
+
# so a client can say "Run #3" and "vs #2".
|
|
10
|
+
module EvaluationSerializer
|
|
11
|
+
module_function
|
|
12
|
+
|
|
13
|
+
# Returns the evaluation with its latest run in full and just enough of
|
|
14
|
+
# the run before it to show movement ("+3 passed vs #2") without a
|
|
15
|
+
# request per evaluation.
|
|
16
|
+
def evaluation(evaluation)
|
|
17
|
+
summary = summary(evaluation)
|
|
18
|
+
latest, previous = recent_runs(evaluation, 2)
|
|
19
|
+
|
|
20
|
+
summary.merge(
|
|
21
|
+
latest_run: latest ? run(latest, number: summary[:run_count]) : nil,
|
|
22
|
+
previous_run: previous ? run_summary(previous, number: summary[:run_count] - 1) : nil
|
|
23
|
+
)
|
|
24
|
+
end
|
|
25
|
+
|
|
26
|
+
# Returns the evaluation's configuration and run count, without its runs.
|
|
27
|
+
def summary(evaluation)
|
|
28
|
+
{
|
|
29
|
+
id: evaluation.id,
|
|
30
|
+
name: evaluation.name,
|
|
31
|
+
agent: { id: evaluation.agent.id, name: evaluation.agent.name, slug: evaluation.agent.slug },
|
|
32
|
+
judge_kind: evaluation.judge_kind,
|
|
33
|
+
judge_model: evaluation.judge_model,
|
|
34
|
+
criteria: evaluation.criteria,
|
|
35
|
+
compare_models: evaluation.compare_models,
|
|
36
|
+
config: evaluation.config,
|
|
37
|
+
sample_size: evaluation.sample_size,
|
|
38
|
+
scenario_suite: evaluation.scenario_suite?,
|
|
39
|
+
scenario_count: evaluation.scenarios.size,
|
|
40
|
+
scenario_groups: evaluation.scenario_suite? ? evaluation.scenario_groups : [],
|
|
41
|
+
created_at: evaluation.created_at.iso8601,
|
|
42
|
+
# size reads a preloaded association and COUNTs otherwise.
|
|
43
|
+
run_count: evaluation.evaluation_runs.size
|
|
44
|
+
}
|
|
45
|
+
end
|
|
46
|
+
|
|
47
|
+
# Returns the run summary plus its scores, selection, models, usage and
|
|
48
|
+
# error.
|
|
49
|
+
def run(run, number: nil)
|
|
50
|
+
run_summary(run, number: number).merge(
|
|
51
|
+
scores: run.scores,
|
|
52
|
+
selection: run.selection,
|
|
53
|
+
models: run.models,
|
|
54
|
+
usage: run.usage,
|
|
55
|
+
error_message: run.error_message
|
|
56
|
+
)
|
|
57
|
+
end
|
|
58
|
+
|
|
59
|
+
# Returns the run's status and headline numbers. `sandbox` is the
|
|
60
|
+
# checkout sandbox the run replayed against, if any: its session id and
|
|
61
|
+
# checkout, never its token.
|
|
62
|
+
def run_summary(run, number: nil)
|
|
63
|
+
{
|
|
64
|
+
id: run.id,
|
|
65
|
+
number: number,
|
|
66
|
+
status: run.status,
|
|
67
|
+
average_score: average_score(run),
|
|
68
|
+
samples_evaluated: run.samples_evaluated,
|
|
69
|
+
samples_passed: run.samples_passed,
|
|
70
|
+
completed_at: run.completed_at&.iso8601,
|
|
71
|
+
created_at: run.created_at.iso8601,
|
|
72
|
+
sandbox: run.sandbox
|
|
73
|
+
}
|
|
74
|
+
end
|
|
75
|
+
|
|
76
|
+
# Returns up to +limit+ of the evaluation's runs, newest first. Sorts the
|
|
77
|
+
# preloaded association when one was loaded rather than issuing one
|
|
78
|
+
# ORDER BY query per evaluation.
|
|
79
|
+
def recent_runs(evaluation, limit)
|
|
80
|
+
runs = evaluation.evaluation_runs
|
|
81
|
+
if runs.loaded?
|
|
82
|
+
runs.sort_by { |run| [ run.created_at, run.id ] }.reverse.first(limit)
|
|
83
|
+
else
|
|
84
|
+
runs.recent.limit(limit).to_a
|
|
85
|
+
end
|
|
86
|
+
end
|
|
87
|
+
|
|
88
|
+
# Returns the run's position in its evaluation's history, oldest = 1.
|
|
89
|
+
def run_number(evaluation, run)
|
|
90
|
+
evaluation.evaluation_runs.where("created_at < ? OR (created_at = ? AND id <= ?)", run.created_at, run.created_at, run.id).count
|
|
91
|
+
end
|
|
92
|
+
|
|
93
|
+
# Returns the run's fix items, or [] when they cannot be built. They are
|
|
94
|
+
# derived from every persisted result's diagnosis, which older runs
|
|
95
|
+
# recorded in earlier shapes, so a run they fail for still serves its
|
|
96
|
+
# results. Without +links+ the item paths are relative to the mount.
|
|
97
|
+
def fix_items(run, links: nil)
|
|
98
|
+
links ? run.fix_items(links: links) : run.fix_items
|
|
99
|
+
rescue StandardError => e
|
|
100
|
+
Rails.logger.warn(
|
|
101
|
+
"[ActionAgent] evaluation run #{run.id} fix_items failed: #{e.class}: #{e.message}"
|
|
102
|
+
)
|
|
103
|
+
[]
|
|
104
|
+
end
|
|
105
|
+
|
|
106
|
+
# Returns the run's average score, or nil when its scores payload cannot
|
|
107
|
+
# be averaged, so one bad run degrades its own headline number rather
|
|
108
|
+
# than failing a whole listing.
|
|
109
|
+
def average_score(run)
|
|
110
|
+
run.average_score
|
|
111
|
+
rescue StandardError => e
|
|
112
|
+
Rails.logger.warn(
|
|
113
|
+
"[ActionAgent] evaluation run #{run.id} average_score failed: #{e.class}: #{e.message}"
|
|
114
|
+
)
|
|
115
|
+
nil
|
|
116
|
+
end
|
|
117
|
+
end
|
|
118
|
+
end
|
|
@@ -21,7 +21,16 @@ module ActionAgent
|
|
|
21
21
|
@trace = trace
|
|
22
22
|
end
|
|
23
23
|
|
|
24
|
+
def self.row(trace)
|
|
25
|
+
new(trace).row
|
|
26
|
+
end
|
|
27
|
+
|
|
24
28
|
def summary
|
|
29
|
+
row.merge(spans: serialized_spans)
|
|
30
|
+
end
|
|
31
|
+
|
|
32
|
+
# The summary without its spans: one line of a trace listing.
|
|
33
|
+
def row
|
|
25
34
|
{
|
|
26
35
|
id: @trace.id,
|
|
27
36
|
trace_id: @trace.trace_id,
|
|
@@ -48,8 +57,7 @@ module ActionAgent
|
|
|
48
57
|
model: @trace.model,
|
|
49
58
|
input_tokens: @trace.total_input_tokens,
|
|
50
59
|
output_tokens: @trace.total_output_tokens
|
|
51
|
-
)
|
|
52
|
-
spans: serialized_spans
|
|
60
|
+
)
|
|
53
61
|
}
|
|
54
62
|
end
|
|
55
63
|
|
|
@@ -113,6 +113,7 @@ module ActionAgent
|
|
|
113
113
|
|
|
114
114
|
def call
|
|
115
115
|
@agent_record.ensure_executable!
|
|
116
|
+
mcp_dispatcher.ensure_extra_servers_live!
|
|
116
117
|
root_span = @root_span = build_root_span
|
|
117
118
|
record_prompt_span(root_span)
|
|
118
119
|
llm_span = root_span.add_span(
|
|
@@ -415,9 +416,10 @@ module ActionAgent
|
|
|
415
416
|
# its reply, so agents can delegate to each other as a tool call. The
|
|
416
417
|
# sub-run is a real AgentRun with its own trace.
|
|
417
418
|
# One dispatcher per run, so every tool call shares the MCP sessions the
|
|
418
|
-
# first call opens.
|
|
419
|
+
# first call opens. A run given a checkout sandbox (an evaluation or a
|
|
420
|
+
# runner run against it) reaches that runtime too.
|
|
419
421
|
def mcp_dispatcher
|
|
420
|
-
@mcp_dispatcher ||= MCPToolDispatcher.new(@agent_record)
|
|
422
|
+
@mcp_dispatcher ||= MCPToolDispatcher.new(@agent_record, extra_server_keys: [ @run.try(:sandbox_server_key) ].compact)
|
|
421
423
|
end
|
|
422
424
|
|
|
423
425
|
# Splits the offered schemas the way `tool_schemas` assembles them, so the
|
|
@@ -25,7 +25,9 @@ module ActionAgent
|
|
|
25
25
|
# offers every agent. These are the roster: +agent.tools+ is what
|
|
26
26
|
# AgentToolbox turns into function schemas at generation time.
|
|
27
27
|
# * **MCP** — never stored on the roster. Computed from the services the
|
|
28
|
-
# agent enables, which is where they are edited.
|
|
28
|
+
# agent enables, which is where they are edited. The services are the
|
|
29
|
+
# catalog plus the live checkout sandbox runtimes the caller hands in
|
|
30
|
+
# (+runtimes:+), which an agent enables under their "sandbox:<id>" keys.
|
|
29
31
|
#
|
|
30
32
|
# Enablement reads the agent's own configuration: +tools+ for the dashboard
|
|
31
33
|
# capabilities and the schema tools, +mcp_servers+ for services and their
|
|
@@ -40,13 +42,16 @@ module ActionAgent
|
|
|
40
42
|
# workspace already talks to, then the rest of the catalog.
|
|
41
43
|
STATUS_RANK = { "active" => 0, "configured" => 1, "available" => 2, "idle" => 3 }.freeze
|
|
42
44
|
|
|
43
|
-
attr_reader :agent, :discovery
|
|
45
|
+
attr_reader :agent, :discovery, :runtimes
|
|
44
46
|
|
|
45
47
|
# @param agent [ActionAgent::Agent] the agent being edited
|
|
46
48
|
# @param traces [ActiveRecord::Relation] the traces the caller may read
|
|
47
49
|
# @param hours [Integer] the window the usage columns are scoped to
|
|
48
|
-
|
|
50
|
+
# @param runtimes [Array<Hash>] live checkout sandbox runtimes this agent
|
|
51
|
+
# can be given, as SandboxSession.runtime_server_listings returns them
|
|
52
|
+
def initialize(agent:, traces:, hours: ToolDiscovery::DEFAULT_WINDOW_HOURS, runtimes: [])
|
|
49
53
|
@agent = agent
|
|
54
|
+
@runtimes = Array(runtimes).index_by { |runtime| runtime[:key] }
|
|
50
55
|
@discovery = ToolDiscovery.new(
|
|
51
56
|
traces: agent.telemetry_traces(traces),
|
|
52
57
|
agents: Agent.where(id: agent.id),
|
|
@@ -89,10 +94,10 @@ module ActionAgent
|
|
|
89
94
|
end
|
|
90
95
|
end
|
|
91
96
|
|
|
92
|
-
# The catalog, plus anything this agent's traffic or
|
|
93
|
-
#
|
|
97
|
+
# The catalog and the live runtimes, plus anything this agent's traffic or
|
|
98
|
+
# configuration names that neither describes.
|
|
94
99
|
def service_keys
|
|
95
|
-
(MCPCatalog.keys + detected_by_server.keys + configured_servers.keys).uniq
|
|
100
|
+
(MCPCatalog.keys + runtimes.keys + detected_by_server.keys + configured_servers.keys).uniq
|
|
96
101
|
end
|
|
97
102
|
|
|
98
103
|
def detected_by_server
|
|
@@ -100,7 +105,8 @@ module ActionAgent
|
|
|
100
105
|
end
|
|
101
106
|
|
|
102
107
|
def service_row(key)
|
|
103
|
-
|
|
108
|
+
# A live runtime reads as a known service; its listing carries no token.
|
|
109
|
+
catalog = MCPCatalog.find(key) || runtimes[key]
|
|
104
110
|
used = detected_by_server[key] || []
|
|
105
111
|
calls = used.sum { |tool| tool[:calls] }
|
|
106
112
|
enabled = configured_servers.key?(key)
|
|
@@ -112,6 +118,7 @@ module ActionAgent
|
|
|
112
118
|
docs_url: catalog&.dig(:docs_url),
|
|
113
119
|
first_party: catalog ? catalog[:first_party] : false,
|
|
114
120
|
known: !catalog.nil?,
|
|
121
|
+
runtime: SandboxSession.runtime_server_key?(key),
|
|
115
122
|
transport: transport_label(catalog),
|
|
116
123
|
status: service_status(calls, enabled, catalog),
|
|
117
124
|
enabled: enabled,
|
|
@@ -125,12 +132,16 @@ module ActionAgent
|
|
|
125
132
|
# How to reach the server, in the one line the expanded panel shows:
|
|
126
133
|
# "Streamable HTTP · <url>" for the ones the dashboard can call,
|
|
127
134
|
# "sandbox · <command>" for the ones it can start, "stdio · <command>"
|
|
128
|
-
# for the rest.
|
|
135
|
+
# for the rest. Every transport MCPToolDispatcher calls reads the same,
|
|
136
|
+
# because MCPClient reaches all of them over Streamable HTTP — a checkout
|
|
137
|
+
# runtime ("streamable_http") included.
|
|
129
138
|
def transport_label(catalog)
|
|
130
139
|
return nil if catalog.nil?
|
|
131
140
|
|
|
132
141
|
transport = catalog[:transport].to_s
|
|
133
|
-
|
|
142
|
+
if transport.in?(MCPToolDispatcher::HTTP_TRANSPORTS)
|
|
143
|
+
return [ "Streamable HTTP", catalog[:url].presence ].compact.join(" · ")
|
|
144
|
+
end
|
|
134
145
|
|
|
135
146
|
prefix = catalog[:sandbox] ? "sandbox" : transport.presence
|
|
136
147
|
[ prefix, catalog[:command] ].compact.join(" · ").presence
|
|
@@ -0,0 +1,86 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module ActionAgent
|
|
4
|
+
# How Claude Code sessions authenticate (ActionAgent.claude_code_auth), and
|
|
5
|
+
# the one rule for whether an owner's Claude Code is "connected", shared by
|
|
6
|
+
# the sandbox listing, the assistant's configuration and the session API so
|
|
7
|
+
# they never disagree.
|
|
8
|
+
#
|
|
9
|
+
# :api_key the owner's Anthropic API key, stored as a ProviderKey. A
|
|
10
|
+
# Claude subscription token stored by an earlier version does
|
|
11
|
+
# not count (ProviderKey#needs_replacing?).
|
|
12
|
+
# :local_login the login of the machine the dashboard runs on, as
|
|
13
|
+
# `claude auth status` reports it (see
|
|
14
|
+
# LocalSandboxBackend.claude_login_status). The dashboard never
|
|
15
|
+
# sees that credential.
|
|
16
|
+
#
|
|
17
|
+
# Anthropic does not let third-party products store or route requests
|
|
18
|
+
# through Claude.ai subscription credentials, so there is no third mode:
|
|
19
|
+
# https://code.claude.com/docs/en/legal-and-compliance.md
|
|
20
|
+
module ClaudeCodeAuth
|
|
21
|
+
module_function
|
|
22
|
+
|
|
23
|
+
# @return [String] "api_key" or "local_login"
|
|
24
|
+
def mode
|
|
25
|
+
ActionAgent.claude_code_auth.to_s
|
|
26
|
+
end
|
|
27
|
+
|
|
28
|
+
def local_login?
|
|
29
|
+
mode == "local_login"
|
|
30
|
+
end
|
|
31
|
+
|
|
32
|
+
# Why +orchestrator+'s backend cannot run Claude Code sessions with the
|
|
33
|
+
# configured authentication, or nil. A machine's own login is the
|
|
34
|
+
# dashboard user's, so only a backend running sessions as that user, on
|
|
35
|
+
# that machine, may use it: a container or a remote host would either
|
|
36
|
+
# find no login or need it copied there, which is what this mode exists
|
|
37
|
+
# to never do.
|
|
38
|
+
def backend_refusal(orchestrator)
|
|
39
|
+
return unless local_login?
|
|
40
|
+
return if orchestrator.local?
|
|
41
|
+
|
|
42
|
+
"ActionAgent.claude_code_auth = :local_login uses this machine's own Claude Code login, so it works only " \
|
|
43
|
+
"with the :local sandbox backend, not #{orchestrator.backend_name}"
|
|
44
|
+
end
|
|
45
|
+
|
|
46
|
+
# Whether the owner of +provider_keys+ (a ProviderKey scope already
|
|
47
|
+
# narrowed to them) has an API key Claude Code can run on.
|
|
48
|
+
def api_key_connected?(provider_keys)
|
|
49
|
+
provider_keys.where(provider: "claude_code").any? { |key| key.runtime_environment.present? }
|
|
50
|
+
end
|
|
51
|
+
|
|
52
|
+
# What the dashboard reports about Claude Code for the owner of
|
|
53
|
+
# +provider_keys+: booleans and the login's method, never a credential.
|
|
54
|
+
#
|
|
55
|
+
# @return [Hash] { mode:, connected:, login: } (login for :local_login only)
|
|
56
|
+
def status(provider_keys)
|
|
57
|
+
if local_login?
|
|
58
|
+
# Only the :local backend uses the login, and no other one needs
|
|
59
|
+
# this machine's CLI asked about it.
|
|
60
|
+
login = local_backend? ? LocalSandboxBackend.claude_login_status : LocalSandboxBackend::LOGGED_OUT
|
|
61
|
+
{ mode: mode, connected: login[:logged_in], login: login.slice(:logged_in, :auth_method) }
|
|
62
|
+
else
|
|
63
|
+
{ mode: mode, connected: api_key_connected?(provider_keys) }
|
|
64
|
+
end
|
|
65
|
+
end
|
|
66
|
+
|
|
67
|
+
# A backend that cannot even be loaded is not the :local one.
|
|
68
|
+
def local_backend?
|
|
69
|
+
SandboxOrchestrator.new.local?
|
|
70
|
+
rescue StandardError, LoadError
|
|
71
|
+
false
|
|
72
|
+
end
|
|
73
|
+
|
|
74
|
+
# Why a Claude Code session cannot start in +sandbox+ for want of
|
|
75
|
+
# credentials, or nil.
|
|
76
|
+
def credential_refusal(sandbox)
|
|
77
|
+
if local_login?
|
|
78
|
+
return if LocalSandboxBackend.claude_login_status[:logged_in]
|
|
79
|
+
|
|
80
|
+
"Claude Code is not logged in on this machine: run `claude /login` as the user the dashboard runs as"
|
|
81
|
+
elsif sandbox.runtime_environment.blank?
|
|
82
|
+
"Claude Code is not connected: connect an Anthropic API key in Settings -> Integrations first"
|
|
83
|
+
end
|
|
84
|
+
end
|
|
85
|
+
end
|
|
86
|
+
end
|