actionagent 1.7.2 → 1.8.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/app/assets/builds/action_agent.css +1 -1
- data/app/assets/builds/action_agent.js +59 -54
- data/app/controllers/action_agent/api/agents_controller.rb +19 -3
- data/app/controllers/action_agent/api/code_sessions_controller.rb +156 -0
- data/app/controllers/action_agent/api/evaluations_controller.rb +27 -130
- data/app/controllers/action_agent/api/github_connections_controller.rb +147 -0
- data/app/controllers/action_agent/api/mcp_controller.rb +38 -5
- data/app/controllers/action_agent/api/mcp_servers_controller.rb +10 -1
- data/app/controllers/action_agent/api/provider_keys_controller.rb +54 -3
- data/app/controllers/action_agent/api/provider_models_controller.rb +10 -6
- data/app/controllers/action_agent/api/sandboxes_controller.rb +97 -6
- data/app/controllers/concerns/action_agent/api/evaluation_run_starting.rb +93 -0
- data/app/controllers/concerns/action_agent/api/mcp_dashboard_tools.rb +507 -0
- data/app/controllers/concerns/action_agent/api/run_sandbox.rb +65 -0
- data/app/jobs/action_agent/code_session_job.rb +166 -0
- data/app/jobs/action_agent/sandbox_cleanup_job.rb +80 -11
- data/app/jobs/action_agent/sandbox_provision_job.rb +122 -14
- data/app/jobs/action_agent/sandbox_run_job.rb +10 -3
- data/app/models/action_agent/agent.rb +16 -6
- data/app/models/action_agent/agent_run.rb +20 -1
- data/app/models/action_agent/code_session.rb +141 -0
- data/app/models/action_agent/evaluation_run.rb +23 -1
- data/app/models/action_agent/github_connection.rb +75 -0
- data/app/models/action_agent/provider_key.rb +142 -10
- data/app/models/action_agent/sandbox_session.rb +193 -17
- data/app/serializers/action_agent/evaluation_serializer.rb +118 -0
- data/app/serializers/action_agent/telemetry_trace_serializer.rb +10 -2
- data/app/services/action_agent/agent_execution_service.rb +4 -2
- data/app/services/action_agent/agent_tool_roster.rb +20 -9
- data/app/services/action_agent/claude_code_auth.rb +86 -0
- data/app/services/action_agent/dashboard_assistant_service.rb +47 -5
- data/app/services/action_agent/evaluation_tool_resolver.rb +18 -0
- data/app/services/action_agent/github_client.rb +111 -0
- data/app/services/action_agent/local_sandbox_backend.rb +1689 -0
- data/app/services/action_agent/local_sandbox_databases.rb +257 -0
- data/app/services/action_agent/mcp_client.rb +5 -1
- data/app/services/action_agent/mcp_tool_dispatcher.rb +137 -17
- data/app/services/action_agent/mock_sandbox_backend.rb +39 -0
- data/app/services/action_agent/ollama_host_probe.rb +75 -0
- data/app/services/action_agent/payload_bounds.rb +36 -0
- data/app/services/action_agent/sandbox_manifest.rb +67 -0
- data/app/services/action_agent/sandbox_orchestrator.rb +69 -14
- data/app/services/action_agent/scenario_evaluation_runner.rb +50 -4
- data/app/services/action_agent/secret_scrubber.rb +37 -0
- data/app/services/action_agent/tool_discovery.rb +19 -5
- data/config/routes.rb +22 -3
- data/lib/action_agent/engine.rb +1 -0
- data/lib/action_agent/version.rb +1 -1
- data/lib/action_agent.rb +147 -3
- data/lib/generators/action_agent/install_generator.rb +30 -3
- data/lib/generators/action_agent/templates/action_agent.rb.erb +44 -0
- data/lib/generators/action_agent/templates/add_provider_key_api_key.rb.erb +25 -0
- data/lib/generators/action_agent/templates/create_active_agent_code_sessions.rb.erb +59 -0
- data/lib/generators/action_agent/templates/create_active_agent_dashboard_tables.rb.erb +2 -0
- data/lib/generators/action_agent/templates/create_active_agent_github_connections.rb.erb +61 -0
- data/lib/tasks/claude_code.rake +16 -0
- data/lib/tasks/sandbox.rake +26 -0
- metadata +23 -1
|
@@ -0,0 +1,166 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module ActionAgent
|
|
4
|
+
# Runs one queued Claude Code session (CodeSession) through the sandbox
|
|
5
|
+
# backend, storing its stream-json transcript as it arrives, then the
|
|
6
|
+
# outcome Claude Code reported and the diff it left in the checkout.
|
|
7
|
+
#
|
|
8
|
+
# Not retried: a session edits the checkout and spends the owner's Claude
|
|
9
|
+
# Code usage, so running it again is not the same as running it once.
|
|
10
|
+
class CodeSessionJob < ApplicationJob
|
|
11
|
+
queue_as :sandboxes
|
|
12
|
+
|
|
13
|
+
# Longest error_message kept: a failed session's own report can be long.
|
|
14
|
+
MAX_ERROR_MESSAGE = 2_000
|
|
15
|
+
|
|
16
|
+
def perform(code_session_id)
|
|
17
|
+
code_session = CodeSession.find_by(id: code_session_id)
|
|
18
|
+
return unless code_session&.queued?
|
|
19
|
+
# Claimed atomically: a session cancelled after it was loaded, or a
|
|
20
|
+
# second job for the same one, finds it no longer queued.
|
|
21
|
+
return unless start(code_session)
|
|
22
|
+
|
|
23
|
+
# Computed once rather than per event: finding them reads the GitHub
|
|
24
|
+
# connection and the Claude Code key.
|
|
25
|
+
secrets = code_session.secrets
|
|
26
|
+
result_event = nil
|
|
27
|
+
stop_sent = false
|
|
28
|
+
orchestrator = SandboxOrchestrator.new
|
|
29
|
+
sandbox = code_session.sandbox_session
|
|
30
|
+
|
|
31
|
+
# Queued behind a Stop (the cleanup job shares this queue): never start
|
|
32
|
+
# Claude Code, with the owner's credential, in a stopped sandbox.
|
|
33
|
+
unless sandbox.reload.ready? && sandbox.active?
|
|
34
|
+
return fail!(code_session, "The sandbox was stopped before the session started")
|
|
35
|
+
end
|
|
36
|
+
|
|
37
|
+
outcome = orchestrator.run_code_session(sandbox, code_session) do |event|
|
|
38
|
+
# append_event! reloads the row under its lock, so this sees a cancel
|
|
39
|
+
# made since. The cancel's own stop can land after this job claimed
|
|
40
|
+
# the session but before the backend started Claude Code, and then
|
|
41
|
+
# has no process to stop; an event means the process exists now, so
|
|
42
|
+
# the stop is sent again, once.
|
|
43
|
+
code_session.append_event!(event, secrets: secrets)
|
|
44
|
+
if code_session.cancelled? && !stop_sent
|
|
45
|
+
stop_sent = true
|
|
46
|
+
stop(orchestrator, sandbox, code_session)
|
|
47
|
+
end
|
|
48
|
+
next unless event.is_a?(Hash) && event["type"] == "result"
|
|
49
|
+
|
|
50
|
+
result_event = event
|
|
51
|
+
code_session.record_result!(event)
|
|
52
|
+
end
|
|
53
|
+
|
|
54
|
+
finish(code_session, outcome.to_h, result_event, secrets)
|
|
55
|
+
rescue StandardError => e
|
|
56
|
+
message = SecretScrubber.scrub(e.message.to_s, secrets || safe_secrets(code_session))
|
|
57
|
+
Rails.logger.error("Claude Code session #{code_session_id} failed: #{message}")
|
|
58
|
+
fail!(code_session, message) if code_session
|
|
59
|
+
end
|
|
60
|
+
|
|
61
|
+
private
|
|
62
|
+
|
|
63
|
+
def start(code_session)
|
|
64
|
+
now = Time.current
|
|
65
|
+
claimed = CodeSession.where(id: code_session.id, status: CodeSession.statuses[:queued])
|
|
66
|
+
.update_all(status: CodeSession.statuses[:running], started_at: now, updated_at: now)
|
|
67
|
+
return false if claimed.zero?
|
|
68
|
+
|
|
69
|
+
code_session.reload
|
|
70
|
+
end
|
|
71
|
+
|
|
72
|
+
# Under the row lock, which reloads the session: a cancel that landed
|
|
73
|
+
# while it ran must not be overwritten by the outcome.
|
|
74
|
+
def finish(code_session, outcome, result_event, secrets)
|
|
75
|
+
code_session.with_lock do
|
|
76
|
+
code_session.diff = outcome[:diff]
|
|
77
|
+
|
|
78
|
+
if code_session.cancelled?
|
|
79
|
+
# Settled now, with its diff (see CodeSession#diff_pending?).
|
|
80
|
+
code_session.finished_at ||= Time.current
|
|
81
|
+
elsif succeeded?(result_event, outcome)
|
|
82
|
+
code_session.assign_attributes(status: :succeeded, finished_at: Time.current)
|
|
83
|
+
else
|
|
84
|
+
code_session.assign_attributes(
|
|
85
|
+
status: :failed,
|
|
86
|
+
error_message: failure_message(result_event, outcome, secrets),
|
|
87
|
+
finished_at: Time.current
|
|
88
|
+
)
|
|
89
|
+
end
|
|
90
|
+
code_session.save!
|
|
91
|
+
end
|
|
92
|
+
end
|
|
93
|
+
|
|
94
|
+
# Claude Code reported success and the process agreed.
|
|
95
|
+
def succeeded?(result_event, outcome)
|
|
96
|
+
result_event.present? && !reported_error?(result_event) && outcome[:exit_status] == 0
|
|
97
|
+
end
|
|
98
|
+
|
|
99
|
+
# A result event reports a failure through is_error, or through an error
|
|
100
|
+
# subtype (error_max_turns, error_during_execution): a session that ran
|
|
101
|
+
# out of turns did not finish the task, whatever its is_error says.
|
|
102
|
+
def reported_error?(result_event)
|
|
103
|
+
result_event["is_error"] != false || (result_event["subtype"] || "success") != "success"
|
|
104
|
+
end
|
|
105
|
+
|
|
106
|
+
# What went wrong, in Claude Code's own words when it reported a failure
|
|
107
|
+
# and from the process otherwise.
|
|
108
|
+
def failure_message(result_event, outcome, secrets)
|
|
109
|
+
message = reported_failure(result_event) || process_failure(result_event, outcome)
|
|
110
|
+
SecretScrubber.scrub(message, secrets).truncate(MAX_ERROR_MESSAGE)
|
|
111
|
+
end
|
|
112
|
+
|
|
113
|
+
def reported_failure(result_event)
|
|
114
|
+
return nil unless result_event && reported_error?(result_event)
|
|
115
|
+
|
|
116
|
+
errors = Array(result_event["errors"]).map { |error| error.is_a?(Hash) ? (error["message"] || error.to_json) : error.to_s }
|
|
117
|
+
result_event["result"].to_s.presence ||
|
|
118
|
+
errors.join("; ").presence ||
|
|
119
|
+
"Claude Code stopped: #{result_event['subtype']}"
|
|
120
|
+
end
|
|
121
|
+
|
|
122
|
+
def process_failure(result_event, outcome)
|
|
123
|
+
status = outcome[:exit_status]
|
|
124
|
+
headline =
|
|
125
|
+
if !status.nil? && status != 0 then "Claude Code exited with status #{status}"
|
|
126
|
+
elsif result_event.nil? then "Claude Code ended without reporting a result"
|
|
127
|
+
else "Claude Code did not finish"
|
|
128
|
+
end
|
|
129
|
+
|
|
130
|
+
[ headline, outcome[:stderr_tail].to_s.strip.presence ].compact.join(": ")
|
|
131
|
+
end
|
|
132
|
+
|
|
133
|
+
def stop(orchestrator, sandbox, code_session)
|
|
134
|
+
orchestrator.cancel_code_session(sandbox, code_session)
|
|
135
|
+
rescue StandardError => e
|
|
136
|
+
Rails.logger.warn("Failed to stop cancelled Claude Code session #{code_session.id}: #{e.message}")
|
|
137
|
+
end
|
|
138
|
+
|
|
139
|
+
# Settles a session that ended without an outcome from the backend: it
|
|
140
|
+
# raised, refused to start Claude Code, or was never asked to. A session
|
|
141
|
+
# cancelled meanwhile stays cancelled, but is settled too: no diff is
|
|
142
|
+
# coming for it (see CodeSession#diff_pending?).
|
|
143
|
+
def fail!(code_session, message)
|
|
144
|
+
# A save that raised leaves unsaved changes behind, and locking a dirty
|
|
145
|
+
# record raises: start from the row as stored.
|
|
146
|
+
code_session.reload
|
|
147
|
+
code_session.with_lock do
|
|
148
|
+
if code_session.cancelled?
|
|
149
|
+
code_session.update!(finished_at: Time.current) if code_session.finished_at.nil?
|
|
150
|
+
next
|
|
151
|
+
end
|
|
152
|
+
next if code_session.finished?
|
|
153
|
+
|
|
154
|
+
code_session.update!(status: :failed, error_message: message.truncate(MAX_ERROR_MESSAGE), finished_at: Time.current)
|
|
155
|
+
end
|
|
156
|
+
rescue ActiveRecord::RecordNotFound
|
|
157
|
+
nil
|
|
158
|
+
end
|
|
159
|
+
|
|
160
|
+
def safe_secrets(code_session)
|
|
161
|
+
code_session&.secrets || []
|
|
162
|
+
rescue StandardError
|
|
163
|
+
[]
|
|
164
|
+
end
|
|
165
|
+
end
|
|
166
|
+
end
|
|
@@ -11,35 +11,104 @@ module ActionAgent
|
|
|
11
11
|
|
|
12
12
|
Rails.logger.info("Cleaning up sandbox: #{sandbox.session_id}")
|
|
13
13
|
|
|
14
|
-
#
|
|
15
|
-
|
|
16
|
-
|
|
17
|
-
|
|
14
|
+
# The handle is kept until the backend confirms it let go: cleared
|
|
15
|
+
# after a failed terminate, nothing would ever know the process (or
|
|
16
|
+
# container) was still there. cleanup_expired! retries these.
|
|
17
|
+
handle = sandbox.cloud_run_job_id.presence || derived_handle(sandbox)
|
|
18
|
+
return if handle.present? && !released?(sandbox, handle)
|
|
18
19
|
|
|
19
20
|
# Optionally delete old sandbox records
|
|
20
21
|
# For now, keep for analytics
|
|
21
22
|
sandbox.update!(cloud_run_url: nil, cloud_run_job_id: nil)
|
|
22
23
|
end
|
|
23
24
|
|
|
24
|
-
# Periodic cleanup of all expired sandboxes
|
|
25
|
+
# Periodic cleanup of all expired sandboxes (rake
|
|
26
|
+
# action_agent:sandbox:reap). Expires the sessions past their expiry that
|
|
27
|
+
# are still pending, provisioning, ready or running — each one's
|
|
28
|
+
# resource is released by a job of its own — and returns how many.
|
|
29
|
+
#
|
|
30
|
+
# @return [Integer]
|
|
25
31
|
def self.cleanup_expired!
|
|
32
|
+
# Expired earlier, but their backend failed to terminate them, so they
|
|
33
|
+
# still hold a handle: try again. Collected first, so the sessions
|
|
34
|
+
# expired below are not enqueued twice.
|
|
35
|
+
unreleased = SandboxSession.expired.where.not(cloud_run_job_id: [ nil, "" ]).pluck(:id)
|
|
36
|
+
unreleased += unrecorded_checkouts
|
|
37
|
+
|
|
38
|
+
count = 0
|
|
26
39
|
SandboxSession.expired_sessions.active.find_each do |sandbox|
|
|
27
40
|
sandbox.expire!
|
|
41
|
+
count += 1
|
|
28
42
|
end
|
|
43
|
+
|
|
44
|
+
unreleased.each { |id| perform_later(id) }
|
|
45
|
+
count
|
|
46
|
+
end
|
|
47
|
+
|
|
48
|
+
# Checkouts whose boot never recorded a handle can still hold processes
|
|
49
|
+
# (see #derived_handle), and a failed terminate of one left nothing
|
|
50
|
+
# behind that says so. Retried only where the backend derives handles,
|
|
51
|
+
# and only for a day after the row last changed: a released one is
|
|
52
|
+
# indistinguishable from an unreleased one, and a terminate of a
|
|
53
|
+
# sandbox that is gone is a cheap no-op, so the window is what bounds
|
|
54
|
+
# the retries.
|
|
55
|
+
RETRY_UNRECORDED_FOR = 1.day
|
|
56
|
+
RETRY_UNRECORDED_LIMIT = 100
|
|
57
|
+
|
|
58
|
+
def self.unrecorded_checkouts
|
|
59
|
+
return [] unless SandboxOrchestrator.new.derives_handles?
|
|
60
|
+
|
|
61
|
+
SandboxSession.expired.by_type("app_runtime").where(cloud_run_job_id: [ nil, "" ])
|
|
62
|
+
.where(updated_at: RETRY_UNRECORDED_FOR.ago..)
|
|
63
|
+
.order(updated_at: :desc).limit(RETRY_UNRECORDED_LIMIT).pluck(:id)
|
|
64
|
+
rescue StandardError, LoadError => e
|
|
65
|
+
Rails.logger.warn("[ActionAgent] sandbox backend unavailable to the reaper: #{e.message}")
|
|
66
|
+
[]
|
|
29
67
|
end
|
|
68
|
+
private_class_method :unrecorded_checkouts
|
|
30
69
|
|
|
31
70
|
private
|
|
32
71
|
|
|
72
|
+
# A checkout whose provisioning never recorded a handle (the job died
|
|
73
|
+
# while the backend was booting it) can still hold processes; a backend
|
|
74
|
+
# that can name a session's sandbox without being told answers here.
|
|
75
|
+
def derived_handle(sandbox)
|
|
76
|
+
return nil unless sandbox.app_runtime?
|
|
77
|
+
|
|
78
|
+
SandboxOrchestrator.new.handle_for(sandbox)
|
|
79
|
+
rescue StandardError => e
|
|
80
|
+
Rails.logger.warn("[ActionAgent] could not derive a sandbox handle: #{e.message}")
|
|
81
|
+
nil
|
|
82
|
+
end
|
|
83
|
+
|
|
84
|
+
# Whether the backend behind +handle+ let go of it.
|
|
85
|
+
#
|
|
86
|
+
# In development a sandbox other than a checkout was only simulated (see
|
|
87
|
+
# SandboxProvisionJob#simulate_provisioning): no backend holds its
|
|
88
|
+
# made-up handle, so there is nothing to terminate. A checkout is real
|
|
89
|
+
# everywhere — the :local backend runs it as child processes of this
|
|
90
|
+
# app, which skipping terminate in development used to orphan.
|
|
91
|
+
def released?(sandbox, handle)
|
|
92
|
+
return true if Rails.env.development? && !sandbox.app_runtime?
|
|
93
|
+
|
|
94
|
+
terminate_backend_sandbox(handle)
|
|
95
|
+
end
|
|
96
|
+
|
|
33
97
|
# Through the orchestrator, like provisioning: whichever backend the
|
|
34
|
-
# host registered (Incus, Kubernetes, Cloud Run, or the
|
|
35
|
-
# reclaims its own resource. This used to require
|
|
36
|
-
# directly, which the engine does not depend on — a
|
|
37
|
-
# ScriptError, not a StandardError, so it escaped the
|
|
38
|
-
# failed on every host but the one that happened to
|
|
98
|
+
# host registered (Incus, Kubernetes, Cloud Run, the :local one or the
|
|
99
|
+
# built-in mock) reclaims its own resource. This used to require
|
|
100
|
+
# google/cloud/run/v2 directly, which the engine does not depend on — a
|
|
101
|
+
# LoadError is a ScriptError, not a StandardError, so it escaped the
|
|
102
|
+
# rescue and the job failed on every host but the one that happened to
|
|
103
|
+
# bundle the SDK.
|
|
104
|
+
#
|
|
105
|
+
# Backends disagree on what terminate returns; only an explicit false
|
|
106
|
+
# (or an error) counts as not released.
|
|
39
107
|
def terminate_backend_sandbox(sandbox_id)
|
|
40
|
-
SandboxOrchestrator.new.terminate(sandbox_id)
|
|
108
|
+
SandboxOrchestrator.new.terminate(sandbox_id) != false
|
|
41
109
|
rescue StandardError => e
|
|
42
110
|
Rails.logger.warn("Failed to terminate sandbox #{sandbox_id}: #{e.message}")
|
|
111
|
+
false
|
|
43
112
|
end
|
|
44
113
|
end
|
|
45
114
|
end
|
|
@@ -2,36 +2,65 @@
|
|
|
2
2
|
|
|
3
3
|
module ActionAgent
|
|
4
4
|
class SandboxProvisionJob < ApplicationJob
|
|
5
|
+
MAX_ERROR_MESSAGE = 8_000
|
|
5
6
|
queue_as :sandboxes
|
|
6
7
|
|
|
7
8
|
# Provision a Cloud Run sandbox for the session
|
|
8
9
|
# Each sandbox is an instance of the ActiveAgents application running in sandbox mode
|
|
9
10
|
def perform(sandbox_session_id)
|
|
10
|
-
|
|
11
|
-
|
|
11
|
+
# Deleted before the job ran: nothing to provision. `find` raised here,
|
|
12
|
+
# and the rescue below then called update! on nil.
|
|
13
|
+
sandbox = SandboxSession.find_by(id: sandbox_session_id)
|
|
14
|
+
return if sandbox.nil?
|
|
12
15
|
|
|
13
|
-
#
|
|
14
|
-
|
|
16
|
+
# SandboxSession#provision! moves a session to provisioning and hands it
|
|
17
|
+
# over exactly once. Any other status means it was provisioned, stopped
|
|
18
|
+
# or has failed since, and provisioning it again would boot a second
|
|
19
|
+
# sandbox for one session (and orphan the first).
|
|
20
|
+
return unless sandbox.provisioning?
|
|
21
|
+
|
|
22
|
+
# In development/test, simulate provisioning. A checkout sandbox always
|
|
23
|
+
# goes to a backend: the simulation has no checkout to boot.
|
|
24
|
+
if (Rails.env.development? || Rails.env.test?) && !sandbox.app_runtime?
|
|
15
25
|
simulate_provisioning(sandbox)
|
|
16
26
|
return
|
|
17
27
|
end
|
|
18
28
|
|
|
19
29
|
# Hand off to whichever backend this install registered — the engine
|
|
20
|
-
# ships
|
|
21
|
-
# host app's backend (see ActionAgent.sandbox_backends).
|
|
22
|
-
|
|
30
|
+
# ships the in-memory one and :local, so a real container/job comes
|
|
31
|
+
# from the host app's backend (see ActionAgent.sandbox_backends).
|
|
32
|
+
ensure_checkout_available!(sandbox) if sandbox.app_runtime?
|
|
23
33
|
|
|
24
|
-
|
|
25
|
-
|
|
26
|
-
|
|
27
|
-
|
|
34
|
+
orchestrator = SandboxOrchestrator.new
|
|
35
|
+
result = orchestrator.create_sandbox(sandbox)
|
|
36
|
+
# From here on the backend runs a sandbox for this session: whatever
|
|
37
|
+
# goes wrong below, the rescue releases it unless the session recorded
|
|
38
|
+
# its handle.
|
|
39
|
+
handle = result[:sandbox_id]
|
|
40
|
+
|
|
41
|
+
# Stopped (or expired) while the backend was booting it: release what
|
|
42
|
+
# was just started rather than reviving the session as ready.
|
|
43
|
+
recorded = mark_ready_unless_stopped(sandbox, result)
|
|
44
|
+
unless recorded
|
|
45
|
+
release(orchestrator, handle, sandbox.id)
|
|
46
|
+
return
|
|
47
|
+
end
|
|
28
48
|
|
|
29
49
|
# Broadcast status update
|
|
30
50
|
broadcast_sandbox_update(sandbox)
|
|
31
51
|
rescue StandardError => e
|
|
32
|
-
|
|
33
|
-
|
|
34
|
-
|
|
52
|
+
# A backend's error can carry the checkout token (a clone URL, a git
|
|
53
|
+
# error echoing its header) or the Claude Code credential; the message
|
|
54
|
+
# is stored and served to the dashboard, so it is scrubbed first.
|
|
55
|
+
# Kept whole up to a bound (SandboxSession#error_summary shortens it
|
|
56
|
+
# for display, keeping the log tail that names the actual error).
|
|
57
|
+
message = SecretScrubber.scrub(e.message.to_s, secrets_for(sandbox)).truncate(MAX_ERROR_MESSAGE)
|
|
58
|
+
Rails.logger.error("Sandbox provision failed: #{message}")
|
|
59
|
+
# Booted, but the session never recorded the handle (marking it ready
|
|
60
|
+
# raised): nothing else knows the sandbox exists, so nothing would
|
|
61
|
+
# ever terminate it.
|
|
62
|
+
release(orchestrator, handle, sandbox.id) if handle && !recorded
|
|
63
|
+
fail_unless_stopped(sandbox, message) if sandbox
|
|
35
64
|
end
|
|
36
65
|
|
|
37
66
|
private
|
|
@@ -46,6 +75,85 @@ module ActionAgent
|
|
|
46
75
|
)
|
|
47
76
|
end
|
|
48
77
|
|
|
78
|
+
# The owner can disconnect GitHub, or drop the repository from their
|
|
79
|
+
# selection, between creating the session and this job running.
|
|
80
|
+
# checkout_spec answers nil for the first and raises ArgumentError for
|
|
81
|
+
# the second; both mean the same thing to the owner.
|
|
82
|
+
def ensure_checkout_available!(sandbox)
|
|
83
|
+
available = begin
|
|
84
|
+
sandbox.checkout_spec.present?
|
|
85
|
+
rescue ArgumentError
|
|
86
|
+
false
|
|
87
|
+
end
|
|
88
|
+
return if available
|
|
89
|
+
|
|
90
|
+
raise "#{sandbox.repository} is no longer available: reconnect GitHub or reselect it in Settings -> Integrations"
|
|
91
|
+
end
|
|
92
|
+
|
|
93
|
+
# Marks the session ready with the backend's endpoint, under a row lock
|
|
94
|
+
# so a concurrent DELETE either lands first (and this reports false) or
|
|
95
|
+
# finds the session ready with a handle to terminate: SandboxSession#expire!
|
|
96
|
+
# re-reads the row under the same lock rather than trusting the copy it
|
|
97
|
+
# loaded before the boot finished.
|
|
98
|
+
def mark_ready_unless_stopped(sandbox, result)
|
|
99
|
+
sandbox.with_lock do
|
|
100
|
+
next false unless sandbox.provisioning?
|
|
101
|
+
|
|
102
|
+
sandbox.mark_ready!(
|
|
103
|
+
cloud_run_url: result[:url],
|
|
104
|
+
cloud_run_job_id: result[:sandbox_id],
|
|
105
|
+
runtime_mcp_url: result[:mcp_url],
|
|
106
|
+
runtime_mcp_token: result[:mcp_token]
|
|
107
|
+
)
|
|
108
|
+
true
|
|
109
|
+
end
|
|
110
|
+
rescue ActiveRecord::RecordNotFound
|
|
111
|
+
false
|
|
112
|
+
end
|
|
113
|
+
|
|
114
|
+
def release(orchestrator, handle, sandbox_id)
|
|
115
|
+
return if handle.blank?
|
|
116
|
+
|
|
117
|
+
released = orchestrator.terminate(handle)
|
|
118
|
+
keep_handle(sandbox_id, handle) if released == false
|
|
119
|
+
rescue StandardError => e
|
|
120
|
+
Rails.logger.warn("Failed to release sandbox #{handle}: #{e.message}")
|
|
121
|
+
keep_handle(sandbox_id, handle)
|
|
122
|
+
end
|
|
123
|
+
|
|
124
|
+
# The backend could not release it now: record the handle on the
|
|
125
|
+
# (expired) session so the reaper retries, as SandboxCleanupJob does.
|
|
126
|
+
def keep_handle(sandbox_id, handle)
|
|
127
|
+
SandboxSession.where(id: sandbox_id).update_all(cloud_run_job_id: handle)
|
|
128
|
+
end
|
|
129
|
+
|
|
130
|
+
# A session stopped while provisioning stays expired: the failure is
|
|
131
|
+
# moot, and marking it failed would bring it back into the owner's list.
|
|
132
|
+
def fail_unless_stopped(sandbox, message)
|
|
133
|
+
sandbox.reload
|
|
134
|
+
return unless sandbox.provisioning?
|
|
135
|
+
|
|
136
|
+
sandbox.update!(status: :failed, error_message: message)
|
|
137
|
+
broadcast_sandbox_update(sandbox)
|
|
138
|
+
rescue ActiveRecord::RecordNotFound
|
|
139
|
+
nil
|
|
140
|
+
end
|
|
141
|
+
|
|
142
|
+
# What must never reach error_message: the checkout token and the Claude
|
|
143
|
+
# Code credential this session boots with.
|
|
144
|
+
def secrets_for(sandbox)
|
|
145
|
+
return [] if sandbox.nil?
|
|
146
|
+
|
|
147
|
+
spec = begin
|
|
148
|
+
sandbox.checkout_spec
|
|
149
|
+
rescue StandardError
|
|
150
|
+
nil
|
|
151
|
+
end
|
|
152
|
+
[ spec&.dig(:token), *sandbox.runtime_environment.values ].compact
|
|
153
|
+
rescue StandardError
|
|
154
|
+
[]
|
|
155
|
+
end
|
|
156
|
+
|
|
49
157
|
def broadcast_sandbox_update(sandbox)
|
|
50
158
|
ActionCable.server.broadcast(
|
|
51
159
|
"sandbox_#{sandbox.session_id}",
|
|
@@ -31,13 +31,20 @@ module ActionAgent
|
|
|
31
31
|
end
|
|
32
32
|
rescue => e
|
|
33
33
|
Rails.logger.error("Sandbox run failed: #{e.message}")
|
|
34
|
-
sandbox
|
|
34
|
+
back_to_ready(sandbox, error_message: e.message)
|
|
35
35
|
broadcast_run_error(sandbox, run_id, provider, e.message)
|
|
36
36
|
end
|
|
37
37
|
end
|
|
38
38
|
|
|
39
39
|
private
|
|
40
40
|
|
|
41
|
+
# Only from running: a sandbox stopped while the run was in flight stays
|
|
42
|
+
# expired rather than coming back.
|
|
43
|
+
def back_to_ready(sandbox, **attributes)
|
|
44
|
+
SandboxSession.where(id: sandbox.id, status: SandboxSession.statuses[:running])
|
|
45
|
+
.update_all(attributes.merge(status: SandboxSession.statuses[:ready], updated_at: Time.current))
|
|
46
|
+
end
|
|
47
|
+
|
|
41
48
|
def execute_with_active_agent(sandbox, run_id, task, provider, started_at)
|
|
42
49
|
# Use generate_now here since we're already in a background job
|
|
43
50
|
# The ActiveAgent will handle the API call and callbacks
|
|
@@ -67,7 +74,7 @@ module ActionAgent
|
|
|
67
74
|
provider: provider
|
|
68
75
|
)
|
|
69
76
|
|
|
70
|
-
sandbox
|
|
77
|
+
back_to_ready(sandbox)
|
|
71
78
|
broadcast_run_complete(sandbox, run_id, run)
|
|
72
79
|
|
|
73
80
|
rescue => e
|
|
@@ -100,7 +107,7 @@ module ActionAgent
|
|
|
100
107
|
provider: provider
|
|
101
108
|
)
|
|
102
109
|
|
|
103
|
-
sandbox
|
|
110
|
+
back_to_ready(sandbox)
|
|
104
111
|
broadcast_run_complete(sandbox, run_id, run)
|
|
105
112
|
end
|
|
106
113
|
|
|
@@ -367,11 +367,16 @@ module ActionAgent
|
|
|
367
367
|
# before the job is enqueued, so a worker on another machine finds them
|
|
368
368
|
# attached. +params+ (provider/model overrides, the context_id of a
|
|
369
369
|
# conversation to continue) are kept on the run as input_params.
|
|
370
|
-
|
|
370
|
+
#
|
|
371
|
+
# +runtime_sandbox+ is a "sandbox:<session_id>" key whose app runtime this
|
|
372
|
+
# run reaches as if the agent had it in mcp_servers, without saving it on
|
|
373
|
+
# the agent. The caller checks the sandbox is theirs and live; the
|
|
374
|
+
# dispatcher still resolves it among this agent's owner's sessions only.
|
|
375
|
+
def execute(input_prompt, action: nil, attachments: [], actor: nil, runtime_sandbox: nil, **params)
|
|
371
376
|
ensure_executable!
|
|
372
377
|
run = create_run(
|
|
373
378
|
input_prompt, action: action, attachments: attachments, params: params,
|
|
374
|
-
actor: actor, status: :pending
|
|
379
|
+
actor: actor, runtime_sandbox: runtime_sandbox, status: :pending
|
|
375
380
|
)
|
|
376
381
|
|
|
377
382
|
# Queue the execution job
|
|
@@ -381,11 +386,11 @@ module ActionAgent
|
|
|
381
386
|
end
|
|
382
387
|
|
|
383
388
|
# Quick test execution (synchronous)
|
|
384
|
-
def test_execute(input_prompt, action: nil, attachments: [], actor: nil, **params)
|
|
389
|
+
def test_execute(input_prompt, action: nil, attachments: [], actor: nil, runtime_sandbox: nil, **params)
|
|
385
390
|
ensure_executable!
|
|
386
391
|
run = create_run(
|
|
387
392
|
input_prompt, action: action, attachments: attachments, params: params,
|
|
388
|
-
actor: actor, status: :running, started_at: Time.current
|
|
393
|
+
actor: actor, runtime_sandbox: runtime_sandbox, status: :running, started_at: Time.current
|
|
389
394
|
)
|
|
390
395
|
run.actor = actor
|
|
391
396
|
|
|
@@ -445,16 +450,21 @@ module ActionAgent
|
|
|
445
450
|
|
|
446
451
|
# Refuses files before creating anything: a run that exists but lost
|
|
447
452
|
# its attachments would execute against the wrong prompt.
|
|
448
|
-
def create_run(input_prompt, action:, attachments:, params:, actor: nil, **attributes)
|
|
453
|
+
def create_run(input_prompt, action:, attachments:, params:, actor: nil, runtime_sandbox: nil, **attributes)
|
|
449
454
|
files = Array.wrap(attachments).compact
|
|
450
455
|
raise AgentRun::AttachmentsUnavailable if files.any? && !AgentRun.attachments_available?
|
|
451
456
|
|
|
457
|
+
input_params = AgentRun.params_with_actor(params, actor)
|
|
458
|
+
if SandboxSession.runtime_server_key?(runtime_sandbox)
|
|
459
|
+
input_params = input_params.merge(AgentRun::SANDBOX_PARAM => runtime_sandbox.to_s)
|
|
460
|
+
end
|
|
461
|
+
|
|
452
462
|
run = agent_runs.create!(
|
|
453
463
|
input_prompt: input_prompt,
|
|
454
464
|
action_name: normalized_action(action),
|
|
455
465
|
# The caller is recorded beside the run's own parameters rather than
|
|
456
466
|
# among them: a client may send provider overrides, never an actor.
|
|
457
|
-
input_params:
|
|
467
|
+
input_params: input_params,
|
|
458
468
|
trace_id: SecureRandom.uuid,
|
|
459
469
|
**attributes
|
|
460
470
|
)
|
|
@@ -28,6 +28,11 @@ module ActionAgent
|
|
|
28
28
|
# Underscored so it cannot collide with a provider override, and
|
|
29
29
|
# stripped from anything a client sends (see Api::AgentsController).
|
|
30
30
|
ACTOR_PARAM = "_actor_gid"
|
|
31
|
+
# The key a checkout sandbox's app runtime this one run also reaches is
|
|
32
|
+
# recorded under ("sandbox:<session_id>"; never its token). Set only by
|
|
33
|
+
# the server, after it checked the caller owns that sandbox, and
|
|
34
|
+
# stripped from anything a client sends, as the actor is.
|
|
35
|
+
SANDBOX_PARAM = "_sandbox_server"
|
|
31
36
|
|
|
32
37
|
# +input_params+ with the caller recorded alongside them.
|
|
33
38
|
#
|
|
@@ -42,7 +47,7 @@ module ActionAgent
|
|
|
42
47
|
# @param actor [Object, nil] the caller
|
|
43
48
|
# @return [Hash]
|
|
44
49
|
def self.params_with_actor(params, actor)
|
|
45
|
-
params = (params || {}).to_h.except(ACTOR_PARAM, ACTOR_PARAM.to_sym)
|
|
50
|
+
params = (params || {}).to_h.except(ACTOR_PARAM, ACTOR_PARAM.to_sym, SANDBOX_PARAM, SANDBOX_PARAM.to_sym)
|
|
46
51
|
gid = actor.respond_to?(:to_global_id) ? actor.to_global_id.to_s : nil
|
|
47
52
|
gid ? params.merge(ACTOR_PARAM => gid) : params
|
|
48
53
|
rescue StandardError => e
|
|
@@ -65,6 +70,19 @@ module ActionAgent
|
|
|
65
70
|
|
|
66
71
|
attr_writer :actor
|
|
67
72
|
|
|
73
|
+
# The "sandbox:<session_id>" runtime this run reaches beside the agent's
|
|
74
|
+
# own MCP servers, or nil.
|
|
75
|
+
# @return [String, nil]
|
|
76
|
+
def sandbox_server_key
|
|
77
|
+
key = input_params[SANDBOX_PARAM] if input_params.is_a?(Hash)
|
|
78
|
+
key.to_s.presence if SandboxSession.runtime_server_key?(key)
|
|
79
|
+
end
|
|
80
|
+
|
|
81
|
+
# The session id of that sandbox, for a summary.
|
|
82
|
+
def sandbox_id
|
|
83
|
+
sandbox_server_key&.delete_prefix(SandboxSession::RUNTIME_SERVER_PREFIX)
|
|
84
|
+
end
|
|
85
|
+
|
|
68
86
|
# Whether this run knows who it is for. A run with a recorded actor that
|
|
69
87
|
# no longer resolves is *not* unattributed — it is broken, and callers
|
|
70
88
|
# that care can tell the two apart.
|
|
@@ -247,6 +265,7 @@ module ActionAgent
|
|
|
247
265
|
instructions_preview: output_metadata&.dig("instructions")&.truncate(120),
|
|
248
266
|
attachments: attachment_manifest,
|
|
249
267
|
context_id: context_id,
|
|
268
|
+
sandbox_id: sandbox_id,
|
|
250
269
|
created_at: created_at,
|
|
251
270
|
error: error_message
|
|
252
271
|
}
|