actionagent 1.7.2 → 1.8.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (59) hide show
  1. checksums.yaml +4 -4
  2. data/app/assets/builds/action_agent.css +1 -1
  3. data/app/assets/builds/action_agent.js +59 -54
  4. data/app/controllers/action_agent/api/agents_controller.rb +19 -3
  5. data/app/controllers/action_agent/api/code_sessions_controller.rb +156 -0
  6. data/app/controllers/action_agent/api/evaluations_controller.rb +27 -130
  7. data/app/controllers/action_agent/api/github_connections_controller.rb +147 -0
  8. data/app/controllers/action_agent/api/mcp_controller.rb +38 -5
  9. data/app/controllers/action_agent/api/mcp_servers_controller.rb +10 -1
  10. data/app/controllers/action_agent/api/provider_keys_controller.rb +54 -3
  11. data/app/controllers/action_agent/api/provider_models_controller.rb +10 -6
  12. data/app/controllers/action_agent/api/sandboxes_controller.rb +97 -6
  13. data/app/controllers/concerns/action_agent/api/evaluation_run_starting.rb +93 -0
  14. data/app/controllers/concerns/action_agent/api/mcp_dashboard_tools.rb +507 -0
  15. data/app/controllers/concerns/action_agent/api/run_sandbox.rb +65 -0
  16. data/app/jobs/action_agent/code_session_job.rb +166 -0
  17. data/app/jobs/action_agent/sandbox_cleanup_job.rb +80 -11
  18. data/app/jobs/action_agent/sandbox_provision_job.rb +122 -14
  19. data/app/jobs/action_agent/sandbox_run_job.rb +10 -3
  20. data/app/models/action_agent/agent.rb +16 -6
  21. data/app/models/action_agent/agent_run.rb +20 -1
  22. data/app/models/action_agent/code_session.rb +141 -0
  23. data/app/models/action_agent/evaluation_run.rb +23 -1
  24. data/app/models/action_agent/github_connection.rb +75 -0
  25. data/app/models/action_agent/provider_key.rb +142 -10
  26. data/app/models/action_agent/sandbox_session.rb +193 -17
  27. data/app/serializers/action_agent/evaluation_serializer.rb +118 -0
  28. data/app/serializers/action_agent/telemetry_trace_serializer.rb +10 -2
  29. data/app/services/action_agent/agent_execution_service.rb +4 -2
  30. data/app/services/action_agent/agent_tool_roster.rb +20 -9
  31. data/app/services/action_agent/claude_code_auth.rb +86 -0
  32. data/app/services/action_agent/dashboard_assistant_service.rb +47 -5
  33. data/app/services/action_agent/evaluation_tool_resolver.rb +18 -0
  34. data/app/services/action_agent/github_client.rb +111 -0
  35. data/app/services/action_agent/local_sandbox_backend.rb +1689 -0
  36. data/app/services/action_agent/local_sandbox_databases.rb +257 -0
  37. data/app/services/action_agent/mcp_client.rb +5 -1
  38. data/app/services/action_agent/mcp_tool_dispatcher.rb +137 -17
  39. data/app/services/action_agent/mock_sandbox_backend.rb +39 -0
  40. data/app/services/action_agent/ollama_host_probe.rb +75 -0
  41. data/app/services/action_agent/payload_bounds.rb +36 -0
  42. data/app/services/action_agent/sandbox_manifest.rb +67 -0
  43. data/app/services/action_agent/sandbox_orchestrator.rb +69 -14
  44. data/app/services/action_agent/scenario_evaluation_runner.rb +50 -4
  45. data/app/services/action_agent/secret_scrubber.rb +37 -0
  46. data/app/services/action_agent/tool_discovery.rb +19 -5
  47. data/config/routes.rb +22 -3
  48. data/lib/action_agent/engine.rb +1 -0
  49. data/lib/action_agent/version.rb +1 -1
  50. data/lib/action_agent.rb +147 -3
  51. data/lib/generators/action_agent/install_generator.rb +30 -3
  52. data/lib/generators/action_agent/templates/action_agent.rb.erb +44 -0
  53. data/lib/generators/action_agent/templates/add_provider_key_api_key.rb.erb +25 -0
  54. data/lib/generators/action_agent/templates/create_active_agent_code_sessions.rb.erb +59 -0
  55. data/lib/generators/action_agent/templates/create_active_agent_dashboard_tables.rb.erb +2 -0
  56. data/lib/generators/action_agent/templates/create_active_agent_github_connections.rb.erb +61 -0
  57. data/lib/tasks/claude_code.rake +16 -0
  58. data/lib/tasks/sandbox.rake +26 -0
  59. metadata +23 -1
@@ -0,0 +1,1689 @@
1
+ # frozen_string_literal: true
2
+
3
+ require "io/wait"
4
+ require "net/http"
5
+ require "open3"
6
+ require "socket"
7
+ require "tmpdir"
8
+
9
+ module ActionAgent
10
+ # Boots app_runtime checkouts as child processes of the dashboard itself,
11
+ # with no containers: the sandbox backend for a developer's own machine
12
+ # (see ActionAgent.local_sandboxes_enabled?). Each sandbox is a workspace
13
+ # under ActionAgent.local_sandbox_root:
14
+ #
15
+ # <session_id>/
16
+ # app/ the checkout
17
+ # db/ its SQLite databases, if it uses SQLite (see
18
+ # LocalSandboxDatabases; state.json records the
19
+ # database variables it was given)
20
+ # runtime.json the manifest the checkout wrote (see SandboxManifest)
21
+ # state.json { pid, port, started_at, code_sessions: { "<id>" => pid } },
22
+ # plus the commit checked out, the boot step running
23
+ # (step_pid), when each recorded process started, cancels
24
+ # sent and cancels that found nothing to stop yet, and
25
+ # whether a terminate is under way
26
+ # state.lock what changes to state.json are serialized on
27
+ # logs/ checkout, setup, manifest, server and claude-<id> logs
28
+ # claude/ CLAUDE_CONFIG_DIR for Claude Code sessions (with
29
+ # ActionAgent.claude_code_auth = :api_key; with
30
+ # :local_login they use the user's own configuration)
31
+ #
32
+ # The orchestrator builds a new backend for every call, and the dashboard
33
+ # may restart while a sandbox runs, so whatever a later call needs lives in
34
+ # state.json rather than in instance variables. Calls race each other
35
+ # through that file (a cancel or a terminate while Claude Code starts), so
36
+ # it is only ever changed under an exclusive lock, and replaced whole (see
37
+ # #update_state) so a crash never leaves it half written.
38
+ #
39
+ # How a checkout boots is up to its .activeagents/sandbox.yml (see Config).
40
+ # Every process starts from a sanitized copy of the dashboard's environment
41
+ # (see .sanitized_environment), so a checkout never sees the dashboard's
42
+ # database or secrets. The GitHub token reaches only the fetch, and the
43
+ # Claude Code API key only Claude Code. With
44
+ # ActionAgent.claude_code_auth = :local_login no credential is passed at
45
+ # all: Claude Code runs on the machine's own login.
46
+ class LocalSandboxBackend
47
+ class Error < RuntimeError; end
48
+
49
+ HANDLE_PREFIX = "local-"
50
+ # Session ids are generated UUIDs; anything else is refused before it can
51
+ # name a path outside the sandbox root.
52
+ SESSION_ID = /\A[A-Za-z0-9][A-Za-z0-9_-]{0,127}\z/
53
+ # Set in every process a sandbox starts. Where state.json has no start
54
+ # time for a recorded pid, this is how it is told apart from an
55
+ # unrelated process that later reused it.
56
+ SESSION_ID_ENV = "ACTION_AGENT_SANDBOX_SESSION_ID"
57
+ COMMIT_ID = /\A\h{40}(?:\h{24})?\z/
58
+
59
+ # A GET on the MCP path answers 405 once the engine is mounted and
60
+ # serving; 401 and 200 also mean something is up and answering there.
61
+ READY_STATUSES = [ 405, 401, 200 ].freeze
62
+ POLL_INTERVAL = 0.25
63
+ # Between SIGTERM and SIGKILL when stopping a sandbox, or a cancelled
64
+ # Claude Code session.
65
+ STOP_GRACE = 10
66
+ LOG_TAIL_LINES = 20
67
+ LOG_TAIL_BYTES = 64 * 1024
68
+
69
+ # Bounds on the git commands run in a checkout (rev-parse after the
70
+ # fetch, add and diff after a session) and on the diff a Claude Code
71
+ # session hands back. The diff limit is above CodeSession's own, so its
72
+ # truncation notice still shows.
73
+ GIT_TIMEOUT = 60
74
+ MAX_DIFF_BYTES = 1_000_000
75
+ # How long a cancel that found no Claude Code process is remembered. The
76
+ # session it names starts within seconds of being marked running, or
77
+ # never; this only keeps old ones from piling up in state.json.
78
+ CANCEL_MEMORY = 3600
79
+ # One stream-json line can carry a whole file; past this it is cut.
80
+ MAX_EVENT_LINE_BYTES = 8 * 1024 * 1024
81
+ # How long output may keep arriving after Claude Code itself exited.
82
+ OUTPUT_DRAIN_GRACE = 2
83
+ HELP_TIMEOUT = 15
84
+ PS_ENVIRONMENT = { "TZ" => "UTC", "LC_ALL" => "C", "LANG" => "C" }.freeze
85
+ # How often a running Claude Code session looks for a cancel in
86
+ # state.json.
87
+ CANCEL_CHECK_INTERVAL = 0.5
88
+ # How long terminate waits on the checkout's database cleanup.
89
+ DATABASE_DROP_TIMEOUT = 60
90
+ DATABASE_DROP_TARGETS_ENV = "ACTION_AGENT_SANDBOX_DATABASE_DROP_TARGETS"
91
+ # Runs in the checkout's Rails bundle. Filter its resolved configurations
92
+ # before either checking protection or dropping anything. A url: added
93
+ # after boot overrides DATABASE_URL in Rails, so require the recorded URL
94
+ # itself as well as its adapter/database to match. No matching config
95
+ # means no drop; legacy state without this allow-list is never inferred.
96
+ DATABASE_DROP_SCRIPT = <<~'RUBY'.freeze
97
+ require "json"
98
+ targets = JSON.parse(ENV.fetch("ACTION_AGENT_SANDBOX_DATABASE_DROP_TARGETS"))
99
+ selected = ActiveRecord::Base.configurations.configs_for(env_name: Rails.env).select do |config|
100
+ url = targets[config.name]
101
+ next false unless url.is_a?(String)
102
+
103
+ expected = ActiveRecord::DatabaseConfigurations::UrlConfig.new(Rails.env, config.name, url, {})
104
+ config.is_a?(ActiveRecord::DatabaseConfigurations::UrlConfig) && config.url == url &&
105
+ config.adapter == expected.adapter && config.database == expected.database
106
+ end
107
+ ActiveRecord::Base.configurations = selected
108
+ ActiveRecord::Tasks::DatabaseTasks.check_protected_environments!(Rails.env)
109
+ selected.each { |config| ActiveRecord::Tasks::DatabaseTasks.drop(config) }
110
+ RUBY
111
+ MODEL_NAME = %r{\A[A-Za-z0-9][A-Za-z0-9._:/@\[\]-]{0,127}\z}
112
+ # `claude auth status`: how long its answer is trusted, how long it may
113
+ # take, and what its authMethod and apiProvider may look like to be
114
+ # shown (a short label, never free text).
115
+ LOGIN_STATUS_TTL = 60
116
+ # A logged-out answer is re-asked soon: someone who just ran
117
+ # `claude /login` clicks "Check again" and expects it to turn green.
118
+ LOGGED_OUT_STATUS_TTL = 3
119
+ LOGIN_STATUS_TIMEOUT = 10
120
+ LOGIN_LABEL = /\A[A-Za-z0-9][A-Za-z0-9._ -]{0,63}\z/
121
+ LOGGED_OUT = { logged_in: false, auth_method: nil, api_provider: nil }.freeze
122
+
123
+ # Never inherited from the dashboard: its database, its keys, and the
124
+ # Ruby/Bundler setup of its own bundle (a checkout has its own Gemfile).
125
+ DROPPED_VARIABLES = %w[
126
+ DATABASE_URL REDIS_URL SECRET_KEY_BASE RAILS_MASTER_KEY RAILS_ENV RACK_ENV PORT
127
+ BUNDLE_GEMFILE RUBYOPT RUBYLIB
128
+ SSH_AUTH_SOCK
129
+ ].freeze
130
+ DROPPED_VARIABLE_PATTERN = /
131
+ _DATABASE_URL\z | \AACTIVE_RECORD_ENCRYPTION_ | \ABUNDLER?_ |
132
+ # Where git finds a repository, and its configuration. Set when the
133
+ # dashboard runs under a git hook, and they would point the checkout's
134
+ # git at the dashboard's own repository or config.
135
+ \AGIT_(?:DIR|WORK_TREE|INDEX_FILE|OBJECT_DIRECTORY|ALTERNATE_OBJECT_DIRECTORIES|COMMON_DIR|
136
+ NAMESPACE|PREFIX|QUARANTINE_PATH|CONFIG|CONFIG_PARAMETERS|CONFIG_COUNT|CONFIG_KEY_\d+|CONFIG_VALUE_\d+|
137
+ CONFIG_GLOBAL|CONFIG_SYSTEM|CONFIG_NOSYSTEM)\z |
138
+ # The dashboard's own model-provider and Claude Code settings. A
139
+ # developer often runs the dashboard from inside Claude Code, which
140
+ # exports CLAUDECODE, CLAUDE_CODE_* (its own session id among them) and
141
+ # ANTHROPIC_BASE_URL; a session inheriting those joins the developer's
142
+ # session, and a base URL redirects the owner's credential. A session
143
+ # gets exactly the Claude Code variables the backend sets.
144
+ \A(?:ANTHROPIC|CLAUDE|OPENAI|OPEN_AI|OPENROUTER|OPEN_ROUTER|OLLAMA)(?:_|\z) | \ACLAUDECODE\z
145
+ /x
146
+ SECRET_VARIABLE = /
147
+ SECRET | TOKEN | PASSWORD | PASSWD | PASSPHRASE | API_KEY | APIKEY | PRIVATE_KEY | CREDENTIAL | ACCESS_KEY |
148
+ # DB_PASS, MYSQL_PWD, LOCKBOX_MASTER_KEY, SENTRY_DSN, SLACK_WEBHOOK_URL, GITHUB_PAT
149
+ (?:\A|_)PASS\z | (?:\A|_)PWD\z | _KEY\z | DSN\z | WEBHOOK | (?:\A|_)PAT\z
150
+ /xi
151
+ # A URL carrying credentials, whatever its name: a password
152
+ # (redis://:secret@host) or a token as the username alone
153
+ # (https://ghp_x@github.com). Any userinfo counts, anywhere in the value.
154
+ CREDENTIALED_URL = %r{[a-z][a-z0-9+.-]*://[^/\s@]+@}i
155
+
156
+ # Fetches the checkout with the token in this process's environment only.
157
+ # Git gets the credential through GIT_CONFIG_* for the one fetch: never in
158
+ # argv (which `ps` shows every user) and never in .git/config (which the
159
+ # checked-out app and Claude Code can read). No credential helper is asked,
160
+ # so the fetch uses this token or nothing.
161
+ CHECKOUT_SCRIPT = <<~'SH'
162
+ set -eu
163
+ git init -q "$APP_DIR"
164
+ cd "$APP_DIR"
165
+ git remote add origin "$CHECKOUT_URL"
166
+ header=""
167
+ if [ -n "${CHECKOUT_TOKEN:-}" ]; then
168
+ header="Authorization: Basic $(printf '%s:%s' "$CHECKOUT_USERNAME" "$CHECKOUT_TOKEN" | base64 | tr -d '\n')"
169
+ fi
170
+ unset CHECKOUT_TOKEN
171
+ GIT_CONFIG_COUNT=2 \
172
+ GIT_CONFIG_KEY_0=credential.helper GIT_CONFIG_VALUE_0= \
173
+ GIT_CONFIG_KEY_1=http.extraHeader GIT_CONFIG_VALUE_1="$header" \
174
+ git fetch -q --depth 1 origin "$CHECKOUT_REF"
175
+ git checkout -q --detach FETCH_HEAD
176
+ SH
177
+
178
+ # .activeagents/sandbox.yml, the checkout's say in how it boots. Every key
179
+ # is optional: a Rails app that mounts this engine boots without the file.
180
+ #
181
+ # env: extra environment for setup, manifest and server
182
+ # setup: commands run once after checkout, in order
183
+ # manifest: writes the manifest JSON to $ACTION_AGENT_SANDBOX_MANIFEST
184
+ # start: serves on 127.0.0.1:$PORT and keeps running
185
+ #
186
+ # Unknown keys are ignored, so a newer file still boots here.
187
+ class Config
188
+ PATH = File.join(".activeagents", "sandbox.yml")
189
+ DEFAULT_SETUP = [ "bundle install", "bin/rails db:prepare" ].freeze
190
+ DEFAULT_MANIFEST = "bin/rails action_agent:sandbox:manifest"
191
+ DEFAULT_START = "bin/rails server -b 127.0.0.1 -p $PORT"
192
+ ENV_NAME = /\A[A-Za-z_][A-Za-z0-9_]*\z/
193
+
194
+ attr_reader :env, :setup, :manifest, :start
195
+
196
+ def self.load(app_dir)
197
+ file = Pathname(app_dir).join(PATH)
198
+ data = file.file? ? YAML.safe_load(file.read, aliases: false) : nil
199
+ data = {} if data.nil?
200
+ invalid!("it must be a mapping of settings") unless data.is_a?(Hash)
201
+
202
+ new(
203
+ env: parse_env(data["env"]),
204
+ setup: parse_setup(data["setup"]),
205
+ manifest: parse_command(data, "manifest", DEFAULT_MANIFEST),
206
+ start: parse_command(data, "start", DEFAULT_START)
207
+ )
208
+ rescue Psych::Exception => e
209
+ invalid!("it is not valid YAML (#{e.message.truncate(200)})")
210
+ end
211
+
212
+ def self.parse_env(value)
213
+ return {} if value.nil?
214
+ invalid!("`env` must be a mapping of names to strings") unless value.is_a?(Hash)
215
+
216
+ value.to_h do |name, setting|
217
+ scalar = case setting
218
+ when String, Integer, Float, true, false then true
219
+ else false
220
+ end
221
+ unless scalar && name.is_a?(String) && ENV_NAME.match?(name)
222
+ invalid!("`env` must map variable names to strings (#{name.inspect} does not)")
223
+ end
224
+
225
+ [ name, setting.to_s ]
226
+ end
227
+ end
228
+
229
+ def self.parse_setup(value)
230
+ return DEFAULT_SETUP if value.nil?
231
+
232
+ commands = value.is_a?(String) ? [ value ] : value
233
+ unless commands.is_a?(Array) && commands.all? { |command| command.is_a?(String) && command.present? }
234
+ invalid!("`setup` must be a list of commands")
235
+ end
236
+ commands
237
+ end
238
+
239
+ def self.parse_command(data, key, default)
240
+ return default if data[key].nil?
241
+ invalid!("`#{key}` must be a command") unless data[key].is_a?(String) && data[key].present?
242
+
243
+ data[key]
244
+ end
245
+
246
+ def self.invalid!(problem)
247
+ raise Error, "Sandbox configuration failed: #{PATH} is malformed: #{problem}"
248
+ end
249
+
250
+ def initialize(env:, setup:, manifest:, start:)
251
+ @env = env
252
+ @setup = setup
253
+ @manifest = manifest
254
+ @start = start
255
+ end
256
+ end
257
+
258
+ class << self
259
+ # The environment every sandbox process starts from: the dashboard's
260
+ # own, as it was before Bundler set it up, minus the dashboard's
261
+ # database, keys and anything named like a credential. Processes are
262
+ # spawned with exactly this (plus what the step adds) and
263
+ # unsetenv_others, so nothing else leaks through.
264
+ #
265
+ # Bundler.unbundled_env is what Bundler.with_unbundled_env swaps into
266
+ # ENV; reading it directly leaves the process-wide ENV alone, which other
267
+ # threads of the dashboard are reading at the same time.
268
+ #
269
+ # @param source [Hash, nil] the environment to sanitize (for tests)
270
+ # @return [Hash{String => String}]
271
+ def sanitized_environment(source = nil)
272
+ source ||= defined?(::Bundler) ? ::Bundler.unbundled_env : ENV.to_h
273
+
274
+ source.each_with_object({}) do |(name, value), env|
275
+ name = name.to_s
276
+ next if value.nil? || DROPPED_VARIABLES.include?(name)
277
+ next if DROPPED_VARIABLE_PATTERN.match?(name) || SECRET_VARIABLE.match?(name)
278
+ next if CREDENTIALED_URL.match?(value.to_s)
279
+
280
+ env[name] = value.to_s
281
+ end
282
+ end
283
+
284
+ # Whether +command+ (a Claude Code executable) takes
285
+ # `--permission-prompts`, read from its --help once per process: the
286
+ # answer only changes when the CLI is upgraded. The block runs the help
287
+ # and returns its output, or nil when it could not run (which is not
288
+ # remembered, so installing the CLI later is noticed).
289
+ def permission_prompts_supported?(command)
290
+ @cli_support_lock.synchronize do
291
+ return @cli_support[command] if @cli_support.key?(command)
292
+
293
+ help = yield
294
+ help.nil? ? false : (@cli_support[command] = help.include?("--permission-prompts"))
295
+ end
296
+ end
297
+ end
298
+
299
+ @cli_support = {}
300
+ @cli_support_lock = Mutex.new
301
+ @login_status = {}
302
+ @login_status_lock = Mutex.new
303
+
304
+ class << self
305
+ # Whether this machine's Claude Code is logged in, for
306
+ # ActionAgent.claude_code_auth = :local_login: what
307
+ # `<claude_code_command> auth status --json` says, run with the
308
+ # sanitized environment (so it reads the user's own ~/.claude, as a
309
+ # session will) and bounded in time.
310
+ #
311
+ # Only whether it is logged in and how (authMethod, apiProvider) is
312
+ # kept. The rest of that output (an account's email, its organization)
313
+ # is never logged, stored or returned, and the credential itself is
314
+ # never in it. A CLI that is missing, fails or answers something else
315
+ # reads as logged out.
316
+ #
317
+ # Asked at most once a LOGIN_STATUS_TTL per command (a logged-out
318
+ # answer, LOGGED_OUT_STATUS_TTL): the listing that shows it is polled.
319
+ #
320
+ # @return [Hash] { logged_in: Boolean, auth_method: String?, api_provider: String? }
321
+ def claude_login_status
322
+ command = ActionAgent.claude_code_command.to_s
323
+ @login_status_lock.synchronize do
324
+ cached = @login_status[command]
325
+ return cached[:status] if cached && Process.clock_gettime(Process::CLOCK_MONOTONIC) < cached[:until]
326
+ end
327
+
328
+ status = new.send(:read_claude_login_status, command)
329
+ @login_status_lock.synchronize do
330
+ @login_status[command] = {
331
+ status: status,
332
+ until: Process.clock_gettime(Process::CLOCK_MONOTONIC) +
333
+ (status[:logged_in] ? LOGIN_STATUS_TTL : LOGGED_OUT_STATUS_TTL)
334
+ }
335
+ end
336
+ status
337
+ end
338
+
339
+ # Forgets the cached login status, so the next call asks the CLI again.
340
+ def reset_claude_login_status!
341
+ @login_status_lock.synchronize { @login_status.clear }
342
+ end
343
+
344
+ # The fields of `claude auth status --json` the dashboard keeps.
345
+ def parse_login_status(output)
346
+ data = JSON.parse(output.to_s.strip)
347
+ return LOGGED_OUT unless data.is_a?(Hash) && data["loggedIn"] == true
348
+
349
+ {
350
+ logged_in: true,
351
+ auth_method: login_label(data["authMethod"]),
352
+ api_provider: login_label(data["apiProvider"])
353
+ }
354
+ rescue JSON::ParserError
355
+ LOGGED_OUT
356
+ end
357
+
358
+ private
359
+
360
+ def login_label(value)
361
+ value if value.is_a?(String) && LOGIN_LABEL.match?(value)
362
+ end
363
+ end
364
+
365
+ # Clones the session's checkout, boots it as sandbox.yml says, and waits
366
+ # until its MCP facade answers.
367
+ #
368
+ # @return [Hash] the handle (container_name), url, mcp_url and mcp_token
369
+ def create_sandbox(session, instance_tier: nil)
370
+ ensure_enabled!
371
+ unless session.app_runtime?
372
+ raise Error, "The local sandbox backend only boots app_runtime checkouts, not #{session.sandbox_type} sandboxes"
373
+ end
374
+
375
+ spec = session.checkout_spec
376
+ raise Error, "Sandbox #{session.session_id} has no checkout to boot (is GitHub still connected?)" if spec.blank?
377
+
378
+ session_id = session_id!(session.session_id)
379
+ # A retried provision (say the dashboard restarted mid-boot) starts
380
+ # clean rather than on a half-built checkout next to a stray server.
381
+ unless discard(session_id)
382
+ raise Error, "Sandbox #{session_id} still has processes from an earlier boot that could not be stopped; " \
383
+ "see the dashboard log"
384
+ end
385
+ boot(session_id, spec, [ spec[:token], *session.runtime_environment.values ])
386
+ end
387
+
388
+ # A sandbox's handle follows from its session id, so one whose boot was
389
+ # never recorded can still be found and stopped.
390
+ def handle_for(session)
391
+ "local-#{session.session_id}"
392
+ end
393
+
394
+ # Stops the sandbox's server, its boot step and any Claude Code sessions
395
+ # it recorded, then removes its workspace. Returns true, also when there
396
+ # was nothing to stop.
397
+ #
398
+ # False when something the sandbox recorded is still alive and could
399
+ # not be stopped (or told apart from an unrelated process): its
400
+ # workspace, and state.json with it, is kept, so the handle is kept for
401
+ # the reaper to try again.
402
+ def terminate(handle)
403
+ session_id = session_id_from(handle)
404
+ session_id ? discard(session_id) : true
405
+ end
406
+
407
+ # @return [Hash] { status: "running" | "stopped" | "not_found", pid:, port: }
408
+ def status(handle)
409
+ session_id = session_id_from(handle)
410
+ workspace = session_id && workspace_for(session_id)
411
+ return { status: "not_found", pid: nil, port: nil } unless workspace&.directory?
412
+
413
+ state = read_state(workspace)
414
+ pid = state["pid"]
415
+ running = recorded_group?(pid, session_id, state) && group_alive?(pid)
416
+ { status: running ? "running" : "stopped", pid: pid, port: state["port"] }
417
+ end
418
+
419
+ # One entry per workspace on disk.
420
+ def list_sandboxes
421
+ root = ActionAgent.local_sandbox_root
422
+ return [] unless root.directory?
423
+
424
+ root.children.select(&:directory?).filter_map do |dir|
425
+ session_id = dir.basename.to_s
426
+ next unless SESSION_ID.match?(session_id)
427
+
428
+ handle = "#{HANDLE_PREFIX}#{session_id}"
429
+ status(handle).merge(container_name: handle, session_id: session_id)
430
+ end
431
+ end
432
+
433
+ # The engine reaps expired sandboxes itself (SandboxCleanupJob), one
434
+ # terminate at a time.
435
+ def cleanup_expired
436
+ 0
437
+ end
438
+
439
+ # Runs Claude Code headless in the sandbox's checkout, yielding each
440
+ # stream-json event (a Hash, already scrubbed of the sandbox's secrets)
441
+ # as it arrives.
442
+ #
443
+ # @return [Hash] { exit_status: Integer, diff: String, stderr_tail: String }
444
+ def run_code_session(sandbox, code_session, &on_event)
445
+ ensure_enabled!
446
+ session_id = session_id!(sandbox.session_id)
447
+ workspace = workspace_for(session_id)
448
+ app = workspace.join("app")
449
+ raise Error, "Sandbox #{session_id} has no local checkout: start the sandbox again" unless app.directory?
450
+
451
+ credentials = session_credentials(sandbox)
452
+ secrets = sandbox_secrets(sandbox, credentials)
453
+ argv = claude_argv(code_session)
454
+ database_env = read_state(workspace)["database_env"]
455
+ env = self.class.sanitized_environment
456
+ .merge(database_env.is_a?(Hash) ? database_env.transform_values(&:to_s) : {})
457
+ .merge(credentials.to_h { |name, value| [ name.to_s, value.to_s ] })
458
+ .merge(
459
+ SESSION_ID_ENV => session_id,
460
+ "DISABLE_AUTOUPDATER" => "1",
461
+ "DISABLE_TELEMETRY" => "1",
462
+ "DISABLE_ERROR_REPORTING" => "1",
463
+ "CLAUDE_CODE_DISABLE_NONESSENTIAL_TRAFFIC" => "1"
464
+ )
465
+ # With an API key a session gets a Claude Code configuration of its
466
+ # own, in the workspace. With the machine's own login it must use the
467
+ # user's: that is where `claude /login` left the credentials (HOME,
468
+ # which the sanitized environment keeps, or the keychain).
469
+ unless ClaudeCodeAuth.local_login?
470
+ env["CLAUDE_CONFIG_DIR"] = workspace.join("claude").to_s
471
+ FileUtils.mkdir_p(workspace.join("claude"), mode: 0o700)
472
+ end
473
+
474
+ run_claude(workspace, code_session, argv, env, secrets, &on_event)
475
+ end
476
+
477
+ # Stops a running Claude Code session: SIGTERM to its process group. The
478
+ # run_code_session that started it then finishes with what it has. The
479
+ # cancel is noted in state.json too, and run_code_session, which watches
480
+ # for it, sends SIGKILL once STOP_GRACE has passed: a CLI that ignores
481
+ # SIGTERM would otherwise run on until claude_code_timeout.
482
+ #
483
+ # The session is marked running before Claude Code starts (the --help
484
+ # probe alone can take seconds), so a cancel can find no process yet.
485
+ # It is then remembered in state.json, under the lock run_code_session
486
+ # records its process with, and the session stops as soon as it starts
487
+ # (see #record_code_session), or never starts at all.
488
+ def cancel_code_session(sandbox, code_session)
489
+ session_id = session_id!(sandbox.session_id)
490
+ workspace = workspace_for(session_id)
491
+ return true unless workspace.directory?
492
+
493
+ key = code_session.id.to_s
494
+ update_state(workspace, create: false) do |state|
495
+ pid = state_hash(state, "code_sessions")[key]
496
+ if pid
497
+ if recorded_group?(pid, session_id, state)
498
+ signal_group(pid, "TERM")
499
+ # Wall-clock time: the session may run in another process.
500
+ state_hash(state, "cancelling_code_sessions")[key] ||= Time.now.to_f
501
+ end
502
+ else
503
+ cancels = state_hash(state, "cancelled_code_sessions")
504
+ cancels.delete_if { |_id, at| !at.is_a?(Numeric) || at < Time.now.to_f - CANCEL_MEMORY }
505
+ cancels[key] = Time.now.to_i
506
+ end
507
+ end
508
+ true
509
+ rescue Errno::ENOENT
510
+ # No state.json, so the sandbox never booted, or it was removed
511
+ # meanwhile: either way no Claude Code session runs there.
512
+ true
513
+ end
514
+
515
+ private
516
+
517
+ def ensure_enabled!
518
+ return if ActionAgent.local_sandboxes_enabled?
519
+
520
+ raise Error, "Local sandboxes are disabled. They run checkouts and Claude Code as processes on this machine; " \
521
+ "set ActionAgent.local_sandboxes_enabled = true to allow them"
522
+ end
523
+
524
+ # --- Boot -------------------------------------------------------------
525
+
526
+ def boot(session_id, spec, secrets)
527
+ workspace = workspace_for(session_id)
528
+ deadline = deadline_after(ActionAgent.local_sandbox_boot_timeout)
529
+ booted = false
530
+ server_pid = nil
531
+
532
+ prepare_workspace(workspace)
533
+ checkout!(workspace, spec, secrets, deadline)
534
+
535
+ app = workspace.join("app")
536
+ config = Config.load(app)
537
+ databases = assign_databases(workspace, app, config, spec)
538
+ env = self.class.sanitized_environment.merge(databases).merge(config.env).merge(
539
+ # Merged after the file's env, so a checkout cannot move them.
540
+ SandboxManifest::PATH_ENV => workspace.join("runtime.json").to_s,
541
+ SESSION_ID_ENV => session_id
542
+ )
543
+
544
+ config.setup.each do |command|
545
+ run_step!(workspace, "setup", command, env: env, chdir: app, deadline: deadline, secrets: secrets)
546
+ end
547
+
548
+ # Picked after setup, which can take minutes, so that the port is
549
+ # likely still free when the server binds it. Nothing reserves it
550
+ # meanwhile: #wait_until_ready! only accepts an answer from a listener
551
+ # of the server's own process group.
552
+ port = free_port
553
+ env = env.merge("PORT" => port.to_s)
554
+ manifest = run_manifest!(workspace, config, env, deadline, secrets)
555
+
556
+ server_pid, waiter = start_server(workspace, config, env, port)
557
+ wait_until_ready!(workspace, port, manifest, server_pid, waiter, deadline, secrets)
558
+ booted = true
559
+
560
+ {
561
+ container_name: "#{HANDLE_PREFIX}#{session_id}",
562
+ url: "http://127.0.0.1:#{port}",
563
+ container_ip: "127.0.0.1",
564
+ mcp_url: "http://127.0.0.1:#{port}#{manifest["mcp_path"]}",
565
+ mcp_token: manifest["mcp_token"],
566
+ created_at: Time.current
567
+ }
568
+ rescue Error
569
+ raise
570
+ rescue StandardError => e
571
+ raise Error, SecretScrubber.scrub("Sandbox boot failed: #{e.class.name}: #{e.message}", secrets)
572
+ ensure
573
+ # A failed boot leaves nothing running and nothing on disk: no handle
574
+ # was reported, so nothing would ever reap it. The error carries the
575
+ # failing step's log tail.
576
+ unless booted
577
+ stop_groups([ server_pid ].compact)
578
+ # A setup that got as far as db:prepare created the databases.
579
+ drop_databases(workspace) if workspace
580
+ remove_workspace(workspace) if workspace
581
+ end
582
+ end
583
+
584
+ # The sandbox's own databases (see LocalSandboxDatabases), recorded in
585
+ # state.json before setup can create them: a terminate, in this process
586
+ # or after a restart, drops what is recorded there. Claude Code sessions
587
+ # get them too, so a `bin/rails db:migrate` a session runs lands in the
588
+ # sandbox's database rather than the developer's.
589
+ def assign_databases(workspace, app, config, spec)
590
+ plan = LocalSandboxDatabases.plan(
591
+ app: app, workspace: workspace, session_id: workspace.basename.to_s, overrides: config.env,
592
+ fallback_name: spec[:repository].to_s.split("/").last
593
+ )
594
+ if plan.notes.any?
595
+ File.open(log_path(workspace, "setup"), "a") do |file|
596
+ plan.notes.each { |line| file.puts("# sandbox database: #{line}") }
597
+ end
598
+ end
599
+ return {} if plan.empty?
600
+
601
+ update_state(workspace) do |state|
602
+ state["database_env"] = plan.env
603
+ state["drop_databases"] = plan.drop
604
+ state["database_drop_targets"] = plan.drop_targets
605
+ state["database_rails_env"] = plan.rails_env
606
+ end
607
+ plan.env
608
+ end
609
+
610
+ # Drops the server databases a sandbox was given (PostgreSQL, MySQL),
611
+ # with the checkout's own Rails database tasks: the adapter, its gem and
612
+ # its credentials are the checkout's. Best effort and bounded: a drop
613
+ # that fails or hangs is logged and the sandbox goes anyway. Only ever
614
+ # for the named URLs recorded here. Overridden databases and newly added
615
+ # configurations are never selected, even after the checkout is edited.
616
+ def drop_databases(workspace)
617
+ state = read_state(workspace)
618
+ database_env = state["database_env"]
619
+ targets = state["database_drop_targets"]
620
+ return unless targets.is_a?(Hash) && targets.any? && database_env.is_a?(Hash)
621
+
622
+ app = workspace.join("app")
623
+ return unless app.join("bin", "rails").file?
624
+
625
+ file_env = begin
626
+ Config.load(app).env
627
+ rescue Error
628
+ {}
629
+ end
630
+ env = self.class.sanitized_environment.merge(file_env).merge(database_env.transform_values(&:to_s)).merge(
631
+ SESSION_ID_ENV => workspace.basename.to_s,
632
+ "RAILS_ENV" => state.fetch("database_rails_env"),
633
+ DATABASE_DROP_TARGETS_ENV => JSON.generate(targets)
634
+ )
635
+ output, status = capture(env, [ "bin/rails", "runner", DATABASE_DROP_SCRIPT ], chdir: app, limit: 64 * 1024, timeout: database_drop_timeout,
636
+ err: [ :child, :out ])
637
+ return if status&.success?
638
+
639
+ Rails.logger.warn("[ActionAgent] sandbox #{workspace.basename}: could not drop its databases " \
640
+ "(#{status ? describe(status) : "timed out"}): #{output.to_s.force_encoding(Encoding::UTF_8).scrub.lines.last(5).join.strip}")
641
+ rescue StandardError => e
642
+ Rails.logger.warn("[ActionAgent] sandbox #{workspace.basename}: could not drop its databases: #{e.class.name}: #{e.message}")
643
+ end
644
+
645
+ # A method, so the tests can shorten it.
646
+ def database_drop_timeout
647
+ DATABASE_DROP_TIMEOUT
648
+ end
649
+
650
+ def prepare_workspace(workspace)
651
+ FileUtils.mkdir_p(ActionAgent.local_sandbox_root)
652
+ # Owner-only: the checkout and runtime.json hold the app's own secrets.
653
+ FileUtils.mkdir_p(workspace, mode: 0o700)
654
+ FileUtils.mkdir_p(workspace.join("logs"), mode: 0o700)
655
+ FileUtils.mkdir_p(workspace.join("claude"), mode: 0o700)
656
+ end
657
+
658
+ def checkout!(workspace, spec, secrets, deadline)
659
+ env = self.class.sanitized_environment.merge(
660
+ "APP_DIR" => workspace.join("app").to_s,
661
+ "CHECKOUT_URL" => spec[:clone_url].to_s,
662
+ "CHECKOUT_REF" => spec[:ref].presence || "HEAD",
663
+ "CHECKOUT_USERNAME" => spec[:username].presence || "x-access-token",
664
+ "CHECKOUT_TOKEN" => spec[:token].to_s,
665
+ SESSION_ID_ENV => workspace.basename.to_s,
666
+ # Fail rather than prompt when the token is refused.
667
+ "GIT_TERMINAL_PROMPT" => "0",
668
+ "GIT_ASKPASS" => ""
669
+ )
670
+ run_step!(workspace, "checkout", CHECKOUT_SCRIPT, env: env, chdir: workspace, deadline: deadline, secrets: secrets,
671
+ label: "git fetch #{spec[:clone_url]} #{spec[:ref]}")
672
+ record_checkout_commit(workspace)
673
+ end
674
+
675
+ # What a Claude Code session's diff is taken against, so it shows the
676
+ # session's changes even where the session committed them, or staged a
677
+ # removal, itself. Without it the diff falls back to HEAD.
678
+ def record_checkout_commit(workspace)
679
+ argv = [ *git_command(workspace.join("app")), "rev-parse", "--verify", "HEAD^{commit}" ]
680
+ output, status = capture(self.class.sanitized_environment, argv, chdir: workspace, limit: 1024, timeout: GIT_TIMEOUT)
681
+ commit = output.strip
682
+ update_state(workspace) { |state| state["checkout_commit"] = commit } if status&.success? && COMMIT_ID.match?(commit)
683
+ end
684
+
685
+ def run_manifest!(workspace, config, env, deadline, secrets)
686
+ path = workspace.join("runtime.json")
687
+ FileUtils.rm_f(path)
688
+ run_step!(workspace, "manifest", config.manifest, env: env, chdir: workspace.join("app"), deadline: deadline, secrets: secrets)
689
+
690
+ log = log_path(workspace, "manifest")
691
+ fail_step!("manifest", "`#{config.manifest}` wrote nothing to $#{SandboxManifest::PATH_ENV}", log, secrets) unless path.file?
692
+
693
+ # Written by the checkout's command, with its umask: the MCP token in
694
+ # it is for the dashboard alone.
695
+ File.chmod(0o600, path)
696
+ manifest = SandboxManifest.parse(path.read)
697
+ URI.parse("http://127.0.0.1#{manifest["mcp_path"]}")
698
+ manifest
699
+ rescue SandboxManifest::Error, URI::InvalidURIError => e
700
+ fail_step!("manifest", e.message, log_path(workspace, "manifest"), secrets)
701
+ end
702
+
703
+ # Runs one boot command to completion in its own process group, with its
704
+ # output in logs/<step>.log. Anything it left running in that group is
705
+ # stopped too: a setup command is not a way to start services the
706
+ # sandbox never records (that is what `start` is for).
707
+ #
708
+ # The group is stopped however the wait ends: the deadline, or any
709
+ # exception at all (a worker shutting down raises into this thread with
710
+ # Thread#raise, which is no StandardError). While it runs its pid is in
711
+ # state.json as step_pid, so a terminate after this process itself died
712
+ # (a crashed dashboard, a killed worker) stops a hung `bundle install`
713
+ # too; nothing else would, the deadline having died with this process.
714
+ def run_step!(workspace, step, command, env:, chdir:, deadline:, secrets:, label: command)
715
+ log = log_path(workspace, step)
716
+ File.open(log, "a") { |file| file.puts("$ #{label}") }
717
+
718
+ pid = nil
719
+ waiter = nil
720
+ finished = false
721
+ # Deferred until the group is recorded or stopped, so an interrupt
722
+ # never lands between the spawn and the ensure that stops it.
723
+ Thread.handle_interrupt(Object => :never) do
724
+ pid = spawn_group(env, "sh", "-c", command, chdir: chdir, in: File::NULL, out: [ log.to_s, "a" ], err: [ :child, :out ])
725
+ begin
726
+ Thread.handle_interrupt(Object => :immediate) do
727
+ record_step(workspace, pid)
728
+ waiter = Process.detach(pid)
729
+ finished = waiter.join(time_left(deadline))
730
+ end
731
+ ensure
732
+ stop_groups([ pid ], grace: finished ? 1 : stop_grace)
733
+ forget_step(workspace, pid)
734
+ end
735
+ end
736
+
737
+ unless finished
738
+ fail_step!(step, "`#{label}` did not finish within the boot timeout (#{ActionAgent.local_sandbox_boot_timeout}s)", log, secrets)
739
+ end
740
+ status = waiter.value
741
+ fail_step!(step, "`#{label}` exited with #{describe(status)}", log, secrets) unless status.success?
742
+ end
743
+
744
+ def record_step(workspace, pid)
745
+ update_state(workspace) do |state|
746
+ state["step_pid"] = pid
747
+ record_process_start(state, pid)
748
+ end
749
+ end
750
+
751
+ def forget_step(workspace, pid)
752
+ update_state(workspace, create: false) do |state|
753
+ next unless state["step_pid"] == pid
754
+
755
+ state.delete("step_pid")
756
+ state_hash(state, "process_starts").delete(pid.to_s)
757
+ end
758
+ rescue SystemCallError
759
+ # The workspace is gone: nothing left to forget it in.
760
+ end
761
+
762
+ def start_server(workspace, config, env, port)
763
+ log = log_path(workspace, "server")
764
+ File.open(log, "a") { |file| file.puts("$ #{config.start}") }
765
+
766
+ pid = spawn_group(env, "sh", "-c", config.start,
767
+ chdir: workspace.join("app"), in: File::NULL, out: [ log.to_s, "a" ], err: [ :child, :out ])
768
+ # Reaped by this thread for as long as the dashboard lives; after a
769
+ # restart the recorded pid is all that is left, hence state.json.
770
+ waiter = Process.detach(pid)
771
+ recorded = false
772
+ begin
773
+ update_state(workspace) do |state|
774
+ state.merge!("pid" => pid, "port" => port, "started_at" => Time.current.iso8601(3))
775
+ state_hash(state, "code_sessions")
776
+ record_process_start(state, pid)
777
+ end
778
+ recorded = true
779
+ ensure
780
+ # Unrecorded, nothing could ever stop it later, whatever the
781
+ # exception (see #run_step!).
782
+ stop_groups([ pid ]) unless recorded
783
+ end
784
+ [ pid, waiter ]
785
+ end
786
+
787
+ # Polls until the MCP path answers, and the answer comes from the
788
+ # server this sandbox started. The port was free when it was picked, but
789
+ # nothing held it after: another process can bind it first, and then
790
+ # answers in the server's place — and would be handed the MCP token.
791
+ def wait_until_ready!(workspace, port, manifest, server_pid, waiter, deadline, secrets)
792
+ log = log_path(workspace, "server")
793
+ mcp_path = manifest["mcp_path"]
794
+ uri = URI.parse("http://127.0.0.1:#{port}#{mcp_path}")
795
+ last_status = nil
796
+ foreign = false
797
+
798
+ loop do
799
+ unless waiter.alive?
800
+ taken = foreign ? " (another process was listening on port #{port})" : ""
801
+ fail_step!("server", "the server exited with #{describe(waiter.value)} before it answered on port #{port}#{taken}",
802
+ log, secrets)
803
+ end
804
+
805
+ status = probe(uri)
806
+ if READY_STATUSES.include?(status)
807
+ return if served_by_sandbox?(uri, server_pid, workspace.basename.to_s, manifest["mcp_token"])
808
+
809
+ foreign = true
810
+ end
811
+
812
+ last_status = status if status
813
+ if monotonic >= deadline
814
+ answered =
815
+ if foreign then " (port #{port} answered, but not from the sandbox's server)"
816
+ elsif last_status then " (last answer: #{last_status})"
817
+ else ""
818
+ end
819
+ fail_step!("server", "the server did not answer #{mcp_path} on port #{port} within the boot timeout " \
820
+ "(#{ActionAgent.local_sandbox_boot_timeout}s)#{answered}", log, secrets)
821
+ end
822
+ sleep POLL_INTERVAL
823
+ end
824
+ end
825
+
826
+ # Whether what listens on +uri+'s port is the sandbox's server (process
827
+ # group +pgid+). Asked of the system where it can say (/proc on Linux,
828
+ # lsof elsewhere); the token is never sent before. Where it cannot say,
829
+ # the listener must answer as only the sandbox's own MCP facade would:
830
+ # refuse a request without the manifest's token and accept one with it.
831
+ def served_by_sandbox?(uri, pgid, session_id, token)
832
+ owned = sandbox_listener?(uri.port, pgid, session_id)
833
+ return owned unless owned.nil?
834
+
835
+ rpc_status(uri, nil) == 401 && rpc_status(uri, token).to_i.between?(200, 299)
836
+ end
837
+
838
+ # true or false when the system can say whether the sandbox listens on
839
+ # +port+: a process of the server's group +pgid+ or, where /proc shows
840
+ # environments, one carrying the sandbox's session id (a server that
841
+ # starts its workers in groups of their own). nil when it cannot say.
842
+ def sandbox_listener?(port, pgid, session_id)
843
+ unless procfs?
844
+ listeners = lsof_listeners(port)
845
+ return listeners&.any? { |pid| group_of(pid) == pgid }
846
+ end
847
+
848
+ inodes = listening_socket_inodes(port)
849
+ return nil if inodes.blank?
850
+ return true if group_members(pgid).any? { |pid| socket_inodes(pid).intersect?(inodes) }
851
+
852
+ # Another user's process does not show its descriptors: whoever holds
853
+ # the socket then is not the sandbox, whose processes are this one's.
854
+ marker = "#{SESSION_ID_ENV}=#{session_id}"
855
+ Dir.children("/proc").any? do |entry|
856
+ next false unless entry.match?(/\A\d+\z/) && socket_inodes(entry).intersect?(inodes)
857
+
858
+ File.binread("/proc/#{entry}/environ").split("\0").include?(marker)
859
+ rescue SystemCallError
860
+ false
861
+ end
862
+ end
863
+
864
+ # The inodes of the TCP sockets listening on +port+, from
865
+ # /proc/net/tcp and tcp6; nil when neither can be read.
866
+ def listening_socket_inodes(port)
867
+ inodes = nil
868
+ %w[/proc/net/tcp /proc/net/tcp6].each do |table|
869
+ lines = File.readlines(table).drop(1)
870
+ inodes ||= Set.new
871
+ lines.each do |line|
872
+ # sl local_address rem_address st tx:rx tr:when retrnsmt uid timeout inode
873
+ fields = line.split
874
+ next unless fields[3] == "0A" && fields[1].to_s.split(":").last.to_i(16) == port
875
+
876
+ inodes << fields[9]
877
+ end
878
+ rescue SystemCallError
879
+ next
880
+ end
881
+ inodes
882
+ end
883
+
884
+ def group_members(pgid)
885
+ Dir.children("/proc").select do |entry|
886
+ next false unless entry.match?(/\A\d+\z/)
887
+
888
+ state, _ppid, group = proc_stat(entry)
889
+ group.to_i == pgid && !%w[Z X].include?(state)
890
+ end
891
+ end
892
+
893
+ def socket_inodes(pid)
894
+ Dir.children("/proc/#{pid}/fd").filter_map do |fd|
895
+ File.readlink("/proc/#{pid}/fd/#{fd}")[/\Asocket:\[(\d+)\]\z/, 1]
896
+ rescue SystemCallError
897
+ nil
898
+ end.to_set
899
+ rescue SystemCallError
900
+ Set.new
901
+ end
902
+
903
+ # The pids listening on +port+ as lsof reports them; nil without lsof.
904
+ def lsof_listeners(port)
905
+ output, _status = Open3.capture2("lsof", "-nP", "-a", "-iTCP:#{port}", "-sTCP:LISTEN", "-Fp", err: File::NULL)
906
+ output.lines.filter_map { |line| Integer(line[/\Ap(\d+)/, 1], exception: false) }
907
+ rescue SystemCallError
908
+ nil
909
+ end
910
+
911
+ def group_of(pid)
912
+ Process.getpgid(pid)
913
+ rescue SystemCallError
914
+ nil
915
+ end
916
+
917
+ # The HTTP status a JSON-RPC ping to +uri+ answers with, carrying
918
+ # +token+ as its bearer when given; nil while nothing answers.
919
+ def rpc_status(uri, token)
920
+ request = Net::HTTP::Post.new(uri, "Content-Type" => "application/json", "Accept" => "application/json, text/event-stream")
921
+ request["Authorization"] = "Bearer #{token}" if token
922
+ request.body = JSON.generate(jsonrpc: "2.0", id: "readiness", method: "ping")
923
+ http = Net::HTTP.new(uri.host, uri.port, nil)
924
+ http.open_timeout = 1
925
+ http.read_timeout = 2
926
+ http.start { |connection| connection.request(request).code.to_i }
927
+ rescue SystemCallError, IOError, Timeout::Error, Net::HTTPBadResponse
928
+ nil
929
+ end
930
+
931
+ # The HTTP status a GET on the MCP path answers with, or nil while
932
+ # nothing answers.
933
+ def probe(uri)
934
+ # No proxy: the dashboard's HTTP(S)_PROXY does not know this loopback.
935
+ http = Net::HTTP.new(uri.host, uri.port, nil)
936
+ http.open_timeout = 1
937
+ http.read_timeout = 2
938
+ http.start { |connection| connection.request(Net::HTTP::Get.new(uri, "Accept" => "application/json")).code.to_i }
939
+ rescue SystemCallError, IOError, Timeout::Error, Net::HTTPBadResponse
940
+ nil
941
+ end
942
+
943
+ def free_port
944
+ server = TCPServer.new("127.0.0.1", 0)
945
+ server.addr[1]
946
+ ensure
947
+ server&.close
948
+ end
949
+
950
+ # Raises the boot failure: the step, what went wrong, and the end of
951
+ # that step's log, all scrubbed of the sandbox's secrets.
952
+ def fail_step!(step, problem, log, secrets)
953
+ tail = log_tail(log, secrets)
954
+ message = "Sandbox #{step} failed: #{problem}"
955
+ message += "\n--- last lines of logs/#{File.basename(log)} ---\n#{tail}" if tail.present?
956
+ raise Error, SecretScrubber.scrub(message, secrets)
957
+ end
958
+
959
+ def log_tail(log, secrets)
960
+ return "" unless File.file?(log)
961
+
962
+ data = File.open(log, "rb") do |file|
963
+ file.seek([ file.size - LOG_TAIL_BYTES, 0 ].max)
964
+ file.read
965
+ end
966
+ lines = data.force_encoding(Encoding::UTF_8).scrub.lines.last(LOG_TAIL_LINES)
967
+ SecretScrubber.scrub(lines.map { |line| line.chomp.truncate(500) }.join("\n"), secrets)
968
+ end
969
+
970
+ # --- Claude Code --------------------------------------------------------
971
+
972
+ def claude_argv(code_session)
973
+ command = ActionAgent.claude_code_command.to_s
974
+ argv = [
975
+ command, "-p", "--output-format", "stream-json", "--verbose",
976
+ "--permission-mode", ActionAgent.claude_code_permission_mode.to_s, "--no-session-persistence"
977
+ ]
978
+ # Nobody is there to answer a permission prompt: with this, anything
979
+ # that would prompt is denied instead of stalling the session.
980
+ argv += [ "--permission-prompts", "none" ] if permission_prompts_supported?(command)
981
+ argv += [ "--max-turns", ActionAgent.claude_code_max_turns.to_i.to_s ] if ActionAgent.claude_code_max_turns.present?
982
+
983
+ if (model = code_session.model.presence)
984
+ # One argv entry, but one that must not read as a flag.
985
+ raise Error, "#{model.inspect} is not a model name" unless MODEL_NAME.match?(model.to_s)
986
+
987
+ argv += [ "--model", model.to_s ]
988
+ end
989
+ argv
990
+ end
991
+
992
+ def permission_prompts_supported?(command)
993
+ self.class.permission_prompts_supported?(command) do
994
+ output, status = capture(self.class.sanitized_environment, [ command, "--help" ],
995
+ chdir: Dir.tmpdir, limit: 256 * 1024, timeout: HELP_TIMEOUT, err: [ :child, :out ])
996
+ status ? output : nil
997
+ rescue SystemCallError
998
+ nil
999
+ end
1000
+ end
1001
+
1002
+ def read_claude_login_status(command)
1003
+ output, status = capture(self.class.sanitized_environment, [ command, "auth", "status", "--json" ],
1004
+ chdir: Dir.tmpdir, limit: 64 * 1024, timeout: LOGIN_STATUS_TIMEOUT)
1005
+ # A logged-out CLI may exit non-zero and still say so; one stopped at
1006
+ # the timeout said nothing to trust.
1007
+ status.nil? ? LOGGED_OUT : self.class.parse_login_status(output)
1008
+ rescue SystemCallError
1009
+ LOGGED_OUT
1010
+ end
1011
+
1012
+ def run_claude(workspace, code_session, argv, env, secrets, &on_event)
1013
+ key = code_session.id.to_s
1014
+ log = log_path(workspace, "claude-#{key}")
1015
+ deadline = deadline_after(ActionAgent.claude_code_timeout)
1016
+ # Checked again once Claude Code is recorded; this saves starting it.
1017
+ state = read_state(workspace)
1018
+ refuse_stopped_session!(state, key)
1019
+ # A cancelled session frees its slot as soon as it is marked cancelled,
1020
+ # while its Claude Code may still be exiting (and diffing). Two in one
1021
+ # checkout would edit the same files.
1022
+ if other_session_running?(state, key, workspace.basename.to_s)
1023
+ raise Error, "The previous Claude Code session in this sandbox is still stopping; try again in a moment"
1024
+ end
1025
+ stdin_read, stdin_write = IO.pipe
1026
+ stdout_read, stdout_write = IO.pipe
1027
+ stderr_read, stderr_write = IO.pipe
1028
+
1029
+ begin
1030
+ pid = spawn_group(env, *argv, chdir: workspace.join("app"), in: stdin_read, out: stdout_write, err: stderr_write)
1031
+ rescue SystemCallError => e
1032
+ raise Error, "Could not start Claude Code (#{argv.first}): #{e.message}"
1033
+ ensure
1034
+ [ stdin_read, stdout_write, stderr_write ].each(&:close)
1035
+ end
1036
+ waiter = Process.detach(pid)
1037
+ # When a cancel was first seen here (monotonic): from then on the
1038
+ # session gets STOP_GRACE to end on SIGTERM, then SIGKILL.
1039
+ cancel_seen = nil
1040
+ case record_code_session(workspace, key, pid)
1041
+ when :terminating
1042
+ # Stopped by the ensure below, before it had the prompt.
1043
+ raise Error, "The sandbox is being stopped, so Claude Code did not run"
1044
+ when :cancelled
1045
+ # Runs its course like any cancelled session: it ends on SIGTERM.
1046
+ signal_group(pid, "TERM")
1047
+ cancel_seen = monotonic
1048
+ end
1049
+
1050
+ # The prompt goes in on stdin, never argv, where `ps` would show it.
1051
+ writer = background { write_prompt(stdin_write, code_session.prompt) }
1052
+ stderr_tail = []
1053
+ reader = background { copy_stderr(stderr_read, log, secrets, stderr_tail) }
1054
+
1055
+ # A cancel sent from another call (or process) is read from
1056
+ # state.json, where cancel_code_session notes it.
1057
+ next_check = monotonic
1058
+ stop_by = lambda do
1059
+ if cancel_seen.nil? && monotonic >= next_check
1060
+ next_check = monotonic + CANCEL_CHECK_INTERVAL
1061
+ cancel_seen = cancel_noted(workspace, key)
1062
+ end
1063
+ cancel_seen ? [ deadline, cancel_seen + stop_grace ].min : deadline
1064
+ end
1065
+
1066
+ finished = stream_events(stdout_read, waiter, stop_by, secrets, &on_event) && waiter.join(time_left(stop_by.call))
1067
+ unless finished
1068
+ # Cancelled, and SIGTERM did not end it within the grace: SIGKILL,
1069
+ # and the session finishes like any cancelled one.
1070
+ raise Error, "Claude Code did not finish within #{ActionAgent.claude_code_timeout}s and was stopped" unless cancel_seen
1071
+
1072
+ stop_groups([ pid ], grace: 0)
1073
+ end
1074
+
1075
+ # Whatever the session left running in the background goes too.
1076
+ stop_groups([ pid ], grace: 1)
1077
+ reader.join(OUTPUT_DRAIN_GRACE)
1078
+
1079
+ {
1080
+ exit_status: waiter.join(OUTPUT_DRAIN_GRACE) ? exit_code(waiter.value) : 128 + Signal.list.fetch("KILL"),
1081
+ diff: capture_diff(workspace, secrets),
1082
+ stderr_tail: SecretScrubber.scrub(stderr_tail.join("\n"), secrets)
1083
+ }
1084
+ ensure
1085
+ stop_groups([ pid ], grace: cancel_seen ? 0 : stop_grace) if pid
1086
+ [ stdin_write, stdout_read, stderr_read ].each { |io| io&.close unless io&.closed? }
1087
+ writer&.join(1)
1088
+ reader&.join(1)
1089
+ forget_code_session(workspace, key)
1090
+ end
1091
+
1092
+ # Why a session must not start: a terminate under way, or a cancel that
1093
+ # came before there was a process to stop.
1094
+ def refuse_stopped_session!(state, key)
1095
+ raise Error, "The sandbox is being stopped, so Claude Code did not run" if state["terminating"]
1096
+
1097
+ cancels = state["cancelled_code_sessions"]
1098
+ raise Error, "Claude Code session #{key} was cancelled before it started" if cancels.is_a?(Hash) && cancels.key?(key)
1099
+ end
1100
+
1101
+ # Records Claude Code's pid (and start time) under the state.json lock,
1102
+ # the one cancel_code_session and terminate take too, so each either
1103
+ # finds this pid or left a mark here first. Returns :terminating when a
1104
+ # terminate is under way (or already removed the workspace), :cancelled
1105
+ # when a cancel came first, and nil otherwise.
1106
+ def other_session_running?(state, key, session_id)
1107
+ sessions = state["code_sessions"].is_a?(Hash) ? state["code_sessions"] : {}
1108
+ sessions.any? do |other, pid|
1109
+ other != key && pid.is_a?(Integer) && group_alive?(pid) && recorded_group?(pid, session_id, state)
1110
+ end
1111
+ end
1112
+
1113
+ def record_code_session(workspace, key, pid)
1114
+ mark = nil
1115
+ update_state(workspace) do |state|
1116
+ state_hash(state, "code_sessions")[key] = pid
1117
+ record_process_start(state, pid)
1118
+ mark =
1119
+ if state["terminating"] then :terminating
1120
+ elsif state_hash(state, "cancelled_code_sessions").delete(key) then :cancelled
1121
+ end
1122
+ end
1123
+ mark
1124
+ rescue Errno::ENOENT
1125
+ :terminating
1126
+ end
1127
+
1128
+ # When a cancel_code_session for session +key+ signalled it, as a
1129
+ # monotonic time; nil while none has.
1130
+ def cancel_noted(workspace, key)
1131
+ at = read_state(workspace).dig("cancelling_code_sessions", key)
1132
+ at.is_a?(Numeric) ? monotonic - (Time.now.to_f - at).clamp(0, Float::INFINITY) : nil
1133
+ rescue SystemCallError
1134
+ nil
1135
+ end
1136
+
1137
+ # Reads stream-json from +io+ until it closes, yielding each line as an
1138
+ # event. Returns false when the deadline +stop_by+ answers (it can move
1139
+ # earlier, on a cancel) passed first.
1140
+ def stream_events(io, waiter, stop_by, secrets, &on_event)
1141
+ buffer = String.new(encoding: Encoding::BINARY)
1142
+ skipping = false
1143
+ exited_at = nil
1144
+
1145
+ loop do
1146
+ deadline = stop_by.call
1147
+ return false if monotonic >= deadline
1148
+
1149
+ # Claude Code exited but something it started still holds stdout.
1150
+ unless waiter.alive?
1151
+ exited_at ||= monotonic
1152
+ break if monotonic - exited_at > OUTPUT_DRAIN_GRACE
1153
+ end
1154
+
1155
+ next unless io.wait_readable([ POLL_INTERVAL, time_left(deadline) || POLL_INTERVAL ].min)
1156
+
1157
+ chunk = io.read_nonblock(64 * 1024, exception: false)
1158
+ break if chunk.nil?
1159
+ next if chunk == :wait_readable
1160
+
1161
+ buffer << chunk
1162
+ while (newline = buffer.index("\n"))
1163
+ line = buffer.slice!(0, newline + 1)
1164
+ if skipping
1165
+ skipping = false
1166
+ else
1167
+ emit_event(line, secrets, &on_event)
1168
+ end
1169
+ end
1170
+
1171
+ next unless buffer.bytesize > MAX_EVENT_LINE_BYTES
1172
+
1173
+ unless skipping
1174
+ head = buffer.byteslice(0, CodeSession::MAX_EVENT_STRING).force_encoding(Encoding::UTF_8).scrub
1175
+ emit_event("#{head}… (line truncated)", secrets, &on_event)
1176
+ end
1177
+ buffer.clear
1178
+ skipping = true
1179
+ end
1180
+
1181
+ emit_event(buffer, secrets, &on_event) unless skipping || buffer.empty?
1182
+ true
1183
+ end
1184
+
1185
+ def emit_event(line, secrets)
1186
+ text = line.dup.force_encoding(Encoding::UTF_8).scrub.chomp
1187
+ return if text.strip.empty?
1188
+
1189
+ event = begin
1190
+ parsed = JSON.parse(text)
1191
+ parsed.is_a?(Hash) ? parsed : nil
1192
+ rescue JSON::ParserError
1193
+ nil
1194
+ end
1195
+ yield SecretScrubber.scrub(event || { "type" => "raw", "text" => text }, secrets) if block_given?
1196
+ end
1197
+
1198
+ def write_prompt(io, prompt)
1199
+ io.write(prompt.to_s)
1200
+ rescue IOError, SystemCallError
1201
+ # The CLI exited without reading it; its exit status says why.
1202
+ ensure
1203
+ io.close unless io.closed?
1204
+ end
1205
+
1206
+ def copy_stderr(io, log, secrets, tail)
1207
+ File.open(log, "a") do |file|
1208
+ io.each_line(64 * 1024) do |line|
1209
+ line = SecretScrubber.scrub(line.scrub, secrets)
1210
+ file.write(line)
1211
+ file.flush
1212
+ tail << line.chomp.truncate(500)
1213
+ tail.shift while tail.size > LOG_TAIL_LINES
1214
+ end
1215
+ end
1216
+ rescue IOError
1217
+ # Closed under us once the session is over.
1218
+ end
1219
+
1220
+ # The checkout's changes since it was fetched: this session's, and any
1221
+ # earlier session's. Untracked files are marked intent-to-add so new
1222
+ # files show up alongside edits. `add --all` also stages removals, which
1223
+ # a plain `git diff` (worktree against index) would then leave out, so
1224
+ # the worktree is compared with the commit checked out instead; that also
1225
+ # keeps changes the session committed itself.
1226
+ def capture_diff(workspace, secrets)
1227
+ app = workspace.join("app")
1228
+ env = self.class.sanitized_environment
1229
+ git = git_command(app)
1230
+ # A filter driver in the checkout's git config (which the session could
1231
+ # have written) runs its command on `git add` and `git diff`, as the
1232
+ # dashboard's user. Rather than run it, report no diff.
1233
+ drivers, = capture(env, [ *git, "config", "--local", "--includes", "--name-only", "--get-regexp", "^filter\\." ],
1234
+ chdir: app, limit: 64 * 1024, timeout: GIT_TIMEOUT)
1235
+ if drivers.to_s.strip.present?
1236
+ return "(diff not recorded: the checkout's git config defines filter drivers, which would run commands)"
1237
+ end
1238
+ capture(env, [ *git, "add", "--intent-to-add", "--all" ], chdir: app, limit: 64 * 1024, timeout: GIT_TIMEOUT)
1239
+
1240
+ base = read_state(workspace)["checkout_commit"]
1241
+ base = "HEAD" unless base.is_a?(String) && COMMIT_ID.match?(base)
1242
+ diff = ->(commit) do
1243
+ capture(env, [ *git, "diff", "--no-color", "--no-ext-diff", "--no-textconv", commit, "--" ],
1244
+ chdir: app, limit: MAX_DIFF_BYTES, timeout: GIT_TIMEOUT)
1245
+ end
1246
+ output, status = diff.call(base)
1247
+ # The session pruned the commit away (a gc after moving HEAD, say).
1248
+ output, = diff.call("HEAD") if base != "HEAD" && output.empty? && status && !status.success?
1249
+
1250
+ SecretScrubber.scrub(output.force_encoding(Encoding::UTF_8).scrub, secrets)
1251
+ rescue SystemCallError => e
1252
+ Rails.logger.warn("[ActionAgent] could not diff the sandbox checkout: #{e.message}")
1253
+ ""
1254
+ end
1255
+
1256
+ # git in the checkout. A Claude Code session could have edited
1257
+ # .git/config: whatever hooks or filesystem monitor it set up there do
1258
+ # not run here.
1259
+ def git_command(app)
1260
+ [ "git", "-C", app.to_s, "-c", "core.fsmonitor=false", "-c", "core.hooksPath=/dev/null" ]
1261
+ end
1262
+
1263
+ # The Claude Code variables a session runs with. An API key comes from
1264
+ # the owner's connection. The machine's own login needs none: `claude`
1265
+ # finds it itself, and the dashboard neither reads nor passes it on.
1266
+ def session_credentials(sandbox)
1267
+ return {} if ClaudeCodeAuth.local_login?
1268
+
1269
+ credentials = sandbox.runtime_environment.to_h
1270
+ if credentials.empty?
1271
+ raise Error, "Claude Code is not connected: connect an Anthropic API key in Settings → Integrations"
1272
+ end
1273
+
1274
+ credentials
1275
+ end
1276
+
1277
+ def sandbox_secrets(sandbox, credentials)
1278
+ token = begin
1279
+ sandbox.checkout_spec&.dig(:token)
1280
+ rescue StandardError
1281
+ nil
1282
+ end
1283
+ [ token, *credentials.values ].compact.map(&:to_s)
1284
+ end
1285
+
1286
+ # Drops the session's pid, and the cancels it may have left.
1287
+ def forget_code_session(workspace, key)
1288
+ update_state(workspace, create: false) do |state|
1289
+ pid = state_hash(state, "code_sessions").delete(key)
1290
+ state_hash(state, "process_starts").delete(pid.to_s) if pid
1291
+ state_hash(state, "cancelled_code_sessions").delete(key)
1292
+ state_hash(state, "cancelling_code_sessions").delete(key)
1293
+ end
1294
+ rescue Errno::ENOENT
1295
+ # Terminated meanwhile: the workspace, and its state, are gone.
1296
+ end
1297
+
1298
+ def exit_code(status)
1299
+ status.exitstatus || (status.termsig ? 128 + status.termsig : 1)
1300
+ end
1301
+
1302
+ # --- Processes ----------------------------------------------------------
1303
+
1304
+ # Every sandbox process leads its own process group, so stopping one
1305
+ # reaches whatever it started, and gets exactly the environment given.
1306
+ def spawn_group(env, *argv, chdir:, **redirects)
1307
+ Process.spawn(env, *argv, chdir: chdir.to_s, pgroup: true, unsetenv_others: true, close_others: true, **redirects)
1308
+ end
1309
+
1310
+ # Runs +argv+ to completion, bounded in time and output. Returns the
1311
+ # output and the exit status (nil when it had to be stopped).
1312
+ def capture(env, argv, chdir:, limit:, timeout:, err: File::NULL)
1313
+ output = String.new(encoding: Encoding::BINARY)
1314
+ reader, writer = IO.pipe
1315
+ pid = spawn_group(env, *argv, chdir: chdir, in: File::NULL, out: writer, err: err)
1316
+ writer.close
1317
+ waiter = Process.detach(pid)
1318
+ deadline = deadline_after(timeout)
1319
+
1320
+ while output.bytesize < limit && monotonic < deadline
1321
+ next unless reader.wait_readable([ POLL_INTERVAL, time_left(deadline) || POLL_INTERVAL ].min)
1322
+
1323
+ chunk = reader.read_nonblock(64 * 1024, exception: false)
1324
+ break if chunk.nil?
1325
+
1326
+ output << chunk.byteslice(0, limit - output.bytesize) unless chunk == :wait_readable
1327
+ end
1328
+
1329
+ [ output, waiter.join(output.bytesize < limit ? time_left(deadline) : 0) && waiter.value ]
1330
+ ensure
1331
+ stop_groups([ pid ], grace: 1) if pid
1332
+ [ reader, writer ].each { |io| io&.close unless io&.closed? }
1333
+ end
1334
+
1335
+ # TERM to each group, KILL to whatever is left after +grace+ (STOP_GRACE
1336
+ # by default), then a moment for the kernel to finish them off.
1337
+ def stop_groups(pids, grace: nil)
1338
+ grace ||= stop_grace
1339
+ live = pids.select { |pid| group_alive?(pid) }
1340
+ return if live.empty?
1341
+
1342
+ live.each { |pid| signal_group(pid, "TERM") }
1343
+ # Not deadline_after, for which 0 means no limit: here it means none.
1344
+ deadline = monotonic + grace
1345
+ sleep 0.1 while live.any? { |pid| group_alive?(pid) } && monotonic < deadline
1346
+
1347
+ live.each { |pid| signal_group(pid, "KILL") if group_alive?(pid) }
1348
+ deadline = deadline_after(2)
1349
+ sleep 0.05 while live.any? { |pid| group_alive?(pid) } && monotonic < deadline
1350
+ end
1351
+
1352
+ # A method, not the constant alone, so the tests can shorten it.
1353
+ def stop_grace
1354
+ STOP_GRACE
1355
+ end
1356
+
1357
+ def signal_group(pgid, signal)
1358
+ return false unless signalable?(pgid)
1359
+
1360
+ Process.kill(signal, -pgid)
1361
+ true
1362
+ rescue Errno::ESRCH, Errno::EPERM
1363
+ false
1364
+ end
1365
+
1366
+ # Never 0 or 1 (kill(-1) signals every process the user owns) and never
1367
+ # the dashboard's own group.
1368
+ def signalable?(pgid)
1369
+ pgid.is_a?(Integer) && pgid > 1 && pgid != Process.getpgrp
1370
+ end
1371
+
1372
+ # Whether any live process remains in group +pgid+. Where /proc exists it
1373
+ # is read directly: kill(0) also succeeds for zombies, which an init that
1374
+ # does not reap (a container's) keeps around indefinitely.
1375
+ def group_alive?(pgid)
1376
+ return false unless signalable?(pgid)
1377
+ return proc_group_alive?(pgid) if procfs?
1378
+
1379
+ Process.kill(0, -pgid)
1380
+ true
1381
+ rescue Errno::ESRCH
1382
+ false
1383
+ rescue Errno::EPERM
1384
+ true
1385
+ end
1386
+
1387
+ def proc_group_alive?(pgid)
1388
+ Dir.each_child("/proc").any? do |entry|
1389
+ next false unless entry.match?(/\A\d+\z/)
1390
+
1391
+ state, _ppid, group = proc_stat(entry)
1392
+ group.to_i == pgid && !%w[Z X].include?(state)
1393
+ end
1394
+ end
1395
+
1396
+ # The fields of /proc/<pid>/stat after the command name, which may itself
1397
+ # contain spaces and parentheses.
1398
+ def proc_stat(pid)
1399
+ stat = File.read("/proc/#{pid}/stat")
1400
+ stat[(stat.rindex(")") + 2)..].split
1401
+ rescue SystemCallError
1402
+ []
1403
+ end
1404
+
1405
+ def procfs?
1406
+ File.exist?("/proc/self/stat")
1407
+ end
1408
+
1409
+ # When process +pid+ started, in clock ticks since boot: what tells it
1410
+ # apart from a later process that reused its pid. Unlike its environment
1411
+ # the process cannot rewrite it, as a long enough process title (Ruby's
1412
+ # `$0=`, setproctitle) does. nil where /proc cannot say.
1413
+ def process_start(pid)
1414
+ # Field 22 of the stat line; proc_stat starts at field 3.
1415
+ return Integer(proc_stat(pid)[19], exception: false) if procfs?
1416
+
1417
+ # No /proc (macOS): ps reports the start time to the second, enough to
1418
+ # tell a reused pid apart. nil when the process is gone. Written in the
1419
+ # local time zone and language, so pinned to UTC and C: a dashboard
1420
+ # restarted under another TZ would otherwise read every recorded
1421
+ # process as a stranger.
1422
+ output, status = Open3.capture2(PS_ENVIRONMENT, "ps", "-o", "lstart=", "-p", pid.to_s)
1423
+ status.success? ? output.strip.presence : nil
1424
+ rescue SystemCallError
1425
+ nil
1426
+ end
1427
+
1428
+ def record_process_start(state, pid)
1429
+ started = process_start(pid)
1430
+ state_hash(state, "process_starts")[pid.to_s] = started if started
1431
+ end
1432
+
1433
+ # Whether +pid+, read from state.json, may be signalled as this
1434
+ # sandbox's process group (see #group_identity).
1435
+ def recorded_group?(pid, session_id, state)
1436
+ group_identity(pid, session_id, state) == :ours
1437
+ end
1438
+
1439
+ # What +pid+, read from state.json, is now: :ours, :stranger or
1440
+ # :unknown. A pid is reused once its process is gone, so where the
1441
+ # process is there it must be the one recorded: started when state.json
1442
+ # says, or, for a pid recorded without a start time, carrying this
1443
+ # sandbox's session id in its environment. Where it is gone there is
1444
+ # nothing to confuse: a group id is not reused while any process of the
1445
+ # group lives. :unknown when nothing can tell (no start time recorded,
1446
+ # and no /proc to read its environment, or no permission to read it):
1447
+ # such a pid is never signalled, and never forgotten either.
1448
+ def group_identity(pid, session_id, state)
1449
+ return :stranger unless signalable?(pid)
1450
+
1451
+ started = state["process_starts"][pid.to_s] if state["process_starts"].is_a?(Hash)
1452
+ if started
1453
+ current = process_start(pid)
1454
+ return current.nil? || current == started ? :ours : :stranger
1455
+ end
1456
+ return :unknown unless procfs?
1457
+
1458
+ environ = File.binread("/proc/#{pid}/environ")
1459
+ if environ.empty?
1460
+ # Zombies have no environment and still own their pid. A live
1461
+ # process can also have an empty environment, including during exec:
1462
+ # without a recorded start time it cannot be identified safely.
1463
+ state, = proc_stat(pid)
1464
+ return %w[Z X].include?(state) ? :ours : :unknown
1465
+ end
1466
+ environ.split("\0").include?("#{SESSION_ID_ENV}=#{session_id}") ? :ours : :stranger
1467
+ rescue Errno::ENOENT, Errno::ESRCH
1468
+ :ours
1469
+ rescue Errno::EACCES, Errno::EPERM
1470
+ :unknown
1471
+ end
1472
+
1473
+ # Stops everything the workspace recorded, then removes it. The
1474
+ # terminating mark goes in under the same lock that reads the pids: a
1475
+ # Claude Code session that records itself later finds the mark and stops
1476
+ # on its own (see #record_code_session), and one that recorded itself
1477
+ # earlier is among the pids stopped here.
1478
+ #
1479
+ # Returns false, keeping the workspace, when a recorded group is still
1480
+ # alive afterwards and is not known to be a stranger's: one that could
1481
+ # not be stopped, or not told apart from an unrelated process. Removing
1482
+ # state.json would drop the only record of it; kept, the next terminate
1483
+ # tries again.
1484
+ def discard(session_id)
1485
+ workspace = workspace_for(session_id)
1486
+ return true unless workspace.exist?
1487
+
1488
+ state = begin
1489
+ update_state(workspace) { |current| current["terminating"] = true }
1490
+ rescue Errno::ENOENT, Errno::ENOTDIR
1491
+ {}
1492
+ end
1493
+ sessions = state["code_sessions"].is_a?(Hash) ? state["code_sessions"].values : []
1494
+ recorded = [ state["pid"], state["step_pid"], *sessions ].uniq.select { |pid| signalable?(pid) }
1495
+ identities = recorded.index_with { |pid| group_identity(pid, session_id, state) }
1496
+ stop_groups(recorded.select { |pid| identities[pid] == :ours })
1497
+ stop_escaped(session_id)
1498
+
1499
+ left = recorded.select { |pid| identities[pid] != :stranger && group_alive?(pid) }
1500
+ if left.any?
1501
+ Rails.logger.error("[ActionAgent] sandbox #{session_id}: process groups #{left.join(", ")} are still alive and " \
1502
+ "could not be #{left.any? { |pid| identities[pid] == :unknown } ? "identified" : "stopped"}; " \
1503
+ "keeping #{workspace} so a later terminate can try again")
1504
+ return false
1505
+ end
1506
+
1507
+ # After the server is gone: PostgreSQL refuses to drop a database
1508
+ # while anything is connected to it.
1509
+ drop_databases(workspace)
1510
+ remove_workspace(workspace)
1511
+ true
1512
+ end
1513
+
1514
+ # Processes of this sandbox that left its process groups (setsid, a
1515
+ # daemonizing server) and so escaped #stop_groups, found where /proc
1516
+ # shows every process's environment: exactly this sandbox's session id,
1517
+ # never this process or its group. One that also rewrote its
1518
+ # environment (a long process title) is not found.
1519
+ def stop_escaped(session_id)
1520
+ return unless procfs?
1521
+
1522
+ marker = "#{SESSION_ID_ENV}=#{session_id}"
1523
+ own_group = Process.getpgrp
1524
+ escaped = Dir.children("/proc").filter_map do |entry|
1525
+ next unless entry.match?(/\A\d+\z/)
1526
+
1527
+ pid = entry.to_i
1528
+ next if pid == Process.pid
1529
+
1530
+ state, _ppid, group = proc_stat(entry)
1531
+ next if state.nil? || %w[Z X].include?(state) || group.to_i == own_group
1532
+
1533
+ pid if File.binread("/proc/#{pid}/environ").split("\0").include?(marker)
1534
+ rescue SystemCallError
1535
+ nil
1536
+ end
1537
+ return if escaped.empty?
1538
+
1539
+ escaped.each { |pid| signal_process(pid, "TERM") }
1540
+ deadline = deadline_after(stop_grace)
1541
+ sleep 0.1 while escaped.any? { |pid| process_alive?(pid) } && monotonic < deadline
1542
+ escaped.each { |pid| signal_process(pid, "KILL") if process_alive?(pid) }
1543
+ end
1544
+
1545
+ def signal_process(pid, signal)
1546
+ return false unless pid.is_a?(Integer) && pid > 1 && pid != Process.pid
1547
+
1548
+ Process.kill(signal, pid)
1549
+ true
1550
+ rescue Errno::ESRCH, Errno::EPERM
1551
+ false
1552
+ end
1553
+
1554
+ def process_alive?(pid)
1555
+ state, = proc_stat(pid)
1556
+ !state.nil? && !%w[Z X].include?(state)
1557
+ end
1558
+
1559
+ # Moved aside before it is deleted: a Claude Code session finishing in
1560
+ # this process may still be diffing or updating state.json there, and
1561
+ # has to find the workspace gone rather than recreate parts of it.
1562
+ def remove_workspace(workspace)
1563
+ return unless workspace.exist?
1564
+
1565
+ doomed = workspace.dirname.join(".#{workspace.basename}.removed-#{SecureRandom.hex(4)}")
1566
+ File.rename(workspace, doomed)
1567
+ 3.times do
1568
+ FileUtils.rm_rf(doomed)
1569
+ break unless doomed.exist?
1570
+
1571
+ sleep 0.2
1572
+ end
1573
+ rescue Errno::ENOENT
1574
+ # Removed meanwhile.
1575
+ end
1576
+
1577
+ def background(&block)
1578
+ Thread.new(&block).tap { |thread| thread.report_on_exception = false }
1579
+ end
1580
+
1581
+ def describe(status)
1582
+ status.exitstatus ? "status #{status.exitstatus}" : "signal #{status.termsig}"
1583
+ end
1584
+
1585
+ # --- Workspace ----------------------------------------------------------
1586
+
1587
+ def workspace_for(session_id)
1588
+ ActionAgent.local_sandbox_root.join(session_id)
1589
+ end
1590
+
1591
+ def log_path(workspace, name)
1592
+ workspace.join("logs", "#{name}.log")
1593
+ end
1594
+
1595
+ def session_id!(value)
1596
+ value = value.to_s
1597
+ raise Error, "#{value.inspect} is not a sandbox session id" unless SESSION_ID.match?(value)
1598
+
1599
+ value
1600
+ end
1601
+
1602
+ # The session id in a handle this backend issued, or nil for anything
1603
+ # else.
1604
+ def session_id_from(handle)
1605
+ handle = handle.to_s
1606
+ return nil unless handle.start_with?(HANDLE_PREFIX)
1607
+
1608
+ session_id = handle.delete_prefix(HANDLE_PREFIX)
1609
+ SESSION_ID.match?(session_id) ? session_id : nil
1610
+ end
1611
+
1612
+ # No lock needed: state.json is only ever replaced whole (see
1613
+ # #update_state), so a read sees one version or the next.
1614
+ def read_state(workspace)
1615
+ parse_state(File.read(workspace.join("state.json")))
1616
+ rescue Errno::ENOENT, Errno::ENOTDIR
1617
+ {}
1618
+ end
1619
+
1620
+ # Read-modify-write under an exclusive lock: a Claude Code session
1621
+ # records its pid while terminate may be reading the same file.
1622
+ #
1623
+ # The new state goes to a temporary file in the workspace, which is then
1624
+ # renamed over state.json: a crash mid-write leaves the old version, not
1625
+ # a truncated one. The lock is on state.lock, which is never replaced; a
1626
+ # lock on state.json itself would stay on the inode the rename
1627
+ # unlinked, and cover nothing. Raises Errno::ENOENT when there is no
1628
+ # state.json and +create+ is false, or no workspace at all.
1629
+ def update_state(workspace, create: true)
1630
+ path = workspace.join("state.json")
1631
+ File.open(workspace.join("state.lock"), File::RDWR | File::CREAT, 0o600) do |lock|
1632
+ lock.flock(File::LOCK_EX)
1633
+ current = begin
1634
+ File.read(path)
1635
+ rescue Errno::ENOENT
1636
+ raise unless create
1637
+
1638
+ nil
1639
+ end
1640
+ state = parse_state(current)
1641
+ yield state
1642
+
1643
+ temporary = workspace.join(".state.json.#{SecureRandom.hex(4)}")
1644
+ begin
1645
+ File.open(temporary, File::WRONLY | File::CREAT | File::EXCL, 0o600) do |file|
1646
+ file.write(JSON.generate(state))
1647
+ file.flush
1648
+ file.fsync
1649
+ end
1650
+ File.rename(temporary, path)
1651
+ ensure
1652
+ FileUtils.rm_f(temporary)
1653
+ end
1654
+ state
1655
+ end
1656
+ end
1657
+
1658
+ def parse_state(json)
1659
+ state = JSON.parse(json.presence || "{}")
1660
+ state.is_a?(Hash) ? state : {}
1661
+ rescue JSON::ParserError
1662
+ {}
1663
+ end
1664
+
1665
+ # state[key] as a Hash, in place of whatever a damaged file held there.
1666
+ def state_hash(state, key)
1667
+ state[key] = {} unless state[key].is_a?(Hash)
1668
+ state[key]
1669
+ end
1670
+
1671
+ # --- Time ---------------------------------------------------------------
1672
+
1673
+ def monotonic
1674
+ Process.clock_gettime(Process::CLOCK_MONOTONIC)
1675
+ end
1676
+
1677
+ # No (or a non-positive) timeout means none.
1678
+ def deadline_after(seconds)
1679
+ seconds = seconds.to_f
1680
+ seconds.positive? ? monotonic + seconds : Float::INFINITY
1681
+ end
1682
+
1683
+ # Seconds until +deadline+, or nil for no limit (what Thread#join and
1684
+ # IO#wait_readable take for "forever").
1685
+ def time_left(deadline)
1686
+ deadline.infinite? ? nil : [ deadline - monotonic, 0 ].max
1687
+ end
1688
+ end
1689
+ end