actionagent 1.7.2 → 1.8.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (59) hide show
  1. checksums.yaml +4 -4
  2. data/app/assets/builds/action_agent.css +1 -1
  3. data/app/assets/builds/action_agent.js +59 -54
  4. data/app/controllers/action_agent/api/agents_controller.rb +19 -3
  5. data/app/controllers/action_agent/api/code_sessions_controller.rb +156 -0
  6. data/app/controllers/action_agent/api/evaluations_controller.rb +27 -130
  7. data/app/controllers/action_agent/api/github_connections_controller.rb +147 -0
  8. data/app/controllers/action_agent/api/mcp_controller.rb +38 -5
  9. data/app/controllers/action_agent/api/mcp_servers_controller.rb +10 -1
  10. data/app/controllers/action_agent/api/provider_keys_controller.rb +54 -3
  11. data/app/controllers/action_agent/api/provider_models_controller.rb +10 -6
  12. data/app/controllers/action_agent/api/sandboxes_controller.rb +97 -6
  13. data/app/controllers/concerns/action_agent/api/evaluation_run_starting.rb +93 -0
  14. data/app/controllers/concerns/action_agent/api/mcp_dashboard_tools.rb +507 -0
  15. data/app/controllers/concerns/action_agent/api/run_sandbox.rb +65 -0
  16. data/app/jobs/action_agent/code_session_job.rb +166 -0
  17. data/app/jobs/action_agent/sandbox_cleanup_job.rb +80 -11
  18. data/app/jobs/action_agent/sandbox_provision_job.rb +122 -14
  19. data/app/jobs/action_agent/sandbox_run_job.rb +10 -3
  20. data/app/models/action_agent/agent.rb +16 -6
  21. data/app/models/action_agent/agent_run.rb +20 -1
  22. data/app/models/action_agent/code_session.rb +141 -0
  23. data/app/models/action_agent/evaluation_run.rb +23 -1
  24. data/app/models/action_agent/github_connection.rb +75 -0
  25. data/app/models/action_agent/provider_key.rb +142 -10
  26. data/app/models/action_agent/sandbox_session.rb +193 -17
  27. data/app/serializers/action_agent/evaluation_serializer.rb +118 -0
  28. data/app/serializers/action_agent/telemetry_trace_serializer.rb +10 -2
  29. data/app/services/action_agent/agent_execution_service.rb +4 -2
  30. data/app/services/action_agent/agent_tool_roster.rb +20 -9
  31. data/app/services/action_agent/claude_code_auth.rb +86 -0
  32. data/app/services/action_agent/dashboard_assistant_service.rb +47 -5
  33. data/app/services/action_agent/evaluation_tool_resolver.rb +18 -0
  34. data/app/services/action_agent/github_client.rb +111 -0
  35. data/app/services/action_agent/local_sandbox_backend.rb +1689 -0
  36. data/app/services/action_agent/local_sandbox_databases.rb +257 -0
  37. data/app/services/action_agent/mcp_client.rb +5 -1
  38. data/app/services/action_agent/mcp_tool_dispatcher.rb +137 -17
  39. data/app/services/action_agent/mock_sandbox_backend.rb +39 -0
  40. data/app/services/action_agent/ollama_host_probe.rb +75 -0
  41. data/app/services/action_agent/payload_bounds.rb +36 -0
  42. data/app/services/action_agent/sandbox_manifest.rb +67 -0
  43. data/app/services/action_agent/sandbox_orchestrator.rb +69 -14
  44. data/app/services/action_agent/scenario_evaluation_runner.rb +50 -4
  45. data/app/services/action_agent/secret_scrubber.rb +37 -0
  46. data/app/services/action_agent/tool_discovery.rb +19 -5
  47. data/config/routes.rb +22 -3
  48. data/lib/action_agent/engine.rb +1 -0
  49. data/lib/action_agent/version.rb +1 -1
  50. data/lib/action_agent.rb +147 -3
  51. data/lib/generators/action_agent/install_generator.rb +30 -3
  52. data/lib/generators/action_agent/templates/action_agent.rb.erb +44 -0
  53. data/lib/generators/action_agent/templates/add_provider_key_api_key.rb.erb +25 -0
  54. data/lib/generators/action_agent/templates/create_active_agent_code_sessions.rb.erb +59 -0
  55. data/lib/generators/action_agent/templates/create_active_agent_dashboard_tables.rb.erb +2 -0
  56. data/lib/generators/action_agent/templates/create_active_agent_github_connections.rb.erb +61 -0
  57. data/lib/tasks/claude_code.rake +16 -0
  58. data/lib/tasks/sandbox.rake +26 -0
  59. metadata +23 -1
@@ -0,0 +1,67 @@
1
+ # frozen_string_literal: true
2
+
3
+ module ActionAgent
4
+ # The runtime manifest: how a booted checkout tells the sandbox backend
5
+ # where its MCP facade answers and which bearer token opens it.
6
+ #
7
+ # { "mcp_path": "/activeagents/mcp", "mcp_token": "aa_..." }
8
+ #
9
+ # The checked-out app writes it with `bin/rails action_agent:sandbox:manifest`
10
+ # (it mounts this engine, so the task ships with it), and a backend reads it
11
+ # back with .parse. Both halves live here so they cannot drift.
12
+ module SandboxManifest
13
+ # Where the manifest task writes, when set; stdout otherwise.
14
+ PATH_ENV = "ACTION_AGENT_SANDBOX_MANIFEST"
15
+ # The dashboard API key the manifest hands out, created once per checkout.
16
+ KEY_NAME = "Checkout sandbox runtime"
17
+
18
+ class Error < StandardError; end
19
+
20
+ module_function
21
+
22
+ # Builds the manifest inside the booted app: the engine's MCP path under
23
+ # wherever the app mounts it, and an API key for the MCP facade.
24
+ #
25
+ # @param routes [ActionDispatch::Routing::RouteSet] the app's routes
26
+ # @return [Hash{String => String}]
27
+ def generate(routes: Rails.application.routes)
28
+ # Rails 8 loads routes lazily in development and test; the mount is
29
+ # invisible until they are.
30
+ Rails.application.reload_routes_unless_loaded if Rails.application.respond_to?(:reload_routes_unless_loaded)
31
+
32
+ mount = ActiveAgent::Telemetry::Configuration.new.mount_path_in(routes)
33
+ raise Error, "ActionAgent::Engine is not mounted in this app's routes" if mount.nil?
34
+
35
+ { "mcp_path" => "#{mount}/mcp", "mcp_token" => api_key.token }
36
+ end
37
+
38
+ # The checkout's own dashboard API key for the facade. Reused across
39
+ # runs so a re-run manifest does not mint a key per boot. A host that
40
+ # owns keys by account or user gets a key with no owner, which reaches
41
+ # no agents over MCP — the manifest says so rather than failing.
42
+ def api_key
43
+ ActionAgent::ApiKey.find_or_create_by!(name: KEY_NAME)
44
+ end
45
+
46
+ # Reads a manifest a checkout wrote.
47
+ #
48
+ # @param json [String]
49
+ # @return [Hash{String => String}] with "mcp_path" and "mcp_token"
50
+ def parse(json)
51
+ data = JSON.parse(json.to_s)
52
+ raise Error, "the manifest is not a JSON object" unless data.is_a?(Hash)
53
+
54
+ path = data["mcp_path"]
55
+ unless path.is_a?(String) && path.start_with?("/")
56
+ raise Error, "the manifest names no mcp_path (expected a path such as /activeagents/mcp)"
57
+ end
58
+
59
+ token = data["mcp_token"]
60
+ raise Error, "the manifest's mcp_token is not a string" unless token.nil? || token.is_a?(String)
61
+
62
+ { "mcp_path" => path, "mcp_token" => token }
63
+ rescue JSON::ParserError => e
64
+ raise Error, "the manifest is not JSON (#{e.message.truncate(120)})"
65
+ end
66
+ end
67
+ end
@@ -3,16 +3,14 @@
3
3
  module ActionAgent
4
4
  # SandboxOrchestrator
5
5
  #
6
- # Unified interface for managing agent sandbox sessions.
7
- # Supports multiple backends for cloud-agnostic deployment:
8
- #
9
- # - incus: Self-hosted Incus containers (any Linux host)
10
- # - cloud_run: Google Cloud Run Jobs (serverless)
11
- # - kubernetes: Kubernetes pods (GKE, EKS, self-hosted k8s)
6
+ # Unified interface for managing agent sandbox sessions. The engine ships
7
+ # two backends — :mock (in-memory, runs nothing) and :local (checkouts as
8
+ # child processes of the dashboard) — and a host registers the rest
9
+ # (Incus, Cloud Run, Kubernetes) in ActionAgent.sandbox_backends.
12
10
  #
13
11
  # Configuration:
14
- # Set SANDBOX_BACKEND environment variable to choose backend.
15
- # Default: "incus" for simplicity
12
+ # ActionAgent.sandbox_service, or the SANDBOX_BACKEND environment
13
+ # variable, which wins. Default: :mock.
16
14
  #
17
15
  # Usage:
18
16
  # orchestrator = SandboxOrchestrator.new
@@ -21,14 +19,21 @@ module ActionAgent
21
19
  # orchestrator.terminate(container_id)
22
20
  #
23
21
  class SandboxOrchestrator
24
- # The engine ships only the in-memory backend. Anything that talks to
25
- # real infrastructure (Incus, Kubernetes, Cloud Run) is registered by
26
- # the app that operates it, so the engine carries none of those SDKs:
22
+ # Anything that talks to real infrastructure (Incus, Kubernetes, Cloud
23
+ # Run) is registered by the app that operates it, so the engine carries
24
+ # none of those SDKs:
27
25
  #
28
26
  # ActionAgent.sandbox_backends = {
29
27
  # "cloud_run" => "CloudRunService"
30
28
  # }
31
- BUILT_IN_BACKENDS = { "mock" => "ActionAgent::MockSandboxBackend" }.freeze
29
+ #
30
+ # The engine also ships :local, which boots app_runtime checkouts as
31
+ # child processes of the dashboard itself (see LocalSandboxBackend and
32
+ # ActionAgent.local_sandboxes_enabled?).
33
+ BUILT_IN_BACKENDS = {
34
+ "mock" => "ActionAgent::MockSandboxBackend",
35
+ "local" => "ActionAgent::LocalSandboxBackend"
36
+ }.freeze
32
37
 
33
38
  # Backends disagree on what to call each verb. Candidates are tried in
34
39
  # order and the first the backend responds to wins, so a host-registered
@@ -38,7 +43,11 @@ module ActionAgent
38
43
  status: %i[status container_status pod_status job_status],
39
44
  terminate: %i[terminate terminate_pod cancel_job],
40
45
  list: %i[list_sandboxes list_sandbox_pods list_jobs],
41
- cleanup: %i[cleanup_expired cleanup_expired_pods cleanup_expired_jobs]
46
+ cleanup: %i[cleanup_expired cleanup_expired_pods cleanup_expired_jobs],
47
+ # Claude Code sessions inside an app_runtime checkout. Optional: a
48
+ # backend without them simply cannot run sessions (see #supports?).
49
+ code_session: %i[run_code_session],
50
+ cancel_code_session: %i[cancel_code_session]
42
51
  }.freeze
43
52
 
44
53
  class UnsupportedBackendError < StandardError; end
@@ -101,7 +110,12 @@ module ActionAgent
101
110
  instance_tier: result[:instance_tier] || tier&.id,
102
111
  resources: result[:resources],
103
112
  hourly_cost: result[:hourly_cost] || tier&.hourly_cost&.to_f,
104
- created_at: result[:created_at] || Time.current
113
+ created_at: result[:created_at] || Time.current,
114
+ # An app_runtime sandbox's backend clones sandbox_session.checkout_spec,
115
+ # boots the app, and reports where its MCP facade answers (and the
116
+ # bearer token it expects) so agents can use the checkout's tools.
117
+ mcp_url: result[:mcp_url],
118
+ mcp_token: result[:mcp_token]
105
119
  }
106
120
  end
107
121
 
@@ -153,6 +167,47 @@ module ActionAgent
153
167
  @backend.public_send(adapter_method(:cleanup))
154
168
  end
155
169
 
170
+ # The handle the backend would give +sandbox_session+'s sandbox, for a
171
+ # backend that derives it from the session (nil otherwise).
172
+ def handle_for(sandbox_session)
173
+ @backend.respond_to?(:handle_for) ? @backend.handle_for(sandbox_session) : nil
174
+ end
175
+
176
+ # Whether the backend can name a session's sandbox without a recorded
177
+ # handle (see #handle_for).
178
+ def derives_handles?
179
+ @backend.respond_to?(:handle_for)
180
+ end
181
+
182
+ # Whether the backend runs sandboxes as processes of the dashboard, on
183
+ # its own machine and as its own user (LocalSandboxBackend): the only
184
+ # place ActionAgent.claude_code_auth = :local_login can work.
185
+ def local?
186
+ @backend.is_a?(LocalSandboxBackend)
187
+ end
188
+
189
+ # Whether the backend implements +verb+ (an ADAPTER_METHODS key).
190
+ def supports?(verb)
191
+ ADAPTER_METHODS.fetch(verb).any? { |m| @backend.respond_to?(m) }
192
+ end
193
+
194
+ # Runs a Claude Code session in +sandbox_session+'s checkout, yielding
195
+ # each stream-json event (a Hash) as it arrives. Returns the backend's
196
+ # outcome: { exit_status:, diff: }.
197
+ def run_code_session(sandbox_session, code_session, &on_event)
198
+ # Checked when a session is requested too; this covers one queued
199
+ # before the configuration changed.
200
+ refusal = ClaudeCodeAuth.backend_refusal(self)
201
+ raise UnsupportedBackendError, refusal if refusal
202
+
203
+ @backend.public_send(adapter_method(:code_session), sandbox_session, code_session, &on_event)
204
+ end
205
+
206
+ # Stops a running Claude Code session.
207
+ def cancel_code_session(sandbox_session, code_session)
208
+ @backend.public_send(adapter_method(:cancel_code_session), sandbox_session, code_session)
209
+ end
210
+
156
211
  # Check if the backend is healthy
157
212
  #
158
213
  # @return [Boolean] true if backend is reachable
@@ -17,7 +17,16 @@ module ActionAgent
17
17
  # "_models" — per model: pass rate, mean score, latency, tokens, cost, fault counts
18
18
  # "_recommendations" — faults grouped across scenarios with the fix each calls for
19
19
  # "_verdict" — the best model and why (judge-written when a judge is available)
20
- # "_selection" — the scenarios and models this run covered
20
+ # "_selection" — the scenarios and models this run covered, and
21
+ # the checkout sandbox it replayed against, if any
22
+ #
23
+ # A selection's `sandbox_id` names a checkout sandbox whose app runtime
24
+ # every replay reaches beside the agent's own MCP servers, as if the agent
25
+ # listed it (see Api::RunSandbox, which checked the caller owns it). The
26
+ # agent is not changed. The runtime is resolved again here, among the
27
+ # agent's owner's sessions, since the run may start well after it was
28
+ # asked for; one that is no longer live fails the run rather than
29
+ # replaying without the tools it was meant to test.
21
30
  class ScenarioEvaluationRunner < EvaluationRunnerService
22
31
  Evals = ActiveAgent::Evals
23
32
 
@@ -43,6 +52,7 @@ module ActionAgent
43
52
  specs = model_specs
44
53
  run = @run || @evaluation.evaluation_runs.create!(status: :pending)
45
54
  run.update!(status: :running, selection: selection_summary(scenarios, specs))
55
+ ensure_sandbox_live!
46
56
 
47
57
  if scenarios.empty?
48
58
  run.update!(status: :failed, error_message: "No scenarios selected — add scenarios to the evaluation or widen the selection",
@@ -151,10 +161,45 @@ module ActionAgent
151
161
  "scenario_ids" => scenarios.map(&:id),
152
162
  "scenario_keys" => scenarios.map(&:key),
153
163
  "group" => @selection[:group].presence,
154
- "models" => specs.map(&:to_h)
164
+ "models" => specs.map(&:to_h),
165
+ "sandbox" => sandbox_summary
155
166
  }.compact
156
167
  end
157
168
 
169
+ # --- sandbox ----------------------------------------------------------
170
+
171
+ # "sandbox:<session_id>" for a run against a checkout sandbox, or nil.
172
+ def sandbox_server_key
173
+ id = @selection[:sandbox_id]
174
+ "#{SandboxSession::RUNTIME_SERVER_PREFIX}#{id}" if id.is_a?(String) && id.present?
175
+ end
176
+
177
+ def sandbox_session
178
+ return @sandbox_session if defined?(@sandbox_session)
179
+
180
+ @sandbox_session = sandbox_server_key && SandboxSession.for_owner(owner).find_by(session_id: @selection[:sandbox_id])
181
+ end
182
+
183
+ # What the run records about its sandbox: which checkout, never its
184
+ # token.
185
+ def sandbox_summary
186
+ return nil unless sandbox_server_key
187
+
188
+ {
189
+ "session_id" => @selection[:sandbox_id],
190
+ "server_key" => sandbox_server_key,
191
+ "repository" => sandbox_session&.repository,
192
+ "repository_ref" => sandbox_session&.repository_ref
193
+ }.compact
194
+ end
195
+
196
+ def ensure_sandbox_live!
197
+ return unless sandbox_server_key
198
+ return if SandboxSession.runtime_server_entry(sandbox_server_key, owner: owner)
199
+
200
+ raise ArgumentError, "Sandbox #{@selection[:sandbox_id]} is no longer running; start it again, or run without it"
201
+ end
202
+
158
203
  # --- replay -----------------------------------------------------------
159
204
 
160
205
  def replay(scenario, spec)
@@ -167,7 +212,8 @@ module ActionAgent
167
212
  scenario.prompt,
168
213
  model_override: spec.model,
169
214
  provider_override: spec.provider,
170
- actor: replay_actor
215
+ actor: replay_actor,
216
+ runtime_sandbox: sandbox_server_key
171
217
  )
172
218
 
173
219
  Evals::Replay.new(
@@ -303,7 +349,7 @@ module ActionAgent
303
349
  end
304
350
 
305
351
  def mcp_dispatcher
306
- @mcp_dispatcher ||= MCPToolDispatcher.new(@evaluation.agent)
352
+ @mcp_dispatcher ||= MCPToolDispatcher.new(@evaluation.agent, extra_server_keys: [ sandbox_server_key ].compact)
307
353
  end
308
354
 
309
355
  # A run whose every declared MCP server failed discovery scored an agent
@@ -0,0 +1,37 @@
1
+ # frozen_string_literal: true
2
+
3
+ module ActionAgent
4
+ # Replaces known secrets in text or nested JSON-like data before it is
5
+ # stored or shown: a sandbox's GitHub token and Claude Code credential can
6
+ # surface in a process's output (an error echoing a URL, a session that
7
+ # prints its environment), and none of that may reach a transcript, a log
8
+ # tail or an API response.
9
+ module SecretScrubber
10
+ MASK = "[REDACTED]"
11
+ # Shorter values are not credentials, and masking them would mangle text.
12
+ MIN_SECRET_LENGTH = 8
13
+
14
+ module_function
15
+
16
+ # @param value [String, Hash, Array, Object] what to scrub
17
+ # @param secrets [Array<String>] the values to mask
18
+ # @return a copy of +value+ with every secret masked
19
+ def scrub(value, secrets)
20
+ secrets = Array(secrets).compact.map(&:to_s).select { |secret| secret.length >= MIN_SECRET_LENGTH }.uniq
21
+ return value if secrets.empty?
22
+
23
+ # Longest first, so a secret that contains another is masked whole.
24
+ pattern = Regexp.union(secrets.sort_by { |secret| -secret.length })
25
+ deep_scrub(value, pattern)
26
+ end
27
+
28
+ def deep_scrub(value, pattern)
29
+ case value
30
+ when String then value.gsub(pattern, MASK)
31
+ when Hash then value.to_h { |key, item| [ key, deep_scrub(item, pattern) ] }
32
+ when Array then value.map { |item| deep_scrub(item, pattern) }
33
+ else value
34
+ end
35
+ end
36
+ end
37
+ end
@@ -35,6 +35,9 @@ module ActionAgent
35
35
  #
36
36
  # Scopes are passed in rather than derived, so the caller's ownership
37
37
  # rules (single-user, per-user, or multi-tenant) decide what is visible.
38
+ # The same goes for checkout sandbox runtimes: the caller hands in the live
39
+ # ones it may see (SandboxSession.runtime_server_listings), and they are
40
+ # listed beside the catalog under their "sandbox:<session_id>" keys.
38
41
  class ToolDiscovery
39
42
  DEFAULT_WINDOW_HOURS = 24 * 7
40
43
  MAX_WINDOW_HOURS = 24 * 90
@@ -53,16 +56,20 @@ module ActionAgent
53
56
  ORIGIN_BUILTIN = "builtin"
54
57
  ORIGIN_AGENT = "agent"
55
58
 
56
- attr_reader :traces, :agents, :window_hours, :since
59
+ attr_reader :traces, :agents, :window_hours, :since, :runtimes
57
60
 
58
61
  # @param traces [ActiveRecord::Relation] the traces the caller may read
59
62
  # @param agents [ActiveRecord::Relation] the agents the caller may reach
60
63
  # @param hours [Integer] how far back to look
61
- def initialize(traces:, agents:, hours: DEFAULT_WINDOW_HOURS)
64
+ # @param runtimes [Array<Hash>] the live checkout sandbox runtimes the
65
+ # caller may see, as SandboxSession.runtime_server_listings returns
66
+ # them — token-free catalog-shaped entries
67
+ def initialize(traces:, agents:, hours: DEFAULT_WINDOW_HOURS, runtimes: [])
62
68
  @traces = traces
63
69
  @agents = agents
64
70
  @window_hours = hours.to_i.clamp(1, MAX_WINDOW_HOURS)
65
71
  @since = @window_hours.hours.ago
72
+ @runtimes = Array(runtimes).index_by { |runtime| runtime[:key] }
66
73
  end
67
74
 
68
75
  # The full inventory: every tool seen in the window, plus the MCP servers
@@ -500,7 +507,7 @@ module ActionAgent
500
507
  end
501
508
  end
502
509
 
503
- keys = (MCPCatalog.keys + detected.keys + configured_servers.keys).uniq
510
+ keys = (MCPCatalog.keys + runtimes.keys + detected.keys + configured_servers.keys).uniq
504
511
 
505
512
  # detected has a default block that would materialize a bucket on
506
513
  # lookup, so unseen servers are passed through as an explicit nil.
@@ -509,7 +516,11 @@ module ActionAgent
509
516
  end
510
517
 
511
518
  def server_row(key, bucket)
512
- catalog = MCPCatalog.find(key)
519
+ # A live checkout runtime reads like a catalog entry: known, named after
520
+ # its repository and ref, with the endpoint it serves on. Its listing
521
+ # carries no bearer token, so nothing below can render one.
522
+ catalog = MCPCatalog.find(key) || runtimes[key]
523
+ runtime = SandboxSession.runtime_server_key?(key)
513
524
  configured = configured_servers[key].to_a.sort
514
525
  calls = bucket ? bucket[:calls] : 0
515
526
 
@@ -524,11 +535,14 @@ module ActionAgent
524
535
  docs_url: catalog&.fetch(:docs_url, nil),
525
536
  first_party: catalog ? catalog[:first_party] : false,
526
537
  requires_credentials: catalog ? catalog[:requires_credentials] : [],
527
- launchable: catalog ? catalog[:sandbox] : false,
538
+ # A runtime is already running — it was started from Settings ->
539
+ # Integrations — so there is nothing here to launch.
540
+ launchable: catalog && !runtime ? catalog[:sandbox] : false,
528
541
  sandbox_type: catalog&.fetch(:sandbox_type, nil),
529
542
  # Catalog membership is what "known" means — a server detected purely
530
543
  # from traffic is real but undocumented here, and the view says so.
531
544
  known: !catalog.nil?,
545
+ runtime: runtime,
532
546
  status: server_status(calls, configured, catalog),
533
547
  calls: calls,
534
548
  errors: bucket ? bucket[:errors] : 0,
data/config/routes.rb CHANGED
@@ -71,8 +71,8 @@ ActionAgent::Engine.routes.draw do
71
71
  end
72
72
  end
73
73
 
74
- # Sandboxes. The engine ships the in-memory backend; an operator registers
75
- # real ones (see ActionAgent.sandbox_backends).
74
+ # Sandboxes. The engine ships the in-memory and local backends; an
75
+ # operator registers the rest (see ActionAgent.sandbox_backends).
76
76
  resources :sandboxes, param: :id, only: [ :index, :create, :show, :destroy ] do
77
77
  collection do
78
78
  post :compare
@@ -80,6 +80,12 @@ ActionAgent::Engine.routes.draw do
80
80
  member do
81
81
  post :run
82
82
  end
83
+ # Claude Code sessions in an app_runtime sandbox's checkout.
84
+ resources :code_sessions, only: [ :index, :create, :show ] do
85
+ member do
86
+ post :cancel
87
+ end
88
+ end
83
89
  end
84
90
 
85
91
  # Tool inventory — auto-detected from the tool roster each generation
@@ -151,7 +157,20 @@ ActionAgent::Engine.routes.draw do
151
157
  # Credentials: dashboard API keys (token shown once on create) and the
152
158
  # owner's own LLM provider credentials, both encrypted at rest.
153
159
  resources :api_keys, only: [ :index, :create, :destroy ]
154
- resources :provider_keys, only: [ :index, :create, :destroy ], param: :provider
160
+ resources :provider_keys, only: [ :index, :create, :destroy ], param: :provider do
161
+ # Reachability + model list for host-based providers (Ollama), for a
162
+ # submitted or the stored host. Read-only.
163
+ post :test, on: :collection
164
+ end
165
+
166
+ # The owner's GitHub connection: the OAuth web flow (connect redirects to
167
+ # GitHub, which returns to callback) and the repositories it makes
168
+ # available to checkout sandboxes.
169
+ resource :github_connection, only: [ :show, :update, :destroy ], controller: "github_connections" do
170
+ get :repositories
171
+ get :connect
172
+ get :callback
173
+ end
155
174
 
156
175
  # Model catalogs for the agent builder (Ollama queried live from the
157
176
  # configured host; hosted providers curated).
@@ -24,6 +24,7 @@ module ActionAgent
24
24
  "mcp_client" => "MCPClient",
25
25
  "mcp_tool_dispatcher" => "MCPToolDispatcher",
26
26
  "mcp_controller" => "MCPController",
27
+ "mcp_dashboard_tools" => "MCPDashboardTools",
27
28
  "mcp_recording_middleware" => "MCPRecordingMiddleware",
28
29
  "mcp_servers_controller" => "MCPServersController",
29
30
  "playwright_mcp_client" => "PlaywrightMCPClient"
@@ -1,5 +1,5 @@
1
1
  # frozen_string_literal: true
2
2
 
3
3
  module ActionAgent
4
- VERSION = "1.7.2"
4
+ VERSION = "1.8.0"
5
5
  end
data/lib/action_agent.rb CHANGED
@@ -104,6 +104,9 @@ require "action_agent/compatibility"
104
104
  # end
105
105
  #
106
106
  module ActionAgent
107
+ # What ActionAgent.claude_code_auth may be set to.
108
+ CLAUDE_CODE_AUTH_MODES = %i[api_key local_login].freeze
109
+
107
110
  class << self
108
111
  # Deprecation warnings for this gem, routed through Rails' machinery so a
109
112
  # host app can silence or escalate them like any other.
@@ -194,9 +197,10 @@ module ActionAgent
194
197
  # @return [String, nil]
195
198
  attr_accessor :layout
196
199
 
197
- # Which sandbox backend to provision with: :mock (the only one the
198
- # engine ships — an in-memory fake that runs nothing) or the name of a
199
- # backend the host registered in sandbox_backends. An unregistered name
200
+ # Which sandbox backend to provision with: :mock (an in-memory fake that
201
+ # runs nothing), :local (checkouts cloned and booted as child processes
202
+ # of the dashboard itself — see local_sandboxes_enabled), or the name of
203
+ # a backend the host registered in sandbox_backends. An unregistered name
200
204
  # falls back to :mock with a logged warning.
201
205
  # @return [Symbol]
202
206
  attr_accessor :sandbox_service
@@ -304,6 +308,74 @@ module ActionAgent
304
308
  # @return [Hash{String => String}]
305
309
  attr_accessor :sandbox_backends
306
310
 
311
+ # Whether the :local sandbox backend may run. It clones the owner's
312
+ # repository onto the dashboard's own machine and runs its setup and
313
+ # server as child processes — the owner's code, with the dashboard's
314
+ # privileges — so it is for a developer's machine or a single-user
315
+ # install. Unset, it follows the environment: on in development and
316
+ # test, off everywhere else.
317
+ # @return [Boolean, nil]
318
+ attr_writer :local_sandboxes_enabled
319
+
320
+ # Where the :local backend keeps each sandbox's checkout, logs and
321
+ # process state (one directory per session). Unset, tmp/action_agent/sandboxes
322
+ # under the host app.
323
+ # @return [String, Pathname, nil]
324
+ attr_writer :local_sandbox_root
325
+
326
+ # How long the :local backend waits for a checkout's setup and server to
327
+ # come up before giving up, in seconds.
328
+ # @return [Integer]
329
+ attr_accessor :local_sandbox_boot_timeout
330
+
331
+ # The Claude Code executable a sandbox backend runs headless sessions
332
+ # with. The :local backend runs it on the dashboard's machine.
333
+ # @return [String]
334
+ attr_accessor :claude_code_command
335
+
336
+ # The permission mode Claude Code sessions run in. "acceptEdits" lets a
337
+ # session edit files in the checkout and run filesystem commands; with
338
+ # nobody to answer prompts, anything else that would ask is denied.
339
+ # @return [String]
340
+ attr_accessor :claude_code_permission_mode
341
+
342
+ # A cap on agentic turns per Claude Code session (nil for Claude Code's
343
+ # own default).
344
+ # @return [Integer, nil]
345
+ attr_accessor :claude_code_max_turns
346
+
347
+ # How long a Claude Code session may run before it is stopped, in
348
+ # seconds.
349
+ # @return [Integer]
350
+ attr_accessor :claude_code_timeout
351
+
352
+ # How Claude Code sessions authenticate.
353
+ #
354
+ # :api_key (the default) runs them on the Anthropic API key the owner
355
+ # connected in Settings -> Integrations, handed to the session as
356
+ # ANTHROPIC_API_KEY. It is the only credential the dashboard stores:
357
+ # Anthropic does not let third-party products collect, store or route
358
+ # requests through Claude.ai subscription credentials
359
+ # (https://code.claude.com/docs/en/legal-and-compliance.md).
360
+ #
361
+ # :local_login runs `claude` on whatever login this machine's user set up
362
+ # with `claude /login` (or `claude auth login`), which Claude Code keeps
363
+ # under ~/.claude or in the keychain. The dashboard never reads, copies
364
+ # or stores it; it only asks `claude auth status` whether there is one.
365
+ # That login is the dashboard user's own, so this works with the :local
366
+ # sandbox backend only, and other backends refuse Claude Code sessions.
367
+ # @return [Symbol] :api_key or :local_login
368
+ attr_reader :claude_code_auth
369
+
370
+ def claude_code_auth=(value)
371
+ mode = value.to_s.to_sym
372
+ unless CLAUDE_CODE_AUTH_MODES.include?(mode)
373
+ raise ArgumentError, "ActionAgent.claude_code_auth must be :api_key or :local_login, not #{value.inspect}"
374
+ end
375
+
376
+ @claude_code_auth = mode
377
+ end
378
+
307
379
  # Whether the dashboard may execute agents against real providers.
308
380
  # Disable to run the dashboard as a read-only observability surface.
309
381
  # @return [Boolean]
@@ -415,6 +487,20 @@ module ActionAgent
415
487
  # @return [Boolean]
416
488
  attr_accessor :encrypt_credentials
417
489
 
490
+ # The GitHub OAuth App the dashboard's "Connect GitHub" flow authorizes
491
+ # against (Settings -> Integrations). Unset, each falls back to
492
+ # GITHUB_CLIENT_ID / GITHUB_CLIENT_SECRET, and the dashboard offers no
493
+ # connection when neither is present. Register the app's callback URL as
494
+ # <mount>/api/github_connection/callback.
495
+ # @return [String, nil]
496
+ attr_writer :github_client_id, :github_client_secret
497
+
498
+ # OAuth scopes requested on connect. +repo+ reaches private repositories
499
+ # so a sandbox can clone them; narrow it to "public_repo read:user" for
500
+ # public checkouts only.
501
+ # @return [String]
502
+ attr_accessor :github_oauth_scopes
503
+
418
504
  # MCP servers the host app itself serves or connects, appended to the
419
505
  # built-in catalog (MCPCatalog) so the MCP Services view lists them and
420
506
  # telemetry traffic attributes to them. Each entry is a hash shaped like
@@ -452,6 +538,17 @@ module ActionAgent
452
538
  # @return [Boolean]
453
539
  attr_accessor :mcp_schema_tools
454
540
 
541
+ # Whether the MCP facade (POST <mount>/mcp) offers the dashboard's own
542
+ # evaluation and telemetry tools — evaluations_list, evaluations_get,
543
+ # evaluations_run, evaluation_runs_get, evaluation_runs_compare,
544
+ # traces_search, traces_get — so a client's coding harness can run an
545
+ # agent's evaluations and read its traces while it edits the agent. Each
546
+ # reads under the key's owner, as the dashboard's JSON API reads under the
547
+ # signed-in owner. On by default; set it to false to leave the facade
548
+ # serving agents and schema tools only.
549
+ # @return [Boolean]
550
+ attr_accessor :mcp_dashboard_tools
551
+
455
552
  # Directory scanned for SchemaTools subclasses when {#schema_tools} is
456
553
  # unset. Relative to the host's root. Set to nil to disable discovery and
457
554
  # require an explicit declaration. Classes built at runtime with
@@ -471,6 +568,20 @@ module ActionAgent
471
568
  # Returns whether multi-tenant mode is enabled.
472
569
  #
473
570
  # @return [Boolean]
571
+ def github_client_id
572
+ @github_client_id.presence || ENV["GITHUB_CLIENT_ID"].presence
573
+ end
574
+
575
+ def github_client_secret
576
+ @github_client_secret.presence || ENV["GITHUB_CLIENT_SECRET"].presence
577
+ end
578
+
579
+ # Whether the GitHub OAuth flow can run on this install.
580
+ # @return [Boolean]
581
+ def github_oauth_configured?
582
+ github_client_id.present? && github_client_secret.present?
583
+ end
584
+
474
585
  def multi_tenant?
475
586
  @multi_tenant == true
476
587
  end
@@ -482,6 +593,14 @@ module ActionAgent
482
593
  @mcp_schema_tools != false
483
594
  end
484
595
 
596
+ # Whether the MCP facade serves the dashboard's evaluation and telemetry
597
+ # tools.
598
+ #
599
+ # @return [Boolean]
600
+ def mcp_dashboard_tools?
601
+ @mcp_dashboard_tools != false
602
+ end
603
+
485
604
  # Returns whether agent execution is permitted.
486
605
  #
487
606
  # @return [Boolean]
@@ -499,6 +618,19 @@ module ActionAgent
499
618
  Rails.env.local?
500
619
  end
501
620
 
621
+ # Whether the :local sandbox backend may run on this install.
622
+ # @return [Boolean]
623
+ def local_sandboxes_enabled?
624
+ return @local_sandboxes_enabled == true unless @local_sandboxes_enabled.nil?
625
+
626
+ Rails.env.local?
627
+ end
628
+
629
+ # @return [Pathname]
630
+ def local_sandbox_root
631
+ Pathname.new(@local_sandbox_root.presence || Rails.root.join("tmp", "action_agent", "sandboxes"))
632
+ end
633
+
502
634
  # Tells the host app that +owner+ performed +kind+. Never raises: a
503
635
  # bookkeeping failure must not fail the action that was already taken.
504
636
  def record_usage(owner, kind)
@@ -633,6 +765,14 @@ module ActionAgent
633
765
  @quota_checker = nil
634
766
  @provider_credentials_resolver = nil
635
767
  @sandbox_backends = {}
768
+ @local_sandboxes_enabled = nil
769
+ @local_sandbox_root = nil
770
+ @local_sandbox_boot_timeout = 600
771
+ @claude_code_command = "claude"
772
+ @claude_code_permission_mode = "acceptEdits"
773
+ @claude_code_max_turns = nil
774
+ @claude_code_timeout = 1800
775
+ @claude_code_auth = :api_key
636
776
  @execution_enabled = true
637
777
  @run_host_agent_classes = false
638
778
  @assistant_enabled = nil
@@ -641,6 +781,9 @@ module ActionAgent
641
781
  @table_name_prefix = "active_agent_"
642
782
  @agent_polymorphic_name = nil
643
783
  @encrypt_credentials = true
784
+ @github_client_id = nil
785
+ @github_client_secret = nil
786
+ @github_oauth_scopes = "repo read:user"
644
787
  @trace_retention = nil
645
788
  @trace_owner_resolver = nil
646
789
  @usage_recorder = nil
@@ -653,6 +796,7 @@ module ActionAgent
653
796
  @schema_tools = nil
654
797
  @schema_tools_path = "app/agent_tools"
655
798
  @mcp_schema_tools = nil
799
+ @mcp_dashboard_tools = nil
656
800
  end
657
801
 
658
802
  # Host-declared schema tool classes, resolved from names and filtered to