actionagent 1.7.2 → 1.8.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/app/assets/builds/action_agent.css +1 -1
- data/app/assets/builds/action_agent.js +59 -54
- data/app/controllers/action_agent/api/agents_controller.rb +19 -3
- data/app/controllers/action_agent/api/code_sessions_controller.rb +156 -0
- data/app/controllers/action_agent/api/evaluations_controller.rb +27 -130
- data/app/controllers/action_agent/api/github_connections_controller.rb +147 -0
- data/app/controllers/action_agent/api/mcp_controller.rb +38 -5
- data/app/controllers/action_agent/api/mcp_servers_controller.rb +10 -1
- data/app/controllers/action_agent/api/provider_keys_controller.rb +54 -3
- data/app/controllers/action_agent/api/provider_models_controller.rb +10 -6
- data/app/controllers/action_agent/api/sandboxes_controller.rb +97 -6
- data/app/controllers/concerns/action_agent/api/evaluation_run_starting.rb +93 -0
- data/app/controllers/concerns/action_agent/api/mcp_dashboard_tools.rb +507 -0
- data/app/controllers/concerns/action_agent/api/run_sandbox.rb +65 -0
- data/app/jobs/action_agent/code_session_job.rb +166 -0
- data/app/jobs/action_agent/sandbox_cleanup_job.rb +80 -11
- data/app/jobs/action_agent/sandbox_provision_job.rb +122 -14
- data/app/jobs/action_agent/sandbox_run_job.rb +10 -3
- data/app/models/action_agent/agent.rb +16 -6
- data/app/models/action_agent/agent_run.rb +20 -1
- data/app/models/action_agent/code_session.rb +141 -0
- data/app/models/action_agent/evaluation_run.rb +23 -1
- data/app/models/action_agent/github_connection.rb +75 -0
- data/app/models/action_agent/provider_key.rb +142 -10
- data/app/models/action_agent/sandbox_session.rb +193 -17
- data/app/serializers/action_agent/evaluation_serializer.rb +118 -0
- data/app/serializers/action_agent/telemetry_trace_serializer.rb +10 -2
- data/app/services/action_agent/agent_execution_service.rb +4 -2
- data/app/services/action_agent/agent_tool_roster.rb +20 -9
- data/app/services/action_agent/claude_code_auth.rb +86 -0
- data/app/services/action_agent/dashboard_assistant_service.rb +47 -5
- data/app/services/action_agent/evaluation_tool_resolver.rb +18 -0
- data/app/services/action_agent/github_client.rb +111 -0
- data/app/services/action_agent/local_sandbox_backend.rb +1689 -0
- data/app/services/action_agent/local_sandbox_databases.rb +257 -0
- data/app/services/action_agent/mcp_client.rb +5 -1
- data/app/services/action_agent/mcp_tool_dispatcher.rb +137 -17
- data/app/services/action_agent/mock_sandbox_backend.rb +39 -0
- data/app/services/action_agent/ollama_host_probe.rb +75 -0
- data/app/services/action_agent/payload_bounds.rb +36 -0
- data/app/services/action_agent/sandbox_manifest.rb +67 -0
- data/app/services/action_agent/sandbox_orchestrator.rb +69 -14
- data/app/services/action_agent/scenario_evaluation_runner.rb +50 -4
- data/app/services/action_agent/secret_scrubber.rb +37 -0
- data/app/services/action_agent/tool_discovery.rb +19 -5
- data/config/routes.rb +22 -3
- data/lib/action_agent/engine.rb +1 -0
- data/lib/action_agent/version.rb +1 -1
- data/lib/action_agent.rb +147 -3
- data/lib/generators/action_agent/install_generator.rb +30 -3
- data/lib/generators/action_agent/templates/action_agent.rb.erb +44 -0
- data/lib/generators/action_agent/templates/add_provider_key_api_key.rb.erb +25 -0
- data/lib/generators/action_agent/templates/create_active_agent_code_sessions.rb.erb +59 -0
- data/lib/generators/action_agent/templates/create_active_agent_dashboard_tables.rb.erb +2 -0
- data/lib/generators/action_agent/templates/create_active_agent_github_connections.rb.erb +61 -0
- data/lib/tasks/claude_code.rake +16 -0
- data/lib/tasks/sandbox.rake +26 -0
- metadata +23 -1
|
@@ -0,0 +1,67 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module ActionAgent
|
|
4
|
+
# The runtime manifest: how a booted checkout tells the sandbox backend
|
|
5
|
+
# where its MCP facade answers and which bearer token opens it.
|
|
6
|
+
#
|
|
7
|
+
# { "mcp_path": "/activeagents/mcp", "mcp_token": "aa_..." }
|
|
8
|
+
#
|
|
9
|
+
# The checked-out app writes it with `bin/rails action_agent:sandbox:manifest`
|
|
10
|
+
# (it mounts this engine, so the task ships with it), and a backend reads it
|
|
11
|
+
# back with .parse. Both halves live here so they cannot drift.
|
|
12
|
+
module SandboxManifest
|
|
13
|
+
# Where the manifest task writes, when set; stdout otherwise.
|
|
14
|
+
PATH_ENV = "ACTION_AGENT_SANDBOX_MANIFEST"
|
|
15
|
+
# The dashboard API key the manifest hands out, created once per checkout.
|
|
16
|
+
KEY_NAME = "Checkout sandbox runtime"
|
|
17
|
+
|
|
18
|
+
class Error < StandardError; end
|
|
19
|
+
|
|
20
|
+
module_function
|
|
21
|
+
|
|
22
|
+
# Builds the manifest inside the booted app: the engine's MCP path under
|
|
23
|
+
# wherever the app mounts it, and an API key for the MCP facade.
|
|
24
|
+
#
|
|
25
|
+
# @param routes [ActionDispatch::Routing::RouteSet] the app's routes
|
|
26
|
+
# @return [Hash{String => String}]
|
|
27
|
+
def generate(routes: Rails.application.routes)
|
|
28
|
+
# Rails 8 loads routes lazily in development and test; the mount is
|
|
29
|
+
# invisible until they are.
|
|
30
|
+
Rails.application.reload_routes_unless_loaded if Rails.application.respond_to?(:reload_routes_unless_loaded)
|
|
31
|
+
|
|
32
|
+
mount = ActiveAgent::Telemetry::Configuration.new.mount_path_in(routes)
|
|
33
|
+
raise Error, "ActionAgent::Engine is not mounted in this app's routes" if mount.nil?
|
|
34
|
+
|
|
35
|
+
{ "mcp_path" => "#{mount}/mcp", "mcp_token" => api_key.token }
|
|
36
|
+
end
|
|
37
|
+
|
|
38
|
+
# The checkout's own dashboard API key for the facade. Reused across
|
|
39
|
+
# runs so a re-run manifest does not mint a key per boot. A host that
|
|
40
|
+
# owns keys by account or user gets a key with no owner, which reaches
|
|
41
|
+
# no agents over MCP — the manifest says so rather than failing.
|
|
42
|
+
def api_key
|
|
43
|
+
ActionAgent::ApiKey.find_or_create_by!(name: KEY_NAME)
|
|
44
|
+
end
|
|
45
|
+
|
|
46
|
+
# Reads a manifest a checkout wrote.
|
|
47
|
+
#
|
|
48
|
+
# @param json [String]
|
|
49
|
+
# @return [Hash{String => String}] with "mcp_path" and "mcp_token"
|
|
50
|
+
def parse(json)
|
|
51
|
+
data = JSON.parse(json.to_s)
|
|
52
|
+
raise Error, "the manifest is not a JSON object" unless data.is_a?(Hash)
|
|
53
|
+
|
|
54
|
+
path = data["mcp_path"]
|
|
55
|
+
unless path.is_a?(String) && path.start_with?("/")
|
|
56
|
+
raise Error, "the manifest names no mcp_path (expected a path such as /activeagents/mcp)"
|
|
57
|
+
end
|
|
58
|
+
|
|
59
|
+
token = data["mcp_token"]
|
|
60
|
+
raise Error, "the manifest's mcp_token is not a string" unless token.nil? || token.is_a?(String)
|
|
61
|
+
|
|
62
|
+
{ "mcp_path" => path, "mcp_token" => token }
|
|
63
|
+
rescue JSON::ParserError => e
|
|
64
|
+
raise Error, "the manifest is not JSON (#{e.message.truncate(120)})"
|
|
65
|
+
end
|
|
66
|
+
end
|
|
67
|
+
end
|
|
@@ -3,16 +3,14 @@
|
|
|
3
3
|
module ActionAgent
|
|
4
4
|
# SandboxOrchestrator
|
|
5
5
|
#
|
|
6
|
-
# Unified interface for managing agent sandbox sessions.
|
|
7
|
-
#
|
|
8
|
-
#
|
|
9
|
-
#
|
|
10
|
-
# - cloud_run: Google Cloud Run Jobs (serverless)
|
|
11
|
-
# - kubernetes: Kubernetes pods (GKE, EKS, self-hosted k8s)
|
|
6
|
+
# Unified interface for managing agent sandbox sessions. The engine ships
|
|
7
|
+
# two backends — :mock (in-memory, runs nothing) and :local (checkouts as
|
|
8
|
+
# child processes of the dashboard) — and a host registers the rest
|
|
9
|
+
# (Incus, Cloud Run, Kubernetes) in ActionAgent.sandbox_backends.
|
|
12
10
|
#
|
|
13
11
|
# Configuration:
|
|
14
|
-
#
|
|
15
|
-
# Default:
|
|
12
|
+
# ActionAgent.sandbox_service, or the SANDBOX_BACKEND environment
|
|
13
|
+
# variable, which wins. Default: :mock.
|
|
16
14
|
#
|
|
17
15
|
# Usage:
|
|
18
16
|
# orchestrator = SandboxOrchestrator.new
|
|
@@ -21,14 +19,21 @@ module ActionAgent
|
|
|
21
19
|
# orchestrator.terminate(container_id)
|
|
22
20
|
#
|
|
23
21
|
class SandboxOrchestrator
|
|
24
|
-
#
|
|
25
|
-
#
|
|
26
|
-
#
|
|
22
|
+
# Anything that talks to real infrastructure (Incus, Kubernetes, Cloud
|
|
23
|
+
# Run) is registered by the app that operates it, so the engine carries
|
|
24
|
+
# none of those SDKs:
|
|
27
25
|
#
|
|
28
26
|
# ActionAgent.sandbox_backends = {
|
|
29
27
|
# "cloud_run" => "CloudRunService"
|
|
30
28
|
# }
|
|
31
|
-
|
|
29
|
+
#
|
|
30
|
+
# The engine also ships :local, which boots app_runtime checkouts as
|
|
31
|
+
# child processes of the dashboard itself (see LocalSandboxBackend and
|
|
32
|
+
# ActionAgent.local_sandboxes_enabled?).
|
|
33
|
+
BUILT_IN_BACKENDS = {
|
|
34
|
+
"mock" => "ActionAgent::MockSandboxBackend",
|
|
35
|
+
"local" => "ActionAgent::LocalSandboxBackend"
|
|
36
|
+
}.freeze
|
|
32
37
|
|
|
33
38
|
# Backends disagree on what to call each verb. Candidates are tried in
|
|
34
39
|
# order and the first the backend responds to wins, so a host-registered
|
|
@@ -38,7 +43,11 @@ module ActionAgent
|
|
|
38
43
|
status: %i[status container_status pod_status job_status],
|
|
39
44
|
terminate: %i[terminate terminate_pod cancel_job],
|
|
40
45
|
list: %i[list_sandboxes list_sandbox_pods list_jobs],
|
|
41
|
-
cleanup: %i[cleanup_expired cleanup_expired_pods cleanup_expired_jobs]
|
|
46
|
+
cleanup: %i[cleanup_expired cleanup_expired_pods cleanup_expired_jobs],
|
|
47
|
+
# Claude Code sessions inside an app_runtime checkout. Optional: a
|
|
48
|
+
# backend without them simply cannot run sessions (see #supports?).
|
|
49
|
+
code_session: %i[run_code_session],
|
|
50
|
+
cancel_code_session: %i[cancel_code_session]
|
|
42
51
|
}.freeze
|
|
43
52
|
|
|
44
53
|
class UnsupportedBackendError < StandardError; end
|
|
@@ -101,7 +110,12 @@ module ActionAgent
|
|
|
101
110
|
instance_tier: result[:instance_tier] || tier&.id,
|
|
102
111
|
resources: result[:resources],
|
|
103
112
|
hourly_cost: result[:hourly_cost] || tier&.hourly_cost&.to_f,
|
|
104
|
-
created_at: result[:created_at] || Time.current
|
|
113
|
+
created_at: result[:created_at] || Time.current,
|
|
114
|
+
# An app_runtime sandbox's backend clones sandbox_session.checkout_spec,
|
|
115
|
+
# boots the app, and reports where its MCP facade answers (and the
|
|
116
|
+
# bearer token it expects) so agents can use the checkout's tools.
|
|
117
|
+
mcp_url: result[:mcp_url],
|
|
118
|
+
mcp_token: result[:mcp_token]
|
|
105
119
|
}
|
|
106
120
|
end
|
|
107
121
|
|
|
@@ -153,6 +167,47 @@ module ActionAgent
|
|
|
153
167
|
@backend.public_send(adapter_method(:cleanup))
|
|
154
168
|
end
|
|
155
169
|
|
|
170
|
+
# The handle the backend would give +sandbox_session+'s sandbox, for a
|
|
171
|
+
# backend that derives it from the session (nil otherwise).
|
|
172
|
+
def handle_for(sandbox_session)
|
|
173
|
+
@backend.respond_to?(:handle_for) ? @backend.handle_for(sandbox_session) : nil
|
|
174
|
+
end
|
|
175
|
+
|
|
176
|
+
# Whether the backend can name a session's sandbox without a recorded
|
|
177
|
+
# handle (see #handle_for).
|
|
178
|
+
def derives_handles?
|
|
179
|
+
@backend.respond_to?(:handle_for)
|
|
180
|
+
end
|
|
181
|
+
|
|
182
|
+
# Whether the backend runs sandboxes as processes of the dashboard, on
|
|
183
|
+
# its own machine and as its own user (LocalSandboxBackend): the only
|
|
184
|
+
# place ActionAgent.claude_code_auth = :local_login can work.
|
|
185
|
+
def local?
|
|
186
|
+
@backend.is_a?(LocalSandboxBackend)
|
|
187
|
+
end
|
|
188
|
+
|
|
189
|
+
# Whether the backend implements +verb+ (an ADAPTER_METHODS key).
|
|
190
|
+
def supports?(verb)
|
|
191
|
+
ADAPTER_METHODS.fetch(verb).any? { |m| @backend.respond_to?(m) }
|
|
192
|
+
end
|
|
193
|
+
|
|
194
|
+
# Runs a Claude Code session in +sandbox_session+'s checkout, yielding
|
|
195
|
+
# each stream-json event (a Hash) as it arrives. Returns the backend's
|
|
196
|
+
# outcome: { exit_status:, diff: }.
|
|
197
|
+
def run_code_session(sandbox_session, code_session, &on_event)
|
|
198
|
+
# Checked when a session is requested too; this covers one queued
|
|
199
|
+
# before the configuration changed.
|
|
200
|
+
refusal = ClaudeCodeAuth.backend_refusal(self)
|
|
201
|
+
raise UnsupportedBackendError, refusal if refusal
|
|
202
|
+
|
|
203
|
+
@backend.public_send(adapter_method(:code_session), sandbox_session, code_session, &on_event)
|
|
204
|
+
end
|
|
205
|
+
|
|
206
|
+
# Stops a running Claude Code session.
|
|
207
|
+
def cancel_code_session(sandbox_session, code_session)
|
|
208
|
+
@backend.public_send(adapter_method(:cancel_code_session), sandbox_session, code_session)
|
|
209
|
+
end
|
|
210
|
+
|
|
156
211
|
# Check if the backend is healthy
|
|
157
212
|
#
|
|
158
213
|
# @return [Boolean] true if backend is reachable
|
|
@@ -17,7 +17,16 @@ module ActionAgent
|
|
|
17
17
|
# "_models" — per model: pass rate, mean score, latency, tokens, cost, fault counts
|
|
18
18
|
# "_recommendations" — faults grouped across scenarios with the fix each calls for
|
|
19
19
|
# "_verdict" — the best model and why (judge-written when a judge is available)
|
|
20
|
-
# "_selection" — the scenarios and models this run covered
|
|
20
|
+
# "_selection" — the scenarios and models this run covered, and
|
|
21
|
+
# the checkout sandbox it replayed against, if any
|
|
22
|
+
#
|
|
23
|
+
# A selection's `sandbox_id` names a checkout sandbox whose app runtime
|
|
24
|
+
# every replay reaches beside the agent's own MCP servers, as if the agent
|
|
25
|
+
# listed it (see Api::RunSandbox, which checked the caller owns it). The
|
|
26
|
+
# agent is not changed. The runtime is resolved again here, among the
|
|
27
|
+
# agent's owner's sessions, since the run may start well after it was
|
|
28
|
+
# asked for; one that is no longer live fails the run rather than
|
|
29
|
+
# replaying without the tools it was meant to test.
|
|
21
30
|
class ScenarioEvaluationRunner < EvaluationRunnerService
|
|
22
31
|
Evals = ActiveAgent::Evals
|
|
23
32
|
|
|
@@ -43,6 +52,7 @@ module ActionAgent
|
|
|
43
52
|
specs = model_specs
|
|
44
53
|
run = @run || @evaluation.evaluation_runs.create!(status: :pending)
|
|
45
54
|
run.update!(status: :running, selection: selection_summary(scenarios, specs))
|
|
55
|
+
ensure_sandbox_live!
|
|
46
56
|
|
|
47
57
|
if scenarios.empty?
|
|
48
58
|
run.update!(status: :failed, error_message: "No scenarios selected — add scenarios to the evaluation or widen the selection",
|
|
@@ -151,10 +161,45 @@ module ActionAgent
|
|
|
151
161
|
"scenario_ids" => scenarios.map(&:id),
|
|
152
162
|
"scenario_keys" => scenarios.map(&:key),
|
|
153
163
|
"group" => @selection[:group].presence,
|
|
154
|
-
"models" => specs.map(&:to_h)
|
|
164
|
+
"models" => specs.map(&:to_h),
|
|
165
|
+
"sandbox" => sandbox_summary
|
|
155
166
|
}.compact
|
|
156
167
|
end
|
|
157
168
|
|
|
169
|
+
# --- sandbox ----------------------------------------------------------
|
|
170
|
+
|
|
171
|
+
# "sandbox:<session_id>" for a run against a checkout sandbox, or nil.
|
|
172
|
+
def sandbox_server_key
|
|
173
|
+
id = @selection[:sandbox_id]
|
|
174
|
+
"#{SandboxSession::RUNTIME_SERVER_PREFIX}#{id}" if id.is_a?(String) && id.present?
|
|
175
|
+
end
|
|
176
|
+
|
|
177
|
+
def sandbox_session
|
|
178
|
+
return @sandbox_session if defined?(@sandbox_session)
|
|
179
|
+
|
|
180
|
+
@sandbox_session = sandbox_server_key && SandboxSession.for_owner(owner).find_by(session_id: @selection[:sandbox_id])
|
|
181
|
+
end
|
|
182
|
+
|
|
183
|
+
# What the run records about its sandbox: which checkout, never its
|
|
184
|
+
# token.
|
|
185
|
+
def sandbox_summary
|
|
186
|
+
return nil unless sandbox_server_key
|
|
187
|
+
|
|
188
|
+
{
|
|
189
|
+
"session_id" => @selection[:sandbox_id],
|
|
190
|
+
"server_key" => sandbox_server_key,
|
|
191
|
+
"repository" => sandbox_session&.repository,
|
|
192
|
+
"repository_ref" => sandbox_session&.repository_ref
|
|
193
|
+
}.compact
|
|
194
|
+
end
|
|
195
|
+
|
|
196
|
+
def ensure_sandbox_live!
|
|
197
|
+
return unless sandbox_server_key
|
|
198
|
+
return if SandboxSession.runtime_server_entry(sandbox_server_key, owner: owner)
|
|
199
|
+
|
|
200
|
+
raise ArgumentError, "Sandbox #{@selection[:sandbox_id]} is no longer running; start it again, or run without it"
|
|
201
|
+
end
|
|
202
|
+
|
|
158
203
|
# --- replay -----------------------------------------------------------
|
|
159
204
|
|
|
160
205
|
def replay(scenario, spec)
|
|
@@ -167,7 +212,8 @@ module ActionAgent
|
|
|
167
212
|
scenario.prompt,
|
|
168
213
|
model_override: spec.model,
|
|
169
214
|
provider_override: spec.provider,
|
|
170
|
-
actor: replay_actor
|
|
215
|
+
actor: replay_actor,
|
|
216
|
+
runtime_sandbox: sandbox_server_key
|
|
171
217
|
)
|
|
172
218
|
|
|
173
219
|
Evals::Replay.new(
|
|
@@ -303,7 +349,7 @@ module ActionAgent
|
|
|
303
349
|
end
|
|
304
350
|
|
|
305
351
|
def mcp_dispatcher
|
|
306
|
-
@mcp_dispatcher ||= MCPToolDispatcher.new(@evaluation.agent)
|
|
352
|
+
@mcp_dispatcher ||= MCPToolDispatcher.new(@evaluation.agent, extra_server_keys: [ sandbox_server_key ].compact)
|
|
307
353
|
end
|
|
308
354
|
|
|
309
355
|
# A run whose every declared MCP server failed discovery scored an agent
|
|
@@ -0,0 +1,37 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module ActionAgent
|
|
4
|
+
# Replaces known secrets in text or nested JSON-like data before it is
|
|
5
|
+
# stored or shown: a sandbox's GitHub token and Claude Code credential can
|
|
6
|
+
# surface in a process's output (an error echoing a URL, a session that
|
|
7
|
+
# prints its environment), and none of that may reach a transcript, a log
|
|
8
|
+
# tail or an API response.
|
|
9
|
+
module SecretScrubber
|
|
10
|
+
MASK = "[REDACTED]"
|
|
11
|
+
# Shorter values are not credentials, and masking them would mangle text.
|
|
12
|
+
MIN_SECRET_LENGTH = 8
|
|
13
|
+
|
|
14
|
+
module_function
|
|
15
|
+
|
|
16
|
+
# @param value [String, Hash, Array, Object] what to scrub
|
|
17
|
+
# @param secrets [Array<String>] the values to mask
|
|
18
|
+
# @return a copy of +value+ with every secret masked
|
|
19
|
+
def scrub(value, secrets)
|
|
20
|
+
secrets = Array(secrets).compact.map(&:to_s).select { |secret| secret.length >= MIN_SECRET_LENGTH }.uniq
|
|
21
|
+
return value if secrets.empty?
|
|
22
|
+
|
|
23
|
+
# Longest first, so a secret that contains another is masked whole.
|
|
24
|
+
pattern = Regexp.union(secrets.sort_by { |secret| -secret.length })
|
|
25
|
+
deep_scrub(value, pattern)
|
|
26
|
+
end
|
|
27
|
+
|
|
28
|
+
def deep_scrub(value, pattern)
|
|
29
|
+
case value
|
|
30
|
+
when String then value.gsub(pattern, MASK)
|
|
31
|
+
when Hash then value.to_h { |key, item| [ key, deep_scrub(item, pattern) ] }
|
|
32
|
+
when Array then value.map { |item| deep_scrub(item, pattern) }
|
|
33
|
+
else value
|
|
34
|
+
end
|
|
35
|
+
end
|
|
36
|
+
end
|
|
37
|
+
end
|
|
@@ -35,6 +35,9 @@ module ActionAgent
|
|
|
35
35
|
#
|
|
36
36
|
# Scopes are passed in rather than derived, so the caller's ownership
|
|
37
37
|
# rules (single-user, per-user, or multi-tenant) decide what is visible.
|
|
38
|
+
# The same goes for checkout sandbox runtimes: the caller hands in the live
|
|
39
|
+
# ones it may see (SandboxSession.runtime_server_listings), and they are
|
|
40
|
+
# listed beside the catalog under their "sandbox:<session_id>" keys.
|
|
38
41
|
class ToolDiscovery
|
|
39
42
|
DEFAULT_WINDOW_HOURS = 24 * 7
|
|
40
43
|
MAX_WINDOW_HOURS = 24 * 90
|
|
@@ -53,16 +56,20 @@ module ActionAgent
|
|
|
53
56
|
ORIGIN_BUILTIN = "builtin"
|
|
54
57
|
ORIGIN_AGENT = "agent"
|
|
55
58
|
|
|
56
|
-
attr_reader :traces, :agents, :window_hours, :since
|
|
59
|
+
attr_reader :traces, :agents, :window_hours, :since, :runtimes
|
|
57
60
|
|
|
58
61
|
# @param traces [ActiveRecord::Relation] the traces the caller may read
|
|
59
62
|
# @param agents [ActiveRecord::Relation] the agents the caller may reach
|
|
60
63
|
# @param hours [Integer] how far back to look
|
|
61
|
-
|
|
64
|
+
# @param runtimes [Array<Hash>] the live checkout sandbox runtimes the
|
|
65
|
+
# caller may see, as SandboxSession.runtime_server_listings returns
|
|
66
|
+
# them — token-free catalog-shaped entries
|
|
67
|
+
def initialize(traces:, agents:, hours: DEFAULT_WINDOW_HOURS, runtimes: [])
|
|
62
68
|
@traces = traces
|
|
63
69
|
@agents = agents
|
|
64
70
|
@window_hours = hours.to_i.clamp(1, MAX_WINDOW_HOURS)
|
|
65
71
|
@since = @window_hours.hours.ago
|
|
72
|
+
@runtimes = Array(runtimes).index_by { |runtime| runtime[:key] }
|
|
66
73
|
end
|
|
67
74
|
|
|
68
75
|
# The full inventory: every tool seen in the window, plus the MCP servers
|
|
@@ -500,7 +507,7 @@ module ActionAgent
|
|
|
500
507
|
end
|
|
501
508
|
end
|
|
502
509
|
|
|
503
|
-
keys = (MCPCatalog.keys + detected.keys + configured_servers.keys).uniq
|
|
510
|
+
keys = (MCPCatalog.keys + runtimes.keys + detected.keys + configured_servers.keys).uniq
|
|
504
511
|
|
|
505
512
|
# detected has a default block that would materialize a bucket on
|
|
506
513
|
# lookup, so unseen servers are passed through as an explicit nil.
|
|
@@ -509,7 +516,11 @@ module ActionAgent
|
|
|
509
516
|
end
|
|
510
517
|
|
|
511
518
|
def server_row(key, bucket)
|
|
512
|
-
catalog
|
|
519
|
+
# A live checkout runtime reads like a catalog entry: known, named after
|
|
520
|
+
# its repository and ref, with the endpoint it serves on. Its listing
|
|
521
|
+
# carries no bearer token, so nothing below can render one.
|
|
522
|
+
catalog = MCPCatalog.find(key) || runtimes[key]
|
|
523
|
+
runtime = SandboxSession.runtime_server_key?(key)
|
|
513
524
|
configured = configured_servers[key].to_a.sort
|
|
514
525
|
calls = bucket ? bucket[:calls] : 0
|
|
515
526
|
|
|
@@ -524,11 +535,14 @@ module ActionAgent
|
|
|
524
535
|
docs_url: catalog&.fetch(:docs_url, nil),
|
|
525
536
|
first_party: catalog ? catalog[:first_party] : false,
|
|
526
537
|
requires_credentials: catalog ? catalog[:requires_credentials] : [],
|
|
527
|
-
|
|
538
|
+
# A runtime is already running — it was started from Settings ->
|
|
539
|
+
# Integrations — so there is nothing here to launch.
|
|
540
|
+
launchable: catalog && !runtime ? catalog[:sandbox] : false,
|
|
528
541
|
sandbox_type: catalog&.fetch(:sandbox_type, nil),
|
|
529
542
|
# Catalog membership is what "known" means — a server detected purely
|
|
530
543
|
# from traffic is real but undocumented here, and the view says so.
|
|
531
544
|
known: !catalog.nil?,
|
|
545
|
+
runtime: runtime,
|
|
532
546
|
status: server_status(calls, configured, catalog),
|
|
533
547
|
calls: calls,
|
|
534
548
|
errors: bucket ? bucket[:errors] : 0,
|
data/config/routes.rb
CHANGED
|
@@ -71,8 +71,8 @@ ActionAgent::Engine.routes.draw do
|
|
|
71
71
|
end
|
|
72
72
|
end
|
|
73
73
|
|
|
74
|
-
# Sandboxes. The engine ships the in-memory
|
|
75
|
-
#
|
|
74
|
+
# Sandboxes. The engine ships the in-memory and local backends; an
|
|
75
|
+
# operator registers the rest (see ActionAgent.sandbox_backends).
|
|
76
76
|
resources :sandboxes, param: :id, only: [ :index, :create, :show, :destroy ] do
|
|
77
77
|
collection do
|
|
78
78
|
post :compare
|
|
@@ -80,6 +80,12 @@ ActionAgent::Engine.routes.draw do
|
|
|
80
80
|
member do
|
|
81
81
|
post :run
|
|
82
82
|
end
|
|
83
|
+
# Claude Code sessions in an app_runtime sandbox's checkout.
|
|
84
|
+
resources :code_sessions, only: [ :index, :create, :show ] do
|
|
85
|
+
member do
|
|
86
|
+
post :cancel
|
|
87
|
+
end
|
|
88
|
+
end
|
|
83
89
|
end
|
|
84
90
|
|
|
85
91
|
# Tool inventory — auto-detected from the tool roster each generation
|
|
@@ -151,7 +157,20 @@ ActionAgent::Engine.routes.draw do
|
|
|
151
157
|
# Credentials: dashboard API keys (token shown once on create) and the
|
|
152
158
|
# owner's own LLM provider credentials, both encrypted at rest.
|
|
153
159
|
resources :api_keys, only: [ :index, :create, :destroy ]
|
|
154
|
-
resources :provider_keys, only: [ :index, :create, :destroy ], param: :provider
|
|
160
|
+
resources :provider_keys, only: [ :index, :create, :destroy ], param: :provider do
|
|
161
|
+
# Reachability + model list for host-based providers (Ollama), for a
|
|
162
|
+
# submitted or the stored host. Read-only.
|
|
163
|
+
post :test, on: :collection
|
|
164
|
+
end
|
|
165
|
+
|
|
166
|
+
# The owner's GitHub connection: the OAuth web flow (connect redirects to
|
|
167
|
+
# GitHub, which returns to callback) and the repositories it makes
|
|
168
|
+
# available to checkout sandboxes.
|
|
169
|
+
resource :github_connection, only: [ :show, :update, :destroy ], controller: "github_connections" do
|
|
170
|
+
get :repositories
|
|
171
|
+
get :connect
|
|
172
|
+
get :callback
|
|
173
|
+
end
|
|
155
174
|
|
|
156
175
|
# Model catalogs for the agent builder (Ollama queried live from the
|
|
157
176
|
# configured host; hosted providers curated).
|
data/lib/action_agent/engine.rb
CHANGED
|
@@ -24,6 +24,7 @@ module ActionAgent
|
|
|
24
24
|
"mcp_client" => "MCPClient",
|
|
25
25
|
"mcp_tool_dispatcher" => "MCPToolDispatcher",
|
|
26
26
|
"mcp_controller" => "MCPController",
|
|
27
|
+
"mcp_dashboard_tools" => "MCPDashboardTools",
|
|
27
28
|
"mcp_recording_middleware" => "MCPRecordingMiddleware",
|
|
28
29
|
"mcp_servers_controller" => "MCPServersController",
|
|
29
30
|
"playwright_mcp_client" => "PlaywrightMCPClient"
|
data/lib/action_agent/version.rb
CHANGED
data/lib/action_agent.rb
CHANGED
|
@@ -104,6 +104,9 @@ require "action_agent/compatibility"
|
|
|
104
104
|
# end
|
|
105
105
|
#
|
|
106
106
|
module ActionAgent
|
|
107
|
+
# What ActionAgent.claude_code_auth may be set to.
|
|
108
|
+
CLAUDE_CODE_AUTH_MODES = %i[api_key local_login].freeze
|
|
109
|
+
|
|
107
110
|
class << self
|
|
108
111
|
# Deprecation warnings for this gem, routed through Rails' machinery so a
|
|
109
112
|
# host app can silence or escalate them like any other.
|
|
@@ -194,9 +197,10 @@ module ActionAgent
|
|
|
194
197
|
# @return [String, nil]
|
|
195
198
|
attr_accessor :layout
|
|
196
199
|
|
|
197
|
-
# Which sandbox backend to provision with: :mock (
|
|
198
|
-
#
|
|
199
|
-
#
|
|
200
|
+
# Which sandbox backend to provision with: :mock (an in-memory fake that
|
|
201
|
+
# runs nothing), :local (checkouts cloned and booted as child processes
|
|
202
|
+
# of the dashboard itself — see local_sandboxes_enabled), or the name of
|
|
203
|
+
# a backend the host registered in sandbox_backends. An unregistered name
|
|
200
204
|
# falls back to :mock with a logged warning.
|
|
201
205
|
# @return [Symbol]
|
|
202
206
|
attr_accessor :sandbox_service
|
|
@@ -304,6 +308,74 @@ module ActionAgent
|
|
|
304
308
|
# @return [Hash{String => String}]
|
|
305
309
|
attr_accessor :sandbox_backends
|
|
306
310
|
|
|
311
|
+
# Whether the :local sandbox backend may run. It clones the owner's
|
|
312
|
+
# repository onto the dashboard's own machine and runs its setup and
|
|
313
|
+
# server as child processes — the owner's code, with the dashboard's
|
|
314
|
+
# privileges — so it is for a developer's machine or a single-user
|
|
315
|
+
# install. Unset, it follows the environment: on in development and
|
|
316
|
+
# test, off everywhere else.
|
|
317
|
+
# @return [Boolean, nil]
|
|
318
|
+
attr_writer :local_sandboxes_enabled
|
|
319
|
+
|
|
320
|
+
# Where the :local backend keeps each sandbox's checkout, logs and
|
|
321
|
+
# process state (one directory per session). Unset, tmp/action_agent/sandboxes
|
|
322
|
+
# under the host app.
|
|
323
|
+
# @return [String, Pathname, nil]
|
|
324
|
+
attr_writer :local_sandbox_root
|
|
325
|
+
|
|
326
|
+
# How long the :local backend waits for a checkout's setup and server to
|
|
327
|
+
# come up before giving up, in seconds.
|
|
328
|
+
# @return [Integer]
|
|
329
|
+
attr_accessor :local_sandbox_boot_timeout
|
|
330
|
+
|
|
331
|
+
# The Claude Code executable a sandbox backend runs headless sessions
|
|
332
|
+
# with. The :local backend runs it on the dashboard's machine.
|
|
333
|
+
# @return [String]
|
|
334
|
+
attr_accessor :claude_code_command
|
|
335
|
+
|
|
336
|
+
# The permission mode Claude Code sessions run in. "acceptEdits" lets a
|
|
337
|
+
# session edit files in the checkout and run filesystem commands; with
|
|
338
|
+
# nobody to answer prompts, anything else that would ask is denied.
|
|
339
|
+
# @return [String]
|
|
340
|
+
attr_accessor :claude_code_permission_mode
|
|
341
|
+
|
|
342
|
+
# A cap on agentic turns per Claude Code session (nil for Claude Code's
|
|
343
|
+
# own default).
|
|
344
|
+
# @return [Integer, nil]
|
|
345
|
+
attr_accessor :claude_code_max_turns
|
|
346
|
+
|
|
347
|
+
# How long a Claude Code session may run before it is stopped, in
|
|
348
|
+
# seconds.
|
|
349
|
+
# @return [Integer]
|
|
350
|
+
attr_accessor :claude_code_timeout
|
|
351
|
+
|
|
352
|
+
# How Claude Code sessions authenticate.
|
|
353
|
+
#
|
|
354
|
+
# :api_key (the default) runs them on the Anthropic API key the owner
|
|
355
|
+
# connected in Settings -> Integrations, handed to the session as
|
|
356
|
+
# ANTHROPIC_API_KEY. It is the only credential the dashboard stores:
|
|
357
|
+
# Anthropic does not let third-party products collect, store or route
|
|
358
|
+
# requests through Claude.ai subscription credentials
|
|
359
|
+
# (https://code.claude.com/docs/en/legal-and-compliance.md).
|
|
360
|
+
#
|
|
361
|
+
# :local_login runs `claude` on whatever login this machine's user set up
|
|
362
|
+
# with `claude /login` (or `claude auth login`), which Claude Code keeps
|
|
363
|
+
# under ~/.claude or in the keychain. The dashboard never reads, copies
|
|
364
|
+
# or stores it; it only asks `claude auth status` whether there is one.
|
|
365
|
+
# That login is the dashboard user's own, so this works with the :local
|
|
366
|
+
# sandbox backend only, and other backends refuse Claude Code sessions.
|
|
367
|
+
# @return [Symbol] :api_key or :local_login
|
|
368
|
+
attr_reader :claude_code_auth
|
|
369
|
+
|
|
370
|
+
def claude_code_auth=(value)
|
|
371
|
+
mode = value.to_s.to_sym
|
|
372
|
+
unless CLAUDE_CODE_AUTH_MODES.include?(mode)
|
|
373
|
+
raise ArgumentError, "ActionAgent.claude_code_auth must be :api_key or :local_login, not #{value.inspect}"
|
|
374
|
+
end
|
|
375
|
+
|
|
376
|
+
@claude_code_auth = mode
|
|
377
|
+
end
|
|
378
|
+
|
|
307
379
|
# Whether the dashboard may execute agents against real providers.
|
|
308
380
|
# Disable to run the dashboard as a read-only observability surface.
|
|
309
381
|
# @return [Boolean]
|
|
@@ -415,6 +487,20 @@ module ActionAgent
|
|
|
415
487
|
# @return [Boolean]
|
|
416
488
|
attr_accessor :encrypt_credentials
|
|
417
489
|
|
|
490
|
+
# The GitHub OAuth App the dashboard's "Connect GitHub" flow authorizes
|
|
491
|
+
# against (Settings -> Integrations). Unset, each falls back to
|
|
492
|
+
# GITHUB_CLIENT_ID / GITHUB_CLIENT_SECRET, and the dashboard offers no
|
|
493
|
+
# connection when neither is present. Register the app's callback URL as
|
|
494
|
+
# <mount>/api/github_connection/callback.
|
|
495
|
+
# @return [String, nil]
|
|
496
|
+
attr_writer :github_client_id, :github_client_secret
|
|
497
|
+
|
|
498
|
+
# OAuth scopes requested on connect. +repo+ reaches private repositories
|
|
499
|
+
# so a sandbox can clone them; narrow it to "public_repo read:user" for
|
|
500
|
+
# public checkouts only.
|
|
501
|
+
# @return [String]
|
|
502
|
+
attr_accessor :github_oauth_scopes
|
|
503
|
+
|
|
418
504
|
# MCP servers the host app itself serves or connects, appended to the
|
|
419
505
|
# built-in catalog (MCPCatalog) so the MCP Services view lists them and
|
|
420
506
|
# telemetry traffic attributes to them. Each entry is a hash shaped like
|
|
@@ -452,6 +538,17 @@ module ActionAgent
|
|
|
452
538
|
# @return [Boolean]
|
|
453
539
|
attr_accessor :mcp_schema_tools
|
|
454
540
|
|
|
541
|
+
# Whether the MCP facade (POST <mount>/mcp) offers the dashboard's own
|
|
542
|
+
# evaluation and telemetry tools — evaluations_list, evaluations_get,
|
|
543
|
+
# evaluations_run, evaluation_runs_get, evaluation_runs_compare,
|
|
544
|
+
# traces_search, traces_get — so a client's coding harness can run an
|
|
545
|
+
# agent's evaluations and read its traces while it edits the agent. Each
|
|
546
|
+
# reads under the key's owner, as the dashboard's JSON API reads under the
|
|
547
|
+
# signed-in owner. On by default; set it to false to leave the facade
|
|
548
|
+
# serving agents and schema tools only.
|
|
549
|
+
# @return [Boolean]
|
|
550
|
+
attr_accessor :mcp_dashboard_tools
|
|
551
|
+
|
|
455
552
|
# Directory scanned for SchemaTools subclasses when {#schema_tools} is
|
|
456
553
|
# unset. Relative to the host's root. Set to nil to disable discovery and
|
|
457
554
|
# require an explicit declaration. Classes built at runtime with
|
|
@@ -471,6 +568,20 @@ module ActionAgent
|
|
|
471
568
|
# Returns whether multi-tenant mode is enabled.
|
|
472
569
|
#
|
|
473
570
|
# @return [Boolean]
|
|
571
|
+
def github_client_id
|
|
572
|
+
@github_client_id.presence || ENV["GITHUB_CLIENT_ID"].presence
|
|
573
|
+
end
|
|
574
|
+
|
|
575
|
+
def github_client_secret
|
|
576
|
+
@github_client_secret.presence || ENV["GITHUB_CLIENT_SECRET"].presence
|
|
577
|
+
end
|
|
578
|
+
|
|
579
|
+
# Whether the GitHub OAuth flow can run on this install.
|
|
580
|
+
# @return [Boolean]
|
|
581
|
+
def github_oauth_configured?
|
|
582
|
+
github_client_id.present? && github_client_secret.present?
|
|
583
|
+
end
|
|
584
|
+
|
|
474
585
|
def multi_tenant?
|
|
475
586
|
@multi_tenant == true
|
|
476
587
|
end
|
|
@@ -482,6 +593,14 @@ module ActionAgent
|
|
|
482
593
|
@mcp_schema_tools != false
|
|
483
594
|
end
|
|
484
595
|
|
|
596
|
+
# Whether the MCP facade serves the dashboard's evaluation and telemetry
|
|
597
|
+
# tools.
|
|
598
|
+
#
|
|
599
|
+
# @return [Boolean]
|
|
600
|
+
def mcp_dashboard_tools?
|
|
601
|
+
@mcp_dashboard_tools != false
|
|
602
|
+
end
|
|
603
|
+
|
|
485
604
|
# Returns whether agent execution is permitted.
|
|
486
605
|
#
|
|
487
606
|
# @return [Boolean]
|
|
@@ -499,6 +618,19 @@ module ActionAgent
|
|
|
499
618
|
Rails.env.local?
|
|
500
619
|
end
|
|
501
620
|
|
|
621
|
+
# Whether the :local sandbox backend may run on this install.
|
|
622
|
+
# @return [Boolean]
|
|
623
|
+
def local_sandboxes_enabled?
|
|
624
|
+
return @local_sandboxes_enabled == true unless @local_sandboxes_enabled.nil?
|
|
625
|
+
|
|
626
|
+
Rails.env.local?
|
|
627
|
+
end
|
|
628
|
+
|
|
629
|
+
# @return [Pathname]
|
|
630
|
+
def local_sandbox_root
|
|
631
|
+
Pathname.new(@local_sandbox_root.presence || Rails.root.join("tmp", "action_agent", "sandboxes"))
|
|
632
|
+
end
|
|
633
|
+
|
|
502
634
|
# Tells the host app that +owner+ performed +kind+. Never raises: a
|
|
503
635
|
# bookkeeping failure must not fail the action that was already taken.
|
|
504
636
|
def record_usage(owner, kind)
|
|
@@ -633,6 +765,14 @@ module ActionAgent
|
|
|
633
765
|
@quota_checker = nil
|
|
634
766
|
@provider_credentials_resolver = nil
|
|
635
767
|
@sandbox_backends = {}
|
|
768
|
+
@local_sandboxes_enabled = nil
|
|
769
|
+
@local_sandbox_root = nil
|
|
770
|
+
@local_sandbox_boot_timeout = 600
|
|
771
|
+
@claude_code_command = "claude"
|
|
772
|
+
@claude_code_permission_mode = "acceptEdits"
|
|
773
|
+
@claude_code_max_turns = nil
|
|
774
|
+
@claude_code_timeout = 1800
|
|
775
|
+
@claude_code_auth = :api_key
|
|
636
776
|
@execution_enabled = true
|
|
637
777
|
@run_host_agent_classes = false
|
|
638
778
|
@assistant_enabled = nil
|
|
@@ -641,6 +781,9 @@ module ActionAgent
|
|
|
641
781
|
@table_name_prefix = "active_agent_"
|
|
642
782
|
@agent_polymorphic_name = nil
|
|
643
783
|
@encrypt_credentials = true
|
|
784
|
+
@github_client_id = nil
|
|
785
|
+
@github_client_secret = nil
|
|
786
|
+
@github_oauth_scopes = "repo read:user"
|
|
644
787
|
@trace_retention = nil
|
|
645
788
|
@trace_owner_resolver = nil
|
|
646
789
|
@usage_recorder = nil
|
|
@@ -653,6 +796,7 @@ module ActionAgent
|
|
|
653
796
|
@schema_tools = nil
|
|
654
797
|
@schema_tools_path = "app/agent_tools"
|
|
655
798
|
@mcp_schema_tools = nil
|
|
799
|
+
@mcp_dashboard_tools = nil
|
|
656
800
|
end
|
|
657
801
|
|
|
658
802
|
# Host-declared schema tool classes, resolved from names and filtered to
|