actionagent 1.7.2 → 1.8.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (59) hide show
  1. checksums.yaml +4 -4
  2. data/app/assets/builds/action_agent.css +1 -1
  3. data/app/assets/builds/action_agent.js +59 -54
  4. data/app/controllers/action_agent/api/agents_controller.rb +19 -3
  5. data/app/controllers/action_agent/api/code_sessions_controller.rb +156 -0
  6. data/app/controllers/action_agent/api/evaluations_controller.rb +27 -130
  7. data/app/controllers/action_agent/api/github_connections_controller.rb +147 -0
  8. data/app/controllers/action_agent/api/mcp_controller.rb +38 -5
  9. data/app/controllers/action_agent/api/mcp_servers_controller.rb +10 -1
  10. data/app/controllers/action_agent/api/provider_keys_controller.rb +54 -3
  11. data/app/controllers/action_agent/api/provider_models_controller.rb +10 -6
  12. data/app/controllers/action_agent/api/sandboxes_controller.rb +97 -6
  13. data/app/controllers/concerns/action_agent/api/evaluation_run_starting.rb +93 -0
  14. data/app/controllers/concerns/action_agent/api/mcp_dashboard_tools.rb +507 -0
  15. data/app/controllers/concerns/action_agent/api/run_sandbox.rb +65 -0
  16. data/app/jobs/action_agent/code_session_job.rb +166 -0
  17. data/app/jobs/action_agent/sandbox_cleanup_job.rb +80 -11
  18. data/app/jobs/action_agent/sandbox_provision_job.rb +122 -14
  19. data/app/jobs/action_agent/sandbox_run_job.rb +10 -3
  20. data/app/models/action_agent/agent.rb +16 -6
  21. data/app/models/action_agent/agent_run.rb +20 -1
  22. data/app/models/action_agent/code_session.rb +141 -0
  23. data/app/models/action_agent/evaluation_run.rb +23 -1
  24. data/app/models/action_agent/github_connection.rb +75 -0
  25. data/app/models/action_agent/provider_key.rb +142 -10
  26. data/app/models/action_agent/sandbox_session.rb +193 -17
  27. data/app/serializers/action_agent/evaluation_serializer.rb +118 -0
  28. data/app/serializers/action_agent/telemetry_trace_serializer.rb +10 -2
  29. data/app/services/action_agent/agent_execution_service.rb +4 -2
  30. data/app/services/action_agent/agent_tool_roster.rb +20 -9
  31. data/app/services/action_agent/claude_code_auth.rb +86 -0
  32. data/app/services/action_agent/dashboard_assistant_service.rb +47 -5
  33. data/app/services/action_agent/evaluation_tool_resolver.rb +18 -0
  34. data/app/services/action_agent/github_client.rb +111 -0
  35. data/app/services/action_agent/local_sandbox_backend.rb +1689 -0
  36. data/app/services/action_agent/local_sandbox_databases.rb +257 -0
  37. data/app/services/action_agent/mcp_client.rb +5 -1
  38. data/app/services/action_agent/mcp_tool_dispatcher.rb +137 -17
  39. data/app/services/action_agent/mock_sandbox_backend.rb +39 -0
  40. data/app/services/action_agent/ollama_host_probe.rb +75 -0
  41. data/app/services/action_agent/payload_bounds.rb +36 -0
  42. data/app/services/action_agent/sandbox_manifest.rb +67 -0
  43. data/app/services/action_agent/sandbox_orchestrator.rb +69 -14
  44. data/app/services/action_agent/scenario_evaluation_runner.rb +50 -4
  45. data/app/services/action_agent/secret_scrubber.rb +37 -0
  46. data/app/services/action_agent/tool_discovery.rb +19 -5
  47. data/config/routes.rb +22 -3
  48. data/lib/action_agent/engine.rb +1 -0
  49. data/lib/action_agent/version.rb +1 -1
  50. data/lib/action_agent.rb +147 -3
  51. data/lib/generators/action_agent/install_generator.rb +30 -3
  52. data/lib/generators/action_agent/templates/action_agent.rb.erb +44 -0
  53. data/lib/generators/action_agent/templates/add_provider_key_api_key.rb.erb +25 -0
  54. data/lib/generators/action_agent/templates/create_active_agent_code_sessions.rb.erb +59 -0
  55. data/lib/generators/action_agent/templates/create_active_agent_dashboard_tables.rb.erb +2 -0
  56. data/lib/generators/action_agent/templates/create_active_agent_github_connections.rb.erb +61 -0
  57. data/lib/tasks/claude_code.rake +16 -0
  58. data/lib/tasks/sandbox.rake +26 -0
  59. metadata +23 -1
@@ -9,6 +9,8 @@ module ActionAgent
9
9
  # and solid_agent records (ToolDiscovery), servers an agent declares in
10
10
  # its configuration, and the default catalog (MCPCatalog). An install
11
11
  # therefore sees both what it is already using and what it could turn on.
12
+ # The caller's live checkout sandbox runtimes are listed with the catalog,
13
+ # so the one started from Settings -> Integrations shows up here too.
12
14
  class MCPServersController < BaseController
13
15
  before_action :require_owner!
14
16
  # Launching provisions a sandbox and runs a server in it, so it answers
@@ -108,8 +110,15 @@ module ActionAgent
108
110
 
109
111
  private
110
112
 
113
+ # Runtimes are the caller's own: a listing carries no token, but which
114
+ # repositories another owner has running is still theirs to know.
111
115
  def discovery
112
- ToolDiscovery.new(traces: owned_traces, agents: owner_agents, hours: window_hours)
116
+ ToolDiscovery.new(
117
+ traces: owned_traces,
118
+ agents: owner_agents,
119
+ hours: window_hours,
120
+ runtimes: SandboxSession.runtime_server_listings(owned(SandboxSession))
121
+ )
113
122
  end
114
123
 
115
124
  # The catalog names this dashboard's own MCP endpoint as "<mount>/mcp",
@@ -4,7 +4,9 @@ module ActionAgent
4
4
  module Api
5
5
  # Per-account LLM provider credentials (Settings -> Provider API Keys).
6
6
  # API keys are write-only: responses carry a masked hint, never the key.
7
- # Ollama's credential is a host URL and is echoed back in full.
7
+ # Ollama's credential is a host URL and is echoed back in full; its
8
+ # optional API key (remote servers) is masked like the others. Claude
9
+ # Code's connection API key is stored here too, write-only like a key.
8
10
  class ProviderKeysController < BaseController
9
11
  before_action :require_owner!
10
12
 
@@ -20,16 +22,42 @@ module ActionAgent
20
22
  end
21
23
 
22
24
  # POST /api/provider_keys — upserts the credential for a provider.
25
+ # Host-based providers may also carry an optional api_key; omitting it
26
+ # keeps the stored one, sending an empty string clears it.
23
27
  def create
24
28
  provider = params.require(:provider)
25
29
  credential = params.require(:credential)
26
30
 
27
31
  record = owned(ProviderKey).find_or_initialize_by(provider: provider)
28
- record.update!(credential: credential)
32
+ attributes = { credential: credential }
33
+ attributes[:api_key] = params[:api_key].presence if params.key?(:api_key) && record.host_based?
34
+ record.update!(**attributes)
29
35
 
30
36
  render json: { provider_key: serialize(provider, record) }, status: :created
31
37
  end
32
38
 
39
+ # POST /api/provider_keys/test — checks that a host-based provider is
40
+ # reachable and lists the models it serves. Tests the submitted host
41
+ # (and api_key) when given, so a URL can be checked before saving;
42
+ # otherwise the stored credential, else the host app's default. Never
43
+ # persists anything.
44
+ def test
45
+ provider = params.require(:provider)
46
+ unless ProviderKey::HOST_PROVIDERS.include?(provider)
47
+ return render json: { error: "#{provider} is not a host-based provider" }, status: :unprocessable_entity
48
+ end
49
+
50
+ stored = owned(ProviderKey).find_by(provider: provider)
51
+ host = params[:credential].presence || stored&.credential || platform_host(provider)
52
+ api_key = params.key?(:api_key) ? params[:api_key].presence : stored&.api_key
53
+
54
+ if host.blank?
55
+ return render json: { ok: false, host: nil, models: [], latency_ms: nil, error: "No host configured" }
56
+ end
57
+
58
+ render json: OllamaHostProbe.call(host: host, api_key: api_key).to_h
59
+ end
60
+
33
61
  # DELETE /api/provider_keys/:provider
34
62
  def destroy
35
63
  owned(ProviderKey).find_by!(provider: params[:provider]).destroy!
@@ -38,12 +66,35 @@ module ActionAgent
38
66
 
39
67
  private
40
68
 
69
+ # The host app's default (config/active_agent.yml, e.g. OLLAMA_HOST)
70
+ # that applies when the owner has not configured their own.
71
+ def platform_host(provider)
72
+ return nil unless ProviderKey::HOST_PROVIDERS.include?(provider)
73
+
74
+ config = ActiveAgent.configuration[provider.to_sym]
75
+ config.respond_to?(:[]) ? config[:host].presence : nil
76
+ rescue StandardError
77
+ nil
78
+ end
79
+
41
80
  def serialize(provider, record)
81
+ host_based = ProviderKey::HOST_PROVIDERS.include?(provider)
82
+
42
83
  {
43
84
  provider: provider,
44
- host_based: ProviderKey::HOST_PROVIDERS.include?(provider),
85
+ host_based: host_based,
86
+ # "key", "host", or "connection" (Settings -> Integrations rather
87
+ # than Provider API Keys).
88
+ kind: ProviderKey.kind_of_provider(provider),
45
89
  configured: record.present?,
46
90
  hint: record&.display_hint,
91
+ api_key_configured: record&.api_key? || false,
92
+ api_key_hint: record&.api_key_hint,
93
+ platform_default: host_based && record.nil? ? platform_host(provider) : nil,
94
+ # A Claude Code connection still holding a Claude subscription token
95
+ # from an earlier version: never used, and the UI asks for an API
96
+ # key in its place.
97
+ needs_replacing: record.present? && record.needs_replacing?,
47
98
  updated_at: record&.updated_at&.iso8601
48
99
  }
49
100
  end
@@ -67,16 +67,20 @@ module ActionAgent
67
67
  owned(ProviderKey).find_by(provider: provider)
68
68
  end
69
69
 
70
+ # Same probe as Settings -> "Test connection", so the builder's dropdown
71
+ # and the settings page agree on what the host serves (including the
72
+ # optional Bearer key for remote servers).
70
73
  def live_ollama_models
71
74
  host = ollama_host
72
75
  return nil unless host
73
76
 
74
- data = fetch_json(URI.join("#{host.chomp('/')}/", "models"))
75
- ids = Array(data&.dig("data")).filter_map { |model| model["id"] }
76
- [ ids.sort, "live" ] if ids.any?
77
- rescue StandardError => e
78
- Rails.logger.warn("[ProviderModels] ollama lookup failed: #{e.message}")
79
- nil
77
+ result = OllamaHostProbe.call(host: host, api_key: owner_provider_key("ollama")&.api_key)
78
+ unless result.ok
79
+ Rails.logger.warn("[ProviderModels] ollama lookup failed: #{result.error}")
80
+ return nil
81
+ end
82
+
83
+ [ result.models, "live" ] if result.models.any?
80
84
  end
81
85
 
82
86
  # Queries the Anthropic Models API with the account's key (newest first,
@@ -12,6 +12,9 @@ module ActionAgent
12
12
  # development, so it bypassed that safeguard too.
13
13
 
14
14
  before_action :require_execution_enabled!, only: [ :run, :compare ]
15
+ # A checkout runs the owner's code (setup, server): the same gates as
16
+ # running an agent, like MCPServersController#launch.
17
+ before_action :gate_checkout!, only: [ :create ]
15
18
  before_action :enforce_execution_quota!, only: [ :compare ]
16
19
  before_action :set_sandbox, only: [ :show, :run, :destroy ]
17
20
 
@@ -83,16 +86,32 @@ module ActionAgent
83
86
  end
84
87
 
85
88
  # GET /api/sandboxes
86
- # List available sandbox types and sample tasks
89
+ # List available sandbox types and sample tasks, and the caller's own
90
+ # sandboxes (?sandbox_type= narrows them), with what the Settings ->
91
+ # Integrations view needs to offer Claude Code sessions in a checkout.
87
92
  def index
88
93
  render json: {
89
94
  sandbox_types: SandboxSession::SANDBOX_TYPES,
90
95
  free_tier_limits: SandboxSession::FREE_TIER_LIMITS,
91
96
  templates: free_tier_templates,
92
- sample_tasks: sample_tasks
97
+ sample_tasks: sample_tasks,
98
+ sandboxes: listed_sandboxes.map(&:summary),
99
+ code_sessions_supported: code_sessions_supported?,
100
+ **claude_code_status
93
101
  }
94
102
  end
95
103
 
104
+ # Refuses a checkout the owner may not start. Usage is recorded by
105
+ # #create once the sandbox saved: a request refused for its own
106
+ # content (a repository that is not selected, say) runs nothing, so it
107
+ # must not spend a plan run.
108
+ def gate_checkout!
109
+ return unless checkout_requested?
110
+
111
+ require_execution_enabled!
112
+ enforce_execution_quota! unless performed?
113
+ end
114
+
96
115
  # POST /api/sandboxes
97
116
  # Create a new sandbox session, owned by whoever opened it.
98
117
  def create
@@ -100,11 +119,17 @@ module ActionAgent
100
119
  # Guarded: the association only exists when the host app configured a
101
120
  # user model, and a single-user install configures none.
102
121
  @sandbox.user = current_user if @sandbox.respond_to?(:user=)
122
+ # A checkout is validated against the owner's GitHub connection, which
123
+ # an account-owned install finds through the account.
124
+ @sandbox.account_id = current_account.id if current_account && @sandbox.has_attribute?(:account_id)
103
125
  @sandbox.agent_template = AgentTemplate.find_by(slug: params[:template_slug]) if params[:template_slug]
104
126
 
105
127
  if @sandbox.save
106
128
  @sandbox.provision!
107
129
  @sandbox.reload # Reload to get updated status after provisioning
130
+ # Counted once the checkout exists, as MCPServersController#launch
131
+ # counts a launched server.
132
+ record_execution_usage if checkout_requested?
108
133
  render json: { sandbox: @sandbox.summary }, status: :created
109
134
  else
110
135
  render json: { errors: @sandbox.errors.full_messages }, status: :unprocessable_entity
@@ -162,20 +187,86 @@ module ActionAgent
162
187
  end
163
188
 
164
189
  # DELETE /api/sandboxes/:session_id
165
- # End sandbox session
190
+ # End sandbox session. Expiring it enqueues SandboxCleanupJob, which
191
+ # terminates whatever the backend runs for it (a checkout's processes
192
+ # included); one still provisioning is released by
193
+ # SandboxProvisionJob when its backend returns.
166
194
  def destroy
167
195
  @sandbox.expire!
168
- render json: { deleted: true }
196
+ render json: { deleted: true, sandbox: @sandbox.summary }
169
197
  end
170
198
 
171
199
  private
172
200
 
173
201
  def set_sandbox
174
- @sandbox = owned(SandboxSession).find_by!(session_id: params[:id])
202
+ @sandbox = account_scoped(owned(SandboxSession)).find_by!(session_id: params[:id])
203
+ end
204
+
205
+ def checkout_requested?
206
+ params[:sandbox_type].to_s == "app_runtime"
207
+ end
208
+
209
+ # A checkout runs on its account's GitHub token and Claude Code
210
+ # credential, so in a multi-tenant install it is listed, shown and
211
+ # stopped only within the caller's current account, as
212
+ # CodeSessionsController finds it: listing another account's checkout
213
+ # as Ready offered a Claude Code panel that then answered 404. Other
214
+ # sandbox types are the caller's own wherever they were opened.
215
+ def account_scoped(scope)
216
+ return scope unless current_account
217
+
218
+ scope.where.not(sandbox_type: "app_runtime").or(scope.where(account_id: current_account.id))
219
+ end
220
+
221
+ # Whether the configured backend can run Claude Code sessions. A
222
+ # backend that cannot even be loaded (a misspelled class in
223
+ # ActionAgent.sandbox_backends, or a class file requiring an SDK the
224
+ # host doesn't bundle, which raises LoadError) cannot, and must not
225
+ # take the rest of this listing down with it.
226
+ #
227
+ # Nor can one whose Claude Code authentication does not work there
228
+ # (ActionAgent.claude_code_auth = :local_login needs :local).
229
+ def code_sessions_supported?
230
+ orchestrator = SandboxOrchestrator.new
231
+ orchestrator.supports?(:code_session) && ClaudeCodeAuth.backend_refusal(orchestrator).nil?
232
+ rescue StandardError, LoadError => e
233
+ Rails.logger.warn("[ActionAgent] sandbox backend unavailable: #{e.message}")
234
+ false
235
+ end
236
+
237
+ # Whether the caller's Claude Code can run sessions, by
238
+ # ClaudeCodeAuth's rule: an API key they connected, or this machine's
239
+ # own login. Never a credential.
240
+ #
241
+ # claude_code_auth "api_key" | "local_login"
242
+ # claude_code_connected Boolean
243
+ # claude_code_login { logged_in:, auth_method: } (local_login only)
244
+ #
245
+ # The key is found through its own owner column, as
246
+ # SandboxSession#runtime_environment finds the credential it hands a
247
+ # checkout: a provider key is account-owned before user-owned.
248
+ def claude_code_status
249
+ status = ClaudeCodeAuth.status(owned(ProviderKey))
250
+ {
251
+ claude_code_auth: status[:mode],
252
+ claude_code_connected: status[:connected],
253
+ claude_code_login: status[:login]
254
+ }.compact
255
+ end
256
+
257
+ # The caller's sandboxes that have not expired, newest first. A failed
258
+ # one stays listed so its error can be read; one past its expiry but
259
+ # not reaped yet stays listed so it can still be stopped.
260
+ def listed_sandboxes
261
+ scope = account_scoped(owned(SandboxSession)).where.not(status: :expired).recent.limit(20)
262
+ type = params[:sandbox_type]
263
+ type.is_a?(String) && type.present? ? scope.by_type(type) : scope
175
264
  end
176
265
 
266
+ # An app_runtime sandbox also names the checkout: one of the owner's
267
+ # selected GitHub repositories and, optionally, a ref.
177
268
  def sandbox_params
178
- params.permit(:sandbox_type)
269
+ params.permit(:sandbox_type, :repository, :repository_ref)
179
270
  end
180
271
 
181
272
  def free_tier_templates
@@ -0,0 +1,93 @@
1
+ # frozen_string_literal: true
2
+
3
+ module ActionAgent
4
+ module Api
5
+ # Starting an evaluation run on the caller's behalf, shared by the
6
+ # evaluations API (POST /api/evaluations/:id/run) and the MCP facade's
7
+ # `evaluations_run` tool so both check a run the same way.
8
+ #
9
+ # A scenario suite accepts a selection that narrows the run:
10
+ # - scenario_ids: the scenarios to replay, by id
11
+ # - keys: the scenarios to replay, by key
12
+ # - group: one scenario group
13
+ # - models: the candidate models, as an array or a comma-separated string
14
+ # - sandbox_id: a checkout sandbox of the caller's that every replay reaches (see RunSandbox)
15
+ module EvaluationRunStarting
16
+ extend ActiveSupport::Concern
17
+
18
+ include RunSandbox
19
+
20
+ private
21
+
22
+ # Starts a run of +evaluation+ and returns it. A scenario suite replays
23
+ # through the provider once per scenario and model, so it runs in the
24
+ # background and the run comes back pending. A generation-sampling
25
+ # evaluation scores recorded data and finishes inline.
26
+ #
27
+ # A failure is recorded on the run rather than raised, so the caller
28
+ # always gets a run to report.
29
+ #
30
+ # @return [EvaluationRun]
31
+ def start_evaluation_run(evaluation, selection)
32
+ return evaluation.run_later!(**selection) if evaluation.scenario_suite?
33
+
34
+ evaluation.run!
35
+ rescue StandardError => e
36
+ Rails.logger.warn(
37
+ "[ActionAgent] evaluation #{evaluation.id} run failed: #{e.class}: #{e.message}"
38
+ )
39
+ # EvaluationRunnerService records the failure on the run before
40
+ # re-raising. A failure before the run record existed is recorded here.
41
+ evaluation.evaluation_runs.recent.first ||
42
+ evaluation.evaluation_runs.create!(status: :failed, error_message: e.message, completed_at: Time.current)
43
+ end
44
+
45
+ # Returns the selection +source+ (params or tool arguments) asks for,
46
+ # without the sandbox, which evaluation_run_sandbox checks separately.
47
+ def evaluation_run_selection(source)
48
+ models = source[:models]
49
+ models = models.to_s.split(",") unless models.is_a?(Array)
50
+
51
+ {
52
+ scenario_ids: Array(source[:scenario_ids]).map(&:to_s).reject(&:blank?),
53
+ keys: Array(source[:keys]).map(&:to_s).reject(&:blank?),
54
+ group: source[:group].to_s.presence,
55
+ models: models.map(&:to_s).map(&:strip).reject(&:blank?)
56
+ }.compact_blank
57
+ end
58
+
59
+ # Returns the checked sandbox +sandbox_id+ names for a run of
60
+ # +evaluation+, or nil when none was asked for. Raises RunSandbox::Refused.
61
+ #
62
+ # Only a scenario suite's own replay executes the agent: a sampling run
63
+ # scores recorded generations, and a host adapter replays in the host's
64
+ # runtime, where no dashboard dispatcher runs.
65
+ #
66
+ # @return [SandboxSession, nil]
67
+ def evaluation_run_sandbox(evaluation, sandbox_id)
68
+ return nil if sandbox_id.blank?
69
+
70
+ unless evaluation.scenario_suite?
71
+ raise RunSandbox::Refused, "Only a scenario evaluation runs the agent; this one scores recorded generations, " \
72
+ "so it cannot run against a sandbox"
73
+ end
74
+ if scenario_evaluation_adapter?(evaluation)
75
+ raise RunSandbox::Refused, "This install replays scenarios through its own adapter, which cannot reach a " \
76
+ "dashboard sandbox"
77
+ end
78
+
79
+ run_sandbox_for(evaluation.agent, sandbox_id)
80
+ end
81
+
82
+ # Whether a run of +evaluation+ cannot replay its scenarios: its agent
83
+ # is observed (read-only), and no host adapter replays it instead.
84
+ def unexecutable_scenario_run?(evaluation)
85
+ evaluation.agent.observed? && !scenario_evaluation_adapter?(evaluation)
86
+ end
87
+
88
+ def scenario_evaluation_adapter?(evaluation)
89
+ ActionAgent.scenario_evaluation_adapter_resolver&.call(evaluation).respond_to?(:call)
90
+ end
91
+ end
92
+ end
93
+ end