actionagent 1.7.2 → 1.8.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (59) hide show
  1. checksums.yaml +4 -4
  2. data/app/assets/builds/action_agent.css +1 -1
  3. data/app/assets/builds/action_agent.js +59 -54
  4. data/app/controllers/action_agent/api/agents_controller.rb +19 -3
  5. data/app/controllers/action_agent/api/code_sessions_controller.rb +156 -0
  6. data/app/controllers/action_agent/api/evaluations_controller.rb +27 -130
  7. data/app/controllers/action_agent/api/github_connections_controller.rb +147 -0
  8. data/app/controllers/action_agent/api/mcp_controller.rb +38 -5
  9. data/app/controllers/action_agent/api/mcp_servers_controller.rb +10 -1
  10. data/app/controllers/action_agent/api/provider_keys_controller.rb +54 -3
  11. data/app/controllers/action_agent/api/provider_models_controller.rb +10 -6
  12. data/app/controllers/action_agent/api/sandboxes_controller.rb +97 -6
  13. data/app/controllers/concerns/action_agent/api/evaluation_run_starting.rb +93 -0
  14. data/app/controllers/concerns/action_agent/api/mcp_dashboard_tools.rb +507 -0
  15. data/app/controllers/concerns/action_agent/api/run_sandbox.rb +65 -0
  16. data/app/jobs/action_agent/code_session_job.rb +166 -0
  17. data/app/jobs/action_agent/sandbox_cleanup_job.rb +80 -11
  18. data/app/jobs/action_agent/sandbox_provision_job.rb +122 -14
  19. data/app/jobs/action_agent/sandbox_run_job.rb +10 -3
  20. data/app/models/action_agent/agent.rb +16 -6
  21. data/app/models/action_agent/agent_run.rb +20 -1
  22. data/app/models/action_agent/code_session.rb +141 -0
  23. data/app/models/action_agent/evaluation_run.rb +23 -1
  24. data/app/models/action_agent/github_connection.rb +75 -0
  25. data/app/models/action_agent/provider_key.rb +142 -10
  26. data/app/models/action_agent/sandbox_session.rb +193 -17
  27. data/app/serializers/action_agent/evaluation_serializer.rb +118 -0
  28. data/app/serializers/action_agent/telemetry_trace_serializer.rb +10 -2
  29. data/app/services/action_agent/agent_execution_service.rb +4 -2
  30. data/app/services/action_agent/agent_tool_roster.rb +20 -9
  31. data/app/services/action_agent/claude_code_auth.rb +86 -0
  32. data/app/services/action_agent/dashboard_assistant_service.rb +47 -5
  33. data/app/services/action_agent/evaluation_tool_resolver.rb +18 -0
  34. data/app/services/action_agent/github_client.rb +111 -0
  35. data/app/services/action_agent/local_sandbox_backend.rb +1689 -0
  36. data/app/services/action_agent/local_sandbox_databases.rb +257 -0
  37. data/app/services/action_agent/mcp_client.rb +5 -1
  38. data/app/services/action_agent/mcp_tool_dispatcher.rb +137 -17
  39. data/app/services/action_agent/mock_sandbox_backend.rb +39 -0
  40. data/app/services/action_agent/ollama_host_probe.rb +75 -0
  41. data/app/services/action_agent/payload_bounds.rb +36 -0
  42. data/app/services/action_agent/sandbox_manifest.rb +67 -0
  43. data/app/services/action_agent/sandbox_orchestrator.rb +69 -14
  44. data/app/services/action_agent/scenario_evaluation_runner.rb +50 -4
  45. data/app/services/action_agent/secret_scrubber.rb +37 -0
  46. data/app/services/action_agent/tool_discovery.rb +19 -5
  47. data/config/routes.rb +22 -3
  48. data/lib/action_agent/engine.rb +1 -0
  49. data/lib/action_agent/version.rb +1 -1
  50. data/lib/action_agent.rb +147 -3
  51. data/lib/generators/action_agent/install_generator.rb +30 -3
  52. data/lib/generators/action_agent/templates/action_agent.rb.erb +44 -0
  53. data/lib/generators/action_agent/templates/add_provider_key_api_key.rb.erb +25 -0
  54. data/lib/generators/action_agent/templates/create_active_agent_code_sessions.rb.erb +59 -0
  55. data/lib/generators/action_agent/templates/create_active_agent_dashboard_tables.rb.erb +2 -0
  56. data/lib/generators/action_agent/templates/create_active_agent_github_connections.rb.erb +61 -0
  57. data/lib/tasks/claude_code.rake +16 -0
  58. data/lib/tasks/sandbox.rake +26 -0
  59. metadata +23 -1
@@ -0,0 +1,141 @@
1
+ # frozen_string_literal: true
2
+
3
+ module ActionAgent
4
+ # A headless Claude Code session run inside an app_runtime sandbox's
5
+ # checkout: the prompt, the stream-json transcript as it arrived, the
6
+ # outcome Claude Code reported, and the diff the session left in the
7
+ # working tree.
8
+ #
9
+ # The transcript is scrubbed of the sandbox's secrets (its GitHub token and
10
+ # Claude Code credential) before it is stored, and bounded: long strings
11
+ # are truncated and events past MAX_EVENTS are counted rather than kept.
12
+ class CodeSession < ApplicationRecord
13
+ include Ownable
14
+ owned_by :user, :account
15
+
16
+ belongs_to :sandbox_session
17
+
18
+ enum :status, { queued: 0, running: 1, succeeded: 2, failed: 3, cancelled: 4 }
19
+
20
+ MAX_PROMPT_CHARACTERS = 20_000
21
+ MAX_EVENTS = 1_000
22
+ # Per string inside an event: tool results can be whole files.
23
+ MAX_EVENT_STRING = 4_000
24
+ MAX_DIFF_BYTES = 500_000
25
+
26
+ validates :prompt, presence: true, length: { maximum: MAX_PROMPT_CHARACTERS }
27
+
28
+ scope :recent, -> { order(created_at: :desc) }
29
+
30
+ # JSON columns carry no default on MySQL or SQLite (see the migration).
31
+ def events
32
+ Array(super)
33
+ end
34
+
35
+ def finished?
36
+ succeeded? || failed? || cancelled?
37
+ end
38
+
39
+ # Whether an outcome and a diff may still be recorded for this session.
40
+ # finished_at is set once nothing more will be: by CodeSessionJob when it
41
+ # settles a session it ran (whatever the backend returned, raised or
42
+ # refused), or by the cancel of a session that never left the queue. A
43
+ # session cancelled while running is finished at once but settled only
44
+ # when its Claude Code has stopped, so its diff may still come, and its
45
+ # transcript still grow, until then.
46
+ def diff_pending?
47
+ finished_at.nil?
48
+ end
49
+
50
+ # The values that must never be stored: the checkout token and the
51
+ # Claude Code credential this session ran with.
52
+ def secrets
53
+ spec = sandbox_session.checkout_spec rescue nil
54
+ [ spec&.dig(:token), *sandbox_session.runtime_environment.values ].compact
55
+ end
56
+
57
+ # Appends one stream-json event, scrubbed and bounded. Past MAX_EVENTS
58
+ # only the count grows, so a runaway session cannot grow the row without
59
+ # limit.
60
+ def append_event!(event, secrets: self.secrets)
61
+ stored = truncate_strings(SecretScrubber.scrub(event.to_h, secrets))
62
+
63
+ with_lock do
64
+ list = events
65
+ if list.size < MAX_EVENTS
66
+ self.events = list + [ stored ]
67
+ else
68
+ self.dropped_events_count = dropped_events_count.to_i + 1
69
+ end
70
+ save!
71
+ end
72
+ end
73
+
74
+ # Records Claude Code's final "result" event.
75
+ def record_result!(event)
76
+ usage = event["usage"].is_a?(Hash) ? event["usage"] : {}
77
+
78
+ update!(
79
+ result: event["result"].to_s.presence && SecretScrubber.scrub(event["result"].to_s, secrets).truncate(MAX_EVENT_STRING * 4),
80
+ claude_session_id: event["session_id"],
81
+ num_turns: event["num_turns"],
82
+ duration_ms: event["duration_ms"],
83
+ total_cost_usd: event["total_cost_usd"],
84
+ input_tokens: usage["input_tokens"],
85
+ output_tokens: usage["output_tokens"]
86
+ )
87
+ end
88
+
89
+ def diff=(value)
90
+ text = SecretScrubber.scrub(value.to_s, secrets)
91
+ super(text.bytesize > MAX_DIFF_BYTES ? "#{text.byteslice(0, MAX_DIFF_BYTES).scrub}\n… diff truncated" : text)
92
+ end
93
+
94
+ def summary
95
+ {
96
+ id: id,
97
+ sandbox_session_id: sandbox_session.session_id,
98
+ status: status,
99
+ prompt: prompt,
100
+ model: model,
101
+ result: result,
102
+ error_message: error_message,
103
+ num_turns: num_turns,
104
+ duration_ms: duration_ms,
105
+ total_cost_usd: total_cost_usd&.to_f,
106
+ input_tokens: input_tokens,
107
+ output_tokens: output_tokens,
108
+ event_count: events.size + dropped_events_count.to_i,
109
+ # What a client polls until: false once nothing about the session
110
+ # changes any more.
111
+ diff_pending: diff_pending?,
112
+ started_at: started_at&.iso8601,
113
+ finished_at: finished_at&.iso8601,
114
+ created_at: created_at&.iso8601
115
+ }
116
+ end
117
+
118
+ # The summary plus the transcript from event index +after+ onward (for
119
+ # incremental polling) and the diff once finished.
120
+ def details(after: 0)
121
+ after = after.to_i.clamp(0, events.size)
122
+ summary.merge(
123
+ events: events.drop(after),
124
+ events_offset: after,
125
+ dropped_events_count: dropped_events_count.to_i,
126
+ diff: finished? ? diff : nil
127
+ )
128
+ end
129
+
130
+ private
131
+
132
+ def truncate_strings(value)
133
+ case value
134
+ when String then value.length > MAX_EVENT_STRING ? "#{value[0, MAX_EVENT_STRING]}… (truncated)" : value
135
+ when Hash then value.to_h { |key, item| [ key, truncate_strings(item) ] }
136
+ when Array then value.map { |item| truncate_strings(item) }
137
+ else value
138
+ end
139
+ end
140
+ end
141
+ end
@@ -25,6 +25,19 @@ module ActionAgent
25
25
  value.is_a?(Hash) ? value : {}
26
26
  end
27
27
 
28
+ # The checkout sandbox a scenario run replayed against
29
+ # (ScenarioEvaluationRunner records it), as { "session_id", "server_key",
30
+ # "repository", "repository_ref" }; nil for a run against the agent's own
31
+ # servers only.
32
+ def sandbox
33
+ value = selection["sandbox"]
34
+ return value if value.is_a?(Hash)
35
+
36
+ # Still pending: run_later! recorded only what was asked for.
37
+ id = selection["sandbox_id"]
38
+ { "session_id" => id } if id.is_a?(String) && id.present?
39
+ end
40
+
28
41
  # The candidate models a scenario run compared, in the order they were
29
42
  # requested; empty for a generation-sampling run.
30
43
  def models
@@ -211,7 +224,8 @@ module ActionAgent
211
224
  "evaluation" => evaluation.name,
212
225
  "agent" => evaluation.agent&.name,
213
226
  "run" => id,
214
- "finished" => completed_at&.iso8601
227
+ "finished" => completed_at&.iso8601,
228
+ "sandbox" => sandbox_label
215
229
  }.compact.merge(report_metadata),
216
230
  verdict: recorded_verdict,
217
231
  judge_label: judge_label,
@@ -226,6 +240,14 @@ module ActionAgent
226
240
  ModelSpec = ActiveAgent::Evals::ModelSpec
227
241
  private_constant :ModelSpec
228
242
 
243
+ # How the report's header names the sandbox: its checkout and session.
244
+ def sandbox_label
245
+ return nil unless sandbox
246
+
247
+ checkout = [ sandbox["repository"], sandbox["repository_ref"] ].compact_blank.join("@")
248
+ [ checkout.presence, "sandbox #{sandbox["session_id"].to_s.first(8)}" ].compact.join(" · ")
249
+ end
250
+
229
251
  # The candidate specs the run was asked to compare, keyed by
230
252
  # [provider, model] in the order requested. ScenarioEvaluationRunner
231
253
  # persists each ModelSpec#to_h in `selection`, and it is that label —
@@ -0,0 +1,75 @@
1
+ # frozen_string_literal: true
2
+
3
+ module ActionAgent
4
+ # An owner's GitHub OAuth grant (Settings -> Integrations) and the
5
+ # repositories they made available to the workspace.
6
+ #
7
+ # The access token is encrypted at rest like a provider key and never
8
+ # rendered back to the client. +repositories+ holds only what GitHub
9
+ # itself listed for this token when the owner chose them (see
10
+ # Api::GithubConnectionsController#update), so a checkout sandbox can
11
+ # trust a name found here without asking GitHub again.
12
+ class GithubConnection < ApplicationRecord
13
+ include Ownable
14
+ owned_by :account, :user
15
+
16
+ encrypts :access_token if ActionAgent.encrypt_credentials
17
+
18
+ validates :access_token, :github_user_id, :login, presence: true
19
+ # One connection per owner; which column that means depends on the
20
+ # configured mode, so it is checked at validation time.
21
+ validate :unique_within_owner
22
+
23
+ # JSON columns carry no default on MySQL or SQLite (see the migration), so
24
+ # an unset column reads as nil.
25
+ def repositories
26
+ Array(super)
27
+ end
28
+
29
+ def repository_names
30
+ repositories.map { |repo| repo["full_name"] }
31
+ end
32
+
33
+ def repository(full_name)
34
+ repositories.find { |repo| repo["full_name"].casecmp?(full_name.to_s) }
35
+ end
36
+
37
+ def client
38
+ GithubClient.new(access_token)
39
+ end
40
+
41
+ # The checkout a sandbox backend clones: repository, ref, and an
42
+ # authenticated HTTPS clone URL. Carries the token — hand it to a backend,
43
+ # never to a response.
44
+ def checkout_spec(full_name, ref: nil)
45
+ repo = repository(full_name) or raise ArgumentError, "#{full_name} is not an available repository"
46
+
47
+ {
48
+ repository: repo["full_name"],
49
+ ref: ref.presence || repo["default_branch"],
50
+ clone_url: "https://github.com/#{repo["full_name"]}.git",
51
+ username: "x-access-token",
52
+ token: access_token
53
+ }
54
+ end
55
+
56
+ def as_summary
57
+ {
58
+ login: login,
59
+ avatar_url: avatar_url,
60
+ scopes: scopes.to_s.split(/[\s,]+/).reject(&:blank?),
61
+ repositories: repositories,
62
+ connected_at: created_at&.iso8601,
63
+ updated_at: updated_at&.iso8601
64
+ }
65
+ end
66
+
67
+ private
68
+
69
+ def unique_within_owner
70
+ siblings = self.class.for_owner(owner)
71
+ siblings = siblings.where.not(id: id) if persisted?
72
+ errors.add(:base, "GitHub is already connected") if siblings.exists?
73
+ end
74
+ end
75
+ end
@@ -5,22 +5,52 @@ module ActionAgent
5
5
  # Generation runs (AgentExecutionService and the evaluation LLM judge)
6
6
  # prefer these over the platform's ENV-configured keys, so users can run
7
7
  # agents with their own OpenAI/Anthropic/OpenRouter accounts — or point
8
- # ollama at their own host (e.g. a tunnel to a locally running instance).
8
+ # ollama at their own host: a locally running instance, a tunnel to one, or
9
+ # a remote/cloud server that additionally needs a Bearer API key.
9
10
  #
10
- # The credential is encrypted at rest with Active Record Encryption. API
11
- # keys are never rendered back to the client — only a masked hint; ollama
12
- # hosts are not secret and are shown in full (see #display_hint).
11
+ # The credential (and the optional api_key) is encrypted at rest with
12
+ # Active Record Encryption. API keys are never rendered back to the client —
13
+ # only a masked hint; ollama hosts are not secret and are shown in full
14
+ # (see #display_hint).
15
+ #
16
+ # A connection credential (Claude Code) is stored the same way but is not a
17
+ # generation provider: no agent runs "on" it. It is handed to runtimes that
18
+ # need it — a checkout sandbox runs Claude Code sessions with it — through
19
+ # #runtime_environment. For Claude Code that is an Anthropic API key only
20
+ # (see CLAUDE_CODE_CREDENTIAL).
13
21
  class ProviderKey < ApplicationRecord
14
22
  # Providers that authenticate with an API key.
15
23
  KEY_PROVIDERS = %w[openai anthropic openrouter].freeze
16
- # Providers addressed by host URL instead of a key.
24
+ # Providers addressed by host URL instead of a key (with an optional key
25
+ # for remote servers).
17
26
  HOST_PROVIDERS = %w[ollama].freeze
18
- PROVIDERS = (KEY_PROVIDERS + HOST_PROVIDERS).freeze
27
+ # Tools connected with a credential, configured beside the providers
28
+ # (Settings -> Integrations) but never offered to the agent builder.
29
+ CONNECTION_PROVIDERS = %w[claude_code].freeze
30
+ PROVIDERS = (KEY_PROVIDERS + HOST_PROVIDERS + CONNECTION_PROVIDERS).freeze
31
+
32
+ # Only an Anthropic API key (sk-ant-api03-…, from the Claude Console or
33
+ # a supported cloud provider). Anthropic does not let third-party
34
+ # products collect, store or route requests through Claude.ai
35
+ # subscription credentials (a `claude setup-token` token, sk-ant-oat…):
36
+ # https://code.claude.com/docs/en/legal-and-compliance.md. A developer
37
+ # who wants their own subscription on their own machine uses
38
+ # ActionAgent.claude_code_auth = :local_login instead, where the
39
+ # dashboard never touches the credential.
40
+ CLAUDE_CODE_CREDENTIAL = /\Ask-ant-api\d{2}-[A-Za-z0-9_-]+\z/
41
+ # A Claude subscription token, as earlier versions stored. Recognized so
42
+ # a stored one is never handed out (see #needs_replacing?).
43
+ SUBSCRIPTION_TOKEN_PREFIX = "sk-ant-oat"
19
44
 
20
45
  include Ownable
21
46
  owned_by :account, :user
22
47
 
23
- encrypts :credential if ActionAgent.encrypt_credentials
48
+ if ActionAgent.encrypt_credentials
49
+ encrypts :credential
50
+ encrypts :api_key
51
+ end
52
+
53
+ before_validation :normalize_host_credential, if: :host_based?
24
54
 
25
55
  validates :provider, presence: true, inclusion: { in: PROVIDERS }
26
56
  # One credential per provider per owner; which column that means
@@ -29,26 +59,128 @@ module ActionAgent
29
59
  validates :credential, presence: true, length: { maximum: 500 }
30
60
  validates :credential, format: { with: %r{\Ahttps?://\S+\z}, message: "must be an http(s):// URL" },
31
61
  if: :host_based?
62
+ validates :credential, format: {
63
+ with: CLAUDE_CODE_CREDENTIAL,
64
+ message: "must be an Anthropic API key (sk-ant-api…) from the Claude Console (https://platform.claude.com). " \
65
+ "Claude subscription tokens (`claude setup-token`, sk-ant-oat…) cannot be stored: Anthropic does not allow " \
66
+ "third-party apps to hold Claude.ai credentials. To use your own Claude login on this machine, set " \
67
+ "ActionAgent.claude_code_auth = :local_login with the :local sandbox backend instead"
68
+ }, if: -> { provider == "claude_code" }
69
+ validates :api_key, length: { maximum: 500 }, allow_nil: true
70
+
71
+ # Deletes every Claude Code connection that still holds a Claude
72
+ # subscription token (see #needs_replacing?), whoever owns it. The
73
+ # credential is encrypted, so each is read to tell. Their owners see
74
+ # Claude Code as not connected, and connect an API key again.
75
+ #
76
+ # @return [Integer] how many were deleted
77
+ def self.purge_subscription_tokens!
78
+ where(provider: "claude_code").find_each.count do |key|
79
+ key.needs_replacing? && key.destroy!
80
+ end
81
+ end
82
+
83
+ def self.kind_of_provider(provider)
84
+ if HOST_PROVIDERS.include?(provider) then "host"
85
+ elsif CONNECTION_PROVIDERS.include?(provider) then "connection"
86
+ else "key"
87
+ end
88
+ end
89
+
90
+ # Ollama's OpenAI-compatible API lives under /v1. Accept the bare server
91
+ # address people naturally paste (http://localhost:11434, a tunnel
92
+ # hostname) and add the path; trailing slashes are dropped so the client
93
+ # can join paths cleanly. An explicit non-root path is left alone, for
94
+ # servers behind a reverse proxy.
95
+ def self.normalize_host(value)
96
+ host = value.to_s.strip.chomp("/")
97
+ return host if host.blank?
98
+
99
+ uri = URI.parse(host)
100
+ return host unless uri.is_a?(URI::HTTP)
101
+
102
+ uri.path = "/v1" if uri.path.blank? || uri.path == "/"
103
+ uri.to_s.chomp("/")
104
+ rescue URI::InvalidURIError
105
+ host
106
+ end
32
107
 
33
108
  def host_based?
34
109
  HOST_PROVIDERS.include?(provider)
35
110
  end
36
111
 
112
+ def connection?
113
+ CONNECTION_PROVIDERS.include?(provider)
114
+ end
115
+
116
+ # Only host-based providers carry an optional key (a remote Ollama behind
117
+ # an authenticating proxy, or Ollama Cloud).
118
+ def api_key?
119
+ host_based? && api_key.present?
120
+ end
121
+
37
122
  # Options merged into generate_with for runs owned by this key's owner,
38
- # overriding the host app's config/active_agent.yml credentials.
123
+ # overriding the host app's config/active_agent.yml credentials. A
124
+ # connection credential configures no generation.
39
125
  def generation_options
40
- host_based? ? { host: credential } : { access_token: credential }
126
+ return {} if connection?
127
+ return { access_token: credential } unless host_based?
128
+
129
+ { host: credential, access_token: api_key.presence }.compact
130
+ end
131
+
132
+ # Environment variables a runtime needs to use this credential, for the
133
+ # credentials that are consumed by a process rather than a provider
134
+ # client: Claude Code reads an API key from ANTHROPIC_API_KEY.
135
+ #
136
+ # A subscription token stored before those were refused is never handed
137
+ # to a process: it yields nothing, as if Claude Code were not connected,
138
+ # until it is replaced with an API key.
139
+ #
140
+ # @return [Hash{String => String}]
141
+ def runtime_environment
142
+ return {} unless provider == "claude_code"
143
+ return {} if needs_replacing? || !CLAUDE_CODE_CREDENTIAL.match?(credential.to_s)
144
+
145
+ { "ANTHROPIC_API_KEY" => credential }
146
+ end
147
+
148
+ # A Claude Code connection that still holds a Claude subscription token
149
+ # (sk-ant-oat…), stored by an earlier version. It is no longer used and
150
+ # must be replaced with an API key; `bin/rails
151
+ # action_agent:claude_code:purge_subscription_tokens` deletes them all.
152
+ def needs_replacing?
153
+ provider == "claude_code" && credential.to_s.start_with?(SUBSCRIPTION_TOKEN_PREFIX)
41
154
  end
42
155
 
43
156
  # "sk-a…Q2z9" for keys; hosts are shown in full.
44
157
  def display_hint
45
158
  return credential if host_based?
46
159
 
47
- "#{credential.first(4)}…#{credential.last(4)}"
160
+ mask(credential)
161
+ end
162
+
163
+ # Masked hint for the optional host-provider key, nil when none is set.
164
+ def api_key_hint
165
+ api_key? ? mask(api_key) : nil
166
+ end
167
+
168
+ # Reachability + served models for host-based providers.
169
+ def probe
170
+ OllamaHostProbe.call(host: credential, api_key: api_key)
48
171
  end
49
172
 
50
173
  private
51
174
 
175
+ def mask(value)
176
+ "#{value.first(4)}…#{value.last(4)}"
177
+ end
178
+
179
+ def normalize_host_credential
180
+ self.credential = self.class.normalize_host(credential) if credential.present?
181
+ self.api_key = api_key.presence&.strip
182
+ end
183
+
52
184
  def provider_unique_within_owner
53
185
  return if provider.blank?
54
186