actionagent 1.7.2 → 1.8.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/app/assets/builds/action_agent.css +1 -1
- data/app/assets/builds/action_agent.js +59 -54
- data/app/controllers/action_agent/api/agents_controller.rb +19 -3
- data/app/controllers/action_agent/api/code_sessions_controller.rb +156 -0
- data/app/controllers/action_agent/api/evaluations_controller.rb +27 -130
- data/app/controllers/action_agent/api/github_connections_controller.rb +147 -0
- data/app/controllers/action_agent/api/mcp_controller.rb +38 -5
- data/app/controllers/action_agent/api/mcp_servers_controller.rb +10 -1
- data/app/controllers/action_agent/api/provider_keys_controller.rb +54 -3
- data/app/controllers/action_agent/api/provider_models_controller.rb +10 -6
- data/app/controllers/action_agent/api/sandboxes_controller.rb +97 -6
- data/app/controllers/concerns/action_agent/api/evaluation_run_starting.rb +93 -0
- data/app/controllers/concerns/action_agent/api/mcp_dashboard_tools.rb +507 -0
- data/app/controllers/concerns/action_agent/api/run_sandbox.rb +65 -0
- data/app/jobs/action_agent/code_session_job.rb +166 -0
- data/app/jobs/action_agent/sandbox_cleanup_job.rb +80 -11
- data/app/jobs/action_agent/sandbox_provision_job.rb +122 -14
- data/app/jobs/action_agent/sandbox_run_job.rb +10 -3
- data/app/models/action_agent/agent.rb +16 -6
- data/app/models/action_agent/agent_run.rb +20 -1
- data/app/models/action_agent/code_session.rb +141 -0
- data/app/models/action_agent/evaluation_run.rb +23 -1
- data/app/models/action_agent/github_connection.rb +75 -0
- data/app/models/action_agent/provider_key.rb +142 -10
- data/app/models/action_agent/sandbox_session.rb +193 -17
- data/app/serializers/action_agent/evaluation_serializer.rb +118 -0
- data/app/serializers/action_agent/telemetry_trace_serializer.rb +10 -2
- data/app/services/action_agent/agent_execution_service.rb +4 -2
- data/app/services/action_agent/agent_tool_roster.rb +20 -9
- data/app/services/action_agent/claude_code_auth.rb +86 -0
- data/app/services/action_agent/dashboard_assistant_service.rb +47 -5
- data/app/services/action_agent/evaluation_tool_resolver.rb +18 -0
- data/app/services/action_agent/github_client.rb +111 -0
- data/app/services/action_agent/local_sandbox_backend.rb +1689 -0
- data/app/services/action_agent/local_sandbox_databases.rb +257 -0
- data/app/services/action_agent/mcp_client.rb +5 -1
- data/app/services/action_agent/mcp_tool_dispatcher.rb +137 -17
- data/app/services/action_agent/mock_sandbox_backend.rb +39 -0
- data/app/services/action_agent/ollama_host_probe.rb +75 -0
- data/app/services/action_agent/payload_bounds.rb +36 -0
- data/app/services/action_agent/sandbox_manifest.rb +67 -0
- data/app/services/action_agent/sandbox_orchestrator.rb +69 -14
- data/app/services/action_agent/scenario_evaluation_runner.rb +50 -4
- data/app/services/action_agent/secret_scrubber.rb +37 -0
- data/app/services/action_agent/tool_discovery.rb +19 -5
- data/config/routes.rb +22 -3
- data/lib/action_agent/engine.rb +1 -0
- data/lib/action_agent/version.rb +1 -1
- data/lib/action_agent.rb +147 -3
- data/lib/generators/action_agent/install_generator.rb +30 -3
- data/lib/generators/action_agent/templates/action_agent.rb.erb +44 -0
- data/lib/generators/action_agent/templates/add_provider_key_api_key.rb.erb +25 -0
- data/lib/generators/action_agent/templates/create_active_agent_code_sessions.rb.erb +59 -0
- data/lib/generators/action_agent/templates/create_active_agent_dashboard_tables.rb.erb +2 -0
- data/lib/generators/action_agent/templates/create_active_agent_github_connections.rb.erb +61 -0
- data/lib/tasks/claude_code.rake +16 -0
- data/lib/tasks/sandbox.rake +26 -0
- metadata +23 -1
|
@@ -9,6 +9,8 @@ module ActionAgent
|
|
|
9
9
|
# and solid_agent records (ToolDiscovery), servers an agent declares in
|
|
10
10
|
# its configuration, and the default catalog (MCPCatalog). An install
|
|
11
11
|
# therefore sees both what it is already using and what it could turn on.
|
|
12
|
+
# The caller's live checkout sandbox runtimes are listed with the catalog,
|
|
13
|
+
# so the one started from Settings -> Integrations shows up here too.
|
|
12
14
|
class MCPServersController < BaseController
|
|
13
15
|
before_action :require_owner!
|
|
14
16
|
# Launching provisions a sandbox and runs a server in it, so it answers
|
|
@@ -108,8 +110,15 @@ module ActionAgent
|
|
|
108
110
|
|
|
109
111
|
private
|
|
110
112
|
|
|
113
|
+
# Runtimes are the caller's own: a listing carries no token, but which
|
|
114
|
+
# repositories another owner has running is still theirs to know.
|
|
111
115
|
def discovery
|
|
112
|
-
ToolDiscovery.new(
|
|
116
|
+
ToolDiscovery.new(
|
|
117
|
+
traces: owned_traces,
|
|
118
|
+
agents: owner_agents,
|
|
119
|
+
hours: window_hours,
|
|
120
|
+
runtimes: SandboxSession.runtime_server_listings(owned(SandboxSession))
|
|
121
|
+
)
|
|
113
122
|
end
|
|
114
123
|
|
|
115
124
|
# The catalog names this dashboard's own MCP endpoint as "<mount>/mcp",
|
|
@@ -4,7 +4,9 @@ module ActionAgent
|
|
|
4
4
|
module Api
|
|
5
5
|
# Per-account LLM provider credentials (Settings -> Provider API Keys).
|
|
6
6
|
# API keys are write-only: responses carry a masked hint, never the key.
|
|
7
|
-
# Ollama's credential is a host URL and is echoed back in full
|
|
7
|
+
# Ollama's credential is a host URL and is echoed back in full; its
|
|
8
|
+
# optional API key (remote servers) is masked like the others. Claude
|
|
9
|
+
# Code's connection API key is stored here too, write-only like a key.
|
|
8
10
|
class ProviderKeysController < BaseController
|
|
9
11
|
before_action :require_owner!
|
|
10
12
|
|
|
@@ -20,16 +22,42 @@ module ActionAgent
|
|
|
20
22
|
end
|
|
21
23
|
|
|
22
24
|
# POST /api/provider_keys — upserts the credential for a provider.
|
|
25
|
+
# Host-based providers may also carry an optional api_key; omitting it
|
|
26
|
+
# keeps the stored one, sending an empty string clears it.
|
|
23
27
|
def create
|
|
24
28
|
provider = params.require(:provider)
|
|
25
29
|
credential = params.require(:credential)
|
|
26
30
|
|
|
27
31
|
record = owned(ProviderKey).find_or_initialize_by(provider: provider)
|
|
28
|
-
|
|
32
|
+
attributes = { credential: credential }
|
|
33
|
+
attributes[:api_key] = params[:api_key].presence if params.key?(:api_key) && record.host_based?
|
|
34
|
+
record.update!(**attributes)
|
|
29
35
|
|
|
30
36
|
render json: { provider_key: serialize(provider, record) }, status: :created
|
|
31
37
|
end
|
|
32
38
|
|
|
39
|
+
# POST /api/provider_keys/test — checks that a host-based provider is
|
|
40
|
+
# reachable and lists the models it serves. Tests the submitted host
|
|
41
|
+
# (and api_key) when given, so a URL can be checked before saving;
|
|
42
|
+
# otherwise the stored credential, else the host app's default. Never
|
|
43
|
+
# persists anything.
|
|
44
|
+
def test
|
|
45
|
+
provider = params.require(:provider)
|
|
46
|
+
unless ProviderKey::HOST_PROVIDERS.include?(provider)
|
|
47
|
+
return render json: { error: "#{provider} is not a host-based provider" }, status: :unprocessable_entity
|
|
48
|
+
end
|
|
49
|
+
|
|
50
|
+
stored = owned(ProviderKey).find_by(provider: provider)
|
|
51
|
+
host = params[:credential].presence || stored&.credential || platform_host(provider)
|
|
52
|
+
api_key = params.key?(:api_key) ? params[:api_key].presence : stored&.api_key
|
|
53
|
+
|
|
54
|
+
if host.blank?
|
|
55
|
+
return render json: { ok: false, host: nil, models: [], latency_ms: nil, error: "No host configured" }
|
|
56
|
+
end
|
|
57
|
+
|
|
58
|
+
render json: OllamaHostProbe.call(host: host, api_key: api_key).to_h
|
|
59
|
+
end
|
|
60
|
+
|
|
33
61
|
# DELETE /api/provider_keys/:provider
|
|
34
62
|
def destroy
|
|
35
63
|
owned(ProviderKey).find_by!(provider: params[:provider]).destroy!
|
|
@@ -38,12 +66,35 @@ module ActionAgent
|
|
|
38
66
|
|
|
39
67
|
private
|
|
40
68
|
|
|
69
|
+
# The host app's default (config/active_agent.yml, e.g. OLLAMA_HOST)
|
|
70
|
+
# that applies when the owner has not configured their own.
|
|
71
|
+
def platform_host(provider)
|
|
72
|
+
return nil unless ProviderKey::HOST_PROVIDERS.include?(provider)
|
|
73
|
+
|
|
74
|
+
config = ActiveAgent.configuration[provider.to_sym]
|
|
75
|
+
config.respond_to?(:[]) ? config[:host].presence : nil
|
|
76
|
+
rescue StandardError
|
|
77
|
+
nil
|
|
78
|
+
end
|
|
79
|
+
|
|
41
80
|
def serialize(provider, record)
|
|
81
|
+
host_based = ProviderKey::HOST_PROVIDERS.include?(provider)
|
|
82
|
+
|
|
42
83
|
{
|
|
43
84
|
provider: provider,
|
|
44
|
-
host_based:
|
|
85
|
+
host_based: host_based,
|
|
86
|
+
# "key", "host", or "connection" (Settings -> Integrations rather
|
|
87
|
+
# than Provider API Keys).
|
|
88
|
+
kind: ProviderKey.kind_of_provider(provider),
|
|
45
89
|
configured: record.present?,
|
|
46
90
|
hint: record&.display_hint,
|
|
91
|
+
api_key_configured: record&.api_key? || false,
|
|
92
|
+
api_key_hint: record&.api_key_hint,
|
|
93
|
+
platform_default: host_based && record.nil? ? platform_host(provider) : nil,
|
|
94
|
+
# A Claude Code connection still holding a Claude subscription token
|
|
95
|
+
# from an earlier version: never used, and the UI asks for an API
|
|
96
|
+
# key in its place.
|
|
97
|
+
needs_replacing: record.present? && record.needs_replacing?,
|
|
47
98
|
updated_at: record&.updated_at&.iso8601
|
|
48
99
|
}
|
|
49
100
|
end
|
|
@@ -67,16 +67,20 @@ module ActionAgent
|
|
|
67
67
|
owned(ProviderKey).find_by(provider: provider)
|
|
68
68
|
end
|
|
69
69
|
|
|
70
|
+
# Same probe as Settings -> "Test connection", so the builder's dropdown
|
|
71
|
+
# and the settings page agree on what the host serves (including the
|
|
72
|
+
# optional Bearer key for remote servers).
|
|
70
73
|
def live_ollama_models
|
|
71
74
|
host = ollama_host
|
|
72
75
|
return nil unless host
|
|
73
76
|
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
|
|
77
|
+
result = OllamaHostProbe.call(host: host, api_key: owner_provider_key("ollama")&.api_key)
|
|
78
|
+
unless result.ok
|
|
79
|
+
Rails.logger.warn("[ProviderModels] ollama lookup failed: #{result.error}")
|
|
80
|
+
return nil
|
|
81
|
+
end
|
|
82
|
+
|
|
83
|
+
[ result.models, "live" ] if result.models.any?
|
|
80
84
|
end
|
|
81
85
|
|
|
82
86
|
# Queries the Anthropic Models API with the account's key (newest first,
|
|
@@ -12,6 +12,9 @@ module ActionAgent
|
|
|
12
12
|
# development, so it bypassed that safeguard too.
|
|
13
13
|
|
|
14
14
|
before_action :require_execution_enabled!, only: [ :run, :compare ]
|
|
15
|
+
# A checkout runs the owner's code (setup, server): the same gates as
|
|
16
|
+
# running an agent, like MCPServersController#launch.
|
|
17
|
+
before_action :gate_checkout!, only: [ :create ]
|
|
15
18
|
before_action :enforce_execution_quota!, only: [ :compare ]
|
|
16
19
|
before_action :set_sandbox, only: [ :show, :run, :destroy ]
|
|
17
20
|
|
|
@@ -83,16 +86,32 @@ module ActionAgent
|
|
|
83
86
|
end
|
|
84
87
|
|
|
85
88
|
# GET /api/sandboxes
|
|
86
|
-
# List available sandbox types and sample tasks
|
|
89
|
+
# List available sandbox types and sample tasks, and the caller's own
|
|
90
|
+
# sandboxes (?sandbox_type= narrows them), with what the Settings ->
|
|
91
|
+
# Integrations view needs to offer Claude Code sessions in a checkout.
|
|
87
92
|
def index
|
|
88
93
|
render json: {
|
|
89
94
|
sandbox_types: SandboxSession::SANDBOX_TYPES,
|
|
90
95
|
free_tier_limits: SandboxSession::FREE_TIER_LIMITS,
|
|
91
96
|
templates: free_tier_templates,
|
|
92
|
-
sample_tasks: sample_tasks
|
|
97
|
+
sample_tasks: sample_tasks,
|
|
98
|
+
sandboxes: listed_sandboxes.map(&:summary),
|
|
99
|
+
code_sessions_supported: code_sessions_supported?,
|
|
100
|
+
**claude_code_status
|
|
93
101
|
}
|
|
94
102
|
end
|
|
95
103
|
|
|
104
|
+
# Refuses a checkout the owner may not start. Usage is recorded by
|
|
105
|
+
# #create once the sandbox saved: a request refused for its own
|
|
106
|
+
# content (a repository that is not selected, say) runs nothing, so it
|
|
107
|
+
# must not spend a plan run.
|
|
108
|
+
def gate_checkout!
|
|
109
|
+
return unless checkout_requested?
|
|
110
|
+
|
|
111
|
+
require_execution_enabled!
|
|
112
|
+
enforce_execution_quota! unless performed?
|
|
113
|
+
end
|
|
114
|
+
|
|
96
115
|
# POST /api/sandboxes
|
|
97
116
|
# Create a new sandbox session, owned by whoever opened it.
|
|
98
117
|
def create
|
|
@@ -100,11 +119,17 @@ module ActionAgent
|
|
|
100
119
|
# Guarded: the association only exists when the host app configured a
|
|
101
120
|
# user model, and a single-user install configures none.
|
|
102
121
|
@sandbox.user = current_user if @sandbox.respond_to?(:user=)
|
|
122
|
+
# A checkout is validated against the owner's GitHub connection, which
|
|
123
|
+
# an account-owned install finds through the account.
|
|
124
|
+
@sandbox.account_id = current_account.id if current_account && @sandbox.has_attribute?(:account_id)
|
|
103
125
|
@sandbox.agent_template = AgentTemplate.find_by(slug: params[:template_slug]) if params[:template_slug]
|
|
104
126
|
|
|
105
127
|
if @sandbox.save
|
|
106
128
|
@sandbox.provision!
|
|
107
129
|
@sandbox.reload # Reload to get updated status after provisioning
|
|
130
|
+
# Counted once the checkout exists, as MCPServersController#launch
|
|
131
|
+
# counts a launched server.
|
|
132
|
+
record_execution_usage if checkout_requested?
|
|
108
133
|
render json: { sandbox: @sandbox.summary }, status: :created
|
|
109
134
|
else
|
|
110
135
|
render json: { errors: @sandbox.errors.full_messages }, status: :unprocessable_entity
|
|
@@ -162,20 +187,86 @@ module ActionAgent
|
|
|
162
187
|
end
|
|
163
188
|
|
|
164
189
|
# DELETE /api/sandboxes/:session_id
|
|
165
|
-
# End sandbox session
|
|
190
|
+
# End sandbox session. Expiring it enqueues SandboxCleanupJob, which
|
|
191
|
+
# terminates whatever the backend runs for it (a checkout's processes
|
|
192
|
+
# included); one still provisioning is released by
|
|
193
|
+
# SandboxProvisionJob when its backend returns.
|
|
166
194
|
def destroy
|
|
167
195
|
@sandbox.expire!
|
|
168
|
-
render json: { deleted: true }
|
|
196
|
+
render json: { deleted: true, sandbox: @sandbox.summary }
|
|
169
197
|
end
|
|
170
198
|
|
|
171
199
|
private
|
|
172
200
|
|
|
173
201
|
def set_sandbox
|
|
174
|
-
@sandbox = owned(SandboxSession).find_by!(session_id: params[:id])
|
|
202
|
+
@sandbox = account_scoped(owned(SandboxSession)).find_by!(session_id: params[:id])
|
|
203
|
+
end
|
|
204
|
+
|
|
205
|
+
def checkout_requested?
|
|
206
|
+
params[:sandbox_type].to_s == "app_runtime"
|
|
207
|
+
end
|
|
208
|
+
|
|
209
|
+
# A checkout runs on its account's GitHub token and Claude Code
|
|
210
|
+
# credential, so in a multi-tenant install it is listed, shown and
|
|
211
|
+
# stopped only within the caller's current account, as
|
|
212
|
+
# CodeSessionsController finds it: listing another account's checkout
|
|
213
|
+
# as Ready offered a Claude Code panel that then answered 404. Other
|
|
214
|
+
# sandbox types are the caller's own wherever they were opened.
|
|
215
|
+
def account_scoped(scope)
|
|
216
|
+
return scope unless current_account
|
|
217
|
+
|
|
218
|
+
scope.where.not(sandbox_type: "app_runtime").or(scope.where(account_id: current_account.id))
|
|
219
|
+
end
|
|
220
|
+
|
|
221
|
+
# Whether the configured backend can run Claude Code sessions. A
|
|
222
|
+
# backend that cannot even be loaded (a misspelled class in
|
|
223
|
+
# ActionAgent.sandbox_backends, or a class file requiring an SDK the
|
|
224
|
+
# host doesn't bundle, which raises LoadError) cannot, and must not
|
|
225
|
+
# take the rest of this listing down with it.
|
|
226
|
+
#
|
|
227
|
+
# Nor can one whose Claude Code authentication does not work there
|
|
228
|
+
# (ActionAgent.claude_code_auth = :local_login needs :local).
|
|
229
|
+
def code_sessions_supported?
|
|
230
|
+
orchestrator = SandboxOrchestrator.new
|
|
231
|
+
orchestrator.supports?(:code_session) && ClaudeCodeAuth.backend_refusal(orchestrator).nil?
|
|
232
|
+
rescue StandardError, LoadError => e
|
|
233
|
+
Rails.logger.warn("[ActionAgent] sandbox backend unavailable: #{e.message}")
|
|
234
|
+
false
|
|
235
|
+
end
|
|
236
|
+
|
|
237
|
+
# Whether the caller's Claude Code can run sessions, by
|
|
238
|
+
# ClaudeCodeAuth's rule: an API key they connected, or this machine's
|
|
239
|
+
# own login. Never a credential.
|
|
240
|
+
#
|
|
241
|
+
# claude_code_auth "api_key" | "local_login"
|
|
242
|
+
# claude_code_connected Boolean
|
|
243
|
+
# claude_code_login { logged_in:, auth_method: } (local_login only)
|
|
244
|
+
#
|
|
245
|
+
# The key is found through its own owner column, as
|
|
246
|
+
# SandboxSession#runtime_environment finds the credential it hands a
|
|
247
|
+
# checkout: a provider key is account-owned before user-owned.
|
|
248
|
+
def claude_code_status
|
|
249
|
+
status = ClaudeCodeAuth.status(owned(ProviderKey))
|
|
250
|
+
{
|
|
251
|
+
claude_code_auth: status[:mode],
|
|
252
|
+
claude_code_connected: status[:connected],
|
|
253
|
+
claude_code_login: status[:login]
|
|
254
|
+
}.compact
|
|
255
|
+
end
|
|
256
|
+
|
|
257
|
+
# The caller's sandboxes that have not expired, newest first. A failed
|
|
258
|
+
# one stays listed so its error can be read; one past its expiry but
|
|
259
|
+
# not reaped yet stays listed so it can still be stopped.
|
|
260
|
+
def listed_sandboxes
|
|
261
|
+
scope = account_scoped(owned(SandboxSession)).where.not(status: :expired).recent.limit(20)
|
|
262
|
+
type = params[:sandbox_type]
|
|
263
|
+
type.is_a?(String) && type.present? ? scope.by_type(type) : scope
|
|
175
264
|
end
|
|
176
265
|
|
|
266
|
+
# An app_runtime sandbox also names the checkout: one of the owner's
|
|
267
|
+
# selected GitHub repositories and, optionally, a ref.
|
|
177
268
|
def sandbox_params
|
|
178
|
-
params.permit(:sandbox_type)
|
|
269
|
+
params.permit(:sandbox_type, :repository, :repository_ref)
|
|
179
270
|
end
|
|
180
271
|
|
|
181
272
|
def free_tier_templates
|
|
@@ -0,0 +1,93 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module ActionAgent
|
|
4
|
+
module Api
|
|
5
|
+
# Starting an evaluation run on the caller's behalf, shared by the
|
|
6
|
+
# evaluations API (POST /api/evaluations/:id/run) and the MCP facade's
|
|
7
|
+
# `evaluations_run` tool so both check a run the same way.
|
|
8
|
+
#
|
|
9
|
+
# A scenario suite accepts a selection that narrows the run:
|
|
10
|
+
# - scenario_ids: the scenarios to replay, by id
|
|
11
|
+
# - keys: the scenarios to replay, by key
|
|
12
|
+
# - group: one scenario group
|
|
13
|
+
# - models: the candidate models, as an array or a comma-separated string
|
|
14
|
+
# - sandbox_id: a checkout sandbox of the caller's that every replay reaches (see RunSandbox)
|
|
15
|
+
module EvaluationRunStarting
|
|
16
|
+
extend ActiveSupport::Concern
|
|
17
|
+
|
|
18
|
+
include RunSandbox
|
|
19
|
+
|
|
20
|
+
private
|
|
21
|
+
|
|
22
|
+
# Starts a run of +evaluation+ and returns it. A scenario suite replays
|
|
23
|
+
# through the provider once per scenario and model, so it runs in the
|
|
24
|
+
# background and the run comes back pending. A generation-sampling
|
|
25
|
+
# evaluation scores recorded data and finishes inline.
|
|
26
|
+
#
|
|
27
|
+
# A failure is recorded on the run rather than raised, so the caller
|
|
28
|
+
# always gets a run to report.
|
|
29
|
+
#
|
|
30
|
+
# @return [EvaluationRun]
|
|
31
|
+
def start_evaluation_run(evaluation, selection)
|
|
32
|
+
return evaluation.run_later!(**selection) if evaluation.scenario_suite?
|
|
33
|
+
|
|
34
|
+
evaluation.run!
|
|
35
|
+
rescue StandardError => e
|
|
36
|
+
Rails.logger.warn(
|
|
37
|
+
"[ActionAgent] evaluation #{evaluation.id} run failed: #{e.class}: #{e.message}"
|
|
38
|
+
)
|
|
39
|
+
# EvaluationRunnerService records the failure on the run before
|
|
40
|
+
# re-raising. A failure before the run record existed is recorded here.
|
|
41
|
+
evaluation.evaluation_runs.recent.first ||
|
|
42
|
+
evaluation.evaluation_runs.create!(status: :failed, error_message: e.message, completed_at: Time.current)
|
|
43
|
+
end
|
|
44
|
+
|
|
45
|
+
# Returns the selection +source+ (params or tool arguments) asks for,
|
|
46
|
+
# without the sandbox, which evaluation_run_sandbox checks separately.
|
|
47
|
+
def evaluation_run_selection(source)
|
|
48
|
+
models = source[:models]
|
|
49
|
+
models = models.to_s.split(",") unless models.is_a?(Array)
|
|
50
|
+
|
|
51
|
+
{
|
|
52
|
+
scenario_ids: Array(source[:scenario_ids]).map(&:to_s).reject(&:blank?),
|
|
53
|
+
keys: Array(source[:keys]).map(&:to_s).reject(&:blank?),
|
|
54
|
+
group: source[:group].to_s.presence,
|
|
55
|
+
models: models.map(&:to_s).map(&:strip).reject(&:blank?)
|
|
56
|
+
}.compact_blank
|
|
57
|
+
end
|
|
58
|
+
|
|
59
|
+
# Returns the checked sandbox +sandbox_id+ names for a run of
|
|
60
|
+
# +evaluation+, or nil when none was asked for. Raises RunSandbox::Refused.
|
|
61
|
+
#
|
|
62
|
+
# Only a scenario suite's own replay executes the agent: a sampling run
|
|
63
|
+
# scores recorded generations, and a host adapter replays in the host's
|
|
64
|
+
# runtime, where no dashboard dispatcher runs.
|
|
65
|
+
#
|
|
66
|
+
# @return [SandboxSession, nil]
|
|
67
|
+
def evaluation_run_sandbox(evaluation, sandbox_id)
|
|
68
|
+
return nil if sandbox_id.blank?
|
|
69
|
+
|
|
70
|
+
unless evaluation.scenario_suite?
|
|
71
|
+
raise RunSandbox::Refused, "Only a scenario evaluation runs the agent; this one scores recorded generations, " \
|
|
72
|
+
"so it cannot run against a sandbox"
|
|
73
|
+
end
|
|
74
|
+
if scenario_evaluation_adapter?(evaluation)
|
|
75
|
+
raise RunSandbox::Refused, "This install replays scenarios through its own adapter, which cannot reach a " \
|
|
76
|
+
"dashboard sandbox"
|
|
77
|
+
end
|
|
78
|
+
|
|
79
|
+
run_sandbox_for(evaluation.agent, sandbox_id)
|
|
80
|
+
end
|
|
81
|
+
|
|
82
|
+
# Whether a run of +evaluation+ cannot replay its scenarios: its agent
|
|
83
|
+
# is observed (read-only), and no host adapter replays it instead.
|
|
84
|
+
def unexecutable_scenario_run?(evaluation)
|
|
85
|
+
evaluation.agent.observed? && !scenario_evaluation_adapter?(evaluation)
|
|
86
|
+
end
|
|
87
|
+
|
|
88
|
+
def scenario_evaluation_adapter?(evaluation)
|
|
89
|
+
ActionAgent.scenario_evaluation_adapter_resolver&.call(evaluation).respond_to?(:call)
|
|
90
|
+
end
|
|
91
|
+
end
|
|
92
|
+
end
|
|
93
|
+
end
|