actionagent 1.2.2 → 1.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (54) hide show
  1. checksums.yaml +4 -4
  2. data/README.md +14 -3
  3. data/app/assets/builds/action_agent.css +1 -1
  4. data/app/assets/builds/action_agent.js +48 -44
  5. data/app/controllers/action_agent/api/agent_runs_controller.rb +25 -7
  6. data/app/controllers/action_agent/api/agents_controller.rb +76 -48
  7. data/app/controllers/action_agent/api/analytics_controller.rb +31 -9
  8. data/app/controllers/action_agent/api/base_controller.rb +16 -0
  9. data/app/controllers/action_agent/api/evaluations_controller.rb +237 -7
  10. data/app/controllers/action_agent/api/mcp_controller.rb +13 -3
  11. data/app/controllers/action_agent/api/mcp_servers_controller.rb +28 -8
  12. data/app/controllers/action_agent/api/metrics_controller.rb +44 -11
  13. data/app/controllers/action_agent/api/provider_models_controller.rb +1 -1
  14. data/app/controllers/action_agent/api/sandboxes_controller.rb +6 -0
  15. data/app/controllers/action_agent/api/session_recordings_controller.rb +34 -12
  16. data/app/controllers/action_agent/api/templates_controller.rb +25 -21
  17. data/app/controllers/action_agent/api/traces_controller.rb +25 -5
  18. data/app/controllers/action_agent/api/usage_controller.rb +20 -0
  19. data/app/controllers/action_agent/application_controller.rb +25 -2
  20. data/app/controllers/action_agent/dashboard_controller.rb +2 -1
  21. data/app/controllers/concerns/action_agent/api/agent_serialization.rb +53 -0
  22. data/app/jobs/action_agent/agent_execution_job.rb +40 -20
  23. data/app/jobs/action_agent/application_job.rb +7 -3
  24. data/app/jobs/action_agent/evaluation_run_job.rb +18 -0
  25. data/app/jobs/action_agent/sandbox_cleanup_job.rb +13 -10
  26. data/app/models/action_agent/agent.rb +14 -5
  27. data/app/models/action_agent/agent_template.rb +22 -7
  28. data/app/models/action_agent/evaluation.rb +64 -4
  29. data/app/models/action_agent/evaluation_run.rb +182 -2
  30. data/app/models/action_agent/evaluation_scenario.rb +59 -0
  31. data/app/models/action_agent/evaluation_scenario_result.rb +66 -0
  32. data/app/models/action_agent/recording_action.rb +11 -7
  33. data/app/models/action_agent/sandbox_session.rb +1 -1
  34. data/app/models/action_agent/session_recording.rb +31 -8
  35. data/app/models/action_agent/telemetry_trace.rb +110 -2
  36. data/app/models/concerns/action_agent/adapter_aware.rb +19 -0
  37. data/app/models/concerns/action_agent/ownable.rb +15 -2
  38. data/app/queries/action_agent/metrics_report.rb +498 -0
  39. data/app/services/action_agent/agent_toolbox.rb +4 -4
  40. data/app/services/action_agent/evaluation_tool_resolver.rb +154 -0
  41. data/app/services/action_agent/mcp_catalog.rb +46 -8
  42. data/app/services/action_agent/mcp_recording_middleware.rb +2 -2
  43. data/app/services/action_agent/playwright_mcp_client.rb +6 -6
  44. data/app/services/action_agent/sandbox_orchestrator.rb +12 -1
  45. data/app/services/action_agent/scenario_evaluation_runner.rb +226 -0
  46. data/app/services/action_agent/tool_discovery.rb +22 -8
  47. data/config/routes.rb +25 -2
  48. data/lib/action_agent/engine.rb +101 -19
  49. data/lib/action_agent/version.rb +1 -1
  50. data/lib/action_agent.rb +72 -6
  51. data/lib/generators/action_agent/install_generator.rb +20 -7
  52. data/lib/generators/action_agent/templates/create_active_agent_evaluation_scenarios.rb.erb +79 -0
  53. data/lib/tasks/action_agent.rake +9 -0
  54. metadata +19 -6
@@ -46,7 +46,7 @@ module ActionAgent
46
46
  }
47
47
  }
48
48
  ],
49
- # A real browser via a Playwright MCP server (PlaywrightMcpClient).
49
+ # A real browser via a Playwright MCP server (PlaywrightMCPClient).
50
50
  # Stateful: navigate changes what snapshot/click see, so these bypass
51
51
  # the toolbox result cache.
52
52
  "playwright_mcp" => [
@@ -288,17 +288,17 @@ module ActionAgent
288
288
  SNAPSHOT_LINK = /\[Snapshot\]\(([^)]+)\)/
289
289
 
290
290
  def playwright_mcp(tool, arguments, retried: false)
291
- result = PlaywrightMcpClient.instance.call_tool(tool, arguments)
291
+ result = PlaywrightMCPClient.instance.call_tool(tool, arguments)
292
292
  text = inline_snapshot(result[:text].to_s)
293
293
  if text.length > PLAYWRIGHT_RESULT_LIMIT
294
294
  text = "#{text[0, PLAYWRIGHT_RESULT_LIMIT]}\n…(truncated, #{text.length} chars total)"
295
295
  end
296
296
  result[:is_error] ? { error: text.presence || "browser tool failed" } : { text: text }
297
- rescue PlaywrightMcpClient::Error => e
297
+ rescue PlaywrightMCPClient::Error => e
298
298
  # One fresh-session retry: the first call after a server (re)start can
299
299
  # race the browser launch.
300
300
  unless retried
301
- PlaywrightMcpClient.reset!
301
+ PlaywrightMCPClient.reset!
302
302
  return playwright_mcp(tool, arguments, retried: true)
303
303
  end
304
304
  { error: e.message }
@@ -0,0 +1,154 @@
1
+ # frozen_string_literal: true
2
+
3
+ module ActionAgent
4
+ # Names the MCP server behind a tool an evaluation run needs, for the
5
+ # report's fix items (ActiveAgent::Evals::Report#fix_items): a scenario
6
+ # that expected +search_slots+ and never got it is fixed by enabling the
7
+ # server that serves it, and the item can only say so — and deep-link to
8
+ # MCP Services — when something here can name that server.
9
+ #
10
+ # Resolution follows the order ToolDiscovery attributes traffic in: an
11
+ # explicit +mcp__server__tool+ namespace wins because the name said so,
12
+ # then MCPCatalog's hints for the bare names of well-known servers, then
13
+ # the servers the agent itself declares in +mcp_servers+ when one of them
14
+ # lists the tool. Nothing here reads telemetry: the resolver runs inside
15
+ # the request that serializes a run, and every lookup is a constant or a
16
+ # single agent attribute.
17
+ #
18
+ # The status is what the fix item's action turns on:
19
+ #
20
+ # "enabled" — the agent's mcp_servers configuration names the server
21
+ # "available" — the catalog (built-in or ActionAgent.mcp_catalog) knows
22
+ # the server and the agent has not enabled it
23
+ # nil — the namespace named a server nothing here knows; the
24
+ # report renders that as "unknown"
25
+ #
26
+ # @example
27
+ # resolver = EvaluationToolResolver.new(agent)
28
+ # resolver.call("browser_navigate")
29
+ # # => { "key" => "playwright", "name" => "Playwright", "status" => "available" }
30
+ class EvaluationToolResolver
31
+ ENABLED = "enabled"
32
+ AVAILABLE = "available"
33
+
34
+ attr_reader :agent
35
+
36
+ # @param agent [ActionAgent::Agent, nil] the agent the run evaluated; nil
37
+ # resolves against the catalog alone
38
+ def initialize(agent)
39
+ @agent = agent
40
+ end
41
+
42
+ # The Report's +tool_resolver+ contract.
43
+ #
44
+ # @param tool_name [String, Symbol] a tool name as a scenario expected it
45
+ # or the model called it
46
+ # @return [Hash, nil] +{ "key", "name", "status" }+, or nil when no
47
+ # server can be named for the tool
48
+ def call(tool_name)
49
+ key = server_key_for(tool_name)
50
+ return nil if key.nil?
51
+
52
+ { "key" => key, "name" => display_name_for(key), "status" => status_for(key) }
53
+ end
54
+
55
+ # @param tool_name [String, Symbol]
56
+ # @return [String, nil] the server key the tool belongs to
57
+ def server_key_for(tool_name)
58
+ name = tool_name.to_s.strip
59
+ return nil if name.blank?
60
+
61
+ ActiveAgent::Telemetry::ToolOrigin.server_for(name).presence ||
62
+ MCPCatalog.server_for_tool(name).presence ||
63
+ configured_tools[name]
64
+ end
65
+
66
+ # @param key [String] a server key
67
+ # @return [String, nil] ENABLED, AVAILABLE, or nil when unknown
68
+ def status_for(key)
69
+ return ENABLED if configured_keys.include?(normalize(key))
70
+ return AVAILABLE if MCPCatalog.find(key)
71
+
72
+ nil
73
+ end
74
+
75
+ private
76
+
77
+ # The catalog's name when it has one; otherwise the name the agent's
78
+ # own configuration gives the server, and the key as a last resort
79
+ # (which is what MCPCatalog.display_name falls back to as well).
80
+ def display_name_for(key)
81
+ return MCPCatalog.display_name(key) if MCPCatalog.find(key)
82
+
83
+ configured_names[normalize(key)] || MCPCatalog.display_name(key)
84
+ end
85
+
86
+ # Server keys the agent declares, normalized for comparison.
87
+ def configured_keys
88
+ @configured_keys ||= configured_entries.filter_map { |entry| normalize(entry_key(entry)) }.to_set
89
+ end
90
+
91
+ # normalized key => the display name a configured hash entry carries
92
+ # alongside its key ({"key" => "sparkle", "name" => "Sparkle Match"}).
93
+ def configured_names
94
+ @configured_names ||= configured_entries.each_with_object({}) do |entry, map|
95
+ next unless entry.respond_to?(:key?)
96
+
97
+ key = normalize(entry_key(entry))
98
+ name = (entry["name"] || entry[:name]).to_s.strip
99
+ next if key.nil? || name.blank? || name.downcase == key
100
+
101
+ map[key] ||= name
102
+ end
103
+ end
104
+
105
+ # bare tool name => server key, from configured entries that list the
106
+ # tools they serve ({"name" => "sparkle", "tools" => ["search_slots"]}),
107
+ # in the catalog's own +tool_hints+ spelling or as tool hashes.
108
+ def configured_tools
109
+ @configured_tools ||= configured_entries.each_with_object({}) do |entry, map|
110
+ next unless entry.respond_to?(:key?)
111
+
112
+ key = entry_key(entry)
113
+ next if key.nil?
114
+
115
+ Array(entry["tools"] || entry[:tools] || entry["tool_hints"] || entry[:tool_hints]).each do |tool|
116
+ name = (tool.respond_to?(:key?) ? tool["name"] || tool[:name] : tool).to_s.strip
117
+ map[name] ||= key unless name.blank?
118
+ end
119
+ end
120
+ end
121
+
122
+ # The agent's mcp_servers as a list of entries. Agents store an Array of
123
+ # bare names or builder hashes, but an agent seeded from an older
124
+ # template carries a top-level Hash keyed by server name
125
+ # ({"playwright" => {"command" => ...}}) — the same shape ToolDiscovery
126
+ # tolerates — whose values become entries carrying that key.
127
+ def configured_entries
128
+ @configured_entries ||= begin
129
+ servers = agent&.mcp_servers
130
+
131
+ if servers.is_a?(Hash)
132
+ servers.map do |key, value|
133
+ value.respond_to?(:key?) ? value.to_h.stringify_keys.merge("key" => key.to_s) : key.to_s
134
+ end
135
+ else
136
+ Array(servers)
137
+ end
138
+ end
139
+ end
140
+
141
+ # An entry names its server as a bare string, or under +key+ or +name+
142
+ # in a builder hash. Anything else (a stray Array, a number) is skipped.
143
+ def entry_key(entry)
144
+ return entry.to_s.strip.presence if entry.is_a?(String) || entry.is_a?(Symbol)
145
+ return nil unless entry.respond_to?(:key?)
146
+
147
+ (entry["key"] || entry[:key] || entry["name"] || entry[:name]).to_s.strip.presence
148
+ end
149
+
150
+ def normalize(key)
151
+ key.to_s.strip.downcase.presence
152
+ end
153
+ end
154
+ end
@@ -19,7 +19,7 @@ module ActionAgent
19
19
  # shows up in the MCP Services view (as +known: false+) the moment a
20
20
  # namespaced tool call from it is ingested. The catalog only adds names,
21
21
  # descriptions, and the ability to launch.
22
- class McpCatalog
22
+ class MCPCatalog
23
23
  # Whether a server can be started inside a sandbox session. Servers that
24
24
  # need workspace-specific credentials (github, slack, postgres) are
25
25
  # listable and attributable but not launchable from the dashboard — there
@@ -156,7 +156,12 @@ module ActionAgent
156
156
  sandbox: false,
157
157
  first_party: true,
158
158
  requires_credentials: [ "Platform API key" ],
159
- tool_hints: %w[call_agent list_agents]
159
+ # No tool_hints: the facade's tools are run_<slug> (one per agent, see
160
+ # Api::MCPController#tools_list), which no static list can name. The
161
+ # old hints advertised call_agent and list_agents, which the facade
162
+ # never exposed, and would have misattributed an agent-defined tool
163
+ # of either name to this server.
164
+ tool_hints: []
160
165
  }
161
166
  ].freeze
162
167
 
@@ -172,11 +177,12 @@ module ActionAgent
172
177
  end.freeze
173
178
 
174
179
  class << self
175
- # Every catalog entry, as API-shaped hashes.
180
+ # Every catalog entry built-ins plus the host's
181
+ # +ActionAgent.mcp_catalog+ registrations — as API-shaped hashes.
176
182
  #
177
183
  # @return [Array<Hash>]
178
184
  def all
179
- SERVERS.map { |server| present(server) }
185
+ entries.map { |server| present(server) }
180
186
  end
181
187
 
182
188
  # Entries that can be started inside a sandbox session.
@@ -189,14 +195,21 @@ module ActionAgent
189
195
  # @param key [String, Symbol]
190
196
  # @return [Hash, nil] the catalog entry, or nil when unknown
191
197
  def find(key)
192
- entry = BY_KEY[key.to_s]
198
+ entry = index[key.to_s]
193
199
  present(entry) if entry
194
200
  end
195
201
 
202
+ # Every catalog key, built-ins first.
203
+ #
204
+ # @return [Array<String>]
205
+ def keys
206
+ entries.map { |server| server[:key] }
207
+ end
208
+
196
209
  # @param key [String, Symbol]
197
210
  # @return [Boolean] whether the server can be started in a sandbox
198
211
  def launchable?(key)
199
- BY_KEY[key.to_s]&.fetch(:sandbox, false) || false
212
+ index[key.to_s]&.fetch(:sandbox, false) || false
200
213
  end
201
214
 
202
215
  # The catalog server a bare (non-namespaced) tool name belongs to.
@@ -208,7 +221,7 @@ module ActionAgent
208
221
  # @param tool_name [String, Symbol]
209
222
  # @return [String, nil] the server key
210
223
  def server_for_tool(tool_name)
211
- TOOL_HINTS[tool_name.to_s]
224
+ tool_hints[tool_name.to_s]
212
225
  end
213
226
 
214
227
  # The display name for a server key, falling back to the key itself for
@@ -217,11 +230,36 @@ module ActionAgent
217
230
  # @param key [String, Symbol]
218
231
  # @return [String]
219
232
  def display_name(key)
220
- BY_KEY[key.to_s]&.fetch(:name) || key.to_s
233
+ index[key.to_s]&.[](:name) || key.to_s
221
234
  end
222
235
 
223
236
  private
224
237
 
238
+ # Built-ins plus host registrations, deduplicated by key (first
239
+ # declaration wins, so a built-in keeps its key). Derived per call
240
+ # rather than memoized: the host list is tiny and tests reconfigure it.
241
+ def entries
242
+ (SERVERS + host_entries).uniq { |server| server[:key] }
243
+ end
244
+
245
+ def host_entries
246
+ Array(ActionAgent.mcp_catalog).filter_map do |entry|
247
+ normalized = entry.to_h.symbolize_keys
248
+ normalized[:key] = normalized[:key].to_s
249
+ normalized unless normalized[:key].blank?
250
+ end
251
+ end
252
+
253
+ def index
254
+ entries.index_by { |server| server[:key] }
255
+ end
256
+
257
+ def tool_hints
258
+ entries.each_with_object({}) do |server, map|
259
+ Array(server[:tool_hints]).each { |tool| map[tool.to_s] ||= server[:key] }
260
+ end
261
+ end
262
+
225
263
  # Drops the internal-only tool_hints and normalizes optional keys so
226
264
  # every entry serializes with the same shape.
227
265
  def present(server)
@@ -7,7 +7,7 @@ module ActionAgent
7
7
  # to a SessionRecording for playback and debugging.
8
8
  #
9
9
  # Usage:
10
- # middleware = McpRecordingMiddleware.new(session_recording: recording)
10
+ # middleware = MCPRecordingMiddleware.new(session_recording: recording)
11
11
  #
12
12
  # # Process a tool call
13
13
  # result = middleware.intercept(tool_call) do
@@ -15,7 +15,7 @@ module ActionAgent
15
15
  # mcp_client.call_tool(tool_call)
16
16
  # end
17
17
  #
18
- class McpRecordingMiddleware
18
+ class MCPRecordingMiddleware
19
19
  # Map of Playwright MCP tool names to our action types
20
20
  PLAYWRIGHT_TOOLS = {
21
21
  "browser_navigate" => "navigate",
@@ -9,7 +9,7 @@ module ActionAgent
9
9
  # app in development, or a sandbox-provisioned browser container in
10
10
  # production. Speaks just enough JSON-RPC for tools/call: initialize once
11
11
  # per process, then call tools under the session id the server hands back.
12
- class PlaywrightMcpClient
12
+ class PlaywrightMCPClient
13
13
  DEFAULT_URL = ENV.fetch("PLAYWRIGHT_MCP_URL", "http://host.orb.internal:8931/mcp")
14
14
  OPEN_TIMEOUT_SECONDS = 5
15
15
  READ_TIMEOUT_SECONDS = 60
@@ -31,7 +31,7 @@ module ActionAgent
31
31
 
32
32
  # Returns { text:, is_error: } — the tool result's text content.
33
33
  def call_tool(name, arguments = {})
34
- Rails.logger.debug("[PlaywrightMcpClient] call #{name} args=#{arguments.inspect[0, 200]}")
34
+ Rails.logger.debug("[PlaywrightMCPClient] call #{name} args=#{arguments.inspect[0, 200]}")
35
35
  ensure_session!
36
36
  response = post(
37
37
  { jsonrpc: "2.0", id: next_id, method: "tools/call",
@@ -40,7 +40,7 @@ module ActionAgent
40
40
  )
41
41
  result = response["result"]
42
42
  unless result
43
- Rails.logger.warn("[PlaywrightMcpClient] #{name} unexpected response: #{response.inspect[0, 500]}")
43
+ Rails.logger.warn("[PlaywrightMCPClient] #{name} unexpected response: #{response.inspect[0, 500]}")
44
44
  raise Error, (response.dig("error", "message") || "empty MCP response")
45
45
  end
46
46
 
@@ -100,17 +100,17 @@ module ActionAgent
100
100
 
101
101
  response = http.request(request)
102
102
  Rails.logger.debug(
103
- "[PlaywrightMcpClient] #{payload[:method]} -> #{response.code} " \
103
+ "[PlaywrightMCPClient] #{payload[:method]} -> #{response.code} " \
104
104
  "ct=#{response['Content-Type']} bytes=#{response.body.to_s.bytesize} session=#{session ? 'yes' : 'no'}"
105
105
  )
106
106
  unless response.code.to_i.between?(200, 299)
107
- Rails.logger.warn("[PlaywrightMcpClient] HTTP #{response.code}: #{response.body.to_s[0, 300]}")
107
+ Rails.logger.warn("[PlaywrightMCPClient] HTTP #{response.code}: #{response.body.to_s[0, 300]}")
108
108
  raise Error, "MCP server returned HTTP #{response.code}"
109
109
  end
110
110
 
111
111
  parsed = parse_body(response)
112
112
  if parsed.empty? && payload[:id]
113
- Rails.logger.warn("[PlaywrightMcpClient] unparsed body (#{response['Content-Type']}): #{response.body.to_s[0, 500]}")
113
+ Rails.logger.warn("[PlaywrightMCPClient] unparsed body (#{response['Content-Type']}): #{response.body.to_s[0, 500]}")
114
114
  end
115
115
  [ parsed, response ]
116
116
  end
@@ -52,7 +52,18 @@ module ActionAgent
52
52
  # as sandbox_service, falling back to the in-memory one.
53
53
  def self.default_backend
54
54
  name = ENV["SANDBOX_BACKEND"].presence || ActionAgent.sandbox_service.to_s
55
- backends.key?(name) ? name : "mock"
55
+ return name if backends.key?(name)
56
+
57
+ # Substituting the mock silently made a misconfigured operator's
58
+ # sandbox "runs" succeed against nothing real.
59
+ if name.present? && name != "mock"
60
+ Rails.logger.warn(
61
+ "[ActionAgent] sandbox backend #{name.inspect} is not registered " \
62
+ "(ActionAgent.sandbox_backends knows #{backends.keys.inspect}); " \
63
+ "using the in-memory mock backend, which runs nothing."
64
+ )
65
+ end
66
+ "mock"
56
67
  end
57
68
 
58
69
  def initialize(backend: nil)
@@ -0,0 +1,226 @@
1
+ # frozen_string_literal: true
2
+
3
+ module ActionAgent
4
+ # Runs a scenario evaluation: replays every selected scenario through the
5
+ # agent once per candidate model and writes one EvaluationScenarioResult per
6
+ # scenario × model plus a per-model summary and verdict on the run.
7
+ #
8
+ # The scoring, fault diagnosis and roll-up are ActiveAgent::Evals'; this
9
+ # class supplies what only the dashboard knows — how to run the agent
10
+ # (Agent#test_execute with a model override), where to persist each result,
11
+ # how to price tokens, and which judge model the owner has credentials for.
12
+ #
13
+ # The run's `scores` keep the shape the Evaluations UI renders — criterion
14
+ # => stats, or criterion => { model => stats } when comparing — and add
15
+ # underscore-prefixed summaries:
16
+ #
17
+ # "_models" — per model: pass rate, mean score, latency, tokens, cost, fault counts
18
+ # "_recommendations" — faults grouped across scenarios with the fix each calls for
19
+ # "_verdict" — the best model and why (judge-written when a judge is available)
20
+ # "_selection" — the scenarios and models this run covered
21
+ class ScenarioEvaluationRunner < EvaluationRunnerService
22
+ Evals = ActiveAgent::Evals
23
+
24
+ # `run` is an EvaluationRun created ahead of time (by run_later!, so the
25
+ # UI can show it pending while the job waits); absent, one is created here.
26
+ def self.call(evaluation, selection: {}, run: nil)
27
+ new(evaluation, selection: selection, run: run).call
28
+ end
29
+
30
+ def initialize(evaluation, selection: {}, run: nil)
31
+ super(evaluation)
32
+ @selection = (selection || {}).to_h.with_indifferent_access
33
+ @run = run
34
+ end
35
+
36
+ def call
37
+ scenarios = selected_scenarios
38
+ specs = model_specs
39
+ run = @run || @evaluation.evaluation_runs.create!(status: :pending)
40
+ run.update!(status: :running, selection: selection_summary(scenarios, specs))
41
+
42
+ if scenarios.empty?
43
+ run.update!(status: :failed, error_message: "No scenarios selected — add scenarios to the evaluation or widen the selection",
44
+ completed_at: Time.current)
45
+ return run
46
+ end
47
+
48
+ ensure_judge_defined_kpis! if @evaluation.judge_defined?
49
+
50
+ records = scenarios.index_by(&:key)
51
+ report = Evals::Runner.new(
52
+ scenarios: scenarios.map { |scenario| Evals::Scenario.from_hash(scenario.as_json_summary) },
53
+ models: specs,
54
+ criteria: sample_criteria,
55
+ judge: evals_judge,
56
+ available_tools: tool_roster,
57
+ instructions: @evaluation.agent.instructions,
58
+ agent_name: @evaluation.agent.name,
59
+ threshold: PASS_THRESHOLD,
60
+ replay: ->(scenario, spec) { replay(scenario, spec) },
61
+ on_result: ->(result) { persist(run, records.fetch(result.scenario.key), result) }
62
+ ).call
63
+
64
+ run.update!(
65
+ status: :complete,
66
+ scores: scores_for(report, run),
67
+ samples_evaluated: report.results.size,
68
+ samples_passed: report.results.count(&:passed?),
69
+ completed_at: Time.current
70
+ )
71
+ run
72
+ rescue StandardError => e
73
+ run&.update!(status: :failed, error_message: e.message, completed_at: Time.current)
74
+ raise
75
+ end
76
+
77
+ private
78
+
79
+ # --- selection --------------------------------------------------------
80
+
81
+ def selected_scenarios
82
+ scope = @evaluation.scenarios.enabled.ordered
83
+ scope = scope.where(id: Array(@selection[:scenario_ids])) if @selection[:scenario_ids].present?
84
+ scope = scope.where(key: Array(@selection[:keys])) if @selection[:keys].present?
85
+ scope = scope.in_group(@selection[:group]) if @selection[:group].present?
86
+ scope.to_a
87
+ end
88
+
89
+ # The models to compare: an explicit selection, else the evaluation's
90
+ # compare_models, else the agent as configured. `mock` is the framework's
91
+ # test double, accepted so the test suite can compare cohorts offline.
92
+ def model_specs
93
+ names = Array(@selection[:models]).presence || @evaluation.compare_models
94
+ specs = Evals::ModelSpec.parse_all(names, default_provider: @evaluation.agent.provider, providers: Agent::PROVIDERS + %w[mock])
95
+ return specs if specs.any?
96
+
97
+ [ Evals::ModelSpec.new(label: @evaluation.agent.model, provider: @evaluation.agent.provider, model: @evaluation.agent.model) ]
98
+ end
99
+
100
+ def selection_summary(scenarios, specs)
101
+ {
102
+ "scenario_ids" => scenarios.map(&:id),
103
+ "scenario_keys" => scenarios.map(&:key),
104
+ "group" => @selection[:group].presence,
105
+ "models" => specs.map(&:to_h)
106
+ }.compact
107
+ end
108
+
109
+ # --- replay -----------------------------------------------------------
110
+
111
+ def replay(scenario, spec)
112
+ # One execution per replay, reported to the host before the run starts
113
+ # (the order SandboxesController#compare uses), so it is counted even
114
+ # when the run fails.
115
+ ActionAgent.record_usage(owner, :execution)
116
+
117
+ agent_run = @evaluation.agent.test_execute(
118
+ scenario.prompt,
119
+ model_override: spec.model,
120
+ provider_override: spec.provider
121
+ )
122
+
123
+ Evals::Replay.new(
124
+ answer: agent_run.output,
125
+ tool_calls: tool_calls_for(agent_run),
126
+ duration_ms: agent_run.calculated_duration_ms,
127
+ input_tokens: agent_run.input_tokens,
128
+ output_tokens: agent_run.output_tokens,
129
+ error: agent_run.failed? ? agent_run.error_message.presence || "run failed" : nil,
130
+ cost: ModelPricing.estimate(model: spec.model, input_tokens: agent_run.input_tokens, output_tokens: agent_run.output_tokens),
131
+ metadata: { "agent_run_id" => agent_run.id }
132
+ )
133
+ end
134
+
135
+ # Each tool call the run made, rebuilt from the run's progress events
136
+ # (AgentRun#append_event pairs a "started" event with its "done"/"error"
137
+ # by eid). Falls back to the bare names in the run's metadata for a run
138
+ # recorded without events.
139
+ def tool_calls_for(agent_run)
140
+ events = Array(agent_run.logs).select { |event| event.is_a?(Hash) && %w[tool agent].include?(event["kind"]) }
141
+ if events.empty?
142
+ return Array(agent_run.output_metadata&.dig("tool_calls")).map { |name| { "name" => name.to_s } }
143
+ end
144
+
145
+ events.group_by { |event| event["eid"] }.values.map do |group|
146
+ started = group.find { |event| event["status"] == "started" }
147
+ finished = group.find { |event| %w[done error].include?(event["status"]) }
148
+ label = (started || finished)["label"].to_s
149
+
150
+ {
151
+ "name" => label.sub(/\s*→.*\z/, ""),
152
+ "arguments" => parse_json(started&.dig("detail")),
153
+ "error" => finished&.dig("status") == "error",
154
+ "detail" => finished&.dig("detail"),
155
+ "duration_ms" => finished&.dig("duration_ms")
156
+ }.compact
157
+ end
158
+ end
159
+
160
+ def parse_json(text)
161
+ return nil if text.blank?
162
+
163
+ JSON.parse(text)
164
+ rescue JSON::ParserError
165
+ text
166
+ end
167
+
168
+ # --- persistence ------------------------------------------------------
169
+
170
+ def persist(run, scenario, result)
171
+ run.scenario_results.create!(
172
+ scenario: scenario,
173
+ agent_run_id: result.replay.metadata["agent_run_id"],
174
+ model: result.model,
175
+ provider: result.provider,
176
+ status: result.status,
177
+ score: result.score,
178
+ scores: result.scores,
179
+ output: result.replay.answer.to_s.byteslice(0, 20_000).to_s.scrub.presence,
180
+ tool_calls: result.replay.tool_calls,
181
+ duration_ms: result.replay.duration_ms,
182
+ input_tokens: result.replay.input_tokens,
183
+ output_tokens: result.replay.output_tokens,
184
+ cost: result.replay.cost,
185
+ fault: result.fault,
186
+ recommendation: result.recommendation,
187
+ diagnosis: result.diagnosis || {},
188
+ error_message: result.replay.error
189
+ )
190
+ end
191
+
192
+ def scores_for(report, run)
193
+ scores = report.criterion_scores
194
+ scores["_models"] = report.summary_by_model
195
+ scores["_recommendations"] = report.recommendations
196
+ scores["_verdict"] = report.verdict if report.comparing?
197
+ scores["_selection"] = run.selection
198
+ scores
199
+ end
200
+
201
+ # --- judge ------------------------------------------------------------
202
+
203
+ def sample_criteria
204
+ @sample_criteria ||= @evaluation.criteria.reject do |criterion|
205
+ Evaluation::TELEMETRY_CRITERION_TYPES.include?(criterion["type"])
206
+ end
207
+ end
208
+
209
+ def tool_roster
210
+ @tool_roster ||= AgentToolbox.definitions_for(@evaluation.agent.tools).to_h do |definition|
211
+ [ definition[:name].to_s, definition[:description].to_s ]
212
+ end
213
+ end
214
+
215
+ # The judge the evaluation's owner has credentials for, wrapped for the
216
+ # evaluation core; nil when none is configured, in which case scoring
217
+ # stays on rules and expectations.
218
+ def evals_judge
219
+ return nil unless judge_available?
220
+
221
+ @evals_judge ||= Evals::Judge.new(label: @evaluation.judge_model.presence || judge_provider.to_s) do |instructions:, prompt:|
222
+ judge_class.prompt(message: prompt, instructions: instructions).generate_now.message&.content
223
+ end
224
+ end
225
+ end
226
+ end
@@ -30,7 +30,7 @@ module ActionAgent
30
30
  #
31
31
  # MCP attribution comes from ActiveAgent::Telemetry::ToolOrigin (the
32
32
  # +mcp__server__tool+ convention, tagged onto spans at instrumentation
33
- # time), then McpCatalog's hints for bare tool names, then the tool is
33
+ # time), then MCPCatalog's hints for bare tool names, then the tool is
34
34
  # treated as a method the agent class defines.
35
35
  #
36
36
  # Scopes are passed in rather than derived, so the caller's ownership
@@ -259,7 +259,7 @@ module ActionAgent
259
259
  end
260
260
  end
261
261
 
262
- Array(agent.mcp_servers).each do |server|
262
+ configured_mcp_servers(agent).each do |server|
263
263
  key = mcp_server_key(server)
264
264
  next if key.blank?
265
265
 
@@ -404,7 +404,7 @@ module ActionAgent
404
404
  classification = ActiveAgent::Telemetry::ToolOrigin.classify(name)
405
405
  return { origin: ORIGIN_MCP, server: classification[:server] } if classification[:server].present?
406
406
 
407
- if (hinted = McpCatalog.server_for_tool(name))
407
+ if (hinted = MCPCatalog.server_for_tool(name))
408
408
  # A catalog hint is weaker evidence than a namespaced name: the tool
409
409
  # is *probably* this server's, but a builtin of the same name is the
410
410
  # dashboard's own implementation, so builtins win the tie.
@@ -474,7 +474,7 @@ module ActionAgent
474
474
 
475
475
  def source_label(origin, server)
476
476
  case origin
477
- when ORIGIN_MCP then server.present? ? "MCP · #{McpCatalog.display_name(server)}" : "MCP"
477
+ when ORIGIN_MCP then server.present? ? "MCP · #{MCPCatalog.display_name(server)}" : "MCP"
478
478
  when ORIGIN_BUILTIN then "Dashboard toolbox"
479
479
  else "Agent-defined"
480
480
  end
@@ -500,7 +500,7 @@ module ActionAgent
500
500
  end
501
501
  end
502
502
 
503
- keys = (McpCatalog::BY_KEY.keys + detected.keys + configured_servers.keys).uniq
503
+ keys = (MCPCatalog.keys + detected.keys + configured_servers.keys).uniq
504
504
 
505
505
  # detected has a default block that would materialize a bucket on
506
506
  # lookup, so unseen servers are passed through as an explicit nil.
@@ -509,7 +509,7 @@ module ActionAgent
509
509
  end
510
510
 
511
511
  def server_row(key, bucket)
512
- catalog = McpCatalog.find(key)
512
+ catalog = MCPCatalog.find(key)
513
513
  configured = configured_servers[key].to_a.sort
514
514
  calls = bucket ? bucket[:calls] : 0
515
515
 
@@ -559,11 +559,25 @@ module ActionAgent
559
559
  @configured_servers ||= Hash.new { |hash, key| hash[key] = Set.new }
560
560
  end
561
561
 
562
+ # The servers an agent declares, as a list. Agents store an Array, but
563
+ # an agent created from an older template seed carried a top-level Hash
564
+ # keyed by server name ({"playwright" => {"command" => ...}}); Array()
565
+ # turned that into [key, value] pairs and the key lookup below raised
566
+ # TypeError on the Array, taking down /api/tools and /api/mcp_servers
567
+ # for the whole workspace.
568
+ def configured_mcp_servers(agent)
569
+ servers = agent.mcp_servers
570
+ return servers.keys if servers.is_a?(Hash)
571
+
572
+ Array(servers)
573
+ end
574
+
562
575
  # An agent's mcp_servers entries are free-form: a bare string name, or a
563
576
  # hash from the builder ({"name" => "playwright", "url" => ...}).
577
+ # Anything else (a stray Array, a number) is skipped rather than raised on.
564
578
  def mcp_server_key(server)
565
- return server.to_s.strip if server.is_a?(String)
566
- return nil unless server.respond_to?(:[])
579
+ return server.to_s.strip.presence if server.is_a?(String) || server.is_a?(Symbol)
580
+ return nil unless server.respond_to?(:key?)
567
581
 
568
582
  (server["key"] || server[:key] || server["name"] || server[:name]).to_s.strip.presence
569
583
  end