actionagent 1.8.0 → 1.8.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (32) hide show
  1. checksums.yaml +4 -4
  2. data/app/assets/builds/action_agent.js +58 -58
  3. data/app/controllers/action_agent/api/code_sessions_controller.rb +19 -7
  4. data/app/controllers/action_agent/api/evaluations_controller.rb +49 -3
  5. data/app/controllers/action_agent/api/sandboxes_controller.rb +8 -0
  6. data/app/controllers/concerns/action_agent/api/mcp_dashboard_tools.rb +4 -2
  7. data/app/jobs/action_agent/code_session_job.rb +13 -8
  8. data/app/models/action_agent/agent.rb +33 -0
  9. data/app/models/action_agent/code_session.rb +6 -2
  10. data/app/models/action_agent/evaluation.rb +64 -0
  11. data/app/models/action_agent/evaluation_run.rb +143 -53
  12. data/app/models/action_agent/evaluation_scenario_result.rb +15 -3
  13. data/app/models/action_agent/model_pricing.rb +214 -34
  14. data/app/models/action_agent/provider_key.rb +7 -1
  15. data/app/models/action_agent/sandbox_session.rb +3 -2
  16. data/app/serializers/action_agent/evaluation_serializer.rb +41 -10
  17. data/app/services/action_agent/agent_scorecard.rb +50 -21
  18. data/app/services/action_agent/codex_session_events.rb +37 -0
  19. data/app/services/action_agent/evaluation_report_import.rb +84 -5
  20. data/app/services/action_agent/evaluation_run_cost.rb +609 -0
  21. data/app/services/action_agent/evaluation_runner_service.rb +25 -5
  22. data/app/services/action_agent/evaluation_standing.rb +145 -0
  23. data/app/services/action_agent/local_sandbox_backend.rb +47 -21
  24. data/app/services/action_agent/sandbox_orchestrator.rb +14 -1
  25. data/app/services/action_agent/scenario_evaluation_runner.rb +2 -1
  26. data/config/routes.rb +1 -1
  27. data/lib/action_agent/version.rb +1 -1
  28. data/lib/action_agent.rb +6 -0
  29. data/lib/generators/action_agent/install_generator.rb +9 -6
  30. data/lib/generators/action_agent/templates/action_agent.rb.erb +3 -0
  31. data/lib/generators/action_agent/templates/add_code_session_runner.rb.erb +9 -0
  32. metadata +6 -2
@@ -29,7 +29,11 @@ module ActionAgent
29
29
 
30
30
  # POST /api/sandboxes/:sandbox_id/code_sessions
31
31
  def create
32
- if (refusal = refusal_for(@sandbox))
32
+ runner = params[:runner].nil? ? "claude_code" : params[:runner]
33
+ unless CodeSession::RUNNERS.include?(runner)
34
+ return render json: { error: "runner must be claude_code or codex" }, status: :unprocessable_entity
35
+ end
36
+ if (refusal = refusal_for(@sandbox, runner: runner))
33
37
  return render json: { error: refusal }, status: :unprocessable_entity
34
38
  end
35
39
 
@@ -41,6 +45,7 @@ module ActionAgent
41
45
  sandbox_session: @sandbox,
42
46
  prompt: string_param(:prompt),
43
47
  model: string_param(:model).presence,
48
+ runner: runner,
44
49
  # Owned like the sandbox it runs in.
45
50
  user_id: @sandbox.try(:user_id),
46
51
  account_id: @sandbox.try(:account_id)
@@ -111,16 +116,21 @@ module ActionAgent
111
116
  end
112
117
 
113
118
  # Why +sandbox+ cannot take a Claude Code session now, or nil.
114
- def refusal_for(sandbox)
115
- return "Claude Code sessions run only in a checkout (app_runtime) sandbox" unless sandbox.app_runtime?
119
+ def refusal_for(sandbox, runner: "claude_code")
120
+ label = runner == "codex" ? "Codex" : "Claude Code"
121
+ return "#{label} sessions run only in a checkout (app_runtime) sandbox" unless sandbox.app_runtime?
116
122
  return "The sandbox has expired; start a new one" if sandbox.ready? && !sandbox.active?
117
123
  return "The sandbox is #{sandbox.status}; wait until it is ready" unless sandbox.ready?
118
124
 
119
125
  orchestrator = SandboxOrchestrator.new
120
- unless orchestrator.supports?(:code_session)
121
- return "The #{orchestrator.backend_name} sandbox backend cannot run Claude Code sessions"
126
+ unless orchestrator.supports_code_runner?(runner)
127
+ return "The #{orchestrator.backend_name} sandbox backend cannot run #{label} sessions"
122
128
  end
123
129
 
130
+ if runner == "codex"
131
+ return "Codex is not connected: connect an OpenAI API key in Settings -> Integrations first" if sandbox.runtime_environment(runner: runner).blank?
132
+ return
133
+ end
124
134
  ClaudeCodeAuth.backend_refusal(orchestrator) || ClaudeCodeAuth.credential_refusal(sandbox)
125
135
  end
126
136
 
@@ -131,11 +141,13 @@ module ActionAgent
131
141
  end
132
142
 
133
143
  def busy_body(current)
134
- { error: "A Claude Code session is already #{current.status} in this sandbox", code_session: current.summary }
144
+ label = current.runner == "codex" ? "Codex" : "Claude Code"
145
+ { error: "A #{label} session is already #{current.status} in this sandbox", code_session: current.summary }
135
146
  end
136
147
 
137
148
  def invalid_request(code_session)
138
- return "model is not a Claude Code model name" if code_session.model && !code_session.model.match?(MODEL_NAME)
149
+ label = code_session.runner == "codex" ? "Codex" : "Claude Code"
150
+ return "model is not a #{label} model name" if code_session.model && !code_session.model.match?(MODEL_NAME)
139
151
 
140
152
  code_session.errors.full_messages.to_sentence unless code_session.valid?
141
153
  end
@@ -41,6 +41,11 @@ module ActionAgent
41
41
  # 50 most recent. The scope is already restricted to the current user's
42
42
  # agents, so an id outside it simply returns nothing.
43
43
  #
44
+ # Archived evaluations are left out the same way, before the limit,
45
+ # unless `archived=1` asks for them; `archived_count` says how many
46
+ # the page left out. Each evaluation carries its standing against the
47
+ # agent's current version and its headline run (EvaluationSerializer).
48
+ #
44
49
  # Three fields feed the dashboard's model pickers. They describe the
45
50
  # credentials of #picker_credentials_owner:
46
51
  # - judge_provider: the provider a judge model runs on, null
@@ -54,16 +59,36 @@ module ActionAgent
54
59
  def index
55
60
  scope = evaluations_scope
56
61
  scope = scope.where(agent_id: params[:agent_id]) if params[:agent_id].present?
57
- evaluations = scope.includes(:agent, :evaluation_runs, :scenarios).recent.limit(50)
62
+ archived_count = scope.archived.count
63
+ scope = scope.unarchived unless include_archived?
64
+ evaluations = scope.includes(:agent, :scenarios, evaluation_runs: :agent_version).recent.limit(50).to_a
65
+ preload_standing_and_costs(evaluations)
58
66
  owner = picker_credentials_owner
59
67
 
60
68
  render json: {
61
69
  evaluations: evaluations.map { |evaluation| serialize(evaluation) },
70
+ archived_count: archived_count,
62
71
  **judge_provider_fields(owner),
63
72
  model_providers: AgentExecutionService.available_providers(owner)
64
73
  }
65
74
  end
66
75
 
76
+ # PATCH /api/evaluations/:id
77
+ # `evaluation: { archived: true | false }` archives the evaluation —
78
+ # it keeps its runs but leaves the index, the pooled pass rate and the
79
+ # agent's scorecard — or brings it back. A new run brings it back too.
80
+ def update
81
+ evaluation = evaluations_scope.find(params[:id])
82
+ archived = params.require(:evaluation)[:archived]
83
+ if archived.nil?
84
+ return render json: { errors: [ "evaluation.archived must be true or false" ] }, status: :unprocessable_entity
85
+ end
86
+
87
+ ActiveModel::Type::Boolean.new.cast(archived) ? evaluation.archive! : evaluation.unarchive!
88
+
89
+ render json: { evaluation: serialize(evaluation.reload) }
90
+ end
91
+
67
92
  # Runs listed per evaluation on GET /api/evaluations/:id. The rest of
68
93
  # the history stays reachable by run id; `run_count` says how long it is.
69
94
  RUN_HISTORY_LIMIT = 20
@@ -72,7 +97,8 @@ module ActionAgent
72
97
  def show
73
98
  evaluation = evaluations_scope.find(params[:id])
74
99
  run_count = evaluation.evaluation_runs.count
75
- runs = evaluation.evaluation_runs.recent.limit(RUN_HISTORY_LIMIT).to_a
100
+ runs = evaluation.evaluation_runs.includes(:agent_version).recent.limit(RUN_HISTORY_LIMIT).to_a
101
+ EvaluationRunCost.preload(runs).each { |id, breakdown| runs.find { |run| run.id == id }&.cost_breakdown = breakdown }
76
102
 
77
103
  render json: {
78
104
  evaluation: serialize(evaluation).merge(
@@ -143,16 +169,21 @@ module ActionAgent
143
169
  # fix items — the faults grouped with the tools, MCP server and
144
170
  # dashboard action that address each. Fix item paths are relative to
145
171
  # the mount: the React app resolves them itself (dashboardPath).
172
+ # `costs` is what each scenario cost across the models, judge apart,
173
+ # and the run's total (EvaluationRunCost#costs); every result carries
174
+ # its effective cost and how it was priced.
146
175
  def show_run
147
176
  evaluation = evaluations_scope.find(params[:id])
148
177
  run = evaluation.evaluation_runs.find(params[:run_id])
149
178
  results = run.scenario_results.includes(:scenario).joins(:scenario)
150
179
  .order(EvaluationScenario.arel_table[:position], EvaluationScenario.arel_table[:id], :model)
180
+ breakdown = run.cost_breakdown
151
181
 
152
182
  render json: {
153
183
  evaluation: serialize(evaluation),
154
184
  run: serialize_run(run, number: run_number(evaluation, run))
155
- .merge(results: results.map(&:as_json_summary), fix_items: safe_fix_items(run))
185
+ .merge(results: results.map { |result| result.as_json_summary(costs: breakdown.result(result)) },
186
+ fix_items: safe_fix_items(run), costs: breakdown.costs)
156
187
  }
157
188
  end
158
189
 
@@ -225,6 +256,21 @@ module ActionAgent
225
256
 
226
257
  private
227
258
 
259
+ def include_archived?
260
+ ActiveModel::Type::Boolean.new.cast(params[:archived]) == true
261
+ end
262
+
263
+ # One query per table for the page's standings and for the costs of
264
+ # the runs it serializes in full (each evaluation's latest and headline
265
+ # run), rather than one per evaluation.
266
+ def preload_standing_and_costs(evaluations)
267
+ EvaluationStanding.preload(evaluations)
268
+ runs = evaluations.flat_map do |evaluation|
269
+ [ EvaluationSerializer.recent_runs(evaluation, 1).first, evaluation.standing_info.headline_run ]
270
+ end.compact.uniq(&:id)
271
+ EvaluationRunCost.preload(runs).each { |id, breakdown| runs.find { |run| run.id == id }&.cost_breakdown = breakdown }
272
+ end
273
+
228
274
  # Returns whose credentials the index's model picker fields describe.
229
275
  # Agent runs and their judge use the evaluated agent's owner's
230
276
  # credentials, so a list scoped to one agent reads that agent's owner,
@@ -97,6 +97,8 @@ module ActionAgent
97
97
  sample_tasks: sample_tasks,
98
98
  sandboxes: listed_sandboxes.map(&:summary),
99
99
  code_sessions_supported: code_sessions_supported?,
100
+ codex_sessions_supported: codex_sessions_supported?,
101
+ codex_connected: owned(ProviderKey).where(provider: "codex").any? { |key| key.runtime_environment.present? },
100
102
  **claude_code_status
101
103
  }
102
104
  end
@@ -234,6 +236,12 @@ module ActionAgent
234
236
  false
235
237
  end
236
238
 
239
+ def codex_sessions_supported?
240
+ SandboxOrchestrator.new.supports_code_runner?("codex")
241
+ rescue StandardError, LoadError
242
+ false
243
+ end
244
+
237
245
  # Whether the caller's Claude Code can run sessions, by
238
246
  # ClaudeCodeAuth's rule: an API key they connected, or this machine's
239
247
  # own login. Never a credential.
@@ -202,13 +202,15 @@ module ActionAgent
202
202
  scope = scope.where(agent: resolve_tool_agent!(agent))
203
203
  end
204
204
  limit = integer_argument(:limit, default: LIST_LIMIT, min: 1, max: MAX_LIST_LIMIT)
205
- evaluations = scope.includes(:agent, :evaluation_runs, :scenarios).recent.limit(limit)
205
+ evaluations = scope.includes(:agent, :scenarios, evaluation_runs: :agent_version).recent.limit(limit).to_a
206
+ EvaluationStanding.preload(evaluations)
206
207
 
207
208
  {
208
209
  evaluations: evaluations.map do |evaluation|
209
210
  summary = EvaluationSerializer.summary(evaluation)
210
211
  latest = EvaluationSerializer.recent_runs(evaluation, 1).first
211
- summary.slice(:id, :name, :agent, :judge_kind, :scenario_suite, :scenario_count, :scenario_groups, :created_at, :run_count)
212
+ summary.slice(:id, :name, :agent, :judge_kind, :scenario_suite, :scenario_count, :scenario_groups, :created_at, :run_count,
213
+ :headline_run_id, :standing, :archived_at)
212
214
  .merge(latest_run: latest && EvaluationSerializer.run_summary(latest, number: summary[:run_count]))
213
215
  end
214
216
  }
@@ -24,6 +24,7 @@ module ActionAgent
24
24
  # connection and the Claude Code key.
25
25
  secrets = code_session.secrets
26
26
  result_event = nil
27
+ codex_events = CodexSessionEvents.new if code_session.runner == "codex"
27
28
  stop_sent = false
28
29
  orchestrator = SandboxOrchestrator.new
29
30
  sandbox = code_session.sandbox_session
@@ -41,6 +42,10 @@ module ActionAgent
41
42
  # has no process to stop; an event means the process exists now, so
42
43
  # the stop is sent again, once.
43
44
  code_session.append_event!(event, secrets: secrets)
45
+ if codex_events
46
+ event = codex_events.consume(event)
47
+ code_session.append_event!(event, secrets: secrets) if event
48
+ end
44
49
  if code_session.cancelled? && !stop_sent
45
50
  stop_sent = true
46
51
  stop(orchestrator, sandbox, code_session)
@@ -54,7 +59,7 @@ module ActionAgent
54
59
  finish(code_session, outcome.to_h, result_event, secrets)
55
60
  rescue StandardError => e
56
61
  message = SecretScrubber.scrub(e.message.to_s, secrets || safe_secrets(code_session))
57
- Rails.logger.error("Claude Code session #{code_session_id} failed: #{message}")
62
+ Rails.logger.error("Code session #{code_session_id} failed: #{message}")
58
63
  fail!(code_session, message) if code_session
59
64
  end
60
65
 
@@ -83,7 +88,7 @@ module ActionAgent
83
88
  else
84
89
  code_session.assign_attributes(
85
90
  status: :failed,
86
- error_message: failure_message(result_event, outcome, secrets),
91
+ error_message: failure_message(result_event, outcome, secrets, label: code_session.runner == "codex" ? "Codex" : "Claude Code"),
87
92
  finished_at: Time.current
88
93
  )
89
94
  end
@@ -105,8 +110,8 @@ module ActionAgent
105
110
 
106
111
  # What went wrong, in Claude Code's own words when it reported a failure
107
112
  # and from the process otherwise.
108
- def failure_message(result_event, outcome, secrets)
109
- message = reported_failure(result_event) || process_failure(result_event, outcome)
113
+ def failure_message(result_event, outcome, secrets, label: "Claude Code")
114
+ message = reported_failure(result_event) || process_failure(result_event, outcome, label: label)
110
115
  SecretScrubber.scrub(message, secrets).truncate(MAX_ERROR_MESSAGE)
111
116
  end
112
117
 
@@ -119,12 +124,12 @@ module ActionAgent
119
124
  "Claude Code stopped: #{result_event['subtype']}"
120
125
  end
121
126
 
122
- def process_failure(result_event, outcome)
127
+ def process_failure(result_event, outcome, label: "Claude Code")
123
128
  status = outcome[:exit_status]
124
129
  headline =
125
- if !status.nil? && status != 0 then "Claude Code exited with status #{status}"
126
- elsif result_event.nil? then "Claude Code ended without reporting a result"
127
- else "Claude Code did not finish"
130
+ if !status.nil? && status != 0 then "#{label} exited with status #{status}"
131
+ elsif result_event.nil? then "#{label} ended without reporting a result"
132
+ else "#{label} did not finish"
128
133
  end
129
134
 
130
135
  [ headline, outcome[:stderr_tail].to_s.strip.presence ].compact.join(": ")
@@ -313,6 +313,39 @@ module ActionAgent
313
313
  version
314
314
  end
315
315
 
316
+ # The version carrying a release an application reports having
317
+ # evaluated (EvaluationReportImport reads it from `report.release`),
318
+ # found among every version of this agent — a report may describe a
319
+ # deploy older than the latest — or recorded now when none does. A
320
+ # version recorded this way carries no manifest: the report names the
321
+ # digest, not what went into it.
322
+ #
323
+ # The agent's own `release_digest` is what its deploys last recorded
324
+ # (#record_release!); a report never moves it, except to set it for an
325
+ # agent that never recorded a deploy at all.
326
+ #
327
+ # @param digest [String] the ActiveAgent::Release digest the report names
328
+ # @param revision [String, nil] the deploy the report names
329
+ # @param label [String, nil] how the report labels the release
330
+ # @return [AgentVersion]
331
+ def find_or_record_release!(digest:, revision: nil, label: nil)
332
+ existing = agent_versions.find_by(release_digest: digest)
333
+ return existing if existing
334
+
335
+ version = agent_versions.create!(
336
+ version_number: (latest_version&.version_number || 0) + 1,
337
+ change_summary: [ "Release #{digest}", revision.presence, label.presence ].compact.join(" · ") + ": reported by an evaluation",
338
+ configuration_snapshot: configuration_snapshot.merge("release" => {}),
339
+ release_digest: digest,
340
+ revision: revision,
341
+ created_by: "evaluation-report"
342
+ )
343
+ update_columns(release_digest: digest) if release_digest.blank?
344
+ version
345
+ rescue ActiveRecord::RecordNotUnique
346
+ agent_versions.find_by!(release_digest: digest)
347
+ end
348
+
316
349
  # Maps each historical instructions digest to the first version that
317
350
  # introduced it ("v3"), so run cohorts can label instruction changes with
318
351
  # real agent versions instead of raw hashes.
@@ -22,8 +22,10 @@ module ActionAgent
22
22
  # Per string inside an event: tool results can be whole files.
23
23
  MAX_EVENT_STRING = 4_000
24
24
  MAX_DIFF_BYTES = 500_000
25
+ RUNNERS = %w[claude_code codex].freeze
25
26
 
26
27
  validates :prompt, presence: true, length: { maximum: MAX_PROMPT_CHARACTERS }
28
+ validates :runner, inclusion: { in: RUNNERS }
27
29
 
28
30
  scope :recent, -> { order(created_at: :desc) }
29
31
 
@@ -51,7 +53,7 @@ module ActionAgent
51
53
  # Claude Code credential this session ran with.
52
54
  def secrets
53
55
  spec = sandbox_session.checkout_spec rescue nil
54
- [ spec&.dig(:token), *sandbox_session.runtime_environment.values ].compact
56
+ [ spec&.dig(:token), *sandbox_session.runtime_environment(runner: runner).values ].compact
55
57
  end
56
58
 
57
59
  # Appends one stream-json event, scrubbed and bounded. Past MAX_EVENTS
@@ -77,7 +79,8 @@ module ActionAgent
77
79
 
78
80
  update!(
79
81
  result: event["result"].to_s.presence && SecretScrubber.scrub(event["result"].to_s, secrets).truncate(MAX_EVENT_STRING * 4),
80
- claude_session_id: event["session_id"],
82
+ claude_session_id: runner == "claude_code" ? event["session_id"] : nil,
83
+ runner_session_id: event["session_id"],
81
84
  num_turns: event["num_turns"],
82
85
  duration_ms: event["duration_ms"],
83
86
  total_cost_usd: event["total_cost_usd"],
@@ -98,6 +101,7 @@ module ActionAgent
98
101
  status: status,
99
102
  prompt: prompt,
100
103
  model: model,
104
+ runner: runner,
101
105
  result: result,
102
106
  error_message: error_message,
103
107
  num_turns: num_turns,
@@ -34,6 +34,20 @@ module ActionAgent
34
34
  validate :validate_criteria
35
35
 
36
36
  scope :recent, -> { order(updated_at: :desc) }
37
+ # Archived evaluations keep their runs but leave the index and its
38
+ # tiles unless asked for (see #archive!). The mark lives in the config
39
+ # JSON, so the filter reads it with each database's own JSON path.
40
+ scope :archived, -> { where("#{archived_at_sql} IS NOT NULL") }
41
+ scope :unarchived, -> { where("#{archived_at_sql} IS NULL") }
42
+
43
+ # SQL reading config["archived_at"] on the connected database.
44
+ def self.archived_at_sql
45
+ case connection.adapter_name.to_s.downcase
46
+ when /postgres/ then "#{quoted_table_name}.config ->> 'archived_at'"
47
+ when /mysql|trilogy/ then "JSON_UNQUOTE(JSON_EXTRACT(#{quoted_table_name}.config, '$.archived_at'))"
48
+ else "json_extract(#{quoted_table_name}.config, '$.archived_at')"
49
+ end
50
+ end
37
51
 
38
52
  # MySQL cannot give a JSON column a default, so a row inserted there
39
53
  # without `criteria` or `config` reads back nil. Both readers answer with
@@ -50,6 +64,56 @@ module ActionAgent
50
64
  evaluation_runs.order(created_at: :desc).first
51
65
  end
52
66
 
67
+ # The newest run that finished: the one the evaluation's pass rate
68
+ # describes. A newer run still pending, or one that failed, shows
69
+ # beside it and never in its place. Read from the loaded association
70
+ # when the index preloaded it.
71
+ def headline_run
72
+ if evaluation_runs.loaded?
73
+ evaluation_runs.select(&:complete?).max_by { |run| [ run.created_at, run.id ] }
74
+ else
75
+ evaluation_runs.complete.order(created_at: :desc, id: :desc).first
76
+ end
77
+ end
78
+
79
+ # --- archiving ---------------------------------------------------------
80
+ #
81
+ # An evaluation nobody maintains — its suite superseded, its agent
82
+ # retired — keeps its history but stops counting: the index leaves it
83
+ # out, and so do the pooled pass rate and the agent's scorecard. The
84
+ # mark is config["archived_at"]; a new run or a published report clears
85
+ # it, since either says the evaluation is alive after all.
86
+
87
+ def archived_at
88
+ value = config["archived_at"]
89
+ value.present? ? Time.zone.parse(value.to_s) : nil
90
+ rescue ArgumentError
91
+ nil
92
+ end
93
+
94
+ def archived?
95
+ config["archived_at"].present?
96
+ end
97
+
98
+ def archive!
99
+ update!(config: config.merge("archived_at" => Time.current.iso8601))
100
+ end
101
+
102
+ def unarchive!
103
+ return unless archived?
104
+
105
+ update!(config: config.except("archived_at"))
106
+ end
107
+
108
+ # Where this evaluation stands against the agent as it is now: whether
109
+ # its headline run scored the current version (EvaluationStanding).
110
+ # Memoized per instance; the index preloads what it needs.
111
+ def standing_info
112
+ @standing_info ||= EvaluationStanding.new(self)
113
+ end
114
+
115
+ attr_writer :standing_info
116
+
53
117
  def judge_defined?
54
118
  judge_kind == "judge_defined"
55
119
  end