actionagent 1.8.0 → 1.8.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/app/assets/builds/action_agent.js +58 -58
- data/app/controllers/action_agent/api/code_sessions_controller.rb +19 -7
- data/app/controllers/action_agent/api/evaluations_controller.rb +49 -3
- data/app/controllers/action_agent/api/sandboxes_controller.rb +8 -0
- data/app/controllers/concerns/action_agent/api/mcp_dashboard_tools.rb +4 -2
- data/app/jobs/action_agent/code_session_job.rb +13 -8
- data/app/models/action_agent/agent.rb +33 -0
- data/app/models/action_agent/code_session.rb +6 -2
- data/app/models/action_agent/evaluation.rb +64 -0
- data/app/models/action_agent/evaluation_run.rb +143 -53
- data/app/models/action_agent/evaluation_scenario_result.rb +15 -3
- data/app/models/action_agent/model_pricing.rb +214 -34
- data/app/models/action_agent/provider_key.rb +7 -1
- data/app/models/action_agent/sandbox_session.rb +3 -2
- data/app/serializers/action_agent/evaluation_serializer.rb +41 -10
- data/app/services/action_agent/agent_scorecard.rb +50 -21
- data/app/services/action_agent/codex_session_events.rb +37 -0
- data/app/services/action_agent/evaluation_report_import.rb +84 -5
- data/app/services/action_agent/evaluation_run_cost.rb +609 -0
- data/app/services/action_agent/evaluation_runner_service.rb +25 -5
- data/app/services/action_agent/evaluation_standing.rb +145 -0
- data/app/services/action_agent/local_sandbox_backend.rb +47 -21
- data/app/services/action_agent/sandbox_orchestrator.rb +14 -1
- data/app/services/action_agent/scenario_evaluation_runner.rb +2 -1
- data/config/routes.rb +1 -1
- data/lib/action_agent/version.rb +1 -1
- data/lib/action_agent.rb +6 -0
- data/lib/generators/action_agent/install_generator.rb +9 -6
- data/lib/generators/action_agent/templates/action_agent.rb.erb +3 -0
- data/lib/generators/action_agent/templates/add_code_session_runner.rb.erb +9 -0
- metadata +6 -2
|
@@ -0,0 +1,145 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module ActionAgent
|
|
4
|
+
# Where an evaluation stands against the agent as it is now, so a pass
|
|
5
|
+
# rate on the Evaluations page describes the current agent rather than
|
|
6
|
+
# pooling every suite's last run, however old the code it scored.
|
|
7
|
+
#
|
|
8
|
+
# The evaluation's headline run is its newest complete run; a newer run
|
|
9
|
+
# still pending or failed shows beside it, never in its place. Its
|
|
10
|
+
# standing is one of:
|
|
11
|
+
#
|
|
12
|
+
# current — the headline run scored the agent's current version
|
|
13
|
+
# stale — it scored an earlier version, or a release the dashboard
|
|
14
|
+
# has not matched to the agent's current release
|
|
15
|
+
# unrecorded — the run recorded no version and the agent has no release
|
|
16
|
+
# versions at all, so nothing says which code it scored
|
|
17
|
+
# archived — the evaluation is archived (Evaluation#archive!)
|
|
18
|
+
# none — no complete run
|
|
19
|
+
#
|
|
20
|
+
# Only edits the model can see count as a new version: the instructions,
|
|
21
|
+
# the action prompts, the tools, the MCP servers, the model config and
|
|
22
|
+
# the response format. A dashboard edit to the agent's appearance cuts an
|
|
23
|
+
# AgentVersion like any other edit, but the run before it scored the same
|
|
24
|
+
# agent and stays current.
|
|
25
|
+
class EvaluationStanding
|
|
26
|
+
STANDINGS = %w[current stale unrecorded archived none].freeze
|
|
27
|
+
VERSION_STATES = %w[current earlier unrecorded].freeze
|
|
28
|
+
# The configuration snapshot keys the model is given.
|
|
29
|
+
MODEL_FACING = %w[instructions action_prompts tools mcp_servers model_config response_format].freeze
|
|
30
|
+
|
|
31
|
+
# Gives every evaluation its standing with one query per table rather
|
|
32
|
+
# than one per evaluation: the agents' latest versions and whether each
|
|
33
|
+
# has a release are read for the page at once.
|
|
34
|
+
def self.preload(evaluations)
|
|
35
|
+
evaluations = Array(evaluations)
|
|
36
|
+
agent_ids = evaluations.filter_map(&:agent_id).uniq
|
|
37
|
+
return if agent_ids.empty?
|
|
38
|
+
|
|
39
|
+
newest = AgentVersion.where(agent_id: agent_ids).group(:agent_id).maximum(:version_number)
|
|
40
|
+
latest = AgentVersion.where(agent_id: agent_ids).where(version_number: newest.values.uniq)
|
|
41
|
+
.select { |version| newest[version.agent_id] == version.version_number }.index_by(&:agent_id)
|
|
42
|
+
releases = AgentVersion.releases.where(agent_id: agent_ids).distinct.pluck(:agent_id).to_set
|
|
43
|
+
|
|
44
|
+
evaluations.each do |evaluation|
|
|
45
|
+
evaluation.standing_info = new(evaluation, latest_version: latest[evaluation.agent_id], has_releases: releases.include?(evaluation.agent_id))
|
|
46
|
+
end
|
|
47
|
+
end
|
|
48
|
+
|
|
49
|
+
attr_reader :evaluation
|
|
50
|
+
|
|
51
|
+
# @param headline_run [EvaluationRun, nil] the headline run when the
|
|
52
|
+
# caller already holds it (AgentScorecard selects one per evaluation)
|
|
53
|
+
def initialize(evaluation, latest_version: :unknown, has_releases: :unknown, headline_run: :unknown)
|
|
54
|
+
@evaluation = evaluation
|
|
55
|
+
@latest_version = latest_version
|
|
56
|
+
@has_releases = has_releases
|
|
57
|
+
@headline_run = headline_run
|
|
58
|
+
end
|
|
59
|
+
|
|
60
|
+
def headline_run
|
|
61
|
+
return @headline_run unless @headline_run == :unknown
|
|
62
|
+
|
|
63
|
+
@headline_run = evaluation.headline_run
|
|
64
|
+
end
|
|
65
|
+
|
|
66
|
+
# The same standing read against +run+ as the headline run.
|
|
67
|
+
def with_headline(run)
|
|
68
|
+
self.class.new(evaluation, latest_version: latest_version, has_releases: has_releases?, headline_run: run)
|
|
69
|
+
end
|
|
70
|
+
|
|
71
|
+
def standing
|
|
72
|
+
return "archived" if evaluation.archived?
|
|
73
|
+
return "none" unless headline_run
|
|
74
|
+
|
|
75
|
+
case version_state(headline_run)
|
|
76
|
+
when "current" then "current"
|
|
77
|
+
when "unrecorded" then has_releases? ? "stale" : "unrecorded"
|
|
78
|
+
else "stale"
|
|
79
|
+
end
|
|
80
|
+
end
|
|
81
|
+
|
|
82
|
+
# Whether +run+ scored the agent's current version: "current",
|
|
83
|
+
# "earlier", or "unrecorded" when it recorded no version.
|
|
84
|
+
def version_state(run)
|
|
85
|
+
version = run&.agent_version
|
|
86
|
+
return "unrecorded" unless version
|
|
87
|
+
|
|
88
|
+
# The agent's last recorded deploy is the truth about the code that
|
|
89
|
+
# runs: a release version matching it is current whatever the
|
|
90
|
+
# dashboard edited since, and one that does not is earlier even when
|
|
91
|
+
# it was recorded last (a report of an older deploy published late).
|
|
92
|
+
if version.release? && (deployed = evaluation.agent&.release_digest.presence)
|
|
93
|
+
return deployed == version.release_digest ? "current" : "earlier"
|
|
94
|
+
end
|
|
95
|
+
|
|
96
|
+
latest = latest_version
|
|
97
|
+
return "current" if latest.nil? || latest.id == version.id
|
|
98
|
+
return "current" if model_facing(version) == model_facing(latest)
|
|
99
|
+
|
|
100
|
+
"earlier"
|
|
101
|
+
end
|
|
102
|
+
|
|
103
|
+
# Passes per model label of the headline run, `{ label => { passed, total } }`,
|
|
104
|
+
# from its recorded summaries; empty without them.
|
|
105
|
+
def per_model
|
|
106
|
+
run = headline_run
|
|
107
|
+
return {} unless run
|
|
108
|
+
|
|
109
|
+
summaries = run.scores.is_a?(Hash) ? (run.scores["_models"].presence || run.scores["_cohorts"]) : nil
|
|
110
|
+
return {} unless summaries.is_a?(Hash)
|
|
111
|
+
|
|
112
|
+
summaries.filter_map do |label, stats|
|
|
113
|
+
next unless stats.is_a?(Hash)
|
|
114
|
+
|
|
115
|
+
total = (stats["scenarios"] || stats["samples"]).to_i
|
|
116
|
+
[ label, { passed: stats["passed"].to_i, total: total } ]
|
|
117
|
+
end.to_h
|
|
118
|
+
end
|
|
119
|
+
|
|
120
|
+
# What the index and the scorecards pool: a current or unrecorded
|
|
121
|
+
# headline run; a stale or archived evaluation is left out and said so.
|
|
122
|
+
def counted?
|
|
123
|
+
%w[current unrecorded].include?(standing)
|
|
124
|
+
end
|
|
125
|
+
|
|
126
|
+
private
|
|
127
|
+
|
|
128
|
+
def latest_version
|
|
129
|
+
return @latest_version unless @latest_version == :unknown
|
|
130
|
+
|
|
131
|
+
@latest_version = evaluation.agent&.latest_version
|
|
132
|
+
end
|
|
133
|
+
|
|
134
|
+
def has_releases?
|
|
135
|
+
return @has_releases unless @has_releases == :unknown
|
|
136
|
+
|
|
137
|
+
@has_releases = evaluation.agent&.agent_versions&.releases&.exists? || false
|
|
138
|
+
end
|
|
139
|
+
|
|
140
|
+
def model_facing(version)
|
|
141
|
+
snapshot = version.configuration_snapshot.to_h.stringify_keys
|
|
142
|
+
MODEL_FACING.to_h { |key| [ key, snapshot[key] ] }
|
|
143
|
+
end
|
|
144
|
+
end
|
|
145
|
+
end
|
|
@@ -141,7 +141,7 @@ module ActionAgent
|
|
|
141
141
|
# ANTHROPIC_BASE_URL; a session inheriting those joins the developer's
|
|
142
142
|
# session, and a base URL redirects the owner's credential. A session
|
|
143
143
|
# gets exactly the Claude Code variables the backend sets.
|
|
144
|
-
\A(?:ANTHROPIC|CLAUDE|OPENAI|OPEN_AI|OPENROUTER|OPEN_ROUTER|OLLAMA)(?:_|\z) | \ACLAUDECODE\z
|
|
144
|
+
\A(?:ANTHROPIC|CLAUDE|CODEX|OPENAI|OPEN_AI|OPENROUTER|OPEN_ROUTER|OLLAMA)(?:_|\z) | \ACLAUDECODE\z
|
|
145
145
|
/x
|
|
146
146
|
SECRET_VARIABLE = /
|
|
147
147
|
SECRET | TOKEN | PASSWORD | PASSWD | PASSPHRASE | API_KEY | APIKEY | PRIVATE_KEY | CREDENTIAL | ACCESS_KEY |
|
|
@@ -436,7 +436,11 @@ module ActionAgent
|
|
|
436
436
|
0
|
|
437
437
|
end
|
|
438
438
|
|
|
439
|
-
|
|
439
|
+
def code_runners
|
|
440
|
+
%w[claude_code codex]
|
|
441
|
+
end
|
|
442
|
+
|
|
443
|
+
# Runs a coding agent headless in the sandbox's checkout, yielding each
|
|
440
444
|
# stream-json event (a Hash, already scrubbed of the sandbox's secrets)
|
|
441
445
|
# as it arrives.
|
|
442
446
|
#
|
|
@@ -448,9 +452,12 @@ module ActionAgent
|
|
|
448
452
|
app = workspace.join("app")
|
|
449
453
|
raise Error, "Sandbox #{session_id} has no local checkout: start the sandbox again" unless app.directory?
|
|
450
454
|
|
|
451
|
-
|
|
455
|
+
runner = code_session.try(:runner) || "claude_code"
|
|
456
|
+
raise Error, "Unsupported code runner: #{runner}" unless code_runners.include?(runner)
|
|
457
|
+
|
|
458
|
+
credentials = session_credentials(sandbox, runner: runner)
|
|
452
459
|
secrets = sandbox_secrets(sandbox, credentials)
|
|
453
|
-
argv = claude_argv(code_session)
|
|
460
|
+
argv = runner == "codex" ? codex_argv(code_session) : claude_argv(code_session)
|
|
454
461
|
database_env = read_state(workspace)["database_env"]
|
|
455
462
|
env = self.class.sanitized_environment
|
|
456
463
|
.merge(database_env.is_a?(Hash) ? database_env.transform_values(&:to_s) : {})
|
|
@@ -466,12 +473,15 @@ module ActionAgent
|
|
|
466
473
|
# own, in the workspace. With the machine's own login it must use the
|
|
467
474
|
# user's: that is where `claude /login` left the credentials (HOME,
|
|
468
475
|
# which the sanitized environment keeps, or the keychain).
|
|
469
|
-
|
|
476
|
+
if runner == "codex"
|
|
477
|
+
env["CODEX_HOME"] = workspace.join("codex").to_s
|
|
478
|
+
FileUtils.mkdir_p(workspace.join("codex"), mode: 0o700)
|
|
479
|
+
elsif !ClaudeCodeAuth.local_login?
|
|
470
480
|
env["CLAUDE_CONFIG_DIR"] = workspace.join("claude").to_s
|
|
471
481
|
FileUtils.mkdir_p(workspace.join("claude"), mode: 0o700)
|
|
472
482
|
end
|
|
473
483
|
|
|
474
|
-
run_claude(workspace, code_session, argv, env, secrets, &on_event)
|
|
484
|
+
run_claude(workspace, code_session, argv, env, secrets, runner: runner, &on_event)
|
|
475
485
|
end
|
|
476
486
|
|
|
477
487
|
# Stops a running Claude Code session: SIGTERM to its process group. The
|
|
@@ -969,6 +979,19 @@ module ActionAgent
|
|
|
969
979
|
|
|
970
980
|
# --- Claude Code --------------------------------------------------------
|
|
971
981
|
|
|
982
|
+
def codex_argv(code_session)
|
|
983
|
+
argv = [
|
|
984
|
+
ActionAgent.codex_command.to_s, "exec", "--json", "--ephemeral",
|
|
985
|
+
"--sandbox", "workspace-write", "--config", 'approval_policy="never"', "--color", "never"
|
|
986
|
+
]
|
|
987
|
+
if (model = code_session.model.presence)
|
|
988
|
+
raise Error, "#{model.inspect} is not a model name" unless MODEL_NAME.match?(model.to_s)
|
|
989
|
+
|
|
990
|
+
argv += [ "--model", model.to_s ]
|
|
991
|
+
end
|
|
992
|
+
argv + [ "-" ]
|
|
993
|
+
end
|
|
994
|
+
|
|
972
995
|
def claude_argv(code_session)
|
|
973
996
|
command = ActionAgent.claude_code_command.to_s
|
|
974
997
|
argv = [
|
|
@@ -1009,18 +1032,20 @@ module ActionAgent
|
|
|
1009
1032
|
LOGGED_OUT
|
|
1010
1033
|
end
|
|
1011
1034
|
|
|
1012
|
-
def run_claude(workspace, code_session, argv, env, secrets, &on_event)
|
|
1035
|
+
def run_claude(workspace, code_session, argv, env, secrets, runner: "claude_code", &on_event)
|
|
1036
|
+
label = runner == "codex" ? "Codex" : "Claude Code"
|
|
1037
|
+
timeout = runner == "codex" ? ActionAgent.codex_timeout : ActionAgent.claude_code_timeout
|
|
1013
1038
|
key = code_session.id.to_s
|
|
1014
|
-
log = log_path(workspace, "claude-#{key}")
|
|
1015
|
-
deadline = deadline_after(
|
|
1039
|
+
log = log_path(workspace, "#{runner == 'codex' ? 'codex' : 'claude'}-#{key}")
|
|
1040
|
+
deadline = deadline_after(timeout)
|
|
1016
1041
|
# Checked again once Claude Code is recorded; this saves starting it.
|
|
1017
1042
|
state = read_state(workspace)
|
|
1018
|
-
refuse_stopped_session!(state, key)
|
|
1043
|
+
refuse_stopped_session!(state, key, label: label)
|
|
1019
1044
|
# A cancelled session frees its slot as soon as it is marked cancelled,
|
|
1020
1045
|
# while its Claude Code may still be exiting (and diffing). Two in one
|
|
1021
1046
|
# checkout would edit the same files.
|
|
1022
1047
|
if other_session_running?(state, key, workspace.basename.to_s)
|
|
1023
|
-
raise Error, "The previous
|
|
1048
|
+
raise Error, "The previous code session in this sandbox is still stopping; try again in a moment"
|
|
1024
1049
|
end
|
|
1025
1050
|
stdin_read, stdin_write = IO.pipe
|
|
1026
1051
|
stdout_read, stdout_write = IO.pipe
|
|
@@ -1029,7 +1054,7 @@ module ActionAgent
|
|
|
1029
1054
|
begin
|
|
1030
1055
|
pid = spawn_group(env, *argv, chdir: workspace.join("app"), in: stdin_read, out: stdout_write, err: stderr_write)
|
|
1031
1056
|
rescue SystemCallError => e
|
|
1032
|
-
raise Error, "Could not start
|
|
1057
|
+
raise Error, "Could not start #{label} (#{argv.first}): #{e.message}"
|
|
1033
1058
|
ensure
|
|
1034
1059
|
[ stdin_read, stdout_write, stderr_write ].each(&:close)
|
|
1035
1060
|
end
|
|
@@ -1040,7 +1065,7 @@ module ActionAgent
|
|
|
1040
1065
|
case record_code_session(workspace, key, pid)
|
|
1041
1066
|
when :terminating
|
|
1042
1067
|
# Stopped by the ensure below, before it had the prompt.
|
|
1043
|
-
raise Error, "The sandbox is being stopped, so
|
|
1068
|
+
raise Error, "The sandbox is being stopped, so #{label} did not run"
|
|
1044
1069
|
when :cancelled
|
|
1045
1070
|
# Runs its course like any cancelled session: it ends on SIGTERM.
|
|
1046
1071
|
signal_group(pid, "TERM")
|
|
@@ -1067,7 +1092,7 @@ module ActionAgent
|
|
|
1067
1092
|
unless finished
|
|
1068
1093
|
# Cancelled, and SIGTERM did not end it within the grace: SIGKILL,
|
|
1069
1094
|
# and the session finishes like any cancelled one.
|
|
1070
|
-
raise Error, "
|
|
1095
|
+
raise Error, "#{label} did not finish within #{timeout}s and was stopped" unless cancel_seen
|
|
1071
1096
|
|
|
1072
1097
|
stop_groups([ pid ], grace: 0)
|
|
1073
1098
|
end
|
|
@@ -1091,11 +1116,11 @@ module ActionAgent
|
|
|
1091
1116
|
|
|
1092
1117
|
# Why a session must not start: a terminate under way, or a cancel that
|
|
1093
1118
|
# came before there was a process to stop.
|
|
1094
|
-
def refuse_stopped_session!(state, key)
|
|
1095
|
-
raise Error, "The sandbox is being stopped, so
|
|
1119
|
+
def refuse_stopped_session!(state, key, label: "Claude Code")
|
|
1120
|
+
raise Error, "The sandbox is being stopped, so #{label} did not run" if state["terminating"]
|
|
1096
1121
|
|
|
1097
1122
|
cancels = state["cancelled_code_sessions"]
|
|
1098
|
-
raise Error, "
|
|
1123
|
+
raise Error, "#{label} session #{key} was cancelled before it started" if cancels.is_a?(Hash) && cancels.key?(key)
|
|
1099
1124
|
end
|
|
1100
1125
|
|
|
1101
1126
|
# Records Claude Code's pid (and start time) under the state.json lock,
|
|
@@ -1263,12 +1288,13 @@ module ActionAgent
|
|
|
1263
1288
|
# The Claude Code variables a session runs with. An API key comes from
|
|
1264
1289
|
# the owner's connection. The machine's own login needs none: `claude`
|
|
1265
1290
|
# finds it itself, and the dashboard neither reads nor passes it on.
|
|
1266
|
-
def session_credentials(sandbox)
|
|
1267
|
-
return {} if ClaudeCodeAuth.local_login?
|
|
1291
|
+
def session_credentials(sandbox, runner: "claude_code")
|
|
1292
|
+
return {} if runner == "claude_code" && ClaudeCodeAuth.local_login?
|
|
1268
1293
|
|
|
1269
|
-
credentials = sandbox.runtime_environment.to_h
|
|
1294
|
+
credentials = (runner == "codex" ? sandbox.runtime_environment(runner: runner) : sandbox.runtime_environment).to_h
|
|
1270
1295
|
if credentials.empty?
|
|
1271
|
-
|
|
1296
|
+
label = runner == "codex" ? "Codex is not connected: connect an OpenAI API key" : "Claude Code is not connected: connect an Anthropic API key"
|
|
1297
|
+
raise Error, "#{label} in Settings → Integrations"
|
|
1272
1298
|
end
|
|
1273
1299
|
|
|
1274
1300
|
credentials
|
|
@@ -191,13 +191,26 @@ module ActionAgent
|
|
|
191
191
|
ADAPTER_METHODS.fetch(verb).any? { |m| @backend.respond_to?(m) }
|
|
192
192
|
end
|
|
193
193
|
|
|
194
|
+
# Existing backends predate runner selection and support Claude only.
|
|
195
|
+
# A backend must explicitly advertise Codex before accepting its keys.
|
|
196
|
+
def supports_code_runner?(runner)
|
|
197
|
+
return false unless supports?(:code_session)
|
|
198
|
+
|
|
199
|
+
runners = @backend.respond_to?(:code_runners) ? @backend.code_runners : [ "claude_code" ]
|
|
200
|
+
Array(runners).include?(runner)
|
|
201
|
+
end
|
|
202
|
+
|
|
194
203
|
# Runs a Claude Code session in +sandbox_session+'s checkout, yielding
|
|
195
204
|
# each stream-json event (a Hash) as it arrives. Returns the backend's
|
|
196
205
|
# outcome: { exit_status:, diff: }.
|
|
197
206
|
def run_code_session(sandbox_session, code_session, &on_event)
|
|
198
207
|
# Checked when a session is requested too; this covers one queued
|
|
199
208
|
# before the configuration changed.
|
|
200
|
-
|
|
209
|
+
runner = code_session.try(:runner) || "claude_code"
|
|
210
|
+
unless supports_code_runner?(runner)
|
|
211
|
+
raise UnsupportedBackendError, "The #{backend_name} sandbox backend cannot run #{runner} sessions"
|
|
212
|
+
end
|
|
213
|
+
refusal = ClaudeCodeAuth.backend_refusal(self) unless runner == "codex"
|
|
201
214
|
raise UnsupportedBackendError, refusal if refusal
|
|
202
215
|
|
|
203
216
|
@backend.public_send(adapter_method(:code_session), sandbox_session, code_session, &on_event)
|
|
@@ -223,7 +223,8 @@ module ActionAgent
|
|
|
223
223
|
input_tokens: agent_run.input_tokens,
|
|
224
224
|
output_tokens: agent_run.output_tokens,
|
|
225
225
|
error: agent_run.failed? ? agent_run.error_message.presence || "run failed" : nil,
|
|
226
|
-
cost: ModelPricing.estimate(model: spec.model,
|
|
226
|
+
cost: ModelPricing.estimate(model: spec.model, provider: spec.provider, input_tokens: agent_run.input_tokens,
|
|
227
|
+
output_tokens: agent_run.output_tokens),
|
|
227
228
|
metadata: { "agent_run_id" => agent_run.id }
|
|
228
229
|
)
|
|
229
230
|
end
|
data/config/routes.rb
CHANGED
|
@@ -142,7 +142,7 @@ ActionAgent::Engine.routes.draw do
|
|
|
142
142
|
|
|
143
143
|
# Agent output evaluations. A scenario suite also manages its scenarios
|
|
144
144
|
# here, and exposes each run's per-scenario, per-model results.
|
|
145
|
-
resources :evaluations, only: [ :index, :show, :create, :destroy ] do
|
|
145
|
+
resources :evaluations, only: [ :index, :show, :create, :update, :destroy ] do
|
|
146
146
|
member do
|
|
147
147
|
post :run
|
|
148
148
|
get "runs/:run_id", action: :show_run, as: :run_result
|
data/lib/action_agent/version.rb
CHANGED
data/lib/action_agent.rb
CHANGED
|
@@ -349,6 +349,10 @@ module ActionAgent
|
|
|
349
349
|
# @return [Integer]
|
|
350
350
|
attr_accessor :claude_code_timeout
|
|
351
351
|
|
|
352
|
+
# Codex CLI sessions use the owner's OpenAI API key and workspace-write
|
|
353
|
+
# sandboxing. They share checkout lifecycle and cancellation with Claude.
|
|
354
|
+
attr_accessor :codex_command, :codex_timeout
|
|
355
|
+
|
|
352
356
|
# How Claude Code sessions authenticate.
|
|
353
357
|
#
|
|
354
358
|
# :api_key (the default) runs them on the Anthropic API key the owner
|
|
@@ -772,6 +776,8 @@ module ActionAgent
|
|
|
772
776
|
@claude_code_permission_mode = "acceptEdits"
|
|
773
777
|
@claude_code_max_turns = nil
|
|
774
778
|
@claude_code_timeout = 1800
|
|
779
|
+
@codex_command = "codex"
|
|
780
|
+
@codex_timeout = 1800
|
|
775
781
|
@claude_code_auth = :api_key
|
|
776
782
|
@execution_enabled = true
|
|
777
783
|
@run_host_agent_classes = false
|
|
@@ -129,12 +129,15 @@ module ActionAgent
|
|
|
129
129
|
end
|
|
130
130
|
|
|
131
131
|
# Claude Code sessions inside those checkouts.
|
|
132
|
-
|
|
133
|
-
|
|
134
|
-
|
|
135
|
-
|
|
136
|
-
|
|
137
|
-
|
|
132
|
+
unless existing_migration?("create_active_agent_code_sessions")
|
|
133
|
+
migration_template(
|
|
134
|
+
"create_active_agent_code_sessions.rb.erb",
|
|
135
|
+
"db/migrate/create_active_agent_code_sessions.rb"
|
|
136
|
+
)
|
|
137
|
+
end
|
|
138
|
+
unless existing_migration?("add_code_session_runner")
|
|
139
|
+
migration_template("add_code_session_runner.rb.erb", "db/migrate/add_code_session_runner.rb")
|
|
140
|
+
end
|
|
138
141
|
end
|
|
139
142
|
|
|
140
143
|
def add_route
|
|
@@ -136,6 +136,9 @@ ActionAgent.configure do |config|
|
|
|
136
136
|
# config.claude_code_permission_mode = "acceptEdits"
|
|
137
137
|
# config.claude_code_max_turns = nil # Claude Code's own default
|
|
138
138
|
# config.claude_code_timeout = 1800 # seconds
|
|
139
|
+
# Codex uses a separate API-key connection in Settings -> Integrations.
|
|
140
|
+
# config.codex_command = "codex"
|
|
141
|
+
# config.codex_timeout = 1800 # seconds
|
|
139
142
|
#
|
|
140
143
|
# How sessions authenticate. :api_key (the default) uses the Anthropic API
|
|
141
144
|
# key connected in Settings -> Integrations (from the Claude Console,
|
|
@@ -0,0 +1,9 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
class AddCodeSessionRunner < ActiveRecord::Migration<%= migration_version %>
|
|
4
|
+
def change
|
|
5
|
+
table = "#{ActionAgent.table_name_prefix}code_sessions"
|
|
6
|
+
add_column table, :runner, :string, null: false, default: "claude_code" unless column_exists?(table, :runner)
|
|
7
|
+
add_column table, :runner_session_id, :string unless column_exists?(table, :runner_session_id)
|
|
8
|
+
end
|
|
9
|
+
end
|
metadata
CHANGED
|
@@ -1,14 +1,14 @@
|
|
|
1
1
|
--- !ruby/object:Gem::Specification
|
|
2
2
|
name: actionagent
|
|
3
3
|
version: !ruby/object:Gem::Version
|
|
4
|
-
version: 1.8.
|
|
4
|
+
version: 1.8.1
|
|
5
5
|
platform: ruby
|
|
6
6
|
authors:
|
|
7
7
|
- Justin Bowen
|
|
8
8
|
autorequire:
|
|
9
9
|
bindir: bin
|
|
10
10
|
cert_chain: []
|
|
11
|
-
date: 2026-
|
|
11
|
+
date: 2026-10-01 00:00:00.000000000 Z
|
|
12
12
|
dependencies:
|
|
13
13
|
- !ruby/object:Gem::Dependency
|
|
14
14
|
name: activeagent
|
|
@@ -210,10 +210,13 @@ files:
|
|
|
210
210
|
- app/services/action_agent/agent_tool_roster.rb
|
|
211
211
|
- app/services/action_agent/agent_toolbox.rb
|
|
212
212
|
- app/services/action_agent/claude_code_auth.rb
|
|
213
|
+
- app/services/action_agent/codex_session_events.rb
|
|
213
214
|
- app/services/action_agent/dashboard_assistant_service.rb
|
|
214
215
|
- app/services/action_agent/evaluation_evidence.rb
|
|
215
216
|
- app/services/action_agent/evaluation_report_import.rb
|
|
217
|
+
- app/services/action_agent/evaluation_run_cost.rb
|
|
216
218
|
- app/services/action_agent/evaluation_runner_service.rb
|
|
219
|
+
- app/services/action_agent/evaluation_standing.rb
|
|
217
220
|
- app/services/action_agent/evaluation_tool_resolver.rb
|
|
218
221
|
- app/services/action_agent/github_client.rb
|
|
219
222
|
- app/services/action_agent/local_sandbox_backend.rb
|
|
@@ -250,6 +253,7 @@ files:
|
|
|
250
253
|
- lib/generators/action_agent/templates/action_agent.rb.erb
|
|
251
254
|
- lib/generators/action_agent/templates/add_agent_id_to_active_agent_telemetry_traces.rb.erb
|
|
252
255
|
- lib/generators/action_agent/templates/add_agent_releases.rb.erb
|
|
256
|
+
- lib/generators/action_agent/templates/add_code_session_runner.rb.erb
|
|
253
257
|
- lib/generators/action_agent/templates/add_evaluation_report_identity.rb.erb
|
|
254
258
|
- lib/generators/action_agent/templates/add_provider_key_api_key.rb.erb
|
|
255
259
|
- lib/generators/action_agent/templates/create_active_agent_code_sessions.rb.erb
|