actionagent 1.8.0 → 1.8.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (32) hide show
  1. checksums.yaml +4 -4
  2. data/app/assets/builds/action_agent.js +58 -58
  3. data/app/controllers/action_agent/api/code_sessions_controller.rb +19 -7
  4. data/app/controllers/action_agent/api/evaluations_controller.rb +49 -3
  5. data/app/controllers/action_agent/api/sandboxes_controller.rb +8 -0
  6. data/app/controllers/concerns/action_agent/api/mcp_dashboard_tools.rb +4 -2
  7. data/app/jobs/action_agent/code_session_job.rb +13 -8
  8. data/app/models/action_agent/agent.rb +33 -0
  9. data/app/models/action_agent/code_session.rb +6 -2
  10. data/app/models/action_agent/evaluation.rb +64 -0
  11. data/app/models/action_agent/evaluation_run.rb +143 -53
  12. data/app/models/action_agent/evaluation_scenario_result.rb +15 -3
  13. data/app/models/action_agent/model_pricing.rb +214 -34
  14. data/app/models/action_agent/provider_key.rb +7 -1
  15. data/app/models/action_agent/sandbox_session.rb +3 -2
  16. data/app/serializers/action_agent/evaluation_serializer.rb +41 -10
  17. data/app/services/action_agent/agent_scorecard.rb +50 -21
  18. data/app/services/action_agent/codex_session_events.rb +37 -0
  19. data/app/services/action_agent/evaluation_report_import.rb +84 -5
  20. data/app/services/action_agent/evaluation_run_cost.rb +609 -0
  21. data/app/services/action_agent/evaluation_runner_service.rb +25 -5
  22. data/app/services/action_agent/evaluation_standing.rb +145 -0
  23. data/app/services/action_agent/local_sandbox_backend.rb +47 -21
  24. data/app/services/action_agent/sandbox_orchestrator.rb +14 -1
  25. data/app/services/action_agent/scenario_evaluation_runner.rb +2 -1
  26. data/config/routes.rb +1 -1
  27. data/lib/action_agent/version.rb +1 -1
  28. data/lib/action_agent.rb +6 -0
  29. data/lib/generators/action_agent/install_generator.rb +9 -6
  30. data/lib/generators/action_agent/templates/action_agent.rb.erb +3 -0
  31. data/lib/generators/action_agent/templates/add_code_session_runner.rb.erb +9 -0
  32. metadata +6 -2
@@ -9,15 +9,29 @@ module ActionAgent
9
9
  class EvaluationRun < ApplicationRecord
10
10
  belongs_to :evaluation
11
11
  # The version of the evaluated agent this run scored, so a pass rate is
12
- # a statement about a release rather than about "the agent".
12
+ # a statement about a release rather than about "the agent". A run the
13
+ # engine executes scored the agent as it is now; an imported run scored
14
+ # the publishing application's code, which only its report can name
15
+ # (EvaluationReportImport sets the version from `report.release`), so
16
+ # a release-less import stays unrecorded rather than claiming the
17
+ # dashboard's latest version.
13
18
  belongs_to :agent_version, optional: true
14
- before_create { self.agent_version_id ||= evaluation&.agent&.latest_version&.id }
19
+ before_create { self.agent_version_id ||= evaluation&.agent&.latest_version&.id unless imported? }
20
+ # A run started on an archived evaluation brings it back: archiving says
21
+ # "no longer maintained", which a new run contradicts.
22
+ after_create { evaluation.unarchive! if evaluation&.archived? }
15
23
  has_many :scenario_results, class_name: "EvaluationScenarioResult", dependent: :destroy
16
24
 
17
25
  enum :status, { pending: 0, running: 1, complete: 2, failed: 3 }
18
26
 
19
27
  scope :recent, -> { order(created_at: :desc) }
20
28
 
29
+ # Whether this run was published by an application that ran it itself
30
+ # (EvaluationReportImport) rather than executed here.
31
+ def imported?
32
+ external_run_id.present?
33
+ end
34
+
21
35
  # Which scenarios and models a scenario run covered; empty for a
22
36
  # generation-sampling run.
23
37
  def selection
@@ -92,14 +106,37 @@ module ActionAgent
92
106
  (values.sum.to_f / values.size).round(3)
93
107
  end
94
108
 
95
- # The judge's own spend on this run — calls, tokens, estimated cost and
96
- # how many calls served each purpose — as the runner recorded it; nil
97
- # for a run that never asked a judge.
98
- def judge_usage
109
+ # The judge's own spend on this run as the engine's meter recorded it —
110
+ # calls, tokens, estimated cost and how many calls served each purpose
111
+ # — or nil for a run the engine did not judge (a rules-only run, or an
112
+ # imported one, whose judge the application ran).
113
+ def judge_usage_meter
99
114
  value = scores&.dig("_judge_usage")
100
115
  value.is_a?(Hash) ? value : nil
101
116
  end
102
117
 
118
+ # The judge's spend on this run from whatever recorded it: the meter,
119
+ # the publishing application's figures, or the judge traces priced
120
+ # (EvaluationRunCost). nil for a run no judge was asked about.
121
+ def judge_usage
122
+ cost_breakdown.judge_usage
123
+ end
124
+
125
+ # Every cost figure of this run — per result, per scenario, per model
126
+ # and in total — worked out once (EvaluationRunCost). A controller that
127
+ # lists runs preloads it (EvaluationRunCost.preload) and hands it in.
128
+ def cost_breakdown
129
+ @cost_breakdown ||= EvaluationRunCost.for(self)
130
+ end
131
+
132
+ attr_writer :cost_breakdown
133
+
134
+ # Forgets the breakdown, so a run whose results just changed is priced again.
135
+ def reload(*)
136
+ @cost_breakdown = nil
137
+ super
138
+ end
139
+
103
140
  # Per-model summaries of a generation-sampling run's cohorts, keyed by
104
141
  # model; empty for a scenario run or a run recorded before they were.
105
142
  def cohorts
@@ -107,59 +144,51 @@ module ActionAgent
107
144
  value.is_a?(Hash) ? value : {}
108
145
  end
109
146
 
147
+ # Per-model summaries of a scenario run, keyed by label, as the runner
148
+ # recorded them under "_models", each with its cost as it stands now:
149
+ # the effective "cost" over its results (reported, else estimated), the
150
+ # "priced", "reported" and "estimated" counts behind it, and the
151
+ # judge's "judge_cost" and "judge_calls" on them (EvaluationRunCost). A
152
+ # recorded summary no result maps to is served as recorded. Empty for a
153
+ # generation-sampling run.
154
+ def model_summaries
155
+ summaries = scores&.dig("_models")
156
+ return {} unless summaries.is_a?(Hash)
157
+
158
+ costed = cost_breakdown.by_label(summaries.keys)
159
+ summaries.to_h do |label, stats|
160
+ next [ label, stats ] unless stats.is_a?(Hash) && costed.key?(label)
161
+
162
+ [ label, stats.merge(costed[label]) ]
163
+ end
164
+ end
165
+
110
166
  # What the run spent, for display after it: the agent's side and the
111
167
  # judge's, kept apart because they answer different questions.
112
168
  #
113
169
  # The agent's side is the operating figure — what the interactions cost
114
- # to serve. For a scenario run that is the replays' estimated cost,
115
- # tokens and summed model time (`replays` of them); for a
116
- # generation-sampling run it is the sampled generations' (`samples`),
117
- # which were served before the run and cost it nothing. `per_interaction`
118
- # is that cost spread over the interactions, the number a per-conversation
119
- # budget is set against.
170
+ # to serve. For a scenario run that is its replays' cost, tokens and
171
+ # summed model time (`replays` of them); for a generation-sampling run
172
+ # it is the sampled generations' (`samples`), which were served before
173
+ # the run and cost it nothing.
174
+ #
175
+ # Every interaction with tokens is priced (EvaluationRunCost): `cost`
176
+ # sums the reported costs and, where none was reported, the estimates
177
+ # from tokens × model rates. `priced` and `unpriced` count the
178
+ # interactions either way, `reported` and `estimated` say how the priced
179
+ # ones were priced, and `cost_basis` sums that up as "reported",
180
+ # "estimated" or "mixed". `per_interaction` is the cost per priced
181
+ # interaction, the number a per-conversation budget is set against.
120
182
  #
121
183
  # `judge` is the evaluation's own overhead: the judge model's calls
122
184
  # (scoring, recommending, the verdict, authoring KPIs), which run
123
- # agent-to-agent and offline. It is present only when a judge was asked.
185
+ # agent-to-agent and offline — from the engine's meter, the publishing
186
+ # application's figures or the judge traces, with `source` naming which
187
+ # and `run` the calls no result owns. `total` is the two sides together.
124
188
  #
125
189
  # Returns nil for a run that recorded nothing on either side.
126
190
  def usage
127
- totals = scenario_results.pick(
128
- Arel.sql("COUNT(*)"), Arel.sql("SUM(cost)"), Arel.sql("SUM(input_tokens)"),
129
- Arel.sql("SUM(output_tokens)"), Arel.sql("SUM(duration_ms)")
130
- )
131
- replays = totals&.first.to_i
132
- judge = judge_usage
133
- runtime_ms = completed_at.present? ? ((completed_at - created_at) * 1000).round : nil
134
-
135
- if replays.positive?
136
- cost = totals[1]&.to_f
137
- {
138
- replays: replays,
139
- cost: cost,
140
- per_interaction: cost && (cost / replays).round(6),
141
- input_tokens: totals[2].to_i,
142
- output_tokens: totals[3].to_i,
143
- model_time_ms: totals[4].to_i,
144
- runtime_ms: runtime_ms,
145
- judge: judge
146
- }.compact
147
- elsif cohorts.any?
148
- samples = cohorts.values.sum { |cohort| cohort["samples"].to_i }
149
- costs = cohorts.values.filter_map { |cohort| cohort["cost"] }
150
- cost = costs.any? ? costs.sum.to_f.round(6) : nil
151
- {
152
- samples: samples,
153
- cost: cost,
154
- per_interaction: cost && samples.positive? ? (cost / samples).round(6) : nil,
155
- input_tokens: cohorts.values.sum { |cohort| cohort["input_tokens"].to_i },
156
- output_tokens: cohorts.values.sum { |cohort| cohort["output_tokens"].to_i },
157
- runtime_ms: runtime_ms,
158
- judge: judge
159
- }.compact
160
- elsif judge
161
- { runtime_ms: runtime_ms, judge: judge }.compact
162
- end
191
+ cost_breakdown.usage
163
192
  end
164
193
 
165
194
  # Route templates for the report's fix item actions, relative to the
@@ -194,10 +223,21 @@ module ActionAgent
194
223
  # the suite panel does. Raises ActiveRecord::RecordNotFound via the
195
224
  # caller for a run of a generation-sampling evaluation, which has no
196
225
  # scenario results to report on.
197
- def to_report(links: report_links)
226
+ #
227
+ # With `estimate` (the default) each replay carries its effective cost
228
+ # and how it was priced (EvaluationRunCost), and the report is told the
229
+ # judge's run-level spend, so the page shows every cost the dashboard
230
+ # does. `estimate: false` rebuilds the report from what was recorded
231
+ # alone — the import summarizes a published run that way, so the
232
+ # summaries it stores stay the application's own figures.
233
+ #
234
+ # The engine runs against every activeagent since 1.4, so the report
235
+ # kwargs that arrived later are passed only when Report.new takes them.
236
+ def to_report(links: report_links, estimate: true)
198
237
  rows = scenario_results.includes(:scenario).sort_by do |row|
199
238
  [ row.evaluated_scenario["position"].to_i, row.evaluation_scenario_id, row.model ]
200
239
  end
240
+ breakdown = cost_breakdown if estimate
201
241
  selected = selected_specs
202
242
  specs = {}
203
243
  results = rows.map do |row|
@@ -210,14 +250,15 @@ module ActionAgent
210
250
  replay: ActiveAgent::Evals::Replay.new(
211
251
  answer: row.output, tool_calls: Array(row.tool_calls), duration_ms: row.duration_ms,
212
252
  input_tokens: row.input_tokens, output_tokens: row.output_tokens,
213
- cost: row.cost&.to_f, error: row.error_message, metadata: row.replay_metadata
253
+ cost: breakdown ? breakdown.result(row)["cost"] : row.cost&.to_f, error: row.error_message,
254
+ metadata: replay_metadata_for(row, breakdown)
214
255
  ),
215
256
  scores: row.scores.to_h, score: row.score, status: row.status,
216
257
  diagnosis: row.evaluation_diagnosis.presence
217
258
  )
218
259
  end
219
260
 
220
- ActiveAgent::Evals::Report.new(
261
+ kwargs = {
221
262
  results: results,
222
263
  models: (selected.values & specs.values) + (specs.values - selected.values),
223
264
  metadata: {
@@ -232,7 +273,40 @@ module ActionAgent
232
273
  tool_resolver: EvaluationToolResolver.new(evaluation.agent),
233
274
  agent_name: evaluation.agent&.name,
234
275
  links: links
235
- )
276
+ }
277
+ kwargs[:judge_usage] = breakdown.judge_usage_run if breakdown && self.class.report_accepts?(:judge_usage)
278
+ kwargs[:release] = release_summary if release_summary && self.class.report_accepts?(:release)
279
+ self.class.report_class.new(**kwargs)
280
+ end
281
+
282
+ # The framework's Report, as installed.
283
+ def self.report_class
284
+ ActiveAgent::Evals::Report
285
+ end
286
+
287
+ # Whether the installed framework's Report.new declares +keyword+.
288
+ def self.report_accepts?(keyword)
289
+ report_class.instance_method(:initialize).parameters.any? { |type, name| name == keyword && %i[key keyreq].include?(type) }
290
+ end
291
+
292
+ # The release this run scored, as the report names it — `{ "digest",
293
+ # "revision", "label" }` — or nil for a run pinned to a dashboard edit
294
+ # or to no version.
295
+ def release_summary
296
+ version = agent_version
297
+ return nil unless version&.release?
298
+
299
+ { "digest" => version.release_digest, "revision" => version.revision, "label" => "v#{version.version_number}" }.compact
300
+ end
301
+
302
+ # The agent version this run scored, for the run's JSON: `{ id, number,
303
+ # release_digest, revision, release }`, or nil when none was recorded.
304
+ def agent_version_summary
305
+ version = agent_version
306
+ return nil unless version
307
+
308
+ { id: version.id, number: version.version_number, release_digest: version.release_digest,
309
+ revision: version.revision, release: version.release? }
236
310
  end
237
311
 
238
312
  private
@@ -240,6 +314,22 @@ module ActionAgent
240
314
  ModelSpec = ActiveAgent::Evals::ModelSpec
241
315
  private_constant :ModelSpec
242
316
 
317
+ # The replay metadata the rebuilt report reads a result's costs from:
318
+ # what the result recorded, the judge usage the application reported
319
+ # for it, and — when estimating — how its cost was priced.
320
+ def replay_metadata_for(row, breakdown)
321
+ metadata = row.replay_metadata.dup
322
+ reported_judge = row.diagnosis.is_a?(Hash) ? row.diagnosis["_judge_usage"] : nil
323
+ metadata["judge_usage"] = reported_judge if reported_judge.is_a?(Hash)
324
+ return metadata unless breakdown
325
+
326
+ entry = breakdown.result(row)
327
+ metadata["cost_source"] = entry["cost_source"]
328
+ metadata["cost_rate"] = entry["cost_rate"] if entry["cost_rate"]
329
+ metadata["judge_usage"] = entry["judge_usage"] if entry["judge_usage"]
330
+ metadata
331
+ end
332
+
243
333
  # How the report's header names the sandbox: its checkout and session.
244
334
  def sandbox_label
245
335
  return nil unless sandbox
@@ -44,8 +44,11 @@ module ActionAgent
44
44
  value.is_a?(Hash) ? value : {}
45
45
  end
46
46
 
47
+ # The diagnosis as the dashboard shows it: without the storage keys —
48
+ # the replay metadata, the scenario snapshot and the judge usage the
49
+ # publishing application reported (served as `judge_usage` instead).
47
50
  def evaluation_diagnosis
48
- (diagnosis || {}).except("_replay_metadata", "_scenario_snapshot")
51
+ (diagnosis || {}).except("_replay_metadata", "_scenario_snapshot", "_judge_usage")
49
52
  end
50
53
 
51
54
  # A catalog can be refreshed without changing what an earlier run asked
@@ -55,7 +58,12 @@ module ActionAgent
55
58
  snapshot.is_a?(Hash) ? snapshot : scenario.as_json_summary.stringify_keys
56
59
  end
57
60
 
58
- def as_json_summary
61
+ # The result for the JSON API. Its cost is the effective one — reported,
62
+ # else estimated — with `reported_cost`, `cost_source`, `cost_rate` and
63
+ # the judge's `judge_usage` on it (EvaluationRunCost). A caller listing
64
+ # a run's results hands in the run's breakdown so nothing is priced per row.
65
+ def as_json_summary(costs: nil)
66
+ costs ||= evaluation_run.cost_breakdown.result(self)
59
67
  {
60
68
  id: id,
61
69
  scenario_id: evaluation_scenario_id,
@@ -73,7 +81,11 @@ module ActionAgent
73
81
  duration_ms: duration_ms,
74
82
  input_tokens: input_tokens,
75
83
  output_tokens: output_tokens,
76
- cost: cost&.to_f,
84
+ cost: costs["cost"],
85
+ reported_cost: costs["reported_cost"],
86
+ cost_source: costs["cost_source"],
87
+ cost_rate: costs["cost_rate"],
88
+ judge_usage: costs["judge_usage"],
77
89
  fault: fault,
78
90
  recommendation: recommendation,
79
91
  diagnosis: evaluation_diagnosis,
@@ -1,16 +1,40 @@
1
1
  # frozen_string_literal: true
2
2
 
3
+ require "digest"
4
+
3
5
  module ActionAgent
4
6
  # Estimates LLM spend from token counts. The activeagent gem's telemetry
5
7
  # records tokens only; the platform layers pricing on top for the cost
6
- # figures shown in Traces and Metrics.
8
+ # figures shown in Traces, Metrics and Evaluations.
7
9
  #
8
10
  # Rates come from RubyLLM's model registry (USD per million tokens,
9
- # maintained upstream per model) when the model is known there; the static
10
- # pattern table below is the fallback for aliases/self-hosted models, and
11
- # a conservative blended rate covers everything else so totals stay
12
- # meaningful. Costs are always presented as estimates.
11
+ # maintained upstream per model) when the model is known there — looked
12
+ # up under the provider it ran on when the caller knows it, since
13
+ # `gpt-5.5` through OpenRouter and `gpt-5.5` at OpenAI are two entries.
14
+ # The static tables below are the fallback for aliases, self-hosted
15
+ # models and an install without RubyLLM: exact rows (STATIC_RATES) for
16
+ # the current frontier models, then name patterns (PRICES), then a
17
+ # conservative blended rate so totals stay meaningful. Costs are always
18
+ # presented as estimates, and every rate says where it came from
19
+ # (`source`), so a figure worked out at a fallback rate can say so.
13
20
  class ModelPricing
21
+ # Exact rates for models the pattern table would otherwise misprice —
22
+ # claude-sonnet-5 is not a $3/$15 Sonnet, and the gpt-5 family is not
23
+ # one price — keyed by the name with its vendor prefix and date suffix
24
+ # removed and its dots and dashes normalised (see .normalize).
25
+ STATIC_RATES = {
26
+ "claude-sonnet-5" => [ 2.00, 10.00 ],
27
+ "gpt-5-5" => [ 5.00, 30.00 ],
28
+ "gpt-5-5-mini" => [ 0.25, 2.00 ],
29
+ "gpt-5-5-nano" => [ 0.05, 0.40 ],
30
+ "gpt-5-1" => [ 1.25, 10.00 ],
31
+ "gpt-5-1-mini" => [ 0.25, 2.00 ],
32
+ "gpt-5-1-nano" => [ 0.05, 0.40 ],
33
+ "gpt-5" => [ 1.25, 10.00 ],
34
+ "gpt-5-mini" => [ 0.25, 2.00 ],
35
+ "gpt-5-nano" => [ 0.05, 0.40 ]
36
+ }.freeze
37
+
14
38
  PRICES = [
15
39
  # [pattern, input $/1M, output $/1M]
16
40
  [ /gpt-4o-mini/i, 0.15, 0.60 ],
@@ -18,10 +42,15 @@ module ActionAgent
18
42
  [ /gpt-4\.1-nano/i, 0.10, 0.40 ],
19
43
  [ /gpt-4\.1-mini/i, 0.40, 1.60 ],
20
44
  [ /gpt-4\.1/i, 2.00, 8.00 ],
45
+ [ /gpt-5\.?5/i, 5.00, 30.00 ],
46
+ [ /gpt-5(\.\d+)?-nano/i, 0.05, 0.40 ],
47
+ [ /gpt-5(\.\d+)?-mini/i, 0.25, 2.00 ],
48
+ [ /gpt-5/i, 1.25, 10.00 ],
21
49
  [ /o3-mini|o4-mini/i, 1.10, 4.40 ],
22
50
  [ /claude.*(fable|mythos)/i, 10.00, 50.00 ],
23
51
  [ /claude.*haiku-?4/i, 1.00, 5.00 ],
24
52
  [ /claude.*(haiku)/i, 0.80, 4.00 ],
53
+ [ /claude.*sonnet-?5/i, 2.00, 10.00 ],
25
54
  [ /claude.*(sonnet)/i, 3.00, 15.00 ],
26
55
  [ /claude.*opus-(5|4-[5-9])/i, 5.00, 25.00 ],
27
56
  [ /claude.*(opus)/i, 15.00, 75.00 ],
@@ -36,45 +65,196 @@ module ActionAgent
36
65
  # Fallback blended rate for unknown models ($/1M input, $/1M output)
37
66
  DEFAULT_RATE = [ 1.00, 4.00 ].freeze
38
67
 
39
- # @return [Float, nil] estimated USD cost, nil when there is nothing to price
40
- def self.estimate(model:, input_tokens:, output_tokens:)
41
- input = input_tokens.to_i
42
- output = output_tokens.to_i
43
- return nil if input.zero? && output.zero?
68
+ # Where a rate came from: RubyLLM's bundled catalog, a registry the host
69
+ # keeps itself (RubyLLM's model table or a refreshed listing), a row of
70
+ # STATIC_RATES or PRICES, or DEFAULT_RATE. The last two are a guess at
71
+ # the model's price rather than its listing.
72
+ SOURCES = %w[catalog remote pattern default].freeze
73
+ FALLBACK_SOURCES = %w[pattern default].freeze
44
74
 
45
- input_rate, output_rate = rate_for(model)
46
- ((input * input_rate) + (output * output_rate)) / 1_000_000.0
47
- end
75
+ # A gateway's prefix on a model name, which names the provider rather
76
+ # than the model.
77
+ GATEWAY_PREFIXES = %w[openrouter requesty].freeze
78
+ # Vendor prefixes a gateway (or a host) puts before a model name, and
79
+ # the provider the bare name is listed under.
80
+ VENDOR_PROVIDERS = {
81
+ "openai" => "openai", "anthropic" => "anthropic", "google" => "gemini", "gemini" => "gemini",
82
+ "meta-llama" => nil, "mistralai" => "mistral", "mistral" => "mistral", "deepseek" => "deepseek",
83
+ "qwen" => nil, "x-ai" => "xai", "xai" => "xai", "cohere" => nil, "perplexity" => "perplexity"
84
+ }.freeze
48
85
 
49
- def self.rate_for(model)
50
- return DEFAULT_RATE if model.blank?
86
+ class << self
87
+ # @return [Float, nil] estimated USD cost, nil when there is nothing to price
88
+ def estimate(model:, input_tokens:, output_tokens:, provider: nil)
89
+ estimate_detailed(model: model, input_tokens: input_tokens, output_tokens: output_tokens, provider: provider)&.fetch(:cost)
90
+ end
51
91
 
52
- registry_rate(model) || static_rate(model)
53
- end
92
+ # The estimate with its working: `{ cost:, input_rate:, output_rate:,
93
+ # source: }`, rates in $ per million tokens. nil when both token
94
+ # counts are zero — a $0.00 is the caller's to decide on, since
95
+ # nothing generated and nothing recorded look alike here.
96
+ def estimate_detailed(model:, input_tokens:, output_tokens:, provider: nil)
97
+ input = input_tokens.to_i
98
+ output = output_tokens.to_i
99
+ return nil if input.zero? && output.zero?
100
+
101
+ rate = rate_detail(model, provider: provider)
102
+ {
103
+ cost: ((input * rate[:input]) + (output * rate[:output])) / 1_000_000.0,
104
+ input_rate: rate[:input],
105
+ output_rate: rate[:output],
106
+ source: rate[:source]
107
+ }
108
+ end
109
+
110
+ # @return [Array(Float, Float)] input and output rates in $ per million tokens
111
+ def rate_for(model, provider = nil)
112
+ rate = rate_detail(model, provider: provider)
113
+ [ rate[:input], rate[:output] ]
114
+ end
115
+
116
+ # `{ input:, output:, source: }` for a model under a provider, memoized
117
+ # per [provider, model]: the registry scan is not free and trace
118
+ # serialization asks per row.
119
+ def rate_detail(model, provider: nil)
120
+ return { input: DEFAULT_RATE[0], output: DEFAULT_RATE[1], source: "default" } if model.blank?
121
+
122
+ key = [ provider.to_s, model.to_s ]
123
+ @rates ||= {}
124
+ return @rates[key] if @rates.key?(key)
125
+
126
+ @rates[key] = registry_rate(model.to_s, provider.to_s.presence) || static_rate(model.to_s)
127
+ end
54
128
 
55
- # Exact per-model rates from RubyLLM's registry. Lookups are memoized —
56
- # the registry scan is not free and trace serialization calls this per
57
- # row.
58
- def self.registry_rate(model)
59
- @registry_rates ||= {}
60
- return @registry_rates[model] if @registry_rates.key?(model)
61
-
62
- @registry_rates[model] = begin
63
- info = RubyLLM.models.find(model.to_s)
64
- tokens = info&.pricing&.text_tokens
65
- if tokens&.input && tokens&.output
66
- [ tokens.input, tokens.output ]
129
+ # Forgets memoized rates — after RubyLLM is loaded, or in tests.
130
+ def reset!
131
+ @rates = {}
132
+ end
133
+
134
+ # Names the rate tables in force, so a cache of figures worked out at
135
+ # them is invalidated when they change: the static rows, the default,
136
+ # and the version of the registry the install has (or none).
137
+ def fingerprint
138
+ registry = defined?(::RubyLLM) && ::RubyLLM.const_defined?(:VERSION) ? ::RubyLLM::VERSION.to_s : "none"
139
+ tables = [ STATIC_RATES, PRICES.map { |pattern, input, output| [ pattern.source, input, output ] }, DEFAULT_RATE, registry ]
140
+ Digest::SHA256.hexdigest(JSON.generate(tables))[0, 12]
141
+ end
142
+
143
+ # "openrouter/anthropic/claude-sonnet-4-5-20250929" → "claude-sonnet-4-5":
144
+ # the name without a gateway or vendor prefix, a date suffix or a
145
+ # colon-tagged size, with dots as dashes, so the same model under two
146
+ # spellings keys one row.
147
+ def normalize(model)
148
+ name = model.to_s.strip.downcase
149
+ name = name.split("/").last.to_s
150
+ name = name.sub(/-\d{8}\z/, "").sub(/-\d{4}-\d{2}-\d{2}\z/, "")
151
+ name.tr(".", "-")
152
+ end
153
+
154
+ private
155
+
156
+ # The registry's rate for the first spelling it knows: the name as
157
+ # given under the provider it ran on, then bare, then with a gateway
158
+ # or vendor prefix stripped (under the vendor's own provider), each
159
+ # with its dots and dashes swapped and its date suffix dropped.
160
+ def registry_rate(model, provider)
161
+ return nil unless registry_available?
162
+
163
+ lookup_candidates(model, provider).each do |id, candidate_provider|
164
+ info = registry_find(id, candidate_provider)
165
+ tokens = info&.pricing&.text_tokens
166
+ next unless tokens.respond_to?(:input) && tokens.input && tokens.output
167
+
168
+ return { input: tokens.input.to_f, output: tokens.output.to_f, source: registry_source }
67
169
  end
170
+ nil
68
171
  rescue StandardError
69
172
  nil
70
173
  end
71
- end
72
174
 
73
- def self.static_rate(model)
74
- PRICES.each do |pattern, input_rate, output_rate|
75
- return [ input_rate, output_rate ] if model.to_s.match?(pattern)
175
+ def registry_available?
176
+ defined?(::RubyLLM) && ::RubyLLM.respond_to?(:models)
177
+ end
178
+
179
+ # A host that keeps its own model registry (RubyLLM's model table or
180
+ # a refreshed listing) prices from it rather than from the gem's
181
+ # bundled catalog.
182
+ def registry_source
183
+ config = ::RubyLLM.respond_to?(:config) ? ::RubyLLM.config : nil
184
+ return "remote" if config.respond_to?(:model_registry_class) && config.model_registry_class.present?
185
+
186
+ "catalog"
187
+ end
188
+
189
+ # RubyLLM 2 takes the provider as a keyword; 1.x took it positionally.
190
+ def registry_find(id, provider)
191
+ models = ::RubyLLM.models
192
+ return models.find(id) if provider.blank?
193
+
194
+ if registry_find_keyword?(models)
195
+ models.find(id, provider: provider)
196
+ else
197
+ models.find(id, provider)
198
+ end
199
+ rescue StandardError
200
+ nil
201
+ end
202
+
203
+ def registry_find_keyword?(models)
204
+ models.method(:find).parameters.any? { |type, name| name == :provider && %i[key keyreq].include?(type) }
205
+ end
206
+
207
+ # [[id, provider], ...] to try in order, without repeats.
208
+ def lookup_candidates(model, provider)
209
+ name = model.strip
210
+ provider = provider&.downcase
211
+ candidates = []
212
+
213
+ head, rest = name.split("/", 2)
214
+ if rest.present? && GATEWAY_PREFIXES.include?(head.downcase)
215
+ provider ||= head.downcase
216
+ name = rest
217
+ head, rest = name.split("/", 2)
218
+ end
219
+
220
+ names = [ name ]
221
+ if rest.present? && VENDOR_PROVIDERS.key?(head.downcase)
222
+ names << rest
223
+ vendor_provider = VENDOR_PROVIDERS[head.downcase]
224
+ end
225
+
226
+ names.each do |candidate|
227
+ spellings(candidate).each do |spelling|
228
+ candidates << [ spelling, provider ] if provider
229
+ candidates << [ spelling, vendor_provider ] if vendor_provider && vendor_provider != provider && candidate == rest
230
+ candidates << [ spelling, nil ]
231
+ end
232
+ end
233
+ candidates.uniq
234
+ end
235
+
236
+ # The name as given, then with its date suffix dropped, each with its
237
+ # dots and dashes swapped: "claude-sonnet-4.5" and "claude-sonnet-4-5"
238
+ # are one model.
239
+ def spellings(name)
240
+ undated = name.sub(/-\d{8}\z/, "").sub(/-\d{4}-\d{2}-\d{2}\z/, "")
241
+ [ name, undated ].uniq.flat_map do |spelling|
242
+ [ spelling, spelling.gsub(/(\d)\.(\d)/, '\1-\2'), spelling.gsub(/(\d)-(\d)/, '\1.\2') ]
243
+ end.uniq
244
+ end
245
+
246
+ # An exact row for the normalised name, else the first pattern the
247
+ # name as given matches, else the default.
248
+ def static_rate(model)
249
+ if (exact = STATIC_RATES[normalize(model)])
250
+ return { input: exact[0], output: exact[1], source: "pattern" }
251
+ end
252
+
253
+ PRICES.each do |pattern, input_rate, output_rate|
254
+ return { input: input_rate, output: output_rate, source: "pattern" } if model.match?(pattern)
255
+ end
256
+ { input: DEFAULT_RATE[0], output: DEFAULT_RATE[1], source: "default" }
76
257
  end
77
- DEFAULT_RATE
78
258
  end
79
259
  end
80
260
  end
@@ -26,7 +26,7 @@ module ActionAgent
26
26
  HOST_PROVIDERS = %w[ollama].freeze
27
27
  # Tools connected with a credential, configured beside the providers
28
28
  # (Settings -> Integrations) but never offered to the agent builder.
29
- CONNECTION_PROVIDERS = %w[claude_code].freeze
29
+ CONNECTION_PROVIDERS = %w[claude_code codex].freeze
30
30
  PROVIDERS = (KEY_PROVIDERS + HOST_PROVIDERS + CONNECTION_PROVIDERS).freeze
31
31
 
32
32
  # Only an Anthropic API key (sk-ant-api03-…, from the Claude Console or
@@ -38,6 +38,7 @@ module ActionAgent
38
38
  # ActionAgent.claude_code_auth = :local_login instead, where the
39
39
  # dashboard never touches the credential.
40
40
  CLAUDE_CODE_CREDENTIAL = /\Ask-ant-api\d{2}-[A-Za-z0-9_-]+\z/
41
+ CODEX_CREDENTIAL = /\Ask-(?!ant-|or-)[A-Za-z0-9_-]+\z/
41
42
  # A Claude subscription token, as earlier versions stored. Recognized so
42
43
  # a stored one is never handed out (see #needs_replacing?).
43
44
  SUBSCRIPTION_TOKEN_PREFIX = "sk-ant-oat"
@@ -66,6 +67,8 @@ module ActionAgent
66
67
  "third-party apps to hold Claude.ai credentials. To use your own Claude login on this machine, set " \
67
68
  "ActionAgent.claude_code_auth = :local_login with the :local sandbox backend instead"
68
69
  }, if: -> { provider == "claude_code" }
70
+ validates :credential, format: { with: CODEX_CREDENTIAL, message: "must be an OpenAI API key (sk-…)" },
71
+ if: -> { provider == "codex" }
69
72
  validates :api_key, length: { maximum: 500 }, allow_nil: true
70
73
 
71
74
  # Deletes every Claude Code connection that still holds a Claude
@@ -139,6 +142,9 @@ module ActionAgent
139
142
  #
140
143
  # @return [Hash{String => String}]
141
144
  def runtime_environment
145
+ if provider == "codex"
146
+ return CODEX_CREDENTIAL.match?(credential.to_s) ? { "CODEX_API_KEY" => credential } : {}
147
+ end
142
148
  return {} unless provider == "claude_code"
143
149
  return {} if needs_replacing? || !CLAUDE_CODE_CREDENTIAL.match?(credential.to_s)
144
150
 
@@ -158,10 +158,11 @@ module ActionAgent
158
158
  # connected. Secret — for the backend, never a response.
159
159
  #
160
160
  # @return [Hash{String => String}]
161
- def runtime_environment
161
+ def runtime_environment(runner: "claude_code")
162
162
  return {} unless app_runtime?
163
+ return {} unless ProviderKey::CONNECTION_PROVIDERS.include?(runner)
163
164
 
164
- owners_record(ProviderKey.where(provider: "claude_code"))&.runtime_environment || {}
165
+ owners_record(ProviderKey.where(provider: runner))&.runtime_environment || {}
165
166
  end
166
167
 
167
168
  # Check if session is still valid