actionagent 1.8.0 → 1.8.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/app/assets/builds/action_agent.js +58 -58
- data/app/controllers/action_agent/api/code_sessions_controller.rb +19 -7
- data/app/controllers/action_agent/api/evaluations_controller.rb +49 -3
- data/app/controllers/action_agent/api/sandboxes_controller.rb +8 -0
- data/app/controllers/concerns/action_agent/api/mcp_dashboard_tools.rb +4 -2
- data/app/jobs/action_agent/code_session_job.rb +13 -8
- data/app/models/action_agent/agent.rb +33 -0
- data/app/models/action_agent/code_session.rb +6 -2
- data/app/models/action_agent/evaluation.rb +64 -0
- data/app/models/action_agent/evaluation_run.rb +143 -53
- data/app/models/action_agent/evaluation_scenario_result.rb +15 -3
- data/app/models/action_agent/model_pricing.rb +214 -34
- data/app/models/action_agent/provider_key.rb +7 -1
- data/app/models/action_agent/sandbox_session.rb +3 -2
- data/app/serializers/action_agent/evaluation_serializer.rb +41 -10
- data/app/services/action_agent/agent_scorecard.rb +50 -21
- data/app/services/action_agent/codex_session_events.rb +37 -0
- data/app/services/action_agent/evaluation_report_import.rb +84 -5
- data/app/services/action_agent/evaluation_run_cost.rb +609 -0
- data/app/services/action_agent/evaluation_runner_service.rb +25 -5
- data/app/services/action_agent/evaluation_standing.rb +145 -0
- data/app/services/action_agent/local_sandbox_backend.rb +47 -21
- data/app/services/action_agent/sandbox_orchestrator.rb +14 -1
- data/app/services/action_agent/scenario_evaluation_runner.rb +2 -1
- data/config/routes.rb +1 -1
- data/lib/action_agent/version.rb +1 -1
- data/lib/action_agent.rb +6 -0
- data/lib/generators/action_agent/install_generator.rb +9 -6
- data/lib/generators/action_agent/templates/action_agent.rb.erb +3 -0
- data/lib/generators/action_agent/templates/add_code_session_runner.rb.erb +9 -0
- metadata +6 -2
|
@@ -9,15 +9,29 @@ module ActionAgent
|
|
|
9
9
|
class EvaluationRun < ApplicationRecord
|
|
10
10
|
belongs_to :evaluation
|
|
11
11
|
# The version of the evaluated agent this run scored, so a pass rate is
|
|
12
|
-
# a statement about a release rather than about "the agent".
|
|
12
|
+
# a statement about a release rather than about "the agent". A run the
|
|
13
|
+
# engine executes scored the agent as it is now; an imported run scored
|
|
14
|
+
# the publishing application's code, which only its report can name
|
|
15
|
+
# (EvaluationReportImport sets the version from `report.release`), so
|
|
16
|
+
# a release-less import stays unrecorded rather than claiming the
|
|
17
|
+
# dashboard's latest version.
|
|
13
18
|
belongs_to :agent_version, optional: true
|
|
14
|
-
before_create { self.agent_version_id ||= evaluation&.agent&.latest_version&.id }
|
|
19
|
+
before_create { self.agent_version_id ||= evaluation&.agent&.latest_version&.id unless imported? }
|
|
20
|
+
# A run started on an archived evaluation brings it back: archiving says
|
|
21
|
+
# "no longer maintained", which a new run contradicts.
|
|
22
|
+
after_create { evaluation.unarchive! if evaluation&.archived? }
|
|
15
23
|
has_many :scenario_results, class_name: "EvaluationScenarioResult", dependent: :destroy
|
|
16
24
|
|
|
17
25
|
enum :status, { pending: 0, running: 1, complete: 2, failed: 3 }
|
|
18
26
|
|
|
19
27
|
scope :recent, -> { order(created_at: :desc) }
|
|
20
28
|
|
|
29
|
+
# Whether this run was published by an application that ran it itself
|
|
30
|
+
# (EvaluationReportImport) rather than executed here.
|
|
31
|
+
def imported?
|
|
32
|
+
external_run_id.present?
|
|
33
|
+
end
|
|
34
|
+
|
|
21
35
|
# Which scenarios and models a scenario run covered; empty for a
|
|
22
36
|
# generation-sampling run.
|
|
23
37
|
def selection
|
|
@@ -92,14 +106,37 @@ module ActionAgent
|
|
|
92
106
|
(values.sum.to_f / values.size).round(3)
|
|
93
107
|
end
|
|
94
108
|
|
|
95
|
-
# The judge's own spend on this run
|
|
96
|
-
# how many calls served each purpose
|
|
97
|
-
# for a run
|
|
98
|
-
|
|
109
|
+
# The judge's own spend on this run as the engine's meter recorded it —
|
|
110
|
+
# calls, tokens, estimated cost and how many calls served each purpose
|
|
111
|
+
# — or nil for a run the engine did not judge (a rules-only run, or an
|
|
112
|
+
# imported one, whose judge the application ran).
|
|
113
|
+
def judge_usage_meter
|
|
99
114
|
value = scores&.dig("_judge_usage")
|
|
100
115
|
value.is_a?(Hash) ? value : nil
|
|
101
116
|
end
|
|
102
117
|
|
|
118
|
+
# The judge's spend on this run from whatever recorded it: the meter,
|
|
119
|
+
# the publishing application's figures, or the judge traces priced
|
|
120
|
+
# (EvaluationRunCost). nil for a run no judge was asked about.
|
|
121
|
+
def judge_usage
|
|
122
|
+
cost_breakdown.judge_usage
|
|
123
|
+
end
|
|
124
|
+
|
|
125
|
+
# Every cost figure of this run — per result, per scenario, per model
|
|
126
|
+
# and in total — worked out once (EvaluationRunCost). A controller that
|
|
127
|
+
# lists runs preloads it (EvaluationRunCost.preload) and hands it in.
|
|
128
|
+
def cost_breakdown
|
|
129
|
+
@cost_breakdown ||= EvaluationRunCost.for(self)
|
|
130
|
+
end
|
|
131
|
+
|
|
132
|
+
attr_writer :cost_breakdown
|
|
133
|
+
|
|
134
|
+
# Forgets the breakdown, so a run whose results just changed is priced again.
|
|
135
|
+
def reload(*)
|
|
136
|
+
@cost_breakdown = nil
|
|
137
|
+
super
|
|
138
|
+
end
|
|
139
|
+
|
|
103
140
|
# Per-model summaries of a generation-sampling run's cohorts, keyed by
|
|
104
141
|
# model; empty for a scenario run or a run recorded before they were.
|
|
105
142
|
def cohorts
|
|
@@ -107,59 +144,51 @@ module ActionAgent
|
|
|
107
144
|
value.is_a?(Hash) ? value : {}
|
|
108
145
|
end
|
|
109
146
|
|
|
147
|
+
# Per-model summaries of a scenario run, keyed by label, as the runner
|
|
148
|
+
# recorded them under "_models", each with its cost as it stands now:
|
|
149
|
+
# the effective "cost" over its results (reported, else estimated), the
|
|
150
|
+
# "priced", "reported" and "estimated" counts behind it, and the
|
|
151
|
+
# judge's "judge_cost" and "judge_calls" on them (EvaluationRunCost). A
|
|
152
|
+
# recorded summary no result maps to is served as recorded. Empty for a
|
|
153
|
+
# generation-sampling run.
|
|
154
|
+
def model_summaries
|
|
155
|
+
summaries = scores&.dig("_models")
|
|
156
|
+
return {} unless summaries.is_a?(Hash)
|
|
157
|
+
|
|
158
|
+
costed = cost_breakdown.by_label(summaries.keys)
|
|
159
|
+
summaries.to_h do |label, stats|
|
|
160
|
+
next [ label, stats ] unless stats.is_a?(Hash) && costed.key?(label)
|
|
161
|
+
|
|
162
|
+
[ label, stats.merge(costed[label]) ]
|
|
163
|
+
end
|
|
164
|
+
end
|
|
165
|
+
|
|
110
166
|
# What the run spent, for display after it: the agent's side and the
|
|
111
167
|
# judge's, kept apart because they answer different questions.
|
|
112
168
|
#
|
|
113
169
|
# The agent's side is the operating figure — what the interactions cost
|
|
114
|
-
# to serve. For a scenario run that is
|
|
115
|
-
#
|
|
116
|
-
#
|
|
117
|
-
#
|
|
118
|
-
#
|
|
119
|
-
#
|
|
170
|
+
# to serve. For a scenario run that is its replays' cost, tokens and
|
|
171
|
+
# summed model time (`replays` of them); for a generation-sampling run
|
|
172
|
+
# it is the sampled generations' (`samples`), which were served before
|
|
173
|
+
# the run and cost it nothing.
|
|
174
|
+
#
|
|
175
|
+
# Every interaction with tokens is priced (EvaluationRunCost): `cost`
|
|
176
|
+
# sums the reported costs and, where none was reported, the estimates
|
|
177
|
+
# from tokens × model rates. `priced` and `unpriced` count the
|
|
178
|
+
# interactions either way, `reported` and `estimated` say how the priced
|
|
179
|
+
# ones were priced, and `cost_basis` sums that up as "reported",
|
|
180
|
+
# "estimated" or "mixed". `per_interaction` is the cost per priced
|
|
181
|
+
# interaction, the number a per-conversation budget is set against.
|
|
120
182
|
#
|
|
121
183
|
# `judge` is the evaluation's own overhead: the judge model's calls
|
|
122
184
|
# (scoring, recommending, the verdict, authoring KPIs), which run
|
|
123
|
-
# agent-to-agent and offline
|
|
185
|
+
# agent-to-agent and offline — from the engine's meter, the publishing
|
|
186
|
+
# application's figures or the judge traces, with `source` naming which
|
|
187
|
+
# and `run` the calls no result owns. `total` is the two sides together.
|
|
124
188
|
#
|
|
125
189
|
# Returns nil for a run that recorded nothing on either side.
|
|
126
190
|
def usage
|
|
127
|
-
|
|
128
|
-
Arel.sql("COUNT(*)"), Arel.sql("SUM(cost)"), Arel.sql("SUM(input_tokens)"),
|
|
129
|
-
Arel.sql("SUM(output_tokens)"), Arel.sql("SUM(duration_ms)")
|
|
130
|
-
)
|
|
131
|
-
replays = totals&.first.to_i
|
|
132
|
-
judge = judge_usage
|
|
133
|
-
runtime_ms = completed_at.present? ? ((completed_at - created_at) * 1000).round : nil
|
|
134
|
-
|
|
135
|
-
if replays.positive?
|
|
136
|
-
cost = totals[1]&.to_f
|
|
137
|
-
{
|
|
138
|
-
replays: replays,
|
|
139
|
-
cost: cost,
|
|
140
|
-
per_interaction: cost && (cost / replays).round(6),
|
|
141
|
-
input_tokens: totals[2].to_i,
|
|
142
|
-
output_tokens: totals[3].to_i,
|
|
143
|
-
model_time_ms: totals[4].to_i,
|
|
144
|
-
runtime_ms: runtime_ms,
|
|
145
|
-
judge: judge
|
|
146
|
-
}.compact
|
|
147
|
-
elsif cohorts.any?
|
|
148
|
-
samples = cohorts.values.sum { |cohort| cohort["samples"].to_i }
|
|
149
|
-
costs = cohorts.values.filter_map { |cohort| cohort["cost"] }
|
|
150
|
-
cost = costs.any? ? costs.sum.to_f.round(6) : nil
|
|
151
|
-
{
|
|
152
|
-
samples: samples,
|
|
153
|
-
cost: cost,
|
|
154
|
-
per_interaction: cost && samples.positive? ? (cost / samples).round(6) : nil,
|
|
155
|
-
input_tokens: cohorts.values.sum { |cohort| cohort["input_tokens"].to_i },
|
|
156
|
-
output_tokens: cohorts.values.sum { |cohort| cohort["output_tokens"].to_i },
|
|
157
|
-
runtime_ms: runtime_ms,
|
|
158
|
-
judge: judge
|
|
159
|
-
}.compact
|
|
160
|
-
elsif judge
|
|
161
|
-
{ runtime_ms: runtime_ms, judge: judge }.compact
|
|
162
|
-
end
|
|
191
|
+
cost_breakdown.usage
|
|
163
192
|
end
|
|
164
193
|
|
|
165
194
|
# Route templates for the report's fix item actions, relative to the
|
|
@@ -194,10 +223,21 @@ module ActionAgent
|
|
|
194
223
|
# the suite panel does. Raises ActiveRecord::RecordNotFound via the
|
|
195
224
|
# caller for a run of a generation-sampling evaluation, which has no
|
|
196
225
|
# scenario results to report on.
|
|
197
|
-
|
|
226
|
+
#
|
|
227
|
+
# With `estimate` (the default) each replay carries its effective cost
|
|
228
|
+
# and how it was priced (EvaluationRunCost), and the report is told the
|
|
229
|
+
# judge's run-level spend, so the page shows every cost the dashboard
|
|
230
|
+
# does. `estimate: false` rebuilds the report from what was recorded
|
|
231
|
+
# alone — the import summarizes a published run that way, so the
|
|
232
|
+
# summaries it stores stay the application's own figures.
|
|
233
|
+
#
|
|
234
|
+
# The engine runs against every activeagent since 1.4, so the report
|
|
235
|
+
# kwargs that arrived later are passed only when Report.new takes them.
|
|
236
|
+
def to_report(links: report_links, estimate: true)
|
|
198
237
|
rows = scenario_results.includes(:scenario).sort_by do |row|
|
|
199
238
|
[ row.evaluated_scenario["position"].to_i, row.evaluation_scenario_id, row.model ]
|
|
200
239
|
end
|
|
240
|
+
breakdown = cost_breakdown if estimate
|
|
201
241
|
selected = selected_specs
|
|
202
242
|
specs = {}
|
|
203
243
|
results = rows.map do |row|
|
|
@@ -210,14 +250,15 @@ module ActionAgent
|
|
|
210
250
|
replay: ActiveAgent::Evals::Replay.new(
|
|
211
251
|
answer: row.output, tool_calls: Array(row.tool_calls), duration_ms: row.duration_ms,
|
|
212
252
|
input_tokens: row.input_tokens, output_tokens: row.output_tokens,
|
|
213
|
-
cost: row.cost&.to_f, error: row.error_message,
|
|
253
|
+
cost: breakdown ? breakdown.result(row)["cost"] : row.cost&.to_f, error: row.error_message,
|
|
254
|
+
metadata: replay_metadata_for(row, breakdown)
|
|
214
255
|
),
|
|
215
256
|
scores: row.scores.to_h, score: row.score, status: row.status,
|
|
216
257
|
diagnosis: row.evaluation_diagnosis.presence
|
|
217
258
|
)
|
|
218
259
|
end
|
|
219
260
|
|
|
220
|
-
|
|
261
|
+
kwargs = {
|
|
221
262
|
results: results,
|
|
222
263
|
models: (selected.values & specs.values) + (specs.values - selected.values),
|
|
223
264
|
metadata: {
|
|
@@ -232,7 +273,40 @@ module ActionAgent
|
|
|
232
273
|
tool_resolver: EvaluationToolResolver.new(evaluation.agent),
|
|
233
274
|
agent_name: evaluation.agent&.name,
|
|
234
275
|
links: links
|
|
235
|
-
|
|
276
|
+
}
|
|
277
|
+
kwargs[:judge_usage] = breakdown.judge_usage_run if breakdown && self.class.report_accepts?(:judge_usage)
|
|
278
|
+
kwargs[:release] = release_summary if release_summary && self.class.report_accepts?(:release)
|
|
279
|
+
self.class.report_class.new(**kwargs)
|
|
280
|
+
end
|
|
281
|
+
|
|
282
|
+
# The framework's Report, as installed.
|
|
283
|
+
def self.report_class
|
|
284
|
+
ActiveAgent::Evals::Report
|
|
285
|
+
end
|
|
286
|
+
|
|
287
|
+
# Whether the installed framework's Report.new declares +keyword+.
|
|
288
|
+
def self.report_accepts?(keyword)
|
|
289
|
+
report_class.instance_method(:initialize).parameters.any? { |type, name| name == keyword && %i[key keyreq].include?(type) }
|
|
290
|
+
end
|
|
291
|
+
|
|
292
|
+
# The release this run scored, as the report names it — `{ "digest",
|
|
293
|
+
# "revision", "label" }` — or nil for a run pinned to a dashboard edit
|
|
294
|
+
# or to no version.
|
|
295
|
+
def release_summary
|
|
296
|
+
version = agent_version
|
|
297
|
+
return nil unless version&.release?
|
|
298
|
+
|
|
299
|
+
{ "digest" => version.release_digest, "revision" => version.revision, "label" => "v#{version.version_number}" }.compact
|
|
300
|
+
end
|
|
301
|
+
|
|
302
|
+
# The agent version this run scored, for the run's JSON: `{ id, number,
|
|
303
|
+
# release_digest, revision, release }`, or nil when none was recorded.
|
|
304
|
+
def agent_version_summary
|
|
305
|
+
version = agent_version
|
|
306
|
+
return nil unless version
|
|
307
|
+
|
|
308
|
+
{ id: version.id, number: version.version_number, release_digest: version.release_digest,
|
|
309
|
+
revision: version.revision, release: version.release? }
|
|
236
310
|
end
|
|
237
311
|
|
|
238
312
|
private
|
|
@@ -240,6 +314,22 @@ module ActionAgent
|
|
|
240
314
|
ModelSpec = ActiveAgent::Evals::ModelSpec
|
|
241
315
|
private_constant :ModelSpec
|
|
242
316
|
|
|
317
|
+
# The replay metadata the rebuilt report reads a result's costs from:
|
|
318
|
+
# what the result recorded, the judge usage the application reported
|
|
319
|
+
# for it, and — when estimating — how its cost was priced.
|
|
320
|
+
def replay_metadata_for(row, breakdown)
|
|
321
|
+
metadata = row.replay_metadata.dup
|
|
322
|
+
reported_judge = row.diagnosis.is_a?(Hash) ? row.diagnosis["_judge_usage"] : nil
|
|
323
|
+
metadata["judge_usage"] = reported_judge if reported_judge.is_a?(Hash)
|
|
324
|
+
return metadata unless breakdown
|
|
325
|
+
|
|
326
|
+
entry = breakdown.result(row)
|
|
327
|
+
metadata["cost_source"] = entry["cost_source"]
|
|
328
|
+
metadata["cost_rate"] = entry["cost_rate"] if entry["cost_rate"]
|
|
329
|
+
metadata["judge_usage"] = entry["judge_usage"] if entry["judge_usage"]
|
|
330
|
+
metadata
|
|
331
|
+
end
|
|
332
|
+
|
|
243
333
|
# How the report's header names the sandbox: its checkout and session.
|
|
244
334
|
def sandbox_label
|
|
245
335
|
return nil unless sandbox
|
|
@@ -44,8 +44,11 @@ module ActionAgent
|
|
|
44
44
|
value.is_a?(Hash) ? value : {}
|
|
45
45
|
end
|
|
46
46
|
|
|
47
|
+
# The diagnosis as the dashboard shows it: without the storage keys —
|
|
48
|
+
# the replay metadata, the scenario snapshot and the judge usage the
|
|
49
|
+
# publishing application reported (served as `judge_usage` instead).
|
|
47
50
|
def evaluation_diagnosis
|
|
48
|
-
(diagnosis || {}).except("_replay_metadata", "_scenario_snapshot")
|
|
51
|
+
(diagnosis || {}).except("_replay_metadata", "_scenario_snapshot", "_judge_usage")
|
|
49
52
|
end
|
|
50
53
|
|
|
51
54
|
# A catalog can be refreshed without changing what an earlier run asked
|
|
@@ -55,7 +58,12 @@ module ActionAgent
|
|
|
55
58
|
snapshot.is_a?(Hash) ? snapshot : scenario.as_json_summary.stringify_keys
|
|
56
59
|
end
|
|
57
60
|
|
|
58
|
-
|
|
61
|
+
# The result for the JSON API. Its cost is the effective one — reported,
|
|
62
|
+
# else estimated — with `reported_cost`, `cost_source`, `cost_rate` and
|
|
63
|
+
# the judge's `judge_usage` on it (EvaluationRunCost). A caller listing
|
|
64
|
+
# a run's results hands in the run's breakdown so nothing is priced per row.
|
|
65
|
+
def as_json_summary(costs: nil)
|
|
66
|
+
costs ||= evaluation_run.cost_breakdown.result(self)
|
|
59
67
|
{
|
|
60
68
|
id: id,
|
|
61
69
|
scenario_id: evaluation_scenario_id,
|
|
@@ -73,7 +81,11 @@ module ActionAgent
|
|
|
73
81
|
duration_ms: duration_ms,
|
|
74
82
|
input_tokens: input_tokens,
|
|
75
83
|
output_tokens: output_tokens,
|
|
76
|
-
cost: cost
|
|
84
|
+
cost: costs["cost"],
|
|
85
|
+
reported_cost: costs["reported_cost"],
|
|
86
|
+
cost_source: costs["cost_source"],
|
|
87
|
+
cost_rate: costs["cost_rate"],
|
|
88
|
+
judge_usage: costs["judge_usage"],
|
|
77
89
|
fault: fault,
|
|
78
90
|
recommendation: recommendation,
|
|
79
91
|
diagnosis: evaluation_diagnosis,
|
|
@@ -1,16 +1,40 @@
|
|
|
1
1
|
# frozen_string_literal: true
|
|
2
2
|
|
|
3
|
+
require "digest"
|
|
4
|
+
|
|
3
5
|
module ActionAgent
|
|
4
6
|
# Estimates LLM spend from token counts. The activeagent gem's telemetry
|
|
5
7
|
# records tokens only; the platform layers pricing on top for the cost
|
|
6
|
-
# figures shown in Traces and
|
|
8
|
+
# figures shown in Traces, Metrics and Evaluations.
|
|
7
9
|
#
|
|
8
10
|
# Rates come from RubyLLM's model registry (USD per million tokens,
|
|
9
|
-
# maintained upstream per model) when the model is known there
|
|
10
|
-
#
|
|
11
|
-
#
|
|
12
|
-
#
|
|
11
|
+
# maintained upstream per model) when the model is known there — looked
|
|
12
|
+
# up under the provider it ran on when the caller knows it, since
|
|
13
|
+
# `gpt-5.5` through OpenRouter and `gpt-5.5` at OpenAI are two entries.
|
|
14
|
+
# The static tables below are the fallback for aliases, self-hosted
|
|
15
|
+
# models and an install without RubyLLM: exact rows (STATIC_RATES) for
|
|
16
|
+
# the current frontier models, then name patterns (PRICES), then a
|
|
17
|
+
# conservative blended rate so totals stay meaningful. Costs are always
|
|
18
|
+
# presented as estimates, and every rate says where it came from
|
|
19
|
+
# (`source`), so a figure worked out at a fallback rate can say so.
|
|
13
20
|
class ModelPricing
|
|
21
|
+
# Exact rates for models the pattern table would otherwise misprice —
|
|
22
|
+
# claude-sonnet-5 is not a $3/$15 Sonnet, and the gpt-5 family is not
|
|
23
|
+
# one price — keyed by the name with its vendor prefix and date suffix
|
|
24
|
+
# removed and its dots and dashes normalised (see .normalize).
|
|
25
|
+
STATIC_RATES = {
|
|
26
|
+
"claude-sonnet-5" => [ 2.00, 10.00 ],
|
|
27
|
+
"gpt-5-5" => [ 5.00, 30.00 ],
|
|
28
|
+
"gpt-5-5-mini" => [ 0.25, 2.00 ],
|
|
29
|
+
"gpt-5-5-nano" => [ 0.05, 0.40 ],
|
|
30
|
+
"gpt-5-1" => [ 1.25, 10.00 ],
|
|
31
|
+
"gpt-5-1-mini" => [ 0.25, 2.00 ],
|
|
32
|
+
"gpt-5-1-nano" => [ 0.05, 0.40 ],
|
|
33
|
+
"gpt-5" => [ 1.25, 10.00 ],
|
|
34
|
+
"gpt-5-mini" => [ 0.25, 2.00 ],
|
|
35
|
+
"gpt-5-nano" => [ 0.05, 0.40 ]
|
|
36
|
+
}.freeze
|
|
37
|
+
|
|
14
38
|
PRICES = [
|
|
15
39
|
# [pattern, input $/1M, output $/1M]
|
|
16
40
|
[ /gpt-4o-mini/i, 0.15, 0.60 ],
|
|
@@ -18,10 +42,15 @@ module ActionAgent
|
|
|
18
42
|
[ /gpt-4\.1-nano/i, 0.10, 0.40 ],
|
|
19
43
|
[ /gpt-4\.1-mini/i, 0.40, 1.60 ],
|
|
20
44
|
[ /gpt-4\.1/i, 2.00, 8.00 ],
|
|
45
|
+
[ /gpt-5\.?5/i, 5.00, 30.00 ],
|
|
46
|
+
[ /gpt-5(\.\d+)?-nano/i, 0.05, 0.40 ],
|
|
47
|
+
[ /gpt-5(\.\d+)?-mini/i, 0.25, 2.00 ],
|
|
48
|
+
[ /gpt-5/i, 1.25, 10.00 ],
|
|
21
49
|
[ /o3-mini|o4-mini/i, 1.10, 4.40 ],
|
|
22
50
|
[ /claude.*(fable|mythos)/i, 10.00, 50.00 ],
|
|
23
51
|
[ /claude.*haiku-?4/i, 1.00, 5.00 ],
|
|
24
52
|
[ /claude.*(haiku)/i, 0.80, 4.00 ],
|
|
53
|
+
[ /claude.*sonnet-?5/i, 2.00, 10.00 ],
|
|
25
54
|
[ /claude.*(sonnet)/i, 3.00, 15.00 ],
|
|
26
55
|
[ /claude.*opus-(5|4-[5-9])/i, 5.00, 25.00 ],
|
|
27
56
|
[ /claude.*(opus)/i, 15.00, 75.00 ],
|
|
@@ -36,45 +65,196 @@ module ActionAgent
|
|
|
36
65
|
# Fallback blended rate for unknown models ($/1M input, $/1M output)
|
|
37
66
|
DEFAULT_RATE = [ 1.00, 4.00 ].freeze
|
|
38
67
|
|
|
39
|
-
#
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
|
|
68
|
+
# Where a rate came from: RubyLLM's bundled catalog, a registry the host
|
|
69
|
+
# keeps itself (RubyLLM's model table or a refreshed listing), a row of
|
|
70
|
+
# STATIC_RATES or PRICES, or DEFAULT_RATE. The last two are a guess at
|
|
71
|
+
# the model's price rather than its listing.
|
|
72
|
+
SOURCES = %w[catalog remote pattern default].freeze
|
|
73
|
+
FALLBACK_SOURCES = %w[pattern default].freeze
|
|
44
74
|
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
|
|
75
|
+
# A gateway's prefix on a model name, which names the provider rather
|
|
76
|
+
# than the model.
|
|
77
|
+
GATEWAY_PREFIXES = %w[openrouter requesty].freeze
|
|
78
|
+
# Vendor prefixes a gateway (or a host) puts before a model name, and
|
|
79
|
+
# the provider the bare name is listed under.
|
|
80
|
+
VENDOR_PROVIDERS = {
|
|
81
|
+
"openai" => "openai", "anthropic" => "anthropic", "google" => "gemini", "gemini" => "gemini",
|
|
82
|
+
"meta-llama" => nil, "mistralai" => "mistral", "mistral" => "mistral", "deepseek" => "deepseek",
|
|
83
|
+
"qwen" => nil, "x-ai" => "xai", "xai" => "xai", "cohere" => nil, "perplexity" => "perplexity"
|
|
84
|
+
}.freeze
|
|
48
85
|
|
|
49
|
-
|
|
50
|
-
return
|
|
86
|
+
class << self
|
|
87
|
+
# @return [Float, nil] estimated USD cost, nil when there is nothing to price
|
|
88
|
+
def estimate(model:, input_tokens:, output_tokens:, provider: nil)
|
|
89
|
+
estimate_detailed(model: model, input_tokens: input_tokens, output_tokens: output_tokens, provider: provider)&.fetch(:cost)
|
|
90
|
+
end
|
|
51
91
|
|
|
52
|
-
|
|
53
|
-
|
|
92
|
+
# The estimate with its working: `{ cost:, input_rate:, output_rate:,
|
|
93
|
+
# source: }`, rates in $ per million tokens. nil when both token
|
|
94
|
+
# counts are zero — a $0.00 is the caller's to decide on, since
|
|
95
|
+
# nothing generated and nothing recorded look alike here.
|
|
96
|
+
def estimate_detailed(model:, input_tokens:, output_tokens:, provider: nil)
|
|
97
|
+
input = input_tokens.to_i
|
|
98
|
+
output = output_tokens.to_i
|
|
99
|
+
return nil if input.zero? && output.zero?
|
|
100
|
+
|
|
101
|
+
rate = rate_detail(model, provider: provider)
|
|
102
|
+
{
|
|
103
|
+
cost: ((input * rate[:input]) + (output * rate[:output])) / 1_000_000.0,
|
|
104
|
+
input_rate: rate[:input],
|
|
105
|
+
output_rate: rate[:output],
|
|
106
|
+
source: rate[:source]
|
|
107
|
+
}
|
|
108
|
+
end
|
|
109
|
+
|
|
110
|
+
# @return [Array(Float, Float)] input and output rates in $ per million tokens
|
|
111
|
+
def rate_for(model, provider = nil)
|
|
112
|
+
rate = rate_detail(model, provider: provider)
|
|
113
|
+
[ rate[:input], rate[:output] ]
|
|
114
|
+
end
|
|
115
|
+
|
|
116
|
+
# `{ input:, output:, source: }` for a model under a provider, memoized
|
|
117
|
+
# per [provider, model]: the registry scan is not free and trace
|
|
118
|
+
# serialization asks per row.
|
|
119
|
+
def rate_detail(model, provider: nil)
|
|
120
|
+
return { input: DEFAULT_RATE[0], output: DEFAULT_RATE[1], source: "default" } if model.blank?
|
|
121
|
+
|
|
122
|
+
key = [ provider.to_s, model.to_s ]
|
|
123
|
+
@rates ||= {}
|
|
124
|
+
return @rates[key] if @rates.key?(key)
|
|
125
|
+
|
|
126
|
+
@rates[key] = registry_rate(model.to_s, provider.to_s.presence) || static_rate(model.to_s)
|
|
127
|
+
end
|
|
54
128
|
|
|
55
|
-
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
|
|
60
|
-
|
|
61
|
-
|
|
62
|
-
|
|
63
|
-
|
|
64
|
-
|
|
65
|
-
|
|
66
|
-
|
|
129
|
+
# Forgets memoized rates — after RubyLLM is loaded, or in tests.
|
|
130
|
+
def reset!
|
|
131
|
+
@rates = {}
|
|
132
|
+
end
|
|
133
|
+
|
|
134
|
+
# Names the rate tables in force, so a cache of figures worked out at
|
|
135
|
+
# them is invalidated when they change: the static rows, the default,
|
|
136
|
+
# and the version of the registry the install has (or none).
|
|
137
|
+
def fingerprint
|
|
138
|
+
registry = defined?(::RubyLLM) && ::RubyLLM.const_defined?(:VERSION) ? ::RubyLLM::VERSION.to_s : "none"
|
|
139
|
+
tables = [ STATIC_RATES, PRICES.map { |pattern, input, output| [ pattern.source, input, output ] }, DEFAULT_RATE, registry ]
|
|
140
|
+
Digest::SHA256.hexdigest(JSON.generate(tables))[0, 12]
|
|
141
|
+
end
|
|
142
|
+
|
|
143
|
+
# "openrouter/anthropic/claude-sonnet-4-5-20250929" → "claude-sonnet-4-5":
|
|
144
|
+
# the name without a gateway or vendor prefix, a date suffix or a
|
|
145
|
+
# colon-tagged size, with dots as dashes, so the same model under two
|
|
146
|
+
# spellings keys one row.
|
|
147
|
+
def normalize(model)
|
|
148
|
+
name = model.to_s.strip.downcase
|
|
149
|
+
name = name.split("/").last.to_s
|
|
150
|
+
name = name.sub(/-\d{8}\z/, "").sub(/-\d{4}-\d{2}-\d{2}\z/, "")
|
|
151
|
+
name.tr(".", "-")
|
|
152
|
+
end
|
|
153
|
+
|
|
154
|
+
private
|
|
155
|
+
|
|
156
|
+
# The registry's rate for the first spelling it knows: the name as
|
|
157
|
+
# given under the provider it ran on, then bare, then with a gateway
|
|
158
|
+
# or vendor prefix stripped (under the vendor's own provider), each
|
|
159
|
+
# with its dots and dashes swapped and its date suffix dropped.
|
|
160
|
+
def registry_rate(model, provider)
|
|
161
|
+
return nil unless registry_available?
|
|
162
|
+
|
|
163
|
+
lookup_candidates(model, provider).each do |id, candidate_provider|
|
|
164
|
+
info = registry_find(id, candidate_provider)
|
|
165
|
+
tokens = info&.pricing&.text_tokens
|
|
166
|
+
next unless tokens.respond_to?(:input) && tokens.input && tokens.output
|
|
167
|
+
|
|
168
|
+
return { input: tokens.input.to_f, output: tokens.output.to_f, source: registry_source }
|
|
67
169
|
end
|
|
170
|
+
nil
|
|
68
171
|
rescue StandardError
|
|
69
172
|
nil
|
|
70
173
|
end
|
|
71
|
-
end
|
|
72
174
|
|
|
73
|
-
|
|
74
|
-
|
|
75
|
-
|
|
175
|
+
def registry_available?
|
|
176
|
+
defined?(::RubyLLM) && ::RubyLLM.respond_to?(:models)
|
|
177
|
+
end
|
|
178
|
+
|
|
179
|
+
# A host that keeps its own model registry (RubyLLM's model table or
|
|
180
|
+
# a refreshed listing) prices from it rather than from the gem's
|
|
181
|
+
# bundled catalog.
|
|
182
|
+
def registry_source
|
|
183
|
+
config = ::RubyLLM.respond_to?(:config) ? ::RubyLLM.config : nil
|
|
184
|
+
return "remote" if config.respond_to?(:model_registry_class) && config.model_registry_class.present?
|
|
185
|
+
|
|
186
|
+
"catalog"
|
|
187
|
+
end
|
|
188
|
+
|
|
189
|
+
# RubyLLM 2 takes the provider as a keyword; 1.x took it positionally.
|
|
190
|
+
def registry_find(id, provider)
|
|
191
|
+
models = ::RubyLLM.models
|
|
192
|
+
return models.find(id) if provider.blank?
|
|
193
|
+
|
|
194
|
+
if registry_find_keyword?(models)
|
|
195
|
+
models.find(id, provider: provider)
|
|
196
|
+
else
|
|
197
|
+
models.find(id, provider)
|
|
198
|
+
end
|
|
199
|
+
rescue StandardError
|
|
200
|
+
nil
|
|
201
|
+
end
|
|
202
|
+
|
|
203
|
+
def registry_find_keyword?(models)
|
|
204
|
+
models.method(:find).parameters.any? { |type, name| name == :provider && %i[key keyreq].include?(type) }
|
|
205
|
+
end
|
|
206
|
+
|
|
207
|
+
# [[id, provider], ...] to try in order, without repeats.
|
|
208
|
+
def lookup_candidates(model, provider)
|
|
209
|
+
name = model.strip
|
|
210
|
+
provider = provider&.downcase
|
|
211
|
+
candidates = []
|
|
212
|
+
|
|
213
|
+
head, rest = name.split("/", 2)
|
|
214
|
+
if rest.present? && GATEWAY_PREFIXES.include?(head.downcase)
|
|
215
|
+
provider ||= head.downcase
|
|
216
|
+
name = rest
|
|
217
|
+
head, rest = name.split("/", 2)
|
|
218
|
+
end
|
|
219
|
+
|
|
220
|
+
names = [ name ]
|
|
221
|
+
if rest.present? && VENDOR_PROVIDERS.key?(head.downcase)
|
|
222
|
+
names << rest
|
|
223
|
+
vendor_provider = VENDOR_PROVIDERS[head.downcase]
|
|
224
|
+
end
|
|
225
|
+
|
|
226
|
+
names.each do |candidate|
|
|
227
|
+
spellings(candidate).each do |spelling|
|
|
228
|
+
candidates << [ spelling, provider ] if provider
|
|
229
|
+
candidates << [ spelling, vendor_provider ] if vendor_provider && vendor_provider != provider && candidate == rest
|
|
230
|
+
candidates << [ spelling, nil ]
|
|
231
|
+
end
|
|
232
|
+
end
|
|
233
|
+
candidates.uniq
|
|
234
|
+
end
|
|
235
|
+
|
|
236
|
+
# The name as given, then with its date suffix dropped, each with its
|
|
237
|
+
# dots and dashes swapped: "claude-sonnet-4.5" and "claude-sonnet-4-5"
|
|
238
|
+
# are one model.
|
|
239
|
+
def spellings(name)
|
|
240
|
+
undated = name.sub(/-\d{8}\z/, "").sub(/-\d{4}-\d{2}-\d{2}\z/, "")
|
|
241
|
+
[ name, undated ].uniq.flat_map do |spelling|
|
|
242
|
+
[ spelling, spelling.gsub(/(\d)\.(\d)/, '\1-\2'), spelling.gsub(/(\d)-(\d)/, '\1.\2') ]
|
|
243
|
+
end.uniq
|
|
244
|
+
end
|
|
245
|
+
|
|
246
|
+
# An exact row for the normalised name, else the first pattern the
|
|
247
|
+
# name as given matches, else the default.
|
|
248
|
+
def static_rate(model)
|
|
249
|
+
if (exact = STATIC_RATES[normalize(model)])
|
|
250
|
+
return { input: exact[0], output: exact[1], source: "pattern" }
|
|
251
|
+
end
|
|
252
|
+
|
|
253
|
+
PRICES.each do |pattern, input_rate, output_rate|
|
|
254
|
+
return { input: input_rate, output: output_rate, source: "pattern" } if model.match?(pattern)
|
|
255
|
+
end
|
|
256
|
+
{ input: DEFAULT_RATE[0], output: DEFAULT_RATE[1], source: "default" }
|
|
76
257
|
end
|
|
77
|
-
DEFAULT_RATE
|
|
78
258
|
end
|
|
79
259
|
end
|
|
80
260
|
end
|
|
@@ -26,7 +26,7 @@ module ActionAgent
|
|
|
26
26
|
HOST_PROVIDERS = %w[ollama].freeze
|
|
27
27
|
# Tools connected with a credential, configured beside the providers
|
|
28
28
|
# (Settings -> Integrations) but never offered to the agent builder.
|
|
29
|
-
CONNECTION_PROVIDERS = %w[claude_code].freeze
|
|
29
|
+
CONNECTION_PROVIDERS = %w[claude_code codex].freeze
|
|
30
30
|
PROVIDERS = (KEY_PROVIDERS + HOST_PROVIDERS + CONNECTION_PROVIDERS).freeze
|
|
31
31
|
|
|
32
32
|
# Only an Anthropic API key (sk-ant-api03-…, from the Claude Console or
|
|
@@ -38,6 +38,7 @@ module ActionAgent
|
|
|
38
38
|
# ActionAgent.claude_code_auth = :local_login instead, where the
|
|
39
39
|
# dashboard never touches the credential.
|
|
40
40
|
CLAUDE_CODE_CREDENTIAL = /\Ask-ant-api\d{2}-[A-Za-z0-9_-]+\z/
|
|
41
|
+
CODEX_CREDENTIAL = /\Ask-(?!ant-|or-)[A-Za-z0-9_-]+\z/
|
|
41
42
|
# A Claude subscription token, as earlier versions stored. Recognized so
|
|
42
43
|
# a stored one is never handed out (see #needs_replacing?).
|
|
43
44
|
SUBSCRIPTION_TOKEN_PREFIX = "sk-ant-oat"
|
|
@@ -66,6 +67,8 @@ module ActionAgent
|
|
|
66
67
|
"third-party apps to hold Claude.ai credentials. To use your own Claude login on this machine, set " \
|
|
67
68
|
"ActionAgent.claude_code_auth = :local_login with the :local sandbox backend instead"
|
|
68
69
|
}, if: -> { provider == "claude_code" }
|
|
70
|
+
validates :credential, format: { with: CODEX_CREDENTIAL, message: "must be an OpenAI API key (sk-…)" },
|
|
71
|
+
if: -> { provider == "codex" }
|
|
69
72
|
validates :api_key, length: { maximum: 500 }, allow_nil: true
|
|
70
73
|
|
|
71
74
|
# Deletes every Claude Code connection that still holds a Claude
|
|
@@ -139,6 +142,9 @@ module ActionAgent
|
|
|
139
142
|
#
|
|
140
143
|
# @return [Hash{String => String}]
|
|
141
144
|
def runtime_environment
|
|
145
|
+
if provider == "codex"
|
|
146
|
+
return CODEX_CREDENTIAL.match?(credential.to_s) ? { "CODEX_API_KEY" => credential } : {}
|
|
147
|
+
end
|
|
142
148
|
return {} unless provider == "claude_code"
|
|
143
149
|
return {} if needs_replacing? || !CLAUDE_CODE_CREDENTIAL.match?(credential.to_s)
|
|
144
150
|
|
|
@@ -158,10 +158,11 @@ module ActionAgent
|
|
|
158
158
|
# connected. Secret — for the backend, never a response.
|
|
159
159
|
#
|
|
160
160
|
# @return [Hash{String => String}]
|
|
161
|
-
def runtime_environment
|
|
161
|
+
def runtime_environment(runner: "claude_code")
|
|
162
162
|
return {} unless app_runtime?
|
|
163
|
+
return {} unless ProviderKey::CONNECTION_PROVIDERS.include?(runner)
|
|
163
164
|
|
|
164
|
-
owners_record(ProviderKey.where(provider:
|
|
165
|
+
owners_record(ProviderKey.where(provider: runner))&.runtime_environment || {}
|
|
165
166
|
end
|
|
166
167
|
|
|
167
168
|
# Check if session is still valid
|