actionagent 1.8.0 → 1.8.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/app/assets/builds/action_agent.js +58 -58
- data/app/controllers/action_agent/api/code_sessions_controller.rb +19 -7
- data/app/controllers/action_agent/api/evaluations_controller.rb +49 -3
- data/app/controllers/action_agent/api/sandboxes_controller.rb +8 -0
- data/app/controllers/concerns/action_agent/api/mcp_dashboard_tools.rb +4 -2
- data/app/jobs/action_agent/code_session_job.rb +13 -8
- data/app/models/action_agent/agent.rb +33 -0
- data/app/models/action_agent/code_session.rb +6 -2
- data/app/models/action_agent/evaluation.rb +64 -0
- data/app/models/action_agent/evaluation_run.rb +143 -53
- data/app/models/action_agent/evaluation_scenario_result.rb +15 -3
- data/app/models/action_agent/model_pricing.rb +214 -34
- data/app/models/action_agent/provider_key.rb +7 -1
- data/app/models/action_agent/sandbox_session.rb +3 -2
- data/app/serializers/action_agent/evaluation_serializer.rb +41 -10
- data/app/services/action_agent/agent_scorecard.rb +50 -21
- data/app/services/action_agent/codex_session_events.rb +37 -0
- data/app/services/action_agent/evaluation_report_import.rb +84 -5
- data/app/services/action_agent/evaluation_run_cost.rb +609 -0
- data/app/services/action_agent/evaluation_runner_service.rb +25 -5
- data/app/services/action_agent/evaluation_standing.rb +145 -0
- data/app/services/action_agent/local_sandbox_backend.rb +47 -21
- data/app/services/action_agent/sandbox_orchestrator.rb +14 -1
- data/app/services/action_agent/scenario_evaluation_runner.rb +2 -1
- data/config/routes.rb +1 -1
- data/lib/action_agent/version.rb +1 -1
- data/lib/action_agent.rb +6 -0
- data/lib/generators/action_agent/install_generator.rb +9 -6
- data/lib/generators/action_agent/templates/action_agent.rb.erb +3 -0
- data/lib/generators/action_agent/templates/add_code_session_runner.rb.erb +9 -0
- metadata +6 -2
|
@@ -0,0 +1,609 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module ActionAgent
|
|
4
|
+
# What an evaluation run cost, worked out from everything the run and its
|
|
5
|
+
# results recorded, so that no cost shows blank where tokens exist.
|
|
6
|
+
#
|
|
7
|
+
# Each scenario result is priced down a chain, the first link that applies
|
|
8
|
+
# winning, and says which link it was (`cost_source`):
|
|
9
|
+
#
|
|
10
|
+
# reported — a cost the publishing application sent with the result
|
|
11
|
+
# estimated — the cost the engine stored when it ran the replay itself
|
|
12
|
+
# (tokens × the model's rate at the time), else the
|
|
13
|
+
# result's tokens × the model's rate now, else the tokens
|
|
14
|
+
# of the telemetry trace the result links to × that rate,
|
|
15
|
+
# else the prompt and answer at ~4 characters a token — a
|
|
16
|
+
# lower bound, for a result that recorded no tokens at all
|
|
17
|
+
# no_usage — both token counts recorded as zero: nothing was
|
|
18
|
+
# generated, so the result cost $0.00 (an errored replay)
|
|
19
|
+
# unpriced — no tokens, no trace and no text: nothing to price from
|
|
20
|
+
#
|
|
21
|
+
# An estimate carries its rate (`cost_rate`: $ per million tokens, where
|
|
22
|
+
# it came from, and the tokens it was applied to), so a figure can show
|
|
23
|
+
# its working. A trace's thinking tokens are already inside its output
|
|
24
|
+
# tokens and are never priced again.
|
|
25
|
+
#
|
|
26
|
+
# The judge's spend is kept apart from the agent's and found down its own
|
|
27
|
+
# chain: the engine's own meter (`scores._judge_usage`), else what the
|
|
28
|
+
# application reported per result (`diagnosis._judge_usage`) and for the
|
|
29
|
+
# run (`scores._judge_usage_run`), else the judge traces the results and
|
|
30
|
+
# the run link to, priced from their input and output tokens. Traces are
|
|
31
|
+
# read within the run's tenant: the account its report was published
|
|
32
|
+
# under, or the agent owner's account.
|
|
33
|
+
#
|
|
34
|
+
# Nothing here writes. A finished run's figures are cached in Rails.cache
|
|
35
|
+
# under the run, its updated_at and the pricing tables in force, so a
|
|
36
|
+
# page of runs prices its traces once; a run still landing results is
|
|
37
|
+
# worked out fresh each time. `preload` prices several runs with one
|
|
38
|
+
# query per table.
|
|
39
|
+
class EvaluationRunCost
|
|
40
|
+
CACHE_VERSION = 2
|
|
41
|
+
CACHE_TTL = 7.days
|
|
42
|
+
# The approximation the context meter applies to text it sizes itself.
|
|
43
|
+
CHARS_PER_TOKEN = 4
|
|
44
|
+
SOURCES = %w[reported estimated no_usage unpriced].freeze
|
|
45
|
+
# Judge calls no result owns: the verdict across cohorts and, for a
|
|
46
|
+
# judge_defined evaluation, authoring the criteria.
|
|
47
|
+
RUN_LEVEL_KINDS = %w[verdict define].freeze
|
|
48
|
+
|
|
49
|
+
attr_reader :run
|
|
50
|
+
|
|
51
|
+
# The breakdown for one run, from the cache when it is finished and
|
|
52
|
+
# cached, else computed.
|
|
53
|
+
def self.for(run)
|
|
54
|
+
new(run).tap(&:data)
|
|
55
|
+
end
|
|
56
|
+
|
|
57
|
+
# { run.id => EvaluationRunCost } for every run in +runs+, loading the
|
|
58
|
+
# uncached runs' results and traces in one query each.
|
|
59
|
+
def self.preload(runs)
|
|
60
|
+
runs = Array(runs).compact.uniq(&:id)
|
|
61
|
+
return {} if runs.empty?
|
|
62
|
+
|
|
63
|
+
breakdowns = runs.index_with { |run| new(run) }
|
|
64
|
+
pending = breakdowns.values.reject(&:cached?)
|
|
65
|
+
rows = EvaluationScenarioResult.where(evaluation_run_id: pending.map { |breakdown| breakdown.run.id })
|
|
66
|
+
.includes(:scenario).group_by(&:evaluation_run_id)
|
|
67
|
+
pending.each { |breakdown| breakdown.rows = rows.fetch(breakdown.run.id, []) }
|
|
68
|
+
TraceReader.preload(pending)
|
|
69
|
+
pending.each(&:data)
|
|
70
|
+
breakdowns.transform_keys(&:id)
|
|
71
|
+
end
|
|
72
|
+
|
|
73
|
+
def initialize(run)
|
|
74
|
+
@run = run
|
|
75
|
+
end
|
|
76
|
+
|
|
77
|
+
# The results' rows, loaded once; `preload` hands them in.
|
|
78
|
+
attr_writer :rows
|
|
79
|
+
|
|
80
|
+
def rows
|
|
81
|
+
@rows ||= run.scenario_results.includes(:scenario).to_a
|
|
82
|
+
end
|
|
83
|
+
|
|
84
|
+
def cached?
|
|
85
|
+
return false unless cacheable?
|
|
86
|
+
|
|
87
|
+
@data ||= Rails.cache.read(cache_key)
|
|
88
|
+
@data.present?
|
|
89
|
+
end
|
|
90
|
+
|
|
91
|
+
# Every figure, as one serializable hash:
|
|
92
|
+
# { "results" => { id => entry }, "judge" => usage | nil, "usage" => ..., "costs" => ..., "by_label" => ... }
|
|
93
|
+
def data
|
|
94
|
+
return @data if @data
|
|
95
|
+
|
|
96
|
+
@data = cacheable? ? Rails.cache.fetch(cache_key, expires_in: CACHE_TTL) { compute } : compute
|
|
97
|
+
end
|
|
98
|
+
|
|
99
|
+
# The result's figures: `{ "cost", "reported_cost", "cost_source",
|
|
100
|
+
# "cost_rate", "judge_usage" }`, with `cost` the effective one. A
|
|
101
|
+
# result this run does not hold reads as unpriced.
|
|
102
|
+
def result(result)
|
|
103
|
+
data["results"][result.id.to_s] || unpriced_entry
|
|
104
|
+
end
|
|
105
|
+
|
|
106
|
+
# The run's usage, the agent's side and the judge's apart, or nil when
|
|
107
|
+
# the run recorded nothing on either side (see EvaluationRun#usage).
|
|
108
|
+
def usage
|
|
109
|
+
data["usage"]&.transform_keys(&:to_sym)
|
|
110
|
+
end
|
|
111
|
+
|
|
112
|
+
# The judge's spend on the run, or nil when no judge was asked.
|
|
113
|
+
def judge_usage
|
|
114
|
+
data["judge"]
|
|
115
|
+
end
|
|
116
|
+
|
|
117
|
+
# The judge's calls no result owns, in the shape Report.new(judge_usage:)
|
|
118
|
+
# takes. The engine's meter knows no result's share, so a metered run
|
|
119
|
+
# hands the report the whole meter: from the report's side, none of it
|
|
120
|
+
# is any result's.
|
|
121
|
+
def judge_usage_run
|
|
122
|
+
judge = judge_usage
|
|
123
|
+
return nil unless judge
|
|
124
|
+
return judge.except("run", "estimated") if judge["source"] == "meter"
|
|
125
|
+
return nil unless judge["run"]
|
|
126
|
+
|
|
127
|
+
judge["run"].merge("model" => judge["model"], "source" => judge["source"]).compact
|
|
128
|
+
end
|
|
129
|
+
|
|
130
|
+
# Costs per scenario and for the run (see Api::EvaluationsController#show_run).
|
|
131
|
+
def costs
|
|
132
|
+
data["costs"]
|
|
133
|
+
end
|
|
134
|
+
|
|
135
|
+
# Per model label: the effective cost and the counts behind it, to merge
|
|
136
|
+
# into the run's recorded `_models` summaries.
|
|
137
|
+
def by_label(labels)
|
|
138
|
+
data["by_label"].select { |label, _| labels.include?(label) }
|
|
139
|
+
end
|
|
140
|
+
|
|
141
|
+
# The trace ids this run refers to: each result's own replay trace and
|
|
142
|
+
# judge traces, and the run's judge traces (the verdict's). TraceReader
|
|
143
|
+
# reads them and hands the traces back through `traces=`.
|
|
144
|
+
def trace_ids
|
|
145
|
+
ids = rows.flat_map { |row| [ row.replay_metadata["trace_id"], *Array(row.replay_metadata["judge_trace_ids"]) ] }
|
|
146
|
+
ids.concat(Array(run.report_metadata["judge_trace_ids"]))
|
|
147
|
+
ids.filter_map { |id| id.to_s.presence }.uniq
|
|
148
|
+
end
|
|
149
|
+
|
|
150
|
+
attr_writer :traces
|
|
151
|
+
|
|
152
|
+
private
|
|
153
|
+
|
|
154
|
+
def cacheable?
|
|
155
|
+
run.persisted? && run.complete? && run.updated_at.present?
|
|
156
|
+
end
|
|
157
|
+
|
|
158
|
+
def cache_key
|
|
159
|
+
[ "action_agent", "evaluation_run_cost", CACHE_VERSION, run.id, run.updated_at.utc.strftime("%Y%m%d%H%M%S%6N"), ModelPricing.fingerprint ]
|
|
160
|
+
end
|
|
161
|
+
|
|
162
|
+
def imported?
|
|
163
|
+
run.external_run_id.present?
|
|
164
|
+
end
|
|
165
|
+
|
|
166
|
+
def unpriced_entry
|
|
167
|
+
{ "cost" => nil, "reported_cost" => nil, "cost_source" => "unpriced", "cost_rate" => nil, "judge_usage" => nil }
|
|
168
|
+
end
|
|
169
|
+
|
|
170
|
+
# --- computation ---------------------------------------------------------
|
|
171
|
+
|
|
172
|
+
def compute
|
|
173
|
+
traces = @traces || TraceReader.new(self).read
|
|
174
|
+
entries = rows.to_h { |row| [ row.id.to_s, result_entry(row, traces).merge("judge_usage" => result_judge(row, traces)) ] }
|
|
175
|
+
judge = run_judge_usage(entries, traces)
|
|
176
|
+
{
|
|
177
|
+
"results" => entries,
|
|
178
|
+
"judge" => judge,
|
|
179
|
+
"usage" => usage_for(entries, judge),
|
|
180
|
+
"costs" => costs_for(entries, judge),
|
|
181
|
+
"by_label" => by_label_for(entries)
|
|
182
|
+
}
|
|
183
|
+
end
|
|
184
|
+
|
|
185
|
+
# The agent's cost of one result, down the chain in the class comment.
|
|
186
|
+
def result_entry(row, traces)
|
|
187
|
+
if !row.cost.nil? && imported?
|
|
188
|
+
return { "cost" => row.cost.to_f, "reported_cost" => row.cost.to_f, "cost_source" => "reported", "cost_rate" => nil }
|
|
189
|
+
end
|
|
190
|
+
|
|
191
|
+
if !row.cost.nil?
|
|
192
|
+
rate = ModelPricing.rate_detail(row.model, provider: row.provider)
|
|
193
|
+
return estimated(row.cost.to_f, rate, "tokens", row.input_tokens, row.output_tokens)
|
|
194
|
+
end
|
|
195
|
+
|
|
196
|
+
input = row.input_tokens
|
|
197
|
+
output = row.output_tokens
|
|
198
|
+
return { "cost" => 0.0, "reported_cost" => nil, "cost_source" => "no_usage", "cost_rate" => nil } if input == 0 && output == 0
|
|
199
|
+
|
|
200
|
+
if input.to_i.positive? || output.to_i.positive?
|
|
201
|
+
return priced("tokens", row.model, row.provider, input, output)
|
|
202
|
+
end
|
|
203
|
+
|
|
204
|
+
trace = traces[row.replay_metadata["trace_id"].to_s]
|
|
205
|
+
if trace
|
|
206
|
+
return { "cost" => 0.0, "reported_cost" => nil, "cost_source" => "no_usage", "cost_rate" => nil } if trace.input_tokens.zero? && trace.output_tokens.zero?
|
|
207
|
+
|
|
208
|
+
return priced("trace", trace.model.presence || row.model, trace.provider.presence || row.provider, trace.input_tokens, trace.output_tokens)
|
|
209
|
+
end
|
|
210
|
+
|
|
211
|
+
prompt_chars = row.evaluated_scenario["prompt"].to_s.length
|
|
212
|
+
answer_chars = row.output.to_s.length
|
|
213
|
+
if prompt_chars.positive? || answer_chars.positive?
|
|
214
|
+
return priced("chars", row.model, row.provider, prompt_chars / CHARS_PER_TOKEN, answer_chars / CHARS_PER_TOKEN)
|
|
215
|
+
end
|
|
216
|
+
|
|
217
|
+
unpriced_entry.except("judge_usage")
|
|
218
|
+
end
|
|
219
|
+
|
|
220
|
+
def priced(basis, model, provider, input_tokens, output_tokens)
|
|
221
|
+
detail = ModelPricing.estimate_detailed(model: model, provider: provider, input_tokens: input_tokens, output_tokens: output_tokens)
|
|
222
|
+
return { "cost" => 0.0, "reported_cost" => nil, "cost_source" => "no_usage", "cost_rate" => nil } unless detail
|
|
223
|
+
|
|
224
|
+
rate = { input: detail[:input_rate], output: detail[:output_rate], source: detail[:source] }
|
|
225
|
+
estimated(detail[:cost], rate, basis, input_tokens, output_tokens)
|
|
226
|
+
end
|
|
227
|
+
|
|
228
|
+
def estimated(cost, rate, basis, input_tokens, output_tokens)
|
|
229
|
+
{
|
|
230
|
+
"cost" => cost.to_f.round(6),
|
|
231
|
+
"reported_cost" => nil,
|
|
232
|
+
"cost_source" => "estimated",
|
|
233
|
+
"cost_rate" => {
|
|
234
|
+
"input" => rate[:input], "output" => rate[:output], "source" => rate[:source],
|
|
235
|
+
"basis" => basis, "input_tokens" => input_tokens.to_i, "output_tokens" => output_tokens.to_i
|
|
236
|
+
}
|
|
237
|
+
}
|
|
238
|
+
end
|
|
239
|
+
|
|
240
|
+
# --- judge -----------------------------------------------------------------
|
|
241
|
+
|
|
242
|
+
# What the judge spent on one result: the application's own figures when
|
|
243
|
+
# it reported them, else the result's judge traces priced; nil when the
|
|
244
|
+
# engine metered the run (the meter is per run, not per result) or
|
|
245
|
+
# nothing links a judge call to the result.
|
|
246
|
+
def result_judge(row, traces)
|
|
247
|
+
return nil if run.judge_usage_meter
|
|
248
|
+
|
|
249
|
+
reported = row.diagnosis.is_a?(Hash) ? row.diagnosis["_judge_usage"] : nil
|
|
250
|
+
return normalize_reported_judge(reported) if reported.is_a?(Hash)
|
|
251
|
+
|
|
252
|
+
judge_traces = Array(row.replay_metadata["judge_trace_ids"]).filter_map { |id| traces[id.to_s] }
|
|
253
|
+
traces_usage(judge_traces)
|
|
254
|
+
end
|
|
255
|
+
|
|
256
|
+
# The run's judge usage: `{ "calls", "input_tokens", "output_tokens",
|
|
257
|
+
# "cost", "model", "by_kind", "source", "estimated", "run" => { "calls",
|
|
258
|
+
# "cost", "by_kind" } }`, the engine meter first, then what the
|
|
259
|
+
# application reported, then the judge traces.
|
|
260
|
+
def run_judge_usage(entries, traces)
|
|
261
|
+
if (meter = run.judge_usage_meter)
|
|
262
|
+
return metered_judge(meter)
|
|
263
|
+
end
|
|
264
|
+
|
|
265
|
+
parts = entries.values.filter_map { |entry| entry["judge_usage"] }
|
|
266
|
+
reported_run = run.scores.is_a?(Hash) ? run.scores["_judge_usage_run"] : nil
|
|
267
|
+
run_part = normalize_reported_judge(reported_run) if reported_run.is_a?(Hash)
|
|
268
|
+
if parts.any? { |part| part["source"] == "reported" } || run_part
|
|
269
|
+
return sum_judge(parts + [ run_part ].compact, run_part, "reported")
|
|
270
|
+
end
|
|
271
|
+
|
|
272
|
+
result_ids = rows.flat_map { |row| Array(row.replay_metadata["judge_trace_ids"]) }.map(&:to_s)
|
|
273
|
+
run_traces = Array(run.report_metadata["judge_trace_ids"]).map(&:to_s).uniq.reject { |id| result_ids.include?(id) }.filter_map { |id| traces[id] }
|
|
274
|
+
run_part = traces_usage(run_traces)
|
|
275
|
+
return nil if parts.empty? && run_part.nil?
|
|
276
|
+
|
|
277
|
+
sum_judge(parts + [ run_part ].compact, run_part, "traces")
|
|
278
|
+
end
|
|
279
|
+
|
|
280
|
+
# The engine's meter knows the run's total and what each call was for,
|
|
281
|
+
# not which result each call served: the run-level part is the calls
|
|
282
|
+
# of a run-level kind, their cost unknown.
|
|
283
|
+
def metered_judge(meter)
|
|
284
|
+
by_kind = (meter["by_kind"] || {}).to_h.transform_keys(&:to_s)
|
|
285
|
+
run_kinds = by_kind.select { |kind, _| RUN_LEVEL_KINDS.include?(kind) }
|
|
286
|
+
{
|
|
287
|
+
"calls" => meter["calls"].to_i,
|
|
288
|
+
"input_tokens" => meter["input_tokens"].to_i,
|
|
289
|
+
"output_tokens" => meter["output_tokens"].to_i,
|
|
290
|
+
"cost" => meter["cost"]&.to_f&.round(6),
|
|
291
|
+
"model" => meter["model"].presence,
|
|
292
|
+
"by_kind" => by_kind,
|
|
293
|
+
"source" => "meter",
|
|
294
|
+
"estimated" => true,
|
|
295
|
+
"run" => { "calls" => run_kinds.values.sum(&:to_i), "cost" => nil, "by_kind" => run_kinds }
|
|
296
|
+
}
|
|
297
|
+
end
|
|
298
|
+
|
|
299
|
+
# A usage the application reported — per result, or for the run —
|
|
300
|
+
# bounded to the keys the dashboard reads. A usage with tokens but no
|
|
301
|
+
# cost is priced at its model's rate and says so.
|
|
302
|
+
def normalize_reported_judge(usage)
|
|
303
|
+
usage = usage.to_h.transform_keys(&:to_s)
|
|
304
|
+
input = usage["input_tokens"].to_i
|
|
305
|
+
output = usage["output_tokens"].to_i
|
|
306
|
+
model = usage["model"].presence
|
|
307
|
+
cost = usage["cost"].is_a?(Numeric) ? usage["cost"].to_f : nil
|
|
308
|
+
estimated = false
|
|
309
|
+
if cost.nil? && (input.positive? || output.positive?)
|
|
310
|
+
cost = ModelPricing.estimate(model: model, input_tokens: input, output_tokens: output)
|
|
311
|
+
estimated = true
|
|
312
|
+
end
|
|
313
|
+
by_kind = usage["by_kind"].is_a?(Hash) ? usage["by_kind"].to_h { |kind, count| [ kind.to_s, count.to_i ] } : {}
|
|
314
|
+
{
|
|
315
|
+
"calls" => [ usage["calls"].to_i, 1 ].max,
|
|
316
|
+
"input_tokens" => input,
|
|
317
|
+
"output_tokens" => output,
|
|
318
|
+
"cost" => cost&.round(6),
|
|
319
|
+
"model" => model,
|
|
320
|
+
"by_kind" => by_kind,
|
|
321
|
+
"source" => "reported",
|
|
322
|
+
"estimated" => estimated
|
|
323
|
+
}
|
|
324
|
+
end
|
|
325
|
+
|
|
326
|
+
# Judge traces priced from their input and output tokens — never their
|
|
327
|
+
# thinking tokens, which the output count already holds. nil without a trace.
|
|
328
|
+
def traces_usage(judge_traces)
|
|
329
|
+
return nil if judge_traces.empty?
|
|
330
|
+
|
|
331
|
+
costs = judge_traces.filter_map do |trace|
|
|
332
|
+
ModelPricing.estimate(model: trace.model, provider: trace.provider, input_tokens: trace.input_tokens, output_tokens: trace.output_tokens)
|
|
333
|
+
end
|
|
334
|
+
{
|
|
335
|
+
"calls" => judge_traces.size,
|
|
336
|
+
"input_tokens" => judge_traces.sum(&:input_tokens),
|
|
337
|
+
"output_tokens" => judge_traces.sum(&:output_tokens),
|
|
338
|
+
"cost" => costs.any? ? costs.sum.round(6) : 0.0,
|
|
339
|
+
"model" => judge_traces.filter_map { |trace| trace.model.presence }.first,
|
|
340
|
+
"by_kind" => judge_traces.filter_map { |trace| trace.action.presence }.tally,
|
|
341
|
+
"source" => "traces",
|
|
342
|
+
"estimated" => true
|
|
343
|
+
}
|
|
344
|
+
end
|
|
345
|
+
|
|
346
|
+
def sum_judge(parts, run_part, source)
|
|
347
|
+
costs = parts.filter_map { |part| part["cost"] }
|
|
348
|
+
{
|
|
349
|
+
"calls" => parts.sum { |part| part["calls"].to_i },
|
|
350
|
+
"input_tokens" => parts.sum { |part| part["input_tokens"].to_i },
|
|
351
|
+
"output_tokens" => parts.sum { |part| part["output_tokens"].to_i },
|
|
352
|
+
"cost" => costs.any? ? costs.sum.round(6) : nil,
|
|
353
|
+
"model" => parts.filter_map { |part| part["model"].presence }.first,
|
|
354
|
+
"by_kind" => parts.each_with_object({}) { |part, tally| (part["by_kind"] || {}).each { |kind, count| tally[kind] = tally.fetch(kind, 0) + count.to_i } },
|
|
355
|
+
"source" => source,
|
|
356
|
+
"estimated" => parts.any? { |part| part["estimated"] },
|
|
357
|
+
"run" => run_part && { "calls" => run_part["calls"].to_i, "cost" => run_part["cost"], "by_kind" => run_part["by_kind"] || {} }
|
|
358
|
+
}.compact
|
|
359
|
+
end
|
|
360
|
+
|
|
361
|
+
# --- roll-ups --------------------------------------------------------------
|
|
362
|
+
|
|
363
|
+
# The run's usage (EvaluationRun#usage): the agent's side summed over
|
|
364
|
+
# the results, the judge's beside it, and their total.
|
|
365
|
+
def usage_for(entries, judge)
|
|
366
|
+
runtime_ms = run.completed_at.present? && run.created_at.present? ? ((run.completed_at - run.created_at) * 1000).round : nil
|
|
367
|
+
if rows.any?
|
|
368
|
+
priced = entries.values.count { |entry| entry["cost"] }
|
|
369
|
+
cost = entries.values.filter_map { |entry| entry["cost"] }.sum.round(6)
|
|
370
|
+
reported = entries.values.count { |entry| entry["cost_source"] == "reported" }
|
|
371
|
+
estimated = entries.values.count { |entry| entry["cost_source"] == "estimated" }
|
|
372
|
+
agent_cost = priced.positive? ? cost : nil
|
|
373
|
+
{
|
|
374
|
+
"replays" => rows.size,
|
|
375
|
+
"priced" => priced,
|
|
376
|
+
"unpriced" => rows.size - priced,
|
|
377
|
+
"reported" => reported,
|
|
378
|
+
"estimated" => estimated,
|
|
379
|
+
"cost" => agent_cost,
|
|
380
|
+
"per_interaction" => agent_cost && priced.positive? ? (agent_cost / priced).round(6) : nil,
|
|
381
|
+
"input_tokens" => rows.sum { |row| row.input_tokens.to_i },
|
|
382
|
+
"output_tokens" => rows.sum { |row| row.output_tokens.to_i },
|
|
383
|
+
"model_time_ms" => rows.sum { |row| row.duration_ms.to_i },
|
|
384
|
+
"runtime_ms" => runtime_ms,
|
|
385
|
+
"cost_basis" => (cost_basis(reported, estimated) if priced.positive?),
|
|
386
|
+
"judge" => judge,
|
|
387
|
+
"total" => total_of(agent_cost, judge&.dig("cost"))
|
|
388
|
+
}.compact
|
|
389
|
+
elsif run.cohorts.any?
|
|
390
|
+
sampling_usage(judge, runtime_ms)
|
|
391
|
+
elsif judge
|
|
392
|
+
{ "runtime_ms" => runtime_ms, "judge" => judge, "total" => judge["cost"] }.compact
|
|
393
|
+
end
|
|
394
|
+
end
|
|
395
|
+
|
|
396
|
+
# A generation-sampling run's side: what the sampled generations cost to
|
|
397
|
+
# serve, as the runner recorded per cohort — always an estimate from
|
|
398
|
+
# tokens. A cohort recorded before `priced` counts all its samples when
|
|
399
|
+
# it has a cost and none when it has not.
|
|
400
|
+
def sampling_usage(judge, runtime_ms)
|
|
401
|
+
cohorts = run.cohorts.values
|
|
402
|
+
samples = cohorts.sum { |cohort| cohort["samples"].to_i }
|
|
403
|
+
priced = cohorts.sum { |cohort| cohort.key?("priced") ? cohort["priced"].to_i : (cohort["cost"].nil? ? 0 : cohort["samples"].to_i) }
|
|
404
|
+
costs = cohorts.filter_map { |cohort| cohort["cost"] }
|
|
405
|
+
cost = costs.any? ? costs.sum.to_f.round(6) : nil
|
|
406
|
+
{
|
|
407
|
+
"samples" => samples,
|
|
408
|
+
"priced" => priced,
|
|
409
|
+
"unpriced" => samples - priced,
|
|
410
|
+
"reported" => 0,
|
|
411
|
+
"estimated" => priced,
|
|
412
|
+
"cost" => cost,
|
|
413
|
+
"per_interaction" => cost && priced.positive? ? (cost / priced).round(6) : nil,
|
|
414
|
+
"input_tokens" => cohorts.sum { |cohort| cohort["input_tokens"].to_i },
|
|
415
|
+
"output_tokens" => cohorts.sum { |cohort| cohort["output_tokens"].to_i },
|
|
416
|
+
"runtime_ms" => runtime_ms,
|
|
417
|
+
"cost_basis" => (cost && "estimated"),
|
|
418
|
+
"judge" => judge,
|
|
419
|
+
"total" => total_of(cost, judge&.dig("cost"))
|
|
420
|
+
}.compact
|
|
421
|
+
end
|
|
422
|
+
|
|
423
|
+
def cost_basis(reported, estimated)
|
|
424
|
+
if reported.positive? && estimated.positive? then "mixed"
|
|
425
|
+
elsif estimated.positive? then "estimated"
|
|
426
|
+
else "reported"
|
|
427
|
+
end
|
|
428
|
+
end
|
|
429
|
+
|
|
430
|
+
def total_of(agent_cost, judge_cost)
|
|
431
|
+
return nil if agent_cost.nil? && judge_cost.nil?
|
|
432
|
+
|
|
433
|
+
(agent_cost.to_f + judge_cost.to_f).round(6)
|
|
434
|
+
end
|
|
435
|
+
|
|
436
|
+
# Per scenario key — the agent's cost summed over the models, the
|
|
437
|
+
# judge's, their total, whether any part is an estimate, and the same
|
|
438
|
+
# per model label — and for the run as a whole.
|
|
439
|
+
def costs_for(entries, judge)
|
|
440
|
+
scenarios = rows.group_by { |row| row.evaluated_scenario["key"].to_s }.to_h do |key, cohort|
|
|
441
|
+
parts = cohort.map { |row| entries[row.id.to_s] }
|
|
442
|
+
costs = parts.filter_map { |entry| entry["cost"] }
|
|
443
|
+
judge_costs = parts.filter_map { |entry| entry.dig("judge_usage", "cost") }
|
|
444
|
+
cost = costs.any? ? costs.sum.round(6) : nil
|
|
445
|
+
judge_cost = judge_costs.any? ? judge_costs.sum.round(6) : nil
|
|
446
|
+
[ key, {
|
|
447
|
+
"cost" => cost,
|
|
448
|
+
"judge_cost" => judge_cost,
|
|
449
|
+
"total" => total_of(cost, judge_cost),
|
|
450
|
+
"estimated" => parts.any? { |entry| entry["cost_source"] == "estimated" || entry.dig("judge_usage", "estimated") } ||
|
|
451
|
+
(costs.any? && costs.size < parts.size),
|
|
452
|
+
"models" => cohort.to_h do |row|
|
|
453
|
+
entry = entries[row.id.to_s]
|
|
454
|
+
[ label_for(row), { "cost" => entry["cost"], "judge_cost" => entry.dig("judge_usage", "cost"), "cost_source" => entry["cost_source"] } ]
|
|
455
|
+
end
|
|
456
|
+
} ]
|
|
457
|
+
end
|
|
458
|
+
agent_costs = entries.values.filter_map { |entry| entry["cost"] }
|
|
459
|
+
agent_cost = agent_costs.any? ? agent_costs.sum.round(6) : nil
|
|
460
|
+
reported = entries.values.count { |entry| entry["cost_source"] == "reported" }
|
|
461
|
+
estimated = entries.values.count { |entry| entry["cost_source"] == "estimated" }
|
|
462
|
+
{
|
|
463
|
+
"scenarios" => scenarios,
|
|
464
|
+
"run" => {
|
|
465
|
+
"agent_cost" => agent_cost,
|
|
466
|
+
"judge_cost" => judge&.dig("cost"),
|
|
467
|
+
"total" => total_of(agent_cost, judge&.dig("cost")),
|
|
468
|
+
"cost_basis" => (cost_basis(reported, estimated) if agent_cost)
|
|
469
|
+
}
|
|
470
|
+
}
|
|
471
|
+
end
|
|
472
|
+
|
|
473
|
+
# Per model label: `{ "cost", "priced", "reported", "estimated",
|
|
474
|
+
# "judge_cost", "judge_calls" }` over that label's results.
|
|
475
|
+
def by_label_for(entries)
|
|
476
|
+
rows.group_by { |row| label_for(row) }.to_h do |label, cohort|
|
|
477
|
+
parts = cohort.map { |row| entries[row.id.to_s] }
|
|
478
|
+
costs = parts.filter_map { |entry| entry["cost"] }
|
|
479
|
+
judge_costs = parts.filter_map { |entry| entry.dig("judge_usage", "cost") }
|
|
480
|
+
[ label, {
|
|
481
|
+
"cost" => costs.any? ? costs.sum.round(6) : nil,
|
|
482
|
+
"priced" => costs.size,
|
|
483
|
+
"reported" => parts.count { |entry| entry["cost_source"] == "reported" },
|
|
484
|
+
"estimated" => parts.count { |entry| entry["cost_source"] == "estimated" },
|
|
485
|
+
"judge_cost" => judge_costs.any? ? judge_costs.sum.round(6) : nil,
|
|
486
|
+
"judge_calls" => parts.sum { |entry| entry.dig("judge_usage", "calls").to_i }
|
|
487
|
+
} ]
|
|
488
|
+
end
|
|
489
|
+
end
|
|
490
|
+
|
|
491
|
+
# The label a result's model runs under in the run's summaries: the one
|
|
492
|
+
# the run's selection gave its provider and model, else — the way the
|
|
493
|
+
# dashboard assigns results to model columns — the recorded summary
|
|
494
|
+
# naming its model bare or as "provider/model", else the bare model.
|
|
495
|
+
def label_for(row)
|
|
496
|
+
@labels ||= {}
|
|
497
|
+
@labels[[ row.provider.to_s, row.model ]] ||= begin
|
|
498
|
+
selected = selected_labels[[ row.provider.to_s, row.model ]]
|
|
499
|
+
recorded = run.scores.is_a?(Hash) && run.scores["_models"].is_a?(Hash) ? run.scores["_models"].keys : []
|
|
500
|
+
selected || [ row.model, [ row.provider.presence, row.model ].compact.join("/") ].find { |name| recorded.include?(name) } || row.model
|
|
501
|
+
end
|
|
502
|
+
end
|
|
503
|
+
|
|
504
|
+
def selected_labels
|
|
505
|
+
@selected_labels ||= Array(run.selection["models"]).filter_map do |entry|
|
|
506
|
+
next unless entry.is_a?(Hash)
|
|
507
|
+
|
|
508
|
+
entry = entry.stringify_keys
|
|
509
|
+
next if entry["model"].blank?
|
|
510
|
+
|
|
511
|
+
[ [ entry["provider"].to_s, entry["model"] ], entry["label"].presence || entry["model"] ]
|
|
512
|
+
end.to_h
|
|
513
|
+
end
|
|
514
|
+
|
|
515
|
+
# The traces a run's results and judge link to, read within the run's
|
|
516
|
+
# tenant and reduced to what pricing needs: tokens, the llm span's
|
|
517
|
+
# model and provider, and the action (a judge trace's action is the
|
|
518
|
+
# kind of call it served — Correlation#judge names it).
|
|
519
|
+
class TraceReader
|
|
520
|
+
Trace = Struct.new(:trace_id, :input_tokens, :output_tokens, :model, :provider, :action, keyword_init: true)
|
|
521
|
+
|
|
522
|
+
# Reads every pending breakdown's traces, one query per tenant.
|
|
523
|
+
def self.preload(breakdowns)
|
|
524
|
+
breakdowns.group_by { |breakdown| new(breakdown).tenant_key }.each_value do |group|
|
|
525
|
+
reader = new(group.first)
|
|
526
|
+
traces = reader.read(group.flat_map(&:trace_ids))
|
|
527
|
+
group.each { |breakdown| breakdown.traces = traces }
|
|
528
|
+
end
|
|
529
|
+
end
|
|
530
|
+
|
|
531
|
+
def initialize(breakdown)
|
|
532
|
+
@breakdown = breakdown
|
|
533
|
+
@run = breakdown.run
|
|
534
|
+
end
|
|
535
|
+
|
|
536
|
+
# { trace_id => Trace } for +ids+ (the breakdown's own by default).
|
|
537
|
+
def read(ids = @breakdown.trace_ids)
|
|
538
|
+
ids = ids.map(&:to_s).uniq
|
|
539
|
+
return {} if ids.empty?
|
|
540
|
+
|
|
541
|
+
scope = tenant_scope
|
|
542
|
+
return {} if scope.nil?
|
|
543
|
+
|
|
544
|
+
table = ActionAgent.trace_model.table_name
|
|
545
|
+
rows = ActionAgent.trace_model.pluck_with_llm_model(
|
|
546
|
+
scope.where(trace_id: ids),
|
|
547
|
+
Arel.sql("#{table}.trace_id"), Arel.sql("#{table}.total_input_tokens"), Arel.sql("#{table}.total_output_tokens"),
|
|
548
|
+
Arel.sql("#{table}.agent_action")
|
|
549
|
+
)
|
|
550
|
+
rows.to_h do |model, trace_id, input, output, action|
|
|
551
|
+
[ trace_id.to_s, Trace.new(trace_id: trace_id.to_s, input_tokens: input.to_i, output_tokens: output.to_i,
|
|
552
|
+
model: model, provider: provider_of(model), action: action) ]
|
|
553
|
+
end
|
|
554
|
+
rescue StandardError => e
|
|
555
|
+
Rails.logger.warn("[ActionAgent] evaluation run #{@run.id} trace pricing failed: #{e.class}: #{e.message}")
|
|
556
|
+
{}
|
|
557
|
+
end
|
|
558
|
+
|
|
559
|
+
# The tenant the run's traces belong to, as a grouping key.
|
|
560
|
+
def tenant_key
|
|
561
|
+
return "install" unless ActionAgent.multi_tenant?
|
|
562
|
+
|
|
563
|
+
tenant_account&.id.to_s.presence || "none"
|
|
564
|
+
end
|
|
565
|
+
|
|
566
|
+
private
|
|
567
|
+
|
|
568
|
+
# A gateway's model names carry the vendor; the model's own name does
|
|
569
|
+
# not say which provider served it, which ModelPricing works around.
|
|
570
|
+
def provider_of(model)
|
|
571
|
+
head, rest = model.to_s.split("/", 2)
|
|
572
|
+
rest.present? && ModelPricing::GATEWAY_PREFIXES.include?(head.downcase) ? head.downcase : nil
|
|
573
|
+
end
|
|
574
|
+
|
|
575
|
+
# The traces this run may read: every trace on a single-tenant
|
|
576
|
+
# install; on a multi-tenant one, the tenant's own and nothing
|
|
577
|
+
# without a tenant.
|
|
578
|
+
def tenant_scope
|
|
579
|
+
traces = ActionAgent.trace_model
|
|
580
|
+
return traces.all unless ActionAgent.multi_tenant?
|
|
581
|
+
|
|
582
|
+
account = tenant_account
|
|
583
|
+
return nil unless account
|
|
584
|
+
|
|
585
|
+
traces.column_names.include?("account_id") ? traces.where(account_id: account.id) : traces.for_account(account)
|
|
586
|
+
end
|
|
587
|
+
|
|
588
|
+
# The account the run's report was published under, else the one the
|
|
589
|
+
# agent belongs to.
|
|
590
|
+
def tenant_account
|
|
591
|
+
return @tenant_account if defined?(@tenant_account)
|
|
592
|
+
|
|
593
|
+
klass = ActionAgent.account_class.to_s.safe_constantize
|
|
594
|
+
@tenant_account =
|
|
595
|
+
if klass.nil?
|
|
596
|
+
nil
|
|
597
|
+
elsif @run.external_tenant.present?
|
|
598
|
+
klass.find_by(id: @run.external_tenant)
|
|
599
|
+
else
|
|
600
|
+
agent = @run.evaluation&.agent
|
|
601
|
+
owner = agent&.owner
|
|
602
|
+
if owner.is_a?(klass) then owner
|
|
603
|
+
elsif agent.respond_to?(:account_id) && agent.account_id.present? then klass.find_by(id: agent.account_id)
|
|
604
|
+
end
|
|
605
|
+
end
|
|
606
|
+
end
|
|
607
|
+
end
|
|
608
|
+
end
|
|
609
|
+
end
|
|
@@ -190,7 +190,9 @@ module ActionAgent
|
|
|
190
190
|
# The cost here is the agent's: what the sampled interactions cost to
|
|
191
191
|
# operate. It is not what this run spent — scoring recorded generations
|
|
192
192
|
# costs nothing until a judge is asked, and the judge's own spend is
|
|
193
|
-
# recorded apart, under "_judge_usage" (see #judge_usage).
|
|
193
|
+
# recorded apart, under "_judge_usage" (see #judge_usage). It sums the
|
|
194
|
+
# generations that could be priced, and "priced" counts them, so a
|
|
195
|
+
# cohort with some unpriced generations reads as partial.
|
|
194
196
|
|
|
195
197
|
def cohort_summaries(samples, per_sample_scores)
|
|
196
198
|
samples
|
|
@@ -207,7 +209,8 @@ module ActionAgent
|
|
|
207
209
|
input_tokens = samples.sum { |generation| generation.input_tokens.to_i }
|
|
208
210
|
output_tokens = samples.sum { |generation| generation.output_tokens.to_i }
|
|
209
211
|
costs = samples.filter_map do |generation|
|
|
210
|
-
ModelPricing.estimate(model: generation.model,
|
|
212
|
+
ModelPricing.estimate(model: generation.model, provider: generation.provider.presence, input_tokens: generation.input_tokens,
|
|
213
|
+
output_tokens: generation.output_tokens)
|
|
211
214
|
end
|
|
212
215
|
|
|
213
216
|
{
|
|
@@ -217,7 +220,8 @@ module ActionAgent
|
|
|
217
220
|
"avg_duration_ms" => durations_ms.any? ? (durations_ms.sum / durations_ms.size).round : nil,
|
|
218
221
|
"input_tokens" => input_tokens,
|
|
219
222
|
"output_tokens" => output_tokens,
|
|
220
|
-
"cost" => costs.any? ? costs.sum.round(6) : nil
|
|
223
|
+
"cost" => costs.any? ? costs.sum.round(6) : nil,
|
|
224
|
+
"priced" => costs.size
|
|
221
225
|
}
|
|
222
226
|
end
|
|
223
227
|
|
|
@@ -254,10 +258,10 @@ module ActionAgent
|
|
|
254
258
|
|
|
255
259
|
def record_judge_call(kind, response)
|
|
256
260
|
usage = response.respond_to?(:usage) ? response.usage : nil
|
|
257
|
-
input_tokens = usage
|
|
261
|
+
input_tokens = judge_input_tokens(usage)
|
|
258
262
|
output_tokens = usage.respond_to?(:output_tokens) ? usage.output_tokens.to_i : 0
|
|
259
263
|
model = (response.respond_to?(:model) && response.model.presence) || @evaluation.judge_model.presence
|
|
260
|
-
cost = ModelPricing.estimate(model: model, input_tokens: input_tokens, output_tokens: output_tokens)
|
|
264
|
+
cost = ModelPricing.estimate(model: model, provider: judge_provider, input_tokens: input_tokens, output_tokens: output_tokens)
|
|
261
265
|
|
|
262
266
|
@judge_usage ||= { "calls" => 0, "input_tokens" => 0, "output_tokens" => 0, "cost" => nil, "model" => nil, "by_kind" => {} }
|
|
263
267
|
@judge_usage["calls"] += 1
|
|
@@ -268,6 +272,22 @@ module ActionAgent
|
|
|
268
272
|
@judge_usage["by_kind"][kind.to_s] = @judge_usage["by_kind"].fetch(kind.to_s, 0) + 1
|
|
269
273
|
end
|
|
270
274
|
|
|
275
|
+
# Every token the judge was billed for reading. Anthropic reports the
|
|
276
|
+
# tokens read from or written to the prompt cache apart from
|
|
277
|
+
# `input_tokens`, so they are added; OpenAI's prompt count already holds
|
|
278
|
+
# its cached tokens. Priced at the input rate, which overstates a cache
|
|
279
|
+
# read: the figure is an upper bound.
|
|
280
|
+
def judge_input_tokens(usage)
|
|
281
|
+
return 0 unless usage.respond_to?(:input_tokens)
|
|
282
|
+
|
|
283
|
+
total = usage.input_tokens.to_i
|
|
284
|
+
return total unless judge_provider == :anthropic
|
|
285
|
+
|
|
286
|
+
total += usage.cached_tokens.to_i if usage.respond_to?(:cached_tokens)
|
|
287
|
+
total += usage.cache_creation_tokens.to_i if usage.respond_to?(:cache_creation_tokens)
|
|
288
|
+
total
|
|
289
|
+
end
|
|
290
|
+
|
|
271
291
|
def sample_generations(model: nil)
|
|
272
292
|
scope = @evaluation.agent.generations
|
|
273
293
|
scope = scope.where(model: model) if model
|