actionagent 1.8.0 → 1.8.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (32) hide show
  1. checksums.yaml +4 -4
  2. data/app/assets/builds/action_agent.js +58 -58
  3. data/app/controllers/action_agent/api/code_sessions_controller.rb +19 -7
  4. data/app/controllers/action_agent/api/evaluations_controller.rb +49 -3
  5. data/app/controllers/action_agent/api/sandboxes_controller.rb +8 -0
  6. data/app/controllers/concerns/action_agent/api/mcp_dashboard_tools.rb +4 -2
  7. data/app/jobs/action_agent/code_session_job.rb +13 -8
  8. data/app/models/action_agent/agent.rb +33 -0
  9. data/app/models/action_agent/code_session.rb +6 -2
  10. data/app/models/action_agent/evaluation.rb +64 -0
  11. data/app/models/action_agent/evaluation_run.rb +143 -53
  12. data/app/models/action_agent/evaluation_scenario_result.rb +15 -3
  13. data/app/models/action_agent/model_pricing.rb +214 -34
  14. data/app/models/action_agent/provider_key.rb +7 -1
  15. data/app/models/action_agent/sandbox_session.rb +3 -2
  16. data/app/serializers/action_agent/evaluation_serializer.rb +41 -10
  17. data/app/services/action_agent/agent_scorecard.rb +50 -21
  18. data/app/services/action_agent/codex_session_events.rb +37 -0
  19. data/app/services/action_agent/evaluation_report_import.rb +84 -5
  20. data/app/services/action_agent/evaluation_run_cost.rb +609 -0
  21. data/app/services/action_agent/evaluation_runner_service.rb +25 -5
  22. data/app/services/action_agent/evaluation_standing.rb +145 -0
  23. data/app/services/action_agent/local_sandbox_backend.rb +47 -21
  24. data/app/services/action_agent/sandbox_orchestrator.rb +14 -1
  25. data/app/services/action_agent/scenario_evaluation_runner.rb +2 -1
  26. data/config/routes.rb +1 -1
  27. data/lib/action_agent/version.rb +1 -1
  28. data/lib/action_agent.rb +6 -0
  29. data/lib/generators/action_agent/install_generator.rb +9 -6
  30. data/lib/generators/action_agent/templates/action_agent.rb.erb +3 -0
  31. data/lib/generators/action_agent/templates/add_code_session_runner.rb.erb +9 -0
  32. metadata +6 -2
@@ -0,0 +1,609 @@
1
+ # frozen_string_literal: true
2
+
3
+ module ActionAgent
4
+ # What an evaluation run cost, worked out from everything the run and its
5
+ # results recorded, so that no cost shows blank where tokens exist.
6
+ #
7
+ # Each scenario result is priced down a chain, the first link that applies
8
+ # winning, and says which link it was (`cost_source`):
9
+ #
10
+ # reported — a cost the publishing application sent with the result
11
+ # estimated — the cost the engine stored when it ran the replay itself
12
+ # (tokens × the model's rate at the time), else the
13
+ # result's tokens × the model's rate now, else the tokens
14
+ # of the telemetry trace the result links to × that rate,
15
+ # else the prompt and answer at ~4 characters a token — a
16
+ # lower bound, for a result that recorded no tokens at all
17
+ # no_usage — both token counts recorded as zero: nothing was
18
+ # generated, so the result cost $0.00 (an errored replay)
19
+ # unpriced — no tokens, no trace and no text: nothing to price from
20
+ #
21
+ # An estimate carries its rate (`cost_rate`: $ per million tokens, where
22
+ # it came from, and the tokens it was applied to), so a figure can show
23
+ # its working. A trace's thinking tokens are already inside its output
24
+ # tokens and are never priced again.
25
+ #
26
+ # The judge's spend is kept apart from the agent's and found down its own
27
+ # chain: the engine's own meter (`scores._judge_usage`), else what the
28
+ # application reported per result (`diagnosis._judge_usage`) and for the
29
+ # run (`scores._judge_usage_run`), else the judge traces the results and
30
+ # the run link to, priced from their input and output tokens. Traces are
31
+ # read within the run's tenant: the account its report was published
32
+ # under, or the agent owner's account.
33
+ #
34
+ # Nothing here writes. A finished run's figures are cached in Rails.cache
35
+ # under the run, its updated_at and the pricing tables in force, so a
36
+ # page of runs prices its traces once; a run still landing results is
37
+ # worked out fresh each time. `preload` prices several runs with one
38
+ # query per table.
39
+ class EvaluationRunCost
40
+ CACHE_VERSION = 2
41
+ CACHE_TTL = 7.days
42
+ # The approximation the context meter applies to text it sizes itself.
43
+ CHARS_PER_TOKEN = 4
44
+ SOURCES = %w[reported estimated no_usage unpriced].freeze
45
+ # Judge calls no result owns: the verdict across cohorts and, for a
46
+ # judge_defined evaluation, authoring the criteria.
47
+ RUN_LEVEL_KINDS = %w[verdict define].freeze
48
+
49
+ attr_reader :run
50
+
51
+ # The breakdown for one run, from the cache when it is finished and
52
+ # cached, else computed.
53
+ def self.for(run)
54
+ new(run).tap(&:data)
55
+ end
56
+
57
+ # { run.id => EvaluationRunCost } for every run in +runs+, loading the
58
+ # uncached runs' results and traces in one query each.
59
+ def self.preload(runs)
60
+ runs = Array(runs).compact.uniq(&:id)
61
+ return {} if runs.empty?
62
+
63
+ breakdowns = runs.index_with { |run| new(run) }
64
+ pending = breakdowns.values.reject(&:cached?)
65
+ rows = EvaluationScenarioResult.where(evaluation_run_id: pending.map { |breakdown| breakdown.run.id })
66
+ .includes(:scenario).group_by(&:evaluation_run_id)
67
+ pending.each { |breakdown| breakdown.rows = rows.fetch(breakdown.run.id, []) }
68
+ TraceReader.preload(pending)
69
+ pending.each(&:data)
70
+ breakdowns.transform_keys(&:id)
71
+ end
72
+
73
+ def initialize(run)
74
+ @run = run
75
+ end
76
+
77
+ # The results' rows, loaded once; `preload` hands them in.
78
+ attr_writer :rows
79
+
80
+ def rows
81
+ @rows ||= run.scenario_results.includes(:scenario).to_a
82
+ end
83
+
84
+ def cached?
85
+ return false unless cacheable?
86
+
87
+ @data ||= Rails.cache.read(cache_key)
88
+ @data.present?
89
+ end
90
+
91
+ # Every figure, as one serializable hash:
92
+ # { "results" => { id => entry }, "judge" => usage | nil, "usage" => ..., "costs" => ..., "by_label" => ... }
93
+ def data
94
+ return @data if @data
95
+
96
+ @data = cacheable? ? Rails.cache.fetch(cache_key, expires_in: CACHE_TTL) { compute } : compute
97
+ end
98
+
99
+ # The result's figures: `{ "cost", "reported_cost", "cost_source",
100
+ # "cost_rate", "judge_usage" }`, with `cost` the effective one. A
101
+ # result this run does not hold reads as unpriced.
102
+ def result(result)
103
+ data["results"][result.id.to_s] || unpriced_entry
104
+ end
105
+
106
+ # The run's usage, the agent's side and the judge's apart, or nil when
107
+ # the run recorded nothing on either side (see EvaluationRun#usage).
108
+ def usage
109
+ data["usage"]&.transform_keys(&:to_sym)
110
+ end
111
+
112
+ # The judge's spend on the run, or nil when no judge was asked.
113
+ def judge_usage
114
+ data["judge"]
115
+ end
116
+
117
+ # The judge's calls no result owns, in the shape Report.new(judge_usage:)
118
+ # takes. The engine's meter knows no result's share, so a metered run
119
+ # hands the report the whole meter: from the report's side, none of it
120
+ # is any result's.
121
+ def judge_usage_run
122
+ judge = judge_usage
123
+ return nil unless judge
124
+ return judge.except("run", "estimated") if judge["source"] == "meter"
125
+ return nil unless judge["run"]
126
+
127
+ judge["run"].merge("model" => judge["model"], "source" => judge["source"]).compact
128
+ end
129
+
130
+ # Costs per scenario and for the run (see Api::EvaluationsController#show_run).
131
+ def costs
132
+ data["costs"]
133
+ end
134
+
135
+ # Per model label: the effective cost and the counts behind it, to merge
136
+ # into the run's recorded `_models` summaries.
137
+ def by_label(labels)
138
+ data["by_label"].select { |label, _| labels.include?(label) }
139
+ end
140
+
141
+ # The trace ids this run refers to: each result's own replay trace and
142
+ # judge traces, and the run's judge traces (the verdict's). TraceReader
143
+ # reads them and hands the traces back through `traces=`.
144
+ def trace_ids
145
+ ids = rows.flat_map { |row| [ row.replay_metadata["trace_id"], *Array(row.replay_metadata["judge_trace_ids"]) ] }
146
+ ids.concat(Array(run.report_metadata["judge_trace_ids"]))
147
+ ids.filter_map { |id| id.to_s.presence }.uniq
148
+ end
149
+
150
+ attr_writer :traces
151
+
152
+ private
153
+
154
+ def cacheable?
155
+ run.persisted? && run.complete? && run.updated_at.present?
156
+ end
157
+
158
+ def cache_key
159
+ [ "action_agent", "evaluation_run_cost", CACHE_VERSION, run.id, run.updated_at.utc.strftime("%Y%m%d%H%M%S%6N"), ModelPricing.fingerprint ]
160
+ end
161
+
162
+ def imported?
163
+ run.external_run_id.present?
164
+ end
165
+
166
+ def unpriced_entry
167
+ { "cost" => nil, "reported_cost" => nil, "cost_source" => "unpriced", "cost_rate" => nil, "judge_usage" => nil }
168
+ end
169
+
170
+ # --- computation ---------------------------------------------------------
171
+
172
+ def compute
173
+ traces = @traces || TraceReader.new(self).read
174
+ entries = rows.to_h { |row| [ row.id.to_s, result_entry(row, traces).merge("judge_usage" => result_judge(row, traces)) ] }
175
+ judge = run_judge_usage(entries, traces)
176
+ {
177
+ "results" => entries,
178
+ "judge" => judge,
179
+ "usage" => usage_for(entries, judge),
180
+ "costs" => costs_for(entries, judge),
181
+ "by_label" => by_label_for(entries)
182
+ }
183
+ end
184
+
185
+ # The agent's cost of one result, down the chain in the class comment.
186
+ def result_entry(row, traces)
187
+ if !row.cost.nil? && imported?
188
+ return { "cost" => row.cost.to_f, "reported_cost" => row.cost.to_f, "cost_source" => "reported", "cost_rate" => nil }
189
+ end
190
+
191
+ if !row.cost.nil?
192
+ rate = ModelPricing.rate_detail(row.model, provider: row.provider)
193
+ return estimated(row.cost.to_f, rate, "tokens", row.input_tokens, row.output_tokens)
194
+ end
195
+
196
+ input = row.input_tokens
197
+ output = row.output_tokens
198
+ return { "cost" => 0.0, "reported_cost" => nil, "cost_source" => "no_usage", "cost_rate" => nil } if input == 0 && output == 0
199
+
200
+ if input.to_i.positive? || output.to_i.positive?
201
+ return priced("tokens", row.model, row.provider, input, output)
202
+ end
203
+
204
+ trace = traces[row.replay_metadata["trace_id"].to_s]
205
+ if trace
206
+ return { "cost" => 0.0, "reported_cost" => nil, "cost_source" => "no_usage", "cost_rate" => nil } if trace.input_tokens.zero? && trace.output_tokens.zero?
207
+
208
+ return priced("trace", trace.model.presence || row.model, trace.provider.presence || row.provider, trace.input_tokens, trace.output_tokens)
209
+ end
210
+
211
+ prompt_chars = row.evaluated_scenario["prompt"].to_s.length
212
+ answer_chars = row.output.to_s.length
213
+ if prompt_chars.positive? || answer_chars.positive?
214
+ return priced("chars", row.model, row.provider, prompt_chars / CHARS_PER_TOKEN, answer_chars / CHARS_PER_TOKEN)
215
+ end
216
+
217
+ unpriced_entry.except("judge_usage")
218
+ end
219
+
220
+ def priced(basis, model, provider, input_tokens, output_tokens)
221
+ detail = ModelPricing.estimate_detailed(model: model, provider: provider, input_tokens: input_tokens, output_tokens: output_tokens)
222
+ return { "cost" => 0.0, "reported_cost" => nil, "cost_source" => "no_usage", "cost_rate" => nil } unless detail
223
+
224
+ rate = { input: detail[:input_rate], output: detail[:output_rate], source: detail[:source] }
225
+ estimated(detail[:cost], rate, basis, input_tokens, output_tokens)
226
+ end
227
+
228
+ def estimated(cost, rate, basis, input_tokens, output_tokens)
229
+ {
230
+ "cost" => cost.to_f.round(6),
231
+ "reported_cost" => nil,
232
+ "cost_source" => "estimated",
233
+ "cost_rate" => {
234
+ "input" => rate[:input], "output" => rate[:output], "source" => rate[:source],
235
+ "basis" => basis, "input_tokens" => input_tokens.to_i, "output_tokens" => output_tokens.to_i
236
+ }
237
+ }
238
+ end
239
+
240
+ # --- judge -----------------------------------------------------------------
241
+
242
+ # What the judge spent on one result: the application's own figures when
243
+ # it reported them, else the result's judge traces priced; nil when the
244
+ # engine metered the run (the meter is per run, not per result) or
245
+ # nothing links a judge call to the result.
246
+ def result_judge(row, traces)
247
+ return nil if run.judge_usage_meter
248
+
249
+ reported = row.diagnosis.is_a?(Hash) ? row.diagnosis["_judge_usage"] : nil
250
+ return normalize_reported_judge(reported) if reported.is_a?(Hash)
251
+
252
+ judge_traces = Array(row.replay_metadata["judge_trace_ids"]).filter_map { |id| traces[id.to_s] }
253
+ traces_usage(judge_traces)
254
+ end
255
+
256
+ # The run's judge usage: `{ "calls", "input_tokens", "output_tokens",
257
+ # "cost", "model", "by_kind", "source", "estimated", "run" => { "calls",
258
+ # "cost", "by_kind" } }`, the engine meter first, then what the
259
+ # application reported, then the judge traces.
260
+ def run_judge_usage(entries, traces)
261
+ if (meter = run.judge_usage_meter)
262
+ return metered_judge(meter)
263
+ end
264
+
265
+ parts = entries.values.filter_map { |entry| entry["judge_usage"] }
266
+ reported_run = run.scores.is_a?(Hash) ? run.scores["_judge_usage_run"] : nil
267
+ run_part = normalize_reported_judge(reported_run) if reported_run.is_a?(Hash)
268
+ if parts.any? { |part| part["source"] == "reported" } || run_part
269
+ return sum_judge(parts + [ run_part ].compact, run_part, "reported")
270
+ end
271
+
272
+ result_ids = rows.flat_map { |row| Array(row.replay_metadata["judge_trace_ids"]) }.map(&:to_s)
273
+ run_traces = Array(run.report_metadata["judge_trace_ids"]).map(&:to_s).uniq.reject { |id| result_ids.include?(id) }.filter_map { |id| traces[id] }
274
+ run_part = traces_usage(run_traces)
275
+ return nil if parts.empty? && run_part.nil?
276
+
277
+ sum_judge(parts + [ run_part ].compact, run_part, "traces")
278
+ end
279
+
280
+ # The engine's meter knows the run's total and what each call was for,
281
+ # not which result each call served: the run-level part is the calls
282
+ # of a run-level kind, their cost unknown.
283
+ def metered_judge(meter)
284
+ by_kind = (meter["by_kind"] || {}).to_h.transform_keys(&:to_s)
285
+ run_kinds = by_kind.select { |kind, _| RUN_LEVEL_KINDS.include?(kind) }
286
+ {
287
+ "calls" => meter["calls"].to_i,
288
+ "input_tokens" => meter["input_tokens"].to_i,
289
+ "output_tokens" => meter["output_tokens"].to_i,
290
+ "cost" => meter["cost"]&.to_f&.round(6),
291
+ "model" => meter["model"].presence,
292
+ "by_kind" => by_kind,
293
+ "source" => "meter",
294
+ "estimated" => true,
295
+ "run" => { "calls" => run_kinds.values.sum(&:to_i), "cost" => nil, "by_kind" => run_kinds }
296
+ }
297
+ end
298
+
299
+ # A usage the application reported — per result, or for the run —
300
+ # bounded to the keys the dashboard reads. A usage with tokens but no
301
+ # cost is priced at its model's rate and says so.
302
+ def normalize_reported_judge(usage)
303
+ usage = usage.to_h.transform_keys(&:to_s)
304
+ input = usage["input_tokens"].to_i
305
+ output = usage["output_tokens"].to_i
306
+ model = usage["model"].presence
307
+ cost = usage["cost"].is_a?(Numeric) ? usage["cost"].to_f : nil
308
+ estimated = false
309
+ if cost.nil? && (input.positive? || output.positive?)
310
+ cost = ModelPricing.estimate(model: model, input_tokens: input, output_tokens: output)
311
+ estimated = true
312
+ end
313
+ by_kind = usage["by_kind"].is_a?(Hash) ? usage["by_kind"].to_h { |kind, count| [ kind.to_s, count.to_i ] } : {}
314
+ {
315
+ "calls" => [ usage["calls"].to_i, 1 ].max,
316
+ "input_tokens" => input,
317
+ "output_tokens" => output,
318
+ "cost" => cost&.round(6),
319
+ "model" => model,
320
+ "by_kind" => by_kind,
321
+ "source" => "reported",
322
+ "estimated" => estimated
323
+ }
324
+ end
325
+
326
+ # Judge traces priced from their input and output tokens — never their
327
+ # thinking tokens, which the output count already holds. nil without a trace.
328
+ def traces_usage(judge_traces)
329
+ return nil if judge_traces.empty?
330
+
331
+ costs = judge_traces.filter_map do |trace|
332
+ ModelPricing.estimate(model: trace.model, provider: trace.provider, input_tokens: trace.input_tokens, output_tokens: trace.output_tokens)
333
+ end
334
+ {
335
+ "calls" => judge_traces.size,
336
+ "input_tokens" => judge_traces.sum(&:input_tokens),
337
+ "output_tokens" => judge_traces.sum(&:output_tokens),
338
+ "cost" => costs.any? ? costs.sum.round(6) : 0.0,
339
+ "model" => judge_traces.filter_map { |trace| trace.model.presence }.first,
340
+ "by_kind" => judge_traces.filter_map { |trace| trace.action.presence }.tally,
341
+ "source" => "traces",
342
+ "estimated" => true
343
+ }
344
+ end
345
+
346
+ def sum_judge(parts, run_part, source)
347
+ costs = parts.filter_map { |part| part["cost"] }
348
+ {
349
+ "calls" => parts.sum { |part| part["calls"].to_i },
350
+ "input_tokens" => parts.sum { |part| part["input_tokens"].to_i },
351
+ "output_tokens" => parts.sum { |part| part["output_tokens"].to_i },
352
+ "cost" => costs.any? ? costs.sum.round(6) : nil,
353
+ "model" => parts.filter_map { |part| part["model"].presence }.first,
354
+ "by_kind" => parts.each_with_object({}) { |part, tally| (part["by_kind"] || {}).each { |kind, count| tally[kind] = tally.fetch(kind, 0) + count.to_i } },
355
+ "source" => source,
356
+ "estimated" => parts.any? { |part| part["estimated"] },
357
+ "run" => run_part && { "calls" => run_part["calls"].to_i, "cost" => run_part["cost"], "by_kind" => run_part["by_kind"] || {} }
358
+ }.compact
359
+ end
360
+
361
+ # --- roll-ups --------------------------------------------------------------
362
+
363
+ # The run's usage (EvaluationRun#usage): the agent's side summed over
364
+ # the results, the judge's beside it, and their total.
365
+ def usage_for(entries, judge)
366
+ runtime_ms = run.completed_at.present? && run.created_at.present? ? ((run.completed_at - run.created_at) * 1000).round : nil
367
+ if rows.any?
368
+ priced = entries.values.count { |entry| entry["cost"] }
369
+ cost = entries.values.filter_map { |entry| entry["cost"] }.sum.round(6)
370
+ reported = entries.values.count { |entry| entry["cost_source"] == "reported" }
371
+ estimated = entries.values.count { |entry| entry["cost_source"] == "estimated" }
372
+ agent_cost = priced.positive? ? cost : nil
373
+ {
374
+ "replays" => rows.size,
375
+ "priced" => priced,
376
+ "unpriced" => rows.size - priced,
377
+ "reported" => reported,
378
+ "estimated" => estimated,
379
+ "cost" => agent_cost,
380
+ "per_interaction" => agent_cost && priced.positive? ? (agent_cost / priced).round(6) : nil,
381
+ "input_tokens" => rows.sum { |row| row.input_tokens.to_i },
382
+ "output_tokens" => rows.sum { |row| row.output_tokens.to_i },
383
+ "model_time_ms" => rows.sum { |row| row.duration_ms.to_i },
384
+ "runtime_ms" => runtime_ms,
385
+ "cost_basis" => (cost_basis(reported, estimated) if priced.positive?),
386
+ "judge" => judge,
387
+ "total" => total_of(agent_cost, judge&.dig("cost"))
388
+ }.compact
389
+ elsif run.cohorts.any?
390
+ sampling_usage(judge, runtime_ms)
391
+ elsif judge
392
+ { "runtime_ms" => runtime_ms, "judge" => judge, "total" => judge["cost"] }.compact
393
+ end
394
+ end
395
+
396
+ # A generation-sampling run's side: what the sampled generations cost to
397
+ # serve, as the runner recorded per cohort — always an estimate from
398
+ # tokens. A cohort recorded before `priced` counts all its samples when
399
+ # it has a cost and none when it has not.
400
+ def sampling_usage(judge, runtime_ms)
401
+ cohorts = run.cohorts.values
402
+ samples = cohorts.sum { |cohort| cohort["samples"].to_i }
403
+ priced = cohorts.sum { |cohort| cohort.key?("priced") ? cohort["priced"].to_i : (cohort["cost"].nil? ? 0 : cohort["samples"].to_i) }
404
+ costs = cohorts.filter_map { |cohort| cohort["cost"] }
405
+ cost = costs.any? ? costs.sum.to_f.round(6) : nil
406
+ {
407
+ "samples" => samples,
408
+ "priced" => priced,
409
+ "unpriced" => samples - priced,
410
+ "reported" => 0,
411
+ "estimated" => priced,
412
+ "cost" => cost,
413
+ "per_interaction" => cost && priced.positive? ? (cost / priced).round(6) : nil,
414
+ "input_tokens" => cohorts.sum { |cohort| cohort["input_tokens"].to_i },
415
+ "output_tokens" => cohorts.sum { |cohort| cohort["output_tokens"].to_i },
416
+ "runtime_ms" => runtime_ms,
417
+ "cost_basis" => (cost && "estimated"),
418
+ "judge" => judge,
419
+ "total" => total_of(cost, judge&.dig("cost"))
420
+ }.compact
421
+ end
422
+
423
+ def cost_basis(reported, estimated)
424
+ if reported.positive? && estimated.positive? then "mixed"
425
+ elsif estimated.positive? then "estimated"
426
+ else "reported"
427
+ end
428
+ end
429
+
430
+ def total_of(agent_cost, judge_cost)
431
+ return nil if agent_cost.nil? && judge_cost.nil?
432
+
433
+ (agent_cost.to_f + judge_cost.to_f).round(6)
434
+ end
435
+
436
+ # Per scenario key — the agent's cost summed over the models, the
437
+ # judge's, their total, whether any part is an estimate, and the same
438
+ # per model label — and for the run as a whole.
439
+ def costs_for(entries, judge)
440
+ scenarios = rows.group_by { |row| row.evaluated_scenario["key"].to_s }.to_h do |key, cohort|
441
+ parts = cohort.map { |row| entries[row.id.to_s] }
442
+ costs = parts.filter_map { |entry| entry["cost"] }
443
+ judge_costs = parts.filter_map { |entry| entry.dig("judge_usage", "cost") }
444
+ cost = costs.any? ? costs.sum.round(6) : nil
445
+ judge_cost = judge_costs.any? ? judge_costs.sum.round(6) : nil
446
+ [ key, {
447
+ "cost" => cost,
448
+ "judge_cost" => judge_cost,
449
+ "total" => total_of(cost, judge_cost),
450
+ "estimated" => parts.any? { |entry| entry["cost_source"] == "estimated" || entry.dig("judge_usage", "estimated") } ||
451
+ (costs.any? && costs.size < parts.size),
452
+ "models" => cohort.to_h do |row|
453
+ entry = entries[row.id.to_s]
454
+ [ label_for(row), { "cost" => entry["cost"], "judge_cost" => entry.dig("judge_usage", "cost"), "cost_source" => entry["cost_source"] } ]
455
+ end
456
+ } ]
457
+ end
458
+ agent_costs = entries.values.filter_map { |entry| entry["cost"] }
459
+ agent_cost = agent_costs.any? ? agent_costs.sum.round(6) : nil
460
+ reported = entries.values.count { |entry| entry["cost_source"] == "reported" }
461
+ estimated = entries.values.count { |entry| entry["cost_source"] == "estimated" }
462
+ {
463
+ "scenarios" => scenarios,
464
+ "run" => {
465
+ "agent_cost" => agent_cost,
466
+ "judge_cost" => judge&.dig("cost"),
467
+ "total" => total_of(agent_cost, judge&.dig("cost")),
468
+ "cost_basis" => (cost_basis(reported, estimated) if agent_cost)
469
+ }
470
+ }
471
+ end
472
+
473
+ # Per model label: `{ "cost", "priced", "reported", "estimated",
474
+ # "judge_cost", "judge_calls" }` over that label's results.
475
+ def by_label_for(entries)
476
+ rows.group_by { |row| label_for(row) }.to_h do |label, cohort|
477
+ parts = cohort.map { |row| entries[row.id.to_s] }
478
+ costs = parts.filter_map { |entry| entry["cost"] }
479
+ judge_costs = parts.filter_map { |entry| entry.dig("judge_usage", "cost") }
480
+ [ label, {
481
+ "cost" => costs.any? ? costs.sum.round(6) : nil,
482
+ "priced" => costs.size,
483
+ "reported" => parts.count { |entry| entry["cost_source"] == "reported" },
484
+ "estimated" => parts.count { |entry| entry["cost_source"] == "estimated" },
485
+ "judge_cost" => judge_costs.any? ? judge_costs.sum.round(6) : nil,
486
+ "judge_calls" => parts.sum { |entry| entry.dig("judge_usage", "calls").to_i }
487
+ } ]
488
+ end
489
+ end
490
+
491
+ # The label a result's model runs under in the run's summaries: the one
492
+ # the run's selection gave its provider and model, else — the way the
493
+ # dashboard assigns results to model columns — the recorded summary
494
+ # naming its model bare or as "provider/model", else the bare model.
495
+ def label_for(row)
496
+ @labels ||= {}
497
+ @labels[[ row.provider.to_s, row.model ]] ||= begin
498
+ selected = selected_labels[[ row.provider.to_s, row.model ]]
499
+ recorded = run.scores.is_a?(Hash) && run.scores["_models"].is_a?(Hash) ? run.scores["_models"].keys : []
500
+ selected || [ row.model, [ row.provider.presence, row.model ].compact.join("/") ].find { |name| recorded.include?(name) } || row.model
501
+ end
502
+ end
503
+
504
+ def selected_labels
505
+ @selected_labels ||= Array(run.selection["models"]).filter_map do |entry|
506
+ next unless entry.is_a?(Hash)
507
+
508
+ entry = entry.stringify_keys
509
+ next if entry["model"].blank?
510
+
511
+ [ [ entry["provider"].to_s, entry["model"] ], entry["label"].presence || entry["model"] ]
512
+ end.to_h
513
+ end
514
+
515
+ # The traces a run's results and judge link to, read within the run's
516
+ # tenant and reduced to what pricing needs: tokens, the llm span's
517
+ # model and provider, and the action (a judge trace's action is the
518
+ # kind of call it served — Correlation#judge names it).
519
+ class TraceReader
520
+ Trace = Struct.new(:trace_id, :input_tokens, :output_tokens, :model, :provider, :action, keyword_init: true)
521
+
522
+ # Reads every pending breakdown's traces, one query per tenant.
523
+ def self.preload(breakdowns)
524
+ breakdowns.group_by { |breakdown| new(breakdown).tenant_key }.each_value do |group|
525
+ reader = new(group.first)
526
+ traces = reader.read(group.flat_map(&:trace_ids))
527
+ group.each { |breakdown| breakdown.traces = traces }
528
+ end
529
+ end
530
+
531
+ def initialize(breakdown)
532
+ @breakdown = breakdown
533
+ @run = breakdown.run
534
+ end
535
+
536
+ # { trace_id => Trace } for +ids+ (the breakdown's own by default).
537
+ def read(ids = @breakdown.trace_ids)
538
+ ids = ids.map(&:to_s).uniq
539
+ return {} if ids.empty?
540
+
541
+ scope = tenant_scope
542
+ return {} if scope.nil?
543
+
544
+ table = ActionAgent.trace_model.table_name
545
+ rows = ActionAgent.trace_model.pluck_with_llm_model(
546
+ scope.where(trace_id: ids),
547
+ Arel.sql("#{table}.trace_id"), Arel.sql("#{table}.total_input_tokens"), Arel.sql("#{table}.total_output_tokens"),
548
+ Arel.sql("#{table}.agent_action")
549
+ )
550
+ rows.to_h do |model, trace_id, input, output, action|
551
+ [ trace_id.to_s, Trace.new(trace_id: trace_id.to_s, input_tokens: input.to_i, output_tokens: output.to_i,
552
+ model: model, provider: provider_of(model), action: action) ]
553
+ end
554
+ rescue StandardError => e
555
+ Rails.logger.warn("[ActionAgent] evaluation run #{@run.id} trace pricing failed: #{e.class}: #{e.message}")
556
+ {}
557
+ end
558
+
559
+ # The tenant the run's traces belong to, as a grouping key.
560
+ def tenant_key
561
+ return "install" unless ActionAgent.multi_tenant?
562
+
563
+ tenant_account&.id.to_s.presence || "none"
564
+ end
565
+
566
+ private
567
+
568
+ # A gateway's model names carry the vendor; the model's own name does
569
+ # not say which provider served it, which ModelPricing works around.
570
+ def provider_of(model)
571
+ head, rest = model.to_s.split("/", 2)
572
+ rest.present? && ModelPricing::GATEWAY_PREFIXES.include?(head.downcase) ? head.downcase : nil
573
+ end
574
+
575
+ # The traces this run may read: every trace on a single-tenant
576
+ # install; on a multi-tenant one, the tenant's own and nothing
577
+ # without a tenant.
578
+ def tenant_scope
579
+ traces = ActionAgent.trace_model
580
+ return traces.all unless ActionAgent.multi_tenant?
581
+
582
+ account = tenant_account
583
+ return nil unless account
584
+
585
+ traces.column_names.include?("account_id") ? traces.where(account_id: account.id) : traces.for_account(account)
586
+ end
587
+
588
+ # The account the run's report was published under, else the one the
589
+ # agent belongs to.
590
+ def tenant_account
591
+ return @tenant_account if defined?(@tenant_account)
592
+
593
+ klass = ActionAgent.account_class.to_s.safe_constantize
594
+ @tenant_account =
595
+ if klass.nil?
596
+ nil
597
+ elsif @run.external_tenant.present?
598
+ klass.find_by(id: @run.external_tenant)
599
+ else
600
+ agent = @run.evaluation&.agent
601
+ owner = agent&.owner
602
+ if owner.is_a?(klass) then owner
603
+ elsif agent.respond_to?(:account_id) && agent.account_id.present? then klass.find_by(id: agent.account_id)
604
+ end
605
+ end
606
+ end
607
+ end
608
+ end
609
+ end
@@ -190,7 +190,9 @@ module ActionAgent
190
190
  # The cost here is the agent's: what the sampled interactions cost to
191
191
  # operate. It is not what this run spent — scoring recorded generations
192
192
  # costs nothing until a judge is asked, and the judge's own spend is
193
- # recorded apart, under "_judge_usage" (see #judge_usage).
193
+ # recorded apart, under "_judge_usage" (see #judge_usage). It sums the
194
+ # generations that could be priced, and "priced" counts them, so a
195
+ # cohort with some unpriced generations reads as partial.
194
196
 
195
197
  def cohort_summaries(samples, per_sample_scores)
196
198
  samples
@@ -207,7 +209,8 @@ module ActionAgent
207
209
  input_tokens = samples.sum { |generation| generation.input_tokens.to_i }
208
210
  output_tokens = samples.sum { |generation| generation.output_tokens.to_i }
209
211
  costs = samples.filter_map do |generation|
210
- ModelPricing.estimate(model: generation.model, input_tokens: generation.input_tokens, output_tokens: generation.output_tokens)
212
+ ModelPricing.estimate(model: generation.model, provider: generation.provider.presence, input_tokens: generation.input_tokens,
213
+ output_tokens: generation.output_tokens)
211
214
  end
212
215
 
213
216
  {
@@ -217,7 +220,8 @@ module ActionAgent
217
220
  "avg_duration_ms" => durations_ms.any? ? (durations_ms.sum / durations_ms.size).round : nil,
218
221
  "input_tokens" => input_tokens,
219
222
  "output_tokens" => output_tokens,
220
- "cost" => costs.any? ? costs.sum.round(6) : nil
223
+ "cost" => costs.any? ? costs.sum.round(6) : nil,
224
+ "priced" => costs.size
221
225
  }
222
226
  end
223
227
 
@@ -254,10 +258,10 @@ module ActionAgent
254
258
 
255
259
  def record_judge_call(kind, response)
256
260
  usage = response.respond_to?(:usage) ? response.usage : nil
257
- input_tokens = usage.respond_to?(:input_tokens) ? usage.input_tokens.to_i : 0
261
+ input_tokens = judge_input_tokens(usage)
258
262
  output_tokens = usage.respond_to?(:output_tokens) ? usage.output_tokens.to_i : 0
259
263
  model = (response.respond_to?(:model) && response.model.presence) || @evaluation.judge_model.presence
260
- cost = ModelPricing.estimate(model: model, input_tokens: input_tokens, output_tokens: output_tokens)
264
+ cost = ModelPricing.estimate(model: model, provider: judge_provider, input_tokens: input_tokens, output_tokens: output_tokens)
261
265
 
262
266
  @judge_usage ||= { "calls" => 0, "input_tokens" => 0, "output_tokens" => 0, "cost" => nil, "model" => nil, "by_kind" => {} }
263
267
  @judge_usage["calls"] += 1
@@ -268,6 +272,22 @@ module ActionAgent
268
272
  @judge_usage["by_kind"][kind.to_s] = @judge_usage["by_kind"].fetch(kind.to_s, 0) + 1
269
273
  end
270
274
 
275
+ # Every token the judge was billed for reading. Anthropic reports the
276
+ # tokens read from or written to the prompt cache apart from
277
+ # `input_tokens`, so they are added; OpenAI's prompt count already holds
278
+ # its cached tokens. Priced at the input rate, which overstates a cache
279
+ # read: the figure is an upper bound.
280
+ def judge_input_tokens(usage)
281
+ return 0 unless usage.respond_to?(:input_tokens)
282
+
283
+ total = usage.input_tokens.to_i
284
+ return total unless judge_provider == :anthropic
285
+
286
+ total += usage.cached_tokens.to_i if usage.respond_to?(:cached_tokens)
287
+ total += usage.cache_creation_tokens.to_i if usage.respond_to?(:cache_creation_tokens)
288
+ total
289
+ end
290
+
271
291
  def sample_generations(model: nil)
272
292
  scope = @evaluation.agent.generations
273
293
  scope = scope.where(model: model) if model