actionagent 1.8.0 → 1.8.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (32) hide show
  1. checksums.yaml +4 -4
  2. data/app/assets/builds/action_agent.js +58 -58
  3. data/app/controllers/action_agent/api/code_sessions_controller.rb +19 -7
  4. data/app/controllers/action_agent/api/evaluations_controller.rb +49 -3
  5. data/app/controllers/action_agent/api/sandboxes_controller.rb +8 -0
  6. data/app/controllers/concerns/action_agent/api/mcp_dashboard_tools.rb +4 -2
  7. data/app/jobs/action_agent/code_session_job.rb +13 -8
  8. data/app/models/action_agent/agent.rb +33 -0
  9. data/app/models/action_agent/code_session.rb +6 -2
  10. data/app/models/action_agent/evaluation.rb +64 -0
  11. data/app/models/action_agent/evaluation_run.rb +143 -53
  12. data/app/models/action_agent/evaluation_scenario_result.rb +15 -3
  13. data/app/models/action_agent/model_pricing.rb +214 -34
  14. data/app/models/action_agent/provider_key.rb +7 -1
  15. data/app/models/action_agent/sandbox_session.rb +3 -2
  16. data/app/serializers/action_agent/evaluation_serializer.rb +41 -10
  17. data/app/services/action_agent/agent_scorecard.rb +50 -21
  18. data/app/services/action_agent/codex_session_events.rb +37 -0
  19. data/app/services/action_agent/evaluation_report_import.rb +84 -5
  20. data/app/services/action_agent/evaluation_run_cost.rb +609 -0
  21. data/app/services/action_agent/evaluation_runner_service.rb +25 -5
  22. data/app/services/action_agent/evaluation_standing.rb +145 -0
  23. data/app/services/action_agent/local_sandbox_backend.rb +47 -21
  24. data/app/services/action_agent/sandbox_orchestrator.rb +14 -1
  25. data/app/services/action_agent/scenario_evaluation_runner.rb +2 -1
  26. data/config/routes.rb +1 -1
  27. data/lib/action_agent/version.rb +1 -1
  28. data/lib/action_agent.rb +6 -0
  29. data/lib/generators/action_agent/install_generator.rb +9 -6
  30. data/lib/generators/action_agent/templates/action_agent.rb.erb +3 -0
  31. data/lib/generators/action_agent/templates/add_code_session_runner.rb.erb +9 -0
  32. metadata +6 -2
@@ -12,19 +12,29 @@ module ActionAgent
12
12
 
13
13
  # Returns the evaluation with its latest run in full and just enough of
14
14
  # the run before it to show movement ("+3 passed vs #2") without a
15
- # request per evaluation.
15
+ # request per evaluation. The headline run — the newest complete one —
16
+ # rides along in full as `headline_run` when a newer run is pending or
17
+ # failed, so a page can describe the finished run while showing the
18
+ # newer one beside it.
16
19
  def evaluation(evaluation)
17
20
  summary = summary(evaluation)
18
21
  latest, previous = recent_runs(evaluation, 2)
22
+ headline = evaluation.standing_info.headline_run
23
+ headline = nil if headline.nil? || headline.id == latest&.id
19
24
 
20
25
  summary.merge(
21
26
  latest_run: latest ? run(latest, number: summary[:run_count]) : nil,
22
- previous_run: previous ? run_summary(previous, number: summary[:run_count] - 1) : nil
27
+ previous_run: previous ? run_summary(previous, number: summary[:run_count] - 1) : nil,
28
+ headline_run: headline ? run(headline, number: run_number(evaluation, headline)) : nil
23
29
  )
24
30
  end
25
31
 
26
- # Returns the evaluation's configuration and run count, without its runs.
32
+ # Returns the evaluation's configuration and run count, without its runs,
33
+ # and where it stands: `headline_run_id` (its newest complete run),
34
+ # `standing` against the agent's current version (EvaluationStanding),
35
+ # `archived_at`, and the headline run's passes `per_model`.
27
36
  def summary(evaluation)
37
+ standing = evaluation.standing_info
28
38
  {
29
39
  id: evaluation.id,
30
40
  name: evaluation.name,
@@ -40,7 +50,11 @@ module ActionAgent
40
50
  scenario_groups: evaluation.scenario_suite? ? evaluation.scenario_groups : [],
41
51
  created_at: evaluation.created_at.iso8601,
42
52
  # size reads a preloaded association and COUNTs otherwise.
43
- run_count: evaluation.evaluation_runs.size
53
+ run_count: evaluation.evaluation_runs.size,
54
+ headline_run_id: standing.headline_run&.id,
55
+ standing: standing.standing,
56
+ archived_at: evaluation.archived_at&.iso8601,
57
+ per_model: standing.per_model
44
58
  }
45
59
  end
46
60
 
@@ -48,7 +62,7 @@ module ActionAgent
48
62
  # error.
49
63
  def run(run, number: nil)
50
64
  run_summary(run, number: number).merge(
51
- scores: run.scores,
65
+ scores: scores(run),
52
66
  selection: run.selection,
53
67
  models: run.models,
54
68
  usage: run.usage,
@@ -56,9 +70,18 @@ module ActionAgent
56
70
  )
57
71
  end
58
72
 
73
+ # Returns the run's scores with each scenario model summary carrying its
74
+ # "priced" count (EvaluationRun#model_summaries).
75
+ def scores(run)
76
+ summaries = run.model_summaries
77
+ summaries.empty? ? run.scores : run.scores.merge("_models" => summaries)
78
+ end
79
+
59
80
  # Returns the run's status and headline numbers. `sandbox` is the
60
81
  # checkout sandbox the run replayed against, if any: its session id and
61
- # checkout, never its token.
82
+ # checkout, never its token. `agent_version` is the version of the
83
+ # agent the run scored, and `version_state` whether that is the agent's
84
+ # current version ("current", "earlier" or "unrecorded").
62
85
  def run_summary(run, number: nil)
63
86
  {
64
87
  id: run.id,
@@ -69,7 +92,9 @@ module ActionAgent
69
92
  samples_passed: run.samples_passed,
70
93
  completed_at: run.completed_at&.iso8601,
71
94
  created_at: run.created_at.iso8601,
72
- sandbox: run.sandbox
95
+ sandbox: run.sandbox,
96
+ agent_version: run.agent_version_summary,
97
+ version_state: run.evaluation.standing_info.version_state(run)
73
98
  }
74
99
  end
75
100
 
@@ -81,13 +106,19 @@ module ActionAgent
81
106
  if runs.loaded?
82
107
  runs.sort_by { |run| [ run.created_at, run.id ] }.reverse.first(limit)
83
108
  else
84
- runs.recent.limit(limit).to_a
109
+ runs.order(created_at: :desc, id: :desc).limit(limit).to_a
85
110
  end
86
111
  end
87
112
 
88
- # Returns the run's position in its evaluation's history, oldest = 1.
113
+ # Returns the run's position in its evaluation's history, oldest = 1 —
114
+ # counted in the preloaded association when one was loaded.
89
115
  def run_number(evaluation, run)
90
- evaluation.evaluation_runs.where("created_at < ? OR (created_at = ? AND id <= ?)", run.created_at, run.created_at, run.id).count
116
+ runs = evaluation.evaluation_runs
117
+ if runs.loaded?
118
+ runs.count { |other| other.created_at < run.created_at || (other.created_at == run.created_at && other.id <= run.id) }
119
+ else
120
+ runs.where("created_at < ? OR (created_at = ? AND id <= ?)", run.created_at, run.created_at, run.id).count
121
+ end
91
122
  end
92
123
 
93
124
  # Returns the run's fix items, or [] when they cannot be built. They are
@@ -32,7 +32,7 @@ module ActionAgent
32
32
  avg_durations = windowed.where.not(duration_ms: nil).group(:agent_id).average(:duration_ms)
33
33
  token_sums = windowed.group(:agent_id).sum("COALESCE(total_tokens, 0)")
34
34
  last_runs = AgentRun.where(agent_id: ids).group(:agent_id).maximum(:created_at)
35
- eval_runs = latest_evaluation_runs(ids)
35
+ evals = evaluation_tiles(ids)
36
36
 
37
37
  trace_stats = telemetry_stats(ids, window_start)
38
38
  trace_last = unclaimed_traces(ids, nil).group(:agent_id).maximum(:timestamp)
@@ -44,7 +44,7 @@ module ActionAgent
44
44
  traced = stats[:count].to_i
45
45
  total = runs + traced
46
46
  succeeded = completed_counts[id].to_i + stats[:ok].to_i
47
- eval_run = eval_runs[id]
47
+ eval_tile = evals[id] || {}
48
48
 
49
49
  {
50
50
  window_days: (WINDOW / 1.day).to_i,
@@ -55,9 +55,13 @@ module ActionAgent
55
55
  avg_duration_ms: blended_duration(avg_durations[id], runs, stats[:avg_duration], traced),
56
56
  tokens: token_sums[id].to_i + stats[:tokens].to_i,
57
57
  cost: costs[id],
58
- eval_score: eval_run&.average_score,
59
- eval_samples_passed: eval_run&.samples_passed,
60
- eval_samples_evaluated: eval_run&.samples_evaluated,
58
+ # The evaluation tile: the pass rate pooled over the headline runs
59
+ # of the agent's evaluations that stand against its current version.
60
+ eval_score: eval_tile[:score],
61
+ eval_samples_passed: eval_tile[:passed],
62
+ eval_samples_evaluated: eval_tile[:evaluated],
63
+ eval_runs: eval_tile[:runs],
64
+ eval_not_counted: eval_tile[:not_counted],
61
65
  last_run_at: [ last_runs[id], trace_last[id] ].compact.max&.iso8601
62
66
  }
63
67
  end
@@ -170,22 +174,47 @@ module ActionAgent
170
174
  end
171
175
  private_class_method :blended_duration
172
176
 
173
- # Latest complete evaluation run per agent. DISTINCT ON would do this in
174
- # one pass on PostgreSQL, but it has no portable equivalent, so the
175
- # highest id per agent (runs are only ever appended) is selected first
176
- # and those rows fetched by id.
177
- def self.latest_evaluation_runs(agent_ids)
178
- evaluations = Evaluation.table_name
179
- runs = EvaluationRun.table_name
180
-
181
- newest = EvaluationRun.complete.joins(:evaluation)
182
- .where(evaluations => { agent_id: agent_ids })
183
- .group("#{evaluations}.agent_id")
184
- .pluck(Arel.sql("#{evaluations}.agent_id"), Arel.sql("MAX(#{runs}.id)"))
185
-
186
- by_run_id = newest.to_h { |agent_id, run_id| [ run_id, agent_id ] }
187
- EvaluationRun.where(id: by_run_id.keys).index_by { |run| by_run_id[run.id] }
177
+ # The evaluation tile per agent: passes pooled over the headline runs
178
+ # (newest complete run) of its evaluations that are current against the
179
+ # agent's version or unrecorded, as the Evaluations page pools them —
180
+ # never a stale suite last run against older code, nor an archived one
181
+ # — so the tile is the pass rate of the agent as it is now, not a mean
182
+ # of criterion scores over whatever ran last.
183
+ #
184
+ # { agent_id => { score:, passed:, evaluated:, runs:, not_counted: } }
185
+ #
186
+ # `score` is the pooled pass rate 0..1 (nil with nothing to count),
187
+ # `runs` the headline runs pooled and `not_counted` the evaluations
188
+ # left out as stale or archived. The newest complete run per evaluation
189
+ # is selected by id first (runs are only ever appended) and those rows
190
+ # fetched by id, which runs the same on every database.
191
+ def self.evaluation_tiles(agent_ids)
192
+ evaluations = Evaluation.where(agent_id: agent_ids).includes(:agent).to_a
193
+ return {} if evaluations.empty?
194
+
195
+ newest = EvaluationRun.complete.where(evaluation_id: evaluations.map(&:id)).group(:evaluation_id).maximum(:id)
196
+ runs = EvaluationRun.where(id: newest.values).includes(:agent_version).index_by(&:evaluation_id)
197
+ EvaluationStanding.preload(evaluations)
198
+
199
+ evaluations.group_by(&:agent_id).to_h do |agent_id, mine|
200
+ tile = { score: nil, passed: 0, evaluated: 0, runs: 0, not_counted: 0 }
201
+ mine.each do |evaluation|
202
+ run = runs[evaluation.id]
203
+ next unless run
204
+
205
+ unless evaluation.standing_info.with_headline(run).counted?
206
+ tile[:not_counted] += 1
207
+ next
208
+ end
209
+
210
+ tile[:runs] += 1
211
+ tile[:passed] += run.samples_passed.to_i
212
+ tile[:evaluated] += run.samples_evaluated.to_i
213
+ end
214
+ tile[:score] = (tile[:passed].to_f / tile[:evaluated]).round(3) if tile[:evaluated].positive?
215
+ [ agent_id, tile ]
216
+ end
188
217
  end
189
- private_class_method :latest_evaluation_runs
218
+ private_class_method :evaluation_tiles
190
219
  end
191
220
  end
@@ -0,0 +1,37 @@
1
+ # frozen_string_literal: true
2
+
3
+ module ActionAgent
4
+ # Keep Codex's native JSONL transcript while adapting its terminal event
5
+ # to CodeSession's shared result contract. A process exit alone is never
6
+ # enough to mark a coding session successful.
7
+ class CodexSessionEvents
8
+ def consume(event)
9
+ return unless event.is_a?(Hash)
10
+
11
+ case event["type"]
12
+ when "thread.started"
13
+ @thread_id = event["thread_id"]
14
+ when "item.completed"
15
+ item = event["item"]
16
+ @answer = item["text"] if item.is_a?(Hash) && item["type"] == "agent_message"
17
+ when "turn.completed"
18
+ return result(false, @answer, event["usage"])
19
+ when "turn.failed", "error"
20
+ error = event["error"]
21
+ message = error.is_a?(Hash) ? error["message"] : event["message"]
22
+ return result(true, message.presence || "Codex reported a failed turn", {})
23
+ end
24
+ nil
25
+ end
26
+
27
+ private
28
+
29
+ def result(failed, answer, usage)
30
+ {
31
+ "type" => "result", "subtype" => failed ? "error_during_execution" : "success",
32
+ "is_error" => failed, "result" => answer, "session_id" => @thread_id,
33
+ "usage" => usage.is_a?(Hash) ? usage.slice("input_tokens", "output_tokens") : {}
34
+ }
35
+ end
36
+ end
37
+ end
@@ -22,7 +22,12 @@ module ActionAgent
22
22
  #
23
23
  # The run's per-model summary, criterion scores and recommendations are
24
24
  # computed from the stored results, not taken from the report. The judge's
25
- # verdict and label are taken from it.
25
+ # verdict and label are taken from it, and so is what the judge spent —
26
+ # per result (`result.judge_usage`, kept beside the result's diagnosis) and
27
+ # for the run (`report.judge_usage.run`) — and the release the report says
28
+ # it evaluated (`report.release`), which pins the run to that version of
29
+ # the agent (Agent#find_or_record_release!). A report naming no release
30
+ # leaves the run unrecorded rather than claiming the dashboard's latest.
26
31
  #
27
32
  # An identical retry returns the stored run, before anything is asked of the
28
33
  # `admit` callable. Different content under a run_id already stored raises
@@ -70,6 +75,11 @@ module ActionAgent
70
75
  IDENTIFIER_PATTERN = /\A[^[:cntrl:]]{1,200}\z/
71
76
  AGENT_NAME_PATTERN = /\A[^[:cntrl:]]{2,100}\z/
72
77
  TRACE_PATTERN = /\A[a-zA-Z0-9_-]{1,128}\z/
78
+ DIGEST_PATTERN = /\A[a-zA-Z0-9._-]{1,64}\z/
79
+ # The judge usage keys a report may send, per result and for the run.
80
+ JUDGE_USAGE_KEYS = %w[calls input_tokens output_tokens cost model by_kind source].freeze
81
+ JUDGE_USAGE_SOURCES = %w[reported traces meter].freeze
82
+ MAX_JUDGE_KINDS = 10
73
83
  SCOPE_PATTERN = %r{\A[\w .:/@-]{1,100}\z}
74
84
  STATUSES = %w[passed failed errored].freeze
75
85
  # Report metadata that tells one evaluation of a suite from another, in the
@@ -253,6 +263,7 @@ module ActionAgent
253
263
  external_tenant: tenant_key,
254
264
  external_run_id: @payload["run_id"],
255
265
  external_report_digest: digest,
266
+ agent_version: reported_release_version(agent),
256
267
  status: :complete,
257
268
  selection: selection,
258
269
  scores: recorded_scores,
@@ -267,6 +278,15 @@ module ActionAgent
267
278
  run
268
279
  end
269
280
 
281
+ # The agent version for the release the report names, or nil for a
282
+ # report that names none.
283
+ def reported_release_version(agent)
284
+ release = report["release"]
285
+ return nil unless release.is_a?(Hash) && release["digest"].present?
286
+
287
+ agent.find_or_record_release!(digest: release["digest"], revision: release["revision"].presence, label: release["label"].presence)
288
+ end
289
+
270
290
  # --- ownership -------------------------------------------------------------
271
291
 
272
292
  # Who a newly observed agent belongs to. Whatever the host's
@@ -492,11 +512,28 @@ module ActionAgent
492
512
  "key" => result["scenario_key"], "group" => result["group"], "prompt" => result["prompt"],
493
513
  "position" => scenario.position, "expectations" => {}
494
514
  }
495
- ),
515
+ ).merge(bounded_judge_usage(result["judge_usage"]).then { |usage| usage ? { "_judge_usage" => usage } : {} }),
496
516
  error_message: truncated(result["error"], TEXT_BYTES)
497
517
  )
498
518
  end
499
519
 
520
+ # A judge usage as the engine stores it: the keys the dashboard reads,
521
+ # each cut to its type, or nil for none.
522
+ def bounded_judge_usage(usage)
523
+ return nil unless usage.is_a?(Hash)
524
+
525
+ by_kind = usage["by_kind"].is_a?(Hash) ? usage["by_kind"].first(MAX_JUDGE_KINDS).to_h { |kind, count| [ kind.to_s.first(40), count.to_i ] } : {}
526
+ {
527
+ "calls" => usage["calls"].to_i,
528
+ "input_tokens" => usage["input_tokens"].to_i,
529
+ "output_tokens" => usage["output_tokens"].to_i,
530
+ "cost" => usage["cost"].is_a?(Numeric) ? usage["cost"].to_f : nil,
531
+ "model" => usage["model"].is_a?(String) ? usage["model"].first(200).presence : nil,
532
+ "by_kind" => by_kind,
533
+ "source" => JUDGE_USAGE_SOURCES.include?(usage["source"]) ? usage["source"] : "reported"
534
+ }
535
+ end
536
+
500
537
  # The first +bytes+ bytes of +text+, dropping a character the cut splits,
501
538
  # or nil for blank text.
502
539
  def truncated(text, bytes)
@@ -521,14 +558,18 @@ module ActionAgent
521
558
  {
522
559
  "_verdict" => report["verdict"],
523
560
  "_selection" => selection,
524
- "_metadata" => metadata
561
+ "_metadata" => metadata,
562
+ "_judge_usage_run" => bounded_judge_usage(report.dig("judge_usage", "run"))
525
563
  }.compact.merge("_judge_label" => judge_label)
526
564
  end
527
565
 
528
566
  # The run's scores in the shape the Evaluations view renders
529
- # (ScenarioEvaluationRunner#scores_for), summarized from the stored results.
567
+ # (ScenarioEvaluationRunner#scores_for), summarized from the stored
568
+ # results as the application reported them — no cost is estimated here,
569
+ # so the summaries stored stay the application's own figures; the API
570
+ # adds the estimates when it serves the run (EvaluationRun#model_summaries).
530
571
  def summarized_scores(run)
531
- rebuilt = run.to_report
572
+ rebuilt = run.to_report(estimate: false)
532
573
  rebuilt.criterion_scores.merge(
533
574
  "_models" => rebuilt.summary_by_model,
534
575
  "_recommendations" => rebuilt.recommendations
@@ -559,6 +600,9 @@ module ActionAgent
559
600
 
560
601
  judge_trace_ids!(metadata["judge_trace_ids"])
561
602
  string!(report["judge"], "report.judge", 200)
603
+ validate_release!
604
+ validate_judge_usage!(report.dig("judge_usage", "run"), "report.judge_usage.run") if report["judge_usage"].is_a?(Hash)
605
+ optional_object!(report["judge_usage"], "report.judge_usage")
562
606
  object!(report["models"], "report.models")
563
607
  raise Invalid, "report.models must contain 1-#{MAX_MODELS} models" unless report["models"].size.between?(1, MAX_MODELS)
564
608
  unless results.is_a?(Array) && results.size.between?(1, MAX_RESULTS)
@@ -605,6 +649,7 @@ module ActionAgent
605
649
  end
606
650
  trace_id!(result_metadata["trace_id"]) if result_metadata["trace_id"]
607
651
  judge_trace_ids!(result_metadata["judge_trace_ids"])
652
+ validate_judge_usage!(result["judge_usage"], "result.judge_usage")
608
653
  end
609
654
  distinct_specs = label_specs.values.uniq
610
655
  raise Invalid, "two model labels name the same provider/model" if distinct_specs.size < label_specs.size
@@ -655,6 +700,40 @@ module ActionAgent
655
700
  string!(judge.dig("suggested_tool", "name"), "result.diagnosis.judge.suggested_tool.name", 200)
656
701
  end
657
702
 
703
+ # The release the report says it evaluated: a digest the dashboard can
704
+ # match to a version cut on deploy, and the deploy and label as text.
705
+ def validate_release!
706
+ release = report["release"]
707
+ return if release.nil?
708
+
709
+ object!(release, "report.release")
710
+ unless release["digest"].nil? || (release["digest"].is_a?(String) && DIGEST_PATTERN.match?(release["digest"]))
711
+ raise Invalid, "report.release.digest must be 1-64 letters, digits or . _ -"
712
+ end
713
+
714
+ string!(release["revision"], "report.release.revision", 200)
715
+ string!(release["label"], "report.release.label", 200)
716
+ end
717
+
718
+ # What the judge spent, as ActiveAgent::Evals::Report writes it per
719
+ # result and for the run.
720
+ def validate_judge_usage!(usage, name)
721
+ return if usage.nil?
722
+
723
+ object!(usage, name)
724
+ %w[calls input_tokens output_tokens].each do |key|
725
+ next if usage[key].nil?
726
+ raise Invalid, "#{name}.#{key} must be a non-negative integer" unless usage[key].is_a?(Integer) && usage[key] >= 0 && usage[key] <= NUMERIC_LIMITS["input_tokens"]
727
+ end
728
+ numeric!(usage["cost"], "#{name}.cost", max: NUMERIC_LIMITS["cost"])
729
+ string!(usage["model"], "#{name}.model", 200)
730
+ string!(usage["source"], "#{name}.source", 20)
731
+ optional_object!(usage["by_kind"], "#{name}.by_kind")
732
+ (usage["by_kind"] || {}).each_value do |count|
733
+ raise Invalid, "#{name}.by_kind counts must be non-negative integers" unless count.is_a?(Integer) && count >= 0
734
+ end
735
+ end
736
+
658
737
  def validate_verdict!
659
738
  verdict = report["verdict"]
660
739
  return if verdict.nil?