actionagent 1.8.0 → 1.8.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/app/assets/builds/action_agent.js +58 -58
- data/app/controllers/action_agent/api/code_sessions_controller.rb +19 -7
- data/app/controllers/action_agent/api/evaluations_controller.rb +49 -3
- data/app/controllers/action_agent/api/sandboxes_controller.rb +8 -0
- data/app/controllers/concerns/action_agent/api/mcp_dashboard_tools.rb +4 -2
- data/app/jobs/action_agent/code_session_job.rb +13 -8
- data/app/models/action_agent/agent.rb +33 -0
- data/app/models/action_agent/code_session.rb +6 -2
- data/app/models/action_agent/evaluation.rb +64 -0
- data/app/models/action_agent/evaluation_run.rb +143 -53
- data/app/models/action_agent/evaluation_scenario_result.rb +15 -3
- data/app/models/action_agent/model_pricing.rb +214 -34
- data/app/models/action_agent/provider_key.rb +7 -1
- data/app/models/action_agent/sandbox_session.rb +3 -2
- data/app/serializers/action_agent/evaluation_serializer.rb +41 -10
- data/app/services/action_agent/agent_scorecard.rb +50 -21
- data/app/services/action_agent/codex_session_events.rb +37 -0
- data/app/services/action_agent/evaluation_report_import.rb +84 -5
- data/app/services/action_agent/evaluation_run_cost.rb +609 -0
- data/app/services/action_agent/evaluation_runner_service.rb +25 -5
- data/app/services/action_agent/evaluation_standing.rb +145 -0
- data/app/services/action_agent/local_sandbox_backend.rb +47 -21
- data/app/services/action_agent/sandbox_orchestrator.rb +14 -1
- data/app/services/action_agent/scenario_evaluation_runner.rb +2 -1
- data/config/routes.rb +1 -1
- data/lib/action_agent/version.rb +1 -1
- data/lib/action_agent.rb +6 -0
- data/lib/generators/action_agent/install_generator.rb +9 -6
- data/lib/generators/action_agent/templates/action_agent.rb.erb +3 -0
- data/lib/generators/action_agent/templates/add_code_session_runner.rb.erb +9 -0
- metadata +6 -2
|
@@ -12,19 +12,29 @@ module ActionAgent
|
|
|
12
12
|
|
|
13
13
|
# Returns the evaluation with its latest run in full and just enough of
|
|
14
14
|
# the run before it to show movement ("+3 passed vs #2") without a
|
|
15
|
-
# request per evaluation.
|
|
15
|
+
# request per evaluation. The headline run — the newest complete one —
|
|
16
|
+
# rides along in full as `headline_run` when a newer run is pending or
|
|
17
|
+
# failed, so a page can describe the finished run while showing the
|
|
18
|
+
# newer one beside it.
|
|
16
19
|
def evaluation(evaluation)
|
|
17
20
|
summary = summary(evaluation)
|
|
18
21
|
latest, previous = recent_runs(evaluation, 2)
|
|
22
|
+
headline = evaluation.standing_info.headline_run
|
|
23
|
+
headline = nil if headline.nil? || headline.id == latest&.id
|
|
19
24
|
|
|
20
25
|
summary.merge(
|
|
21
26
|
latest_run: latest ? run(latest, number: summary[:run_count]) : nil,
|
|
22
|
-
previous_run: previous ? run_summary(previous, number: summary[:run_count] - 1) : nil
|
|
27
|
+
previous_run: previous ? run_summary(previous, number: summary[:run_count] - 1) : nil,
|
|
28
|
+
headline_run: headline ? run(headline, number: run_number(evaluation, headline)) : nil
|
|
23
29
|
)
|
|
24
30
|
end
|
|
25
31
|
|
|
26
|
-
# Returns the evaluation's configuration and run count, without its runs
|
|
32
|
+
# Returns the evaluation's configuration and run count, without its runs,
|
|
33
|
+
# and where it stands: `headline_run_id` (its newest complete run),
|
|
34
|
+
# `standing` against the agent's current version (EvaluationStanding),
|
|
35
|
+
# `archived_at`, and the headline run's passes `per_model`.
|
|
27
36
|
def summary(evaluation)
|
|
37
|
+
standing = evaluation.standing_info
|
|
28
38
|
{
|
|
29
39
|
id: evaluation.id,
|
|
30
40
|
name: evaluation.name,
|
|
@@ -40,7 +50,11 @@ module ActionAgent
|
|
|
40
50
|
scenario_groups: evaluation.scenario_suite? ? evaluation.scenario_groups : [],
|
|
41
51
|
created_at: evaluation.created_at.iso8601,
|
|
42
52
|
# size reads a preloaded association and COUNTs otherwise.
|
|
43
|
-
run_count: evaluation.evaluation_runs.size
|
|
53
|
+
run_count: evaluation.evaluation_runs.size,
|
|
54
|
+
headline_run_id: standing.headline_run&.id,
|
|
55
|
+
standing: standing.standing,
|
|
56
|
+
archived_at: evaluation.archived_at&.iso8601,
|
|
57
|
+
per_model: standing.per_model
|
|
44
58
|
}
|
|
45
59
|
end
|
|
46
60
|
|
|
@@ -48,7 +62,7 @@ module ActionAgent
|
|
|
48
62
|
# error.
|
|
49
63
|
def run(run, number: nil)
|
|
50
64
|
run_summary(run, number: number).merge(
|
|
51
|
-
scores: run
|
|
65
|
+
scores: scores(run),
|
|
52
66
|
selection: run.selection,
|
|
53
67
|
models: run.models,
|
|
54
68
|
usage: run.usage,
|
|
@@ -56,9 +70,18 @@ module ActionAgent
|
|
|
56
70
|
)
|
|
57
71
|
end
|
|
58
72
|
|
|
73
|
+
# Returns the run's scores with each scenario model summary carrying its
|
|
74
|
+
# "priced" count (EvaluationRun#model_summaries).
|
|
75
|
+
def scores(run)
|
|
76
|
+
summaries = run.model_summaries
|
|
77
|
+
summaries.empty? ? run.scores : run.scores.merge("_models" => summaries)
|
|
78
|
+
end
|
|
79
|
+
|
|
59
80
|
# Returns the run's status and headline numbers. `sandbox` is the
|
|
60
81
|
# checkout sandbox the run replayed against, if any: its session id and
|
|
61
|
-
# checkout, never its token.
|
|
82
|
+
# checkout, never its token. `agent_version` is the version of the
|
|
83
|
+
# agent the run scored, and `version_state` whether that is the agent's
|
|
84
|
+
# current version ("current", "earlier" or "unrecorded").
|
|
62
85
|
def run_summary(run, number: nil)
|
|
63
86
|
{
|
|
64
87
|
id: run.id,
|
|
@@ -69,7 +92,9 @@ module ActionAgent
|
|
|
69
92
|
samples_passed: run.samples_passed,
|
|
70
93
|
completed_at: run.completed_at&.iso8601,
|
|
71
94
|
created_at: run.created_at.iso8601,
|
|
72
|
-
sandbox: run.sandbox
|
|
95
|
+
sandbox: run.sandbox,
|
|
96
|
+
agent_version: run.agent_version_summary,
|
|
97
|
+
version_state: run.evaluation.standing_info.version_state(run)
|
|
73
98
|
}
|
|
74
99
|
end
|
|
75
100
|
|
|
@@ -81,13 +106,19 @@ module ActionAgent
|
|
|
81
106
|
if runs.loaded?
|
|
82
107
|
runs.sort_by { |run| [ run.created_at, run.id ] }.reverse.first(limit)
|
|
83
108
|
else
|
|
84
|
-
runs.
|
|
109
|
+
runs.order(created_at: :desc, id: :desc).limit(limit).to_a
|
|
85
110
|
end
|
|
86
111
|
end
|
|
87
112
|
|
|
88
|
-
# Returns the run's position in its evaluation's history, oldest = 1
|
|
113
|
+
# Returns the run's position in its evaluation's history, oldest = 1 —
|
|
114
|
+
# counted in the preloaded association when one was loaded.
|
|
89
115
|
def run_number(evaluation, run)
|
|
90
|
-
|
|
116
|
+
runs = evaluation.evaluation_runs
|
|
117
|
+
if runs.loaded?
|
|
118
|
+
runs.count { |other| other.created_at < run.created_at || (other.created_at == run.created_at && other.id <= run.id) }
|
|
119
|
+
else
|
|
120
|
+
runs.where("created_at < ? OR (created_at = ? AND id <= ?)", run.created_at, run.created_at, run.id).count
|
|
121
|
+
end
|
|
91
122
|
end
|
|
92
123
|
|
|
93
124
|
# Returns the run's fix items, or [] when they cannot be built. They are
|
|
@@ -32,7 +32,7 @@ module ActionAgent
|
|
|
32
32
|
avg_durations = windowed.where.not(duration_ms: nil).group(:agent_id).average(:duration_ms)
|
|
33
33
|
token_sums = windowed.group(:agent_id).sum("COALESCE(total_tokens, 0)")
|
|
34
34
|
last_runs = AgentRun.where(agent_id: ids).group(:agent_id).maximum(:created_at)
|
|
35
|
-
|
|
35
|
+
evals = evaluation_tiles(ids)
|
|
36
36
|
|
|
37
37
|
trace_stats = telemetry_stats(ids, window_start)
|
|
38
38
|
trace_last = unclaimed_traces(ids, nil).group(:agent_id).maximum(:timestamp)
|
|
@@ -44,7 +44,7 @@ module ActionAgent
|
|
|
44
44
|
traced = stats[:count].to_i
|
|
45
45
|
total = runs + traced
|
|
46
46
|
succeeded = completed_counts[id].to_i + stats[:ok].to_i
|
|
47
|
-
|
|
47
|
+
eval_tile = evals[id] || {}
|
|
48
48
|
|
|
49
49
|
{
|
|
50
50
|
window_days: (WINDOW / 1.day).to_i,
|
|
@@ -55,9 +55,13 @@ module ActionAgent
|
|
|
55
55
|
avg_duration_ms: blended_duration(avg_durations[id], runs, stats[:avg_duration], traced),
|
|
56
56
|
tokens: token_sums[id].to_i + stats[:tokens].to_i,
|
|
57
57
|
cost: costs[id],
|
|
58
|
-
|
|
59
|
-
|
|
60
|
-
|
|
58
|
+
# The evaluation tile: the pass rate pooled over the headline runs
|
|
59
|
+
# of the agent's evaluations that stand against its current version.
|
|
60
|
+
eval_score: eval_tile[:score],
|
|
61
|
+
eval_samples_passed: eval_tile[:passed],
|
|
62
|
+
eval_samples_evaluated: eval_tile[:evaluated],
|
|
63
|
+
eval_runs: eval_tile[:runs],
|
|
64
|
+
eval_not_counted: eval_tile[:not_counted],
|
|
61
65
|
last_run_at: [ last_runs[id], trace_last[id] ].compact.max&.iso8601
|
|
62
66
|
}
|
|
63
67
|
end
|
|
@@ -170,22 +174,47 @@ module ActionAgent
|
|
|
170
174
|
end
|
|
171
175
|
private_class_method :blended_duration
|
|
172
176
|
|
|
173
|
-
#
|
|
174
|
-
#
|
|
175
|
-
#
|
|
176
|
-
#
|
|
177
|
-
|
|
178
|
-
|
|
179
|
-
|
|
180
|
-
|
|
181
|
-
|
|
182
|
-
|
|
183
|
-
|
|
184
|
-
|
|
185
|
-
|
|
186
|
-
|
|
187
|
-
|
|
177
|
+
# The evaluation tile per agent: passes pooled over the headline runs
|
|
178
|
+
# (newest complete run) of its evaluations that are current against the
|
|
179
|
+
# agent's version or unrecorded, as the Evaluations page pools them —
|
|
180
|
+
# never a stale suite last run against older code, nor an archived one
|
|
181
|
+
# — so the tile is the pass rate of the agent as it is now, not a mean
|
|
182
|
+
# of criterion scores over whatever ran last.
|
|
183
|
+
#
|
|
184
|
+
# { agent_id => { score:, passed:, evaluated:, runs:, not_counted: } }
|
|
185
|
+
#
|
|
186
|
+
# `score` is the pooled pass rate 0..1 (nil with nothing to count),
|
|
187
|
+
# `runs` the headline runs pooled and `not_counted` the evaluations
|
|
188
|
+
# left out as stale or archived. The newest complete run per evaluation
|
|
189
|
+
# is selected by id first (runs are only ever appended) and those rows
|
|
190
|
+
# fetched by id, which runs the same on every database.
|
|
191
|
+
def self.evaluation_tiles(agent_ids)
|
|
192
|
+
evaluations = Evaluation.where(agent_id: agent_ids).includes(:agent).to_a
|
|
193
|
+
return {} if evaluations.empty?
|
|
194
|
+
|
|
195
|
+
newest = EvaluationRun.complete.where(evaluation_id: evaluations.map(&:id)).group(:evaluation_id).maximum(:id)
|
|
196
|
+
runs = EvaluationRun.where(id: newest.values).includes(:agent_version).index_by(&:evaluation_id)
|
|
197
|
+
EvaluationStanding.preload(evaluations)
|
|
198
|
+
|
|
199
|
+
evaluations.group_by(&:agent_id).to_h do |agent_id, mine|
|
|
200
|
+
tile = { score: nil, passed: 0, evaluated: 0, runs: 0, not_counted: 0 }
|
|
201
|
+
mine.each do |evaluation|
|
|
202
|
+
run = runs[evaluation.id]
|
|
203
|
+
next unless run
|
|
204
|
+
|
|
205
|
+
unless evaluation.standing_info.with_headline(run).counted?
|
|
206
|
+
tile[:not_counted] += 1
|
|
207
|
+
next
|
|
208
|
+
end
|
|
209
|
+
|
|
210
|
+
tile[:runs] += 1
|
|
211
|
+
tile[:passed] += run.samples_passed.to_i
|
|
212
|
+
tile[:evaluated] += run.samples_evaluated.to_i
|
|
213
|
+
end
|
|
214
|
+
tile[:score] = (tile[:passed].to_f / tile[:evaluated]).round(3) if tile[:evaluated].positive?
|
|
215
|
+
[ agent_id, tile ]
|
|
216
|
+
end
|
|
188
217
|
end
|
|
189
|
-
private_class_method :
|
|
218
|
+
private_class_method :evaluation_tiles
|
|
190
219
|
end
|
|
191
220
|
end
|
|
@@ -0,0 +1,37 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module ActionAgent
|
|
4
|
+
# Keep Codex's native JSONL transcript while adapting its terminal event
|
|
5
|
+
# to CodeSession's shared result contract. A process exit alone is never
|
|
6
|
+
# enough to mark a coding session successful.
|
|
7
|
+
class CodexSessionEvents
|
|
8
|
+
def consume(event)
|
|
9
|
+
return unless event.is_a?(Hash)
|
|
10
|
+
|
|
11
|
+
case event["type"]
|
|
12
|
+
when "thread.started"
|
|
13
|
+
@thread_id = event["thread_id"]
|
|
14
|
+
when "item.completed"
|
|
15
|
+
item = event["item"]
|
|
16
|
+
@answer = item["text"] if item.is_a?(Hash) && item["type"] == "agent_message"
|
|
17
|
+
when "turn.completed"
|
|
18
|
+
return result(false, @answer, event["usage"])
|
|
19
|
+
when "turn.failed", "error"
|
|
20
|
+
error = event["error"]
|
|
21
|
+
message = error.is_a?(Hash) ? error["message"] : event["message"]
|
|
22
|
+
return result(true, message.presence || "Codex reported a failed turn", {})
|
|
23
|
+
end
|
|
24
|
+
nil
|
|
25
|
+
end
|
|
26
|
+
|
|
27
|
+
private
|
|
28
|
+
|
|
29
|
+
def result(failed, answer, usage)
|
|
30
|
+
{
|
|
31
|
+
"type" => "result", "subtype" => failed ? "error_during_execution" : "success",
|
|
32
|
+
"is_error" => failed, "result" => answer, "session_id" => @thread_id,
|
|
33
|
+
"usage" => usage.is_a?(Hash) ? usage.slice("input_tokens", "output_tokens") : {}
|
|
34
|
+
}
|
|
35
|
+
end
|
|
36
|
+
end
|
|
37
|
+
end
|
|
@@ -22,7 +22,12 @@ module ActionAgent
|
|
|
22
22
|
#
|
|
23
23
|
# The run's per-model summary, criterion scores and recommendations are
|
|
24
24
|
# computed from the stored results, not taken from the report. The judge's
|
|
25
|
-
# verdict and label are taken from it
|
|
25
|
+
# verdict and label are taken from it, and so is what the judge spent —
|
|
26
|
+
# per result (`result.judge_usage`, kept beside the result's diagnosis) and
|
|
27
|
+
# for the run (`report.judge_usage.run`) — and the release the report says
|
|
28
|
+
# it evaluated (`report.release`), which pins the run to that version of
|
|
29
|
+
# the agent (Agent#find_or_record_release!). A report naming no release
|
|
30
|
+
# leaves the run unrecorded rather than claiming the dashboard's latest.
|
|
26
31
|
#
|
|
27
32
|
# An identical retry returns the stored run, before anything is asked of the
|
|
28
33
|
# `admit` callable. Different content under a run_id already stored raises
|
|
@@ -70,6 +75,11 @@ module ActionAgent
|
|
|
70
75
|
IDENTIFIER_PATTERN = /\A[^[:cntrl:]]{1,200}\z/
|
|
71
76
|
AGENT_NAME_PATTERN = /\A[^[:cntrl:]]{2,100}\z/
|
|
72
77
|
TRACE_PATTERN = /\A[a-zA-Z0-9_-]{1,128}\z/
|
|
78
|
+
DIGEST_PATTERN = /\A[a-zA-Z0-9._-]{1,64}\z/
|
|
79
|
+
# The judge usage keys a report may send, per result and for the run.
|
|
80
|
+
JUDGE_USAGE_KEYS = %w[calls input_tokens output_tokens cost model by_kind source].freeze
|
|
81
|
+
JUDGE_USAGE_SOURCES = %w[reported traces meter].freeze
|
|
82
|
+
MAX_JUDGE_KINDS = 10
|
|
73
83
|
SCOPE_PATTERN = %r{\A[\w .:/@-]{1,100}\z}
|
|
74
84
|
STATUSES = %w[passed failed errored].freeze
|
|
75
85
|
# Report metadata that tells one evaluation of a suite from another, in the
|
|
@@ -253,6 +263,7 @@ module ActionAgent
|
|
|
253
263
|
external_tenant: tenant_key,
|
|
254
264
|
external_run_id: @payload["run_id"],
|
|
255
265
|
external_report_digest: digest,
|
|
266
|
+
agent_version: reported_release_version(agent),
|
|
256
267
|
status: :complete,
|
|
257
268
|
selection: selection,
|
|
258
269
|
scores: recorded_scores,
|
|
@@ -267,6 +278,15 @@ module ActionAgent
|
|
|
267
278
|
run
|
|
268
279
|
end
|
|
269
280
|
|
|
281
|
+
# The agent version for the release the report names, or nil for a
|
|
282
|
+
# report that names none.
|
|
283
|
+
def reported_release_version(agent)
|
|
284
|
+
release = report["release"]
|
|
285
|
+
return nil unless release.is_a?(Hash) && release["digest"].present?
|
|
286
|
+
|
|
287
|
+
agent.find_or_record_release!(digest: release["digest"], revision: release["revision"].presence, label: release["label"].presence)
|
|
288
|
+
end
|
|
289
|
+
|
|
270
290
|
# --- ownership -------------------------------------------------------------
|
|
271
291
|
|
|
272
292
|
# Who a newly observed agent belongs to. Whatever the host's
|
|
@@ -492,11 +512,28 @@ module ActionAgent
|
|
|
492
512
|
"key" => result["scenario_key"], "group" => result["group"], "prompt" => result["prompt"],
|
|
493
513
|
"position" => scenario.position, "expectations" => {}
|
|
494
514
|
}
|
|
495
|
-
),
|
|
515
|
+
).merge(bounded_judge_usage(result["judge_usage"]).then { |usage| usage ? { "_judge_usage" => usage } : {} }),
|
|
496
516
|
error_message: truncated(result["error"], TEXT_BYTES)
|
|
497
517
|
)
|
|
498
518
|
end
|
|
499
519
|
|
|
520
|
+
# A judge usage as the engine stores it: the keys the dashboard reads,
|
|
521
|
+
# each cut to its type, or nil for none.
|
|
522
|
+
def bounded_judge_usage(usage)
|
|
523
|
+
return nil unless usage.is_a?(Hash)
|
|
524
|
+
|
|
525
|
+
by_kind = usage["by_kind"].is_a?(Hash) ? usage["by_kind"].first(MAX_JUDGE_KINDS).to_h { |kind, count| [ kind.to_s.first(40), count.to_i ] } : {}
|
|
526
|
+
{
|
|
527
|
+
"calls" => usage["calls"].to_i,
|
|
528
|
+
"input_tokens" => usage["input_tokens"].to_i,
|
|
529
|
+
"output_tokens" => usage["output_tokens"].to_i,
|
|
530
|
+
"cost" => usage["cost"].is_a?(Numeric) ? usage["cost"].to_f : nil,
|
|
531
|
+
"model" => usage["model"].is_a?(String) ? usage["model"].first(200).presence : nil,
|
|
532
|
+
"by_kind" => by_kind,
|
|
533
|
+
"source" => JUDGE_USAGE_SOURCES.include?(usage["source"]) ? usage["source"] : "reported"
|
|
534
|
+
}
|
|
535
|
+
end
|
|
536
|
+
|
|
500
537
|
# The first +bytes+ bytes of +text+, dropping a character the cut splits,
|
|
501
538
|
# or nil for blank text.
|
|
502
539
|
def truncated(text, bytes)
|
|
@@ -521,14 +558,18 @@ module ActionAgent
|
|
|
521
558
|
{
|
|
522
559
|
"_verdict" => report["verdict"],
|
|
523
560
|
"_selection" => selection,
|
|
524
|
-
"_metadata" => metadata
|
|
561
|
+
"_metadata" => metadata,
|
|
562
|
+
"_judge_usage_run" => bounded_judge_usage(report.dig("judge_usage", "run"))
|
|
525
563
|
}.compact.merge("_judge_label" => judge_label)
|
|
526
564
|
end
|
|
527
565
|
|
|
528
566
|
# The run's scores in the shape the Evaluations view renders
|
|
529
|
-
# (ScenarioEvaluationRunner#scores_for), summarized from the stored
|
|
567
|
+
# (ScenarioEvaluationRunner#scores_for), summarized from the stored
|
|
568
|
+
# results as the application reported them — no cost is estimated here,
|
|
569
|
+
# so the summaries stored stay the application's own figures; the API
|
|
570
|
+
# adds the estimates when it serves the run (EvaluationRun#model_summaries).
|
|
530
571
|
def summarized_scores(run)
|
|
531
|
-
rebuilt = run.to_report
|
|
572
|
+
rebuilt = run.to_report(estimate: false)
|
|
532
573
|
rebuilt.criterion_scores.merge(
|
|
533
574
|
"_models" => rebuilt.summary_by_model,
|
|
534
575
|
"_recommendations" => rebuilt.recommendations
|
|
@@ -559,6 +600,9 @@ module ActionAgent
|
|
|
559
600
|
|
|
560
601
|
judge_trace_ids!(metadata["judge_trace_ids"])
|
|
561
602
|
string!(report["judge"], "report.judge", 200)
|
|
603
|
+
validate_release!
|
|
604
|
+
validate_judge_usage!(report.dig("judge_usage", "run"), "report.judge_usage.run") if report["judge_usage"].is_a?(Hash)
|
|
605
|
+
optional_object!(report["judge_usage"], "report.judge_usage")
|
|
562
606
|
object!(report["models"], "report.models")
|
|
563
607
|
raise Invalid, "report.models must contain 1-#{MAX_MODELS} models" unless report["models"].size.between?(1, MAX_MODELS)
|
|
564
608
|
unless results.is_a?(Array) && results.size.between?(1, MAX_RESULTS)
|
|
@@ -605,6 +649,7 @@ module ActionAgent
|
|
|
605
649
|
end
|
|
606
650
|
trace_id!(result_metadata["trace_id"]) if result_metadata["trace_id"]
|
|
607
651
|
judge_trace_ids!(result_metadata["judge_trace_ids"])
|
|
652
|
+
validate_judge_usage!(result["judge_usage"], "result.judge_usage")
|
|
608
653
|
end
|
|
609
654
|
distinct_specs = label_specs.values.uniq
|
|
610
655
|
raise Invalid, "two model labels name the same provider/model" if distinct_specs.size < label_specs.size
|
|
@@ -655,6 +700,40 @@ module ActionAgent
|
|
|
655
700
|
string!(judge.dig("suggested_tool", "name"), "result.diagnosis.judge.suggested_tool.name", 200)
|
|
656
701
|
end
|
|
657
702
|
|
|
703
|
+
# The release the report says it evaluated: a digest the dashboard can
|
|
704
|
+
# match to a version cut on deploy, and the deploy and label as text.
|
|
705
|
+
def validate_release!
|
|
706
|
+
release = report["release"]
|
|
707
|
+
return if release.nil?
|
|
708
|
+
|
|
709
|
+
object!(release, "report.release")
|
|
710
|
+
unless release["digest"].nil? || (release["digest"].is_a?(String) && DIGEST_PATTERN.match?(release["digest"]))
|
|
711
|
+
raise Invalid, "report.release.digest must be 1-64 letters, digits or . _ -"
|
|
712
|
+
end
|
|
713
|
+
|
|
714
|
+
string!(release["revision"], "report.release.revision", 200)
|
|
715
|
+
string!(release["label"], "report.release.label", 200)
|
|
716
|
+
end
|
|
717
|
+
|
|
718
|
+
# What the judge spent, as ActiveAgent::Evals::Report writes it per
|
|
719
|
+
# result and for the run.
|
|
720
|
+
def validate_judge_usage!(usage, name)
|
|
721
|
+
return if usage.nil?
|
|
722
|
+
|
|
723
|
+
object!(usage, name)
|
|
724
|
+
%w[calls input_tokens output_tokens].each do |key|
|
|
725
|
+
next if usage[key].nil?
|
|
726
|
+
raise Invalid, "#{name}.#{key} must be a non-negative integer" unless usage[key].is_a?(Integer) && usage[key] >= 0 && usage[key] <= NUMERIC_LIMITS["input_tokens"]
|
|
727
|
+
end
|
|
728
|
+
numeric!(usage["cost"], "#{name}.cost", max: NUMERIC_LIMITS["cost"])
|
|
729
|
+
string!(usage["model"], "#{name}.model", 200)
|
|
730
|
+
string!(usage["source"], "#{name}.source", 20)
|
|
731
|
+
optional_object!(usage["by_kind"], "#{name}.by_kind")
|
|
732
|
+
(usage["by_kind"] || {}).each_value do |count|
|
|
733
|
+
raise Invalid, "#{name}.by_kind counts must be non-negative integers" unless count.is_a?(Integer) && count >= 0
|
|
734
|
+
end
|
|
735
|
+
end
|
|
736
|
+
|
|
658
737
|
def validate_verdict!
|
|
659
738
|
verdict = report["verdict"]
|
|
660
739
|
return if verdict.nil?
|