activeagent 1.8.0 → 1.8.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +149 -0
- data/lib/active_agent/evals/diagnosis.rb +3 -3
- data/lib/active_agent/evals/format.rb +115 -0
- data/lib/active_agent/evals/judge.rb +2 -0
- data/lib/active_agent/evals/report.rb +232 -31
- data/lib/active_agent/evals/report_html.rb +316 -149
- data/lib/active_agent/evals/result.rb +40 -0
- data/lib/active_agent/evals/runner.rb +6 -2
- data/lib/active_agent/evals.rb +1 -0
- data/lib/active_agent/version.rb +1 -1
- metadata +3 -2
|
@@ -29,8 +29,17 @@ module ActiveAgent
|
|
|
29
29
|
# and the dashboard never name two different best models.
|
|
30
30
|
# @param judge_label [String, nil] how to name the judge when no Judge
|
|
31
31
|
# instance is at hand — a rebuilt run knows only its label.
|
|
32
|
+
# @param judge_usage [Hash, nil] what the judge spent on calls no result
|
|
33
|
+
# owns — the verdict, authoring criteria — as `{ "calls",
|
|
34
|
+
# "input_tokens", "output_tokens", "cost", "model", "by_kind",
|
|
35
|
+
# "source" }`. Each result's own judge calls travel in its replay
|
|
36
|
+
# metadata (Result#judge_usage); #judge_usage sums the two.
|
|
37
|
+
# @param release [Hash, nil] the release of the agent under evaluation,
|
|
38
|
+
# `{ "digest", "revision", "label" }`, so the report names the code
|
|
39
|
+
# it scored: in `to_h["release"]` and as a header chip.
|
|
32
40
|
def initialize(results:, models:, judge: nil, instructions: nil, threshold: PASS_THRESHOLD, metadata: {},
|
|
33
|
-
tool_resolver: nil, agent_name: nil, links: {}, verdict: nil, judge_label: nil
|
|
41
|
+
tool_resolver: nil, agent_name: nil, links: {}, verdict: nil, judge_label: nil,
|
|
42
|
+
judge_usage: nil, release: nil)
|
|
34
43
|
@results = results
|
|
35
44
|
@models = models
|
|
36
45
|
@judge = judge
|
|
@@ -42,14 +51,26 @@ module ActiveAgent
|
|
|
42
51
|
@agent_name = agent_name.presence || "the agent"
|
|
43
52
|
@links = (links || {}).to_h.stringify_keys
|
|
44
53
|
@recorded_verdict = verdict.is_a?(Hash) ? verdict.to_h.stringify_keys.presence : nil
|
|
54
|
+
@run_judge_usage = judge_usage.is_a?(Hash) ? judge_usage.to_h.stringify_keys.presence : nil
|
|
55
|
+
@release = release.is_a?(Hash) ? release.to_h.stringify_keys.compact.presence : nil
|
|
45
56
|
end
|
|
46
57
|
|
|
58
|
+
# The release of the agent this report scored, `{ "digest",
|
|
59
|
+
# "revision", "label" }`, when the caller named it.
|
|
60
|
+
attr_reader :release
|
|
61
|
+
|
|
47
62
|
def comparing?
|
|
48
63
|
@models.size > 1
|
|
49
64
|
end
|
|
50
65
|
|
|
51
66
|
# Per model, keyed by label: scenario count, passes, errors, pass rate,
|
|
52
|
-
# mean score, mean latency, tokens, cost and fault counts.
|
|
67
|
+
# mean score, mean latency, tokens, cost and fault counts. "cost" sums
|
|
68
|
+
# the replays that carried a cost — "priced" of the "scenarios", of
|
|
69
|
+
# which "reported" came from the caller and "estimated" from tokens ×
|
|
70
|
+
# a model rate (Result#cost_source) — and is nil when none did.
|
|
71
|
+
# "judge_cost" and "judge_calls" sum what the judge spent on the
|
|
72
|
+
# cohort's results (Result#judge_usage); the judge's cost never joins
|
|
73
|
+
# the agent's.
|
|
53
74
|
def summary_by_model
|
|
54
75
|
@summary_by_model ||= @models.to_h do |spec|
|
|
55
76
|
cohort = @results.select { |result| result.label == spec.label }
|
|
@@ -57,6 +78,7 @@ module ActiveAgent
|
|
|
57
78
|
task_scores = cohort.filter_map { |result| result.scores["task_completion"] }
|
|
58
79
|
durations = cohort.filter_map { |result| result.replay.duration_ms }
|
|
59
80
|
costs = cohort.filter_map { |result| result.replay.cost }
|
|
81
|
+
judge_costs = cohort.filter_map { |result| result.judge_usage&.dig("cost") }
|
|
60
82
|
|
|
61
83
|
[ spec.label, {
|
|
62
84
|
"provider" => spec.provider,
|
|
@@ -71,11 +93,95 @@ module ActiveAgent
|
|
|
71
93
|
"input_tokens" => cohort.sum { |result| result.replay.input_tokens.to_i },
|
|
72
94
|
"output_tokens" => cohort.sum { |result| result.replay.output_tokens.to_i },
|
|
73
95
|
"cost" => costs.any? ? costs.sum.to_f.round(6) : nil,
|
|
96
|
+
"priced" => costs.size,
|
|
97
|
+
"reported" => cohort.count { |result| result.cost_source == "reported" },
|
|
98
|
+
"estimated" => cohort.count(&:estimated_cost?),
|
|
99
|
+
"judge_cost" => judge_costs.any? ? judge_costs.sum.to_f.round(6) : nil,
|
|
100
|
+
"judge_calls" => cohort.sum { |result| result.judge_usage&.dig("calls").to_i },
|
|
74
101
|
"faults" => cohort.filter_map(&:fault).tally
|
|
75
102
|
} ]
|
|
76
103
|
end
|
|
77
104
|
end
|
|
78
105
|
|
|
106
|
+
# Per scenario key, in run order: what every model's answer cost
|
|
107
|
+
# together — the agent's cost summed over the models, the judge's, and
|
|
108
|
+
# the two as "total" — plus the same per model under "models", keyed
|
|
109
|
+
# by label, with each result's "cost_source". "estimated" is true when
|
|
110
|
+
# any part was estimated or any model went unpriced (the sum is then a
|
|
111
|
+
# lower bound), which is what a "~" on the figure means. Derived from
|
|
112
|
+
# the results for a matrix column and a cell line; not part of #to_h.
|
|
113
|
+
def scenario_costs
|
|
114
|
+
@scenario_costs ||= @results.group_by { |result| result.scenario.key }.to_h do |key, cohort|
|
|
115
|
+
costs = cohort.filter_map { |result| result.replay.cost }
|
|
116
|
+
judge_costs = cohort.filter_map { |result| result.judge_usage&.dig("cost") }
|
|
117
|
+
cost = costs.any? ? costs.sum.to_f.round(6) : nil
|
|
118
|
+
judge_cost = judge_costs.any? ? judge_costs.sum.to_f.round(6) : nil
|
|
119
|
+
[ key, {
|
|
120
|
+
"cost" => cost,
|
|
121
|
+
"judge_cost" => judge_cost,
|
|
122
|
+
"total" => cost.nil? && judge_cost.nil? ? nil : (cost.to_f + judge_cost.to_f).round(6),
|
|
123
|
+
"estimated" => cohort.any? { |result| result.estimated_cost? || (result.replay.cost.nil? && costs.any?) } ||
|
|
124
|
+
cohort.any? { |result| estimated_usage?(result.judge_usage) },
|
|
125
|
+
"models" => cohort.to_h do |result|
|
|
126
|
+
[ result.label, { "cost" => result.replay.cost&.to_f, "judge_cost" => result.judge_usage&.dig("cost")&.to_f,
|
|
127
|
+
"cost_source" => result.cost_source } ]
|
|
128
|
+
end
|
|
129
|
+
} ]
|
|
130
|
+
end
|
|
131
|
+
end
|
|
132
|
+
|
|
133
|
+
# What the judge spent on the whole run: every result's own calls
|
|
134
|
+
# (Result#judge_usage) plus the run-level calls handed to `judge_usage:`
|
|
135
|
+
# — `{ "calls", "input_tokens", "output_tokens", "cost", "model",
|
|
136
|
+
# "by_kind", "estimated", "run" => { "calls", "cost", "by_kind" } }`,
|
|
137
|
+
# where "run" is that run-level part alone and "estimated" says a cost
|
|
138
|
+
# in the sum was worked out from tokens rather than reported. nil when
|
|
139
|
+
# no judge was asked anything.
|
|
140
|
+
def judge_usage
|
|
141
|
+
return @judge_usage if defined?(@judge_usage)
|
|
142
|
+
|
|
143
|
+
parts = @results.filter_map(&:judge_usage)
|
|
144
|
+
parts << @run_judge_usage if @run_judge_usage
|
|
145
|
+
return @judge_usage = nil if parts.empty?
|
|
146
|
+
|
|
147
|
+
costs = parts.filter_map { |usage| usage["cost"] }
|
|
148
|
+
@judge_usage = {
|
|
149
|
+
"calls" => parts.sum { |usage| usage["calls"].to_i },
|
|
150
|
+
"input_tokens" => parts.sum { |usage| usage["input_tokens"].to_i },
|
|
151
|
+
"output_tokens" => parts.sum { |usage| usage["output_tokens"].to_i },
|
|
152
|
+
"cost" => costs.any? ? costs.sum.to_f.round(6) : nil,
|
|
153
|
+
"model" => parts.filter_map { |usage| usage["model"].presence }.first,
|
|
154
|
+
"by_kind" => usage_by_kind(parts),
|
|
155
|
+
"estimated" => parts.any? { |usage| estimated_usage?(usage) },
|
|
156
|
+
"run" => @run_judge_usage && {
|
|
157
|
+
"calls" => @run_judge_usage["calls"].to_i,
|
|
158
|
+
"cost" => @run_judge_usage["cost"]&.to_f,
|
|
159
|
+
"by_kind" => usage_by_kind([ @run_judge_usage ])
|
|
160
|
+
}
|
|
161
|
+
}.compact
|
|
162
|
+
end
|
|
163
|
+
|
|
164
|
+
# The run's spend: the agent's cost over every result (nil when none
|
|
165
|
+
# was priced), the judge's (nil without a judge), their "total", and
|
|
166
|
+
# whether any of it is an estimate — the "~" of the cost tile and the
|
|
167
|
+
# footer. "priced" and "unpriced" count the results either way.
|
|
168
|
+
def run_costs
|
|
169
|
+
@run_costs ||= begin
|
|
170
|
+
costs = @results.filter_map { |result| result.replay.cost }
|
|
171
|
+
agent = costs.any? ? costs.sum.to_f.round(6) : nil
|
|
172
|
+
judge = judge_usage&.dig("cost")
|
|
173
|
+
{
|
|
174
|
+
"cost" => agent,
|
|
175
|
+
"judge_cost" => judge,
|
|
176
|
+
"total" => agent.nil? && judge.nil? ? nil : (agent.to_f + judge.to_f).round(6),
|
|
177
|
+
"estimated" => @results.any?(&:estimated_cost?) || (costs.any? && costs.size < @results.size) ||
|
|
178
|
+
judge_usage&.dig("estimated") == true,
|
|
179
|
+
"priced" => costs.size,
|
|
180
|
+
"unpriced" => @results.size - costs.size
|
|
181
|
+
}
|
|
182
|
+
end
|
|
183
|
+
end
|
|
184
|
+
|
|
79
185
|
# Per criterion key: `{ "score", "min", "max", "passed", "total" }` over
|
|
80
186
|
# every result, or a map of model label => those stats when comparing
|
|
81
187
|
# models. A criterion nothing could score is `{ "skipped" => true }`.
|
|
@@ -126,21 +232,27 @@ module ActiveAgent
|
|
|
126
232
|
|
|
127
233
|
# The best model when comparing: the verdict the run recorded when one
|
|
128
234
|
# was handed in, else highest pass rate, then mean score, then lowest
|
|
129
|
-
# cost (a model with no cost estimate ranks after
|
|
130
|
-
# judge's rationale when one is available.
|
|
131
|
-
#
|
|
235
|
+
# cost per priced scenario (a model with no cost estimate ranks after
|
|
236
|
+
# one with), with the judge's rationale when one is available. The
|
|
237
|
+
# judge's own spend is no part of the ranking: it measures the
|
|
238
|
+
# evaluation, not the model. `{ "winner", "rationale", "judge" }`, or
|
|
239
|
+
# nil for a single model.
|
|
132
240
|
def verdict
|
|
133
241
|
return @recorded_verdict if @recorded_verdict
|
|
134
242
|
return nil unless comparing?
|
|
135
243
|
|
|
136
244
|
@verdict ||= begin
|
|
137
245
|
ranked = summary_by_model.sort_by do |_label, stats|
|
|
138
|
-
[ -stats["pass_rate"].to_f, -stats["avg_score"].to_f, stats
|
|
246
|
+
[ -stats["pass_rate"].to_f, -stats["avg_score"].to_f, cost_per_priced(stats) || Float::INFINITY ]
|
|
139
247
|
end
|
|
140
248
|
winner, stats = ranked.first
|
|
249
|
+
if stats["cost"]
|
|
250
|
+
cost = " at #{Format.money(stats['cost'], estimated: estimated_cost?(stats))}"
|
|
251
|
+
cost += " (estimated)" if estimated_cost?(stats)
|
|
252
|
+
end
|
|
141
253
|
rationale = "Passed #{stats['passed']} of #{stats['scenarios']} scenarios" \
|
|
142
|
-
"#{
|
|
143
|
-
"#{"
|
|
254
|
+
" (#{Format.percent(stats['scenarios'].to_i.positive? ? stats['passed'].to_f / stats['scenarios'] : nil)})" \
|
|
255
|
+
"#{" with a mean score of #{Format.score(stats['avg_score'])}" if stats['avg_score']}#{cost}."
|
|
144
256
|
judged = @judge&.verdict(summary_by_model, instructions: @instructions)
|
|
145
257
|
|
|
146
258
|
{
|
|
@@ -155,6 +267,8 @@ module ActiveAgent
|
|
|
155
267
|
verdict&.dig("winner")
|
|
156
268
|
end
|
|
157
269
|
|
|
270
|
+
# Adds "judge_usage" and "release" only when there is one to add, so
|
|
271
|
+
# a report built the way it always was serializes exactly as before.
|
|
158
272
|
def to_h
|
|
159
273
|
{
|
|
160
274
|
"models" => summary_by_model,
|
|
@@ -162,6 +276,8 @@ module ActiveAgent
|
|
|
162
276
|
"recommendations" => recommendations,
|
|
163
277
|
"verdict" => verdict,
|
|
164
278
|
"judge" => @judge_label || @judge&.label,
|
|
279
|
+
"judge_usage" => judge_usage,
|
|
280
|
+
"release" => release,
|
|
165
281
|
"metadata" => @metadata.presence,
|
|
166
282
|
"results" => @results.map(&:to_h)
|
|
167
283
|
}.compact
|
|
@@ -179,16 +295,46 @@ module ActiveAgent
|
|
|
179
295
|
lines << ""
|
|
180
296
|
lines.concat(summary_table)
|
|
181
297
|
lines << ""
|
|
298
|
+
lines.concat(total_lines)
|
|
182
299
|
lines << "**Best model: #{winner}**" if winner
|
|
183
300
|
lines << ""
|
|
184
301
|
lines.concat(matrix_table)
|
|
185
|
-
lines.concat(recommendation_lines)
|
|
186
302
|
lines.concat(detail_lines)
|
|
303
|
+
lines.concat(recommendation_lines)
|
|
187
304
|
lines.join("\n")
|
|
188
305
|
end
|
|
189
306
|
|
|
190
307
|
private
|
|
191
308
|
|
|
309
|
+
# Whether a model summary's cost is an estimate — any replay priced
|
|
310
|
+
# from tokens, or some replay unpriced, which leaves the sum a lower
|
|
311
|
+
# bound — and so reads with a "~".
|
|
312
|
+
def estimated_cost?(stats)
|
|
313
|
+
stats["estimated"].to_i.positive? || (stats["cost"] && stats["priced"].to_i < stats["scenarios"].to_i)
|
|
314
|
+
end
|
|
315
|
+
|
|
316
|
+
# A judge usage whose cost was worked out rather than reported: metered
|
|
317
|
+
# or priced from traces by the caller. One with no source is the
|
|
318
|
+
# caller's own figure.
|
|
319
|
+
def estimated_usage?(usage)
|
|
320
|
+
return false unless usage.is_a?(Hash)
|
|
321
|
+
|
|
322
|
+
source = (usage["source"] || usage[:source]).to_s
|
|
323
|
+
source.present? && source != "reported"
|
|
324
|
+
end
|
|
325
|
+
|
|
326
|
+
def usage_by_kind(parts)
|
|
327
|
+
parts.each_with_object({}) do |usage, tally|
|
|
328
|
+
(usage["by_kind"] || {}).each { |kind, count| tally[kind.to_s] = tally.fetch(kind.to_s, 0) + count.to_i }
|
|
329
|
+
end
|
|
330
|
+
end
|
|
331
|
+
|
|
332
|
+
# A model's cost per priced replay, nil when none was priced.
|
|
333
|
+
def cost_per_priced(stats)
|
|
334
|
+
priced = stats["priced"].to_i
|
|
335
|
+
stats["cost"].to_f / priced if stats["cost"] && priced.positive?
|
|
336
|
+
end
|
|
337
|
+
|
|
192
338
|
def criterion_keys
|
|
193
339
|
@results.flat_map { |result| result.scores.keys }.uniq
|
|
194
340
|
end
|
|
@@ -379,58 +525,90 @@ module ActiveAgent
|
|
|
379
525
|
|
|
380
526
|
# --- Markdown ----------------------------------------------------------
|
|
381
527
|
|
|
528
|
+
# The per-model table. A Judge column joins it when a judge spent
|
|
529
|
+
# anything, so the agent's cost and the evaluation's stay two numbers.
|
|
382
530
|
def summary_table
|
|
383
|
-
|
|
384
|
-
|
|
531
|
+
judged = judge_usage.present?
|
|
532
|
+
columns = [ "Model", "Passed", "Mean score", "Mean latency", "Tokens in/out", "Cost", ("Judge" if judged), "Faults" ].compact
|
|
533
|
+
header = [ "| #{columns.join(' | ')} |", "|#{columns.map { '---' }.join('|')}|" ]
|
|
385
534
|
rows = summary_by_model.map do |label, stats|
|
|
386
535
|
faults = stats["faults"].map { |fault, count| "#{fault.tr('_', ' ')} ×#{count}" }.join(", ")
|
|
387
536
|
latency = stats["avg_duration_ms"] ? "#{stats['avg_duration_ms']} ms" : "—"
|
|
388
|
-
|
|
389
|
-
|
|
390
|
-
|
|
537
|
+
cells = [
|
|
538
|
+
"`#{label}`",
|
|
539
|
+
Format.passes(stats["passed"], stats["scenarios"], style: :markdown),
|
|
540
|
+
Format.score(stats["avg_score"]),
|
|
541
|
+
latency,
|
|
542
|
+
"#{stats['input_tokens']}/#{stats['output_tokens']}",
|
|
543
|
+
Format.money(stats["cost"], estimated: estimated_cost?(stats)),
|
|
544
|
+
(judge_cell(stats) if judged),
|
|
545
|
+
faults.presence || "—"
|
|
546
|
+
].compact
|
|
547
|
+
"| #{cells.join(' | ')} |"
|
|
391
548
|
end
|
|
392
549
|
header + rows
|
|
393
550
|
end
|
|
394
551
|
|
|
552
|
+
# "~$0.0030 (3 calls)" for a model's judge spend, "—" when the judge
|
|
553
|
+
# was not asked about its results.
|
|
554
|
+
def judge_cell(stats)
|
|
555
|
+
return "—" if stats["judge_cost"].nil? && stats["judge_calls"].to_i.zero?
|
|
556
|
+
|
|
557
|
+
money = Format.money(stats["judge_cost"], estimated: judge_usage&.dig("estimated") == true)
|
|
558
|
+
"#{money} (#{stats['judge_calls'].to_i} call#{'s' unless stats['judge_calls'].to_i == 1})"
|
|
559
|
+
end
|
|
560
|
+
|
|
561
|
+
# "**Total: ~$0.0412** (agent ~$0.0397 · judge ~$0.0015)" under the
|
|
562
|
+
# table, with the legend for "~" when any figure carries it. Nothing
|
|
563
|
+
# for a run with no cost on either side.
|
|
564
|
+
def total_lines
|
|
565
|
+
costs = run_costs
|
|
566
|
+
return [] if costs["total"].nil?
|
|
567
|
+
|
|
568
|
+
parts = [ "agent #{Format.money(costs['cost'], estimated: costs['estimated'])}" ]
|
|
569
|
+
parts << "judge #{Format.money(costs['judge_cost'], estimated: judge_usage&.dig('estimated') == true)}" if judge_usage
|
|
570
|
+
lines = [ "**Total: #{Format.money(costs['total'], estimated: costs['estimated'])}** (#{parts.join(' · ')})" ]
|
|
571
|
+
lines << "_#{Format::LEGEND}_" if costs["estimated"]
|
|
572
|
+
lines << ""
|
|
573
|
+
end
|
|
574
|
+
|
|
575
|
+
# The scenario matrix, with a trailing Cost column: what the scenario
|
|
576
|
+
# cost across every model, and in brackets what the judge spent on it.
|
|
395
577
|
def matrix_table
|
|
396
578
|
labels = @models.map(&:label)
|
|
397
|
-
header = [ "| Scenario | #{labels.map { |label| "`#{label}`" }.join(' | ')} |", "|---|#{labels.map { '---' }.join('|')}
|
|
579
|
+
header = [ "| Scenario | #{labels.map { |label| "`#{label}`" }.join(' | ')} | Cost |", "|---|#{labels.map { '---' }.join('|')}|---|" ]
|
|
398
580
|
rows = @results.group_by { |result| result.scenario.key }.map do |key, cohort|
|
|
399
581
|
cells = labels.map do |label|
|
|
400
582
|
result = cohort.find { |candidate| candidate.label == label }
|
|
401
583
|
next "—" unless result
|
|
402
584
|
|
|
403
585
|
mark = result.passed? ? "✅" : (result.errored? ? "⚠️" : "❌")
|
|
404
|
-
[ mark, result.score
|
|
586
|
+
[ mark, (Format.score(result.score) if result.score), result.fault&.tr("_", " ") ].compact.join(" ")
|
|
405
587
|
end
|
|
406
|
-
"| `#{key}` #{cell(cohort.first.scenario.prompt.truncate(70))} | #{cells.join(' | ')} |"
|
|
588
|
+
"| `#{key}` #{cell(cohort.first.scenario.prompt.truncate(70))} | #{cells.join(' | ')} | #{scenario_cost_cell(key)} |"
|
|
407
589
|
end
|
|
408
590
|
header + rows
|
|
409
591
|
end
|
|
410
592
|
|
|
593
|
+
# "~$0.0243 (judge ~$0.0015)" for a scenario's total across models.
|
|
594
|
+
def scenario_cost_cell(key)
|
|
595
|
+
costs = scenario_costs[key] || {}
|
|
596
|
+
text = Format.money(costs["cost"], estimated: costs["estimated"])
|
|
597
|
+
text += " (judge #{Format.money(costs['judge_cost'], estimated: costs['estimated'])})" if costs["judge_cost"]
|
|
598
|
+
text
|
|
599
|
+
end
|
|
600
|
+
|
|
411
601
|
# A prompt may contain " | " (ScenarioParser keeps it), which would
|
|
412
602
|
# otherwise split the table cell.
|
|
413
603
|
def cell(text)
|
|
414
604
|
text.to_s.gsub("|") { "\\|" }
|
|
415
605
|
end
|
|
416
606
|
|
|
417
|
-
def recommendation_lines
|
|
418
|
-
return [] if recommendations.empty?
|
|
419
|
-
|
|
420
|
-
lines = [ "", "## Recommendations", "" ]
|
|
421
|
-
recommendations.each do |entry|
|
|
422
|
-
lines << "- **#{entry['fault'].tr('_', ' ')}** ×#{entry['count']} (#{entry['scenario_keys'].join(', ')}): #{entry['recommendation']}"
|
|
423
|
-
entry["suggested_tools"].each do |tool|
|
|
424
|
-
lines << " - suggested tool `#{tool['name']}`: #{tool['description']}"
|
|
425
|
-
end
|
|
426
|
-
end
|
|
427
|
-
lines
|
|
428
|
-
end
|
|
429
|
-
|
|
430
607
|
def detail_lines
|
|
431
608
|
lines = [ "", "## Answers", "" ]
|
|
432
609
|
@results.each do |result|
|
|
433
|
-
lines << "### `#{result.scenario.key}` · `#{result.label}` · #{result.status}
|
|
610
|
+
lines << "### `#{result.scenario.key}` · `#{result.label}` · #{result.status}" \
|
|
611
|
+
"#{" · score #{Format.score(result.score)}" if result.score}#{" · #{answer_cost(result)}" if result.replay.cost}"
|
|
434
612
|
lines << ""
|
|
435
613
|
lines << "> #{result.scenario.prompt}"
|
|
436
614
|
lines << ""
|
|
@@ -445,6 +623,29 @@ module ActiveAgent
|
|
|
445
623
|
end
|
|
446
624
|
lines
|
|
447
625
|
end
|
|
626
|
+
|
|
627
|
+
# "~$0.0243 · judge ~$0.0015": what one answer cost, and what judging
|
|
628
|
+
# it cost.
|
|
629
|
+
def answer_cost(result)
|
|
630
|
+
text = Format.money(result.replay.cost, estimated: result.estimated_cost?)
|
|
631
|
+
judge_cost = result.judge_usage&.dig("cost")
|
|
632
|
+
text += " · judge #{Format.money(judge_cost, estimated: estimated_usage?(result.judge_usage))}" if judge_cost
|
|
633
|
+
text
|
|
634
|
+
end
|
|
635
|
+
|
|
636
|
+
# Renders after `detail_lines`, whose trailing blank line separates the two sections.
|
|
637
|
+
def recommendation_lines
|
|
638
|
+
return [] if recommendations.empty?
|
|
639
|
+
|
|
640
|
+
lines = [ "## Recommendations", "" ]
|
|
641
|
+
recommendations.each do |entry|
|
|
642
|
+
lines << "- **#{entry['fault'].tr('_', ' ')}** ×#{entry['count']} (#{entry['scenario_keys'].join(', ')}): #{entry['recommendation']}"
|
|
643
|
+
entry["suggested_tools"].each do |tool|
|
|
644
|
+
lines << " - suggested tool `#{tool['name']}`: #{tool['description']}"
|
|
645
|
+
end
|
|
646
|
+
end
|
|
647
|
+
lines << ""
|
|
648
|
+
end
|
|
448
649
|
end
|
|
449
650
|
end
|
|
450
651
|
end
|