activeagent 1.8.0 → 1.8.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -29,8 +29,17 @@ module ActiveAgent
29
29
  # and the dashboard never name two different best models.
30
30
  # @param judge_label [String, nil] how to name the judge when no Judge
31
31
  # instance is at hand — a rebuilt run knows only its label.
32
+ # @param judge_usage [Hash, nil] what the judge spent on calls no result
33
+ # owns — the verdict, authoring criteria — as `{ "calls",
34
+ # "input_tokens", "output_tokens", "cost", "model", "by_kind",
35
+ # "source" }`. Each result's own judge calls travel in its replay
36
+ # metadata (Result#judge_usage); #judge_usage sums the two.
37
+ # @param release [Hash, nil] the release of the agent under evaluation,
38
+ # `{ "digest", "revision", "label" }`, so the report names the code
39
+ # it scored: in `to_h["release"]` and as a header chip.
32
40
  def initialize(results:, models:, judge: nil, instructions: nil, threshold: PASS_THRESHOLD, metadata: {},
33
- tool_resolver: nil, agent_name: nil, links: {}, verdict: nil, judge_label: nil)
41
+ tool_resolver: nil, agent_name: nil, links: {}, verdict: nil, judge_label: nil,
42
+ judge_usage: nil, release: nil)
34
43
  @results = results
35
44
  @models = models
36
45
  @judge = judge
@@ -42,14 +51,26 @@ module ActiveAgent
42
51
  @agent_name = agent_name.presence || "the agent"
43
52
  @links = (links || {}).to_h.stringify_keys
44
53
  @recorded_verdict = verdict.is_a?(Hash) ? verdict.to_h.stringify_keys.presence : nil
54
+ @run_judge_usage = judge_usage.is_a?(Hash) ? judge_usage.to_h.stringify_keys.presence : nil
55
+ @release = release.is_a?(Hash) ? release.to_h.stringify_keys.compact.presence : nil
45
56
  end
46
57
 
58
+ # The release of the agent this report scored, `{ "digest",
59
+ # "revision", "label" }`, when the caller named it.
60
+ attr_reader :release
61
+
47
62
  def comparing?
48
63
  @models.size > 1
49
64
  end
50
65
 
51
66
  # Per model, keyed by label: scenario count, passes, errors, pass rate,
52
- # mean score, mean latency, tokens, cost and fault counts.
67
+ # mean score, mean latency, tokens, cost and fault counts. "cost" sums
68
+ # the replays that carried a cost — "priced" of the "scenarios", of
69
+ # which "reported" came from the caller and "estimated" from tokens ×
70
+ # a model rate (Result#cost_source) — and is nil when none did.
71
+ # "judge_cost" and "judge_calls" sum what the judge spent on the
72
+ # cohort's results (Result#judge_usage); the judge's cost never joins
73
+ # the agent's.
53
74
  def summary_by_model
54
75
  @summary_by_model ||= @models.to_h do |spec|
55
76
  cohort = @results.select { |result| result.label == spec.label }
@@ -57,6 +78,7 @@ module ActiveAgent
57
78
  task_scores = cohort.filter_map { |result| result.scores["task_completion"] }
58
79
  durations = cohort.filter_map { |result| result.replay.duration_ms }
59
80
  costs = cohort.filter_map { |result| result.replay.cost }
81
+ judge_costs = cohort.filter_map { |result| result.judge_usage&.dig("cost") }
60
82
 
61
83
  [ spec.label, {
62
84
  "provider" => spec.provider,
@@ -71,11 +93,95 @@ module ActiveAgent
71
93
  "input_tokens" => cohort.sum { |result| result.replay.input_tokens.to_i },
72
94
  "output_tokens" => cohort.sum { |result| result.replay.output_tokens.to_i },
73
95
  "cost" => costs.any? ? costs.sum.to_f.round(6) : nil,
96
+ "priced" => costs.size,
97
+ "reported" => cohort.count { |result| result.cost_source == "reported" },
98
+ "estimated" => cohort.count(&:estimated_cost?),
99
+ "judge_cost" => judge_costs.any? ? judge_costs.sum.to_f.round(6) : nil,
100
+ "judge_calls" => cohort.sum { |result| result.judge_usage&.dig("calls").to_i },
74
101
  "faults" => cohort.filter_map(&:fault).tally
75
102
  } ]
76
103
  end
77
104
  end
78
105
 
106
+ # Per scenario key, in run order: what every model's answer cost
107
+ # together — the agent's cost summed over the models, the judge's, and
108
+ # the two as "total" — plus the same per model under "models", keyed
109
+ # by label, with each result's "cost_source". "estimated" is true when
110
+ # any part was estimated or any model went unpriced (the sum is then a
111
+ # lower bound), which is what a "~" on the figure means. Derived from
112
+ # the results for a matrix column and a cell line; not part of #to_h.
113
+ def scenario_costs
114
+ @scenario_costs ||= @results.group_by { |result| result.scenario.key }.to_h do |key, cohort|
115
+ costs = cohort.filter_map { |result| result.replay.cost }
116
+ judge_costs = cohort.filter_map { |result| result.judge_usage&.dig("cost") }
117
+ cost = costs.any? ? costs.sum.to_f.round(6) : nil
118
+ judge_cost = judge_costs.any? ? judge_costs.sum.to_f.round(6) : nil
119
+ [ key, {
120
+ "cost" => cost,
121
+ "judge_cost" => judge_cost,
122
+ "total" => cost.nil? && judge_cost.nil? ? nil : (cost.to_f + judge_cost.to_f).round(6),
123
+ "estimated" => cohort.any? { |result| result.estimated_cost? || (result.replay.cost.nil? && costs.any?) } ||
124
+ cohort.any? { |result| estimated_usage?(result.judge_usage) },
125
+ "models" => cohort.to_h do |result|
126
+ [ result.label, { "cost" => result.replay.cost&.to_f, "judge_cost" => result.judge_usage&.dig("cost")&.to_f,
127
+ "cost_source" => result.cost_source } ]
128
+ end
129
+ } ]
130
+ end
131
+ end
132
+
133
+ # What the judge spent on the whole run: every result's own calls
134
+ # (Result#judge_usage) plus the run-level calls handed to `judge_usage:`
135
+ # — `{ "calls", "input_tokens", "output_tokens", "cost", "model",
136
+ # "by_kind", "estimated", "run" => { "calls", "cost", "by_kind" } }`,
137
+ # where "run" is that run-level part alone and "estimated" says a cost
138
+ # in the sum was worked out from tokens rather than reported. nil when
139
+ # no judge was asked anything.
140
+ def judge_usage
141
+ return @judge_usage if defined?(@judge_usage)
142
+
143
+ parts = @results.filter_map(&:judge_usage)
144
+ parts << @run_judge_usage if @run_judge_usage
145
+ return @judge_usage = nil if parts.empty?
146
+
147
+ costs = parts.filter_map { |usage| usage["cost"] }
148
+ @judge_usage = {
149
+ "calls" => parts.sum { |usage| usage["calls"].to_i },
150
+ "input_tokens" => parts.sum { |usage| usage["input_tokens"].to_i },
151
+ "output_tokens" => parts.sum { |usage| usage["output_tokens"].to_i },
152
+ "cost" => costs.any? ? costs.sum.to_f.round(6) : nil,
153
+ "model" => parts.filter_map { |usage| usage["model"].presence }.first,
154
+ "by_kind" => usage_by_kind(parts),
155
+ "estimated" => parts.any? { |usage| estimated_usage?(usage) },
156
+ "run" => @run_judge_usage && {
157
+ "calls" => @run_judge_usage["calls"].to_i,
158
+ "cost" => @run_judge_usage["cost"]&.to_f,
159
+ "by_kind" => usage_by_kind([ @run_judge_usage ])
160
+ }
161
+ }.compact
162
+ end
163
+
164
+ # The run's spend: the agent's cost over every result (nil when none
165
+ # was priced), the judge's (nil without a judge), their "total", and
166
+ # whether any of it is an estimate — the "~" of the cost tile and the
167
+ # footer. "priced" and "unpriced" count the results either way.
168
+ def run_costs
169
+ @run_costs ||= begin
170
+ costs = @results.filter_map { |result| result.replay.cost }
171
+ agent = costs.any? ? costs.sum.to_f.round(6) : nil
172
+ judge = judge_usage&.dig("cost")
173
+ {
174
+ "cost" => agent,
175
+ "judge_cost" => judge,
176
+ "total" => agent.nil? && judge.nil? ? nil : (agent.to_f + judge.to_f).round(6),
177
+ "estimated" => @results.any?(&:estimated_cost?) || (costs.any? && costs.size < @results.size) ||
178
+ judge_usage&.dig("estimated") == true,
179
+ "priced" => costs.size,
180
+ "unpriced" => @results.size - costs.size
181
+ }
182
+ end
183
+ end
184
+
79
185
  # Per criterion key: `{ "score", "min", "max", "passed", "total" }` over
80
186
  # every result, or a map of model label => those stats when comparing
81
187
  # models. A criterion nothing could score is `{ "skipped" => true }`.
@@ -126,21 +232,27 @@ module ActiveAgent
126
232
 
127
233
  # The best model when comparing: the verdict the run recorded when one
128
234
  # was handed in, else highest pass rate, then mean score, then lowest
129
- # cost (a model with no cost estimate ranks after one with), with the
130
- # judge's rationale when one is available.
131
- # `{ "winner", "rationale", "judge" }`, or nil for a single model.
235
+ # cost per priced scenario (a model with no cost estimate ranks after
236
+ # one with), with the judge's rationale when one is available. The
237
+ # judge's own spend is no part of the ranking: it measures the
238
+ # evaluation, not the model. `{ "winner", "rationale", "judge" }`, or
239
+ # nil for a single model.
132
240
  def verdict
133
241
  return @recorded_verdict if @recorded_verdict
134
242
  return nil unless comparing?
135
243
 
136
244
  @verdict ||= begin
137
245
  ranked = summary_by_model.sort_by do |_label, stats|
138
- [ -stats["pass_rate"].to_f, -stats["avg_score"].to_f, stats["cost"] || Float::INFINITY ]
246
+ [ -stats["pass_rate"].to_f, -stats["avg_score"].to_f, cost_per_priced(stats) || Float::INFINITY ]
139
247
  end
140
248
  winner, stats = ranked.first
249
+ if stats["cost"]
250
+ cost = " at #{Format.money(stats['cost'], estimated: estimated_cost?(stats))}"
251
+ cost += " (estimated)" if estimated_cost?(stats)
252
+ end
141
253
  rationale = "Passed #{stats['passed']} of #{stats['scenarios']} scenarios" \
142
- "#{" with a mean score of #{stats['avg_score']}" if stats['avg_score']}" \
143
- "#{" at an estimated $#{format('%.4f', stats['cost'])}" if stats['cost']}."
254
+ " (#{Format.percent(stats['scenarios'].to_i.positive? ? stats['passed'].to_f / stats['scenarios'] : nil)})" \
255
+ "#{" with a mean score of #{Format.score(stats['avg_score'])}" if stats['avg_score']}#{cost}."
144
256
  judged = @judge&.verdict(summary_by_model, instructions: @instructions)
145
257
 
146
258
  {
@@ -155,6 +267,8 @@ module ActiveAgent
155
267
  verdict&.dig("winner")
156
268
  end
157
269
 
270
+ # Adds "judge_usage" and "release" only when there is one to add, so
271
+ # a report built the way it always was serializes exactly as before.
158
272
  def to_h
159
273
  {
160
274
  "models" => summary_by_model,
@@ -162,6 +276,8 @@ module ActiveAgent
162
276
  "recommendations" => recommendations,
163
277
  "verdict" => verdict,
164
278
  "judge" => @judge_label || @judge&.label,
279
+ "judge_usage" => judge_usage,
280
+ "release" => release,
165
281
  "metadata" => @metadata.presence,
166
282
  "results" => @results.map(&:to_h)
167
283
  }.compact
@@ -179,16 +295,46 @@ module ActiveAgent
179
295
  lines << ""
180
296
  lines.concat(summary_table)
181
297
  lines << ""
298
+ lines.concat(total_lines)
182
299
  lines << "**Best model: #{winner}**" if winner
183
300
  lines << ""
184
301
  lines.concat(matrix_table)
185
- lines.concat(recommendation_lines)
186
302
  lines.concat(detail_lines)
303
+ lines.concat(recommendation_lines)
187
304
  lines.join("\n")
188
305
  end
189
306
 
190
307
  private
191
308
 
309
+ # Whether a model summary's cost is an estimate — any replay priced
310
+ # from tokens, or some replay unpriced, which leaves the sum a lower
311
+ # bound — and so reads with a "~".
312
+ def estimated_cost?(stats)
313
+ stats["estimated"].to_i.positive? || (stats["cost"] && stats["priced"].to_i < stats["scenarios"].to_i)
314
+ end
315
+
316
+ # A judge usage whose cost was worked out rather than reported: metered
317
+ # or priced from traces by the caller. One with no source is the
318
+ # caller's own figure.
319
+ def estimated_usage?(usage)
320
+ return false unless usage.is_a?(Hash)
321
+
322
+ source = (usage["source"] || usage[:source]).to_s
323
+ source.present? && source != "reported"
324
+ end
325
+
326
+ def usage_by_kind(parts)
327
+ parts.each_with_object({}) do |usage, tally|
328
+ (usage["by_kind"] || {}).each { |kind, count| tally[kind.to_s] = tally.fetch(kind.to_s, 0) + count.to_i }
329
+ end
330
+ end
331
+
332
+ # A model's cost per priced replay, nil when none was priced.
333
+ def cost_per_priced(stats)
334
+ priced = stats["priced"].to_i
335
+ stats["cost"].to_f / priced if stats["cost"] && priced.positive?
336
+ end
337
+
192
338
  def criterion_keys
193
339
  @results.flat_map { |result| result.scores.keys }.uniq
194
340
  end
@@ -379,58 +525,90 @@ module ActiveAgent
379
525
 
380
526
  # --- Markdown ----------------------------------------------------------
381
527
 
528
+ # The per-model table. A Judge column joins it when a judge spent
529
+ # anything, so the agent's cost and the evaluation's stay two numbers.
382
530
  def summary_table
383
- header = [ "| Model | Pass rate | Passed | Mean score | Mean latency | Tokens in/out | Cost | Faults |",
384
- "|---|---|---|---|---|---|---|---|" ]
531
+ judged = judge_usage.present?
532
+ columns = [ "Model", "Passed", "Mean score", "Mean latency", "Tokens in/out", "Cost", ("Judge" if judged), "Faults" ].compact
533
+ header = [ "| #{columns.join(' | ')} |", "|#{columns.map { '---' }.join('|')}|" ]
385
534
  rows = summary_by_model.map do |label, stats|
386
535
  faults = stats["faults"].map { |fault, count| "#{fault.tr('_', ' ')} ×#{count}" }.join(", ")
387
536
  latency = stats["avg_duration_ms"] ? "#{stats['avg_duration_ms']} ms" : "—"
388
- cost = stats["cost"] ? format("$%.4f", stats["cost"]) : "—"
389
- "| `#{label}` | #{stats['pass_rate']}% | #{stats['passed']}/#{stats['scenarios']} | #{stats['avg_score'] || '—'} | " \
390
- "#{latency} | #{stats['input_tokens']}/#{stats['output_tokens']} | #{cost} | #{faults.presence || '—'} |"
537
+ cells = [
538
+ "`#{label}`",
539
+ Format.passes(stats["passed"], stats["scenarios"], style: :markdown),
540
+ Format.score(stats["avg_score"]),
541
+ latency,
542
+ "#{stats['input_tokens']}/#{stats['output_tokens']}",
543
+ Format.money(stats["cost"], estimated: estimated_cost?(stats)),
544
+ (judge_cell(stats) if judged),
545
+ faults.presence || "—"
546
+ ].compact
547
+ "| #{cells.join(' | ')} |"
391
548
  end
392
549
  header + rows
393
550
  end
394
551
 
552
+ # "~$0.0030 (3 calls)" for a model's judge spend, "—" when the judge
553
+ # was not asked about its results.
554
+ def judge_cell(stats)
555
+ return "—" if stats["judge_cost"].nil? && stats["judge_calls"].to_i.zero?
556
+
557
+ money = Format.money(stats["judge_cost"], estimated: judge_usage&.dig("estimated") == true)
558
+ "#{money} (#{stats['judge_calls'].to_i} call#{'s' unless stats['judge_calls'].to_i == 1})"
559
+ end
560
+
561
+ # "**Total: ~$0.0412** (agent ~$0.0397 · judge ~$0.0015)" under the
562
+ # table, with the legend for "~" when any figure carries it. Nothing
563
+ # for a run with no cost on either side.
564
+ def total_lines
565
+ costs = run_costs
566
+ return [] if costs["total"].nil?
567
+
568
+ parts = [ "agent #{Format.money(costs['cost'], estimated: costs['estimated'])}" ]
569
+ parts << "judge #{Format.money(costs['judge_cost'], estimated: judge_usage&.dig('estimated') == true)}" if judge_usage
570
+ lines = [ "**Total: #{Format.money(costs['total'], estimated: costs['estimated'])}** (#{parts.join(' · ')})" ]
571
+ lines << "_#{Format::LEGEND}_" if costs["estimated"]
572
+ lines << ""
573
+ end
574
+
575
+ # The scenario matrix, with a trailing Cost column: what the scenario
576
+ # cost across every model, and in brackets what the judge spent on it.
395
577
  def matrix_table
396
578
  labels = @models.map(&:label)
397
- header = [ "| Scenario | #{labels.map { |label| "`#{label}`" }.join(' | ')} |", "|---|#{labels.map { '---' }.join('|')}|" ]
579
+ header = [ "| Scenario | #{labels.map { |label| "`#{label}`" }.join(' | ')} | Cost |", "|---|#{labels.map { '---' }.join('|')}|---|" ]
398
580
  rows = @results.group_by { |result| result.scenario.key }.map do |key, cohort|
399
581
  cells = labels.map do |label|
400
582
  result = cohort.find { |candidate| candidate.label == label }
401
583
  next "—" unless result
402
584
 
403
585
  mark = result.passed? ? "✅" : (result.errored? ? "⚠️" : "❌")
404
- [ mark, result.score&.round(2), result.fault&.tr("_", " ") ].compact.join(" ")
586
+ [ mark, (Format.score(result.score) if result.score), result.fault&.tr("_", " ") ].compact.join(" ")
405
587
  end
406
- "| `#{key}` #{cell(cohort.first.scenario.prompt.truncate(70))} | #{cells.join(' | ')} |"
588
+ "| `#{key}` #{cell(cohort.first.scenario.prompt.truncate(70))} | #{cells.join(' | ')} | #{scenario_cost_cell(key)} |"
407
589
  end
408
590
  header + rows
409
591
  end
410
592
 
593
+ # "~$0.0243 (judge ~$0.0015)" for a scenario's total across models.
594
+ def scenario_cost_cell(key)
595
+ costs = scenario_costs[key] || {}
596
+ text = Format.money(costs["cost"], estimated: costs["estimated"])
597
+ text += " (judge #{Format.money(costs['judge_cost'], estimated: costs['estimated'])})" if costs["judge_cost"]
598
+ text
599
+ end
600
+
411
601
  # A prompt may contain " | " (ScenarioParser keeps it), which would
412
602
  # otherwise split the table cell.
413
603
  def cell(text)
414
604
  text.to_s.gsub("|") { "\\|" }
415
605
  end
416
606
 
417
- def recommendation_lines
418
- return [] if recommendations.empty?
419
-
420
- lines = [ "", "## Recommendations", "" ]
421
- recommendations.each do |entry|
422
- lines << "- **#{entry['fault'].tr('_', ' ')}** ×#{entry['count']} (#{entry['scenario_keys'].join(', ')}): #{entry['recommendation']}"
423
- entry["suggested_tools"].each do |tool|
424
- lines << " - suggested tool `#{tool['name']}`: #{tool['description']}"
425
- end
426
- end
427
- lines
428
- end
429
-
430
607
  def detail_lines
431
608
  lines = [ "", "## Answers", "" ]
432
609
  @results.each do |result|
433
- lines << "### `#{result.scenario.key}` · `#{result.label}` · #{result.status}#{" · score #{result.score.round(2)}" if result.score}"
610
+ lines << "### `#{result.scenario.key}` · `#{result.label}` · #{result.status}" \
611
+ "#{" · score #{Format.score(result.score)}" if result.score}#{" · #{answer_cost(result)}" if result.replay.cost}"
434
612
  lines << ""
435
613
  lines << "> #{result.scenario.prompt}"
436
614
  lines << ""
@@ -445,6 +623,29 @@ module ActiveAgent
445
623
  end
446
624
  lines
447
625
  end
626
+
627
+ # "~$0.0243 · judge ~$0.0015": what one answer cost, and what judging
628
+ # it cost.
629
+ def answer_cost(result)
630
+ text = Format.money(result.replay.cost, estimated: result.estimated_cost?)
631
+ judge_cost = result.judge_usage&.dig("cost")
632
+ text += " · judge #{Format.money(judge_cost, estimated: estimated_usage?(result.judge_usage))}" if judge_cost
633
+ text
634
+ end
635
+
636
+ # Renders after `detail_lines`, whose trailing blank line separates the two sections.
637
+ def recommendation_lines
638
+ return [] if recommendations.empty?
639
+
640
+ lines = [ "## Recommendations", "" ]
641
+ recommendations.each do |entry|
642
+ lines << "- **#{entry['fault'].tr('_', ' ')}** ×#{entry['count']} (#{entry['scenario_keys'].join(', ')}): #{entry['recommendation']}"
643
+ entry["suggested_tools"].each do |tool|
644
+ lines << " - suggested tool `#{tool['name']}`: #{tool['description']}"
645
+ end
646
+ end
647
+ lines << ""
648
+ end
448
649
  end
449
650
  end
450
651
  end