activeagent 1.8.0 → 1.8.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -12,8 +12,8 @@ module ActiveAgent
12
12
  #
13
13
  # Same content as Report#to_markdown, laid out the way the dashboard's
14
14
  # suite card is: header and stat tiles, the MODELS panel with the judge's
15
- # pick and verdict, WHAT TO FIX cards from Report#fix_items, the
16
- # SCENARIOS matrix, and a per-scenario disclosure with every answer.
15
+ # pick and verdict, the SCENARIOS matrix, a per-scenario disclosure with
16
+ # every answer, and WHAT TO FIX cards from Report#fix_items.
17
17
  module ReportHtml
18
18
  THEMES = %w[light dark].freeze
19
19
  ANSWER_LIMIT = 3_000
@@ -44,9 +44,9 @@ module ActiveAgent
44
44
  #{html_stat_tiles}
45
45
  <section class="card">
46
46
  #{html_models_panel}
47
- #{html_fixes}
48
47
  #{html_matrix}
49
48
  #{html_details}
49
+ #{html_fixes}
50
50
  #{html_footer}
51
51
  </section>
52
52
  </main>
@@ -151,16 +151,50 @@ module ActiveAgent
151
151
  "#{format("%.#{n >= 10_000 ? 1 : 2}f", n / 1_000).sub(/\.?0+\z/, '')}s"
152
152
  end
153
153
 
154
- def fmt_cost(value)
155
- value.nil? ? "—" : format("$%.4f", value)
154
+ # Money reads "~$0.0243" when any part of it was estimated from tokens
155
+ # × a model rate rather than reported (Format.money); the footer's
156
+ # legend explains the mark once.
157
+ def fmt_cost(value, estimated: false)
158
+ Format.money(value, estimated: estimated)
156
159
  end
157
160
 
161
+ # A 0..1 score as a whole percent, "—" when there is none.
158
162
  def fmt_score(value)
159
- value.nil? ? "—" : format("%.2f", value)
163
+ Format.score(value)
164
+ end
165
+
166
+ # "14/16 · 88%" — a fraction always carries its percentage.
167
+ def fmt_passes(passed, total)
168
+ Format.passes(passed, total)
169
+ end
170
+
171
+ # Whether the page shows a "~" anywhere, so the footer carries the legend.
172
+ def estimated_anywhere?
173
+ run_costs["estimated"] || summary_by_model.values.any? { |stats| estimated_cost?(stats) }
174
+ end
175
+
176
+ # The tooltip of an estimated figure, worked out from the results it
177
+ # sums: their tokens at the rate the first estimated one recorded.
178
+ # Nothing for a reported figure.
179
+ def cost_title_attr(results, estimated:)
180
+ return "" unless estimated
181
+
182
+ priced = results.select(&:estimated_cost?)
183
+ rate = priced.filter_map(&:cost_rate).first
184
+ title = Format.cost_title(
185
+ input_tokens: priced.sum { |result| result.replay.input_tokens.to_i },
186
+ output_tokens: priced.sum { |result| result.replay.output_tokens.to_i },
187
+ rate: rate
188
+ )
189
+ %( title="#{h(title)}")
160
190
  end
161
191
 
162
- def fmt_mean_score(value)
163
- value.nil? ? "—" : format("%.3f", value)
192
+ def results_for(label)
193
+ @results.select { |result| result.label == label }
194
+ end
195
+
196
+ def judge_estimated?
197
+ judge_usage&.dig("estimated") == true
164
198
  end
165
199
 
166
200
  # --- page ------------------------------------------------------------
@@ -177,17 +211,21 @@ module ActiveAgent
177
211
  "}",
178
212
  DesignTokens.css(scope: ":root.theme-dark", tokens: DesignTokens::DARK, color_scheme: "dark"),
179
213
  STYLES,
180
- ".mx { grid-template-columns: minmax(240px, 1.6fr) 150px repeat(#{@models.size}, minmax(170px, 1fr)); }",
214
+ ".mx { grid-template-columns: minmax(240px, 1.6fr) 150px repeat(#{@models.size}, minmax(170px, 1fr)) 120px; }",
181
215
  # One rule per model: with that chip checked, hide every fix card
182
216
  # attributed to other models (cards attributed to none stay).
183
217
  *@models.each_index.map { |i| ".fix-section:has(input[value=\"m#{i}\"]:checked) .fix[data-models]:not([data-models~=\"m#{i}\"]) { display: none; }" },
184
- ".matrix .inner { min-width: #{390 + 185 * @models.size}px; }"
218
+ ".matrix .inner { min-width: #{520 + 185 * @models.size}px; }"
185
219
  ].join("\n")
186
220
  end
187
221
 
222
+ # One chip per scalar metadata value — an array or a hash (the judge's
223
+ # trace ids, say) is a record, not a label — then the release when the
224
+ # report names one, and the judge with its calls and spend.
188
225
  def html_header(title)
189
- chips = @metadata.to_h.map { |key, value| html_chip(key, value) }
190
- chips << html_chip("judge", judge_name)
226
+ chips = metadata_chips.map { |key, value| html_chip(key, value) }
227
+ chips << html_chip("release", release_label) if release_label
228
+ chips << html_chip("judge", judge_chip_text)
191
229
 
192
230
  <<~HEADER
193
231
  <header>
@@ -201,24 +239,63 @@ module ActiveAgent
201
239
  %(<span class="chip"><b>#{h(key)}</b>#{h(value)}</span>)
202
240
  end
203
241
 
242
+ # The metadata worth a chip: scalar values, minus the judge's trace
243
+ # ids, which the dashboard follows but nobody reads.
244
+ def metadata_chips
245
+ @metadata.to_h.reject { |key, value| key.to_s == "judge_trace_ids" || value.is_a?(Hash) || value.is_a?(Array) || value.nil? }
246
+ end
247
+
248
+ # "1a2b3c4d5e6f · abc1234", or the label the caller gave the release.
249
+ def release_label
250
+ return nil unless release
251
+
252
+ release["label"].presence || [ release["digest"], release["revision"] ].compact_blank.join(" · ").presence
253
+ end
254
+
255
+ # "gpt-5 · 12 calls · ~$0.0315" when the judge spent anything.
256
+ def judge_chip_text
257
+ usage = judge_usage
258
+ return judge_name unless usage
259
+
260
+ parts = [ judge_name, plural(usage["calls"].to_i, "call") ]
261
+ parts << fmt_cost(usage["cost"], estimated: judge_estimated?) if usage["cost"]
262
+ parts.join(" · ")
263
+ end
264
+
204
265
  def html_stat_tiles
205
266
  total = @results.size
206
267
  passed = @results.count(&:passed?)
207
268
  ratio = total.positive? ? passed.to_f / total : 0.0
208
269
  tiles = [
209
270
  html_tile("Scenario runs", total, "#{plural(scenario_cohorts.size, 'scenario')} × #{plural(@models.size, 'model')}"),
210
- html_tile("Pass rate", "#{(ratio * 100).round}%", "#{passed} / #{total} passed", tone: tone_for(ratio)),
271
+ html_tile("Pass rate", Format.percent(total.positive? ? ratio : nil), "#{passed}/#{total} passed", tone: tone_for(ratio)),
211
272
  html_tile("Open faults", total - passed, plural(fix_items.size, "fix item")),
273
+ html_cost_tile,
212
274
  html_tile("Models", @models.size, models_subline)
213
275
  ]
214
276
  %(<section class="stats">#{tiles.join}</section>)
215
277
  end
216
278
 
217
- def html_tile(label, value, sub, tone: nil)
218
- %(<div class="tile"><div class="micro">#{h(label)}</div>) +
279
+ def html_tile(label, value, sub, tone: nil, title: nil)
280
+ %(<div class="tile"#{%( title="#{h(title)}") if title}><div class="micro">#{h(label)}</div>) +
219
281
  %(<div class="value#{" tone-#{tone}" if tone}">#{h(value)}</div><div class="sub">#{h(sub)}</div></div>)
220
282
  end
221
283
 
284
+ # The run's spend: agent plus judge, with the two apart underneath.
285
+ # A run with nothing priced says so rather than showing a blank.
286
+ def html_cost_tile
287
+ costs = run_costs
288
+ sub =
289
+ if costs["total"].nil?
290
+ "nothing priced"
291
+ elsif judge_usage
292
+ "agent #{fmt_cost(costs['cost'], estimated: costs['estimated'])} · judge #{fmt_cost(costs['judge_cost'], estimated: judge_estimated?)}"
293
+ else
294
+ "agent only · #{plural(costs['priced'], 'scenario run')} priced"
295
+ end
296
+ html_tile("Cost", fmt_cost(costs["total"], estimated: costs["estimated"]), sub)
297
+ end
298
+
222
299
  def models_subline
223
300
  if comparing? && winner
224
301
  "judge's pick · #{short_name(model_by_label(winner))}"
@@ -245,17 +322,20 @@ module ActiveAgent
245
322
 
246
323
  # The comparison read across: one row per model, best first (pass rate,
247
324
  # then mean score) — passed, mean score, average latency, average
248
- # tokens per scenario, cost, and the model's typical fault. The blocks
249
- # under it carry the same figures per model with bars and every fault.
325
+ # tokens per scenario, cost (and per scenario), the judge's spend on
326
+ # the cohort when a judge was asked, and the model's typical fault.
327
+ # The blocks under it carry the same figures per model with bars and
328
+ # every fault.
250
329
  def html_comparison_table
251
330
  rows = summary_by_model.sort_by do |label, stats|
252
331
  total = stats["scenarios"].to_i
253
332
  [ total.positive? ? -stats["passed"].to_f / total : 0.0, -(stats["avg_score"] || -1).to_f, @models.index(model_by_label(label)).to_i ]
254
333
  end
334
+ judge_head = judge_usage ? %(<th class="num" title="What the judge spent scoring this model's answers">Judge</th>) : ""
255
335
 
256
336
  <<~TABLE
257
337
  <div class="compare"><table>
258
- <thead><tr><th>Model</th><th class="num">Passed</th><th class="num">Mean score</th><th class="num">Avg latency</th><th class="num" title="Average input + output tokens per scenario">Avg tokens</th><th class="num" title="Cohort spend, and per scenario">Cost</th><th class="fault">Typical fault</th></tr></thead>
338
+ <thead><tr><th>Model</th><th class="num">Passed</th><th class="num">Mean score</th><th class="num">Avg latency</th><th class="num" title="Average input + output tokens per scenario">Avg tokens</th><th class="num" title="Cohort spend, and per scenario">Cost</th>#{judge_head}<th class="fault">Typical fault</th></tr></thead>
259
339
  <tbody>#{rows.map { |label, stats| html_comparison_row(label, stats) }.join}</tbody>
260
340
  </table></div>
261
341
  TABLE
@@ -270,23 +350,34 @@ module ActiveAgent
270
350
  avg_tokens = per.call(stats["input_tokens"].to_i + stats["output_tokens"].to_i)
271
351
  tokens_cell = avg_tokens ? h(fmt_k(avg_tokens.round)) : "—"
272
352
  tokens_title = avg_tokens ? %( title="#{per.call(stats['input_tokens']).to_f.round} in · #{per.call(stats['output_tokens']).to_f.round} out per scenario") : ""
273
- per_cost = per.call(stats["cost"])
274
- cost_cell = stats["cost"].nil? ? "—" : h(fmt_cost(stats["cost"]))
275
- cost_cell += "<span class=\"per\">#{h(fmt_cost(per_cost))}/scenario</span>" if per_cost
353
+ per_cost = cost_per_priced(stats)
354
+ estimated = estimated_cost?(stats)
355
+ cost_cell = h(fmt_cost(stats["cost"], estimated: estimated))
356
+ cost_cell += "<span class=\"per\">#{h(fmt_cost(per_cost, estimated: estimated))}/scenario</span>" if per_cost
357
+ judge_cell = judge_usage ? %(<td class="num">#{html_judge_cost(stats)}</td>) : ""
276
358
 
277
359
  <<~ROW
278
360
  <tr>
279
361
  <td class="model-cell"><span class="name">#{h(short)}</span>#{pick}<span class="provider">#{h(provider)}</span></td>
280
- <td class="num ratio tone-#{tone_for(ratio)}">#{total.positive? ? "#{stats['passed']}/#{total}" : '—'}</td>
281
- <td class="num">#{h(fmt_mean_score(stats['avg_score']))}</td>
362
+ <td class="num ratio tone-#{tone_for(ratio)}">#{h(fmt_passes(stats['passed'], total))}</td>
363
+ <td class="num">#{h(fmt_score(stats['avg_score']))}</td>
282
364
  <td class="num">#{h(fmt_ms(stats['avg_duration_ms']))}</td>
283
365
  <td class="num"#{tokens_title}>#{tokens_cell}</td>
284
- <td class="num">#{cost_cell}</td>
285
- <td class="fault">#{typical_fault_text(label, stats)}</td>
366
+ <td class="num"#{cost_title_attr(results_for(label), estimated: estimated)}>#{cost_cell}</td>
367
+ #{judge_cell}<td class="fault">#{typical_fault_text(label, stats)}</td>
286
368
  </tr>
287
369
  ROW
288
370
  end
289
371
 
372
+ # "~$0.0030<span class="per">3 calls</span>" — the judge's spend on a
373
+ # model's answers; "—" when it was not asked about them.
374
+ def html_judge_cost(stats)
375
+ calls = stats["judge_calls"].to_i
376
+ return "—" if stats["judge_cost"].nil? && calls.zero?
377
+
378
+ h(fmt_cost(stats["judge_cost"], estimated: judge_estimated?)) + %(<span class="per">#{h(plural(calls, 'call'))}</span>)
379
+ end
380
+
290
381
  # "missing content ×2 · refund_window: The answer is missing expected
291
382
  # content: 30." — the model's most frequent fault, and the diagnosis of
292
383
  # the first result that carries it; "no faults" for a clean cohort.
@@ -317,130 +408,21 @@ module ActiveAgent
317
408
  faults = stats["faults"].map { |fault, count| %(<span class="badge error">#{h(fault_name(fault))} ×#{count}</span>) }
318
409
  faults_html = faults.any? ? faults.join : %(<span class="clean">[+] no faults</span>)
319
410
 
411
+ estimated = estimated_cost?(stats)
412
+ judge_line = ""
413
+ if stats["judge_cost"] || stats["judge_calls"].to_i.positive?
414
+ judge_line = %(<span>judge <b>#{h(fmt_cost(stats['judge_cost'], estimated: judge_estimated?))}</b> · #{h(plural(stats['judge_calls'].to_i, 'call'))}</span>)
415
+ end
416
+
320
417
  <<~BLOCK
321
418
  <div class="model">
322
- <div class="line"><span class="name">#{h(short)}</span><span class="provider">#{h(provider)}</span>#{pick}<span class="pass"><span class="bar bar-#{tone}"><span style="width:#{(ratio * 100).round}%"></span></span><span class="ratio tone-#{tone}">#{stats['passed']}/#{total}</span></span></div>
323
- <div class="stats-line"><span>score <b>#{h(fmt_mean_score(stats['avg_score']))}</b></span><span>latency <b>#{h(fmt_ms(stats['avg_duration_ms']))}</b></span><span class="tok"><span class="in">in</span> #{h(fmt_k(stats['input_tokens']))} · <span class="out">out</span> #{h(fmt_k(stats['output_tokens']))}</span><span>cost <b>#{h(fmt_cost(stats['cost']))}</b></span></div>
419
+ <div class="line"><span class="name">#{h(short)}</span><span class="provider">#{h(provider)}</span>#{pick}<span class="pass"><span class="bar bar-#{tone}"><span style="width:#{(ratio * 100).round}%"></span></span><span class="ratio tone-#{tone}">#{h(fmt_passes(stats['passed'], total))}</span></span></div>
420
+ <div class="stats-line"><span>score <b>#{h(fmt_score(stats['avg_score']))}</b></span><span>latency <b>#{h(fmt_ms(stats['avg_duration_ms']))}</b></span><span class="tok"><span class="in">in</span> #{h(fmt_k(stats['input_tokens']))} · <span class="out">out</span> #{h(fmt_k(stats['output_tokens']))}</span><span#{cost_title_attr(results_for(label), estimated: estimated)}>cost <b>#{h(fmt_cost(stats['cost'], estimated: estimated))}</b></span>#{judge_line}</div>
324
421
  <div class="faults">#{faults_html}</div>
325
422
  </div>
326
423
  BLOCK
327
424
  end
328
425
 
329
- # --- WHAT TO FIX -----------------------------------------------------
330
-
331
- # The section stands even for a run with nothing to fix — the dashboard
332
- # keeps it too, so a clean run reads as clean rather than as a page
333
- # missing a section. A report over no results at all has nothing to say.
334
- def html_fixes
335
- return "" if @results.empty?
336
-
337
- items = fix_items
338
- faulted = @results.reject(&:passed?)
339
- meta = "#{plural(items.size, 'item')} · #{plural(faulted.size, 'fault')} across " \
340
- "#{plural(faulted.map { |result| result.scenario.key }.uniq.size, 'scenario')}"
341
- body =
342
- if items.any?
343
- %(<div class="fixes">#{items.map { |item| html_fix_card(item) }.join}</div>)
344
- else
345
- %(<div class="nothing">[+] nothing to fix</div>)
346
- end
347
-
348
- <<~FIXES
349
- <section class="section fix-section" aria-label="Recommendations">
350
- <div class="section-head"><span class="micro">What to fix</span><span class="meta">#{h(meta)}</span></div>
351
- #{html_fix_filter(items) if comparing? && items.any?}
352
- #{body}
353
- </section>
354
- FIXES
355
- end
356
-
357
- # A model filter for the fix cards — a fault one model keeps making is
358
- # that model's to fix, so the list narrows to what was attributed to
359
- # it. Radio chips and stylesheet rules alone (the page carries no
360
- # script): each card names its models in data-models, and a checked
361
- # model hides every card that does not name it. Cards attributed to no
362
- # model (an older run) stay under every filter.
363
- def html_fix_filter(items)
364
- chips = [ %(<label class="chip pick-model"><input type="radio" name="fix-model" value="all" checked><span>all models #{items.size}</span></label>) ]
365
- @models.each_with_index do |spec, index|
366
- count = items.count { |item| Array(item["models"]).empty? || item["models"].include?(spec.label) }
367
- chips << %(<label class="chip pick-model"><input type="radio" name="fix-model" value="m#{index}"><span>#{h(short_name(spec))} #{count}</span></label>)
368
- end
369
- %(<div class="fix-filter"><span class="micro sm">for</span>#{chips.join}</div>)
370
- end
371
-
372
- def fix_model_tokens(item)
373
- labels = Array(item["models"])
374
- return "" if labels.empty?
375
-
376
- labels.filter_map { |label| (index = @models.index(model_by_label(label))) && "m#{index}" }.join(" ")
377
- end
378
-
379
- def html_fix_card(item)
380
- tone = item["kind"] == "instruction" ? "info" : "error"
381
- glyph = tone == "info" ? "[i]" : "[!]"
382
- title = fault_name(item["fault"]) + (item["count"].to_i > 1 ? " ×#{item['count']}" : "")
383
-
384
- parts = [ %(<div class="head"><span class="glyph tone-#{tone}">#{glyph}</span>) +
385
- %(<span class="badge #{tone}">#{h(title)}</span><span class="scope">#{h(fix_scope(item))}</span></div>) ]
386
- parts << %(<p>#{h(item['recommendation'])}</p>) if item["recommendation"].present?
387
- parts << %(<div class="quote">“#{h(item['quote'])}”</div>) if item["quote"].present?
388
- parts << html_fix_tools(item) if item["tools"].any?
389
- parts << html_fix_server(item["server"]) if item["server"]
390
- parts << %(<div class="note">#{h(item['note'])}</div>) if item["note"].present?
391
- parts << html_fix_action(item["action"]) if item["action"]
392
- models = fix_model_tokens(item)
393
- %(<div class="fix"#{%( data-models="#{models}") if models.present?}>#{parts.join}</div>)
394
- end
395
-
396
- def html_fix_tools(item)
397
- chips = item["tools"].map do |tool|
398
- note = tool["note"].presence
399
- %(<span class="tool"><b>#{h(tool['name'])}</b>#{%(<span class="note">#{h(note)}</span>) if note}</span>)
400
- end
401
- %(<div class="tools"><span class="micro sm">#{h(item['tools_label'])}</span><div class="list">#{chips.join}</div></div>)
402
- end
403
-
404
- # "available · not enabled for Assistant", "unknown · not enabled for
405
- # Assistant" — every status but "enabled" leads with the status word, the
406
- # way the dashboard's fix list reads it.
407
- def html_fix_server(server)
408
- badge =
409
- if server["status"] == "enabled"
410
- %(<span class="badge success xs">enabled for #{h(@agent_name)}</span>)
411
- else
412
- %(<span class="badge warning xs">#{h(server['status'].presence || 'unknown')} · not enabled for #{h(@agent_name)}</span>)
413
- end
414
- %(<div class="served"><span>served by</span><b>#{h(server['name'].presence || server['key'])}</b>#{badge}</div>)
415
- end
416
-
417
- # With a route the action is a button; without one, the page can only
418
- # say where in the dashboard the fix lives. The link targets the top
419
- # window: served in the dashboard's report iframe it would otherwise
420
- # open the whole dashboard inside the frame.
421
- def html_fix_action(action)
422
- button = action["path"].present? ? %(<a class="btn" target="_top" href="#{h(action['path'])}">#{h(action['label'])}</a>) : ""
423
- %(<div class="action">#{button}<span class="hint">#{h(action['hint'])}</span></div>)
424
- end
425
-
426
- # "3 scenarios · both models" — the models are worth naming only on a
427
- # comparison run; on a single-model run the count says it all.
428
- def fix_scope(item)
429
- return "#{item['scenario_keys'].join(', ')} · judge suggestion" if item["kind"] == "instruction"
430
-
431
- scenarios = plural(item["scenario_keys"].size, "scenario")
432
- labels = Array(item["models"])
433
- return scenarios unless comparing? && labels.any?
434
-
435
- models =
436
- if labels.size >= @models.size
437
- @models.size == 2 ? "both models" : "all models"
438
- else
439
- labels.map { |label| short_name(model_by_label(label)) }.join(", ")
440
- end
441
- "#{scenarios} · #{models}"
442
- end
443
-
444
426
  # --- SCENARIOS matrix ------------------------------------------------
445
427
 
446
428
  def html_matrix
@@ -448,7 +430,7 @@ module ActiveAgent
448
430
  short, provider = split_label(spec)
449
431
  %(<span class="col"><span class="name">#{h(short)}</span><span class="provider">#{h(provider)}</span></span>)
450
432
  end
451
- rows = [ %(<div class="mx head"><span class="micro sm">Scenario</span><span class="micro sm">Expects</span>#{columns.join}</div>) ]
433
+ rows = [ %(<div class="mx head"><span class="micro sm">Scenario</span><span class="micro sm">Expects</span>#{columns.join}<span class="micro sm cost-head" title="What the scenario cost across every model, and what judging it cost">Cost</span></div>) ]
452
434
  scenario_groups.each do |cohorts|
453
435
  rows << html_group_row(cohorts) if group_name(cohorts.first.first.scenario)
454
436
  cohorts.each { |cohort| rows << html_scenario_row(cohort) }
@@ -476,10 +458,29 @@ module ActiveAgent
476
458
  elsif passed.zero? then " text-error"
477
459
  else ""
478
460
  end
479
- %(<span class="group-pass#{tone}">#{passed}/#{results.size} passed</span>)
461
+ %(<span class="group-pass#{tone}">#{results.empty? ? '—' : "#{h(fmt_passes(passed, results.size))} passed"}</span>)
480
462
  end
481
463
  %(<div class="mx group"><span class="group-name">#{h(name)}</span>) +
482
- %(<span class="count">#{h(plural(cohorts.size, 'scenario'))}</span>#{passes.join}</div>)
464
+ %(<span class="count">#{h(plural(cohorts.size, 'scenario'))}</span>#{passes.join}#{html_group_cost(cohorts)}</div>)
465
+ end
466
+
467
+ # The group's subtotal — its scenarios' costs summed, judge apart — in
468
+ # the trailing column, so the grid stays aligned under a group row.
469
+ def html_group_cost(cohorts)
470
+ costs = cohorts.map { |cohort| scenario_costs[cohort.first.scenario.key] }.compact
471
+ agent = costs.filter_map { |entry| entry["cost"] }
472
+ judge = costs.filter_map { |entry| entry["judge_cost"] }
473
+ estimated = costs.any? { |entry| entry["estimated"] }
474
+ return %(<span class="cost"><span class="muted">—</span></span>) if agent.empty? && judge.empty?
475
+
476
+ html_cost_block(agent.any? ? agent.sum : nil, judge.any? ? judge.sum : nil, estimated: estimated)
477
+ end
478
+
479
+ # "<b>~$0.0243</b><span class="judge">judge ~$0.0015</span>" — a
480
+ # scenario's (or group's) spend across every model.
481
+ def html_cost_block(cost, judge_cost, estimated:)
482
+ judge = judge_cost ? %(<span class="judge">judge #{h(fmt_cost(judge_cost, estimated: estimated))}</span>) : ""
483
+ %(<span class="cost"><b>#{h(fmt_cost(cost, estimated: estimated))}</b>#{judge}</span>)
483
484
  end
484
485
 
485
486
  def html_scenario_row(cohort)
@@ -489,8 +490,14 @@ module ActiveAgent
489
490
  result = cohort.find { |candidate| candidate.label == spec.label }
490
491
  result ? html_result_cell(result) : %(<div class="cell"><div class="top"><span class="muted">—</span></div></div>)
491
492
  end
493
+ costs = scenario_costs[scenario.key] || {}
494
+ total_cell = if costs["cost"].nil? && costs["judge_cost"].nil?
495
+ %(<span class="cost"><span class="muted">—</span></span>)
496
+ else
497
+ html_cost_block(costs["cost"], costs["judge_cost"], estimated: costs["estimated"])
498
+ end
492
499
  %(<div class="mx"><div><div class="key"><a href="##{h(anchor(scenario))}">#{h(scenario.key)}</a></div>) +
493
- %(<div class="prompt">#{h(scenario.prompt)}</div></div><div class="expects">#{expects}</div>#{cells.join}</div>)
500
+ %(<div class="prompt">#{h(scenario.prompt)}</div></div><div class="expects">#{expects}</div>#{cells.join}#{total_cell}</div>)
494
501
  end
495
502
 
496
503
  def html_result_cell(result)
@@ -499,7 +506,25 @@ module ActiveAgent
499
506
  fault = result.fault ? %(<span class="f">#{h(fault_name(result.fault))}</span>) : ""
500
507
  %(<div class="cell"><div class="top"><span class="g tone-#{tone}">#{glyph}</span>) +
501
508
  %(<span class="s tone-#{tone}">#{h(fmt_score(result.score))}</span>#{fault}</div>) +
502
- %(<div class="calls">#{html_calls(result, empty: 'no tools called')}</div></div>)
509
+ %(<div class="calls">#{html_calls(result, empty: 'no tools called')}</div>#{html_cell_cost(result)}</div>)
510
+ end
511
+
512
+ # "~$0.0243 · judge ~$0.0015" under a cell: what this answer cost and
513
+ # what judging it cost. A result with no cost at all shows no line.
514
+ def html_cell_cost(result)
515
+ text = result_cost_text(result)
516
+ return "" if text.nil?
517
+
518
+ %(<div class="cost-line"#{cost_title_attr([ result ], estimated: result.estimated_cost?)}>#{h(text)}</div>)
519
+ end
520
+
521
+ def result_cost_text(result)
522
+ judge_cost = result.judge_usage&.dig("cost")
523
+ return nil if result.replay.cost.nil? && judge_cost.nil?
524
+
525
+ parts = [ fmt_cost(result.replay.cost, estimated: result.estimated_cost?) ]
526
+ parts << "judge #{fmt_cost(judge_cost, estimated: estimated_usage?(result.judge_usage))}" if judge_cost
527
+ parts.join(" · ")
503
528
  end
504
529
 
505
530
  def html_calls(result, empty:)
@@ -580,20 +605,156 @@ module ActiveAgent
580
605
  [
581
606
  replay.duration_ms && fmt_ms(replay.duration_ms),
582
607
  replay.total_tokens.positive? ? "#{fmt_k(replay.total_tokens)} tokens" : nil,
583
- replay.cost && fmt_cost(replay.cost)
608
+ result_cost_text(result)
584
609
  ].compact.join(" · ")
585
610
  end
586
611
 
612
+ # --- WHAT TO FIX -----------------------------------------------------
613
+
614
+ # The section stands even for a run with nothing to fix — the dashboard
615
+ # keeps it too, so a clean run reads as clean rather than as a page
616
+ # missing a section. A report over no results at all has nothing to say.
617
+ def html_fixes
618
+ return "" if @results.empty?
619
+
620
+ items = fix_items
621
+ faulted = @results.reject(&:passed?)
622
+ meta = "#{plural(items.size, 'item')} · #{plural(faulted.size, 'fault')} across " \
623
+ "#{plural(faulted.map { |result| result.scenario.key }.uniq.size, 'scenario')}"
624
+ body =
625
+ if items.any?
626
+ %(<div class="fixes">#{items.map { |item| html_fix_card(item) }.join}</div>)
627
+ else
628
+ %(<div class="nothing">[+] nothing to fix</div>)
629
+ end
630
+
631
+ <<~FIXES
632
+ <section class="section fix-section" aria-label="Recommendations">
633
+ <div class="section-head"><span class="micro">What to fix</span><span class="meta">#{h(meta)}</span></div>
634
+ #{html_fix_filter(items) if comparing? && items.any?}
635
+ #{body}
636
+ </section>
637
+ FIXES
638
+ end
639
+
640
+ # A model filter for the fix cards — a fault one model keeps making is
641
+ # that model's to fix, so the list narrows to what was attributed to
642
+ # it. Radio chips and stylesheet rules alone (the page carries no
643
+ # script): each card names its models in data-models, and a checked
644
+ # model hides every card that does not name it. Cards attributed to no
645
+ # model (an older run) stay under every filter.
646
+ def html_fix_filter(items)
647
+ chips = [ %(<label class="chip pick-model"><input type="radio" name="fix-model" value="all" checked><span>all models #{items.size}</span></label>) ]
648
+ @models.each_with_index do |spec, index|
649
+ count = items.count { |item| Array(item["models"]).empty? || item["models"].include?(spec.label) }
650
+ chips << %(<label class="chip pick-model"><input type="radio" name="fix-model" value="m#{index}"><span>#{h(short_name(spec))} #{count}</span></label>)
651
+ end
652
+ %(<div class="fix-filter"><span class="micro sm">for</span>#{chips.join}</div>)
653
+ end
654
+
655
+ def fix_model_tokens(item)
656
+ labels = Array(item["models"])
657
+ return "" if labels.empty?
658
+
659
+ labels.filter_map { |label| (index = @models.index(model_by_label(label))) && "m#{index}" }.join(" ")
660
+ end
661
+
662
+ def html_fix_card(item)
663
+ tone = item["kind"] == "instruction" ? "info" : "error"
664
+ glyph = tone == "info" ? "[i]" : "[!]"
665
+ title = fault_name(item["fault"]) + (item["count"].to_i > 1 ? " ×#{item['count']}" : "")
666
+
667
+ parts = [ %(<div class="head"><span class="glyph tone-#{tone}">#{glyph}</span>) +
668
+ %(<span class="badge #{tone}">#{h(title)}</span><span class="scope">#{h(fix_scope(item))}</span></div>) ]
669
+ parts << %(<p>#{h(item['recommendation'])}</p>) if item["recommendation"].present?
670
+ parts << %(<div class="quote">“#{h(item['quote'])}”</div>) if item["quote"].present?
671
+ parts << html_fix_tools(item) if item["tools"].any?
672
+ parts << html_fix_server(item["server"]) if item["server"]
673
+ parts << %(<div class="note">#{h(item['note'])}</div>) if item["note"].present?
674
+ parts << html_fix_action(item["action"]) if item["action"]
675
+ models = fix_model_tokens(item)
676
+ %(<div class="fix"#{%( data-models="#{models}") if models.present?}>#{parts.join}</div>)
677
+ end
678
+
679
+ def html_fix_tools(item)
680
+ chips = item["tools"].map do |tool|
681
+ note = tool["note"].presence
682
+ %(<span class="tool"><b>#{h(tool['name'])}</b>#{%(<span class="note">#{h(note)}</span>) if note}</span>)
683
+ end
684
+ %(<div class="tools"><span class="micro sm">#{h(item['tools_label'])}</span><div class="list">#{chips.join}</div></div>)
685
+ end
686
+
687
+ # "available · not enabled for Assistant", "unknown · not enabled for
688
+ # Assistant" — every status but "enabled" leads with the status word, the
689
+ # way the dashboard's fix list reads it.
690
+ def html_fix_server(server)
691
+ badge =
692
+ if server["status"] == "enabled"
693
+ %(<span class="badge success xs">enabled for #{h(@agent_name)}</span>)
694
+ else
695
+ %(<span class="badge warning xs">#{h(server['status'].presence || 'unknown')} · not enabled for #{h(@agent_name)}</span>)
696
+ end
697
+ %(<div class="served"><span>served by</span><b>#{h(server['name'].presence || server['key'])}</b>#{badge}</div>)
698
+ end
699
+
700
+ # With a route the action is a button; without one, the page can only
701
+ # say where in the dashboard the fix lives. The link targets the top
702
+ # window: served in the dashboard's report iframe it would otherwise
703
+ # open the whole dashboard inside the frame.
704
+ def html_fix_action(action)
705
+ button = action["path"].present? ? %(<a class="btn" target="_top" href="#{h(action['path'])}">#{h(action['label'])}</a>) : ""
706
+ %(<div class="action">#{button}<span class="hint">#{h(action['hint'])}</span></div>)
707
+ end
708
+
709
+ # "3 scenarios · both models" — the models are worth naming only on a
710
+ # comparison run; on a single-model run the count says it all.
711
+ def fix_scope(item)
712
+ return "#{item['scenario_keys'].join(', ')} · judge suggestion" if item["kind"] == "instruction"
713
+
714
+ scenarios = plural(item["scenario_keys"].size, "scenario")
715
+ labels = Array(item["models"])
716
+ return scenarios unless comparing? && labels.any?
717
+
718
+ models =
719
+ if labels.size >= @models.size
720
+ @models.size == 2 ? "both models" : "all models"
721
+ else
722
+ labels.map { |label| short_name(model_by_label(label)) }.join(", ")
723
+ end
724
+ "#{scenarios} · #{models}"
725
+ end
726
+
587
727
  # --- footer ----------------------------------------------------------
588
728
 
729
+ # The run's terms: the judge, the criteria and the pass mark, what it
730
+ # cost on each side, the metadata, and — once, when any figure on the
731
+ # page carries a "~" — what the mark means.
589
732
  def html_footer
590
733
  criteria = criterion_keys.map { |key| key.to_s.tr("_", " ") }.join(" · ")
591
734
  spans = [ %(<span class="nowrap">judge #{h(judge_name)}</span>) ]
592
735
  spans << %(<span class="criteria">criteria #{h(criteria)}</span>) if criteria.present?
593
- spans.concat(@metadata.to_h.map { |key, value| %(<span class="nowrap">#{h(key)} #{h(value)}</span>) })
736
+ spans << %(<span class="nowrap">#{h(Format.threshold(@threshold))}</span>)
737
+ spans << %(<span class="nowrap">#{h(footer_cost_text)}</span>) if footer_cost_text
738
+ spans << %(<span class="nowrap">release #{h(release_label)}</span>) if release_label
739
+ spans.concat(metadata_chips.map { |key, value| %(<span class="nowrap">#{h(key)} #{h(value)}</span>) })
740
+ spans << %(<span class="legend">#{h(Format::LEGEND)}</span>) if estimated_anywhere?
594
741
  %(<footer>#{spans.join}</footer>)
595
742
  end
596
743
 
744
+ # "cost ~$0.0412 · agent ~$0.0397 · judge ~$0.0015", or nil when
745
+ # nothing was priced.
746
+ def footer_cost_text
747
+ costs = run_costs
748
+ return nil if costs["total"].nil?
749
+
750
+ parts = [ "cost #{fmt_cost(costs['total'], estimated: costs['estimated'])}" ]
751
+ if judge_usage
752
+ parts << "agent #{fmt_cost(costs['cost'], estimated: costs['estimated'])}"
753
+ parts << "judge #{fmt_cost(costs['judge_cost'], estimated: judge_estimated?)}"
754
+ end
755
+ parts.join(" · ")
756
+ end
757
+
597
758
  # Colors only through the token variables; radii 4 badges · 6 chips ·
598
759
  # 8 controls · 10 nested panels · 12 cards · 999 bars; no shadows.
599
760
  STYLES = <<~CSS.freeze
@@ -701,6 +862,10 @@ module ActiveAgent
701
862
  .group-name { font-size: 12px; font-weight: 600; }
702
863
  .count { font-family: var(--font-mono); font-size: 11px; color: var(--color-text-muted); }
703
864
  .group-pass { font-family: var(--font-mono); font-size: 11px; font-weight: 600; color: var(--color-text-cell); }
865
+ .mx .cost { display: flex; flex-direction: column; gap: 1px; min-width: 0; font-family: var(--font-mono); font-size: 11px; color: var(--color-text-secondary); text-align: right; }
866
+ .mx .cost b { font-weight: 600; color: var(--color-text-primary); }
867
+ .mx .cost .judge { color: var(--color-text-muted); white-space: nowrap; }
868
+ .mx .cost-head { text-align: right; }
704
869
  .key { font-family: var(--font-mono); font-size: 11px; color: var(--color-text-muted); margin-bottom: 2px; }
705
870
  .key a { color: inherit; }
706
871
  .prompt { font-size: 13px; line-height: 18px; color: var(--color-text-primary); }
@@ -714,6 +879,7 @@ module ActiveAgent
714
879
  .calls { display: flex; flex-wrap: wrap; gap: 2px 8px; font-family: var(--font-mono); font-size: 11px; color: var(--color-text-muted); }
715
880
  .call-hit { color: var(--color-success-text); font-weight: 600; }
716
881
  .call-err { color: var(--color-error); font-weight: 600; }
882
+ .cost-line { font-family: var(--font-mono); font-size: 11px; color: var(--color-text-secondary); white-space: nowrap; }
717
883
  .details { display: flex; flex-direction: column; gap: 8px; }
718
884
  details { border: 1px solid var(--color-border-light); border-radius: 10px; overflow: hidden; }
719
885
  summary { display: flex; align-items: center; gap: 10px; padding: 10px 12px; cursor: pointer; list-style: none; flex-wrap: wrap; }
@@ -739,6 +905,7 @@ module ActiveAgent
739
905
  footer { display: flex; align-items: center; gap: 16px; flex-wrap: wrap; padding-top: 12px; border-top: 1px solid var(--color-border-light); font-family: var(--font-mono); font-size: 11px; color: var(--color-text-muted); }
740
906
  footer .criteria { min-width: 0; }
741
907
  footer .nowrap { white-space: nowrap; }
908
+ footer .legend { margin-left: auto; white-space: nowrap; color: var(--color-text-muted); }
742
909
  CSS
743
910
  end
744
911
  end