activeagent 1.7.2 → 1.8.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -12,8 +12,8 @@ module ActiveAgent
12
12
  #
13
13
  # Same content as Report#to_markdown, laid out the way the dashboard's
14
14
  # suite card is: header and stat tiles, the MODELS panel with the judge's
15
- # pick and verdict, WHAT TO FIX cards from Report#fix_items, the
16
- # SCENARIOS matrix, and a per-scenario disclosure with every answer.
15
+ # pick and verdict, the SCENARIOS matrix, a per-scenario disclosure with
16
+ # every answer, and WHAT TO FIX cards from Report#fix_items.
17
17
  module ReportHtml
18
18
  THEMES = %w[light dark].freeze
19
19
  ANSWER_LIMIT = 3_000
@@ -44,9 +44,9 @@ module ActiveAgent
44
44
  #{html_stat_tiles}
45
45
  <section class="card">
46
46
  #{html_models_panel}
47
- #{html_fixes}
48
47
  #{html_matrix}
49
48
  #{html_details}
49
+ #{html_fixes}
50
50
  #{html_footer}
51
51
  </section>
52
52
  </main>
@@ -151,16 +151,50 @@ module ActiveAgent
151
151
  "#{format("%.#{n >= 10_000 ? 1 : 2}f", n / 1_000).sub(/\.?0+\z/, '')}s"
152
152
  end
153
153
 
154
- def fmt_cost(value)
155
- value.nil? ? "—" : format("$%.4f", value)
154
+ # Money reads "~$0.0243" when any part of it was estimated from tokens
155
+ # × a model rate rather than reported (Format.money); the footer's
156
+ # legend explains the mark once.
157
+ def fmt_cost(value, estimated: false)
158
+ Format.money(value, estimated: estimated)
156
159
  end
157
160
 
161
+ # A 0..1 score as a whole percent, "—" when there is none.
158
162
  def fmt_score(value)
159
- value.nil? ? "—" : format("%.2f", value)
163
+ Format.score(value)
164
+ end
165
+
166
+ # "14/16 · 88%" — a fraction always carries its percentage.
167
+ def fmt_passes(passed, total)
168
+ Format.passes(passed, total)
160
169
  end
161
170
 
162
- def fmt_mean_score(value)
163
- value.nil? ? "—" : format("%.3f", value)
171
+ # Whether the page shows a "~" anywhere, so the footer carries the legend.
172
+ def estimated_anywhere?
173
+ run_costs["estimated"] || summary_by_model.values.any? { |stats| estimated_cost?(stats) }
174
+ end
175
+
176
+ # The tooltip of an estimated figure, worked out from the results it
177
+ # sums: their tokens at the rate the first estimated one recorded.
178
+ # Nothing for a reported figure.
179
+ def cost_title_attr(results, estimated:)
180
+ return "" unless estimated
181
+
182
+ priced = results.select(&:estimated_cost?)
183
+ rate = priced.filter_map(&:cost_rate).first
184
+ title = Format.cost_title(
185
+ input_tokens: priced.sum { |result| result.replay.input_tokens.to_i },
186
+ output_tokens: priced.sum { |result| result.replay.output_tokens.to_i },
187
+ rate: rate
188
+ )
189
+ %( title="#{h(title)}")
190
+ end
191
+
192
+ def results_for(label)
193
+ @results.select { |result| result.label == label }
194
+ end
195
+
196
+ def judge_estimated?
197
+ judge_usage&.dig("estimated") == true
164
198
  end
165
199
 
166
200
  # --- page ------------------------------------------------------------
@@ -177,14 +211,21 @@ module ActiveAgent
177
211
  "}",
178
212
  DesignTokens.css(scope: ":root.theme-dark", tokens: DesignTokens::DARK, color_scheme: "dark"),
179
213
  STYLES,
180
- ".mx { grid-template-columns: minmax(240px, 1.6fr) 150px repeat(#{@models.size}, minmax(170px, 1fr)); }",
181
- ".matrix .inner { min-width: #{390 + 185 * @models.size}px; }"
214
+ ".mx { grid-template-columns: minmax(240px, 1.6fr) 150px repeat(#{@models.size}, minmax(170px, 1fr)) 120px; }",
215
+ # One rule per model: with that chip checked, hide every fix card
216
+ # attributed to other models (cards attributed to none stay).
217
+ *@models.each_index.map { |i| ".fix-section:has(input[value=\"m#{i}\"]:checked) .fix[data-models]:not([data-models~=\"m#{i}\"]) { display: none; }" },
218
+ ".matrix .inner { min-width: #{520 + 185 * @models.size}px; }"
182
219
  ].join("\n")
183
220
  end
184
221
 
222
+ # One chip per scalar metadata value — an array or a hash (the judge's
223
+ # trace ids, say) is a record, not a label — then the release when the
224
+ # report names one, and the judge with its calls and spend.
185
225
  def html_header(title)
186
- chips = @metadata.to_h.map { |key, value| html_chip(key, value) }
187
- chips << html_chip("judge", judge_name)
226
+ chips = metadata_chips.map { |key, value| html_chip(key, value) }
227
+ chips << html_chip("release", release_label) if release_label
228
+ chips << html_chip("judge", judge_chip_text)
188
229
 
189
230
  <<~HEADER
190
231
  <header>
@@ -198,24 +239,63 @@ module ActiveAgent
198
239
  %(<span class="chip"><b>#{h(key)}</b>#{h(value)}</span>)
199
240
  end
200
241
 
242
+ # The metadata worth a chip: scalar values, minus the judge's trace
243
+ # ids, which the dashboard follows but nobody reads.
244
+ def metadata_chips
245
+ @metadata.to_h.reject { |key, value| key.to_s == "judge_trace_ids" || value.is_a?(Hash) || value.is_a?(Array) || value.nil? }
246
+ end
247
+
248
+ # "1a2b3c4d5e6f · abc1234", or the label the caller gave the release.
249
+ def release_label
250
+ return nil unless release
251
+
252
+ release["label"].presence || [ release["digest"], release["revision"] ].compact_blank.join(" · ").presence
253
+ end
254
+
255
+ # "gpt-5 · 12 calls · ~$0.0315" when the judge spent anything.
256
+ def judge_chip_text
257
+ usage = judge_usage
258
+ return judge_name unless usage
259
+
260
+ parts = [ judge_name, plural(usage["calls"].to_i, "call") ]
261
+ parts << fmt_cost(usage["cost"], estimated: judge_estimated?) if usage["cost"]
262
+ parts.join(" · ")
263
+ end
264
+
201
265
  def html_stat_tiles
202
266
  total = @results.size
203
267
  passed = @results.count(&:passed?)
204
268
  ratio = total.positive? ? passed.to_f / total : 0.0
205
269
  tiles = [
206
270
  html_tile("Scenario runs", total, "#{plural(scenario_cohorts.size, 'scenario')} × #{plural(@models.size, 'model')}"),
207
- html_tile("Pass rate", "#{(ratio * 100).round}%", "#{passed} / #{total} passed", tone: tone_for(ratio)),
271
+ html_tile("Pass rate", Format.percent(total.positive? ? ratio : nil), "#{passed}/#{total} passed", tone: tone_for(ratio)),
208
272
  html_tile("Open faults", total - passed, plural(fix_items.size, "fix item")),
273
+ html_cost_tile,
209
274
  html_tile("Models", @models.size, models_subline)
210
275
  ]
211
276
  %(<section class="stats">#{tiles.join}</section>)
212
277
  end
213
278
 
214
- def html_tile(label, value, sub, tone: nil)
215
- %(<div class="tile"><div class="micro">#{h(label)}</div>) +
279
+ def html_tile(label, value, sub, tone: nil, title: nil)
280
+ %(<div class="tile"#{%( title="#{h(title)}") if title}><div class="micro">#{h(label)}</div>) +
216
281
  %(<div class="value#{" tone-#{tone}" if tone}">#{h(value)}</div><div class="sub">#{h(sub)}</div></div>)
217
282
  end
218
283
 
284
+ # The run's spend: agent plus judge, with the two apart underneath.
285
+ # A run with nothing priced says so rather than showing a blank.
286
+ def html_cost_tile
287
+ costs = run_costs
288
+ sub =
289
+ if costs["total"].nil?
290
+ "nothing priced"
291
+ elsif judge_usage
292
+ "agent #{fmt_cost(costs['cost'], estimated: costs['estimated'])} · judge #{fmt_cost(costs['judge_cost'], estimated: judge_estimated?)}"
293
+ else
294
+ "agent only · #{plural(costs['priced'], 'scenario run')} priced"
295
+ end
296
+ html_tile("Cost", fmt_cost(costs["total"], estimated: costs["estimated"]), sub)
297
+ end
298
+
219
299
  def models_subline
220
300
  if comparing? && winner
221
301
  "judge's pick · #{short_name(model_by_label(winner))}"
@@ -233,12 +313,92 @@ module ActiveAgent
233
313
  <<~PANEL
234
314
  <div class="panel">
235
315
  <div class="panel-head"><span class="micro">Models</span><span class="right">judged by #{h(judged_by)}</span></div>
316
+ #{html_comparison_table if comparing?}
236
317
  #{blocks.join}
237
318
  #{verdict_row}
238
319
  </div>
239
320
  PANEL
240
321
  end
241
322
 
323
+ # The comparison read across: one row per model, best first (pass rate,
324
+ # then mean score) — passed, mean score, average latency, average
325
+ # tokens per scenario, cost (and per scenario), the judge's spend on
326
+ # the cohort when a judge was asked, and the model's typical fault.
327
+ # The blocks under it carry the same figures per model with bars and
328
+ # every fault.
329
+ def html_comparison_table
330
+ rows = summary_by_model.sort_by do |label, stats|
331
+ total = stats["scenarios"].to_i
332
+ [ total.positive? ? -stats["passed"].to_f / total : 0.0, -(stats["avg_score"] || -1).to_f, @models.index(model_by_label(label)).to_i ]
333
+ end
334
+ judge_head = judge_usage ? %(<th class="num" title="What the judge spent scoring this model's answers">Judge</th>) : ""
335
+
336
+ <<~TABLE
337
+ <div class="compare"><table>
338
+ <thead><tr><th>Model</th><th class="num">Passed</th><th class="num">Mean score</th><th class="num">Avg latency</th><th class="num" title="Average input + output tokens per scenario">Avg tokens</th><th class="num" title="Cohort spend, and per scenario">Cost</th>#{judge_head}<th class="fault">Typical fault</th></tr></thead>
339
+ <tbody>#{rows.map { |label, stats| html_comparison_row(label, stats) }.join}</tbody>
340
+ </table></div>
341
+ TABLE
342
+ end
343
+
344
+ def html_comparison_row(label, stats)
345
+ short, provider = split_label(model_by_label(label))
346
+ total = stats["scenarios"].to_i
347
+ ratio = total.positive? ? stats["passed"].to_f / total : 0.0
348
+ pick = comparing? && winner == label ? %(<span class="pick" title="picked by the judge">★ pick</span>) : ""
349
+ per = ->(value) { value.nil? || total.zero? ? nil : value.to_f / total }
350
+ avg_tokens = per.call(stats["input_tokens"].to_i + stats["output_tokens"].to_i)
351
+ tokens_cell = avg_tokens ? h(fmt_k(avg_tokens.round)) : "—"
352
+ tokens_title = avg_tokens ? %( title="#{per.call(stats['input_tokens']).to_f.round} in · #{per.call(stats['output_tokens']).to_f.round} out per scenario") : ""
353
+ per_cost = cost_per_priced(stats)
354
+ estimated = estimated_cost?(stats)
355
+ cost_cell = h(fmt_cost(stats["cost"], estimated: estimated))
356
+ cost_cell += "<span class=\"per\">#{h(fmt_cost(per_cost, estimated: estimated))}/scenario</span>" if per_cost
357
+ judge_cell = judge_usage ? %(<td class="num">#{html_judge_cost(stats)}</td>) : ""
358
+
359
+ <<~ROW
360
+ <tr>
361
+ <td class="model-cell"><span class="name">#{h(short)}</span>#{pick}<span class="provider">#{h(provider)}</span></td>
362
+ <td class="num ratio tone-#{tone_for(ratio)}">#{h(fmt_passes(stats['passed'], total))}</td>
363
+ <td class="num">#{h(fmt_score(stats['avg_score']))}</td>
364
+ <td class="num">#{h(fmt_ms(stats['avg_duration_ms']))}</td>
365
+ <td class="num"#{tokens_title}>#{tokens_cell}</td>
366
+ <td class="num"#{cost_title_attr(results_for(label), estimated: estimated)}>#{cost_cell}</td>
367
+ #{judge_cell}<td class="fault">#{typical_fault_text(label, stats)}</td>
368
+ </tr>
369
+ ROW
370
+ end
371
+
372
+ # "~$0.0030<span class="per">3 calls</span>" — the judge's spend on a
373
+ # model's answers; "—" when it was not asked about them.
374
+ def html_judge_cost(stats)
375
+ calls = stats["judge_calls"].to_i
376
+ return "—" if stats["judge_cost"].nil? && calls.zero?
377
+
378
+ h(fmt_cost(stats["judge_cost"], estimated: judge_estimated?)) + %(<span class="per">#{h(plural(calls, 'call'))}</span>)
379
+ end
380
+
381
+ # "missing content ×2 · refund_window: The answer is missing expected
382
+ # content: 30." — the model's most frequent fault, and the diagnosis of
383
+ # the first result that carries it; "no faults" for a clean cohort.
384
+ def typical_fault_text(label, stats)
385
+ tally = stats["faults"] || {}
386
+ return %(<span class="clean">no faults</span>) if tally.empty?
387
+
388
+ mine = @results.select { |result| result.label == label }
389
+ example_of = ->(fault) { mine.find { |result| result.fault == fault } }
390
+ # Most frequent first; between equals, a fault a result can explain,
391
+ # then a specific fault over the judge's catch-all, then the name.
392
+ fault, count = tally.min_by { |name, n| [ -n, example_of.call(name) ? 0 : 1, name == "low_quality" ? 1 : 0, name ] }
393
+ example = example_of.call(fault)
394
+ head = "#{fault_name(fault)} ×#{count}"
395
+ return h(head) unless example&.summary.present?
396
+
397
+ detail = "#{example.scenario.key}: #{example.summary}"
398
+ detail = "#{detail[0, 119]}…" if detail.length > 120
399
+ "#{h(head)} <span class=\"detail\">· #{h(detail)}</span>"
400
+ end
401
+
242
402
  def html_model_block(label, stats)
243
403
  short, provider = split_label(model_by_label(label))
244
404
  total = stats["scenarios"]
@@ -248,106 +408,21 @@ module ActiveAgent
248
408
  faults = stats["faults"].map { |fault, count| %(<span class="badge error">#{h(fault_name(fault))} ×#{count}</span>) }
249
409
  faults_html = faults.any? ? faults.join : %(<span class="clean">[+] no faults</span>)
250
410
 
411
+ estimated = estimated_cost?(stats)
412
+ judge_line = ""
413
+ if stats["judge_cost"] || stats["judge_calls"].to_i.positive?
414
+ judge_line = %(<span>judge <b>#{h(fmt_cost(stats['judge_cost'], estimated: judge_estimated?))}</b> · #{h(plural(stats['judge_calls'].to_i, 'call'))}</span>)
415
+ end
416
+
251
417
  <<~BLOCK
252
418
  <div class="model">
253
- <div class="line"><span class="name">#{h(short)}</span><span class="provider">#{h(provider)}</span>#{pick}<span class="pass"><span class="bar bar-#{tone}"><span style="width:#{(ratio * 100).round}%"></span></span><span class="ratio tone-#{tone}">#{stats['passed']}/#{total}</span></span></div>
254
- <div class="stats-line"><span>score <b>#{h(fmt_mean_score(stats['avg_score']))}</b></span><span>latency <b>#{h(fmt_ms(stats['avg_duration_ms']))}</b></span><span class="tok"><span class="in">in</span> #{h(fmt_k(stats['input_tokens']))} · <span class="out">out</span> #{h(fmt_k(stats['output_tokens']))}</span><span>cost <b>#{h(fmt_cost(stats['cost']))}</b></span></div>
419
+ <div class="line"><span class="name">#{h(short)}</span><span class="provider">#{h(provider)}</span>#{pick}<span class="pass"><span class="bar bar-#{tone}"><span style="width:#{(ratio * 100).round}%"></span></span><span class="ratio tone-#{tone}">#{h(fmt_passes(stats['passed'], total))}</span></span></div>
420
+ <div class="stats-line"><span>score <b>#{h(fmt_score(stats['avg_score']))}</b></span><span>latency <b>#{h(fmt_ms(stats['avg_duration_ms']))}</b></span><span class="tok"><span class="in">in</span> #{h(fmt_k(stats['input_tokens']))} · <span class="out">out</span> #{h(fmt_k(stats['output_tokens']))}</span><span#{cost_title_attr(results_for(label), estimated: estimated)}>cost <b>#{h(fmt_cost(stats['cost'], estimated: estimated))}</b></span>#{judge_line}</div>
255
421
  <div class="faults">#{faults_html}</div>
256
422
  </div>
257
423
  BLOCK
258
424
  end
259
425
 
260
- # --- WHAT TO FIX -----------------------------------------------------
261
-
262
- # The section stands even for a run with nothing to fix — the dashboard
263
- # keeps it too, so a clean run reads as clean rather than as a page
264
- # missing a section. A report over no results at all has nothing to say.
265
- def html_fixes
266
- return "" if @results.empty?
267
-
268
- items = fix_items
269
- faulted = @results.reject(&:passed?)
270
- meta = "#{plural(items.size, 'item')} · #{plural(faulted.size, 'fault')} across " \
271
- "#{plural(faulted.map { |result| result.scenario.key }.uniq.size, 'scenario')}"
272
- body =
273
- if items.any?
274
- %(<div class="fixes">#{items.map { |item| html_fix_card(item) }.join}</div>)
275
- else
276
- %(<div class="nothing">[+] nothing to fix</div>)
277
- end
278
-
279
- <<~FIXES
280
- <section class="section" aria-label="Recommendations">
281
- <div class="section-head"><span class="micro">What to fix</span><span class="meta">#{h(meta)}</span></div>
282
- #{body}
283
- </section>
284
- FIXES
285
- end
286
-
287
- def html_fix_card(item)
288
- tone = item["kind"] == "instruction" ? "info" : "error"
289
- glyph = tone == "info" ? "[i]" : "[!]"
290
- title = fault_name(item["fault"]) + (item["count"].to_i > 1 ? " ×#{item['count']}" : "")
291
-
292
- parts = [ %(<div class="head"><span class="glyph tone-#{tone}">#{glyph}</span>) +
293
- %(<span class="badge #{tone}">#{h(title)}</span><span class="scope">#{h(fix_scope(item))}</span></div>) ]
294
- parts << %(<p>#{h(item['recommendation'])}</p>) if item["recommendation"].present?
295
- parts << %(<div class="quote">“#{h(item['quote'])}”</div>) if item["quote"].present?
296
- parts << html_fix_tools(item) if item["tools"].any?
297
- parts << html_fix_server(item["server"]) if item["server"]
298
- parts << %(<div class="note">#{h(item['note'])}</div>) if item["note"].present?
299
- parts << html_fix_action(item["action"]) if item["action"]
300
- %(<div class="fix">#{parts.join}</div>)
301
- end
302
-
303
- def html_fix_tools(item)
304
- chips = item["tools"].map do |tool|
305
- note = tool["note"].presence
306
- %(<span class="tool"><b>#{h(tool['name'])}</b>#{%(<span class="note">#{h(note)}</span>) if note}</span>)
307
- end
308
- %(<div class="tools"><span class="micro sm">#{h(item['tools_label'])}</span><div class="list">#{chips.join}</div></div>)
309
- end
310
-
311
- # "available · not enabled for Assistant", "unknown · not enabled for
312
- # Assistant" — every status but "enabled" leads with the status word, the
313
- # way the dashboard's fix list reads it.
314
- def html_fix_server(server)
315
- badge =
316
- if server["status"] == "enabled"
317
- %(<span class="badge success xs">enabled for #{h(@agent_name)}</span>)
318
- else
319
- %(<span class="badge warning xs">#{h(server['status'].presence || 'unknown')} · not enabled for #{h(@agent_name)}</span>)
320
- end
321
- %(<div class="served"><span>served by</span><b>#{h(server['name'].presence || server['key'])}</b>#{badge}</div>)
322
- end
323
-
324
- # With a route the action is a button; without one, the page can only
325
- # say where in the dashboard the fix lives. The link targets the top
326
- # window: served in the dashboard's report iframe it would otherwise
327
- # open the whole dashboard inside the frame.
328
- def html_fix_action(action)
329
- button = action["path"].present? ? %(<a class="btn" target="_top" href="#{h(action['path'])}">#{h(action['label'])}</a>) : ""
330
- %(<div class="action">#{button}<span class="hint">#{h(action['hint'])}</span></div>)
331
- end
332
-
333
- # "3 scenarios · both models" — the models are worth naming only on a
334
- # comparison run; on a single-model run the count says it all.
335
- def fix_scope(item)
336
- return "#{item['scenario_keys'].join(', ')} · judge suggestion" if item["kind"] == "instruction"
337
-
338
- scenarios = plural(item["scenario_keys"].size, "scenario")
339
- labels = Array(item["models"])
340
- return scenarios unless comparing? && labels.any?
341
-
342
- models =
343
- if labels.size >= @models.size
344
- @models.size == 2 ? "both models" : "all models"
345
- else
346
- labels.map { |label| short_name(model_by_label(label)) }.join(", ")
347
- end
348
- "#{scenarios} · #{models}"
349
- end
350
-
351
426
  # --- SCENARIOS matrix ------------------------------------------------
352
427
 
353
428
  def html_matrix
@@ -355,7 +430,7 @@ module ActiveAgent
355
430
  short, provider = split_label(spec)
356
431
  %(<span class="col"><span class="name">#{h(short)}</span><span class="provider">#{h(provider)}</span></span>)
357
432
  end
358
- rows = [ %(<div class="mx head"><span class="micro sm">Scenario</span><span class="micro sm">Expects</span>#{columns.join}</div>) ]
433
+ rows = [ %(<div class="mx head"><span class="micro sm">Scenario</span><span class="micro sm">Expects</span>#{columns.join}<span class="micro sm cost-head" title="What the scenario cost across every model, and what judging it cost">Cost</span></div>) ]
359
434
  scenario_groups.each do |cohorts|
360
435
  rows << html_group_row(cohorts) if group_name(cohorts.first.first.scenario)
361
436
  cohorts.each { |cohort| rows << html_scenario_row(cohort) }
@@ -383,10 +458,29 @@ module ActiveAgent
383
458
  elsif passed.zero? then " text-error"
384
459
  else ""
385
460
  end
386
- %(<span class="group-pass#{tone}">#{passed}/#{results.size} passed</span>)
461
+ %(<span class="group-pass#{tone}">#{results.empty? ? '—' : "#{h(fmt_passes(passed, results.size))} passed"}</span>)
387
462
  end
388
463
  %(<div class="mx group"><span class="group-name">#{h(name)}</span>) +
389
- %(<span class="count">#{h(plural(cohorts.size, 'scenario'))}</span>#{passes.join}</div>)
464
+ %(<span class="count">#{h(plural(cohorts.size, 'scenario'))}</span>#{passes.join}#{html_group_cost(cohorts)}</div>)
465
+ end
466
+
467
+ # The group's subtotal — its scenarios' costs summed, judge apart — in
468
+ # the trailing column, so the grid stays aligned under a group row.
469
+ def html_group_cost(cohorts)
470
+ costs = cohorts.map { |cohort| scenario_costs[cohort.first.scenario.key] }.compact
471
+ agent = costs.filter_map { |entry| entry["cost"] }
472
+ judge = costs.filter_map { |entry| entry["judge_cost"] }
473
+ estimated = costs.any? { |entry| entry["estimated"] }
474
+ return %(<span class="cost"><span class="muted">—</span></span>) if agent.empty? && judge.empty?
475
+
476
+ html_cost_block(agent.any? ? agent.sum : nil, judge.any? ? judge.sum : nil, estimated: estimated)
477
+ end
478
+
479
+ # "<b>~$0.0243</b><span class="judge">judge ~$0.0015</span>" — a
480
+ # scenario's (or group's) spend across every model.
481
+ def html_cost_block(cost, judge_cost, estimated:)
482
+ judge = judge_cost ? %(<span class="judge">judge #{h(fmt_cost(judge_cost, estimated: estimated))}</span>) : ""
483
+ %(<span class="cost"><b>#{h(fmt_cost(cost, estimated: estimated))}</b>#{judge}</span>)
390
484
  end
391
485
 
392
486
  def html_scenario_row(cohort)
@@ -396,8 +490,14 @@ module ActiveAgent
396
490
  result = cohort.find { |candidate| candidate.label == spec.label }
397
491
  result ? html_result_cell(result) : %(<div class="cell"><div class="top"><span class="muted">—</span></div></div>)
398
492
  end
493
+ costs = scenario_costs[scenario.key] || {}
494
+ total_cell = if costs["cost"].nil? && costs["judge_cost"].nil?
495
+ %(<span class="cost"><span class="muted">—</span></span>)
496
+ else
497
+ html_cost_block(costs["cost"], costs["judge_cost"], estimated: costs["estimated"])
498
+ end
399
499
  %(<div class="mx"><div><div class="key"><a href="##{h(anchor(scenario))}">#{h(scenario.key)}</a></div>) +
400
- %(<div class="prompt">#{h(scenario.prompt)}</div></div><div class="expects">#{expects}</div>#{cells.join}</div>)
500
+ %(<div class="prompt">#{h(scenario.prompt)}</div></div><div class="expects">#{expects}</div>#{cells.join}#{total_cell}</div>)
401
501
  end
402
502
 
403
503
  def html_result_cell(result)
@@ -406,7 +506,25 @@ module ActiveAgent
406
506
  fault = result.fault ? %(<span class="f">#{h(fault_name(result.fault))}</span>) : ""
407
507
  %(<div class="cell"><div class="top"><span class="g tone-#{tone}">#{glyph}</span>) +
408
508
  %(<span class="s tone-#{tone}">#{h(fmt_score(result.score))}</span>#{fault}</div>) +
409
- %(<div class="calls">#{html_calls(result, empty: 'no tools called')}</div></div>)
509
+ %(<div class="calls">#{html_calls(result, empty: 'no tools called')}</div>#{html_cell_cost(result)}</div>)
510
+ end
511
+
512
+ # "~$0.0243 · judge ~$0.0015" under a cell: what this answer cost and
513
+ # what judging it cost. A result with no cost at all shows no line.
514
+ def html_cell_cost(result)
515
+ text = result_cost_text(result)
516
+ return "" if text.nil?
517
+
518
+ %(<div class="cost-line"#{cost_title_attr([ result ], estimated: result.estimated_cost?)}>#{h(text)}</div>)
519
+ end
520
+
521
+ def result_cost_text(result)
522
+ judge_cost = result.judge_usage&.dig("cost")
523
+ return nil if result.replay.cost.nil? && judge_cost.nil?
524
+
525
+ parts = [ fmt_cost(result.replay.cost, estimated: result.estimated_cost?) ]
526
+ parts << "judge #{fmt_cost(judge_cost, estimated: estimated_usage?(result.judge_usage))}" if judge_cost
527
+ parts.join(" · ")
410
528
  end
411
529
 
412
530
  def html_calls(result, empty:)
@@ -487,20 +605,156 @@ module ActiveAgent
487
605
  [
488
606
  replay.duration_ms && fmt_ms(replay.duration_ms),
489
607
  replay.total_tokens.positive? ? "#{fmt_k(replay.total_tokens)} tokens" : nil,
490
- replay.cost && fmt_cost(replay.cost)
608
+ result_cost_text(result)
491
609
  ].compact.join(" · ")
492
610
  end
493
611
 
612
+ # --- WHAT TO FIX -----------------------------------------------------
613
+
614
+ # The section stands even for a run with nothing to fix — the dashboard
615
+ # keeps it too, so a clean run reads as clean rather than as a page
616
+ # missing a section. A report over no results at all has nothing to say.
617
+ def html_fixes
618
+ return "" if @results.empty?
619
+
620
+ items = fix_items
621
+ faulted = @results.reject(&:passed?)
622
+ meta = "#{plural(items.size, 'item')} · #{plural(faulted.size, 'fault')} across " \
623
+ "#{plural(faulted.map { |result| result.scenario.key }.uniq.size, 'scenario')}"
624
+ body =
625
+ if items.any?
626
+ %(<div class="fixes">#{items.map { |item| html_fix_card(item) }.join}</div>)
627
+ else
628
+ %(<div class="nothing">[+] nothing to fix</div>)
629
+ end
630
+
631
+ <<~FIXES
632
+ <section class="section fix-section" aria-label="Recommendations">
633
+ <div class="section-head"><span class="micro">What to fix</span><span class="meta">#{h(meta)}</span></div>
634
+ #{html_fix_filter(items) if comparing? && items.any?}
635
+ #{body}
636
+ </section>
637
+ FIXES
638
+ end
639
+
640
+ # A model filter for the fix cards — a fault one model keeps making is
641
+ # that model's to fix, so the list narrows to what was attributed to
642
+ # it. Radio chips and stylesheet rules alone (the page carries no
643
+ # script): each card names its models in data-models, and a checked
644
+ # model hides every card that does not name it. Cards attributed to no
645
+ # model (an older run) stay under every filter.
646
+ def html_fix_filter(items)
647
+ chips = [ %(<label class="chip pick-model"><input type="radio" name="fix-model" value="all" checked><span>all models #{items.size}</span></label>) ]
648
+ @models.each_with_index do |spec, index|
649
+ count = items.count { |item| Array(item["models"]).empty? || item["models"].include?(spec.label) }
650
+ chips << %(<label class="chip pick-model"><input type="radio" name="fix-model" value="m#{index}"><span>#{h(short_name(spec))} #{count}</span></label>)
651
+ end
652
+ %(<div class="fix-filter"><span class="micro sm">for</span>#{chips.join}</div>)
653
+ end
654
+
655
+ def fix_model_tokens(item)
656
+ labels = Array(item["models"])
657
+ return "" if labels.empty?
658
+
659
+ labels.filter_map { |label| (index = @models.index(model_by_label(label))) && "m#{index}" }.join(" ")
660
+ end
661
+
662
+ def html_fix_card(item)
663
+ tone = item["kind"] == "instruction" ? "info" : "error"
664
+ glyph = tone == "info" ? "[i]" : "[!]"
665
+ title = fault_name(item["fault"]) + (item["count"].to_i > 1 ? " ×#{item['count']}" : "")
666
+
667
+ parts = [ %(<div class="head"><span class="glyph tone-#{tone}">#{glyph}</span>) +
668
+ %(<span class="badge #{tone}">#{h(title)}</span><span class="scope">#{h(fix_scope(item))}</span></div>) ]
669
+ parts << %(<p>#{h(item['recommendation'])}</p>) if item["recommendation"].present?
670
+ parts << %(<div class="quote">“#{h(item['quote'])}”</div>) if item["quote"].present?
671
+ parts << html_fix_tools(item) if item["tools"].any?
672
+ parts << html_fix_server(item["server"]) if item["server"]
673
+ parts << %(<div class="note">#{h(item['note'])}</div>) if item["note"].present?
674
+ parts << html_fix_action(item["action"]) if item["action"]
675
+ models = fix_model_tokens(item)
676
+ %(<div class="fix"#{%( data-models="#{models}") if models.present?}>#{parts.join}</div>)
677
+ end
678
+
679
+ def html_fix_tools(item)
680
+ chips = item["tools"].map do |tool|
681
+ note = tool["note"].presence
682
+ %(<span class="tool"><b>#{h(tool['name'])}</b>#{%(<span class="note">#{h(note)}</span>) if note}</span>)
683
+ end
684
+ %(<div class="tools"><span class="micro sm">#{h(item['tools_label'])}</span><div class="list">#{chips.join}</div></div>)
685
+ end
686
+
687
+ # "available · not enabled for Assistant", "unknown · not enabled for
688
+ # Assistant" — every status but "enabled" leads with the status word, the
689
+ # way the dashboard's fix list reads it.
690
+ def html_fix_server(server)
691
+ badge =
692
+ if server["status"] == "enabled"
693
+ %(<span class="badge success xs">enabled for #{h(@agent_name)}</span>)
694
+ else
695
+ %(<span class="badge warning xs">#{h(server['status'].presence || 'unknown')} · not enabled for #{h(@agent_name)}</span>)
696
+ end
697
+ %(<div class="served"><span>served by</span><b>#{h(server['name'].presence || server['key'])}</b>#{badge}</div>)
698
+ end
699
+
700
+ # With a route the action is a button; without one, the page can only
701
+ # say where in the dashboard the fix lives. The link targets the top
702
+ # window: served in the dashboard's report iframe it would otherwise
703
+ # open the whole dashboard inside the frame.
704
+ def html_fix_action(action)
705
+ button = action["path"].present? ? %(<a class="btn" target="_top" href="#{h(action['path'])}">#{h(action['label'])}</a>) : ""
706
+ %(<div class="action">#{button}<span class="hint">#{h(action['hint'])}</span></div>)
707
+ end
708
+
709
+ # "3 scenarios · both models" — the models are worth naming only on a
710
+ # comparison run; on a single-model run the count says it all.
711
+ def fix_scope(item)
712
+ return "#{item['scenario_keys'].join(', ')} · judge suggestion" if item["kind"] == "instruction"
713
+
714
+ scenarios = plural(item["scenario_keys"].size, "scenario")
715
+ labels = Array(item["models"])
716
+ return scenarios unless comparing? && labels.any?
717
+
718
+ models =
719
+ if labels.size >= @models.size
720
+ @models.size == 2 ? "both models" : "all models"
721
+ else
722
+ labels.map { |label| short_name(model_by_label(label)) }.join(", ")
723
+ end
724
+ "#{scenarios} · #{models}"
725
+ end
726
+
494
727
  # --- footer ----------------------------------------------------------
495
728
 
729
+ # The run's terms: the judge, the criteria and the pass mark, what it
730
+ # cost on each side, the metadata, and — once, when any figure on the
731
+ # page carries a "~" — what the mark means.
496
732
  def html_footer
497
733
  criteria = criterion_keys.map { |key| key.to_s.tr("_", " ") }.join(" · ")
498
734
  spans = [ %(<span class="nowrap">judge #{h(judge_name)}</span>) ]
499
735
  spans << %(<span class="criteria">criteria #{h(criteria)}</span>) if criteria.present?
500
- spans.concat(@metadata.to_h.map { |key, value| %(<span class="nowrap">#{h(key)} #{h(value)}</span>) })
736
+ spans << %(<span class="nowrap">#{h(Format.threshold(@threshold))}</span>)
737
+ spans << %(<span class="nowrap">#{h(footer_cost_text)}</span>) if footer_cost_text
738
+ spans << %(<span class="nowrap">release #{h(release_label)}</span>) if release_label
739
+ spans.concat(metadata_chips.map { |key, value| %(<span class="nowrap">#{h(key)} #{h(value)}</span>) })
740
+ spans << %(<span class="legend">#{h(Format::LEGEND)}</span>) if estimated_anywhere?
501
741
  %(<footer>#{spans.join}</footer>)
502
742
  end
503
743
 
744
+ # "cost ~$0.0412 · agent ~$0.0397 · judge ~$0.0015", or nil when
745
+ # nothing was priced.
746
+ def footer_cost_text
747
+ costs = run_costs
748
+ return nil if costs["total"].nil?
749
+
750
+ parts = [ "cost #{fmt_cost(costs['total'], estimated: costs['estimated'])}" ]
751
+ if judge_usage
752
+ parts << "agent #{fmt_cost(costs['cost'], estimated: costs['estimated'])}"
753
+ parts << "judge #{fmt_cost(costs['judge_cost'], estimated: judge_estimated?)}"
754
+ end
755
+ parts.join(" · ")
756
+ end
757
+
504
758
  # Colors only through the token variables; radii 4 badges · 6 chips ·
505
759
  # 8 controls · 10 nested panels · 12 cards · 999 bars; no shadows.
506
760
  STYLES = <<~CSS.freeze
@@ -557,6 +811,24 @@ module ActiveAgent
557
811
  .tok .in { color: var(--color-token-in); }
558
812
  .tok .out { color: var(--color-token-out); }
559
813
  .faults { display: flex; gap: 6px; flex-wrap: wrap; }
814
+ .compare { overflow-x: auto; border-top: 1px solid var(--color-border-light); }
815
+ .compare table { width: 100%; border-collapse: collapse; }
816
+ .compare th { padding: 8px 12px; text-align: left; vertical-align: bottom; white-space: nowrap; font-family: var(--font-mono); font-size: 10px; font-weight: 600; letter-spacing: 0.05em; text-transform: uppercase; color: var(--color-text-muted); background: var(--color-muted); }
817
+ .compare td { padding: 9px 12px; vertical-align: top; border-top: 1px solid var(--color-border-light); font-size: 13px; color: var(--color-text-cell); }
818
+ .compare th.num, .compare td.num { text-align: right; }
819
+ .compare td.num { font-family: var(--font-mono); font-size: 12px; white-space: nowrap; }
820
+ .compare td.ratio { font-weight: 600; }
821
+ .compare .per { display: block; font-weight: 400; color: var(--color-text-muted); }
822
+ .compare .model-cell { white-space: nowrap; }
823
+ .compare .model-cell .name { font-family: var(--font-mono); font-size: 12px; font-weight: 600; color: var(--color-text-primary); }
824
+ .compare .model-cell .provider { display: block; font-family: var(--font-mono); font-size: 11px; color: var(--color-text-muted); }
825
+ .compare .pick { margin-left: 6px; font-family: var(--font-mono); font-size: 10px; font-weight: 700; color: var(--color-warning-text); }
826
+ .compare th.fault, .compare td.fault { width: 34%; }
827
+ .compare td.fault .detail { color: var(--color-text-secondary); }
828
+ .fix-filter { display: flex; align-items: center; gap: 6px; flex-wrap: wrap; }
829
+ .pick-model { cursor: pointer; border: 1px solid var(--color-border); background: var(--color-card); }
830
+ .pick-model input { position: absolute; opacity: 0; width: 0; height: 0; }
831
+ .pick-model:has(input:checked) { border-color: var(--color-accent-ui); background: var(--color-accent-ui-tint); color: var(--color-accent-ui); }
560
832
  .clean { font-family: var(--font-mono); font-size: 11px; color: var(--color-success-text); }
561
833
  .verdict { padding: 10px 12px; border-top: 1px solid var(--color-border-light); font-size: 12px; line-height: 18px; color: var(--color-text-cell); }
562
834
  .verdict .micro { margin-right: 8px; }
@@ -590,6 +862,10 @@ module ActiveAgent
590
862
  .group-name { font-size: 12px; font-weight: 600; }
591
863
  .count { font-family: var(--font-mono); font-size: 11px; color: var(--color-text-muted); }
592
864
  .group-pass { font-family: var(--font-mono); font-size: 11px; font-weight: 600; color: var(--color-text-cell); }
865
+ .mx .cost { display: flex; flex-direction: column; gap: 1px; min-width: 0; font-family: var(--font-mono); font-size: 11px; color: var(--color-text-secondary); text-align: right; }
866
+ .mx .cost b { font-weight: 600; color: var(--color-text-primary); }
867
+ .mx .cost .judge { color: var(--color-text-muted); white-space: nowrap; }
868
+ .mx .cost-head { text-align: right; }
593
869
  .key { font-family: var(--font-mono); font-size: 11px; color: var(--color-text-muted); margin-bottom: 2px; }
594
870
  .key a { color: inherit; }
595
871
  .prompt { font-size: 13px; line-height: 18px; color: var(--color-text-primary); }
@@ -603,6 +879,7 @@ module ActiveAgent
603
879
  .calls { display: flex; flex-wrap: wrap; gap: 2px 8px; font-family: var(--font-mono); font-size: 11px; color: var(--color-text-muted); }
604
880
  .call-hit { color: var(--color-success-text); font-weight: 600; }
605
881
  .call-err { color: var(--color-error); font-weight: 600; }
882
+ .cost-line { font-family: var(--font-mono); font-size: 11px; color: var(--color-text-secondary); white-space: nowrap; }
606
883
  .details { display: flex; flex-direction: column; gap: 8px; }
607
884
  details { border: 1px solid var(--color-border-light); border-radius: 10px; overflow: hidden; }
608
885
  summary { display: flex; align-items: center; gap: 10px; padding: 10px 12px; cursor: pointer; list-style: none; flex-wrap: wrap; }
@@ -628,6 +905,7 @@ module ActiveAgent
628
905
  footer { display: flex; align-items: center; gap: 16px; flex-wrap: wrap; padding-top: 12px; border-top: 1px solid var(--color-border-light); font-family: var(--font-mono); font-size: 11px; color: var(--color-text-muted); }
629
906
  footer .criteria { min-width: 0; }
630
907
  footer .nowrap { white-space: nowrap; }
908
+ footer .legend { margin-left: auto; white-space: nowrap; color: var(--color-text-muted); }
631
909
  CSS
632
910
  end
633
911
  end