activeagent 1.8.0 → 1.8.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +149 -0
- data/lib/active_agent/evals/diagnosis.rb +3 -3
- data/lib/active_agent/evals/format.rb +115 -0
- data/lib/active_agent/evals/judge.rb +2 -0
- data/lib/active_agent/evals/report.rb +232 -31
- data/lib/active_agent/evals/report_html.rb +316 -149
- data/lib/active_agent/evals/result.rb +40 -0
- data/lib/active_agent/evals/runner.rb +6 -2
- data/lib/active_agent/evals.rb +1 -0
- data/lib/active_agent/version.rb +1 -1
- metadata +3 -2
|
@@ -12,8 +12,8 @@ module ActiveAgent
|
|
|
12
12
|
#
|
|
13
13
|
# Same content as Report#to_markdown, laid out the way the dashboard's
|
|
14
14
|
# suite card is: header and stat tiles, the MODELS panel with the judge's
|
|
15
|
-
# pick and verdict,
|
|
16
|
-
#
|
|
15
|
+
# pick and verdict, the SCENARIOS matrix, a per-scenario disclosure with
|
|
16
|
+
# every answer, and WHAT TO FIX cards from Report#fix_items.
|
|
17
17
|
module ReportHtml
|
|
18
18
|
THEMES = %w[light dark].freeze
|
|
19
19
|
ANSWER_LIMIT = 3_000
|
|
@@ -44,9 +44,9 @@ module ActiveAgent
|
|
|
44
44
|
#{html_stat_tiles}
|
|
45
45
|
<section class="card">
|
|
46
46
|
#{html_models_panel}
|
|
47
|
-
#{html_fixes}
|
|
48
47
|
#{html_matrix}
|
|
49
48
|
#{html_details}
|
|
49
|
+
#{html_fixes}
|
|
50
50
|
#{html_footer}
|
|
51
51
|
</section>
|
|
52
52
|
</main>
|
|
@@ -151,16 +151,50 @@ module ActiveAgent
|
|
|
151
151
|
"#{format("%.#{n >= 10_000 ? 1 : 2}f", n / 1_000).sub(/\.?0+\z/, '')}s"
|
|
152
152
|
end
|
|
153
153
|
|
|
154
|
-
|
|
155
|
-
|
|
154
|
+
# Money reads "~$0.0243" when any part of it was estimated from tokens
|
|
155
|
+
# × a model rate rather than reported (Format.money); the footer's
|
|
156
|
+
# legend explains the mark once.
|
|
157
|
+
def fmt_cost(value, estimated: false)
|
|
158
|
+
Format.money(value, estimated: estimated)
|
|
156
159
|
end
|
|
157
160
|
|
|
161
|
+
# A 0..1 score as a whole percent, "—" when there is none.
|
|
158
162
|
def fmt_score(value)
|
|
159
|
-
|
|
163
|
+
Format.score(value)
|
|
164
|
+
end
|
|
165
|
+
|
|
166
|
+
# "14/16 · 88%" — a fraction always carries its percentage.
|
|
167
|
+
def fmt_passes(passed, total)
|
|
168
|
+
Format.passes(passed, total)
|
|
169
|
+
end
|
|
170
|
+
|
|
171
|
+
# Whether the page shows a "~" anywhere, so the footer carries the legend.
|
|
172
|
+
def estimated_anywhere?
|
|
173
|
+
run_costs["estimated"] || summary_by_model.values.any? { |stats| estimated_cost?(stats) }
|
|
174
|
+
end
|
|
175
|
+
|
|
176
|
+
# The tooltip of an estimated figure, worked out from the results it
|
|
177
|
+
# sums: their tokens at the rate the first estimated one recorded.
|
|
178
|
+
# Nothing for a reported figure.
|
|
179
|
+
def cost_title_attr(results, estimated:)
|
|
180
|
+
return "" unless estimated
|
|
181
|
+
|
|
182
|
+
priced = results.select(&:estimated_cost?)
|
|
183
|
+
rate = priced.filter_map(&:cost_rate).first
|
|
184
|
+
title = Format.cost_title(
|
|
185
|
+
input_tokens: priced.sum { |result| result.replay.input_tokens.to_i },
|
|
186
|
+
output_tokens: priced.sum { |result| result.replay.output_tokens.to_i },
|
|
187
|
+
rate: rate
|
|
188
|
+
)
|
|
189
|
+
%( title="#{h(title)}")
|
|
160
190
|
end
|
|
161
191
|
|
|
162
|
-
def
|
|
163
|
-
|
|
192
|
+
def results_for(label)
|
|
193
|
+
@results.select { |result| result.label == label }
|
|
194
|
+
end
|
|
195
|
+
|
|
196
|
+
def judge_estimated?
|
|
197
|
+
judge_usage&.dig("estimated") == true
|
|
164
198
|
end
|
|
165
199
|
|
|
166
200
|
# --- page ------------------------------------------------------------
|
|
@@ -177,17 +211,21 @@ module ActiveAgent
|
|
|
177
211
|
"}",
|
|
178
212
|
DesignTokens.css(scope: ":root.theme-dark", tokens: DesignTokens::DARK, color_scheme: "dark"),
|
|
179
213
|
STYLES,
|
|
180
|
-
".mx { grid-template-columns: minmax(240px, 1.6fr) 150px repeat(#{@models.size}, minmax(170px, 1fr)); }",
|
|
214
|
+
".mx { grid-template-columns: minmax(240px, 1.6fr) 150px repeat(#{@models.size}, minmax(170px, 1fr)) 120px; }",
|
|
181
215
|
# One rule per model: with that chip checked, hide every fix card
|
|
182
216
|
# attributed to other models (cards attributed to none stay).
|
|
183
217
|
*@models.each_index.map { |i| ".fix-section:has(input[value=\"m#{i}\"]:checked) .fix[data-models]:not([data-models~=\"m#{i}\"]) { display: none; }" },
|
|
184
|
-
".matrix .inner { min-width: #{
|
|
218
|
+
".matrix .inner { min-width: #{520 + 185 * @models.size}px; }"
|
|
185
219
|
].join("\n")
|
|
186
220
|
end
|
|
187
221
|
|
|
222
|
+
# One chip per scalar metadata value — an array or a hash (the judge's
|
|
223
|
+
# trace ids, say) is a record, not a label — then the release when the
|
|
224
|
+
# report names one, and the judge with its calls and spend.
|
|
188
225
|
def html_header(title)
|
|
189
|
-
chips =
|
|
190
|
-
chips << html_chip("
|
|
226
|
+
chips = metadata_chips.map { |key, value| html_chip(key, value) }
|
|
227
|
+
chips << html_chip("release", release_label) if release_label
|
|
228
|
+
chips << html_chip("judge", judge_chip_text)
|
|
191
229
|
|
|
192
230
|
<<~HEADER
|
|
193
231
|
<header>
|
|
@@ -201,24 +239,63 @@ module ActiveAgent
|
|
|
201
239
|
%(<span class="chip"><b>#{h(key)}</b>#{h(value)}</span>)
|
|
202
240
|
end
|
|
203
241
|
|
|
242
|
+
# The metadata worth a chip: scalar values, minus the judge's trace
|
|
243
|
+
# ids, which the dashboard follows but nobody reads.
|
|
244
|
+
def metadata_chips
|
|
245
|
+
@metadata.to_h.reject { |key, value| key.to_s == "judge_trace_ids" || value.is_a?(Hash) || value.is_a?(Array) || value.nil? }
|
|
246
|
+
end
|
|
247
|
+
|
|
248
|
+
# "1a2b3c4d5e6f · abc1234", or the label the caller gave the release.
|
|
249
|
+
def release_label
|
|
250
|
+
return nil unless release
|
|
251
|
+
|
|
252
|
+
release["label"].presence || [ release["digest"], release["revision"] ].compact_blank.join(" · ").presence
|
|
253
|
+
end
|
|
254
|
+
|
|
255
|
+
# "gpt-5 · 12 calls · ~$0.0315" when the judge spent anything.
|
|
256
|
+
def judge_chip_text
|
|
257
|
+
usage = judge_usage
|
|
258
|
+
return judge_name unless usage
|
|
259
|
+
|
|
260
|
+
parts = [ judge_name, plural(usage["calls"].to_i, "call") ]
|
|
261
|
+
parts << fmt_cost(usage["cost"], estimated: judge_estimated?) if usage["cost"]
|
|
262
|
+
parts.join(" · ")
|
|
263
|
+
end
|
|
264
|
+
|
|
204
265
|
def html_stat_tiles
|
|
205
266
|
total = @results.size
|
|
206
267
|
passed = @results.count(&:passed?)
|
|
207
268
|
ratio = total.positive? ? passed.to_f / total : 0.0
|
|
208
269
|
tiles = [
|
|
209
270
|
html_tile("Scenario runs", total, "#{plural(scenario_cohorts.size, 'scenario')} × #{plural(@models.size, 'model')}"),
|
|
210
|
-
html_tile("Pass rate",
|
|
271
|
+
html_tile("Pass rate", Format.percent(total.positive? ? ratio : nil), "#{passed}/#{total} passed", tone: tone_for(ratio)),
|
|
211
272
|
html_tile("Open faults", total - passed, plural(fix_items.size, "fix item")),
|
|
273
|
+
html_cost_tile,
|
|
212
274
|
html_tile("Models", @models.size, models_subline)
|
|
213
275
|
]
|
|
214
276
|
%(<section class="stats">#{tiles.join}</section>)
|
|
215
277
|
end
|
|
216
278
|
|
|
217
|
-
def html_tile(label, value, sub, tone: nil)
|
|
218
|
-
%(<div class="tile"><div class="micro">#{h(label)}</div>) +
|
|
279
|
+
def html_tile(label, value, sub, tone: nil, title: nil)
|
|
280
|
+
%(<div class="tile"#{%( title="#{h(title)}") if title}><div class="micro">#{h(label)}</div>) +
|
|
219
281
|
%(<div class="value#{" tone-#{tone}" if tone}">#{h(value)}</div><div class="sub">#{h(sub)}</div></div>)
|
|
220
282
|
end
|
|
221
283
|
|
|
284
|
+
# The run's spend: agent plus judge, with the two apart underneath.
|
|
285
|
+
# A run with nothing priced says so rather than showing a blank.
|
|
286
|
+
def html_cost_tile
|
|
287
|
+
costs = run_costs
|
|
288
|
+
sub =
|
|
289
|
+
if costs["total"].nil?
|
|
290
|
+
"nothing priced"
|
|
291
|
+
elsif judge_usage
|
|
292
|
+
"agent #{fmt_cost(costs['cost'], estimated: costs['estimated'])} · judge #{fmt_cost(costs['judge_cost'], estimated: judge_estimated?)}"
|
|
293
|
+
else
|
|
294
|
+
"agent only · #{plural(costs['priced'], 'scenario run')} priced"
|
|
295
|
+
end
|
|
296
|
+
html_tile("Cost", fmt_cost(costs["total"], estimated: costs["estimated"]), sub)
|
|
297
|
+
end
|
|
298
|
+
|
|
222
299
|
def models_subline
|
|
223
300
|
if comparing? && winner
|
|
224
301
|
"judge's pick · #{short_name(model_by_label(winner))}"
|
|
@@ -245,17 +322,20 @@ module ActiveAgent
|
|
|
245
322
|
|
|
246
323
|
# The comparison read across: one row per model, best first (pass rate,
|
|
247
324
|
# then mean score) — passed, mean score, average latency, average
|
|
248
|
-
# tokens per scenario, cost
|
|
249
|
-
#
|
|
325
|
+
# tokens per scenario, cost (and per scenario), the judge's spend on
|
|
326
|
+
# the cohort when a judge was asked, and the model's typical fault.
|
|
327
|
+
# The blocks under it carry the same figures per model with bars and
|
|
328
|
+
# every fault.
|
|
250
329
|
def html_comparison_table
|
|
251
330
|
rows = summary_by_model.sort_by do |label, stats|
|
|
252
331
|
total = stats["scenarios"].to_i
|
|
253
332
|
[ total.positive? ? -stats["passed"].to_f / total : 0.0, -(stats["avg_score"] || -1).to_f, @models.index(model_by_label(label)).to_i ]
|
|
254
333
|
end
|
|
334
|
+
judge_head = judge_usage ? %(<th class="num" title="What the judge spent scoring this model's answers">Judge</th>) : ""
|
|
255
335
|
|
|
256
336
|
<<~TABLE
|
|
257
337
|
<div class="compare"><table>
|
|
258
|
-
<thead><tr><th>Model</th><th class="num">Passed</th><th class="num">Mean score</th><th class="num">Avg latency</th><th class="num" title="Average input + output tokens per scenario">Avg tokens</th><th class="num" title="Cohort spend, and per scenario">Cost</th
|
|
338
|
+
<thead><tr><th>Model</th><th class="num">Passed</th><th class="num">Mean score</th><th class="num">Avg latency</th><th class="num" title="Average input + output tokens per scenario">Avg tokens</th><th class="num" title="Cohort spend, and per scenario">Cost</th>#{judge_head}<th class="fault">Typical fault</th></tr></thead>
|
|
259
339
|
<tbody>#{rows.map { |label, stats| html_comparison_row(label, stats) }.join}</tbody>
|
|
260
340
|
</table></div>
|
|
261
341
|
TABLE
|
|
@@ -270,23 +350,34 @@ module ActiveAgent
|
|
|
270
350
|
avg_tokens = per.call(stats["input_tokens"].to_i + stats["output_tokens"].to_i)
|
|
271
351
|
tokens_cell = avg_tokens ? h(fmt_k(avg_tokens.round)) : "—"
|
|
272
352
|
tokens_title = avg_tokens ? %( title="#{per.call(stats['input_tokens']).to_f.round} in · #{per.call(stats['output_tokens']).to_f.round} out per scenario") : ""
|
|
273
|
-
per_cost =
|
|
274
|
-
|
|
275
|
-
cost_cell
|
|
353
|
+
per_cost = cost_per_priced(stats)
|
|
354
|
+
estimated = estimated_cost?(stats)
|
|
355
|
+
cost_cell = h(fmt_cost(stats["cost"], estimated: estimated))
|
|
356
|
+
cost_cell += "<span class=\"per\">#{h(fmt_cost(per_cost, estimated: estimated))}/scenario</span>" if per_cost
|
|
357
|
+
judge_cell = judge_usage ? %(<td class="num">#{html_judge_cost(stats)}</td>) : ""
|
|
276
358
|
|
|
277
359
|
<<~ROW
|
|
278
360
|
<tr>
|
|
279
361
|
<td class="model-cell"><span class="name">#{h(short)}</span>#{pick}<span class="provider">#{h(provider)}</span></td>
|
|
280
|
-
<td class="num ratio tone-#{tone_for(ratio)}">#{
|
|
281
|
-
<td class="num">#{h(
|
|
362
|
+
<td class="num ratio tone-#{tone_for(ratio)}">#{h(fmt_passes(stats['passed'], total))}</td>
|
|
363
|
+
<td class="num">#{h(fmt_score(stats['avg_score']))}</td>
|
|
282
364
|
<td class="num">#{h(fmt_ms(stats['avg_duration_ms']))}</td>
|
|
283
365
|
<td class="num"#{tokens_title}>#{tokens_cell}</td>
|
|
284
|
-
<td class="num">#{cost_cell}</td>
|
|
285
|
-
<td class="fault">#{typical_fault_text(label, stats)}</td>
|
|
366
|
+
<td class="num"#{cost_title_attr(results_for(label), estimated: estimated)}>#{cost_cell}</td>
|
|
367
|
+
#{judge_cell}<td class="fault">#{typical_fault_text(label, stats)}</td>
|
|
286
368
|
</tr>
|
|
287
369
|
ROW
|
|
288
370
|
end
|
|
289
371
|
|
|
372
|
+
# "~$0.0030<span class="per">3 calls</span>" — the judge's spend on a
|
|
373
|
+
# model's answers; "—" when it was not asked about them.
|
|
374
|
+
def html_judge_cost(stats)
|
|
375
|
+
calls = stats["judge_calls"].to_i
|
|
376
|
+
return "—" if stats["judge_cost"].nil? && calls.zero?
|
|
377
|
+
|
|
378
|
+
h(fmt_cost(stats["judge_cost"], estimated: judge_estimated?)) + %(<span class="per">#{h(plural(calls, 'call'))}</span>)
|
|
379
|
+
end
|
|
380
|
+
|
|
290
381
|
# "missing content ×2 · refund_window: The answer is missing expected
|
|
291
382
|
# content: 30." — the model's most frequent fault, and the diagnosis of
|
|
292
383
|
# the first result that carries it; "no faults" for a clean cohort.
|
|
@@ -317,130 +408,21 @@ module ActiveAgent
|
|
|
317
408
|
faults = stats["faults"].map { |fault, count| %(<span class="badge error">#{h(fault_name(fault))} ×#{count}</span>) }
|
|
318
409
|
faults_html = faults.any? ? faults.join : %(<span class="clean">[+] no faults</span>)
|
|
319
410
|
|
|
411
|
+
estimated = estimated_cost?(stats)
|
|
412
|
+
judge_line = ""
|
|
413
|
+
if stats["judge_cost"] || stats["judge_calls"].to_i.positive?
|
|
414
|
+
judge_line = %(<span>judge <b>#{h(fmt_cost(stats['judge_cost'], estimated: judge_estimated?))}</b> · #{h(plural(stats['judge_calls'].to_i, 'call'))}</span>)
|
|
415
|
+
end
|
|
416
|
+
|
|
320
417
|
<<~BLOCK
|
|
321
418
|
<div class="model">
|
|
322
|
-
<div class="line"><span class="name">#{h(short)}</span><span class="provider">#{h(provider)}</span>#{pick}<span class="pass"><span class="bar bar-#{tone}"><span style="width:#{(ratio * 100).round}%"></span></span><span class="ratio tone-#{tone}">#{stats['passed']
|
|
323
|
-
<div class="stats-line"><span>score <b>#{h(
|
|
419
|
+
<div class="line"><span class="name">#{h(short)}</span><span class="provider">#{h(provider)}</span>#{pick}<span class="pass"><span class="bar bar-#{tone}"><span style="width:#{(ratio * 100).round}%"></span></span><span class="ratio tone-#{tone}">#{h(fmt_passes(stats['passed'], total))}</span></span></div>
|
|
420
|
+
<div class="stats-line"><span>score <b>#{h(fmt_score(stats['avg_score']))}</b></span><span>latency <b>#{h(fmt_ms(stats['avg_duration_ms']))}</b></span><span class="tok"><span class="in">in</span> #{h(fmt_k(stats['input_tokens']))} · <span class="out">out</span> #{h(fmt_k(stats['output_tokens']))}</span><span#{cost_title_attr(results_for(label), estimated: estimated)}>cost <b>#{h(fmt_cost(stats['cost'], estimated: estimated))}</b></span>#{judge_line}</div>
|
|
324
421
|
<div class="faults">#{faults_html}</div>
|
|
325
422
|
</div>
|
|
326
423
|
BLOCK
|
|
327
424
|
end
|
|
328
425
|
|
|
329
|
-
# --- WHAT TO FIX -----------------------------------------------------
|
|
330
|
-
|
|
331
|
-
# The section stands even for a run with nothing to fix — the dashboard
|
|
332
|
-
# keeps it too, so a clean run reads as clean rather than as a page
|
|
333
|
-
# missing a section. A report over no results at all has nothing to say.
|
|
334
|
-
def html_fixes
|
|
335
|
-
return "" if @results.empty?
|
|
336
|
-
|
|
337
|
-
items = fix_items
|
|
338
|
-
faulted = @results.reject(&:passed?)
|
|
339
|
-
meta = "#{plural(items.size, 'item')} · #{plural(faulted.size, 'fault')} across " \
|
|
340
|
-
"#{plural(faulted.map { |result| result.scenario.key }.uniq.size, 'scenario')}"
|
|
341
|
-
body =
|
|
342
|
-
if items.any?
|
|
343
|
-
%(<div class="fixes">#{items.map { |item| html_fix_card(item) }.join}</div>)
|
|
344
|
-
else
|
|
345
|
-
%(<div class="nothing">[+] nothing to fix</div>)
|
|
346
|
-
end
|
|
347
|
-
|
|
348
|
-
<<~FIXES
|
|
349
|
-
<section class="section fix-section" aria-label="Recommendations">
|
|
350
|
-
<div class="section-head"><span class="micro">What to fix</span><span class="meta">#{h(meta)}</span></div>
|
|
351
|
-
#{html_fix_filter(items) if comparing? && items.any?}
|
|
352
|
-
#{body}
|
|
353
|
-
</section>
|
|
354
|
-
FIXES
|
|
355
|
-
end
|
|
356
|
-
|
|
357
|
-
# A model filter for the fix cards — a fault one model keeps making is
|
|
358
|
-
# that model's to fix, so the list narrows to what was attributed to
|
|
359
|
-
# it. Radio chips and stylesheet rules alone (the page carries no
|
|
360
|
-
# script): each card names its models in data-models, and a checked
|
|
361
|
-
# model hides every card that does not name it. Cards attributed to no
|
|
362
|
-
# model (an older run) stay under every filter.
|
|
363
|
-
def html_fix_filter(items)
|
|
364
|
-
chips = [ %(<label class="chip pick-model"><input type="radio" name="fix-model" value="all" checked><span>all models #{items.size}</span></label>) ]
|
|
365
|
-
@models.each_with_index do |spec, index|
|
|
366
|
-
count = items.count { |item| Array(item["models"]).empty? || item["models"].include?(spec.label) }
|
|
367
|
-
chips << %(<label class="chip pick-model"><input type="radio" name="fix-model" value="m#{index}"><span>#{h(short_name(spec))} #{count}</span></label>)
|
|
368
|
-
end
|
|
369
|
-
%(<div class="fix-filter"><span class="micro sm">for</span>#{chips.join}</div>)
|
|
370
|
-
end
|
|
371
|
-
|
|
372
|
-
def fix_model_tokens(item)
|
|
373
|
-
labels = Array(item["models"])
|
|
374
|
-
return "" if labels.empty?
|
|
375
|
-
|
|
376
|
-
labels.filter_map { |label| (index = @models.index(model_by_label(label))) && "m#{index}" }.join(" ")
|
|
377
|
-
end
|
|
378
|
-
|
|
379
|
-
def html_fix_card(item)
|
|
380
|
-
tone = item["kind"] == "instruction" ? "info" : "error"
|
|
381
|
-
glyph = tone == "info" ? "[i]" : "[!]"
|
|
382
|
-
title = fault_name(item["fault"]) + (item["count"].to_i > 1 ? " ×#{item['count']}" : "")
|
|
383
|
-
|
|
384
|
-
parts = [ %(<div class="head"><span class="glyph tone-#{tone}">#{glyph}</span>) +
|
|
385
|
-
%(<span class="badge #{tone}">#{h(title)}</span><span class="scope">#{h(fix_scope(item))}</span></div>) ]
|
|
386
|
-
parts << %(<p>#{h(item['recommendation'])}</p>) if item["recommendation"].present?
|
|
387
|
-
parts << %(<div class="quote">“#{h(item['quote'])}”</div>) if item["quote"].present?
|
|
388
|
-
parts << html_fix_tools(item) if item["tools"].any?
|
|
389
|
-
parts << html_fix_server(item["server"]) if item["server"]
|
|
390
|
-
parts << %(<div class="note">#{h(item['note'])}</div>) if item["note"].present?
|
|
391
|
-
parts << html_fix_action(item["action"]) if item["action"]
|
|
392
|
-
models = fix_model_tokens(item)
|
|
393
|
-
%(<div class="fix"#{%( data-models="#{models}") if models.present?}>#{parts.join}</div>)
|
|
394
|
-
end
|
|
395
|
-
|
|
396
|
-
def html_fix_tools(item)
|
|
397
|
-
chips = item["tools"].map do |tool|
|
|
398
|
-
note = tool["note"].presence
|
|
399
|
-
%(<span class="tool"><b>#{h(tool['name'])}</b>#{%(<span class="note">#{h(note)}</span>) if note}</span>)
|
|
400
|
-
end
|
|
401
|
-
%(<div class="tools"><span class="micro sm">#{h(item['tools_label'])}</span><div class="list">#{chips.join}</div></div>)
|
|
402
|
-
end
|
|
403
|
-
|
|
404
|
-
# "available · not enabled for Assistant", "unknown · not enabled for
|
|
405
|
-
# Assistant" — every status but "enabled" leads with the status word, the
|
|
406
|
-
# way the dashboard's fix list reads it.
|
|
407
|
-
def html_fix_server(server)
|
|
408
|
-
badge =
|
|
409
|
-
if server["status"] == "enabled"
|
|
410
|
-
%(<span class="badge success xs">enabled for #{h(@agent_name)}</span>)
|
|
411
|
-
else
|
|
412
|
-
%(<span class="badge warning xs">#{h(server['status'].presence || 'unknown')} · not enabled for #{h(@agent_name)}</span>)
|
|
413
|
-
end
|
|
414
|
-
%(<div class="served"><span>served by</span><b>#{h(server['name'].presence || server['key'])}</b>#{badge}</div>)
|
|
415
|
-
end
|
|
416
|
-
|
|
417
|
-
# With a route the action is a button; without one, the page can only
|
|
418
|
-
# say where in the dashboard the fix lives. The link targets the top
|
|
419
|
-
# window: served in the dashboard's report iframe it would otherwise
|
|
420
|
-
# open the whole dashboard inside the frame.
|
|
421
|
-
def html_fix_action(action)
|
|
422
|
-
button = action["path"].present? ? %(<a class="btn" target="_top" href="#{h(action['path'])}">#{h(action['label'])}</a>) : ""
|
|
423
|
-
%(<div class="action">#{button}<span class="hint">#{h(action['hint'])}</span></div>)
|
|
424
|
-
end
|
|
425
|
-
|
|
426
|
-
# "3 scenarios · both models" — the models are worth naming only on a
|
|
427
|
-
# comparison run; on a single-model run the count says it all.
|
|
428
|
-
def fix_scope(item)
|
|
429
|
-
return "#{item['scenario_keys'].join(', ')} · judge suggestion" if item["kind"] == "instruction"
|
|
430
|
-
|
|
431
|
-
scenarios = plural(item["scenario_keys"].size, "scenario")
|
|
432
|
-
labels = Array(item["models"])
|
|
433
|
-
return scenarios unless comparing? && labels.any?
|
|
434
|
-
|
|
435
|
-
models =
|
|
436
|
-
if labels.size >= @models.size
|
|
437
|
-
@models.size == 2 ? "both models" : "all models"
|
|
438
|
-
else
|
|
439
|
-
labels.map { |label| short_name(model_by_label(label)) }.join(", ")
|
|
440
|
-
end
|
|
441
|
-
"#{scenarios} · #{models}"
|
|
442
|
-
end
|
|
443
|
-
|
|
444
426
|
# --- SCENARIOS matrix ------------------------------------------------
|
|
445
427
|
|
|
446
428
|
def html_matrix
|
|
@@ -448,7 +430,7 @@ module ActiveAgent
|
|
|
448
430
|
short, provider = split_label(spec)
|
|
449
431
|
%(<span class="col"><span class="name">#{h(short)}</span><span class="provider">#{h(provider)}</span></span>)
|
|
450
432
|
end
|
|
451
|
-
rows = [ %(<div class="mx head"><span class="micro sm">Scenario</span><span class="micro sm">Expects</span>#{columns.join}</div>) ]
|
|
433
|
+
rows = [ %(<div class="mx head"><span class="micro sm">Scenario</span><span class="micro sm">Expects</span>#{columns.join}<span class="micro sm cost-head" title="What the scenario cost across every model, and what judging it cost">Cost</span></div>) ]
|
|
452
434
|
scenario_groups.each do |cohorts|
|
|
453
435
|
rows << html_group_row(cohorts) if group_name(cohorts.first.first.scenario)
|
|
454
436
|
cohorts.each { |cohort| rows << html_scenario_row(cohort) }
|
|
@@ -476,10 +458,29 @@ module ActiveAgent
|
|
|
476
458
|
elsif passed.zero? then " text-error"
|
|
477
459
|
else ""
|
|
478
460
|
end
|
|
479
|
-
%(<span class="group-pass#{tone}">#{passed
|
|
461
|
+
%(<span class="group-pass#{tone}">#{results.empty? ? '—' : "#{h(fmt_passes(passed, results.size))} passed"}</span>)
|
|
480
462
|
end
|
|
481
463
|
%(<div class="mx group"><span class="group-name">#{h(name)}</span>) +
|
|
482
|
-
%(<span class="count">#{h(plural(cohorts.size, 'scenario'))}</span>#{passes.join}</div>)
|
|
464
|
+
%(<span class="count">#{h(plural(cohorts.size, 'scenario'))}</span>#{passes.join}#{html_group_cost(cohorts)}</div>)
|
|
465
|
+
end
|
|
466
|
+
|
|
467
|
+
# The group's subtotal — its scenarios' costs summed, judge apart — in
|
|
468
|
+
# the trailing column, so the grid stays aligned under a group row.
|
|
469
|
+
def html_group_cost(cohorts)
|
|
470
|
+
costs = cohorts.map { |cohort| scenario_costs[cohort.first.scenario.key] }.compact
|
|
471
|
+
agent = costs.filter_map { |entry| entry["cost"] }
|
|
472
|
+
judge = costs.filter_map { |entry| entry["judge_cost"] }
|
|
473
|
+
estimated = costs.any? { |entry| entry["estimated"] }
|
|
474
|
+
return %(<span class="cost"><span class="muted">—</span></span>) if agent.empty? && judge.empty?
|
|
475
|
+
|
|
476
|
+
html_cost_block(agent.any? ? agent.sum : nil, judge.any? ? judge.sum : nil, estimated: estimated)
|
|
477
|
+
end
|
|
478
|
+
|
|
479
|
+
# "<b>~$0.0243</b><span class="judge">judge ~$0.0015</span>" — a
|
|
480
|
+
# scenario's (or group's) spend across every model.
|
|
481
|
+
def html_cost_block(cost, judge_cost, estimated:)
|
|
482
|
+
judge = judge_cost ? %(<span class="judge">judge #{h(fmt_cost(judge_cost, estimated: estimated))}</span>) : ""
|
|
483
|
+
%(<span class="cost"><b>#{h(fmt_cost(cost, estimated: estimated))}</b>#{judge}</span>)
|
|
483
484
|
end
|
|
484
485
|
|
|
485
486
|
def html_scenario_row(cohort)
|
|
@@ -489,8 +490,14 @@ module ActiveAgent
|
|
|
489
490
|
result = cohort.find { |candidate| candidate.label == spec.label }
|
|
490
491
|
result ? html_result_cell(result) : %(<div class="cell"><div class="top"><span class="muted">—</span></div></div>)
|
|
491
492
|
end
|
|
493
|
+
costs = scenario_costs[scenario.key] || {}
|
|
494
|
+
total_cell = if costs["cost"].nil? && costs["judge_cost"].nil?
|
|
495
|
+
%(<span class="cost"><span class="muted">—</span></span>)
|
|
496
|
+
else
|
|
497
|
+
html_cost_block(costs["cost"], costs["judge_cost"], estimated: costs["estimated"])
|
|
498
|
+
end
|
|
492
499
|
%(<div class="mx"><div><div class="key"><a href="##{h(anchor(scenario))}">#{h(scenario.key)}</a></div>) +
|
|
493
|
-
%(<div class="prompt">#{h(scenario.prompt)}</div></div><div class="expects">#{expects}</div>#{cells.join}</div>)
|
|
500
|
+
%(<div class="prompt">#{h(scenario.prompt)}</div></div><div class="expects">#{expects}</div>#{cells.join}#{total_cell}</div>)
|
|
494
501
|
end
|
|
495
502
|
|
|
496
503
|
def html_result_cell(result)
|
|
@@ -499,7 +506,25 @@ module ActiveAgent
|
|
|
499
506
|
fault = result.fault ? %(<span class="f">#{h(fault_name(result.fault))}</span>) : ""
|
|
500
507
|
%(<div class="cell"><div class="top"><span class="g tone-#{tone}">#{glyph}</span>) +
|
|
501
508
|
%(<span class="s tone-#{tone}">#{h(fmt_score(result.score))}</span>#{fault}</div>) +
|
|
502
|
-
%(<div class="calls">#{html_calls(result, empty: 'no tools called')}</div
|
|
509
|
+
%(<div class="calls">#{html_calls(result, empty: 'no tools called')}</div>#{html_cell_cost(result)}</div>)
|
|
510
|
+
end
|
|
511
|
+
|
|
512
|
+
# "~$0.0243 · judge ~$0.0015" under a cell: what this answer cost and
|
|
513
|
+
# what judging it cost. A result with no cost at all shows no line.
|
|
514
|
+
def html_cell_cost(result)
|
|
515
|
+
text = result_cost_text(result)
|
|
516
|
+
return "" if text.nil?
|
|
517
|
+
|
|
518
|
+
%(<div class="cost-line"#{cost_title_attr([ result ], estimated: result.estimated_cost?)}>#{h(text)}</div>)
|
|
519
|
+
end
|
|
520
|
+
|
|
521
|
+
def result_cost_text(result)
|
|
522
|
+
judge_cost = result.judge_usage&.dig("cost")
|
|
523
|
+
return nil if result.replay.cost.nil? && judge_cost.nil?
|
|
524
|
+
|
|
525
|
+
parts = [ fmt_cost(result.replay.cost, estimated: result.estimated_cost?) ]
|
|
526
|
+
parts << "judge #{fmt_cost(judge_cost, estimated: estimated_usage?(result.judge_usage))}" if judge_cost
|
|
527
|
+
parts.join(" · ")
|
|
503
528
|
end
|
|
504
529
|
|
|
505
530
|
def html_calls(result, empty:)
|
|
@@ -580,20 +605,156 @@ module ActiveAgent
|
|
|
580
605
|
[
|
|
581
606
|
replay.duration_ms && fmt_ms(replay.duration_ms),
|
|
582
607
|
replay.total_tokens.positive? ? "#{fmt_k(replay.total_tokens)} tokens" : nil,
|
|
583
|
-
|
|
608
|
+
result_cost_text(result)
|
|
584
609
|
].compact.join(" · ")
|
|
585
610
|
end
|
|
586
611
|
|
|
612
|
+
# --- WHAT TO FIX -----------------------------------------------------
|
|
613
|
+
|
|
614
|
+
# The section stands even for a run with nothing to fix — the dashboard
|
|
615
|
+
# keeps it too, so a clean run reads as clean rather than as a page
|
|
616
|
+
# missing a section. A report over no results at all has nothing to say.
|
|
617
|
+
def html_fixes
|
|
618
|
+
return "" if @results.empty?
|
|
619
|
+
|
|
620
|
+
items = fix_items
|
|
621
|
+
faulted = @results.reject(&:passed?)
|
|
622
|
+
meta = "#{plural(items.size, 'item')} · #{plural(faulted.size, 'fault')} across " \
|
|
623
|
+
"#{plural(faulted.map { |result| result.scenario.key }.uniq.size, 'scenario')}"
|
|
624
|
+
body =
|
|
625
|
+
if items.any?
|
|
626
|
+
%(<div class="fixes">#{items.map { |item| html_fix_card(item) }.join}</div>)
|
|
627
|
+
else
|
|
628
|
+
%(<div class="nothing">[+] nothing to fix</div>)
|
|
629
|
+
end
|
|
630
|
+
|
|
631
|
+
<<~FIXES
|
|
632
|
+
<section class="section fix-section" aria-label="Recommendations">
|
|
633
|
+
<div class="section-head"><span class="micro">What to fix</span><span class="meta">#{h(meta)}</span></div>
|
|
634
|
+
#{html_fix_filter(items) if comparing? && items.any?}
|
|
635
|
+
#{body}
|
|
636
|
+
</section>
|
|
637
|
+
FIXES
|
|
638
|
+
end
|
|
639
|
+
|
|
640
|
+
# A model filter for the fix cards — a fault one model keeps making is
|
|
641
|
+
# that model's to fix, so the list narrows to what was attributed to
|
|
642
|
+
# it. Radio chips and stylesheet rules alone (the page carries no
|
|
643
|
+
# script): each card names its models in data-models, and a checked
|
|
644
|
+
# model hides every card that does not name it. Cards attributed to no
|
|
645
|
+
# model (an older run) stay under every filter.
|
|
646
|
+
def html_fix_filter(items)
|
|
647
|
+
chips = [ %(<label class="chip pick-model"><input type="radio" name="fix-model" value="all" checked><span>all models #{items.size}</span></label>) ]
|
|
648
|
+
@models.each_with_index do |spec, index|
|
|
649
|
+
count = items.count { |item| Array(item["models"]).empty? || item["models"].include?(spec.label) }
|
|
650
|
+
chips << %(<label class="chip pick-model"><input type="radio" name="fix-model" value="m#{index}"><span>#{h(short_name(spec))} #{count}</span></label>)
|
|
651
|
+
end
|
|
652
|
+
%(<div class="fix-filter"><span class="micro sm">for</span>#{chips.join}</div>)
|
|
653
|
+
end
|
|
654
|
+
|
|
655
|
+
def fix_model_tokens(item)
|
|
656
|
+
labels = Array(item["models"])
|
|
657
|
+
return "" if labels.empty?
|
|
658
|
+
|
|
659
|
+
labels.filter_map { |label| (index = @models.index(model_by_label(label))) && "m#{index}" }.join(" ")
|
|
660
|
+
end
|
|
661
|
+
|
|
662
|
+
def html_fix_card(item)
|
|
663
|
+
tone = item["kind"] == "instruction" ? "info" : "error"
|
|
664
|
+
glyph = tone == "info" ? "[i]" : "[!]"
|
|
665
|
+
title = fault_name(item["fault"]) + (item["count"].to_i > 1 ? " ×#{item['count']}" : "")
|
|
666
|
+
|
|
667
|
+
parts = [ %(<div class="head"><span class="glyph tone-#{tone}">#{glyph}</span>) +
|
|
668
|
+
%(<span class="badge #{tone}">#{h(title)}</span><span class="scope">#{h(fix_scope(item))}</span></div>) ]
|
|
669
|
+
parts << %(<p>#{h(item['recommendation'])}</p>) if item["recommendation"].present?
|
|
670
|
+
parts << %(<div class="quote">“#{h(item['quote'])}”</div>) if item["quote"].present?
|
|
671
|
+
parts << html_fix_tools(item) if item["tools"].any?
|
|
672
|
+
parts << html_fix_server(item["server"]) if item["server"]
|
|
673
|
+
parts << %(<div class="note">#{h(item['note'])}</div>) if item["note"].present?
|
|
674
|
+
parts << html_fix_action(item["action"]) if item["action"]
|
|
675
|
+
models = fix_model_tokens(item)
|
|
676
|
+
%(<div class="fix"#{%( data-models="#{models}") if models.present?}>#{parts.join}</div>)
|
|
677
|
+
end
|
|
678
|
+
|
|
679
|
+
def html_fix_tools(item)
|
|
680
|
+
chips = item["tools"].map do |tool|
|
|
681
|
+
note = tool["note"].presence
|
|
682
|
+
%(<span class="tool"><b>#{h(tool['name'])}</b>#{%(<span class="note">#{h(note)}</span>) if note}</span>)
|
|
683
|
+
end
|
|
684
|
+
%(<div class="tools"><span class="micro sm">#{h(item['tools_label'])}</span><div class="list">#{chips.join}</div></div>)
|
|
685
|
+
end
|
|
686
|
+
|
|
687
|
+
# "available · not enabled for Assistant", "unknown · not enabled for
|
|
688
|
+
# Assistant" — every status but "enabled" leads with the status word, the
|
|
689
|
+
# way the dashboard's fix list reads it.
|
|
690
|
+
def html_fix_server(server)
|
|
691
|
+
badge =
|
|
692
|
+
if server["status"] == "enabled"
|
|
693
|
+
%(<span class="badge success xs">enabled for #{h(@agent_name)}</span>)
|
|
694
|
+
else
|
|
695
|
+
%(<span class="badge warning xs">#{h(server['status'].presence || 'unknown')} · not enabled for #{h(@agent_name)}</span>)
|
|
696
|
+
end
|
|
697
|
+
%(<div class="served"><span>served by</span><b>#{h(server['name'].presence || server['key'])}</b>#{badge}</div>)
|
|
698
|
+
end
|
|
699
|
+
|
|
700
|
+
# With a route the action is a button; without one, the page can only
|
|
701
|
+
# say where in the dashboard the fix lives. The link targets the top
|
|
702
|
+
# window: served in the dashboard's report iframe it would otherwise
|
|
703
|
+
# open the whole dashboard inside the frame.
|
|
704
|
+
def html_fix_action(action)
|
|
705
|
+
button = action["path"].present? ? %(<a class="btn" target="_top" href="#{h(action['path'])}">#{h(action['label'])}</a>) : ""
|
|
706
|
+
%(<div class="action">#{button}<span class="hint">#{h(action['hint'])}</span></div>)
|
|
707
|
+
end
|
|
708
|
+
|
|
709
|
+
# "3 scenarios · both models" — the models are worth naming only on a
|
|
710
|
+
# comparison run; on a single-model run the count says it all.
|
|
711
|
+
def fix_scope(item)
|
|
712
|
+
return "#{item['scenario_keys'].join(', ')} · judge suggestion" if item["kind"] == "instruction"
|
|
713
|
+
|
|
714
|
+
scenarios = plural(item["scenario_keys"].size, "scenario")
|
|
715
|
+
labels = Array(item["models"])
|
|
716
|
+
return scenarios unless comparing? && labels.any?
|
|
717
|
+
|
|
718
|
+
models =
|
|
719
|
+
if labels.size >= @models.size
|
|
720
|
+
@models.size == 2 ? "both models" : "all models"
|
|
721
|
+
else
|
|
722
|
+
labels.map { |label| short_name(model_by_label(label)) }.join(", ")
|
|
723
|
+
end
|
|
724
|
+
"#{scenarios} · #{models}"
|
|
725
|
+
end
|
|
726
|
+
|
|
587
727
|
# --- footer ----------------------------------------------------------
|
|
588
728
|
|
|
729
|
+
# The run's terms: the judge, the criteria and the pass mark, what it
|
|
730
|
+
# cost on each side, the metadata, and — once, when any figure on the
|
|
731
|
+
# page carries a "~" — what the mark means.
|
|
589
732
|
def html_footer
|
|
590
733
|
criteria = criterion_keys.map { |key| key.to_s.tr("_", " ") }.join(" · ")
|
|
591
734
|
spans = [ %(<span class="nowrap">judge #{h(judge_name)}</span>) ]
|
|
592
735
|
spans << %(<span class="criteria">criteria #{h(criteria)}</span>) if criteria.present?
|
|
593
|
-
spans
|
|
736
|
+
spans << %(<span class="nowrap">#{h(Format.threshold(@threshold))}</span>)
|
|
737
|
+
spans << %(<span class="nowrap">#{h(footer_cost_text)}</span>) if footer_cost_text
|
|
738
|
+
spans << %(<span class="nowrap">release #{h(release_label)}</span>) if release_label
|
|
739
|
+
spans.concat(metadata_chips.map { |key, value| %(<span class="nowrap">#{h(key)} #{h(value)}</span>) })
|
|
740
|
+
spans << %(<span class="legend">#{h(Format::LEGEND)}</span>) if estimated_anywhere?
|
|
594
741
|
%(<footer>#{spans.join}</footer>)
|
|
595
742
|
end
|
|
596
743
|
|
|
744
|
+
# "cost ~$0.0412 · agent ~$0.0397 · judge ~$0.0015", or nil when
|
|
745
|
+
# nothing was priced.
|
|
746
|
+
def footer_cost_text
|
|
747
|
+
costs = run_costs
|
|
748
|
+
return nil if costs["total"].nil?
|
|
749
|
+
|
|
750
|
+
parts = [ "cost #{fmt_cost(costs['total'], estimated: costs['estimated'])}" ]
|
|
751
|
+
if judge_usage
|
|
752
|
+
parts << "agent #{fmt_cost(costs['cost'], estimated: costs['estimated'])}"
|
|
753
|
+
parts << "judge #{fmt_cost(costs['judge_cost'], estimated: judge_estimated?)}"
|
|
754
|
+
end
|
|
755
|
+
parts.join(" · ")
|
|
756
|
+
end
|
|
757
|
+
|
|
597
758
|
# Colors only through the token variables; radii 4 badges · 6 chips ·
|
|
598
759
|
# 8 controls · 10 nested panels · 12 cards · 999 bars; no shadows.
|
|
599
760
|
STYLES = <<~CSS.freeze
|
|
@@ -701,6 +862,10 @@ module ActiveAgent
|
|
|
701
862
|
.group-name { font-size: 12px; font-weight: 600; }
|
|
702
863
|
.count { font-family: var(--font-mono); font-size: 11px; color: var(--color-text-muted); }
|
|
703
864
|
.group-pass { font-family: var(--font-mono); font-size: 11px; font-weight: 600; color: var(--color-text-cell); }
|
|
865
|
+
.mx .cost { display: flex; flex-direction: column; gap: 1px; min-width: 0; font-family: var(--font-mono); font-size: 11px; color: var(--color-text-secondary); text-align: right; }
|
|
866
|
+
.mx .cost b { font-weight: 600; color: var(--color-text-primary); }
|
|
867
|
+
.mx .cost .judge { color: var(--color-text-muted); white-space: nowrap; }
|
|
868
|
+
.mx .cost-head { text-align: right; }
|
|
704
869
|
.key { font-family: var(--font-mono); font-size: 11px; color: var(--color-text-muted); margin-bottom: 2px; }
|
|
705
870
|
.key a { color: inherit; }
|
|
706
871
|
.prompt { font-size: 13px; line-height: 18px; color: var(--color-text-primary); }
|
|
@@ -714,6 +879,7 @@ module ActiveAgent
|
|
|
714
879
|
.calls { display: flex; flex-wrap: wrap; gap: 2px 8px; font-family: var(--font-mono); font-size: 11px; color: var(--color-text-muted); }
|
|
715
880
|
.call-hit { color: var(--color-success-text); font-weight: 600; }
|
|
716
881
|
.call-err { color: var(--color-error); font-weight: 600; }
|
|
882
|
+
.cost-line { font-family: var(--font-mono); font-size: 11px; color: var(--color-text-secondary); white-space: nowrap; }
|
|
717
883
|
.details { display: flex; flex-direction: column; gap: 8px; }
|
|
718
884
|
details { border: 1px solid var(--color-border-light); border-radius: 10px; overflow: hidden; }
|
|
719
885
|
summary { display: flex; align-items: center; gap: 10px; padding: 10px 12px; cursor: pointer; list-style: none; flex-wrap: wrap; }
|
|
@@ -739,6 +905,7 @@ module ActiveAgent
|
|
|
739
905
|
footer { display: flex; align-items: center; gap: 16px; flex-wrap: wrap; padding-top: 12px; border-top: 1px solid var(--color-border-light); font-family: var(--font-mono); font-size: 11px; color: var(--color-text-muted); }
|
|
740
906
|
footer .criteria { min-width: 0; }
|
|
741
907
|
footer .nowrap { white-space: nowrap; }
|
|
908
|
+
footer .legend { margin-left: auto; white-space: nowrap; color: var(--color-text-muted); }
|
|
742
909
|
CSS
|
|
743
910
|
end
|
|
744
911
|
end
|