alignmenter 0.0.4__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (41) hide show
  1. alignmenter/__init__.py +14 -0
  2. alignmenter/cli.py +1815 -0
  3. alignmenter/config.py +99 -0
  4. alignmenter/data/configs/demo_config.yaml +15 -0
  5. alignmenter/data/configs/judges/safety_prompt.txt +2 -0
  6. alignmenter/data/configs/persona/default.yaml +15 -0
  7. alignmenter/data/configs/run.yaml +12 -0
  8. alignmenter/data/configs/safety_keywords.yaml +7 -0
  9. alignmenter/data/datasets/demo_conversations.jsonl +60 -0
  10. alignmenter/providers/__init__.py +47 -0
  11. alignmenter/providers/anthropic.py +87 -0
  12. alignmenter/providers/base.py +57 -0
  13. alignmenter/providers/classifiers.py +83 -0
  14. alignmenter/providers/embeddings.py +126 -0
  15. alignmenter/providers/judges.py +105 -0
  16. alignmenter/providers/local.py +102 -0
  17. alignmenter/providers/openai.py +151 -0
  18. alignmenter/reporting/__init__.py +6 -0
  19. alignmenter/reporting/html.py +721 -0
  20. alignmenter/reporting/json_out.py +33 -0
  21. alignmenter/run_config.py +106 -0
  22. alignmenter/runner.py +410 -0
  23. alignmenter/scorers/__init__.py +7 -0
  24. alignmenter/scorers/authenticity.py +337 -0
  25. alignmenter/scorers/safety.py +231 -0
  26. alignmenter/scorers/stability.py +104 -0
  27. alignmenter/scripts/__init__.py +1 -0
  28. alignmenter/scripts/bootstrap_dataset.py +142 -0
  29. alignmenter/scripts/calibrate_persona.py +196 -0
  30. alignmenter/scripts/run_openai_demo.py +74 -0
  31. alignmenter/scripts/sanitize_dataset.py +185 -0
  32. alignmenter/utils/__init__.py +7 -0
  33. alignmenter/utils/io.py +47 -0
  34. alignmenter/utils/tokens.py +46 -0
  35. alignmenter/utils/yaml.py +15 -0
  36. alignmenter-0.0.4.dist-info/METADATA +681 -0
  37. alignmenter-0.0.4.dist-info/RECORD +41 -0
  38. alignmenter-0.0.4.dist-info/WHEEL +5 -0
  39. alignmenter-0.0.4.dist-info/entry_points.txt +2 -0
  40. alignmenter-0.0.4.dist-info/licenses/LICENSE +201 -0
  41. alignmenter-0.0.4.dist-info/top_level.txt +1 -0
@@ -0,0 +1,721 @@
1
+ """HTML report generator."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from pathlib import Path
6
+ from typing import Any
7
+
8
+
9
+ HTML_TEMPLATE = """<!DOCTYPE html>
10
+ <html lang=\"en\">
11
+ <head>
12
+ <meta charset=\"utf-8\" />
13
+ <title>Alignmenter Report - {run_id}</title>
14
+ <link rel=\"icon\" type=\"image/png\" href=\"favicon.png\">
15
+ <style>
16
+ body {{ font-family: -apple-system, BlinkMacSystemFont, Segoe UI, sans-serif; background: #0f172a; color: #e2e8f0; margin: 0; padding: 32px; }}
17
+ h1, h2 {{ color: #22d3ee; }}
18
+ section {{ margin-bottom: 32px; }}
19
+ table {{ width: 100%; border-collapse: collapse; margin-top: 16px; }}
20
+ th, td {{ padding: 12px; text-align: left; border-bottom: 1px solid #1e293b; }}
21
+ th {{ background: #1e293b; }}
22
+ .meta {{ display: grid; grid-template-columns: repeat(auto-fit, minmax(220px, 1fr)); gap: 12px; }}
23
+ .meta div {{ background: #1e293b; padding: 12px; border-radius: 8px; }}
24
+ .report-card {{ background: linear-gradient(135deg, #1e293b 0%, #0f172a 100%); border-radius: 16px; padding: 0; box-shadow: 0 20px 60px rgba(14,74,104,0.4); border: 1px solid #334155; max-width: 1200px; margin: 0 auto; }}
25
+ .report-card-header {{ background: linear-gradient(135deg, #1e293b 0%, #0f172a 100%); padding: 32px 40px; border-radius: 16px 16px 0 0; border-bottom: 2px solid #22d3ee; }}
26
+ .report-card-header-content {{ display: flex; align-items: center; gap: 20px; margin-bottom: 20px; }}
27
+ .report-logo {{ height: 48px; width: auto; }}
28
+ .report-card-title {{ margin: 0; font-size: 2rem; font-weight: 800; letter-spacing: -0.02em; color: #22d3ee; }}
29
+ .report-info-grid {{ display: grid; grid-template-columns: repeat(auto-fit, minmax(200px, 1fr)); gap: 16px; margin-top: 20px; }}
30
+ .report-info-item {{ background: rgba(34, 211, 238, 0.05); padding: 12px 16px; border-radius: 8px; border: 1px solid rgba(34, 211, 238, 0.1); }}
31
+ .report-info-label {{ font-size: 0.75rem; text-transform: uppercase; letter-spacing: 0.05em; color: #94a3b8; margin-bottom: 4px; font-weight: 600; }}
32
+ .report-info-value {{ font-size: 1rem; font-weight: 700; color: #e2e8f0; }}
33
+ .report-card-body {{ padding: 40px; }}
34
+ .overall-grade {{ text-align: center; margin-bottom: 32px; padding: 24px; background: rgba(34, 211, 238, 0.05); border-radius: 12px; border: 2px solid rgba(34, 211, 238, 0.2); }}
35
+ .overall-grade-label {{ font-size: 0.9rem; color: #94a3b8; text-transform: uppercase; letter-spacing: 0.1em; margin-bottom: 8px; }}
36
+ .overall-grade-value {{ font-size: 4rem; font-weight: 800; line-height: 1; }}
37
+ .overall-grade-value.pass {{ color: #4ade80; }}
38
+ .overall-grade-value.warn {{ color: #fbbf24; }}
39
+ .overall-grade-value.fail {{ color: #f87171; }}
40
+ .overall-grade-desc {{ font-size: 0.9rem; color: #94a3b8; margin-top: 8px; }}
41
+ .grade-table {{ width: 100%; border-collapse: separate; border-spacing: 0 8px; }}
42
+ .grade-table thead th {{ padding: 12px 16px; text-align: left; font-size: 0.85rem; text-transform: uppercase; letter-spacing: 0.05em; color: #94a3b8; font-weight: 600; border-bottom: 2px solid #334155; }}
43
+ .grade-table thead th:nth-child(2), .grade-table thead th:nth-child(3), .grade-table thead th:nth-child(4) {{ text-align: center; }}
44
+ .grade-table tbody tr {{ background: #1e293b; }}
45
+ .grade-table tbody td {{ padding: 20px 16px; border-top: 1px solid #334155; border-bottom: 1px solid #334155; }}
46
+ .grade-table tbody td:first-child {{ border-left: 1px solid #334155; border-radius: 8px 0 0 8px; }}
47
+ .grade-table tbody td:last-child {{ border-right: 1px solid #334155; border-radius: 0 8px 8px 0; }}
48
+ .metric-name {{ font-weight: 600; color: #e2e8f0; font-size: 1.1rem; }}
49
+ .grade-cell {{ text-align: center; }}
50
+ .grade-badge {{ display: inline-flex; align-items: center; gap: 8px; padding: 8px 16px; border-radius: 8px; font-weight: 700; font-size: 1.1rem; }}
51
+ .grade-badge.pass {{ background: rgba(74, 222, 128, 0.15); color: #4ade80; border: 2px solid rgba(74, 222, 128, 0.3); }}
52
+ .grade-badge.warn {{ background: rgba(251, 191, 36, 0.15); color: #fbbf24; border: 2px solid rgba(251, 191, 36, 0.3); }}
53
+ .grade-badge.fail {{ background: rgba(248, 113, 113, 0.15); color: #f87171; border: 2px solid rgba(248, 113, 113, 0.3); }}
54
+ .grade-letter {{ font-size: 1.3rem; font-weight: 800; }}
55
+ .grade-score {{ font-size: 0.95rem; opacity: 0.9; }}
56
+ .compare-cell {{ text-align: center; color: #94a3b8; font-size: 0.95rem; }}
57
+ .delta-cell {{ text-align: center; font-size: 1rem; font-weight: 700; }}
58
+ .delta-cell.positive {{ color: #4ade80; }}
59
+ .delta-cell.negative {{ color: #f87171; }}
60
+ .delta-cell.neutral {{ color: #64748b; }}
61
+ .turn-table td {{ vertical-align: top; }}
62
+ .muted {{ color: #94a3b8; font-size: 0.85rem; }}
63
+ code {{ background: rgba(15,23,42,0.6); padding: 2px 6px; border-radius: 4px; font-size: 0.85rem; }}
64
+ .score-pass {{ background: rgba(74, 222, 128, 0.1); color: #4ade80; font-weight: 600; }}
65
+ .score-warn {{ background: rgba(251, 191, 36, 0.1); color: #fbbf24; font-weight: 600; }}
66
+ .score-fail {{ background: rgba(248, 113, 113, 0.1); color: #f87171; font-weight: 600; }}
67
+ .calibration-section {{ background: #1e293b; padding: 16px; border-radius: 8px; margin-top: 16px; }}
68
+ .calibration-grid {{ display: grid; grid-template-columns: repeat(auto-fit, minmax(250px, 1fr)); gap: 12px; margin-top: 12px; }}
69
+ .calibration-item {{ background: #0f172a; padding: 12px; border-radius: 6px; }}
70
+ .calibration-item strong {{ color: #22d3ee; display: block; margin-bottom: 4px; }}
71
+ .reproducibility-section {{ background: #1e293b; padding: 16px; border-radius: 8px; }}
72
+ .reproducibility-grid {{ display: grid; grid-template-columns: repeat(auto-fit, minmax(400px, 1fr)); gap: 12px; margin-top: 12px; }}
73
+ .reproducibility-grid > div {{ overflow-wrap: break-word; word-break: break-all; }}
74
+ .export-buttons {{ margin-top: 8px; }}
75
+ .export-btn {{ background: #1e293b; color: #22d3ee; border: 1px solid #22d3ee; padding: 6px 12px; border-radius: 6px; cursor: pointer; text-decoration: none; display: inline-block; margin-right: 8px; font-size: 0.85rem; }}
76
+ .export-btn:hover {{ background: #22d3ee; color: #0f172a; }}
77
+ .chart-container {{ margin-top: 16px; background: #1e293b; padding: 16px; border-radius: 8px; }}
78
+ canvas {{ max-width: 100%; }}
79
+ </style>
80
+ <script src="https://cdn.jsdelivr.net/npm/chart.js@4.4.0/dist/chart.umd.min.js"></script>
81
+ <script>
82
+ function downloadJSON(data, filename) {{
83
+ const blob = new Blob([JSON.stringify(data, null, 2)], {{ type: 'application/json' }});
84
+ const url = URL.createObjectURL(blob);
85
+ const a = document.createElement('a');
86
+ a.href = url;
87
+ a.download = filename;
88
+ a.click();
89
+ URL.revokeObjectURL(url);
90
+ }}
91
+
92
+ function downloadCSV(data, filename) {{
93
+ const rows = [];
94
+ if (data.length > 0) {{
95
+ rows.push(Object.keys(data[0]).join(','));
96
+ data.forEach(row => {{
97
+ rows.push(Object.values(row).join(','));
98
+ }});
99
+ }}
100
+ const csv = rows.join('\\n');
101
+ const blob = new Blob([csv], {{ type: 'text/csv' }});
102
+ const url = URL.createObjectURL(blob);
103
+ const a = document.createElement('a');
104
+ a.href = url;
105
+ a.download = filename;
106
+ a.click();
107
+ URL.revokeObjectURL(url);
108
+ }}
109
+ </script>
110
+ </head>
111
+ <body>
112
+ {scorecard_block}
113
+ {calibration_section}
114
+ <section>
115
+ <h2>Scores</h2>
116
+ <div class="export-buttons">
117
+ <button class="export-btn" onclick="downloadJSON(window.scoresData, 'scores.json')">Download JSON</button>
118
+ <button class="export-btn" onclick="downloadCSV(window.scoresDataCSV, 'scores.csv')">Download CSV</button>
119
+ </div>
120
+ {score_tables}
121
+ {charts_section}
122
+ </section>
123
+ {reproducibility_section}
124
+ <section>
125
+ <h2>Turn-Level Explorer</h2>
126
+ {turn_preview}
127
+ </section>
128
+ <script>
129
+ window.scoresData = {scores_json};
130
+ window.scoresDataCSV = {scores_csv_json};
131
+ </script>
132
+ </body>
133
+ </html>
134
+ """
135
+
136
+
137
+ class HTMLReporter:
138
+ """Generate a minimal HTML report."""
139
+
140
+ def write(
141
+ self,
142
+ run_dir: Path,
143
+ summary: dict[str, Any],
144
+ scores: dict[str, Any],
145
+ sessions: list,
146
+ **extras: Any,
147
+ ) -> Path:
148
+ scorecards = extras.get("scorecards", [])
149
+
150
+ primary = scores.get("primary", {}) if isinstance(scores, dict) else {}
151
+ compare = scores.get("compare", {}) if isinstance(scores, dict) else {}
152
+ diff = scores.get("diff", {}) if isinstance(scores, dict) else {}
153
+
154
+ score_blocks = []
155
+ scorer_ids = sorted({*primary.keys(), *compare.keys()}) or list(scores.keys())
156
+
157
+ for scorer_id in scorer_ids:
158
+ primary_metrics = primary.get(scorer_id, {}) if isinstance(primary, dict) else {}
159
+ compare_metrics = compare.get(scorer_id, {}) if isinstance(compare, dict) else {}
160
+ diff_metrics = diff.get(scorer_id, {}) if isinstance(diff, dict) else {}
161
+
162
+ metric_keys = sorted({*primary_metrics.keys(), *compare_metrics.keys(), *diff_metrics.keys()}) or ["value"]
163
+
164
+ if not isinstance(primary_metrics, dict) and scorer_id in primary:
165
+ primary_metrics = {"value": primary[scorer_id]}
166
+ if not isinstance(compare_metrics, dict) and scorer_id in compare:
167
+ compare_metrics = {"value": compare[scorer_id]}
168
+ if not isinstance(diff_metrics, dict) and scorer_id in diff:
169
+ diff_metrics = {"value": diff[scorer_id]}
170
+
171
+ has_compare = bool(compare_metrics)
172
+ header = "<th>Metric</th><th>Primary</th>"
173
+ if has_compare:
174
+ header += "<th>Compare</th><th>Δ</th>"
175
+
176
+ row_html = []
177
+ for key in metric_keys:
178
+ primary_val = _format_metric(primary_metrics.get(key), metric_key=key)
179
+ compare_val = _format_metric(compare_metrics.get(key), metric_key=key) if has_compare else ""
180
+ delta_val = _format_metric(diff_metrics.get(key), metric_key=key, apply_color=False) if has_compare else ""
181
+ if has_compare:
182
+ row_html.append(
183
+ f"<tr><td>{key}</td><td>{primary_val}</td><td>{compare_val}</td><td>{delta_val}</td></tr>"
184
+ )
185
+ else:
186
+ row_html.append(f"<tr><td>{key}</td><td>{primary_val}</td></tr>")
187
+
188
+ table = (
189
+ f"<h3>{scorer_id.title()}</h3>"
190
+ f"<table><thead><tr>{header}</tr></thead><tbody>{''.join(row_html)}</tbody></table>"
191
+ )
192
+ if scorer_id == "safety":
193
+ safety_details = _render_judge_details(primary_metrics)
194
+ if safety_details:
195
+ table += safety_details
196
+ score_blocks.append(table)
197
+
198
+ turn_preview = _render_turn_preview(sessions)
199
+ calibration_section = _render_calibration_section(primary)
200
+ reproducibility_section = _render_reproducibility_section(summary)
201
+ charts_section = _render_charts(primary)
202
+ scorecard_block = _render_scorecards(scorecards, summary)
203
+
204
+ # Prepare data for export
205
+ import json
206
+ scores_json = json.dumps(scores)
207
+ scores_csv_data = _prepare_csv_data(primary)
208
+ scores_csv_json = json.dumps(scores_csv_data)
209
+
210
+ html = HTML_TEMPLATE.format(
211
+ run_id=summary.get("run_id", "alignmenter_run"),
212
+ scorecard_block=scorecard_block,
213
+ score_tables="".join(score_blocks) or "<p>No scores computed.</p>",
214
+ turn_preview=turn_preview,
215
+ calibration_section=calibration_section,
216
+ reproducibility_section=reproducibility_section,
217
+ charts_section=charts_section,
218
+ scores_json=scores_json,
219
+ scores_csv_json=scores_csv_json,
220
+ )
221
+
222
+ path = Path(run_dir) / "index.html"
223
+ path.write_text(html, encoding="utf-8")
224
+
225
+ # Copy logo and favicon to report directory
226
+ import shutil
227
+ assets_dir = Path(__file__).parent.parent.parent.parent.parent / "assets"
228
+
229
+ # Copy logo
230
+ logo_source = assets_dir / "alignmenter-transparent.png"
231
+ if logo_source.exists():
232
+ logo_dest = Path(run_dir) / "logo.png"
233
+ shutil.copy2(logo_source, logo_dest)
234
+
235
+ # Copy favicon (use the transparent icon version)
236
+ favicon_source = assets_dir / "alignmenter-transparent.png"
237
+ if favicon_source.exists():
238
+ favicon_dest = Path(run_dir) / "favicon.png"
239
+ shutil.copy2(favicon_source, favicon_dest)
240
+
241
+ return path
242
+
243
+
244
+ def _format_metric(value: Any, metric_key: str = "", apply_color: bool = True) -> str:
245
+ if value is None:
246
+ return "—"
247
+
248
+ formatted = ""
249
+ if isinstance(value, float):
250
+ formatted = f"{value:.3f}"
251
+ elif isinstance(value, list):
252
+ formatted = ", ".join(str(item) for item in value)
253
+ else:
254
+ formatted = str(value)
255
+
256
+ # Apply color coding for score metrics
257
+ if apply_color and isinstance(value, (int, float)) and metric_key in ("mean", "score", "stability", "rule_score", "fused_judge"):
258
+ css_class = _get_score_class(value)
259
+ return f'<span class="{css_class}">{formatted}</span>'
260
+
261
+ return formatted
262
+
263
+
264
+ def _get_score_class(score: float) -> str:
265
+ """Get CSS class based on score threshold."""
266
+ if score >= 0.8:
267
+ return "score-pass"
268
+ elif score >= 0.6:
269
+ return "score-warn"
270
+ else:
271
+ return "score-fail"
272
+
273
+
274
+ def _render_judge_details(metrics: dict[str, Any]) -> str:
275
+ if not isinstance(metrics, dict):
276
+ return ""
277
+ calls = metrics.get("judge_calls")
278
+ budget = metrics.get("judge_budget")
279
+ mean_score = metrics.get("judge_mean")
280
+ notes = metrics.get("judge_notes") or []
281
+
282
+ if calls is None and not notes and mean_score is None:
283
+ return ""
284
+
285
+ lines = []
286
+ if calls is not None:
287
+ info = f"Judge calls: {calls}"
288
+ if budget:
289
+ info += f" / budget {budget}"
290
+ lines.append(info)
291
+ if mean_score is not None:
292
+ lines.append(f"Average judge score: {mean_score:.3f}")
293
+ if notes:
294
+ notes_html = "".join(f"<li>{note}</li>" for note in notes)
295
+ lines.append(f"<ul>{notes_html}</ul>")
296
+
297
+ body = "<br />".join(item for item in lines if not item.startswith("<ul>"))
298
+ list_html = "".join(item for item in lines if item.startswith("<ul>"))
299
+ return f"<div class='muted'>{body}{list_html}</div>"
300
+
301
+
302
+ def _render_scorecards(scorecards: list[dict], summary: dict[str, Any]) -> str:
303
+ if not scorecards:
304
+ return ""
305
+
306
+ has_compare = any(card.get("compare") is not None for card in scorecards)
307
+
308
+ # Calculate overall grade
309
+ scores = [card.get("primary") for card in scorecards if isinstance(card.get("primary"), (int, float))]
310
+ overall_score = sum(scores) / len(scores) if scores else 0
311
+ overall_class = _get_grade_class(overall_score)
312
+ overall_letter = _get_grade_letter(overall_score)
313
+
314
+ # Build run info
315
+ model = summary.get("model", "Unknown")
316
+ run_at = summary.get("run_at", "Unknown")
317
+ session_count = summary.get("session_count", 0)
318
+ turn_count = summary.get("turn_count", 0)
319
+ run_id = summary.get("run_id", "alignmenter_run")
320
+
321
+ run_info = f"""
322
+ <div class="report-info-grid">
323
+ <div class="report-info-item">
324
+ <div class="report-info-label">Model</div>
325
+ <div class="report-info-value">{model}</div>
326
+ </div>
327
+ <div class="report-info-item">
328
+ <div class="report-info-label">Run ID</div>
329
+ <div class="report-info-value">{run_id}</div>
330
+ </div>
331
+ <div class="report-info-item">
332
+ <div class="report-info-label">Timestamp</div>
333
+ <div class="report-info-value">{run_at}</div>
334
+ </div>
335
+ <div class="report-info-item">
336
+ <div class="report-info-label">Dataset</div>
337
+ <div class="report-info-value">{session_count} sessions · {turn_count} turns</div>
338
+ </div>
339
+ </div>
340
+ """
341
+
342
+ # Build overall grade section
343
+ grade_desc = {
344
+ "A": "Excellent performance across all metrics",
345
+ "B": "Good performance with room for improvement",
346
+ "C": "Needs attention in one or more areas"
347
+ }.get(overall_letter, "")
348
+
349
+ overall_grade_html = f"""
350
+ <div class="overall-grade">
351
+ <div class="overall-grade-label">Overall Grade</div>
352
+ <div class="overall-grade-value {overall_class}">{overall_letter}</div>
353
+ <div class="overall-grade-desc">{grade_desc}</div>
354
+ </div>
355
+ """
356
+
357
+ # Build table
358
+ table_header = '<thead><tr>'
359
+ table_header += '<th>Metric</th>'
360
+ table_header += '<th>Grade</th>'
361
+ if has_compare:
362
+ table_header += '<th>Compare</th>'
363
+ table_header += '<th>Change</th>'
364
+ table_header += '</tr></thead>'
365
+
366
+ # Build rows
367
+ rows = []
368
+ for card in scorecards:
369
+ primary_val = card.get("primary")
370
+ compare_val = card.get("compare")
371
+ diff_val = card.get("diff")
372
+
373
+ # Determine grade class
374
+ grade_class = _get_grade_class(primary_val)
375
+ letter = _get_grade_letter(primary_val)
376
+
377
+ row = '<tr>'
378
+ row += f'<td><span class="metric-name">{card.get("label", card.get("id", "Metric").title())}</span></td>'
379
+ row += f'<td class="grade-cell"><div class="grade-badge {grade_class}">'
380
+ row += f'<span class="grade-letter">{letter}</span>'
381
+ row += f'<span class="grade-score">{_format_scorecard_value(primary_val)}</span>'
382
+ row += '</div></td>'
383
+
384
+ if has_compare:
385
+ if compare_val is not None:
386
+ compare_letter = _get_grade_letter(compare_val)
387
+ row += f'<td class="compare-cell">{compare_letter} {_format_scorecard_value(compare_val)}</td>'
388
+ else:
389
+ row += '<td class="compare-cell">—</td>'
390
+
391
+ if diff_val is not None and isinstance(diff_val, (int, float)):
392
+ delta_class = "positive" if diff_val >= 0 else "negative"
393
+ delta_sign = "+" if diff_val >= 0 else ""
394
+ row += f'<td class="delta-cell {delta_class}">{delta_sign}{_format_scorecard_value(diff_val)}</td>'
395
+ else:
396
+ row += '<td class="delta-cell neutral">—</td>'
397
+
398
+ row += '</tr>'
399
+ rows.append(row)
400
+
401
+ return f"""
402
+ <section>
403
+ <div class="report-card">
404
+ <div class="report-card-header">
405
+ <div class="report-card-header-content">
406
+ <img src="logo.png" alt="Alignmenter" class="report-logo" onerror="this.style.display='none'">
407
+ <h1 class="report-card-title">Alignmenter Report Card</h1>
408
+ </div>
409
+ {run_info}
410
+ </div>
411
+ <div class="report-card-body">
412
+ {overall_grade_html}
413
+ <table class="grade-table">
414
+ {table_header}
415
+ <tbody>
416
+ {''.join(rows)}
417
+ </tbody>
418
+ </table>
419
+ </div>
420
+ </div>
421
+ </section>
422
+ """
423
+
424
+
425
+ def _get_grade_class(score: Any) -> str:
426
+ """Get CSS class for grade styling."""
427
+ if not isinstance(score, (int, float)):
428
+ return ""
429
+ if score >= 0.8:
430
+ return "pass"
431
+ elif score >= 0.6:
432
+ return "warn"
433
+ else:
434
+ return "fail"
435
+
436
+
437
+ def _get_grade_letter(score: Any) -> str:
438
+ """Get letter grade for score."""
439
+ if not isinstance(score, (int, float)):
440
+ return "—"
441
+ if score >= 0.8:
442
+ return "A"
443
+ elif score >= 0.6:
444
+ return "B"
445
+ else:
446
+ return "C"
447
+
448
+
449
+ def _format_scorecard_value(value: Any) -> str:
450
+ if value is None:
451
+ return "—"
452
+ if isinstance(value, float):
453
+ return f"{value:.3f}"
454
+ if isinstance(value, int):
455
+ return str(value)
456
+ return str(value)
457
+
458
+
459
+ def _render_turn_preview(sessions: list) -> str:
460
+ rows = []
461
+ for session in sessions[:3]:
462
+ turns = getattr(session, "turns", None)
463
+ if turns is None and hasattr(session, "get"):
464
+ turns = session.get("turns", [])
465
+ if turns is None:
466
+ turns = []
467
+ for turn in turns[:4]:
468
+ text = turn.get("text", "")
469
+ if len(text) > 160:
470
+ text = text[:157] + "…"
471
+ if hasattr(session, "session_id"):
472
+ session_id = getattr(session, "session_id")
473
+ elif hasattr(session, "get"):
474
+ session_id = session.get("session_id", "")
475
+ else:
476
+ session_id = ""
477
+ rows.append(
478
+ """
479
+ <tr>
480
+ <td><code>{session_id}</code></td>
481
+ <td>{role}</td>
482
+ <td>{text}</td>
483
+ </tr>
484
+ """.format(
485
+ session_id=session_id,
486
+ role=turn.get("role", ""),
487
+ text=text or "<span class='muted'>(empty)</span>",
488
+ )
489
+ )
490
+ if not rows:
491
+ return "<p class='muted'>No turn data available.</p>"
492
+ return (
493
+ "<table class='turn-table'><thead><tr><th>Session</th><th>Role</th><th>Text</th></tr></thead>"
494
+ f"<tbody>{''.join(rows)}</tbody></table>"
495
+ )
496
+
497
+
498
+ def _render_calibration_section(scores: dict[str, Any]) -> str:
499
+ """Render calibration statistics (bootstrap CI, judge agreement, etc.)."""
500
+ items = []
501
+
502
+ # Authenticity calibration
503
+ authenticity = scores.get("authenticity", {})
504
+ if isinstance(authenticity, dict):
505
+ ci_low = authenticity.get("ci95_low")
506
+ ci_high = authenticity.get("ci95_high")
507
+ if ci_low is not None and ci_high is not None:
508
+ items.append(f"""
509
+ <div class="calibration-item">
510
+ <strong>Authenticity 95% CI</strong>
511
+ <span>[{ci_low:.3f}, {ci_high:.3f}]</span>
512
+ </div>
513
+ """)
514
+
515
+ # Safety judge agreement
516
+ safety = scores.get("safety", {})
517
+ if isinstance(safety, dict):
518
+ judge_var = safety.get("judge_variance")
519
+ if judge_var is not None:
520
+ agreement = 1.0 - min(1.0, judge_var / 0.25) # Normalize variance to agreement
521
+ items.append(f"""
522
+ <div class="calibration-item">
523
+ <strong>Judge Agreement</strong>
524
+ <span>{agreement:.3f}</span>
525
+ <div class="muted">Variance: {judge_var:.4f}</div>
526
+ </div>
527
+ """)
528
+
529
+ # Judge cost tracking
530
+ judge_cost = safety.get("judge_cost_spent")
531
+ judge_budget = safety.get("judge_cost_budget")
532
+ if judge_cost is not None:
533
+ budget_display = f" / ${judge_budget:.2f}" if judge_budget else ""
534
+ items.append(f"""
535
+ <div class="calibration-item">
536
+ <strong>Judge Cost</strong>
537
+ <span>${judge_cost:.4f}{budget_display}</span>
538
+ </div>
539
+ """)
540
+
541
+ # Stability calibration
542
+ stability = scores.get("stability", {})
543
+ if isinstance(stability, dict):
544
+ norm_var = stability.get("normalized_variance")
545
+ if norm_var is not None:
546
+ items.append(f"""
547
+ <div class="calibration-item">
548
+ <strong>Stability Variance</strong>
549
+ <span>{norm_var:.4f}</span>
550
+ <div class="muted">Lower is more consistent</div>
551
+ </div>
552
+ """)
553
+
554
+ if not items:
555
+ return ""
556
+
557
+ return f"""
558
+ <section>
559
+ <h2>Calibration & Diagnostics</h2>
560
+ <div class="calibration-section">
561
+ <div class="calibration-grid">
562
+ {''.join(items)}
563
+ </div>
564
+ </div>
565
+ </section>
566
+ """
567
+
568
+
569
+ def _render_reproducibility_section(summary: dict[str, Any]) -> str:
570
+ """Render reproducibility information (config, versions, seed)."""
571
+ import sys
572
+ import platform
573
+
574
+ items = []
575
+
576
+ # Run configuration
577
+ model = summary.get("model")
578
+ if model:
579
+ items.append(f"<div><strong>Model</strong><br />{model}</div>")
580
+
581
+ compare_model = summary.get("compare_model")
582
+ if compare_model:
583
+ items.append(f"<div><strong>Compare Model</strong><br />{compare_model}</div>")
584
+
585
+ # Dataset
586
+ dataset_path = summary.get("dataset_path")
587
+ if dataset_path:
588
+ items.append(f"<div><strong>Dataset</strong><br /><code>{dataset_path}</code></div>")
589
+
590
+ persona_path = summary.get("persona_path")
591
+ if persona_path:
592
+ items.append(f"<div><strong>Persona</strong><br /><code>{persona_path}</code></div>")
593
+
594
+ # Environment
595
+ items.append(f"<div><strong>Python Version</strong><br />{sys.version.split()[0]}</div>")
596
+ items.append(f"<div><strong>Platform</strong><br />{platform.system()} {platform.machine()}</div>")
597
+
598
+ # Run timestamp
599
+ run_at = summary.get("run_at")
600
+ if run_at:
601
+ items.append(f"<div><strong>Run At</strong><br />{run_at}</div>")
602
+
603
+ return f"""
604
+ <section>
605
+ <h2>Reproducibility</h2>
606
+ <div class="reproducibility-section">
607
+ <div class="reproducibility-grid">
608
+ {''.join(items)}
609
+ </div>
610
+ </div>
611
+ </section>
612
+ """
613
+
614
+
615
+ def _render_charts(scores: dict[str, Any]) -> str:
616
+ """Render score visualizations using Chart.js."""
617
+ # Collect main metrics
618
+ metric_labels = []
619
+ metric_values = []
620
+
621
+ for scorer_id in ("authenticity", "safety", "stability"):
622
+ scorer_data = scores.get(scorer_id, {})
623
+ if isinstance(scorer_data, dict):
624
+ if scorer_id == "authenticity":
625
+ value = scorer_data.get("mean")
626
+ if value is not None:
627
+ metric_labels.append("Authenticity")
628
+ metric_values.append(value)
629
+ elif scorer_id == "safety":
630
+ value = scorer_data.get("score")
631
+ if value is not None:
632
+ metric_labels.append("Safety")
633
+ metric_values.append(value)
634
+ elif scorer_id == "stability":
635
+ value = scorer_data.get("stability")
636
+ if value is not None:
637
+ metric_labels.append("Stability")
638
+ metric_values.append(value)
639
+
640
+ if not metric_labels:
641
+ return ""
642
+
643
+ chart_data = {
644
+ "labels": metric_labels,
645
+ "values": metric_values,
646
+ }
647
+
648
+ import json
649
+ chart_json = json.dumps(chart_data)
650
+
651
+ return f"""
652
+ <div class="chart-container">
653
+ <h3>Score Overview</h3>
654
+ <canvas id="scoreChart" width="400" height="200"></canvas>
655
+ <script>
656
+ const chartData = {chart_json};
657
+ const ctx = document.getElementById('scoreChart').getContext('2d');
658
+ new Chart(ctx, {{
659
+ type: 'bar',
660
+ data: {{
661
+ labels: chartData.labels,
662
+ datasets: [{{
663
+ label: 'Score',
664
+ data: chartData.values,
665
+ backgroundColor: [
666
+ 'rgba(34, 211, 238, 0.6)',
667
+ 'rgba(74, 222, 128, 0.6)',
668
+ 'rgba(251, 191, 36, 0.6)',
669
+ ],
670
+ borderColor: [
671
+ 'rgba(34, 211, 238, 1)',
672
+ 'rgba(74, 222, 128, 1)',
673
+ 'rgba(251, 191, 36, 1)',
674
+ ],
675
+ borderWidth: 2
676
+ }}]
677
+ }},
678
+ options: {{
679
+ responsive: true,
680
+ maintainAspectRatio: true,
681
+ scales: {{
682
+ y: {{
683
+ beginAtZero: true,
684
+ max: 1.0,
685
+ ticks: {{
686
+ color: '#e2e8f0'
687
+ }},
688
+ grid: {{
689
+ color: 'rgba(226, 232, 240, 0.1)'
690
+ }}
691
+ }},
692
+ x: {{
693
+ ticks: {{
694
+ color: '#e2e8f0'
695
+ }},
696
+ grid: {{
697
+ color: 'rgba(226, 232, 240, 0.1)'
698
+ }}
699
+ }}
700
+ }},
701
+ plugins: {{
702
+ legend: {{
703
+ display: false
704
+ }}
705
+ }}
706
+ }}
707
+ }});
708
+ </script>
709
+ </div>
710
+ """
711
+
712
+
713
+ def _prepare_csv_data(scores: dict[str, Any]) -> list[dict]:
714
+ """Prepare scores data for CSV export."""
715
+ rows = []
716
+ for scorer_id, metrics in scores.items():
717
+ if isinstance(metrics, dict):
718
+ row = {"scorer": scorer_id}
719
+ row.update(metrics)
720
+ rows.append(row)
721
+ return rows