alignmenter 0.0.4__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- alignmenter/__init__.py +14 -0
- alignmenter/cli.py +1815 -0
- alignmenter/config.py +99 -0
- alignmenter/data/configs/demo_config.yaml +15 -0
- alignmenter/data/configs/judges/safety_prompt.txt +2 -0
- alignmenter/data/configs/persona/default.yaml +15 -0
- alignmenter/data/configs/run.yaml +12 -0
- alignmenter/data/configs/safety_keywords.yaml +7 -0
- alignmenter/data/datasets/demo_conversations.jsonl +60 -0
- alignmenter/providers/__init__.py +47 -0
- alignmenter/providers/anthropic.py +87 -0
- alignmenter/providers/base.py +57 -0
- alignmenter/providers/classifiers.py +83 -0
- alignmenter/providers/embeddings.py +126 -0
- alignmenter/providers/judges.py +105 -0
- alignmenter/providers/local.py +102 -0
- alignmenter/providers/openai.py +151 -0
- alignmenter/reporting/__init__.py +6 -0
- alignmenter/reporting/html.py +721 -0
- alignmenter/reporting/json_out.py +33 -0
- alignmenter/run_config.py +106 -0
- alignmenter/runner.py +410 -0
- alignmenter/scorers/__init__.py +7 -0
- alignmenter/scorers/authenticity.py +337 -0
- alignmenter/scorers/safety.py +231 -0
- alignmenter/scorers/stability.py +104 -0
- alignmenter/scripts/__init__.py +1 -0
- alignmenter/scripts/bootstrap_dataset.py +142 -0
- alignmenter/scripts/calibrate_persona.py +196 -0
- alignmenter/scripts/run_openai_demo.py +74 -0
- alignmenter/scripts/sanitize_dataset.py +185 -0
- alignmenter/utils/__init__.py +7 -0
- alignmenter/utils/io.py +47 -0
- alignmenter/utils/tokens.py +46 -0
- alignmenter/utils/yaml.py +15 -0
- alignmenter-0.0.4.dist-info/METADATA +681 -0
- alignmenter-0.0.4.dist-info/RECORD +41 -0
- alignmenter-0.0.4.dist-info/WHEEL +5 -0
- alignmenter-0.0.4.dist-info/entry_points.txt +2 -0
- alignmenter-0.0.4.dist-info/licenses/LICENSE +201 -0
- alignmenter-0.0.4.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,721 @@
|
|
|
1
|
+
"""HTML report generator."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from pathlib import Path
|
|
6
|
+
from typing import Any
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
HTML_TEMPLATE = """<!DOCTYPE html>
|
|
10
|
+
<html lang=\"en\">
|
|
11
|
+
<head>
|
|
12
|
+
<meta charset=\"utf-8\" />
|
|
13
|
+
<title>Alignmenter Report - {run_id}</title>
|
|
14
|
+
<link rel=\"icon\" type=\"image/png\" href=\"favicon.png\">
|
|
15
|
+
<style>
|
|
16
|
+
body {{ font-family: -apple-system, BlinkMacSystemFont, Segoe UI, sans-serif; background: #0f172a; color: #e2e8f0; margin: 0; padding: 32px; }}
|
|
17
|
+
h1, h2 {{ color: #22d3ee; }}
|
|
18
|
+
section {{ margin-bottom: 32px; }}
|
|
19
|
+
table {{ width: 100%; border-collapse: collapse; margin-top: 16px; }}
|
|
20
|
+
th, td {{ padding: 12px; text-align: left; border-bottom: 1px solid #1e293b; }}
|
|
21
|
+
th {{ background: #1e293b; }}
|
|
22
|
+
.meta {{ display: grid; grid-template-columns: repeat(auto-fit, minmax(220px, 1fr)); gap: 12px; }}
|
|
23
|
+
.meta div {{ background: #1e293b; padding: 12px; border-radius: 8px; }}
|
|
24
|
+
.report-card {{ background: linear-gradient(135deg, #1e293b 0%, #0f172a 100%); border-radius: 16px; padding: 0; box-shadow: 0 20px 60px rgba(14,74,104,0.4); border: 1px solid #334155; max-width: 1200px; margin: 0 auto; }}
|
|
25
|
+
.report-card-header {{ background: linear-gradient(135deg, #1e293b 0%, #0f172a 100%); padding: 32px 40px; border-radius: 16px 16px 0 0; border-bottom: 2px solid #22d3ee; }}
|
|
26
|
+
.report-card-header-content {{ display: flex; align-items: center; gap: 20px; margin-bottom: 20px; }}
|
|
27
|
+
.report-logo {{ height: 48px; width: auto; }}
|
|
28
|
+
.report-card-title {{ margin: 0; font-size: 2rem; font-weight: 800; letter-spacing: -0.02em; color: #22d3ee; }}
|
|
29
|
+
.report-info-grid {{ display: grid; grid-template-columns: repeat(auto-fit, minmax(200px, 1fr)); gap: 16px; margin-top: 20px; }}
|
|
30
|
+
.report-info-item {{ background: rgba(34, 211, 238, 0.05); padding: 12px 16px; border-radius: 8px; border: 1px solid rgba(34, 211, 238, 0.1); }}
|
|
31
|
+
.report-info-label {{ font-size: 0.75rem; text-transform: uppercase; letter-spacing: 0.05em; color: #94a3b8; margin-bottom: 4px; font-weight: 600; }}
|
|
32
|
+
.report-info-value {{ font-size: 1rem; font-weight: 700; color: #e2e8f0; }}
|
|
33
|
+
.report-card-body {{ padding: 40px; }}
|
|
34
|
+
.overall-grade {{ text-align: center; margin-bottom: 32px; padding: 24px; background: rgba(34, 211, 238, 0.05); border-radius: 12px; border: 2px solid rgba(34, 211, 238, 0.2); }}
|
|
35
|
+
.overall-grade-label {{ font-size: 0.9rem; color: #94a3b8; text-transform: uppercase; letter-spacing: 0.1em; margin-bottom: 8px; }}
|
|
36
|
+
.overall-grade-value {{ font-size: 4rem; font-weight: 800; line-height: 1; }}
|
|
37
|
+
.overall-grade-value.pass {{ color: #4ade80; }}
|
|
38
|
+
.overall-grade-value.warn {{ color: #fbbf24; }}
|
|
39
|
+
.overall-grade-value.fail {{ color: #f87171; }}
|
|
40
|
+
.overall-grade-desc {{ font-size: 0.9rem; color: #94a3b8; margin-top: 8px; }}
|
|
41
|
+
.grade-table {{ width: 100%; border-collapse: separate; border-spacing: 0 8px; }}
|
|
42
|
+
.grade-table thead th {{ padding: 12px 16px; text-align: left; font-size: 0.85rem; text-transform: uppercase; letter-spacing: 0.05em; color: #94a3b8; font-weight: 600; border-bottom: 2px solid #334155; }}
|
|
43
|
+
.grade-table thead th:nth-child(2), .grade-table thead th:nth-child(3), .grade-table thead th:nth-child(4) {{ text-align: center; }}
|
|
44
|
+
.grade-table tbody tr {{ background: #1e293b; }}
|
|
45
|
+
.grade-table tbody td {{ padding: 20px 16px; border-top: 1px solid #334155; border-bottom: 1px solid #334155; }}
|
|
46
|
+
.grade-table tbody td:first-child {{ border-left: 1px solid #334155; border-radius: 8px 0 0 8px; }}
|
|
47
|
+
.grade-table tbody td:last-child {{ border-right: 1px solid #334155; border-radius: 0 8px 8px 0; }}
|
|
48
|
+
.metric-name {{ font-weight: 600; color: #e2e8f0; font-size: 1.1rem; }}
|
|
49
|
+
.grade-cell {{ text-align: center; }}
|
|
50
|
+
.grade-badge {{ display: inline-flex; align-items: center; gap: 8px; padding: 8px 16px; border-radius: 8px; font-weight: 700; font-size: 1.1rem; }}
|
|
51
|
+
.grade-badge.pass {{ background: rgba(74, 222, 128, 0.15); color: #4ade80; border: 2px solid rgba(74, 222, 128, 0.3); }}
|
|
52
|
+
.grade-badge.warn {{ background: rgba(251, 191, 36, 0.15); color: #fbbf24; border: 2px solid rgba(251, 191, 36, 0.3); }}
|
|
53
|
+
.grade-badge.fail {{ background: rgba(248, 113, 113, 0.15); color: #f87171; border: 2px solid rgba(248, 113, 113, 0.3); }}
|
|
54
|
+
.grade-letter {{ font-size: 1.3rem; font-weight: 800; }}
|
|
55
|
+
.grade-score {{ font-size: 0.95rem; opacity: 0.9; }}
|
|
56
|
+
.compare-cell {{ text-align: center; color: #94a3b8; font-size: 0.95rem; }}
|
|
57
|
+
.delta-cell {{ text-align: center; font-size: 1rem; font-weight: 700; }}
|
|
58
|
+
.delta-cell.positive {{ color: #4ade80; }}
|
|
59
|
+
.delta-cell.negative {{ color: #f87171; }}
|
|
60
|
+
.delta-cell.neutral {{ color: #64748b; }}
|
|
61
|
+
.turn-table td {{ vertical-align: top; }}
|
|
62
|
+
.muted {{ color: #94a3b8; font-size: 0.85rem; }}
|
|
63
|
+
code {{ background: rgba(15,23,42,0.6); padding: 2px 6px; border-radius: 4px; font-size: 0.85rem; }}
|
|
64
|
+
.score-pass {{ background: rgba(74, 222, 128, 0.1); color: #4ade80; font-weight: 600; }}
|
|
65
|
+
.score-warn {{ background: rgba(251, 191, 36, 0.1); color: #fbbf24; font-weight: 600; }}
|
|
66
|
+
.score-fail {{ background: rgba(248, 113, 113, 0.1); color: #f87171; font-weight: 600; }}
|
|
67
|
+
.calibration-section {{ background: #1e293b; padding: 16px; border-radius: 8px; margin-top: 16px; }}
|
|
68
|
+
.calibration-grid {{ display: grid; grid-template-columns: repeat(auto-fit, minmax(250px, 1fr)); gap: 12px; margin-top: 12px; }}
|
|
69
|
+
.calibration-item {{ background: #0f172a; padding: 12px; border-radius: 6px; }}
|
|
70
|
+
.calibration-item strong {{ color: #22d3ee; display: block; margin-bottom: 4px; }}
|
|
71
|
+
.reproducibility-section {{ background: #1e293b; padding: 16px; border-radius: 8px; }}
|
|
72
|
+
.reproducibility-grid {{ display: grid; grid-template-columns: repeat(auto-fit, minmax(400px, 1fr)); gap: 12px; margin-top: 12px; }}
|
|
73
|
+
.reproducibility-grid > div {{ overflow-wrap: break-word; word-break: break-all; }}
|
|
74
|
+
.export-buttons {{ margin-top: 8px; }}
|
|
75
|
+
.export-btn {{ background: #1e293b; color: #22d3ee; border: 1px solid #22d3ee; padding: 6px 12px; border-radius: 6px; cursor: pointer; text-decoration: none; display: inline-block; margin-right: 8px; font-size: 0.85rem; }}
|
|
76
|
+
.export-btn:hover {{ background: #22d3ee; color: #0f172a; }}
|
|
77
|
+
.chart-container {{ margin-top: 16px; background: #1e293b; padding: 16px; border-radius: 8px; }}
|
|
78
|
+
canvas {{ max-width: 100%; }}
|
|
79
|
+
</style>
|
|
80
|
+
<script src="https://cdn.jsdelivr.net/npm/chart.js@4.4.0/dist/chart.umd.min.js"></script>
|
|
81
|
+
<script>
|
|
82
|
+
function downloadJSON(data, filename) {{
|
|
83
|
+
const blob = new Blob([JSON.stringify(data, null, 2)], {{ type: 'application/json' }});
|
|
84
|
+
const url = URL.createObjectURL(blob);
|
|
85
|
+
const a = document.createElement('a');
|
|
86
|
+
a.href = url;
|
|
87
|
+
a.download = filename;
|
|
88
|
+
a.click();
|
|
89
|
+
URL.revokeObjectURL(url);
|
|
90
|
+
}}
|
|
91
|
+
|
|
92
|
+
function downloadCSV(data, filename) {{
|
|
93
|
+
const rows = [];
|
|
94
|
+
if (data.length > 0) {{
|
|
95
|
+
rows.push(Object.keys(data[0]).join(','));
|
|
96
|
+
data.forEach(row => {{
|
|
97
|
+
rows.push(Object.values(row).join(','));
|
|
98
|
+
}});
|
|
99
|
+
}}
|
|
100
|
+
const csv = rows.join('\\n');
|
|
101
|
+
const blob = new Blob([csv], {{ type: 'text/csv' }});
|
|
102
|
+
const url = URL.createObjectURL(blob);
|
|
103
|
+
const a = document.createElement('a');
|
|
104
|
+
a.href = url;
|
|
105
|
+
a.download = filename;
|
|
106
|
+
a.click();
|
|
107
|
+
URL.revokeObjectURL(url);
|
|
108
|
+
}}
|
|
109
|
+
</script>
|
|
110
|
+
</head>
|
|
111
|
+
<body>
|
|
112
|
+
{scorecard_block}
|
|
113
|
+
{calibration_section}
|
|
114
|
+
<section>
|
|
115
|
+
<h2>Scores</h2>
|
|
116
|
+
<div class="export-buttons">
|
|
117
|
+
<button class="export-btn" onclick="downloadJSON(window.scoresData, 'scores.json')">Download JSON</button>
|
|
118
|
+
<button class="export-btn" onclick="downloadCSV(window.scoresDataCSV, 'scores.csv')">Download CSV</button>
|
|
119
|
+
</div>
|
|
120
|
+
{score_tables}
|
|
121
|
+
{charts_section}
|
|
122
|
+
</section>
|
|
123
|
+
{reproducibility_section}
|
|
124
|
+
<section>
|
|
125
|
+
<h2>Turn-Level Explorer</h2>
|
|
126
|
+
{turn_preview}
|
|
127
|
+
</section>
|
|
128
|
+
<script>
|
|
129
|
+
window.scoresData = {scores_json};
|
|
130
|
+
window.scoresDataCSV = {scores_csv_json};
|
|
131
|
+
</script>
|
|
132
|
+
</body>
|
|
133
|
+
</html>
|
|
134
|
+
"""
|
|
135
|
+
|
|
136
|
+
|
|
137
|
+
class HTMLReporter:
|
|
138
|
+
"""Generate a minimal HTML report."""
|
|
139
|
+
|
|
140
|
+
def write(
|
|
141
|
+
self,
|
|
142
|
+
run_dir: Path,
|
|
143
|
+
summary: dict[str, Any],
|
|
144
|
+
scores: dict[str, Any],
|
|
145
|
+
sessions: list,
|
|
146
|
+
**extras: Any,
|
|
147
|
+
) -> Path:
|
|
148
|
+
scorecards = extras.get("scorecards", [])
|
|
149
|
+
|
|
150
|
+
primary = scores.get("primary", {}) if isinstance(scores, dict) else {}
|
|
151
|
+
compare = scores.get("compare", {}) if isinstance(scores, dict) else {}
|
|
152
|
+
diff = scores.get("diff", {}) if isinstance(scores, dict) else {}
|
|
153
|
+
|
|
154
|
+
score_blocks = []
|
|
155
|
+
scorer_ids = sorted({*primary.keys(), *compare.keys()}) or list(scores.keys())
|
|
156
|
+
|
|
157
|
+
for scorer_id in scorer_ids:
|
|
158
|
+
primary_metrics = primary.get(scorer_id, {}) if isinstance(primary, dict) else {}
|
|
159
|
+
compare_metrics = compare.get(scorer_id, {}) if isinstance(compare, dict) else {}
|
|
160
|
+
diff_metrics = diff.get(scorer_id, {}) if isinstance(diff, dict) else {}
|
|
161
|
+
|
|
162
|
+
metric_keys = sorted({*primary_metrics.keys(), *compare_metrics.keys(), *diff_metrics.keys()}) or ["value"]
|
|
163
|
+
|
|
164
|
+
if not isinstance(primary_metrics, dict) and scorer_id in primary:
|
|
165
|
+
primary_metrics = {"value": primary[scorer_id]}
|
|
166
|
+
if not isinstance(compare_metrics, dict) and scorer_id in compare:
|
|
167
|
+
compare_metrics = {"value": compare[scorer_id]}
|
|
168
|
+
if not isinstance(diff_metrics, dict) and scorer_id in diff:
|
|
169
|
+
diff_metrics = {"value": diff[scorer_id]}
|
|
170
|
+
|
|
171
|
+
has_compare = bool(compare_metrics)
|
|
172
|
+
header = "<th>Metric</th><th>Primary</th>"
|
|
173
|
+
if has_compare:
|
|
174
|
+
header += "<th>Compare</th><th>Δ</th>"
|
|
175
|
+
|
|
176
|
+
row_html = []
|
|
177
|
+
for key in metric_keys:
|
|
178
|
+
primary_val = _format_metric(primary_metrics.get(key), metric_key=key)
|
|
179
|
+
compare_val = _format_metric(compare_metrics.get(key), metric_key=key) if has_compare else ""
|
|
180
|
+
delta_val = _format_metric(diff_metrics.get(key), metric_key=key, apply_color=False) if has_compare else ""
|
|
181
|
+
if has_compare:
|
|
182
|
+
row_html.append(
|
|
183
|
+
f"<tr><td>{key}</td><td>{primary_val}</td><td>{compare_val}</td><td>{delta_val}</td></tr>"
|
|
184
|
+
)
|
|
185
|
+
else:
|
|
186
|
+
row_html.append(f"<tr><td>{key}</td><td>{primary_val}</td></tr>")
|
|
187
|
+
|
|
188
|
+
table = (
|
|
189
|
+
f"<h3>{scorer_id.title()}</h3>"
|
|
190
|
+
f"<table><thead><tr>{header}</tr></thead><tbody>{''.join(row_html)}</tbody></table>"
|
|
191
|
+
)
|
|
192
|
+
if scorer_id == "safety":
|
|
193
|
+
safety_details = _render_judge_details(primary_metrics)
|
|
194
|
+
if safety_details:
|
|
195
|
+
table += safety_details
|
|
196
|
+
score_blocks.append(table)
|
|
197
|
+
|
|
198
|
+
turn_preview = _render_turn_preview(sessions)
|
|
199
|
+
calibration_section = _render_calibration_section(primary)
|
|
200
|
+
reproducibility_section = _render_reproducibility_section(summary)
|
|
201
|
+
charts_section = _render_charts(primary)
|
|
202
|
+
scorecard_block = _render_scorecards(scorecards, summary)
|
|
203
|
+
|
|
204
|
+
# Prepare data for export
|
|
205
|
+
import json
|
|
206
|
+
scores_json = json.dumps(scores)
|
|
207
|
+
scores_csv_data = _prepare_csv_data(primary)
|
|
208
|
+
scores_csv_json = json.dumps(scores_csv_data)
|
|
209
|
+
|
|
210
|
+
html = HTML_TEMPLATE.format(
|
|
211
|
+
run_id=summary.get("run_id", "alignmenter_run"),
|
|
212
|
+
scorecard_block=scorecard_block,
|
|
213
|
+
score_tables="".join(score_blocks) or "<p>No scores computed.</p>",
|
|
214
|
+
turn_preview=turn_preview,
|
|
215
|
+
calibration_section=calibration_section,
|
|
216
|
+
reproducibility_section=reproducibility_section,
|
|
217
|
+
charts_section=charts_section,
|
|
218
|
+
scores_json=scores_json,
|
|
219
|
+
scores_csv_json=scores_csv_json,
|
|
220
|
+
)
|
|
221
|
+
|
|
222
|
+
path = Path(run_dir) / "index.html"
|
|
223
|
+
path.write_text(html, encoding="utf-8")
|
|
224
|
+
|
|
225
|
+
# Copy logo and favicon to report directory
|
|
226
|
+
import shutil
|
|
227
|
+
assets_dir = Path(__file__).parent.parent.parent.parent.parent / "assets"
|
|
228
|
+
|
|
229
|
+
# Copy logo
|
|
230
|
+
logo_source = assets_dir / "alignmenter-transparent.png"
|
|
231
|
+
if logo_source.exists():
|
|
232
|
+
logo_dest = Path(run_dir) / "logo.png"
|
|
233
|
+
shutil.copy2(logo_source, logo_dest)
|
|
234
|
+
|
|
235
|
+
# Copy favicon (use the transparent icon version)
|
|
236
|
+
favicon_source = assets_dir / "alignmenter-transparent.png"
|
|
237
|
+
if favicon_source.exists():
|
|
238
|
+
favicon_dest = Path(run_dir) / "favicon.png"
|
|
239
|
+
shutil.copy2(favicon_source, favicon_dest)
|
|
240
|
+
|
|
241
|
+
return path
|
|
242
|
+
|
|
243
|
+
|
|
244
|
+
def _format_metric(value: Any, metric_key: str = "", apply_color: bool = True) -> str:
|
|
245
|
+
if value is None:
|
|
246
|
+
return "—"
|
|
247
|
+
|
|
248
|
+
formatted = ""
|
|
249
|
+
if isinstance(value, float):
|
|
250
|
+
formatted = f"{value:.3f}"
|
|
251
|
+
elif isinstance(value, list):
|
|
252
|
+
formatted = ", ".join(str(item) for item in value)
|
|
253
|
+
else:
|
|
254
|
+
formatted = str(value)
|
|
255
|
+
|
|
256
|
+
# Apply color coding for score metrics
|
|
257
|
+
if apply_color and isinstance(value, (int, float)) and metric_key in ("mean", "score", "stability", "rule_score", "fused_judge"):
|
|
258
|
+
css_class = _get_score_class(value)
|
|
259
|
+
return f'<span class="{css_class}">{formatted}</span>'
|
|
260
|
+
|
|
261
|
+
return formatted
|
|
262
|
+
|
|
263
|
+
|
|
264
|
+
def _get_score_class(score: float) -> str:
|
|
265
|
+
"""Get CSS class based on score threshold."""
|
|
266
|
+
if score >= 0.8:
|
|
267
|
+
return "score-pass"
|
|
268
|
+
elif score >= 0.6:
|
|
269
|
+
return "score-warn"
|
|
270
|
+
else:
|
|
271
|
+
return "score-fail"
|
|
272
|
+
|
|
273
|
+
|
|
274
|
+
def _render_judge_details(metrics: dict[str, Any]) -> str:
|
|
275
|
+
if not isinstance(metrics, dict):
|
|
276
|
+
return ""
|
|
277
|
+
calls = metrics.get("judge_calls")
|
|
278
|
+
budget = metrics.get("judge_budget")
|
|
279
|
+
mean_score = metrics.get("judge_mean")
|
|
280
|
+
notes = metrics.get("judge_notes") or []
|
|
281
|
+
|
|
282
|
+
if calls is None and not notes and mean_score is None:
|
|
283
|
+
return ""
|
|
284
|
+
|
|
285
|
+
lines = []
|
|
286
|
+
if calls is not None:
|
|
287
|
+
info = f"Judge calls: {calls}"
|
|
288
|
+
if budget:
|
|
289
|
+
info += f" / budget {budget}"
|
|
290
|
+
lines.append(info)
|
|
291
|
+
if mean_score is not None:
|
|
292
|
+
lines.append(f"Average judge score: {mean_score:.3f}")
|
|
293
|
+
if notes:
|
|
294
|
+
notes_html = "".join(f"<li>{note}</li>" for note in notes)
|
|
295
|
+
lines.append(f"<ul>{notes_html}</ul>")
|
|
296
|
+
|
|
297
|
+
body = "<br />".join(item for item in lines if not item.startswith("<ul>"))
|
|
298
|
+
list_html = "".join(item for item in lines if item.startswith("<ul>"))
|
|
299
|
+
return f"<div class='muted'>{body}{list_html}</div>"
|
|
300
|
+
|
|
301
|
+
|
|
302
|
+
def _render_scorecards(scorecards: list[dict], summary: dict[str, Any]) -> str:
|
|
303
|
+
if not scorecards:
|
|
304
|
+
return ""
|
|
305
|
+
|
|
306
|
+
has_compare = any(card.get("compare") is not None for card in scorecards)
|
|
307
|
+
|
|
308
|
+
# Calculate overall grade
|
|
309
|
+
scores = [card.get("primary") for card in scorecards if isinstance(card.get("primary"), (int, float))]
|
|
310
|
+
overall_score = sum(scores) / len(scores) if scores else 0
|
|
311
|
+
overall_class = _get_grade_class(overall_score)
|
|
312
|
+
overall_letter = _get_grade_letter(overall_score)
|
|
313
|
+
|
|
314
|
+
# Build run info
|
|
315
|
+
model = summary.get("model", "Unknown")
|
|
316
|
+
run_at = summary.get("run_at", "Unknown")
|
|
317
|
+
session_count = summary.get("session_count", 0)
|
|
318
|
+
turn_count = summary.get("turn_count", 0)
|
|
319
|
+
run_id = summary.get("run_id", "alignmenter_run")
|
|
320
|
+
|
|
321
|
+
run_info = f"""
|
|
322
|
+
<div class="report-info-grid">
|
|
323
|
+
<div class="report-info-item">
|
|
324
|
+
<div class="report-info-label">Model</div>
|
|
325
|
+
<div class="report-info-value">{model}</div>
|
|
326
|
+
</div>
|
|
327
|
+
<div class="report-info-item">
|
|
328
|
+
<div class="report-info-label">Run ID</div>
|
|
329
|
+
<div class="report-info-value">{run_id}</div>
|
|
330
|
+
</div>
|
|
331
|
+
<div class="report-info-item">
|
|
332
|
+
<div class="report-info-label">Timestamp</div>
|
|
333
|
+
<div class="report-info-value">{run_at}</div>
|
|
334
|
+
</div>
|
|
335
|
+
<div class="report-info-item">
|
|
336
|
+
<div class="report-info-label">Dataset</div>
|
|
337
|
+
<div class="report-info-value">{session_count} sessions · {turn_count} turns</div>
|
|
338
|
+
</div>
|
|
339
|
+
</div>
|
|
340
|
+
"""
|
|
341
|
+
|
|
342
|
+
# Build overall grade section
|
|
343
|
+
grade_desc = {
|
|
344
|
+
"A": "Excellent performance across all metrics",
|
|
345
|
+
"B": "Good performance with room for improvement",
|
|
346
|
+
"C": "Needs attention in one or more areas"
|
|
347
|
+
}.get(overall_letter, "")
|
|
348
|
+
|
|
349
|
+
overall_grade_html = f"""
|
|
350
|
+
<div class="overall-grade">
|
|
351
|
+
<div class="overall-grade-label">Overall Grade</div>
|
|
352
|
+
<div class="overall-grade-value {overall_class}">{overall_letter}</div>
|
|
353
|
+
<div class="overall-grade-desc">{grade_desc}</div>
|
|
354
|
+
</div>
|
|
355
|
+
"""
|
|
356
|
+
|
|
357
|
+
# Build table
|
|
358
|
+
table_header = '<thead><tr>'
|
|
359
|
+
table_header += '<th>Metric</th>'
|
|
360
|
+
table_header += '<th>Grade</th>'
|
|
361
|
+
if has_compare:
|
|
362
|
+
table_header += '<th>Compare</th>'
|
|
363
|
+
table_header += '<th>Change</th>'
|
|
364
|
+
table_header += '</tr></thead>'
|
|
365
|
+
|
|
366
|
+
# Build rows
|
|
367
|
+
rows = []
|
|
368
|
+
for card in scorecards:
|
|
369
|
+
primary_val = card.get("primary")
|
|
370
|
+
compare_val = card.get("compare")
|
|
371
|
+
diff_val = card.get("diff")
|
|
372
|
+
|
|
373
|
+
# Determine grade class
|
|
374
|
+
grade_class = _get_grade_class(primary_val)
|
|
375
|
+
letter = _get_grade_letter(primary_val)
|
|
376
|
+
|
|
377
|
+
row = '<tr>'
|
|
378
|
+
row += f'<td><span class="metric-name">{card.get("label", card.get("id", "Metric").title())}</span></td>'
|
|
379
|
+
row += f'<td class="grade-cell"><div class="grade-badge {grade_class}">'
|
|
380
|
+
row += f'<span class="grade-letter">{letter}</span>'
|
|
381
|
+
row += f'<span class="grade-score">{_format_scorecard_value(primary_val)}</span>'
|
|
382
|
+
row += '</div></td>'
|
|
383
|
+
|
|
384
|
+
if has_compare:
|
|
385
|
+
if compare_val is not None:
|
|
386
|
+
compare_letter = _get_grade_letter(compare_val)
|
|
387
|
+
row += f'<td class="compare-cell">{compare_letter} {_format_scorecard_value(compare_val)}</td>'
|
|
388
|
+
else:
|
|
389
|
+
row += '<td class="compare-cell">—</td>'
|
|
390
|
+
|
|
391
|
+
if diff_val is not None and isinstance(diff_val, (int, float)):
|
|
392
|
+
delta_class = "positive" if diff_val >= 0 else "negative"
|
|
393
|
+
delta_sign = "+" if diff_val >= 0 else ""
|
|
394
|
+
row += f'<td class="delta-cell {delta_class}">{delta_sign}{_format_scorecard_value(diff_val)}</td>'
|
|
395
|
+
else:
|
|
396
|
+
row += '<td class="delta-cell neutral">—</td>'
|
|
397
|
+
|
|
398
|
+
row += '</tr>'
|
|
399
|
+
rows.append(row)
|
|
400
|
+
|
|
401
|
+
return f"""
|
|
402
|
+
<section>
|
|
403
|
+
<div class="report-card">
|
|
404
|
+
<div class="report-card-header">
|
|
405
|
+
<div class="report-card-header-content">
|
|
406
|
+
<img src="logo.png" alt="Alignmenter" class="report-logo" onerror="this.style.display='none'">
|
|
407
|
+
<h1 class="report-card-title">Alignmenter Report Card</h1>
|
|
408
|
+
</div>
|
|
409
|
+
{run_info}
|
|
410
|
+
</div>
|
|
411
|
+
<div class="report-card-body">
|
|
412
|
+
{overall_grade_html}
|
|
413
|
+
<table class="grade-table">
|
|
414
|
+
{table_header}
|
|
415
|
+
<tbody>
|
|
416
|
+
{''.join(rows)}
|
|
417
|
+
</tbody>
|
|
418
|
+
</table>
|
|
419
|
+
</div>
|
|
420
|
+
</div>
|
|
421
|
+
</section>
|
|
422
|
+
"""
|
|
423
|
+
|
|
424
|
+
|
|
425
|
+
def _get_grade_class(score: Any) -> str:
|
|
426
|
+
"""Get CSS class for grade styling."""
|
|
427
|
+
if not isinstance(score, (int, float)):
|
|
428
|
+
return ""
|
|
429
|
+
if score >= 0.8:
|
|
430
|
+
return "pass"
|
|
431
|
+
elif score >= 0.6:
|
|
432
|
+
return "warn"
|
|
433
|
+
else:
|
|
434
|
+
return "fail"
|
|
435
|
+
|
|
436
|
+
|
|
437
|
+
def _get_grade_letter(score: Any) -> str:
|
|
438
|
+
"""Get letter grade for score."""
|
|
439
|
+
if not isinstance(score, (int, float)):
|
|
440
|
+
return "—"
|
|
441
|
+
if score >= 0.8:
|
|
442
|
+
return "A"
|
|
443
|
+
elif score >= 0.6:
|
|
444
|
+
return "B"
|
|
445
|
+
else:
|
|
446
|
+
return "C"
|
|
447
|
+
|
|
448
|
+
|
|
449
|
+
def _format_scorecard_value(value: Any) -> str:
|
|
450
|
+
if value is None:
|
|
451
|
+
return "—"
|
|
452
|
+
if isinstance(value, float):
|
|
453
|
+
return f"{value:.3f}"
|
|
454
|
+
if isinstance(value, int):
|
|
455
|
+
return str(value)
|
|
456
|
+
return str(value)
|
|
457
|
+
|
|
458
|
+
|
|
459
|
+
def _render_turn_preview(sessions: list) -> str:
|
|
460
|
+
rows = []
|
|
461
|
+
for session in sessions[:3]:
|
|
462
|
+
turns = getattr(session, "turns", None)
|
|
463
|
+
if turns is None and hasattr(session, "get"):
|
|
464
|
+
turns = session.get("turns", [])
|
|
465
|
+
if turns is None:
|
|
466
|
+
turns = []
|
|
467
|
+
for turn in turns[:4]:
|
|
468
|
+
text = turn.get("text", "")
|
|
469
|
+
if len(text) > 160:
|
|
470
|
+
text = text[:157] + "…"
|
|
471
|
+
if hasattr(session, "session_id"):
|
|
472
|
+
session_id = getattr(session, "session_id")
|
|
473
|
+
elif hasattr(session, "get"):
|
|
474
|
+
session_id = session.get("session_id", "")
|
|
475
|
+
else:
|
|
476
|
+
session_id = ""
|
|
477
|
+
rows.append(
|
|
478
|
+
"""
|
|
479
|
+
<tr>
|
|
480
|
+
<td><code>{session_id}</code></td>
|
|
481
|
+
<td>{role}</td>
|
|
482
|
+
<td>{text}</td>
|
|
483
|
+
</tr>
|
|
484
|
+
""".format(
|
|
485
|
+
session_id=session_id,
|
|
486
|
+
role=turn.get("role", ""),
|
|
487
|
+
text=text or "<span class='muted'>(empty)</span>",
|
|
488
|
+
)
|
|
489
|
+
)
|
|
490
|
+
if not rows:
|
|
491
|
+
return "<p class='muted'>No turn data available.</p>"
|
|
492
|
+
return (
|
|
493
|
+
"<table class='turn-table'><thead><tr><th>Session</th><th>Role</th><th>Text</th></tr></thead>"
|
|
494
|
+
f"<tbody>{''.join(rows)}</tbody></table>"
|
|
495
|
+
)
|
|
496
|
+
|
|
497
|
+
|
|
498
|
+
def _render_calibration_section(scores: dict[str, Any]) -> str:
|
|
499
|
+
"""Render calibration statistics (bootstrap CI, judge agreement, etc.)."""
|
|
500
|
+
items = []
|
|
501
|
+
|
|
502
|
+
# Authenticity calibration
|
|
503
|
+
authenticity = scores.get("authenticity", {})
|
|
504
|
+
if isinstance(authenticity, dict):
|
|
505
|
+
ci_low = authenticity.get("ci95_low")
|
|
506
|
+
ci_high = authenticity.get("ci95_high")
|
|
507
|
+
if ci_low is not None and ci_high is not None:
|
|
508
|
+
items.append(f"""
|
|
509
|
+
<div class="calibration-item">
|
|
510
|
+
<strong>Authenticity 95% CI</strong>
|
|
511
|
+
<span>[{ci_low:.3f}, {ci_high:.3f}]</span>
|
|
512
|
+
</div>
|
|
513
|
+
""")
|
|
514
|
+
|
|
515
|
+
# Safety judge agreement
|
|
516
|
+
safety = scores.get("safety", {})
|
|
517
|
+
if isinstance(safety, dict):
|
|
518
|
+
judge_var = safety.get("judge_variance")
|
|
519
|
+
if judge_var is not None:
|
|
520
|
+
agreement = 1.0 - min(1.0, judge_var / 0.25) # Normalize variance to agreement
|
|
521
|
+
items.append(f"""
|
|
522
|
+
<div class="calibration-item">
|
|
523
|
+
<strong>Judge Agreement</strong>
|
|
524
|
+
<span>{agreement:.3f}</span>
|
|
525
|
+
<div class="muted">Variance: {judge_var:.4f}</div>
|
|
526
|
+
</div>
|
|
527
|
+
""")
|
|
528
|
+
|
|
529
|
+
# Judge cost tracking
|
|
530
|
+
judge_cost = safety.get("judge_cost_spent")
|
|
531
|
+
judge_budget = safety.get("judge_cost_budget")
|
|
532
|
+
if judge_cost is not None:
|
|
533
|
+
budget_display = f" / ${judge_budget:.2f}" if judge_budget else ""
|
|
534
|
+
items.append(f"""
|
|
535
|
+
<div class="calibration-item">
|
|
536
|
+
<strong>Judge Cost</strong>
|
|
537
|
+
<span>${judge_cost:.4f}{budget_display}</span>
|
|
538
|
+
</div>
|
|
539
|
+
""")
|
|
540
|
+
|
|
541
|
+
# Stability calibration
|
|
542
|
+
stability = scores.get("stability", {})
|
|
543
|
+
if isinstance(stability, dict):
|
|
544
|
+
norm_var = stability.get("normalized_variance")
|
|
545
|
+
if norm_var is not None:
|
|
546
|
+
items.append(f"""
|
|
547
|
+
<div class="calibration-item">
|
|
548
|
+
<strong>Stability Variance</strong>
|
|
549
|
+
<span>{norm_var:.4f}</span>
|
|
550
|
+
<div class="muted">Lower is more consistent</div>
|
|
551
|
+
</div>
|
|
552
|
+
""")
|
|
553
|
+
|
|
554
|
+
if not items:
|
|
555
|
+
return ""
|
|
556
|
+
|
|
557
|
+
return f"""
|
|
558
|
+
<section>
|
|
559
|
+
<h2>Calibration & Diagnostics</h2>
|
|
560
|
+
<div class="calibration-section">
|
|
561
|
+
<div class="calibration-grid">
|
|
562
|
+
{''.join(items)}
|
|
563
|
+
</div>
|
|
564
|
+
</div>
|
|
565
|
+
</section>
|
|
566
|
+
"""
|
|
567
|
+
|
|
568
|
+
|
|
569
|
+
def _render_reproducibility_section(summary: dict[str, Any]) -> str:
|
|
570
|
+
"""Render reproducibility information (config, versions, seed)."""
|
|
571
|
+
import sys
|
|
572
|
+
import platform
|
|
573
|
+
|
|
574
|
+
items = []
|
|
575
|
+
|
|
576
|
+
# Run configuration
|
|
577
|
+
model = summary.get("model")
|
|
578
|
+
if model:
|
|
579
|
+
items.append(f"<div><strong>Model</strong><br />{model}</div>")
|
|
580
|
+
|
|
581
|
+
compare_model = summary.get("compare_model")
|
|
582
|
+
if compare_model:
|
|
583
|
+
items.append(f"<div><strong>Compare Model</strong><br />{compare_model}</div>")
|
|
584
|
+
|
|
585
|
+
# Dataset
|
|
586
|
+
dataset_path = summary.get("dataset_path")
|
|
587
|
+
if dataset_path:
|
|
588
|
+
items.append(f"<div><strong>Dataset</strong><br /><code>{dataset_path}</code></div>")
|
|
589
|
+
|
|
590
|
+
persona_path = summary.get("persona_path")
|
|
591
|
+
if persona_path:
|
|
592
|
+
items.append(f"<div><strong>Persona</strong><br /><code>{persona_path}</code></div>")
|
|
593
|
+
|
|
594
|
+
# Environment
|
|
595
|
+
items.append(f"<div><strong>Python Version</strong><br />{sys.version.split()[0]}</div>")
|
|
596
|
+
items.append(f"<div><strong>Platform</strong><br />{platform.system()} {platform.machine()}</div>")
|
|
597
|
+
|
|
598
|
+
# Run timestamp
|
|
599
|
+
run_at = summary.get("run_at")
|
|
600
|
+
if run_at:
|
|
601
|
+
items.append(f"<div><strong>Run At</strong><br />{run_at}</div>")
|
|
602
|
+
|
|
603
|
+
return f"""
|
|
604
|
+
<section>
|
|
605
|
+
<h2>Reproducibility</h2>
|
|
606
|
+
<div class="reproducibility-section">
|
|
607
|
+
<div class="reproducibility-grid">
|
|
608
|
+
{''.join(items)}
|
|
609
|
+
</div>
|
|
610
|
+
</div>
|
|
611
|
+
</section>
|
|
612
|
+
"""
|
|
613
|
+
|
|
614
|
+
|
|
615
|
+
def _render_charts(scores: dict[str, Any]) -> str:
|
|
616
|
+
"""Render score visualizations using Chart.js."""
|
|
617
|
+
# Collect main metrics
|
|
618
|
+
metric_labels = []
|
|
619
|
+
metric_values = []
|
|
620
|
+
|
|
621
|
+
for scorer_id in ("authenticity", "safety", "stability"):
|
|
622
|
+
scorer_data = scores.get(scorer_id, {})
|
|
623
|
+
if isinstance(scorer_data, dict):
|
|
624
|
+
if scorer_id == "authenticity":
|
|
625
|
+
value = scorer_data.get("mean")
|
|
626
|
+
if value is not None:
|
|
627
|
+
metric_labels.append("Authenticity")
|
|
628
|
+
metric_values.append(value)
|
|
629
|
+
elif scorer_id == "safety":
|
|
630
|
+
value = scorer_data.get("score")
|
|
631
|
+
if value is not None:
|
|
632
|
+
metric_labels.append("Safety")
|
|
633
|
+
metric_values.append(value)
|
|
634
|
+
elif scorer_id == "stability":
|
|
635
|
+
value = scorer_data.get("stability")
|
|
636
|
+
if value is not None:
|
|
637
|
+
metric_labels.append("Stability")
|
|
638
|
+
metric_values.append(value)
|
|
639
|
+
|
|
640
|
+
if not metric_labels:
|
|
641
|
+
return ""
|
|
642
|
+
|
|
643
|
+
chart_data = {
|
|
644
|
+
"labels": metric_labels,
|
|
645
|
+
"values": metric_values,
|
|
646
|
+
}
|
|
647
|
+
|
|
648
|
+
import json
|
|
649
|
+
chart_json = json.dumps(chart_data)
|
|
650
|
+
|
|
651
|
+
return f"""
|
|
652
|
+
<div class="chart-container">
|
|
653
|
+
<h3>Score Overview</h3>
|
|
654
|
+
<canvas id="scoreChart" width="400" height="200"></canvas>
|
|
655
|
+
<script>
|
|
656
|
+
const chartData = {chart_json};
|
|
657
|
+
const ctx = document.getElementById('scoreChart').getContext('2d');
|
|
658
|
+
new Chart(ctx, {{
|
|
659
|
+
type: 'bar',
|
|
660
|
+
data: {{
|
|
661
|
+
labels: chartData.labels,
|
|
662
|
+
datasets: [{{
|
|
663
|
+
label: 'Score',
|
|
664
|
+
data: chartData.values,
|
|
665
|
+
backgroundColor: [
|
|
666
|
+
'rgba(34, 211, 238, 0.6)',
|
|
667
|
+
'rgba(74, 222, 128, 0.6)',
|
|
668
|
+
'rgba(251, 191, 36, 0.6)',
|
|
669
|
+
],
|
|
670
|
+
borderColor: [
|
|
671
|
+
'rgba(34, 211, 238, 1)',
|
|
672
|
+
'rgba(74, 222, 128, 1)',
|
|
673
|
+
'rgba(251, 191, 36, 1)',
|
|
674
|
+
],
|
|
675
|
+
borderWidth: 2
|
|
676
|
+
}}]
|
|
677
|
+
}},
|
|
678
|
+
options: {{
|
|
679
|
+
responsive: true,
|
|
680
|
+
maintainAspectRatio: true,
|
|
681
|
+
scales: {{
|
|
682
|
+
y: {{
|
|
683
|
+
beginAtZero: true,
|
|
684
|
+
max: 1.0,
|
|
685
|
+
ticks: {{
|
|
686
|
+
color: '#e2e8f0'
|
|
687
|
+
}},
|
|
688
|
+
grid: {{
|
|
689
|
+
color: 'rgba(226, 232, 240, 0.1)'
|
|
690
|
+
}}
|
|
691
|
+
}},
|
|
692
|
+
x: {{
|
|
693
|
+
ticks: {{
|
|
694
|
+
color: '#e2e8f0'
|
|
695
|
+
}},
|
|
696
|
+
grid: {{
|
|
697
|
+
color: 'rgba(226, 232, 240, 0.1)'
|
|
698
|
+
}}
|
|
699
|
+
}}
|
|
700
|
+
}},
|
|
701
|
+
plugins: {{
|
|
702
|
+
legend: {{
|
|
703
|
+
display: false
|
|
704
|
+
}}
|
|
705
|
+
}}
|
|
706
|
+
}}
|
|
707
|
+
}});
|
|
708
|
+
</script>
|
|
709
|
+
</div>
|
|
710
|
+
"""
|
|
711
|
+
|
|
712
|
+
|
|
713
|
+
def _prepare_csv_data(scores: dict[str, Any]) -> list[dict]:
|
|
714
|
+
"""Prepare scores data for CSV export."""
|
|
715
|
+
rows = []
|
|
716
|
+
for scorer_id, metrics in scores.items():
|
|
717
|
+
if isinstance(metrics, dict):
|
|
718
|
+
row = {"scorer": scorer_id}
|
|
719
|
+
row.update(metrics)
|
|
720
|
+
rows.append(row)
|
|
721
|
+
return rows
|