evalcore 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- evalcore-0.1.0.dist-info/METADATA +828 -0
- evalcore-0.1.0.dist-info/RECORD +34 -0
- evalcore-0.1.0.dist-info/WHEEL +4 -0
- evalcore-0.1.0.dist-info/entry_points.txt +2 -0
- evalcore-0.1.0.dist-info/licenses/LICENSE +28 -0
- evalkit/__init__.py +39 -0
- evalkit/adapters/__init__.py +15 -0
- evalkit/adapters/_env.py +24 -0
- evalkit/adapters/base.py +40 -0
- evalkit/adapters/http.py +96 -0
- evalkit/adapters/replay.py +41 -0
- evalkit/cli.py +655 -0
- evalkit/compare.py +137 -0
- evalkit/graders/__init__.py +18 -0
- evalkit/graders/base.py +77 -0
- evalkit/graders/classification.py +107 -0
- evalkit/graders/deterministic.py +127 -0
- evalkit/graders/judge.py +635 -0
- evalkit/graders/numeric.py +91 -0
- evalkit/loader.py +129 -0
- evalkit/models.py +390 -0
- evalkit/pairwise.py +324 -0
- evalkit/py.typed +0 -0
- evalkit/rating.py +1323 -0
- evalkit/refs.py +71 -0
- evalkit/report.py +186 -0
- evalkit/reporters/__init__.py +33 -0
- evalkit/reporters/base.py +159 -0
- evalkit/reporters/html.py +426 -0
- evalkit/reporters/markdown.py +74 -0
- evalkit/retry.py +85 -0
- evalkit/runner.py +283 -0
- evalkit/store.py +286 -0
- evalkit/sweep.py +93 -0
evalkit/cli.py
ADDED
|
@@ -0,0 +1,655 @@
|
|
|
1
|
+
"""evalkit command-line interface.
|
|
2
|
+
|
|
3
|
+
python -m evalkit.cli run --suite S --variant V [--mode replay] [--out F]
|
|
4
|
+
[--checkpoint F [--resume]]
|
|
5
|
+
python -m evalkit.cli gate --suite S [--baseline B --candidate C]
|
|
6
|
+
[--mode replay] [--export OUTBOX]
|
|
7
|
+
python -m evalkit.cli compare --suite S --baseline F1 --candidate F2
|
|
8
|
+
python -m evalkit.cli sweep --suite S [--variants a,b,c] [--mode replay]
|
|
9
|
+
python -m evalkit.cli pairwise --suite S --a V1 --b V2 [--mode replay]
|
|
10
|
+
[--preferences P]
|
|
11
|
+
python -m evalkit.cli rate --run R --ratings F --dimensions a,b
|
|
12
|
+
python -m evalkit.cli rank --run-a A --run-b B --preferences P
|
|
13
|
+
--dimensions a,b
|
|
14
|
+
python -m evalkit.cli agreement --run R --ratings F --dimensions a,b
|
|
15
|
+
python -m evalkit.cli preferences --run-a A --run-b B --preferences P
|
|
16
|
+
[--report html] [--report-out F]
|
|
17
|
+
python -m evalkit.cli report --run R [--ratings F --dimensions a,b]
|
|
18
|
+
[--report html] [--report-out F]
|
|
19
|
+
|
|
20
|
+
``gate`` is the CI workhorse: it runs the baseline and candidate variants and
|
|
21
|
+
compares them in one shot, exiting non-zero when the verdict is ``fail``.
|
|
22
|
+
``--plugins mod1,mod2`` imports consumer modules so their custom graders /
|
|
23
|
+
adapters register before the run.
|
|
24
|
+
"""
|
|
25
|
+
|
|
26
|
+
import argparse
|
|
27
|
+
import asyncio
|
|
28
|
+
import datetime
|
|
29
|
+
import importlib
|
|
30
|
+
import os
|
|
31
|
+
import pathlib
|
|
32
|
+
import sys
|
|
33
|
+
|
|
34
|
+
from evalkit import compare as compare_mod
|
|
35
|
+
from evalkit import loader, rating, report, reporters, runner, store
|
|
36
|
+
from evalkit import pairwise as pairwise_mod
|
|
37
|
+
from evalkit import sweep as sweep_mod
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def _load_plugins(spec: str | None) -> None:
|
|
41
|
+
if not spec:
|
|
42
|
+
return
|
|
43
|
+
# Consumers run the CLI from their repo root; the console script
|
|
44
|
+
# (unlike `python -m`) does not put the cwd on sys.path, so add it
|
|
45
|
+
# or `--plugins my_pkg.graders` could never import.
|
|
46
|
+
cwd = os.getcwd()
|
|
47
|
+
if cwd not in sys.path:
|
|
48
|
+
sys.path.insert(0, cwd)
|
|
49
|
+
for name in filter(None, spec.split(',')):
|
|
50
|
+
importlib.import_module(name.strip())
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
def _now() -> str:
|
|
54
|
+
return datetime.datetime.now(datetime.UTC).isoformat()
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def _reporter(args: argparse.Namespace) -> reporters.Reporter:
|
|
58
|
+
"""Build the reporter chosen by ``--report`` (default markdown)."""
|
|
59
|
+
return reporters.build_reporter(
|
|
60
|
+
getattr(args, 'report', None) or 'markdown'
|
|
61
|
+
)
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def _emit_report(args: argparse.Namespace, body: str) -> None:
|
|
65
|
+
"""Write the rendered report to ``--report-out``, else print it."""
|
|
66
|
+
out = getattr(args, 'report_out', None)
|
|
67
|
+
if out:
|
|
68
|
+
pathlib.Path(out).write_text(body, encoding='utf-8')
|
|
69
|
+
print(f'wrote report -> {out}', file=sys.stderr)
|
|
70
|
+
else:
|
|
71
|
+
print(body)
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def _thresholds_variants(suite: loader.SuiteConfig) -> tuple[str, str]:
|
|
75
|
+
variants = suite.thresholds.get('variants', {})
|
|
76
|
+
baseline = variants.get('baseline')
|
|
77
|
+
candidate = variants.get('candidate')
|
|
78
|
+
names = list(suite.variants)
|
|
79
|
+
if not baseline:
|
|
80
|
+
baseline = 'baseline' if 'baseline' in suite.variants else names[0]
|
|
81
|
+
if not candidate:
|
|
82
|
+
candidate = 'candidate' if 'candidate' in suite.variants else names[-1]
|
|
83
|
+
return baseline, candidate
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
def _cmd_run(args: argparse.Namespace) -> int:
|
|
87
|
+
_load_plugins(args.plugins)
|
|
88
|
+
suite = loader.load_suite(args.suite)
|
|
89
|
+
run = runner.run_suite_sync(
|
|
90
|
+
suite,
|
|
91
|
+
args.variant,
|
|
92
|
+
mode=args.mode,
|
|
93
|
+
grader_mode=args.judge_mode,
|
|
94
|
+
revision=args.revision,
|
|
95
|
+
created_at=_now(),
|
|
96
|
+
checkpoint=args.checkpoint,
|
|
97
|
+
resume=args.resume,
|
|
98
|
+
)
|
|
99
|
+
rep = _reporter(args)
|
|
100
|
+
_emit_report(args, reporters.render_run(rep, run))
|
|
101
|
+
if args.out:
|
|
102
|
+
store.write_scorecard(args.out, run.scorecard)
|
|
103
|
+
print(f'\nwrote {args.out}', file=sys.stderr)
|
|
104
|
+
if args.run_out:
|
|
105
|
+
store.write_run(args.run_out, run)
|
|
106
|
+
print(
|
|
107
|
+
f'wrote {args.run_out} ({len(run.results)} results)',
|
|
108
|
+
file=sys.stderr,
|
|
109
|
+
)
|
|
110
|
+
return 0
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
def _cmd_compare(args: argparse.Namespace) -> int:
|
|
114
|
+
suite = loader.load_suite(args.suite)
|
|
115
|
+
# Accept scorecard files (run --out) or full-run files (run --run-out).
|
|
116
|
+
baseline = store.load_scorecard(args.baseline)
|
|
117
|
+
candidate = store.load_scorecard(args.candidate)
|
|
118
|
+
result = compare_mod.compare(baseline, candidate, suite.thresholds)
|
|
119
|
+
rep = _reporter(args)
|
|
120
|
+
_emit_report(args, reporters.wrap_document(rep, rep.comparison(result)))
|
|
121
|
+
return 0 if result.verdict != 'fail' else 1
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
def _cmd_gate(args: argparse.Namespace) -> int:
|
|
125
|
+
_load_plugins(args.plugins)
|
|
126
|
+
suite = loader.load_suite(args.suite)
|
|
127
|
+
baseline_name, candidate_name = _thresholds_variants(suite)
|
|
128
|
+
if args.baseline:
|
|
129
|
+
baseline_name = args.baseline
|
|
130
|
+
if args.candidate:
|
|
131
|
+
candidate_name = args.candidate
|
|
132
|
+
|
|
133
|
+
baseline = runner.run_suite_sync(
|
|
134
|
+
suite,
|
|
135
|
+
baseline_name,
|
|
136
|
+
mode=args.mode,
|
|
137
|
+
grader_mode=args.judge_mode,
|
|
138
|
+
revision=args.revision,
|
|
139
|
+
created_at=_now(),
|
|
140
|
+
)
|
|
141
|
+
candidate = runner.run_suite_sync(
|
|
142
|
+
suite,
|
|
143
|
+
candidate_name,
|
|
144
|
+
mode=args.mode,
|
|
145
|
+
grader_mode=args.judge_mode,
|
|
146
|
+
revision=args.revision,
|
|
147
|
+
created_at=_now(),
|
|
148
|
+
)
|
|
149
|
+
result = compare_mod.compare(
|
|
150
|
+
baseline.scorecard, candidate.scorecard, suite.thresholds
|
|
151
|
+
)
|
|
152
|
+
|
|
153
|
+
rep = _reporter(args)
|
|
154
|
+
body = (
|
|
155
|
+
rep.scorecard(baseline.scorecard)
|
|
156
|
+
+ '\n\n'
|
|
157
|
+
+ rep.scorecard(candidate.scorecard)
|
|
158
|
+
+ '\n\n'
|
|
159
|
+
+ rep.comparison(result)
|
|
160
|
+
)
|
|
161
|
+
_emit_report(args, reporters.wrap_document(rep, body))
|
|
162
|
+
|
|
163
|
+
if args.export:
|
|
164
|
+
exporter = store.JsonlOutboxExporter(args.export)
|
|
165
|
+
exporter.export(baseline.scorecard)
|
|
166
|
+
exporter.export(candidate.scorecard)
|
|
167
|
+
print(f'\nexported scorecards -> {args.export}', file=sys.stderr)
|
|
168
|
+
if args.export_scores:
|
|
169
|
+
exporter = store.JsonlOutboxExporter(args.export_scores)
|
|
170
|
+
rows = exporter.export_scores(baseline)
|
|
171
|
+
rows += exporter.export_scores(candidate)
|
|
172
|
+
print(
|
|
173
|
+
f'exported {rows} score rows -> {args.export_scores}',
|
|
174
|
+
file=sys.stderr,
|
|
175
|
+
)
|
|
176
|
+
|
|
177
|
+
return 0 if result.verdict != 'fail' else 1
|
|
178
|
+
|
|
179
|
+
|
|
180
|
+
def _cmd_sweep(args: argparse.Namespace) -> int:
|
|
181
|
+
_load_plugins(args.plugins)
|
|
182
|
+
suite = loader.load_suite(args.suite)
|
|
183
|
+
names = (
|
|
184
|
+
[v.strip() for v in args.variants.split(',') if v.strip()]
|
|
185
|
+
if args.variants
|
|
186
|
+
else None
|
|
187
|
+
)
|
|
188
|
+
runs = asyncio.run(
|
|
189
|
+
sweep_mod.run_sweep(
|
|
190
|
+
suite,
|
|
191
|
+
names,
|
|
192
|
+
mode=args.mode,
|
|
193
|
+
grader_mode=args.judge_mode,
|
|
194
|
+
revision=args.revision,
|
|
195
|
+
created_at=_now(),
|
|
196
|
+
)
|
|
197
|
+
)
|
|
198
|
+
result = sweep_mod.summarize_sweep(runs, suite.thresholds)
|
|
199
|
+
print(report.render_sweep(result))
|
|
200
|
+
if args.export:
|
|
201
|
+
exporter = store.JsonlOutboxExporter(args.export)
|
|
202
|
+
for run in runs:
|
|
203
|
+
exporter.export(run.scorecard)
|
|
204
|
+
print(
|
|
205
|
+
f'\nexported {len(runs)} scorecards -> {args.export}',
|
|
206
|
+
file=sys.stderr,
|
|
207
|
+
)
|
|
208
|
+
return 0
|
|
209
|
+
|
|
210
|
+
|
|
211
|
+
def _cmd_pairwise(args: argparse.Namespace) -> int:
|
|
212
|
+
_load_plugins(args.plugins)
|
|
213
|
+
suite = loader.load_suite(args.suite)
|
|
214
|
+
config = dict(suite.thresholds.get('pairwise') or {})
|
|
215
|
+
content_ref = args.content_ref or config.get('content_ref')
|
|
216
|
+
if not content_ref:
|
|
217
|
+
raise SystemExit('pairwise needs --content-ref or thresholds.pairwise')
|
|
218
|
+
mode = args.mode or suite.mode_default
|
|
219
|
+
run_a = runner.run_suite_sync(suite, args.a, mode=mode, created_at=_now())
|
|
220
|
+
run_b = runner.run_suite_sync(suite, args.b, mode=mode, created_at=_now())
|
|
221
|
+
client = pairwise_mod.build_pairwise_client(mode, config)
|
|
222
|
+
result = asyncio.run(
|
|
223
|
+
pairwise_mod.judge_pairwise(
|
|
224
|
+
run_a,
|
|
225
|
+
run_b,
|
|
226
|
+
content_ref=content_ref,
|
|
227
|
+
client=client,
|
|
228
|
+
context_refs=config.get('context_refs'),
|
|
229
|
+
rubric=config.get('rubric'),
|
|
230
|
+
judge_version=config.get('judge_version', 'v1'),
|
|
231
|
+
)
|
|
232
|
+
)
|
|
233
|
+
body = report.render_pairwise(result)
|
|
234
|
+
if args.preferences:
|
|
235
|
+
# Fold in how the human panel agreed with this judge, per case.
|
|
236
|
+
prefs = store.read_preferences(args.preferences)
|
|
237
|
+
agreement = rating.compute_pairwise_agreement(prefs, result)
|
|
238
|
+
body += '\n\n' + reporters.render_pairwise_agreement(
|
|
239
|
+
reporters.build_reporter('markdown'), agreement
|
|
240
|
+
)
|
|
241
|
+
print(body)
|
|
242
|
+
return 0
|
|
243
|
+
|
|
244
|
+
|
|
245
|
+
def _parse_views(specs: list[str] | None) -> list[dict] | None:
|
|
246
|
+
"""Parse ``--view`` specs of the form ``label:kind:ref`` or ``label:ref``.
|
|
247
|
+
|
|
248
|
+
``kind`` (image/pdf/html/json/text) is optional; when omitted it is
|
|
249
|
+
inferred from the resolved value. Refs use dots, never colons, so a
|
|
250
|
+
plain split is unambiguous.
|
|
251
|
+
"""
|
|
252
|
+
if not specs:
|
|
253
|
+
return None
|
|
254
|
+
views: list[dict] = []
|
|
255
|
+
for spec in specs:
|
|
256
|
+
parts = spec.split(':')
|
|
257
|
+
if len(parts) == 3:
|
|
258
|
+
views.append(
|
|
259
|
+
{'label': parts[0], 'kind': parts[1], 'ref': parts[2]}
|
|
260
|
+
)
|
|
261
|
+
elif len(parts) == 2:
|
|
262
|
+
views.append({'label': parts[0], 'ref': parts[1]})
|
|
263
|
+
else:
|
|
264
|
+
raise SystemExit(
|
|
265
|
+
f'bad --view {spec!r}; use label:kind:ref or label:ref'
|
|
266
|
+
)
|
|
267
|
+
return views
|
|
268
|
+
|
|
269
|
+
|
|
270
|
+
def _cmd_rate(args: argparse.Namespace) -> int:
|
|
271
|
+
runs = [store.read_run(path) for path in args.run]
|
|
272
|
+
rating.serve(
|
|
273
|
+
runs,
|
|
274
|
+
args.ratings,
|
|
275
|
+
[d.strip() for d in args.dimensions.split(',') if d.strip()],
|
|
276
|
+
scale=args.scale,
|
|
277
|
+
content_ref=args.content_ref,
|
|
278
|
+
screenshot_ref=args.screenshot_ref,
|
|
279
|
+
views=_parse_views(args.view),
|
|
280
|
+
port=args.port,
|
|
281
|
+
open_browser=not args.no_open,
|
|
282
|
+
)
|
|
283
|
+
return 0
|
|
284
|
+
|
|
285
|
+
|
|
286
|
+
def _cmd_rank(args: argparse.Namespace) -> int:
|
|
287
|
+
run_a = store.read_run(args.run_a)
|
|
288
|
+
run_b = store.read_run(args.run_b)
|
|
289
|
+
rating.serve_rank(
|
|
290
|
+
run_a,
|
|
291
|
+
run_b,
|
|
292
|
+
args.preferences,
|
|
293
|
+
[d.strip() for d in args.dimensions.split(',') if d.strip()],
|
|
294
|
+
content_ref=args.content_ref,
|
|
295
|
+
screenshot_ref=args.screenshot_ref,
|
|
296
|
+
views=_parse_views(args.view),
|
|
297
|
+
port=args.port,
|
|
298
|
+
open_browser=not args.no_open,
|
|
299
|
+
)
|
|
300
|
+
return 0
|
|
301
|
+
|
|
302
|
+
|
|
303
|
+
def _cmd_preferences(args: argparse.Namespace) -> int:
|
|
304
|
+
run_a = store.read_run(args.run_a)
|
|
305
|
+
run_b = store.read_run(args.run_b)
|
|
306
|
+
prefs = store.read_preferences(args.preferences)
|
|
307
|
+
dims = (
|
|
308
|
+
[d.strip() for d in args.dimensions.split(',') if d.strip()]
|
|
309
|
+
if args.dimensions
|
|
310
|
+
else None
|
|
311
|
+
)
|
|
312
|
+
result = rating.aggregate_preferences(run_a, run_b, prefs, dims)
|
|
313
|
+
rep = _reporter(args)
|
|
314
|
+
_emit_report(
|
|
315
|
+
args,
|
|
316
|
+
reporters.wrap_document(
|
|
317
|
+
rep, reporters.render_preferences(rep, result)
|
|
318
|
+
),
|
|
319
|
+
)
|
|
320
|
+
return 0
|
|
321
|
+
|
|
322
|
+
|
|
323
|
+
def _cmd_agreement(args: argparse.Namespace) -> int:
|
|
324
|
+
run = store.read_run(args.run)
|
|
325
|
+
ratings = store.read_ratings(args.ratings)
|
|
326
|
+
result = rating.compute_agreement(
|
|
327
|
+
run,
|
|
328
|
+
ratings,
|
|
329
|
+
[d.strip() for d in args.dimensions.split(',') if d.strip()],
|
|
330
|
+
judge_name=args.judge_name,
|
|
331
|
+
scale=args.scale,
|
|
332
|
+
)
|
|
333
|
+
print(report.render_agreement(result))
|
|
334
|
+
return 0
|
|
335
|
+
|
|
336
|
+
|
|
337
|
+
def _cmd_report(args: argparse.Namespace) -> int:
|
|
338
|
+
run = store.read_run(args.run)
|
|
339
|
+
rep = _reporter(args)
|
|
340
|
+
body = reporters.run_body(rep, run)
|
|
341
|
+
if args.ratings:
|
|
342
|
+
if not args.dimensions:
|
|
343
|
+
raise SystemExit('report --ratings also needs --dimensions')
|
|
344
|
+
ratings = store.read_ratings(args.ratings)
|
|
345
|
+
result = rating.compute_agreement(
|
|
346
|
+
run,
|
|
347
|
+
ratings,
|
|
348
|
+
[d.strip() for d in args.dimensions.split(',') if d.strip()],
|
|
349
|
+
judge_name=args.judge_name,
|
|
350
|
+
scale=args.scale,
|
|
351
|
+
)
|
|
352
|
+
body += '\n\n' + reporters.render_agreement(rep, result)
|
|
353
|
+
_emit_report(args, reporters.wrap_document(rep, body))
|
|
354
|
+
return 0
|
|
355
|
+
|
|
356
|
+
|
|
357
|
+
def build_parser() -> argparse.ArgumentParser:
|
|
358
|
+
"""Construct the argument parser (exposed for tests)."""
|
|
359
|
+
parser = argparse.ArgumentParser(prog='evalkit')
|
|
360
|
+
parser.add_argument(
|
|
361
|
+
'--plugins',
|
|
362
|
+
help='comma-separated modules to import (register custom graders)',
|
|
363
|
+
)
|
|
364
|
+
sub = parser.add_subparsers(dest='command', required=True)
|
|
365
|
+
|
|
366
|
+
run = sub.add_parser('run', help='run one variant of a suite')
|
|
367
|
+
run.add_argument('--suite', required=True)
|
|
368
|
+
run.add_argument('--variant', required=True)
|
|
369
|
+
run.add_argument('--mode')
|
|
370
|
+
run.add_argument(
|
|
371
|
+
'--judge-mode',
|
|
372
|
+
dest='judge_mode',
|
|
373
|
+
help='grader mode when it differs from --mode '
|
|
374
|
+
'(e.g. live judges on replayed data)',
|
|
375
|
+
)
|
|
376
|
+
run.add_argument('--out', help='write the scorecard as JSON')
|
|
377
|
+
run.add_argument(
|
|
378
|
+
'--run-out',
|
|
379
|
+
dest='run_out',
|
|
380
|
+
help='write the full run (scorecard + per-sample results) as JSON',
|
|
381
|
+
)
|
|
382
|
+
run.add_argument(
|
|
383
|
+
'--revision',
|
|
384
|
+
help='opaque provenance id stamped on the scorecard '
|
|
385
|
+
'(git SHA, image digest, release label, ...)',
|
|
386
|
+
)
|
|
387
|
+
run.add_argument(
|
|
388
|
+
'--report',
|
|
389
|
+
default='markdown',
|
|
390
|
+
help='report format: markdown (default), html, or a '
|
|
391
|
+
'registered custom type',
|
|
392
|
+
)
|
|
393
|
+
run.add_argument(
|
|
394
|
+
'--report-out',
|
|
395
|
+
dest='report_out',
|
|
396
|
+
help='write the rendered report to a file instead of stdout',
|
|
397
|
+
)
|
|
398
|
+
run.add_argument(
|
|
399
|
+
'--checkpoint',
|
|
400
|
+
help='JSONL checkpoint appended as each (case, sample) completes, so '
|
|
401
|
+
'an interrupted run can be --resumed',
|
|
402
|
+
)
|
|
403
|
+
run.add_argument(
|
|
404
|
+
'--resume',
|
|
405
|
+
action='store_true',
|
|
406
|
+
help='reuse completed results from --checkpoint (same suite/dataset/'
|
|
407
|
+
'variant) and run only what is missing',
|
|
408
|
+
)
|
|
409
|
+
run.set_defaults(func=_cmd_run)
|
|
410
|
+
|
|
411
|
+
cmp_ = sub.add_parser('compare', help='compare two saved scorecards')
|
|
412
|
+
cmp_.add_argument('--suite', required=True)
|
|
413
|
+
cmp_.add_argument('--baseline', required=True)
|
|
414
|
+
cmp_.add_argument('--candidate', required=True)
|
|
415
|
+
cmp_.add_argument(
|
|
416
|
+
'--report',
|
|
417
|
+
default='markdown',
|
|
418
|
+
help='report format: markdown (default), html, or a '
|
|
419
|
+
'registered custom type',
|
|
420
|
+
)
|
|
421
|
+
cmp_.add_argument(
|
|
422
|
+
'--report-out',
|
|
423
|
+
dest='report_out',
|
|
424
|
+
help='write the rendered report to a file instead of stdout',
|
|
425
|
+
)
|
|
426
|
+
cmp_.set_defaults(func=_cmd_compare)
|
|
427
|
+
|
|
428
|
+
gate = sub.add_parser('gate', help='run baseline+candidate and compare')
|
|
429
|
+
gate.add_argument('--suite', required=True)
|
|
430
|
+
gate.add_argument('--baseline')
|
|
431
|
+
gate.add_argument('--candidate')
|
|
432
|
+
gate.add_argument('--mode')
|
|
433
|
+
gate.add_argument(
|
|
434
|
+
'--judge-mode',
|
|
435
|
+
dest='judge_mode',
|
|
436
|
+
help='grader mode when it differs from --mode '
|
|
437
|
+
'(e.g. live judges on replayed data)',
|
|
438
|
+
)
|
|
439
|
+
gate.add_argument('--export', help='append scorecards to a JSONL outbox')
|
|
440
|
+
gate.add_argument(
|
|
441
|
+
'--export-scores',
|
|
442
|
+
dest='export_scores',
|
|
443
|
+
help='append per-case score rows to a JSONL outbox (eval_scores)',
|
|
444
|
+
)
|
|
445
|
+
gate.add_argument(
|
|
446
|
+
'--revision', help='opaque provenance id stamped on both scorecards'
|
|
447
|
+
)
|
|
448
|
+
gate.add_argument(
|
|
449
|
+
'--report',
|
|
450
|
+
default='markdown',
|
|
451
|
+
help='report format: markdown (default), html, or a '
|
|
452
|
+
'registered custom type',
|
|
453
|
+
)
|
|
454
|
+
gate.add_argument(
|
|
455
|
+
'--report-out',
|
|
456
|
+
dest='report_out',
|
|
457
|
+
help='write the rendered report to a file instead of stdout',
|
|
458
|
+
)
|
|
459
|
+
gate.set_defaults(func=_cmd_gate)
|
|
460
|
+
|
|
461
|
+
sweep = sub.add_parser(
|
|
462
|
+
'sweep', help='run many variants and rank them (a leaderboard)'
|
|
463
|
+
)
|
|
464
|
+
sweep.add_argument('--suite', required=True)
|
|
465
|
+
sweep.add_argument(
|
|
466
|
+
'--variants', help='comma-separated subset (default: all)'
|
|
467
|
+
)
|
|
468
|
+
sweep.add_argument('--mode')
|
|
469
|
+
sweep.add_argument(
|
|
470
|
+
'--judge-mode',
|
|
471
|
+
dest='judge_mode',
|
|
472
|
+
help='grader mode when it differs from --mode '
|
|
473
|
+
'(e.g. live judges on replayed data)',
|
|
474
|
+
)
|
|
475
|
+
sweep.add_argument('--export', help='append scorecards to a JSONL outbox')
|
|
476
|
+
sweep.add_argument('--revision')
|
|
477
|
+
sweep.set_defaults(func=_cmd_sweep)
|
|
478
|
+
|
|
479
|
+
pw = sub.add_parser(
|
|
480
|
+
'pairwise', help='A-vs-B win-rate between two variants'
|
|
481
|
+
)
|
|
482
|
+
pw.add_argument('--suite', required=True)
|
|
483
|
+
pw.add_argument('--a', required=True, help='variant A name')
|
|
484
|
+
pw.add_argument('--b', required=True, help='variant B name')
|
|
485
|
+
pw.add_argument('--mode')
|
|
486
|
+
pw.add_argument(
|
|
487
|
+
'--content-ref',
|
|
488
|
+
dest='content_ref',
|
|
489
|
+
help='ref to the text to compare (else thresholds.pairwise)',
|
|
490
|
+
)
|
|
491
|
+
pw.add_argument(
|
|
492
|
+
'--preferences',
|
|
493
|
+
help='JSONL human preferences (from `rank`) to append a '
|
|
494
|
+
'human-panel vs judge agreement section',
|
|
495
|
+
)
|
|
496
|
+
pw.set_defaults(func=_cmd_pairwise)
|
|
497
|
+
|
|
498
|
+
rate = sub.add_parser(
|
|
499
|
+
'rate', help='blind human-rating web app over saved run(s)'
|
|
500
|
+
)
|
|
501
|
+
rate.add_argument(
|
|
502
|
+
'--run',
|
|
503
|
+
action='append',
|
|
504
|
+
required=True,
|
|
505
|
+
help='a run JSON from `run --run-out` (repeat to blind across runs)',
|
|
506
|
+
)
|
|
507
|
+
rate.add_argument('--ratings', required=True, help='JSONL ratings output')
|
|
508
|
+
rate.add_argument(
|
|
509
|
+
'--dimensions', required=True, help='comma-separated rubric keys'
|
|
510
|
+
)
|
|
511
|
+
rate.add_argument('--scale', type=int, default=5)
|
|
512
|
+
rate.add_argument(
|
|
513
|
+
'--content-ref',
|
|
514
|
+
dest='content_ref',
|
|
515
|
+
help='ref to HTML/text to render (e.g. output.html)',
|
|
516
|
+
)
|
|
517
|
+
rate.add_argument(
|
|
518
|
+
'--screenshot-ref',
|
|
519
|
+
dest='screenshot_ref',
|
|
520
|
+
default='artifacts.screenshot',
|
|
521
|
+
help='ref to a screenshot artifact path',
|
|
522
|
+
)
|
|
523
|
+
rate.add_argument(
|
|
524
|
+
'--view',
|
|
525
|
+
action='append',
|
|
526
|
+
help='explicit panel as label:kind:ref (kind: image|pdf|html|json|'
|
|
527
|
+
'text; optional). Repeatable; overrides --content-ref/--screenshot-'
|
|
528
|
+
'ref. Omit all to auto-derive from artifacts + output fields.',
|
|
529
|
+
)
|
|
530
|
+
rate.add_argument('--port', type=int, default=8900)
|
|
531
|
+
rate.add_argument('--no-open', action='store_true', dest='no_open')
|
|
532
|
+
rate.set_defaults(func=_cmd_rate)
|
|
533
|
+
|
|
534
|
+
rank = sub.add_parser(
|
|
535
|
+
'rank',
|
|
536
|
+
help='blind side-by-side A-vs-B ranking web app (human win-rate)',
|
|
537
|
+
)
|
|
538
|
+
rank.add_argument(
|
|
539
|
+
'--run-a', dest='run_a', required=True, help='variant A run JSON'
|
|
540
|
+
)
|
|
541
|
+
rank.add_argument(
|
|
542
|
+
'--run-b', dest='run_b', required=True, help='variant B run JSON'
|
|
543
|
+
)
|
|
544
|
+
rank.add_argument(
|
|
545
|
+
'--preferences', required=True, help='JSONL preferences output'
|
|
546
|
+
)
|
|
547
|
+
rank.add_argument(
|
|
548
|
+
'--dimensions',
|
|
549
|
+
required=True,
|
|
550
|
+
help='comma-separated rubric keys picked per pair (plus overall)',
|
|
551
|
+
)
|
|
552
|
+
rank.add_argument(
|
|
553
|
+
'--content-ref',
|
|
554
|
+
dest='content_ref',
|
|
555
|
+
help='ref to HTML/text to render (e.g. output.html)',
|
|
556
|
+
)
|
|
557
|
+
rank.add_argument(
|
|
558
|
+
'--screenshot-ref',
|
|
559
|
+
dest='screenshot_ref',
|
|
560
|
+
default='artifacts.screenshot',
|
|
561
|
+
help='ref to a screenshot artifact path',
|
|
562
|
+
)
|
|
563
|
+
rank.add_argument(
|
|
564
|
+
'--view',
|
|
565
|
+
action='append',
|
|
566
|
+
help='explicit panel as label:kind:ref (see `rate`); repeatable',
|
|
567
|
+
)
|
|
568
|
+
rank.add_argument('--port', type=int, default=8901)
|
|
569
|
+
rank.add_argument('--no-open', action='store_true', dest='no_open')
|
|
570
|
+
rank.set_defaults(func=_cmd_rank)
|
|
571
|
+
|
|
572
|
+
prefs = sub.add_parser(
|
|
573
|
+
'preferences',
|
|
574
|
+
help='human A-vs-B win-rate from saved runs + a preferences file',
|
|
575
|
+
)
|
|
576
|
+
prefs.add_argument(
|
|
577
|
+
'--run-a', dest='run_a', required=True, help='variant A run JSON'
|
|
578
|
+
)
|
|
579
|
+
prefs.add_argument(
|
|
580
|
+
'--run-b', dest='run_b', required=True, help='variant B run JSON'
|
|
581
|
+
)
|
|
582
|
+
prefs.add_argument(
|
|
583
|
+
'--preferences', required=True, help='a JSONL preferences file'
|
|
584
|
+
)
|
|
585
|
+
prefs.add_argument(
|
|
586
|
+
'--dimensions',
|
|
587
|
+
help='comma-separated rubric keys (default: all seen in the file)',
|
|
588
|
+
)
|
|
589
|
+
prefs.add_argument(
|
|
590
|
+
'--report',
|
|
591
|
+
default='markdown',
|
|
592
|
+
help='report format: markdown (default), html, or a '
|
|
593
|
+
'registered custom type',
|
|
594
|
+
)
|
|
595
|
+
prefs.add_argument(
|
|
596
|
+
'--report-out',
|
|
597
|
+
dest='report_out',
|
|
598
|
+
help='write the rendered report to a file instead of stdout',
|
|
599
|
+
)
|
|
600
|
+
prefs.set_defaults(func=_cmd_preferences)
|
|
601
|
+
|
|
602
|
+
rpt = sub.add_parser(
|
|
603
|
+
'report',
|
|
604
|
+
help='render a report from a saved run (no re-run); optionally '
|
|
605
|
+
'fold in human ratings as a judge<->human agreement section',
|
|
606
|
+
)
|
|
607
|
+
rpt.add_argument(
|
|
608
|
+
'--run', required=True, help='a run JSON from `run --run-out`'
|
|
609
|
+
)
|
|
610
|
+
rpt.add_argument(
|
|
611
|
+
'--ratings',
|
|
612
|
+
help='JSONL ratings to append a judge<->human agreement section',
|
|
613
|
+
)
|
|
614
|
+
rpt.add_argument(
|
|
615
|
+
'--dimensions',
|
|
616
|
+
help='comma-separated rubric keys (required with --ratings)',
|
|
617
|
+
)
|
|
618
|
+
rpt.add_argument('--judge-name', dest='judge_name', default='quality')
|
|
619
|
+
rpt.add_argument('--scale', type=int, default=5)
|
|
620
|
+
rpt.add_argument(
|
|
621
|
+
'--report',
|
|
622
|
+
default='markdown',
|
|
623
|
+
help='report format: markdown (default), html, or a '
|
|
624
|
+
'registered custom type',
|
|
625
|
+
)
|
|
626
|
+
rpt.add_argument(
|
|
627
|
+
'--report-out',
|
|
628
|
+
dest='report_out',
|
|
629
|
+
help='write the rendered report to a file instead of stdout',
|
|
630
|
+
)
|
|
631
|
+
rpt.set_defaults(func=_cmd_report)
|
|
632
|
+
|
|
633
|
+
agree = sub.add_parser(
|
|
634
|
+
'agreement', help='judge<->human agreement over a run + ratings'
|
|
635
|
+
)
|
|
636
|
+
agree.add_argument('--run', required=True, help='a run JSON')
|
|
637
|
+
agree.add_argument('--ratings', required=True, help='a JSONL ratings file')
|
|
638
|
+
agree.add_argument(
|
|
639
|
+
'--dimensions', required=True, help='comma-separated rubric keys'
|
|
640
|
+
)
|
|
641
|
+
agree.add_argument('--judge-name', dest='judge_name', default='quality')
|
|
642
|
+
agree.add_argument('--scale', type=int, default=5)
|
|
643
|
+
agree.set_defaults(func=_cmd_agreement)
|
|
644
|
+
return parser
|
|
645
|
+
|
|
646
|
+
|
|
647
|
+
def main(argv: list[str] | None = None) -> int:
|
|
648
|
+
"""CLI entry point. Returns a process exit code."""
|
|
649
|
+
parser = build_parser()
|
|
650
|
+
args = parser.parse_args(argv)
|
|
651
|
+
return args.func(args)
|
|
652
|
+
|
|
653
|
+
|
|
654
|
+
if __name__ == '__main__':
|
|
655
|
+
raise SystemExit(main())
|