modelduel 0.6.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- modelduel/__init__.py +3 -0
- modelduel/__main__.py +3 -0
- modelduel/cli.py +535 -0
- modelduel/demo.py +26 -0
- modelduel/duel.py +169 -0
- modelduel/examples/replays/alfa/merge_intervals.md +21 -0
- modelduel/examples/replays/alfa/parse_duration.md +28 -0
- modelduel/examples/replays/alfa/slugify.md +17 -0
- modelduel/examples/replays/beta/merge_intervals.md +37 -0
- modelduel/examples/replays/beta/parse_duration.md +43 -0
- modelduel/examples/replays/beta/slugify.md +33 -0
- modelduel/examples/replays/gamma/merge_intervals.md +18 -0
- modelduel/examples/replays/gamma/parse_duration.md +21 -0
- modelduel/examples/replays/gamma/slugify.md +17 -0
- modelduel/examples/tasks/merge_intervals/meta.toml +3 -0
- modelduel/examples/tasks/merge_intervals/task.md +30 -0
- modelduel/examples/tasks/merge_intervals/test_task.py +41 -0
- modelduel/examples/tasks/parse_duration/meta.toml +3 -0
- modelduel/examples/tasks/parse_duration/task.md +27 -0
- modelduel/examples/tasks/parse_duration/test_task.py +42 -0
- modelduel/examples/tasks/slugify/meta.toml +3 -0
- modelduel/examples/tasks/slugify/task.md +24 -0
- modelduel/examples/tasks/slugify/test_task.py +38 -0
- modelduel/extract.py +51 -0
- modelduel/leaderboard.py +238 -0
- modelduel/pricing.py +95 -0
- modelduel/providers/__init__.py +42 -0
- modelduel/providers/base.py +280 -0
- modelduel/providers/gemini.py +91 -0
- modelduel/providers/openai_compat.py +127 -0
- modelduel/providers/replay.py +76 -0
- modelduel/report/__init__.py +37 -0
- modelduel/report/html.py +499 -0
- modelduel/report/leaderboard.html +105 -0
- modelduel/report/league.html +64 -0
- modelduel/report/league.py +293 -0
- modelduel/report/markdown.py +198 -0
- modelduel/report/style.css +223 -0
- modelduel/report/template.html +74 -0
- modelduel/results.py +173 -0
- modelduel/resume.py +89 -0
- modelduel/runner.py +372 -0
- modelduel/tasks.py +157 -0
- modelduel-0.6.0.dist-info/METADATA +277 -0
- modelduel-0.6.0.dist-info/RECORD +48 -0
- modelduel-0.6.0.dist-info/WHEEL +4 -0
- modelduel-0.6.0.dist-info/entry_points.txt +2 -0
- modelduel-0.6.0.dist-info/licenses/LICENSE +21 -0
modelduel/__init__.py
ADDED
modelduel/__main__.py
ADDED
modelduel/cli.py
ADDED
|
@@ -0,0 +1,535 @@
|
|
|
1
|
+
"""Interfaz de línea de órdenes de modelduel."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import argparse
|
|
6
|
+
import math
|
|
7
|
+
import re
|
|
8
|
+
import shutil
|
|
9
|
+
import sys
|
|
10
|
+
from pathlib import Path
|
|
11
|
+
|
|
12
|
+
from modelduel import __version__
|
|
13
|
+
from modelduel.demo import DEMO_MODELS, examples_dir
|
|
14
|
+
from modelduel.duel import run_duel
|
|
15
|
+
from modelduel.leaderboard import build_leaderboard
|
|
16
|
+
from modelduel.pricing import PricingError, load_prices
|
|
17
|
+
from modelduel.providers import ProviderError, get_provider
|
|
18
|
+
from modelduel.providers.base import DEFAULT_RETRIES
|
|
19
|
+
from modelduel.report import write_markdown, write_report
|
|
20
|
+
from modelduel.report.html import fmt_cost, fmt_int, fmt_seconds, plural
|
|
21
|
+
from modelduel.results import (
|
|
22
|
+
ALL_SIDES,
|
|
23
|
+
MAX_CONTENDERS,
|
|
24
|
+
MIN_CONTENDERS,
|
|
25
|
+
ResultsError,
|
|
26
|
+
load_results,
|
|
27
|
+
rank_sides,
|
|
28
|
+
save_results,
|
|
29
|
+
summarize,
|
|
30
|
+
)
|
|
31
|
+
from modelduel.resume import ResumeError, plan_resume
|
|
32
|
+
from modelduel.runner import DEFAULT_TIMEOUT, RunnerError, ensure_pytest_available
|
|
33
|
+
from modelduel.tasks import TaskError, discover_tasks, is_task_dir
|
|
34
|
+
|
|
35
|
+
# Códigos de salida: 0 duelo completado (aunque los modelos fallen tests), 1 no se pudieron
|
|
36
|
+
# escribir los resultados, 2 error de uso o de configuración, 130 interrumpido con Ctrl+C.
|
|
37
|
+
EXIT_OK = 0
|
|
38
|
+
EXIT_ERROR = 1
|
|
39
|
+
EXIT_USAGE = 2
|
|
40
|
+
EXIT_INTERRUPTED = 130
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
FORMATS = ("html", "md")
|
|
44
|
+
_WRITERS = {"html": write_report, "md": write_markdown}
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
class OutputError(Exception):
|
|
48
|
+
"""No se pudo escribir en la carpeta de salida."""
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
# argparse no trae traducciones: se traducen sus mensajes de error más habituales.
|
|
52
|
+
_ARGPARSE_ES = [
|
|
53
|
+
(r"the following arguments are required: (.+)", r"faltan argumentos obligatorios: \1"),
|
|
54
|
+
(
|
|
55
|
+
r"argument orden: invalid choice: (.+?) \(choose from (.+)\)",
|
|
56
|
+
r"orden no válida: \1 (elige entre \2)",
|
|
57
|
+
),
|
|
58
|
+
(
|
|
59
|
+
r"argument (.+?): invalid choice: (.+?) \(choose from (.+)\)",
|
|
60
|
+
r"argumento \1: valor no válido: \2 (elige entre \3)",
|
|
61
|
+
),
|
|
62
|
+
(r"argument (.+?): invalid \w+ value: (.+)", r"argumento \1: valor no válido: \2"),
|
|
63
|
+
(r"argument (.+?): expected one argument", r"argumento \1: necesita un valor"),
|
|
64
|
+
(r"unrecognized arguments: (.+)", r"argumentos no reconocidos: \1"),
|
|
65
|
+
(r"ambiguous option: (.+?) could match (.+)", r"opción ambigua: \1 puede ser \2"),
|
|
66
|
+
]
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def translate_argparse(message: str) -> str:
|
|
70
|
+
for pattern, replacement in _ARGPARSE_ES:
|
|
71
|
+
translated, n = re.subn(f"^{pattern}$", replacement, message)
|
|
72
|
+
if n:
|
|
73
|
+
return translated
|
|
74
|
+
return message
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
class _SpanishHelpFormatter(argparse.HelpFormatter):
|
|
78
|
+
def add_usage(self, usage, actions, groups, prefix=None):
|
|
79
|
+
super().add_usage(usage, actions, groups, prefix="uso: " if prefix is None else prefix)
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
class SpanishArgumentParser(argparse.ArgumentParser):
|
|
83
|
+
"""ArgumentParser con ayuda y errores en español y código de salida ``EXIT_USAGE``."""
|
|
84
|
+
|
|
85
|
+
def __init__(self, *args, **kwargs) -> None:
|
|
86
|
+
kwargs["add_help"] = False
|
|
87
|
+
kwargs.setdefault("formatter_class", _SpanishHelpFormatter)
|
|
88
|
+
super().__init__(*args, **kwargs)
|
|
89
|
+
self._positionals.title = "argumentos posicionales"
|
|
90
|
+
self._optionals.title = "opciones"
|
|
91
|
+
self.add_argument("-h", "--help", action="help", help="muestra esta ayuda y sale")
|
|
92
|
+
|
|
93
|
+
def error(self, message: str): # type: ignore[override]
|
|
94
|
+
self.print_usage(sys.stderr)
|
|
95
|
+
self.exit(EXIT_USAGE, f"{self.prog}: error: {translate_argparse(message)}\n")
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
_FORMAT_HELP = "formatos del informe: html, md o html,md (html); results.json se guarda siempre"
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
def build_parser() -> argparse.ArgumentParser:
|
|
102
|
+
parser = SpanishArgumentParser(
|
|
103
|
+
prog="modelduel",
|
|
104
|
+
description="Dos modelos, una tarea, los mismos tests.",
|
|
105
|
+
epilog="Aviso: el código generado por los modelos se ejecuta en tu máquina.",
|
|
106
|
+
)
|
|
107
|
+
parser.add_argument(
|
|
108
|
+
"--version",
|
|
109
|
+
action="version",
|
|
110
|
+
version=f"modelduel {__version__}",
|
|
111
|
+
help="muestra la versión y sale",
|
|
112
|
+
)
|
|
113
|
+
sub = parser.add_subparsers(dest="command", required=True, metavar="orden")
|
|
114
|
+
|
|
115
|
+
run = sub.add_parser("run", help="enfrenta de 2 a 6 modelos y genera el informe")
|
|
116
|
+
run.add_argument("tasks", type=Path, help="carpeta de tareas o carpeta de una tarea")
|
|
117
|
+
run.add_argument("--a", metavar="PROVEEDOR:MODELO", help="contendiente A")
|
|
118
|
+
run.add_argument("--b", metavar="PROVEEDOR:MODELO", help="contendiente B")
|
|
119
|
+
run.add_argument(
|
|
120
|
+
"--model",
|
|
121
|
+
"-m",
|
|
122
|
+
action="append",
|
|
123
|
+
default=[],
|
|
124
|
+
metavar="PROVEEDOR:MODELO",
|
|
125
|
+
help=f"contendiente de una liga; repítelo ({MIN_CONTENDERS} a {MAX_CONTENDERS} en total, "
|
|
126
|
+
"contando --a y --b)",
|
|
127
|
+
)
|
|
128
|
+
run.add_argument("--runs", type=int, default=1, metavar="N", help="ejecuciones por tarea (1)")
|
|
129
|
+
run.add_argument(
|
|
130
|
+
"--timeout",
|
|
131
|
+
type=float,
|
|
132
|
+
default=DEFAULT_TIMEOUT,
|
|
133
|
+
metavar="S",
|
|
134
|
+
help=f"límite en segundos para los tests de cada respuesta ({DEFAULT_TIMEOUT:g})",
|
|
135
|
+
)
|
|
136
|
+
run.add_argument(
|
|
137
|
+
"--retries",
|
|
138
|
+
type=int,
|
|
139
|
+
default=DEFAULT_RETRIES,
|
|
140
|
+
metavar="N",
|
|
141
|
+
help=f"reintentos ante HTTP 429/5xx y cortes de conexión ({DEFAULT_RETRIES})",
|
|
142
|
+
)
|
|
143
|
+
run.add_argument(
|
|
144
|
+
"--resume",
|
|
145
|
+
action="store_true",
|
|
146
|
+
help="continúa el duelo de --out saltando los intentos ya terminados",
|
|
147
|
+
)
|
|
148
|
+
run.add_argument("--format", default="html", metavar="F", help=_FORMAT_HELP)
|
|
149
|
+
run.add_argument("--prices", type=Path, metavar="F.json", help="tabla de precios adicional")
|
|
150
|
+
run.add_argument(
|
|
151
|
+
"--replays", type=Path, metavar="DIR", help="carpeta de respuestas grabadas para replay"
|
|
152
|
+
)
|
|
153
|
+
run.add_argument("--out", type=Path, required=True, metavar="DIR", help="carpeta de salida")
|
|
154
|
+
|
|
155
|
+
demo = sub.add_parser(
|
|
156
|
+
"demo",
|
|
157
|
+
help="prueba modelduel sin claves, con los ejemplos incluidos y tres modelos ficticios",
|
|
158
|
+
)
|
|
159
|
+
demo.add_argument(
|
|
160
|
+
"--out",
|
|
161
|
+
type=Path,
|
|
162
|
+
default=Path("modelduel-demo"),
|
|
163
|
+
metavar="DIR",
|
|
164
|
+
help="carpeta de salida (modelduel-demo)",
|
|
165
|
+
)
|
|
166
|
+
demo.add_argument(
|
|
167
|
+
"--copy",
|
|
168
|
+
type=Path,
|
|
169
|
+
metavar="DIR",
|
|
170
|
+
help="en vez de ejecutar el duelo, copia las tareas y respuestas de ejemplo a DIR",
|
|
171
|
+
)
|
|
172
|
+
|
|
173
|
+
report = sub.add_parser(
|
|
174
|
+
"report", help="regenera el informe (HTML y/o Markdown) de results.json"
|
|
175
|
+
)
|
|
176
|
+
report.add_argument("results", type=Path, help="ruta a results.json")
|
|
177
|
+
report.add_argument(
|
|
178
|
+
"--format",
|
|
179
|
+
default="html",
|
|
180
|
+
metavar="F",
|
|
181
|
+
help="formatos del informe: html, md o html,md (html)",
|
|
182
|
+
)
|
|
183
|
+
report.add_argument("--out", type=Path, required=True, metavar="DIR", help="carpeta de salida")
|
|
184
|
+
|
|
185
|
+
board = sub.add_parser(
|
|
186
|
+
"leaderboard",
|
|
187
|
+
help="genera la página de clasificación a partir de una carpeta de results.json",
|
|
188
|
+
)
|
|
189
|
+
board.add_argument("results", type=Path, help="carpeta con los results.json (uno por duelo)")
|
|
190
|
+
board.add_argument("--out", type=Path, required=True, metavar="DIR", help="carpeta de salida")
|
|
191
|
+
|
|
192
|
+
lst = sub.add_parser("list-tasks", help="lista las tareas de una carpeta")
|
|
193
|
+
lst.add_argument("tasks", type=Path, help="carpeta de tareas")
|
|
194
|
+
return parser
|
|
195
|
+
|
|
196
|
+
|
|
197
|
+
def main(argv: list[str] | None = None) -> int:
|
|
198
|
+
for stream in (sys.stdout, sys.stderr):
|
|
199
|
+
if hasattr(stream, "reconfigure"):
|
|
200
|
+
stream.reconfigure(encoding="utf-8", errors="replace")
|
|
201
|
+
args = build_parser().parse_args(argv)
|
|
202
|
+
try:
|
|
203
|
+
if args.command == "run":
|
|
204
|
+
return cmd_run(args)
|
|
205
|
+
if args.command == "demo":
|
|
206
|
+
return cmd_demo(args)
|
|
207
|
+
if args.command == "report":
|
|
208
|
+
return cmd_report(args)
|
|
209
|
+
if args.command == "leaderboard":
|
|
210
|
+
return cmd_leaderboard(args)
|
|
211
|
+
return cmd_list_tasks(args)
|
|
212
|
+
except (TaskError, ProviderError, PricingError, ResultsError, RunnerError, ResumeError) as exc:
|
|
213
|
+
print(f"modelduel: error: {exc}", file=sys.stderr)
|
|
214
|
+
return EXIT_USAGE
|
|
215
|
+
except OutputError as exc:
|
|
216
|
+
print(f"modelduel: error: {exc}", file=sys.stderr)
|
|
217
|
+
return EXIT_ERROR
|
|
218
|
+
except KeyboardInterrupt:
|
|
219
|
+
print("\nmodelduel: interrumpido.", file=sys.stderr)
|
|
220
|
+
return EXIT_INTERRUPTED
|
|
221
|
+
|
|
222
|
+
|
|
223
|
+
def parse_formats(text: str) -> tuple[str, ...]:
|
|
224
|
+
"""``--format html,md`` -> ``("html", "md")`` sin repetidos, o error de uso."""
|
|
225
|
+
names = [part.strip() for part in text.split(",")]
|
|
226
|
+
for name in names:
|
|
227
|
+
if name not in FORMATS:
|
|
228
|
+
raise TaskError(
|
|
229
|
+
f"--format: «{name}» no es un formato válido "
|
|
230
|
+
f"(elige entre {' y '.join(FORMATS)}; sepáralos con comas)."
|
|
231
|
+
)
|
|
232
|
+
return tuple(dict.fromkeys(names))
|
|
233
|
+
|
|
234
|
+
|
|
235
|
+
def collect_specs(args: argparse.Namespace) -> list[str]:
|
|
236
|
+
"""Contendientes en orden: ``--a``, ``--b`` y después cada ``--model``."""
|
|
237
|
+
if args.b and not args.a:
|
|
238
|
+
raise TaskError("--b necesita también --a (o usa --model para cada contendiente).")
|
|
239
|
+
specs = [spec for spec in (args.a, args.b) if spec] + list(args.model)
|
|
240
|
+
if len(specs) < MIN_CONTENDERS:
|
|
241
|
+
raise TaskError(
|
|
242
|
+
f"hacen falta al menos {MIN_CONTENDERS} contendientes: usa --a y --b, "
|
|
243
|
+
"o --model varias veces."
|
|
244
|
+
)
|
|
245
|
+
if len(specs) > MAX_CONTENDERS:
|
|
246
|
+
raise TaskError(f"como máximo {MAX_CONTENDERS} contendientes (has puesto {len(specs)}).")
|
|
247
|
+
return specs
|
|
248
|
+
|
|
249
|
+
|
|
250
|
+
def _reject_duplicates(providers: dict) -> None:
|
|
251
|
+
seen: dict[str, str] = {}
|
|
252
|
+
for side, provider in providers.items():
|
|
253
|
+
if provider.spec in seen:
|
|
254
|
+
raise TaskError(
|
|
255
|
+
f"«{provider.spec}» aparece dos veces ({seen[provider.spec].upper()} y "
|
|
256
|
+
f"{side.upper()}). Para repetir un mismo modelo usa --runs."
|
|
257
|
+
)
|
|
258
|
+
seen[provider.spec] = side
|
|
259
|
+
|
|
260
|
+
|
|
261
|
+
def cmd_run(args: argparse.Namespace) -> int:
|
|
262
|
+
if args.runs < 1:
|
|
263
|
+
raise TaskError("--runs debe ser 1 o más.")
|
|
264
|
+
if not math.isfinite(args.timeout) or args.timeout <= 0:
|
|
265
|
+
raise TaskError("--timeout debe ser un número de segundos mayor que 0.")
|
|
266
|
+
if args.retries < 0:
|
|
267
|
+
raise TaskError("--retries debe ser 0 o más.")
|
|
268
|
+
formats = parse_formats(args.format)
|
|
269
|
+
specs = collect_specs(args)
|
|
270
|
+
tasks = discover_tasks(args.tasks)
|
|
271
|
+
prices = load_prices(args.prices)
|
|
272
|
+
# examples/tasks[/<tarea>] -> examples/replays
|
|
273
|
+
tasks_root = args.tasks.parent if is_task_dir(args.tasks) else args.tasks
|
|
274
|
+
replay_dirs = [
|
|
275
|
+
d
|
|
276
|
+
for d in (args.replays, tasks_root.parent / "replays", Path("examples/replays"))
|
|
277
|
+
if d is not None
|
|
278
|
+
]
|
|
279
|
+
|
|
280
|
+
def on_retry(message: str) -> None:
|
|
281
|
+
print(f" ~~ {message}", flush=True)
|
|
282
|
+
|
|
283
|
+
providers = {
|
|
284
|
+
side: get_provider(spec, replay_dirs, retries=args.retries, on_retry=on_retry)
|
|
285
|
+
for side, spec in zip(ALL_SIDES, specs, strict=False)
|
|
286
|
+
}
|
|
287
|
+
_reject_duplicates(providers)
|
|
288
|
+
ensure_pytest_available()
|
|
289
|
+
# Antes de gastar llamadas a las APIs: la carpeta de salida tiene que poder crearse.
|
|
290
|
+
_prepare_out(args.out)
|
|
291
|
+
results_path = args.out / "results.json"
|
|
292
|
+
previous = _load_previous(results_path, args.resume)
|
|
293
|
+
reuse: dict = {}
|
|
294
|
+
warnings: list[str] = []
|
|
295
|
+
if previous is not None:
|
|
296
|
+
specs = {side: provider.spec for side, provider in providers.items()}
|
|
297
|
+
try:
|
|
298
|
+
reuse, warnings = plan_resume(previous, tasks, specs, args.runs, args.timeout)
|
|
299
|
+
except (KeyError, TypeError, ValueError, AttributeError) as exc:
|
|
300
|
+
raise ResumeError(
|
|
301
|
+
f"{results_path} está dañado o incompleto ({type(exc).__name__}: {exc}). "
|
|
302
|
+
"Bórralo o elige otra carpeta --out."
|
|
303
|
+
) from exc
|
|
304
|
+
|
|
305
|
+
print(
|
|
306
|
+
f"modelduel {__version__} · {plural(len(tasks), 'tarea', 'tareas')} · "
|
|
307
|
+
f"{plural(args.runs, 'ejecución', 'ejecuciones')} por tarea"
|
|
308
|
+
)
|
|
309
|
+
for side, provider in providers.items():
|
|
310
|
+
print(f" {side.upper()} {provider.spec}")
|
|
311
|
+
print(" Aviso: el código de los modelos se ejecuta en esta máquina (temporal + límite).")
|
|
312
|
+
if args.resume:
|
|
313
|
+
total = len(tasks) * args.runs * len(providers)
|
|
314
|
+
if previous is None:
|
|
315
|
+
print(f" Reanudar: no hay {results_path}; se empieza de cero.")
|
|
316
|
+
else:
|
|
317
|
+
print(
|
|
318
|
+
f" Reanudando: {plural(len(reuse), 'intento', 'intentos')} ya hecho"
|
|
319
|
+
f"{'' if len(reuse) == 1 else 's'} de {total}; "
|
|
320
|
+
f"quedan {total - len(reuse)}."
|
|
321
|
+
)
|
|
322
|
+
for warning in warnings:
|
|
323
|
+
print(f" Aviso: {warning}")
|
|
324
|
+
print()
|
|
325
|
+
|
|
326
|
+
latest: dict = {}
|
|
327
|
+
save_failed = False
|
|
328
|
+
|
|
329
|
+
def on_update(current: dict) -> None:
|
|
330
|
+
"""Guarda ``results.json`` tras cada intento sin tumbar el duelo si falla el disco."""
|
|
331
|
+
nonlocal save_failed
|
|
332
|
+
latest["results"] = current
|
|
333
|
+
try:
|
|
334
|
+
save_results(current, results_path)
|
|
335
|
+
except OSError as exc:
|
|
336
|
+
if not save_failed:
|
|
337
|
+
save_failed = True
|
|
338
|
+
print(f" Aviso: no se pudo guardar {results_path}: {exc.strerror or exc}.")
|
|
339
|
+
|
|
340
|
+
try:
|
|
341
|
+
results = run_duel(
|
|
342
|
+
tasks,
|
|
343
|
+
providers,
|
|
344
|
+
prices,
|
|
345
|
+
runs=args.runs,
|
|
346
|
+
timeout=args.timeout,
|
|
347
|
+
progress=lambda line: print(line, flush=True),
|
|
348
|
+
on_update=on_update,
|
|
349
|
+
reuse=reuse,
|
|
350
|
+
created_at=(previous or {}).get("created_at"),
|
|
351
|
+
)
|
|
352
|
+
except KeyboardInterrupt:
|
|
353
|
+
return _interrupted(latest.get("results"), args.out, formats)
|
|
354
|
+
report_paths = _write_outputs(results, args.out, formats, with_json=True)
|
|
355
|
+
print()
|
|
356
|
+
print(scoreboard(results))
|
|
357
|
+
print()
|
|
358
|
+
print(f" Resultados {results_path}")
|
|
359
|
+
for path in report_paths:
|
|
360
|
+
print(f" Informe {path}")
|
|
361
|
+
return 0
|
|
362
|
+
|
|
363
|
+
|
|
364
|
+
def _load_previous(results_path: Path, resume: bool) -> dict | None:
|
|
365
|
+
"""``results.json`` previo de la carpeta de salida (o ``None``).
|
|
366
|
+
|
|
367
|
+
Sin ``--resume`` solo importa para no machacar por descuido un duelo a medias, que puede
|
|
368
|
+
haber costado dinero.
|
|
369
|
+
"""
|
|
370
|
+
if not results_path.is_file():
|
|
371
|
+
return None
|
|
372
|
+
if resume:
|
|
373
|
+
return load_results(results_path)
|
|
374
|
+
try:
|
|
375
|
+
previous = load_results(results_path)
|
|
376
|
+
except ResultsError:
|
|
377
|
+
return None
|
|
378
|
+
if previous.get("status") == "in_progress":
|
|
379
|
+
raise ResumeError(
|
|
380
|
+
f"{results_path} es un duelo incompleto. Continúalo con --resume o bórralo, "
|
|
381
|
+
"o elige otra carpeta --out."
|
|
382
|
+
)
|
|
383
|
+
return None
|
|
384
|
+
|
|
385
|
+
|
|
386
|
+
def _interrupted(results: dict | None, out: Path, formats: tuple[str, ...]) -> int:
|
|
387
|
+
"""Ctrl+C: lo hecho ya está en ``results.json``; se deja también el informe parcial."""
|
|
388
|
+
print("\nmodelduel: interrumpido.", file=sys.stderr)
|
|
389
|
+
if results is not None:
|
|
390
|
+
try:
|
|
391
|
+
results["summary"] = summarize(results)
|
|
392
|
+
for name in formats:
|
|
393
|
+
_WRITERS[name](results, out)
|
|
394
|
+
except (OSError, KeyError, TypeError, ValueError, AttributeError, OverflowError):
|
|
395
|
+
pass
|
|
396
|
+
print(
|
|
397
|
+
f"Lo hecho hasta ahora está en {out / 'results.json'}. "
|
|
398
|
+
"Continúa con la misma orden añadiendo --resume.",
|
|
399
|
+
file=sys.stderr,
|
|
400
|
+
)
|
|
401
|
+
return EXIT_INTERRUPTED
|
|
402
|
+
|
|
403
|
+
|
|
404
|
+
def cmd_demo(args: argparse.Namespace) -> int:
|
|
405
|
+
"""``modelduel demo``: liga de tres contendientes ficticios sobre los ejemplos incluidos."""
|
|
406
|
+
examples = examples_dir()
|
|
407
|
+
if args.copy:
|
|
408
|
+
target = args.copy
|
|
409
|
+
if target.exists() and any(target.iterdir()):
|
|
410
|
+
raise TaskError(f"{target} ya existe y no está vacía: elige otra carpeta.")
|
|
411
|
+
try:
|
|
412
|
+
shutil.copytree(
|
|
413
|
+
examples, target, dirs_exist_ok=True, ignore=shutil.ignore_patterns("__pycache__")
|
|
414
|
+
)
|
|
415
|
+
except OSError as exc:
|
|
416
|
+
raise OutputError(f"no se pudo copiar a {target}: {exc.strerror or exc}.") from exc
|
|
417
|
+
print(f"Ejemplos copiados en {target}:")
|
|
418
|
+
print(f" tareas {target / 'tasks'}")
|
|
419
|
+
print(f" respuestas {target / 'replays'}")
|
|
420
|
+
print("Pruébalos con:")
|
|
421
|
+
print(f" modelduel run {target / 'tasks'} --model replay:alfa --model replay:beta")
|
|
422
|
+
return 0
|
|
423
|
+
run_args = argparse.Namespace(
|
|
424
|
+
tasks=examples / "tasks",
|
|
425
|
+
a=None,
|
|
426
|
+
b=None,
|
|
427
|
+
model=list(DEMO_MODELS),
|
|
428
|
+
runs=1,
|
|
429
|
+
timeout=DEFAULT_TIMEOUT,
|
|
430
|
+
retries=0,
|
|
431
|
+
resume=False,
|
|
432
|
+
prices=None,
|
|
433
|
+
replays=examples / "replays",
|
|
434
|
+
format="html",
|
|
435
|
+
out=args.out,
|
|
436
|
+
)
|
|
437
|
+
code = cmd_run(run_args)
|
|
438
|
+
if code == 0:
|
|
439
|
+
print()
|
|
440
|
+
print(" Son respuestas grabadas y precios ficticios. Para un duelo real, mira la guía:")
|
|
441
|
+
print(" https://github.com/BertMarti/modelduel/blob/main/docs/USO.md")
|
|
442
|
+
return code
|
|
443
|
+
|
|
444
|
+
|
|
445
|
+
def cmd_report(args: argparse.Namespace) -> int:
|
|
446
|
+
formats = parse_formats(args.format)
|
|
447
|
+
results = load_results(args.results)
|
|
448
|
+
_prepare_out(args.out)
|
|
449
|
+
try:
|
|
450
|
+
paths = _write_outputs(results, args.out, formats, with_json=False)
|
|
451
|
+
except (KeyError, TypeError, ValueError, AttributeError, OverflowError) as exc:
|
|
452
|
+
raise ResultsError(
|
|
453
|
+
f"{args.results} no parece un results.json de modelduel ({type(exc).__name__}: {exc})."
|
|
454
|
+
) from exc
|
|
455
|
+
for path in paths:
|
|
456
|
+
print(f"Informe regenerado: {path}")
|
|
457
|
+
return 0
|
|
458
|
+
|
|
459
|
+
|
|
460
|
+
def cmd_leaderboard(args: argparse.Namespace) -> int:
|
|
461
|
+
try:
|
|
462
|
+
index = build_leaderboard(args.results, args.out)
|
|
463
|
+
except OSError as exc:
|
|
464
|
+
raise OutputError(f"no se pudo escribir en {args.out}: {exc.strerror or exc}.") from exc
|
|
465
|
+
except (KeyError, TypeError, ValueError, AttributeError, OverflowError) as exc:
|
|
466
|
+
raise ResultsError(
|
|
467
|
+
f"hay un results.json que no parece de modelduel ({type(exc).__name__}: {exc})."
|
|
468
|
+
) from exc
|
|
469
|
+
print(f"Clasificación generada: {index}")
|
|
470
|
+
return 0
|
|
471
|
+
|
|
472
|
+
|
|
473
|
+
def _prepare_out(out: Path) -> None:
|
|
474
|
+
try:
|
|
475
|
+
out.mkdir(parents=True, exist_ok=True)
|
|
476
|
+
except OSError as exc:
|
|
477
|
+
raise OutputError(f"no se pudo escribir en {out}: {exc.strerror or exc}.") from exc
|
|
478
|
+
|
|
479
|
+
|
|
480
|
+
def _write_outputs(
|
|
481
|
+
results: dict, out: Path, formats: tuple[str, ...], with_json: bool
|
|
482
|
+
) -> list[Path]:
|
|
483
|
+
try:
|
|
484
|
+
if with_json:
|
|
485
|
+
save_results(results, out / "results.json")
|
|
486
|
+
return [_WRITERS[name](results, out) for name in formats]
|
|
487
|
+
except OSError as exc:
|
|
488
|
+
raise OutputError(f"no se pudo escribir en {out}: {exc.strerror or exc}.") from exc
|
|
489
|
+
|
|
490
|
+
|
|
491
|
+
def cmd_list_tasks(args: argparse.Namespace) -> int:
|
|
492
|
+
tasks = discover_tasks(args.tasks)
|
|
493
|
+
width = max(len(t.id) for t in tasks)
|
|
494
|
+
for task in tasks:
|
|
495
|
+
extra = f" [{task.difficulty}]" if task.difficulty else ""
|
|
496
|
+
print(f"{task.id:<{width}} {task.expected_tests:>2} tests {task.title}{extra}")
|
|
497
|
+
return 0
|
|
498
|
+
|
|
499
|
+
|
|
500
|
+
def scoreboard(results: dict) -> str:
|
|
501
|
+
"""Clasificación en texto: una fila por contendiente, de mejor a peor."""
|
|
502
|
+
summary = results["summary"]
|
|
503
|
+
header = ("#", "Contendiente", "Tareas", "Tests", "Tiempo", "Tokens (ent/sal)", "Coste")
|
|
504
|
+
rows = [header]
|
|
505
|
+
for rank, side in rank_sides(summary):
|
|
506
|
+
data = summary[side]
|
|
507
|
+
rows.append(
|
|
508
|
+
(
|
|
509
|
+
str(rank),
|
|
510
|
+
f"{side.upper()} {results['contenders'][side]['spec']}",
|
|
511
|
+
f"{data['tasks_solved']}/{data['tasks_total']}",
|
|
512
|
+
f"{data['tests_passed']}/{data['tests_total']}",
|
|
513
|
+
fmt_seconds(data["latency_s"]),
|
|
514
|
+
f"{fmt_int(data['input_tokens'])}/{fmt_int(data['output_tokens'])}",
|
|
515
|
+
fmt_cost(data["cost"], data["currency"]),
|
|
516
|
+
)
|
|
517
|
+
)
|
|
518
|
+
widths = [max(len(row[col]) for row in rows) for col in range(len(header))]
|
|
519
|
+
left = {1} # solo «Contendiente» va alineada a la izquierda
|
|
520
|
+
lines = [
|
|
521
|
+
" "
|
|
522
|
+
+ " ".join(
|
|
523
|
+
f"{cell:<{widths[col]}}" if col in left else f"{cell:>{widths[col]}}"
|
|
524
|
+
for col, cell in enumerate(row)
|
|
525
|
+
)
|
|
526
|
+
for row in rows
|
|
527
|
+
]
|
|
528
|
+
lines.insert(1, " " + "-" * (sum(widths) + 3 * (len(widths) - 1)))
|
|
529
|
+
if any(data["fictitious_price"] for data in summary.values()):
|
|
530
|
+
lines.append(" (precios ficticios de demostración)")
|
|
531
|
+
return "\n".join(lines)
|
|
532
|
+
|
|
533
|
+
|
|
534
|
+
if __name__ == "__main__":
|
|
535
|
+
raise SystemExit(main())
|
modelduel/demo.py
ADDED
|
@@ -0,0 +1,26 @@
|
|
|
1
|
+
"""Ejemplos incluidos en el paquete: ``modelduel demo`` funciona nada más instalarlo."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from pathlib import Path
|
|
6
|
+
|
|
7
|
+
from modelduel.tasks import TaskError
|
|
8
|
+
|
|
9
|
+
# Contendientes ficticios con respuestas grabadas: sin claves y sin coste.
|
|
10
|
+
DEMO_MODELS = ("replay:alfa", "replay:beta", "replay:gamma")
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
def examples_dir() -> Path:
|
|
14
|
+
"""Carpeta con ``tasks/`` y ``replays/``.
|
|
15
|
+
|
|
16
|
+
Instalado desde PyPI están dentro del paquete (``modelduel/examples``); en un clon del
|
|
17
|
+
repositorio (también con ``pip install -e .``) están en ``examples/`` junto a ``src/``.
|
|
18
|
+
"""
|
|
19
|
+
here = Path(__file__).resolve().parent
|
|
20
|
+
for candidate in (here / "examples", here.parents[1] / "examples"):
|
|
21
|
+
if (candidate / "tasks").is_dir() and (candidate / "replays").is_dir():
|
|
22
|
+
return candidate
|
|
23
|
+
raise TaskError(
|
|
24
|
+
"no encuentro los ejemplos de modelduel. Reinstálalo o clona "
|
|
25
|
+
"https://github.com/BertMarti/modelduel."
|
|
26
|
+
)
|