modelduel 0.6.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (48) hide show
  1. modelduel/__init__.py +3 -0
  2. modelduel/__main__.py +3 -0
  3. modelduel/cli.py +535 -0
  4. modelduel/demo.py +26 -0
  5. modelduel/duel.py +169 -0
  6. modelduel/examples/replays/alfa/merge_intervals.md +21 -0
  7. modelduel/examples/replays/alfa/parse_duration.md +28 -0
  8. modelduel/examples/replays/alfa/slugify.md +17 -0
  9. modelduel/examples/replays/beta/merge_intervals.md +37 -0
  10. modelduel/examples/replays/beta/parse_duration.md +43 -0
  11. modelduel/examples/replays/beta/slugify.md +33 -0
  12. modelduel/examples/replays/gamma/merge_intervals.md +18 -0
  13. modelduel/examples/replays/gamma/parse_duration.md +21 -0
  14. modelduel/examples/replays/gamma/slugify.md +17 -0
  15. modelduel/examples/tasks/merge_intervals/meta.toml +3 -0
  16. modelduel/examples/tasks/merge_intervals/task.md +30 -0
  17. modelduel/examples/tasks/merge_intervals/test_task.py +41 -0
  18. modelduel/examples/tasks/parse_duration/meta.toml +3 -0
  19. modelduel/examples/tasks/parse_duration/task.md +27 -0
  20. modelduel/examples/tasks/parse_duration/test_task.py +42 -0
  21. modelduel/examples/tasks/slugify/meta.toml +3 -0
  22. modelduel/examples/tasks/slugify/task.md +24 -0
  23. modelduel/examples/tasks/slugify/test_task.py +38 -0
  24. modelduel/extract.py +51 -0
  25. modelduel/leaderboard.py +238 -0
  26. modelduel/pricing.py +95 -0
  27. modelduel/providers/__init__.py +42 -0
  28. modelduel/providers/base.py +280 -0
  29. modelduel/providers/gemini.py +91 -0
  30. modelduel/providers/openai_compat.py +127 -0
  31. modelduel/providers/replay.py +76 -0
  32. modelduel/report/__init__.py +37 -0
  33. modelduel/report/html.py +499 -0
  34. modelduel/report/leaderboard.html +105 -0
  35. modelduel/report/league.html +64 -0
  36. modelduel/report/league.py +293 -0
  37. modelduel/report/markdown.py +198 -0
  38. modelduel/report/style.css +223 -0
  39. modelduel/report/template.html +74 -0
  40. modelduel/results.py +173 -0
  41. modelduel/resume.py +89 -0
  42. modelduel/runner.py +372 -0
  43. modelduel/tasks.py +157 -0
  44. modelduel-0.6.0.dist-info/METADATA +277 -0
  45. modelduel-0.6.0.dist-info/RECORD +48 -0
  46. modelduel-0.6.0.dist-info/WHEEL +4 -0
  47. modelduel-0.6.0.dist-info/entry_points.txt +2 -0
  48. modelduel-0.6.0.dist-info/licenses/LICENSE +21 -0
modelduel/__init__.py ADDED
@@ -0,0 +1,3 @@
1
+ """modelduel: dos modelos, una tarea, los mismos tests."""
2
+
3
+ __version__ = "0.6.0"
modelduel/__main__.py ADDED
@@ -0,0 +1,3 @@
1
+ from modelduel.cli import main
2
+
3
+ raise SystemExit(main())
modelduel/cli.py ADDED
@@ -0,0 +1,535 @@
1
+ """Interfaz de línea de órdenes de modelduel."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import argparse
6
+ import math
7
+ import re
8
+ import shutil
9
+ import sys
10
+ from pathlib import Path
11
+
12
+ from modelduel import __version__
13
+ from modelduel.demo import DEMO_MODELS, examples_dir
14
+ from modelduel.duel import run_duel
15
+ from modelduel.leaderboard import build_leaderboard
16
+ from modelduel.pricing import PricingError, load_prices
17
+ from modelduel.providers import ProviderError, get_provider
18
+ from modelduel.providers.base import DEFAULT_RETRIES
19
+ from modelduel.report import write_markdown, write_report
20
+ from modelduel.report.html import fmt_cost, fmt_int, fmt_seconds, plural
21
+ from modelduel.results import (
22
+ ALL_SIDES,
23
+ MAX_CONTENDERS,
24
+ MIN_CONTENDERS,
25
+ ResultsError,
26
+ load_results,
27
+ rank_sides,
28
+ save_results,
29
+ summarize,
30
+ )
31
+ from modelduel.resume import ResumeError, plan_resume
32
+ from modelduel.runner import DEFAULT_TIMEOUT, RunnerError, ensure_pytest_available
33
+ from modelduel.tasks import TaskError, discover_tasks, is_task_dir
34
+
35
+ # Códigos de salida: 0 duelo completado (aunque los modelos fallen tests), 1 no se pudieron
36
+ # escribir los resultados, 2 error de uso o de configuración, 130 interrumpido con Ctrl+C.
37
+ EXIT_OK = 0
38
+ EXIT_ERROR = 1
39
+ EXIT_USAGE = 2
40
+ EXIT_INTERRUPTED = 130
41
+
42
+
43
+ FORMATS = ("html", "md")
44
+ _WRITERS = {"html": write_report, "md": write_markdown}
45
+
46
+
47
+ class OutputError(Exception):
48
+ """No se pudo escribir en la carpeta de salida."""
49
+
50
+
51
+ # argparse no trae traducciones: se traducen sus mensajes de error más habituales.
52
+ _ARGPARSE_ES = [
53
+ (r"the following arguments are required: (.+)", r"faltan argumentos obligatorios: \1"),
54
+ (
55
+ r"argument orden: invalid choice: (.+?) \(choose from (.+)\)",
56
+ r"orden no válida: \1 (elige entre \2)",
57
+ ),
58
+ (
59
+ r"argument (.+?): invalid choice: (.+?) \(choose from (.+)\)",
60
+ r"argumento \1: valor no válido: \2 (elige entre \3)",
61
+ ),
62
+ (r"argument (.+?): invalid \w+ value: (.+)", r"argumento \1: valor no válido: \2"),
63
+ (r"argument (.+?): expected one argument", r"argumento \1: necesita un valor"),
64
+ (r"unrecognized arguments: (.+)", r"argumentos no reconocidos: \1"),
65
+ (r"ambiguous option: (.+?) could match (.+)", r"opción ambigua: \1 puede ser \2"),
66
+ ]
67
+
68
+
69
+ def translate_argparse(message: str) -> str:
70
+ for pattern, replacement in _ARGPARSE_ES:
71
+ translated, n = re.subn(f"^{pattern}$", replacement, message)
72
+ if n:
73
+ return translated
74
+ return message
75
+
76
+
77
+ class _SpanishHelpFormatter(argparse.HelpFormatter):
78
+ def add_usage(self, usage, actions, groups, prefix=None):
79
+ super().add_usage(usage, actions, groups, prefix="uso: " if prefix is None else prefix)
80
+
81
+
82
+ class SpanishArgumentParser(argparse.ArgumentParser):
83
+ """ArgumentParser con ayuda y errores en español y código de salida ``EXIT_USAGE``."""
84
+
85
+ def __init__(self, *args, **kwargs) -> None:
86
+ kwargs["add_help"] = False
87
+ kwargs.setdefault("formatter_class", _SpanishHelpFormatter)
88
+ super().__init__(*args, **kwargs)
89
+ self._positionals.title = "argumentos posicionales"
90
+ self._optionals.title = "opciones"
91
+ self.add_argument("-h", "--help", action="help", help="muestra esta ayuda y sale")
92
+
93
+ def error(self, message: str): # type: ignore[override]
94
+ self.print_usage(sys.stderr)
95
+ self.exit(EXIT_USAGE, f"{self.prog}: error: {translate_argparse(message)}\n")
96
+
97
+
98
+ _FORMAT_HELP = "formatos del informe: html, md o html,md (html); results.json se guarda siempre"
99
+
100
+
101
+ def build_parser() -> argparse.ArgumentParser:
102
+ parser = SpanishArgumentParser(
103
+ prog="modelduel",
104
+ description="Dos modelos, una tarea, los mismos tests.",
105
+ epilog="Aviso: el código generado por los modelos se ejecuta en tu máquina.",
106
+ )
107
+ parser.add_argument(
108
+ "--version",
109
+ action="version",
110
+ version=f"modelduel {__version__}",
111
+ help="muestra la versión y sale",
112
+ )
113
+ sub = parser.add_subparsers(dest="command", required=True, metavar="orden")
114
+
115
+ run = sub.add_parser("run", help="enfrenta de 2 a 6 modelos y genera el informe")
116
+ run.add_argument("tasks", type=Path, help="carpeta de tareas o carpeta de una tarea")
117
+ run.add_argument("--a", metavar="PROVEEDOR:MODELO", help="contendiente A")
118
+ run.add_argument("--b", metavar="PROVEEDOR:MODELO", help="contendiente B")
119
+ run.add_argument(
120
+ "--model",
121
+ "-m",
122
+ action="append",
123
+ default=[],
124
+ metavar="PROVEEDOR:MODELO",
125
+ help=f"contendiente de una liga; repítelo ({MIN_CONTENDERS} a {MAX_CONTENDERS} en total, "
126
+ "contando --a y --b)",
127
+ )
128
+ run.add_argument("--runs", type=int, default=1, metavar="N", help="ejecuciones por tarea (1)")
129
+ run.add_argument(
130
+ "--timeout",
131
+ type=float,
132
+ default=DEFAULT_TIMEOUT,
133
+ metavar="S",
134
+ help=f"límite en segundos para los tests de cada respuesta ({DEFAULT_TIMEOUT:g})",
135
+ )
136
+ run.add_argument(
137
+ "--retries",
138
+ type=int,
139
+ default=DEFAULT_RETRIES,
140
+ metavar="N",
141
+ help=f"reintentos ante HTTP 429/5xx y cortes de conexión ({DEFAULT_RETRIES})",
142
+ )
143
+ run.add_argument(
144
+ "--resume",
145
+ action="store_true",
146
+ help="continúa el duelo de --out saltando los intentos ya terminados",
147
+ )
148
+ run.add_argument("--format", default="html", metavar="F", help=_FORMAT_HELP)
149
+ run.add_argument("--prices", type=Path, metavar="F.json", help="tabla de precios adicional")
150
+ run.add_argument(
151
+ "--replays", type=Path, metavar="DIR", help="carpeta de respuestas grabadas para replay"
152
+ )
153
+ run.add_argument("--out", type=Path, required=True, metavar="DIR", help="carpeta de salida")
154
+
155
+ demo = sub.add_parser(
156
+ "demo",
157
+ help="prueba modelduel sin claves, con los ejemplos incluidos y tres modelos ficticios",
158
+ )
159
+ demo.add_argument(
160
+ "--out",
161
+ type=Path,
162
+ default=Path("modelduel-demo"),
163
+ metavar="DIR",
164
+ help="carpeta de salida (modelduel-demo)",
165
+ )
166
+ demo.add_argument(
167
+ "--copy",
168
+ type=Path,
169
+ metavar="DIR",
170
+ help="en vez de ejecutar el duelo, copia las tareas y respuestas de ejemplo a DIR",
171
+ )
172
+
173
+ report = sub.add_parser(
174
+ "report", help="regenera el informe (HTML y/o Markdown) de results.json"
175
+ )
176
+ report.add_argument("results", type=Path, help="ruta a results.json")
177
+ report.add_argument(
178
+ "--format",
179
+ default="html",
180
+ metavar="F",
181
+ help="formatos del informe: html, md o html,md (html)",
182
+ )
183
+ report.add_argument("--out", type=Path, required=True, metavar="DIR", help="carpeta de salida")
184
+
185
+ board = sub.add_parser(
186
+ "leaderboard",
187
+ help="genera la página de clasificación a partir de una carpeta de results.json",
188
+ )
189
+ board.add_argument("results", type=Path, help="carpeta con los results.json (uno por duelo)")
190
+ board.add_argument("--out", type=Path, required=True, metavar="DIR", help="carpeta de salida")
191
+
192
+ lst = sub.add_parser("list-tasks", help="lista las tareas de una carpeta")
193
+ lst.add_argument("tasks", type=Path, help="carpeta de tareas")
194
+ return parser
195
+
196
+
197
+ def main(argv: list[str] | None = None) -> int:
198
+ for stream in (sys.stdout, sys.stderr):
199
+ if hasattr(stream, "reconfigure"):
200
+ stream.reconfigure(encoding="utf-8", errors="replace")
201
+ args = build_parser().parse_args(argv)
202
+ try:
203
+ if args.command == "run":
204
+ return cmd_run(args)
205
+ if args.command == "demo":
206
+ return cmd_demo(args)
207
+ if args.command == "report":
208
+ return cmd_report(args)
209
+ if args.command == "leaderboard":
210
+ return cmd_leaderboard(args)
211
+ return cmd_list_tasks(args)
212
+ except (TaskError, ProviderError, PricingError, ResultsError, RunnerError, ResumeError) as exc:
213
+ print(f"modelduel: error: {exc}", file=sys.stderr)
214
+ return EXIT_USAGE
215
+ except OutputError as exc:
216
+ print(f"modelduel: error: {exc}", file=sys.stderr)
217
+ return EXIT_ERROR
218
+ except KeyboardInterrupt:
219
+ print("\nmodelduel: interrumpido.", file=sys.stderr)
220
+ return EXIT_INTERRUPTED
221
+
222
+
223
+ def parse_formats(text: str) -> tuple[str, ...]:
224
+ """``--format html,md`` -> ``("html", "md")`` sin repetidos, o error de uso."""
225
+ names = [part.strip() for part in text.split(",")]
226
+ for name in names:
227
+ if name not in FORMATS:
228
+ raise TaskError(
229
+ f"--format: «{name}» no es un formato válido "
230
+ f"(elige entre {' y '.join(FORMATS)}; sepáralos con comas)."
231
+ )
232
+ return tuple(dict.fromkeys(names))
233
+
234
+
235
+ def collect_specs(args: argparse.Namespace) -> list[str]:
236
+ """Contendientes en orden: ``--a``, ``--b`` y después cada ``--model``."""
237
+ if args.b and not args.a:
238
+ raise TaskError("--b necesita también --a (o usa --model para cada contendiente).")
239
+ specs = [spec for spec in (args.a, args.b) if spec] + list(args.model)
240
+ if len(specs) < MIN_CONTENDERS:
241
+ raise TaskError(
242
+ f"hacen falta al menos {MIN_CONTENDERS} contendientes: usa --a y --b, "
243
+ "o --model varias veces."
244
+ )
245
+ if len(specs) > MAX_CONTENDERS:
246
+ raise TaskError(f"como máximo {MAX_CONTENDERS} contendientes (has puesto {len(specs)}).")
247
+ return specs
248
+
249
+
250
+ def _reject_duplicates(providers: dict) -> None:
251
+ seen: dict[str, str] = {}
252
+ for side, provider in providers.items():
253
+ if provider.spec in seen:
254
+ raise TaskError(
255
+ f"«{provider.spec}» aparece dos veces ({seen[provider.spec].upper()} y "
256
+ f"{side.upper()}). Para repetir un mismo modelo usa --runs."
257
+ )
258
+ seen[provider.spec] = side
259
+
260
+
261
+ def cmd_run(args: argparse.Namespace) -> int:
262
+ if args.runs < 1:
263
+ raise TaskError("--runs debe ser 1 o más.")
264
+ if not math.isfinite(args.timeout) or args.timeout <= 0:
265
+ raise TaskError("--timeout debe ser un número de segundos mayor que 0.")
266
+ if args.retries < 0:
267
+ raise TaskError("--retries debe ser 0 o más.")
268
+ formats = parse_formats(args.format)
269
+ specs = collect_specs(args)
270
+ tasks = discover_tasks(args.tasks)
271
+ prices = load_prices(args.prices)
272
+ # examples/tasks[/<tarea>] -> examples/replays
273
+ tasks_root = args.tasks.parent if is_task_dir(args.tasks) else args.tasks
274
+ replay_dirs = [
275
+ d
276
+ for d in (args.replays, tasks_root.parent / "replays", Path("examples/replays"))
277
+ if d is not None
278
+ ]
279
+
280
+ def on_retry(message: str) -> None:
281
+ print(f" ~~ {message}", flush=True)
282
+
283
+ providers = {
284
+ side: get_provider(spec, replay_dirs, retries=args.retries, on_retry=on_retry)
285
+ for side, spec in zip(ALL_SIDES, specs, strict=False)
286
+ }
287
+ _reject_duplicates(providers)
288
+ ensure_pytest_available()
289
+ # Antes de gastar llamadas a las APIs: la carpeta de salida tiene que poder crearse.
290
+ _prepare_out(args.out)
291
+ results_path = args.out / "results.json"
292
+ previous = _load_previous(results_path, args.resume)
293
+ reuse: dict = {}
294
+ warnings: list[str] = []
295
+ if previous is not None:
296
+ specs = {side: provider.spec for side, provider in providers.items()}
297
+ try:
298
+ reuse, warnings = plan_resume(previous, tasks, specs, args.runs, args.timeout)
299
+ except (KeyError, TypeError, ValueError, AttributeError) as exc:
300
+ raise ResumeError(
301
+ f"{results_path} está dañado o incompleto ({type(exc).__name__}: {exc}). "
302
+ "Bórralo o elige otra carpeta --out."
303
+ ) from exc
304
+
305
+ print(
306
+ f"modelduel {__version__} · {plural(len(tasks), 'tarea', 'tareas')} · "
307
+ f"{plural(args.runs, 'ejecución', 'ejecuciones')} por tarea"
308
+ )
309
+ for side, provider in providers.items():
310
+ print(f" {side.upper()} {provider.spec}")
311
+ print(" Aviso: el código de los modelos se ejecuta en esta máquina (temporal + límite).")
312
+ if args.resume:
313
+ total = len(tasks) * args.runs * len(providers)
314
+ if previous is None:
315
+ print(f" Reanudar: no hay {results_path}; se empieza de cero.")
316
+ else:
317
+ print(
318
+ f" Reanudando: {plural(len(reuse), 'intento', 'intentos')} ya hecho"
319
+ f"{'' if len(reuse) == 1 else 's'} de {total}; "
320
+ f"quedan {total - len(reuse)}."
321
+ )
322
+ for warning in warnings:
323
+ print(f" Aviso: {warning}")
324
+ print()
325
+
326
+ latest: dict = {}
327
+ save_failed = False
328
+
329
+ def on_update(current: dict) -> None:
330
+ """Guarda ``results.json`` tras cada intento sin tumbar el duelo si falla el disco."""
331
+ nonlocal save_failed
332
+ latest["results"] = current
333
+ try:
334
+ save_results(current, results_path)
335
+ except OSError as exc:
336
+ if not save_failed:
337
+ save_failed = True
338
+ print(f" Aviso: no se pudo guardar {results_path}: {exc.strerror or exc}.")
339
+
340
+ try:
341
+ results = run_duel(
342
+ tasks,
343
+ providers,
344
+ prices,
345
+ runs=args.runs,
346
+ timeout=args.timeout,
347
+ progress=lambda line: print(line, flush=True),
348
+ on_update=on_update,
349
+ reuse=reuse,
350
+ created_at=(previous or {}).get("created_at"),
351
+ )
352
+ except KeyboardInterrupt:
353
+ return _interrupted(latest.get("results"), args.out, formats)
354
+ report_paths = _write_outputs(results, args.out, formats, with_json=True)
355
+ print()
356
+ print(scoreboard(results))
357
+ print()
358
+ print(f" Resultados {results_path}")
359
+ for path in report_paths:
360
+ print(f" Informe {path}")
361
+ return 0
362
+
363
+
364
+ def _load_previous(results_path: Path, resume: bool) -> dict | None:
365
+ """``results.json`` previo de la carpeta de salida (o ``None``).
366
+
367
+ Sin ``--resume`` solo importa para no machacar por descuido un duelo a medias, que puede
368
+ haber costado dinero.
369
+ """
370
+ if not results_path.is_file():
371
+ return None
372
+ if resume:
373
+ return load_results(results_path)
374
+ try:
375
+ previous = load_results(results_path)
376
+ except ResultsError:
377
+ return None
378
+ if previous.get("status") == "in_progress":
379
+ raise ResumeError(
380
+ f"{results_path} es un duelo incompleto. Continúalo con --resume o bórralo, "
381
+ "o elige otra carpeta --out."
382
+ )
383
+ return None
384
+
385
+
386
+ def _interrupted(results: dict | None, out: Path, formats: tuple[str, ...]) -> int:
387
+ """Ctrl+C: lo hecho ya está en ``results.json``; se deja también el informe parcial."""
388
+ print("\nmodelduel: interrumpido.", file=sys.stderr)
389
+ if results is not None:
390
+ try:
391
+ results["summary"] = summarize(results)
392
+ for name in formats:
393
+ _WRITERS[name](results, out)
394
+ except (OSError, KeyError, TypeError, ValueError, AttributeError, OverflowError):
395
+ pass
396
+ print(
397
+ f"Lo hecho hasta ahora está en {out / 'results.json'}. "
398
+ "Continúa con la misma orden añadiendo --resume.",
399
+ file=sys.stderr,
400
+ )
401
+ return EXIT_INTERRUPTED
402
+
403
+
404
+ def cmd_demo(args: argparse.Namespace) -> int:
405
+ """``modelduel demo``: liga de tres contendientes ficticios sobre los ejemplos incluidos."""
406
+ examples = examples_dir()
407
+ if args.copy:
408
+ target = args.copy
409
+ if target.exists() and any(target.iterdir()):
410
+ raise TaskError(f"{target} ya existe y no está vacía: elige otra carpeta.")
411
+ try:
412
+ shutil.copytree(
413
+ examples, target, dirs_exist_ok=True, ignore=shutil.ignore_patterns("__pycache__")
414
+ )
415
+ except OSError as exc:
416
+ raise OutputError(f"no se pudo copiar a {target}: {exc.strerror or exc}.") from exc
417
+ print(f"Ejemplos copiados en {target}:")
418
+ print(f" tareas {target / 'tasks'}")
419
+ print(f" respuestas {target / 'replays'}")
420
+ print("Pruébalos con:")
421
+ print(f" modelduel run {target / 'tasks'} --model replay:alfa --model replay:beta")
422
+ return 0
423
+ run_args = argparse.Namespace(
424
+ tasks=examples / "tasks",
425
+ a=None,
426
+ b=None,
427
+ model=list(DEMO_MODELS),
428
+ runs=1,
429
+ timeout=DEFAULT_TIMEOUT,
430
+ retries=0,
431
+ resume=False,
432
+ prices=None,
433
+ replays=examples / "replays",
434
+ format="html",
435
+ out=args.out,
436
+ )
437
+ code = cmd_run(run_args)
438
+ if code == 0:
439
+ print()
440
+ print(" Son respuestas grabadas y precios ficticios. Para un duelo real, mira la guía:")
441
+ print(" https://github.com/BertMarti/modelduel/blob/main/docs/USO.md")
442
+ return code
443
+
444
+
445
+ def cmd_report(args: argparse.Namespace) -> int:
446
+ formats = parse_formats(args.format)
447
+ results = load_results(args.results)
448
+ _prepare_out(args.out)
449
+ try:
450
+ paths = _write_outputs(results, args.out, formats, with_json=False)
451
+ except (KeyError, TypeError, ValueError, AttributeError, OverflowError) as exc:
452
+ raise ResultsError(
453
+ f"{args.results} no parece un results.json de modelduel ({type(exc).__name__}: {exc})."
454
+ ) from exc
455
+ for path in paths:
456
+ print(f"Informe regenerado: {path}")
457
+ return 0
458
+
459
+
460
+ def cmd_leaderboard(args: argparse.Namespace) -> int:
461
+ try:
462
+ index = build_leaderboard(args.results, args.out)
463
+ except OSError as exc:
464
+ raise OutputError(f"no se pudo escribir en {args.out}: {exc.strerror or exc}.") from exc
465
+ except (KeyError, TypeError, ValueError, AttributeError, OverflowError) as exc:
466
+ raise ResultsError(
467
+ f"hay un results.json que no parece de modelduel ({type(exc).__name__}: {exc})."
468
+ ) from exc
469
+ print(f"Clasificación generada: {index}")
470
+ return 0
471
+
472
+
473
+ def _prepare_out(out: Path) -> None:
474
+ try:
475
+ out.mkdir(parents=True, exist_ok=True)
476
+ except OSError as exc:
477
+ raise OutputError(f"no se pudo escribir en {out}: {exc.strerror or exc}.") from exc
478
+
479
+
480
+ def _write_outputs(
481
+ results: dict, out: Path, formats: tuple[str, ...], with_json: bool
482
+ ) -> list[Path]:
483
+ try:
484
+ if with_json:
485
+ save_results(results, out / "results.json")
486
+ return [_WRITERS[name](results, out) for name in formats]
487
+ except OSError as exc:
488
+ raise OutputError(f"no se pudo escribir en {out}: {exc.strerror or exc}.") from exc
489
+
490
+
491
+ def cmd_list_tasks(args: argparse.Namespace) -> int:
492
+ tasks = discover_tasks(args.tasks)
493
+ width = max(len(t.id) for t in tasks)
494
+ for task in tasks:
495
+ extra = f" [{task.difficulty}]" if task.difficulty else ""
496
+ print(f"{task.id:<{width}} {task.expected_tests:>2} tests {task.title}{extra}")
497
+ return 0
498
+
499
+
500
+ def scoreboard(results: dict) -> str:
501
+ """Clasificación en texto: una fila por contendiente, de mejor a peor."""
502
+ summary = results["summary"]
503
+ header = ("#", "Contendiente", "Tareas", "Tests", "Tiempo", "Tokens (ent/sal)", "Coste")
504
+ rows = [header]
505
+ for rank, side in rank_sides(summary):
506
+ data = summary[side]
507
+ rows.append(
508
+ (
509
+ str(rank),
510
+ f"{side.upper()} {results['contenders'][side]['spec']}",
511
+ f"{data['tasks_solved']}/{data['tasks_total']}",
512
+ f"{data['tests_passed']}/{data['tests_total']}",
513
+ fmt_seconds(data["latency_s"]),
514
+ f"{fmt_int(data['input_tokens'])}/{fmt_int(data['output_tokens'])}",
515
+ fmt_cost(data["cost"], data["currency"]),
516
+ )
517
+ )
518
+ widths = [max(len(row[col]) for row in rows) for col in range(len(header))]
519
+ left = {1} # solo «Contendiente» va alineada a la izquierda
520
+ lines = [
521
+ " "
522
+ + " ".join(
523
+ f"{cell:<{widths[col]}}" if col in left else f"{cell:>{widths[col]}}"
524
+ for col, cell in enumerate(row)
525
+ )
526
+ for row in rows
527
+ ]
528
+ lines.insert(1, " " + "-" * (sum(widths) + 3 * (len(widths) - 1)))
529
+ if any(data["fictitious_price"] for data in summary.values()):
530
+ lines.append(" (precios ficticios de demostración)")
531
+ return "\n".join(lines)
532
+
533
+
534
+ if __name__ == "__main__":
535
+ raise SystemExit(main())
modelduel/demo.py ADDED
@@ -0,0 +1,26 @@
1
+ """Ejemplos incluidos en el paquete: ``modelduel demo`` funciona nada más instalarlo."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from pathlib import Path
6
+
7
+ from modelduel.tasks import TaskError
8
+
9
+ # Contendientes ficticios con respuestas grabadas: sin claves y sin coste.
10
+ DEMO_MODELS = ("replay:alfa", "replay:beta", "replay:gamma")
11
+
12
+
13
+ def examples_dir() -> Path:
14
+ """Carpeta con ``tasks/`` y ``replays/``.
15
+
16
+ Instalado desde PyPI están dentro del paquete (``modelduel/examples``); en un clon del
17
+ repositorio (también con ``pip install -e .``) están en ``examples/`` junto a ``src/``.
18
+ """
19
+ here = Path(__file__).resolve().parent
20
+ for candidate in (here / "examples", here.parents[1] / "examples"):
21
+ if (candidate / "tasks").is_dir() and (candidate / "replays").is_dir():
22
+ return candidate
23
+ raise TaskError(
24
+ "no encuentro los ejemplos de modelduel. Reinstálalo o clona "
25
+ "https://github.com/BertMarti/modelduel."
26
+ )