troy-cli 0.3.0__tar.gz → 0.5.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {troy_cli-0.3.0 → troy_cli-0.5.0}/PKG-INFO +20 -1
- {troy_cli-0.3.0 → troy_cli-0.5.0}/README.md +19 -0
- {troy_cli-0.3.0 → troy_cli-0.5.0}/pyproject.toml +1 -1
- {troy_cli-0.3.0 → troy_cli-0.5.0}/src/troy/__init__.py +1 -1
- {troy_cli-0.3.0 → troy_cli-0.5.0}/src/troy/cli.py +273 -9
- {troy_cli-0.3.0 → troy_cli-0.5.0}/src/troy/data.py +118 -0
- troy_cli-0.5.0/src/troy/export.py +242 -0
- troy_cli-0.5.0/src/troy/mesh.py +427 -0
- troy_cli-0.5.0/src/troy/synth.py +424 -0
- {troy_cli-0.3.0 → troy_cli-0.5.0}/tests/test_data.py +47 -0
- troy_cli-0.5.0/tests/test_export.py +54 -0
- troy_cli-0.5.0/tests/test_mesh.py +123 -0
- troy_cli-0.5.0/tests/test_synth.py +82 -0
- troy_cli-0.5.0/tests/test_synth_tools.py +60 -0
- troy_cli-0.3.0/src/troy/export.py +0 -81
- {troy_cli-0.3.0 → troy_cli-0.5.0}/.gitignore +0 -0
- {troy_cli-0.3.0 → troy_cli-0.5.0}/src/troy/chat.py +0 -0
- {troy_cli-0.3.0 → troy_cli-0.5.0}/src/troy/config.py +0 -0
- {troy_cli-0.3.0 → troy_cli-0.5.0}/src/troy/evaluate.py +0 -0
- {troy_cli-0.3.0 → troy_cli-0.5.0}/src/troy/hardware.py +0 -0
- {troy_cli-0.3.0 → troy_cli-0.5.0}/src/troy/push.py +0 -0
- {troy_cli-0.3.0 → troy_cli-0.5.0}/src/troy/serve.py +0 -0
- {troy_cli-0.3.0 → troy_cli-0.5.0}/src/troy/templates.py +0 -0
- {troy_cli-0.3.0 → troy_cli-0.5.0}/src/troy/train_dpo.py +0 -0
- {troy_cli-0.3.0 → troy_cli-0.5.0}/src/troy/train_orpo.py +0 -0
- {troy_cli-0.3.0 → troy_cli-0.5.0}/src/troy/train_sft.py +0 -0
- {troy_cli-0.3.0 → troy_cli-0.5.0}/src/troy/train_vision.py +0 -0
- {troy_cli-0.3.0 → troy_cli-0.5.0}/tests/test_config.py +0 -0
- {troy_cli-0.3.0 → troy_cli-0.5.0}/tests/test_hardware.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: troy-cli
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.5.0
|
|
4
4
|
Summary: Fine-tune LLMs on your MacBook with one YAML file. Built for Apple Silicon.
|
|
5
5
|
Author: Troy
|
|
6
6
|
License: Apache-2.0
|
|
@@ -88,6 +88,25 @@ output: ./output
|
|
|
88
88
|
| `troy export` | Fuse the adapter; export MLX or GGUF |
|
|
89
89
|
| `troy push` | Upload adapter or fused model to the Hugging Face Hub |
|
|
90
90
|
| `troy data inspect` | Dataset stats and format detection |
|
|
91
|
+
| `troy data synth` | Synthesize a dataset with a local teacher model |
|
|
92
|
+
| `troy mesh serve` | Coordinate a LAN mesh: iPhones and Macs generate the dataset for you |
|
|
93
|
+
| `troy mesh join` | Join a mesh as a worker from any Mac |
|
|
94
|
+
|
|
95
|
+
## The mesh: your idle iPhones generate the dataset
|
|
96
|
+
|
|
97
|
+
`troy data synth` runs the teacher on one Mac. `troy mesh` farms the same job
|
|
98
|
+
out to every Apple device on your network — the coordinator mints prompts and
|
|
99
|
+
validates results (identical parsing to local synth), workers run the teacher:
|
|
100
|
+
|
|
101
|
+
```bash
|
|
102
|
+
troy mesh serve --from ./docs --n 500 # Mac: prints URL + token
|
|
103
|
+
troy mesh join http://mac:8765 --token … # any other Mac
|
|
104
|
+
# iPhones: the TroyWorker app (examples/ios/TroyWorker)
|
|
105
|
+
```
|
|
106
|
+
|
|
107
|
+
Workers can drop out at any time — leased work requeues automatically, and
|
|
108
|
+
duplicates are rejected centrally. The output is a normal `train.jsonl`:
|
|
109
|
+
validate it, then `troy train`.
|
|
91
110
|
|
|
92
111
|
## What Troy can train on your Mac
|
|
93
112
|
|
|
@@ -71,6 +71,25 @@ output: ./output
|
|
|
71
71
|
| `troy export` | Fuse the adapter; export MLX or GGUF |
|
|
72
72
|
| `troy push` | Upload adapter or fused model to the Hugging Face Hub |
|
|
73
73
|
| `troy data inspect` | Dataset stats and format detection |
|
|
74
|
+
| `troy data synth` | Synthesize a dataset with a local teacher model |
|
|
75
|
+
| `troy mesh serve` | Coordinate a LAN mesh: iPhones and Macs generate the dataset for you |
|
|
76
|
+
| `troy mesh join` | Join a mesh as a worker from any Mac |
|
|
77
|
+
|
|
78
|
+
## The mesh: your idle iPhones generate the dataset
|
|
79
|
+
|
|
80
|
+
`troy data synth` runs the teacher on one Mac. `troy mesh` farms the same job
|
|
81
|
+
out to every Apple device on your network — the coordinator mints prompts and
|
|
82
|
+
validates results (identical parsing to local synth), workers run the teacher:
|
|
83
|
+
|
|
84
|
+
```bash
|
|
85
|
+
troy mesh serve --from ./docs --n 500 # Mac: prints URL + token
|
|
86
|
+
troy mesh join http://mac:8765 --token … # any other Mac
|
|
87
|
+
# iPhones: the TroyWorker app (examples/ios/TroyWorker)
|
|
88
|
+
```
|
|
89
|
+
|
|
90
|
+
Workers can drop out at any time — leased work requeues automatically, and
|
|
91
|
+
duplicates are rejected centrally. The output is a normal `train.jsonl`:
|
|
92
|
+
validate it, then `troy train`.
|
|
74
93
|
|
|
75
94
|
## What Troy can train on your Mac
|
|
76
95
|
|
|
@@ -258,14 +258,14 @@ def serve(
|
|
|
258
258
|
@app.command()
|
|
259
259
|
def export(
|
|
260
260
|
config: Path = typer.Option(Path("troy.yaml"), "--config", "-c", help="Config file."),
|
|
261
|
-
fmt: str = typer.Option("mlx", "--format", "-f", help="Export format: mlx or
|
|
261
|
+
fmt: str = typer.Option("mlx", "--format", "-f", help="Export format: mlx, gguf, or ios."),
|
|
262
262
|
save_path: Optional[Path] = typer.Option(None, help="Output directory (default: <output>/fused)."),
|
|
263
263
|
dequantize: bool = typer.Option(False, help="Dequantize when fusing a quantized base."),
|
|
264
264
|
) -> None:
|
|
265
265
|
"""Merge the trained adapter into the base model and export it."""
|
|
266
266
|
_require_apple_silicon()
|
|
267
|
-
if fmt not in ("mlx", "gguf"):
|
|
268
|
-
console.print("[red]--format must be `mlx` or `
|
|
267
|
+
if fmt not in ("mlx", "gguf", "ios"):
|
|
268
|
+
console.print("[red]--format must be `mlx`, `gguf`, or `ios`.[/red]")
|
|
269
269
|
raise typer.Exit(1)
|
|
270
270
|
from .config import load_config
|
|
271
271
|
from .export import run_export
|
|
@@ -319,15 +319,19 @@ def push(
|
|
|
319
319
|
run_push(folder, repo, private=not public)
|
|
320
320
|
|
|
321
321
|
|
|
322
|
-
|
|
323
|
-
|
|
324
|
-
|
|
322
|
+
data_app = typer.Typer(
|
|
323
|
+
name="data",
|
|
324
|
+
help="Inspect, validate, and synthesize datasets.",
|
|
325
|
+
no_args_is_help=True,
|
|
326
|
+
)
|
|
327
|
+
app.add_typer(data_app)
|
|
328
|
+
|
|
329
|
+
|
|
330
|
+
@data_app.command()
|
|
331
|
+
def inspect(
|
|
325
332
|
path: Path = typer.Argument(help="Dataset file (.jsonl, .json, .csv)."),
|
|
326
333
|
) -> None:
|
|
327
334
|
"""Inspect a dataset: record count, detected format, sizes."""
|
|
328
|
-
if action != "inspect":
|
|
329
|
-
console.print("[red]Only `troy data inspect <path>` is supported.[/red]")
|
|
330
|
-
raise typer.Exit(1)
|
|
331
335
|
from .data import inspect_stats
|
|
332
336
|
|
|
333
337
|
stats = inspect_stats(str(path))
|
|
@@ -341,5 +345,265 @@ def data(
|
|
|
341
345
|
console.print(table)
|
|
342
346
|
|
|
343
347
|
|
|
348
|
+
@data_app.command()
|
|
349
|
+
def validate(
|
|
350
|
+
path: Path = typer.Argument(help="Dataset file (.jsonl, .json, .csv)."),
|
|
351
|
+
) -> None:
|
|
352
|
+
"""Lint a dataset: broken records, mixed formats, empty fields, duplicates."""
|
|
353
|
+
from .data import validate_records
|
|
354
|
+
|
|
355
|
+
report = validate_records(str(path))
|
|
356
|
+
console.print(
|
|
357
|
+
f"{report['records']} records, format: [bold]{report['format']}[/bold]"
|
|
358
|
+
)
|
|
359
|
+
if not report["issues"]:
|
|
360
|
+
console.print("[green]No issues found.[/green]")
|
|
361
|
+
return
|
|
362
|
+
for issue in report["issues"]:
|
|
363
|
+
console.print(f" [yellow]•[/yellow] {issue}")
|
|
364
|
+
if report["truncated"]:
|
|
365
|
+
console.print(" [dim]... more issues not shown[/dim]")
|
|
366
|
+
console.print(f"[red]{len(report['issues'])}{'+' if report['truncated'] else ''} issue(s).[/red]")
|
|
367
|
+
raise typer.Exit(1)
|
|
368
|
+
|
|
369
|
+
|
|
370
|
+
@data_app.command()
|
|
371
|
+
def synth(
|
|
372
|
+
source: Optional[Path] = typer.Option(
|
|
373
|
+
None, "--from", help="Ground examples in a file or folder of docs/code."
|
|
374
|
+
),
|
|
375
|
+
seed: Optional[str] = typer.Option(
|
|
376
|
+
None, "--seed", help='Task description, e.g. "customer support bot for Acme".'
|
|
377
|
+
),
|
|
378
|
+
n: int = typer.Option(100, "--n", help="Number of examples to generate."),
|
|
379
|
+
fmt: str = typer.Option(
|
|
380
|
+
"chat", "--format", "-f",
|
|
381
|
+
help="Output format: chat (SFT), preference (DPO/ORPO), or tools (tool calling).",
|
|
382
|
+
),
|
|
383
|
+
tools: Optional[Path] = typer.Option(
|
|
384
|
+
None, "--tools",
|
|
385
|
+
help="For -f tools: JSON file with the tool schemas (OpenAI function format).",
|
|
386
|
+
),
|
|
387
|
+
no_think: bool = typer.Option(
|
|
388
|
+
False, "--no-think",
|
|
389
|
+
help="For -f tools: omit the <think> reasoning traces.",
|
|
390
|
+
),
|
|
391
|
+
teacher: str = typer.Option(
|
|
392
|
+
"auto", help="Teacher model (auto = sized to this Mac's memory)."
|
|
393
|
+
),
|
|
394
|
+
out: Optional[Path] = typer.Option(
|
|
395
|
+
None, "--out", "-o",
|
|
396
|
+
help="Output file (default: data/train.jsonl or data/preferences.jsonl).",
|
|
397
|
+
),
|
|
398
|
+
max_tokens: int = typer.Option(2048, help="Max tokens per teacher call."),
|
|
399
|
+
temperature: float = typer.Option(0.8),
|
|
400
|
+
) -> None:
|
|
401
|
+
"""Synthesize a training dataset with a local teacher model."""
|
|
402
|
+
_require_apple_silicon()
|
|
403
|
+
if source is None and seed is None:
|
|
404
|
+
console.print(
|
|
405
|
+
'[red]Give the teacher something to work from:[/red] '
|
|
406
|
+
'--from ./docs and/or --seed "task description".'
|
|
407
|
+
)
|
|
408
|
+
raise typer.Exit(1)
|
|
409
|
+
if fmt not in ("chat", "preference", "tools"):
|
|
410
|
+
console.print("[red]--format must be `chat`, `preference`, or `tools`.[/red]")
|
|
411
|
+
raise typer.Exit(1)
|
|
412
|
+
if fmt == "tools":
|
|
413
|
+
if tools is None or not tools.exists():
|
|
414
|
+
console.print(
|
|
415
|
+
"[red]-f tools needs --tools schemas.json[/red] — a JSON list of "
|
|
416
|
+
"OpenAI-style function specs the assistant can call."
|
|
417
|
+
)
|
|
418
|
+
raise typer.Exit(1)
|
|
419
|
+
if seed is None:
|
|
420
|
+
console.print(
|
|
421
|
+
'[red]-f tools needs --seed[/red] — it becomes the system prompt, '
|
|
422
|
+
'e.g. "travel planning assistant that books nothing".'
|
|
423
|
+
)
|
|
424
|
+
raise typer.Exit(1)
|
|
425
|
+
|
|
426
|
+
from .synth import pick_teacher, run_synth
|
|
427
|
+
|
|
428
|
+
if teacher == "auto":
|
|
429
|
+
teacher = pick_teacher()
|
|
430
|
+
console.print(f"Teacher: [bold]{teacher}[/bold] (picked for this Mac's memory)")
|
|
431
|
+
out = out or Path("data") / ("preferences.jsonl" if fmt == "preference" else "train.jsonl")
|
|
432
|
+
if out.exists():
|
|
433
|
+
console.print(f"[red]{out} already exists[/red] — pass -o to write elsewhere.")
|
|
434
|
+
raise typer.Exit(1)
|
|
435
|
+
|
|
436
|
+
stats = run_synth(
|
|
437
|
+
n=n, out_path=out, fmt=fmt, teacher=teacher,
|
|
438
|
+
seed_task=seed, source=source,
|
|
439
|
+
max_tokens=max_tokens, temperature=temperature,
|
|
440
|
+
tools_path=tools, think=not no_think,
|
|
441
|
+
)
|
|
442
|
+
console.print(
|
|
443
|
+
f"\nWrote [bold]{stats['records']}[/bold] examples to [bold]{stats['out']}[/bold] "
|
|
444
|
+
f"({stats['teacher_calls']} teacher calls)"
|
|
445
|
+
)
|
|
446
|
+
if stats["records"] < stats["requested"]:
|
|
447
|
+
console.print(
|
|
448
|
+
f"[yellow]Stopped at {stats['records']}/{stats['requested']} — "
|
|
449
|
+
"try a larger --teacher, higher --max-tokens, or more source material.[/yellow]"
|
|
450
|
+
)
|
|
451
|
+
console.print(
|
|
452
|
+
"Review the data before training — spot-check a dozen examples, then: "
|
|
453
|
+
"[bold]troy data validate " + str(out) + "[/bold] and [bold]troy train[/bold]."
|
|
454
|
+
)
|
|
455
|
+
|
|
456
|
+
|
|
457
|
+
mesh_app = typer.Typer(
|
|
458
|
+
name="mesh",
|
|
459
|
+
help="Distribute data synthesis across devices on your LAN.",
|
|
460
|
+
no_args_is_help=True,
|
|
461
|
+
)
|
|
462
|
+
app.add_typer(mesh_app)
|
|
463
|
+
|
|
464
|
+
|
|
465
|
+
@mesh_app.command("serve")
|
|
466
|
+
def mesh_serve(
|
|
467
|
+
source: Optional[Path] = typer.Option(
|
|
468
|
+
None, "--from", help="Ground examples in a file or folder of docs/code."
|
|
469
|
+
),
|
|
470
|
+
seed: Optional[str] = typer.Option(
|
|
471
|
+
None, "--seed", help='Task description, e.g. "customer support bot for Acme".'
|
|
472
|
+
),
|
|
473
|
+
n: int = typer.Option(100, "--n", help="Number of examples to generate."),
|
|
474
|
+
fmt: str = typer.Option(
|
|
475
|
+
"chat", "--format", "-f",
|
|
476
|
+
help="Output format: chat (SFT), preference (DPO/ORPO), or tools (tool calling).",
|
|
477
|
+
),
|
|
478
|
+
tools: Optional[Path] = typer.Option(
|
|
479
|
+
None, "--tools",
|
|
480
|
+
help="For -f tools: JSON file with the tool schemas (OpenAI function format).",
|
|
481
|
+
),
|
|
482
|
+
no_think: bool = typer.Option(
|
|
483
|
+
False, "--no-think",
|
|
484
|
+
help="For -f tools: omit the <think> reasoning traces.",
|
|
485
|
+
),
|
|
486
|
+
out: Optional[Path] = typer.Option(
|
|
487
|
+
None, "--out", "-o",
|
|
488
|
+
help="Output file (default: data/train.jsonl or data/preferences.jsonl).",
|
|
489
|
+
),
|
|
490
|
+
max_tokens: int = typer.Option(2048, help="Max tokens per teacher call."),
|
|
491
|
+
temperature: float = typer.Option(0.8),
|
|
492
|
+
host: str = typer.Option("0.0.0.0", help="Interface to bind."),
|
|
493
|
+
port: int = typer.Option(8765),
|
|
494
|
+
token: Optional[str] = typer.Option(
|
|
495
|
+
None, help="Shared worker token (default: generated at startup)."
|
|
496
|
+
),
|
|
497
|
+
lease_timeout: float = typer.Option(
|
|
498
|
+
300.0, help="Seconds before an unanswered work item is requeued."
|
|
499
|
+
),
|
|
500
|
+
linger: float = typer.Option(
|
|
501
|
+
30.0, help="Seconds to wait for in-flight results after the target is hit."
|
|
502
|
+
),
|
|
503
|
+
) -> None:
|
|
504
|
+
"""Coordinate a mesh: serve synth work to iPhones and Macs on your LAN.
|
|
505
|
+
|
|
506
|
+
Workers run the teacher model; this machine only mints prompts and
|
|
507
|
+
validates results, so it can be any Mac (the model never loads here).
|
|
508
|
+
"""
|
|
509
|
+
if source is None and seed is None:
|
|
510
|
+
console.print(
|
|
511
|
+
'[red]Give the workers something to work from:[/red] '
|
|
512
|
+
'--from ./docs and/or --seed "task description".'
|
|
513
|
+
)
|
|
514
|
+
raise typer.Exit(1)
|
|
515
|
+
if fmt not in ("chat", "preference", "tools"):
|
|
516
|
+
console.print("[red]--format must be `chat`, `preference`, or `tools`.[/red]")
|
|
517
|
+
raise typer.Exit(1)
|
|
518
|
+
if fmt == "tools":
|
|
519
|
+
if tools is None or not tools.exists():
|
|
520
|
+
console.print(
|
|
521
|
+
"[red]-f tools needs --tools schemas.json[/red] — a JSON list of "
|
|
522
|
+
"OpenAI-style function specs the assistant can call."
|
|
523
|
+
)
|
|
524
|
+
raise typer.Exit(1)
|
|
525
|
+
if seed is None:
|
|
526
|
+
console.print(
|
|
527
|
+
'[red]-f tools needs --seed[/red] — it becomes the system prompt.'
|
|
528
|
+
)
|
|
529
|
+
raise typer.Exit(1)
|
|
530
|
+
out = out or Path("data") / ("preferences.jsonl" if fmt == "preference" else "train.jsonl")
|
|
531
|
+
if out.exists():
|
|
532
|
+
console.print(f"[red]{out} already exists[/red] — pass -o to write elsewhere.")
|
|
533
|
+
raise typer.Exit(1)
|
|
534
|
+
|
|
535
|
+
import secrets
|
|
536
|
+
|
|
537
|
+
from rich.panel import Panel
|
|
538
|
+
|
|
539
|
+
from .mesh import MeshState, lan_ip, run_mesh_serve
|
|
540
|
+
from .synth import prepare_synth
|
|
541
|
+
|
|
542
|
+
token = token or secrets.token_urlsafe(16)
|
|
543
|
+
state = MeshState(
|
|
544
|
+
prepare_synth(fmt, seed, source, tools), n, out,
|
|
545
|
+
max_tokens=max_tokens, temperature=temperature,
|
|
546
|
+
lease_timeout=lease_timeout, think=not no_think,
|
|
547
|
+
)
|
|
548
|
+
join_url = f"http://{lan_ip()}:{port}"
|
|
549
|
+
console.print(Panel.fit(
|
|
550
|
+
f"Join from a Mac: [bold]troy mesh join {join_url} --token {token}[/bold]\n"
|
|
551
|
+
f"Join from iPhone: TroyWorker app → {join_url} + token [bold]{token}[/bold]",
|
|
552
|
+
title="troy mesh coordinator",
|
|
553
|
+
))
|
|
554
|
+
|
|
555
|
+
try:
|
|
556
|
+
stats = run_mesh_serve(state, host, port, token, linger=linger)
|
|
557
|
+
except OSError as e:
|
|
558
|
+
state.close()
|
|
559
|
+
if out.exists() and out.stat().st_size == 0:
|
|
560
|
+
out.unlink() # nothing was written; don't block a relaunch
|
|
561
|
+
console.print(f"[red]Can't bind {host}:{port}[/red] ({e.strerror}) — "
|
|
562
|
+
"pass --port to use a different one.")
|
|
563
|
+
raise typer.Exit(1)
|
|
564
|
+
console.print(
|
|
565
|
+
f"\nWrote [bold]{stats['records']}[/bold] examples to [bold]{stats['out']}[/bold] "
|
|
566
|
+
f"across {len(stats['workers'])} worker(s)"
|
|
567
|
+
)
|
|
568
|
+
if stats["records"] < stats["target"]:
|
|
569
|
+
console.print(
|
|
570
|
+
f"[yellow]Stopped at {stats['records']}/{stats['target']}.[/yellow]"
|
|
571
|
+
)
|
|
572
|
+
console.print(
|
|
573
|
+
"Review the data before training — spot-check a dozen examples, then: "
|
|
574
|
+
f"[bold]troy data validate {out}[/bold] and [bold]troy train[/bold]."
|
|
575
|
+
)
|
|
576
|
+
|
|
577
|
+
|
|
578
|
+
@mesh_app.command("join")
|
|
579
|
+
def mesh_join(
|
|
580
|
+
url: str = typer.Argument(..., help="Coordinator URL, e.g. http://192.168.1.5:8765"),
|
|
581
|
+
token: str = typer.Option(..., help="Token printed by `troy mesh serve`."),
|
|
582
|
+
model: str = typer.Option(
|
|
583
|
+
"auto", help="Teacher model to run here (auto = sized to this Mac's memory)."
|
|
584
|
+
),
|
|
585
|
+
name: Optional[str] = typer.Option(
|
|
586
|
+
None, help="Worker name shown on the coordinator (default: hostname)."
|
|
587
|
+
),
|
|
588
|
+
batch: int = typer.Option(2, help="Work items to lease per request."),
|
|
589
|
+
) -> None:
|
|
590
|
+
"""Join a mesh as a worker: run the teacher here, send results back."""
|
|
591
|
+
_require_apple_silicon()
|
|
592
|
+
|
|
593
|
+
import socket
|
|
594
|
+
|
|
595
|
+
from .mesh import run_mesh_join
|
|
596
|
+
from .synth import pick_teacher
|
|
597
|
+
|
|
598
|
+
if model == "auto":
|
|
599
|
+
model = pick_teacher()
|
|
600
|
+
console.print(f"Teacher: [bold]{model}[/bold] (picked for this Mac's memory)")
|
|
601
|
+
stats = run_mesh_join(url, token, model, name or socket.gethostname(), batch=batch)
|
|
602
|
+
console.print(
|
|
603
|
+
f"\nDone: completed [bold]{stats['completed']}[/bold] work item(s) "
|
|
604
|
+
f"({stats.get('records', '?')}/{stats.get('target', '?')} mesh total)."
|
|
605
|
+
)
|
|
606
|
+
|
|
607
|
+
|
|
344
608
|
if __name__ == "__main__":
|
|
345
609
|
app()
|
|
@@ -62,6 +62,10 @@ _SHAREGPT_ROLES = {
|
|
|
62
62
|
def _to_messages(record: Dict[str, Any], fmt: str) -> Dict[str, Any]:
|
|
63
63
|
"""Normalize an SFT record to mlx-lm chat format ({"messages": [...]})."""
|
|
64
64
|
if fmt == "chat":
|
|
65
|
+
# Preserve a per-record `tools` list — mlx-lm passes it to the chat
|
|
66
|
+
# template so tool schemas render exactly as they will at inference.
|
|
67
|
+
if "tools" in record:
|
|
68
|
+
return {"messages": record["messages"], "tools": record["tools"]}
|
|
65
69
|
return {"messages": record["messages"]}
|
|
66
70
|
if fmt == "sharegpt":
|
|
67
71
|
messages = [
|
|
@@ -152,6 +156,120 @@ def load_and_prepare(
|
|
|
152
156
|
return train, valid, fmt
|
|
153
157
|
|
|
154
158
|
|
|
159
|
+
# Required non-empty string fields per format (nested fields checked separately).
|
|
160
|
+
_REQUIRED_FIELDS = {
|
|
161
|
+
"alpaca": ("instruction", "output"),
|
|
162
|
+
"completions": ("prompt", "completion"),
|
|
163
|
+
"preference": ("prompt", "chosen", "rejected"),
|
|
164
|
+
"text": ("text",),
|
|
165
|
+
}
|
|
166
|
+
|
|
167
|
+
|
|
168
|
+
def _record_issues(record: Dict[str, Any], fmt: str, line: int) -> List[str]:
|
|
169
|
+
issues = []
|
|
170
|
+
|
|
171
|
+
def empty(v: Any) -> bool:
|
|
172
|
+
return not (isinstance(v, str) and v.strip())
|
|
173
|
+
|
|
174
|
+
for field in _REQUIRED_FIELDS.get(fmt, ()):
|
|
175
|
+
if empty(record.get(field)):
|
|
176
|
+
issues.append(f"line {line}: empty or missing `{field}`")
|
|
177
|
+
if fmt == "chat":
|
|
178
|
+
msgs = record.get("messages")
|
|
179
|
+
if not isinstance(msgs, list) or not msgs:
|
|
180
|
+
issues.append(f"line {line}: `messages` is not a non-empty list")
|
|
181
|
+
else:
|
|
182
|
+
roles = [m.get("role") for m in msgs if isinstance(m, dict)]
|
|
183
|
+
if "assistant" not in roles:
|
|
184
|
+
issues.append(f"line {line}: no assistant message")
|
|
185
|
+
if any(
|
|
186
|
+
not isinstance(m, dict) or empty(m.get("content"))
|
|
187
|
+
for m in msgs
|
|
188
|
+
):
|
|
189
|
+
issues.append(f"line {line}: message with empty content")
|
|
190
|
+
if fmt == "sharegpt":
|
|
191
|
+
convs = record.get("conversations")
|
|
192
|
+
if not isinstance(convs, list) or not convs:
|
|
193
|
+
issues.append(f"line {line}: `conversations` is not a non-empty list")
|
|
194
|
+
else:
|
|
195
|
+
bad = [
|
|
196
|
+
m.get("from")
|
|
197
|
+
for m in convs
|
|
198
|
+
if not isinstance(m, dict)
|
|
199
|
+
or str(m.get("from", "")).lower() not in _SHAREGPT_ROLES
|
|
200
|
+
]
|
|
201
|
+
if bad:
|
|
202
|
+
issues.append(f"line {line}: unknown speaker role(s): {bad}")
|
|
203
|
+
if fmt == "preference" and not issues:
|
|
204
|
+
if record["chosen"].strip() == record["rejected"].strip():
|
|
205
|
+
issues.append(f"line {line}: chosen == rejected")
|
|
206
|
+
return issues
|
|
207
|
+
|
|
208
|
+
|
|
209
|
+
def validate_records(path: str, max_issues: int = 50) -> Dict[str, Any]:
|
|
210
|
+
"""Lint a dataset: per-record issues, mixed formats, duplicates.
|
|
211
|
+
|
|
212
|
+
Returns {"records", "format", "issues", "duplicates", "truncated"}.
|
|
213
|
+
"""
|
|
214
|
+
p = Path(path).expanduser()
|
|
215
|
+
issues: List[str] = []
|
|
216
|
+
records: List[Tuple[int, Dict[str, Any]]] = []
|
|
217
|
+
|
|
218
|
+
if p.suffix.lower() == ".jsonl":
|
|
219
|
+
with open(p) as f:
|
|
220
|
+
for i, line in enumerate(f, 1):
|
|
221
|
+
if not line.strip():
|
|
222
|
+
continue
|
|
223
|
+
try:
|
|
224
|
+
records.append((i, json.loads(line)))
|
|
225
|
+
except json.JSONDecodeError as e:
|
|
226
|
+
issues.append(f"line {i}: invalid JSON ({e.msg})")
|
|
227
|
+
else:
|
|
228
|
+
records = [(i, r) for i, r in enumerate(_read_records(p), 1)]
|
|
229
|
+
|
|
230
|
+
if not records:
|
|
231
|
+
return {
|
|
232
|
+
"records": 0, "format": "unknown", "issues": issues or ["file has no records"],
|
|
233
|
+
"duplicates": 0, "truncated": False,
|
|
234
|
+
}
|
|
235
|
+
|
|
236
|
+
try:
|
|
237
|
+
fmt = detect_format(records[0][1])
|
|
238
|
+
except ValueError as e:
|
|
239
|
+
return {
|
|
240
|
+
"records": len(records), "format": "unknown",
|
|
241
|
+
"issues": issues + [str(e)], "duplicates": 0, "truncated": False,
|
|
242
|
+
}
|
|
243
|
+
|
|
244
|
+
seen: Dict[str, int] = {}
|
|
245
|
+
duplicates = 0
|
|
246
|
+
for line, record in records:
|
|
247
|
+
try:
|
|
248
|
+
rec_fmt = detect_format(record)
|
|
249
|
+
except ValueError:
|
|
250
|
+
issues.append(f"line {line}: keys match no known format")
|
|
251
|
+
continue
|
|
252
|
+
if rec_fmt != fmt:
|
|
253
|
+
issues.append(f"line {line}: format `{rec_fmt}` (file is `{fmt}`)")
|
|
254
|
+
continue
|
|
255
|
+
issues.extend(_record_issues(record, fmt, line))
|
|
256
|
+
key = json.dumps(record, sort_keys=True)
|
|
257
|
+
if key in seen:
|
|
258
|
+
duplicates += 1
|
|
259
|
+
issues.append(f"line {line}: exact duplicate of line {seen[key]}")
|
|
260
|
+
else:
|
|
261
|
+
seen[key] = line
|
|
262
|
+
|
|
263
|
+
truncated = len(issues) > max_issues
|
|
264
|
+
return {
|
|
265
|
+
"records": len(records),
|
|
266
|
+
"format": fmt,
|
|
267
|
+
"issues": issues[:max_issues],
|
|
268
|
+
"duplicates": duplicates,
|
|
269
|
+
"truncated": truncated,
|
|
270
|
+
}
|
|
271
|
+
|
|
272
|
+
|
|
155
273
|
def inspect_stats(path: str) -> Dict[str, Any]:
|
|
156
274
|
"""Lightweight dataset statistics for `troy data inspect`-style output."""
|
|
157
275
|
records = _read_records(Path(path).expanduser())
|