troy-cli 0.4.0__tar.gz → 0.5.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {troy_cli-0.4.0 → troy_cli-0.5.0}/PKG-INFO +20 -1
- {troy_cli-0.4.0 → troy_cli-0.5.0}/README.md +19 -0
- {troy_cli-0.4.0 → troy_cli-0.5.0}/pyproject.toml +1 -1
- {troy_cli-0.4.0 → troy_cli-0.5.0}/src/troy/__init__.py +1 -1
- {troy_cli-0.4.0 → troy_cli-0.5.0}/src/troy/cli.py +179 -6
- {troy_cli-0.4.0 → troy_cli-0.5.0}/src/troy/data.py +4 -0
- troy_cli-0.5.0/src/troy/export.py +242 -0
- troy_cli-0.5.0/src/troy/mesh.py +427 -0
- troy_cli-0.5.0/src/troy/synth.py +424 -0
- troy_cli-0.5.0/tests/test_export.py +54 -0
- troy_cli-0.5.0/tests/test_mesh.py +123 -0
- troy_cli-0.5.0/tests/test_synth_tools.py +60 -0
- troy_cli-0.4.0/src/troy/export.py +0 -81
- troy_cli-0.4.0/src/troy/synth.py +0 -247
- {troy_cli-0.4.0 → troy_cli-0.5.0}/.gitignore +0 -0
- {troy_cli-0.4.0 → troy_cli-0.5.0}/src/troy/chat.py +0 -0
- {troy_cli-0.4.0 → troy_cli-0.5.0}/src/troy/config.py +0 -0
- {troy_cli-0.4.0 → troy_cli-0.5.0}/src/troy/evaluate.py +0 -0
- {troy_cli-0.4.0 → troy_cli-0.5.0}/src/troy/hardware.py +0 -0
- {troy_cli-0.4.0 → troy_cli-0.5.0}/src/troy/push.py +0 -0
- {troy_cli-0.4.0 → troy_cli-0.5.0}/src/troy/serve.py +0 -0
- {troy_cli-0.4.0 → troy_cli-0.5.0}/src/troy/templates.py +0 -0
- {troy_cli-0.4.0 → troy_cli-0.5.0}/src/troy/train_dpo.py +0 -0
- {troy_cli-0.4.0 → troy_cli-0.5.0}/src/troy/train_orpo.py +0 -0
- {troy_cli-0.4.0 → troy_cli-0.5.0}/src/troy/train_sft.py +0 -0
- {troy_cli-0.4.0 → troy_cli-0.5.0}/src/troy/train_vision.py +0 -0
- {troy_cli-0.4.0 → troy_cli-0.5.0}/tests/test_config.py +0 -0
- {troy_cli-0.4.0 → troy_cli-0.5.0}/tests/test_data.py +0 -0
- {troy_cli-0.4.0 → troy_cli-0.5.0}/tests/test_hardware.py +0 -0
- {troy_cli-0.4.0 → troy_cli-0.5.0}/tests/test_synth.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: troy-cli
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.5.0
|
|
4
4
|
Summary: Fine-tune LLMs on your MacBook with one YAML file. Built for Apple Silicon.
|
|
5
5
|
Author: Troy
|
|
6
6
|
License: Apache-2.0
|
|
@@ -88,6 +88,25 @@ output: ./output
|
|
|
88
88
|
| `troy export` | Fuse the adapter; export MLX or GGUF |
|
|
89
89
|
| `troy push` | Upload adapter or fused model to the Hugging Face Hub |
|
|
90
90
|
| `troy data inspect` | Dataset stats and format detection |
|
|
91
|
+
| `troy data synth` | Synthesize a dataset with a local teacher model |
|
|
92
|
+
| `troy mesh serve` | Coordinate a LAN mesh: iPhones and Macs generate the dataset for you |
|
|
93
|
+
| `troy mesh join` | Join a mesh as a worker from any Mac |
|
|
94
|
+
|
|
95
|
+
## The mesh: your idle iPhones generate the dataset
|
|
96
|
+
|
|
97
|
+
`troy data synth` runs the teacher on one Mac. `troy mesh` farms the same job
|
|
98
|
+
out to every Apple device on your network — the coordinator mints prompts and
|
|
99
|
+
validates results (identical parsing to local synth), workers run the teacher:
|
|
100
|
+
|
|
101
|
+
```bash
|
|
102
|
+
troy mesh serve --from ./docs --n 500 # Mac: prints URL + token
|
|
103
|
+
troy mesh join http://mac:8765 --token … # any other Mac
|
|
104
|
+
# iPhones: the TroyWorker app (examples/ios/TroyWorker)
|
|
105
|
+
```
|
|
106
|
+
|
|
107
|
+
Workers can drop out at any time — leased work requeues automatically, and
|
|
108
|
+
duplicates are rejected centrally. The output is a normal `train.jsonl`:
|
|
109
|
+
validate it, then `troy train`.
|
|
91
110
|
|
|
92
111
|
## What Troy can train on your Mac
|
|
93
112
|
|
|
@@ -71,6 +71,25 @@ output: ./output
|
|
|
71
71
|
| `troy export` | Fuse the adapter; export MLX or GGUF |
|
|
72
72
|
| `troy push` | Upload adapter or fused model to the Hugging Face Hub |
|
|
73
73
|
| `troy data inspect` | Dataset stats and format detection |
|
|
74
|
+
| `troy data synth` | Synthesize a dataset with a local teacher model |
|
|
75
|
+
| `troy mesh serve` | Coordinate a LAN mesh: iPhones and Macs generate the dataset for you |
|
|
76
|
+
| `troy mesh join` | Join a mesh as a worker from any Mac |
|
|
77
|
+
|
|
78
|
+
## The mesh: your idle iPhones generate the dataset
|
|
79
|
+
|
|
80
|
+
`troy data synth` runs the teacher on one Mac. `troy mesh` farms the same job
|
|
81
|
+
out to every Apple device on your network — the coordinator mints prompts and
|
|
82
|
+
validates results (identical parsing to local synth), workers run the teacher:
|
|
83
|
+
|
|
84
|
+
```bash
|
|
85
|
+
troy mesh serve --from ./docs --n 500 # Mac: prints URL + token
|
|
86
|
+
troy mesh join http://mac:8765 --token … # any other Mac
|
|
87
|
+
# iPhones: the TroyWorker app (examples/ios/TroyWorker)
|
|
88
|
+
```
|
|
89
|
+
|
|
90
|
+
Workers can drop out at any time — leased work requeues automatically, and
|
|
91
|
+
duplicates are rejected centrally. The output is a normal `train.jsonl`:
|
|
92
|
+
validate it, then `troy train`.
|
|
74
93
|
|
|
75
94
|
## What Troy can train on your Mac
|
|
76
95
|
|
|
@@ -258,14 +258,14 @@ def serve(
|
|
|
258
258
|
@app.command()
|
|
259
259
|
def export(
|
|
260
260
|
config: Path = typer.Option(Path("troy.yaml"), "--config", "-c", help="Config file."),
|
|
261
|
-
fmt: str = typer.Option("mlx", "--format", "-f", help="Export format: mlx or
|
|
261
|
+
fmt: str = typer.Option("mlx", "--format", "-f", help="Export format: mlx, gguf, or ios."),
|
|
262
262
|
save_path: Optional[Path] = typer.Option(None, help="Output directory (default: <output>/fused)."),
|
|
263
263
|
dequantize: bool = typer.Option(False, help="Dequantize when fusing a quantized base."),
|
|
264
264
|
) -> None:
|
|
265
265
|
"""Merge the trained adapter into the base model and export it."""
|
|
266
266
|
_require_apple_silicon()
|
|
267
|
-
if fmt not in ("mlx", "gguf"):
|
|
268
|
-
console.print("[red]--format must be `mlx` or `
|
|
267
|
+
if fmt not in ("mlx", "gguf", "ios"):
|
|
268
|
+
console.print("[red]--format must be `mlx`, `gguf`, or `ios`.[/red]")
|
|
269
269
|
raise typer.Exit(1)
|
|
270
270
|
from .config import load_config
|
|
271
271
|
from .export import run_export
|
|
@@ -378,7 +378,15 @@ def synth(
|
|
|
378
378
|
n: int = typer.Option(100, "--n", help="Number of examples to generate."),
|
|
379
379
|
fmt: str = typer.Option(
|
|
380
380
|
"chat", "--format", "-f",
|
|
381
|
-
help="Output format: chat (SFT)
|
|
381
|
+
help="Output format: chat (SFT), preference (DPO/ORPO), or tools (tool calling).",
|
|
382
|
+
),
|
|
383
|
+
tools: Optional[Path] = typer.Option(
|
|
384
|
+
None, "--tools",
|
|
385
|
+
help="For -f tools: JSON file with the tool schemas (OpenAI function format).",
|
|
386
|
+
),
|
|
387
|
+
no_think: bool = typer.Option(
|
|
388
|
+
False, "--no-think",
|
|
389
|
+
help="For -f tools: omit the <think> reasoning traces.",
|
|
382
390
|
),
|
|
383
391
|
teacher: str = typer.Option(
|
|
384
392
|
"auto", help="Teacher model (auto = sized to this Mac's memory)."
|
|
@@ -398,9 +406,22 @@ def synth(
|
|
|
398
406
|
'--from ./docs and/or --seed "task description".'
|
|
399
407
|
)
|
|
400
408
|
raise typer.Exit(1)
|
|
401
|
-
if fmt not in ("chat", "preference"):
|
|
402
|
-
console.print("[red]--format must be `chat` or `
|
|
409
|
+
if fmt not in ("chat", "preference", "tools"):
|
|
410
|
+
console.print("[red]--format must be `chat`, `preference`, or `tools`.[/red]")
|
|
403
411
|
raise typer.Exit(1)
|
|
412
|
+
if fmt == "tools":
|
|
413
|
+
if tools is None or not tools.exists():
|
|
414
|
+
console.print(
|
|
415
|
+
"[red]-f tools needs --tools schemas.json[/red] — a JSON list of "
|
|
416
|
+
"OpenAI-style function specs the assistant can call."
|
|
417
|
+
)
|
|
418
|
+
raise typer.Exit(1)
|
|
419
|
+
if seed is None:
|
|
420
|
+
console.print(
|
|
421
|
+
'[red]-f tools needs --seed[/red] — it becomes the system prompt, '
|
|
422
|
+
'e.g. "travel planning assistant that books nothing".'
|
|
423
|
+
)
|
|
424
|
+
raise typer.Exit(1)
|
|
404
425
|
|
|
405
426
|
from .synth import pick_teacher, run_synth
|
|
406
427
|
|
|
@@ -416,6 +437,7 @@ def synth(
|
|
|
416
437
|
n=n, out_path=out, fmt=fmt, teacher=teacher,
|
|
417
438
|
seed_task=seed, source=source,
|
|
418
439
|
max_tokens=max_tokens, temperature=temperature,
|
|
440
|
+
tools_path=tools, think=not no_think,
|
|
419
441
|
)
|
|
420
442
|
console.print(
|
|
421
443
|
f"\nWrote [bold]{stats['records']}[/bold] examples to [bold]{stats['out']}[/bold] "
|
|
@@ -432,5 +454,156 @@ def synth(
|
|
|
432
454
|
)
|
|
433
455
|
|
|
434
456
|
|
|
457
|
+
mesh_app = typer.Typer(
|
|
458
|
+
name="mesh",
|
|
459
|
+
help="Distribute data synthesis across devices on your LAN.",
|
|
460
|
+
no_args_is_help=True,
|
|
461
|
+
)
|
|
462
|
+
app.add_typer(mesh_app)
|
|
463
|
+
|
|
464
|
+
|
|
465
|
+
@mesh_app.command("serve")
|
|
466
|
+
def mesh_serve(
|
|
467
|
+
source: Optional[Path] = typer.Option(
|
|
468
|
+
None, "--from", help="Ground examples in a file or folder of docs/code."
|
|
469
|
+
),
|
|
470
|
+
seed: Optional[str] = typer.Option(
|
|
471
|
+
None, "--seed", help='Task description, e.g. "customer support bot for Acme".'
|
|
472
|
+
),
|
|
473
|
+
n: int = typer.Option(100, "--n", help="Number of examples to generate."),
|
|
474
|
+
fmt: str = typer.Option(
|
|
475
|
+
"chat", "--format", "-f",
|
|
476
|
+
help="Output format: chat (SFT), preference (DPO/ORPO), or tools (tool calling).",
|
|
477
|
+
),
|
|
478
|
+
tools: Optional[Path] = typer.Option(
|
|
479
|
+
None, "--tools",
|
|
480
|
+
help="For -f tools: JSON file with the tool schemas (OpenAI function format).",
|
|
481
|
+
),
|
|
482
|
+
no_think: bool = typer.Option(
|
|
483
|
+
False, "--no-think",
|
|
484
|
+
help="For -f tools: omit the <think> reasoning traces.",
|
|
485
|
+
),
|
|
486
|
+
out: Optional[Path] = typer.Option(
|
|
487
|
+
None, "--out", "-o",
|
|
488
|
+
help="Output file (default: data/train.jsonl or data/preferences.jsonl).",
|
|
489
|
+
),
|
|
490
|
+
max_tokens: int = typer.Option(2048, help="Max tokens per teacher call."),
|
|
491
|
+
temperature: float = typer.Option(0.8),
|
|
492
|
+
host: str = typer.Option("0.0.0.0", help="Interface to bind."),
|
|
493
|
+
port: int = typer.Option(8765),
|
|
494
|
+
token: Optional[str] = typer.Option(
|
|
495
|
+
None, help="Shared worker token (default: generated at startup)."
|
|
496
|
+
),
|
|
497
|
+
lease_timeout: float = typer.Option(
|
|
498
|
+
300.0, help="Seconds before an unanswered work item is requeued."
|
|
499
|
+
),
|
|
500
|
+
linger: float = typer.Option(
|
|
501
|
+
30.0, help="Seconds to wait for in-flight results after the target is hit."
|
|
502
|
+
),
|
|
503
|
+
) -> None:
|
|
504
|
+
"""Coordinate a mesh: serve synth work to iPhones and Macs on your LAN.
|
|
505
|
+
|
|
506
|
+
Workers run the teacher model; this machine only mints prompts and
|
|
507
|
+
validates results, so it can be any Mac (the model never loads here).
|
|
508
|
+
"""
|
|
509
|
+
if source is None and seed is None:
|
|
510
|
+
console.print(
|
|
511
|
+
'[red]Give the workers something to work from:[/red] '
|
|
512
|
+
'--from ./docs and/or --seed "task description".'
|
|
513
|
+
)
|
|
514
|
+
raise typer.Exit(1)
|
|
515
|
+
if fmt not in ("chat", "preference", "tools"):
|
|
516
|
+
console.print("[red]--format must be `chat`, `preference`, or `tools`.[/red]")
|
|
517
|
+
raise typer.Exit(1)
|
|
518
|
+
if fmt == "tools":
|
|
519
|
+
if tools is None or not tools.exists():
|
|
520
|
+
console.print(
|
|
521
|
+
"[red]-f tools needs --tools schemas.json[/red] — a JSON list of "
|
|
522
|
+
"OpenAI-style function specs the assistant can call."
|
|
523
|
+
)
|
|
524
|
+
raise typer.Exit(1)
|
|
525
|
+
if seed is None:
|
|
526
|
+
console.print(
|
|
527
|
+
'[red]-f tools needs --seed[/red] — it becomes the system prompt.'
|
|
528
|
+
)
|
|
529
|
+
raise typer.Exit(1)
|
|
530
|
+
out = out or Path("data") / ("preferences.jsonl" if fmt == "preference" else "train.jsonl")
|
|
531
|
+
if out.exists():
|
|
532
|
+
console.print(f"[red]{out} already exists[/red] — pass -o to write elsewhere.")
|
|
533
|
+
raise typer.Exit(1)
|
|
534
|
+
|
|
535
|
+
import secrets
|
|
536
|
+
|
|
537
|
+
from rich.panel import Panel
|
|
538
|
+
|
|
539
|
+
from .mesh import MeshState, lan_ip, run_mesh_serve
|
|
540
|
+
from .synth import prepare_synth
|
|
541
|
+
|
|
542
|
+
token = token or secrets.token_urlsafe(16)
|
|
543
|
+
state = MeshState(
|
|
544
|
+
prepare_synth(fmt, seed, source, tools), n, out,
|
|
545
|
+
max_tokens=max_tokens, temperature=temperature,
|
|
546
|
+
lease_timeout=lease_timeout, think=not no_think,
|
|
547
|
+
)
|
|
548
|
+
join_url = f"http://{lan_ip()}:{port}"
|
|
549
|
+
console.print(Panel.fit(
|
|
550
|
+
f"Join from a Mac: [bold]troy mesh join {join_url} --token {token}[/bold]\n"
|
|
551
|
+
f"Join from iPhone: TroyWorker app → {join_url} + token [bold]{token}[/bold]",
|
|
552
|
+
title="troy mesh coordinator",
|
|
553
|
+
))
|
|
554
|
+
|
|
555
|
+
try:
|
|
556
|
+
stats = run_mesh_serve(state, host, port, token, linger=linger)
|
|
557
|
+
except OSError as e:
|
|
558
|
+
state.close()
|
|
559
|
+
if out.exists() and out.stat().st_size == 0:
|
|
560
|
+
out.unlink() # nothing was written; don't block a relaunch
|
|
561
|
+
console.print(f"[red]Can't bind {host}:{port}[/red] ({e.strerror}) — "
|
|
562
|
+
"pass --port to use a different one.")
|
|
563
|
+
raise typer.Exit(1)
|
|
564
|
+
console.print(
|
|
565
|
+
f"\nWrote [bold]{stats['records']}[/bold] examples to [bold]{stats['out']}[/bold] "
|
|
566
|
+
f"across {len(stats['workers'])} worker(s)"
|
|
567
|
+
)
|
|
568
|
+
if stats["records"] < stats["target"]:
|
|
569
|
+
console.print(
|
|
570
|
+
f"[yellow]Stopped at {stats['records']}/{stats['target']}.[/yellow]"
|
|
571
|
+
)
|
|
572
|
+
console.print(
|
|
573
|
+
"Review the data before training — spot-check a dozen examples, then: "
|
|
574
|
+
f"[bold]troy data validate {out}[/bold] and [bold]troy train[/bold]."
|
|
575
|
+
)
|
|
576
|
+
|
|
577
|
+
|
|
578
|
+
@mesh_app.command("join")
|
|
579
|
+
def mesh_join(
|
|
580
|
+
url: str = typer.Argument(..., help="Coordinator URL, e.g. http://192.168.1.5:8765"),
|
|
581
|
+
token: str = typer.Option(..., help="Token printed by `troy mesh serve`."),
|
|
582
|
+
model: str = typer.Option(
|
|
583
|
+
"auto", help="Teacher model to run here (auto = sized to this Mac's memory)."
|
|
584
|
+
),
|
|
585
|
+
name: Optional[str] = typer.Option(
|
|
586
|
+
None, help="Worker name shown on the coordinator (default: hostname)."
|
|
587
|
+
),
|
|
588
|
+
batch: int = typer.Option(2, help="Work items to lease per request."),
|
|
589
|
+
) -> None:
|
|
590
|
+
"""Join a mesh as a worker: run the teacher here, send results back."""
|
|
591
|
+
_require_apple_silicon()
|
|
592
|
+
|
|
593
|
+
import socket
|
|
594
|
+
|
|
595
|
+
from .mesh import run_mesh_join
|
|
596
|
+
from .synth import pick_teacher
|
|
597
|
+
|
|
598
|
+
if model == "auto":
|
|
599
|
+
model = pick_teacher()
|
|
600
|
+
console.print(f"Teacher: [bold]{model}[/bold] (picked for this Mac's memory)")
|
|
601
|
+
stats = run_mesh_join(url, token, model, name or socket.gethostname(), batch=batch)
|
|
602
|
+
console.print(
|
|
603
|
+
f"\nDone: completed [bold]{stats['completed']}[/bold] work item(s) "
|
|
604
|
+
f"({stats.get('records', '?')}/{stats.get('target', '?')} mesh total)."
|
|
605
|
+
)
|
|
606
|
+
|
|
607
|
+
|
|
435
608
|
if __name__ == "__main__":
|
|
436
609
|
app()
|
|
@@ -62,6 +62,10 @@ _SHAREGPT_ROLES = {
|
|
|
62
62
|
def _to_messages(record: Dict[str, Any], fmt: str) -> Dict[str, Any]:
|
|
63
63
|
"""Normalize an SFT record to mlx-lm chat format ({"messages": [...]})."""
|
|
64
64
|
if fmt == "chat":
|
|
65
|
+
# Preserve a per-record `tools` list — mlx-lm passes it to the chat
|
|
66
|
+
# template so tool schemas render exactly as they will at inference.
|
|
67
|
+
if "tools" in record:
|
|
68
|
+
return {"messages": record["messages"], "tools": record["tools"]}
|
|
65
69
|
return {"messages": record["messages"]}
|
|
66
70
|
if fmt == "sharegpt":
|
|
67
71
|
messages = [
|
|
@@ -0,0 +1,242 @@
|
|
|
1
|
+
"""Merge adapters and export to deployment formats (MLX, GGUF)."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import subprocess
|
|
6
|
+
import sys
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
def _fuse(base: str, adapter_path: str, save_path: str, dequantize: bool) -> None:
|
|
11
|
+
cmd = [
|
|
12
|
+
sys.executable,
|
|
13
|
+
"-m",
|
|
14
|
+
"mlx_lm",
|
|
15
|
+
"fuse",
|
|
16
|
+
"--model",
|
|
17
|
+
base,
|
|
18
|
+
"--adapter-path",
|
|
19
|
+
adapter_path,
|
|
20
|
+
"--save-path",
|
|
21
|
+
save_path,
|
|
22
|
+
]
|
|
23
|
+
if dequantize:
|
|
24
|
+
cmd.append("--dequantize")
|
|
25
|
+
print("Fusing adapter into base model ...")
|
|
26
|
+
result = subprocess.run(cmd)
|
|
27
|
+
if result.returncode != 0:
|
|
28
|
+
raise SystemExit(result.returncode)
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def _export_gguf(save_path: Path) -> Path:
|
|
32
|
+
"""Convert a fused MLX model directory to GGUF (llama/mistral/mixtral archs)."""
|
|
33
|
+
import json
|
|
34
|
+
|
|
35
|
+
import mlx.core as mx
|
|
36
|
+
from mlx_lm import gguf as gguf_mod
|
|
37
|
+
|
|
38
|
+
with open(save_path / "config.json") as f:
|
|
39
|
+
config = json.load(f)
|
|
40
|
+
|
|
41
|
+
weights = {}
|
|
42
|
+
for part in sorted(save_path.glob("*.safetensors")):
|
|
43
|
+
weights.update(mx.load(str(part)))
|
|
44
|
+
|
|
45
|
+
# mlx-lm's converter permutes attention weights into non-contiguous views,
|
|
46
|
+
# which save_gguf rejects — force contiguity on its output.
|
|
47
|
+
orig_permute = gguf_mod.permute_weights
|
|
48
|
+
gguf_mod.permute_weights = lambda *a, **k: mx.contiguous(orig_permute(*a, **k))
|
|
49
|
+
try:
|
|
50
|
+
out = save_path / "ggml-model-f16.gguf"
|
|
51
|
+
gguf_mod.convert_to_gguf(save_path, weights, config, str(out))
|
|
52
|
+
finally:
|
|
53
|
+
gguf_mod.permute_weights = orig_permute
|
|
54
|
+
return out
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
# model_type values registered in mlx-swift-lm's LLMTypeRegistry (MLXLLM).
|
|
58
|
+
# Fused models outside this set won't load in an iOS app using mlx-swift-lm.
|
|
59
|
+
_IOS_SUPPORTED_MODEL_TYPES = {
|
|
60
|
+
"mistral", "mixtral", "llama", "phi", "phi3", "phimoe",
|
|
61
|
+
"gemma", "gemma2", "gemma3", "gemma3_text", "gemma3n",
|
|
62
|
+
"gemma4", "gemma4_unified", "gemma4_text",
|
|
63
|
+
"qwen2", "qwen3", "qwen3_moe", "qwen3_next",
|
|
64
|
+
"qwen3_5", "qwen3_5_moe", "qwen3_5_text",
|
|
65
|
+
"minicpm", "starcoder2", "cohere", "openelm", "internlm2",
|
|
66
|
+
"deepseek_v2", "deepseek_v3", "granite", "helium", "granitemoehybrid",
|
|
67
|
+
"glm4", "glm4_moe", "glm4_moe_lite", "falcon_h1", "bitnet", "smollm3",
|
|
68
|
+
"ernie4_5", "lfm2", "lfm2_moe", "exaone4", "gpt_oss", "olmoe", "olmo2",
|
|
69
|
+
"olmo3", "nemotron_h", "jamba", "mamba2", "mistral3", "apertus",
|
|
70
|
+
}
|
|
71
|
+
|
|
72
|
+
# On an 8 GB iPhone, an app with the Increased Memory Limit entitlement gets
|
|
73
|
+
# roughly 6 GB resident; weights + KV cache + app overhead must fit inside it.
|
|
74
|
+
_IOS_COMFORTABLE_GB = 2.2 # loads without entitlements on recent iPhones
|
|
75
|
+
_IOS_MAX_GB = 4.0 # needs Increased Memory Limit / Extended Virtual Addressing
|
|
76
|
+
|
|
77
|
+
_IOS_README = """\
|
|
78
|
+
# Run this model on iPhone / iPad with MLX Swift
|
|
79
|
+
|
|
80
|
+
This directory is a fused MLX model in the layout `mlx-swift-lm` loads directly.
|
|
81
|
+
|
|
82
|
+
## Load it
|
|
83
|
+
|
|
84
|
+
Add the Swift package: https://github.com/ml-explore/mlx-swift-lm
|
|
85
|
+
|
|
86
|
+
```swift
|
|
87
|
+
import MLXLLM
|
|
88
|
+
import MLXLMCommon
|
|
89
|
+
|
|
90
|
+
// From a local directory bundled with (or downloaded by) your app:
|
|
91
|
+
let modelDirectory: URL = ... // this folder on device
|
|
92
|
+
let container = try await LLMModelFactory.shared.loadContainer(
|
|
93
|
+
configuration: ModelConfiguration(directory: modelDirectory))
|
|
94
|
+
|
|
95
|
+
let result = try await container.perform { context in
|
|
96
|
+
let input = try await context.processor.prepare(
|
|
97
|
+
input: .init(prompt: "Hello!"))
|
|
98
|
+
return try MLXLMCommon.generate(
|
|
99
|
+
input: input, parameters: .init(), context: context)
|
|
100
|
+
}
|
|
101
|
+
```
|
|
102
|
+
|
|
103
|
+
Weights this large should not ship inside the app bundle — download them on
|
|
104
|
+
first launch (Background Assets, or a direct download from your server or the
|
|
105
|
+
Hugging Face Hub via `troy push`).
|
|
106
|
+
|
|
107
|
+
## Memory entitlements
|
|
108
|
+
|
|
109
|
+
Models over ~2 GB need one of these capabilities in Xcode
|
|
110
|
+
(Signing & Capabilities → + Capability):
|
|
111
|
+
|
|
112
|
+
- **Increased Memory Limit** (`com.apple.developer.kernel.increased-memory-limit`)
|
|
113
|
+
- **Extended Virtual Addressing**
|
|
114
|
+
|
|
115
|
+
Either is usually sufficient; devices with 8 GB RAM run 4-bit models up to
|
|
116
|
+
roughly 4 GB of weights. Test on the oldest device you target.
|
|
117
|
+
"""
|
|
118
|
+
|
|
119
|
+
|
|
120
|
+
def _base_is_quantized(base: str) -> bool:
|
|
121
|
+
import json
|
|
122
|
+
|
|
123
|
+
p = Path(base)
|
|
124
|
+
if not p.exists():
|
|
125
|
+
from huggingface_hub import snapshot_download
|
|
126
|
+
|
|
127
|
+
p = Path(snapshot_download(base, allow_patterns=["config.json"]))
|
|
128
|
+
return "quantization" in json.loads((p / "config.json").read_text())
|
|
129
|
+
|
|
130
|
+
|
|
131
|
+
def _fuse_for_ios(base: str, adapter_path: str, save_path: str) -> None:
|
|
132
|
+
"""Fuse for iOS without destroying the adapter.
|
|
133
|
+
|
|
134
|
+
Fusing a LoRA into 4-bit weights loses the deltas to quantization noise —
|
|
135
|
+
the exported model silently behaves like the base. So for a quantized
|
|
136
|
+
base: dequantize-fuse to fp16, then requantize fresh at 8 bits (verified
|
|
137
|
+
to preserve tuned behavior; 6 bits already degrades it).
|
|
138
|
+
"""
|
|
139
|
+
import shutil
|
|
140
|
+
import tempfile
|
|
141
|
+
|
|
142
|
+
if not _base_is_quantized(base):
|
|
143
|
+
_fuse(base, adapter_path, save_path, dequantize=False)
|
|
144
|
+
return
|
|
145
|
+
from mlx_lm import convert
|
|
146
|
+
|
|
147
|
+
print("Quantized base: dequantize-fusing, then requantizing at 8 bits")
|
|
148
|
+
print("(fusing straight into 4-bit silently erases the adapter).")
|
|
149
|
+
out = Path(save_path)
|
|
150
|
+
if out.exists():
|
|
151
|
+
shutil.rmtree(out)
|
|
152
|
+
with tempfile.TemporaryDirectory() as td:
|
|
153
|
+
_fuse(base, adapter_path, td, dequantize=True)
|
|
154
|
+
convert(td, mlx_path=str(out), quantize=True, q_bits=8)
|
|
155
|
+
|
|
156
|
+
|
|
157
|
+
def _dir_weight_bytes(save_path: Path) -> int:
|
|
158
|
+
return sum(p.stat().st_size for p in save_path.glob("*.safetensors"))
|
|
159
|
+
|
|
160
|
+
|
|
161
|
+
def _export_ios(save_path: Path) -> None:
|
|
162
|
+
"""Validate a fused MLX directory for mlx-swift-lm on iOS and add a how-to."""
|
|
163
|
+
import json
|
|
164
|
+
|
|
165
|
+
with open(save_path / "config.json") as f:
|
|
166
|
+
config = json.load(f)
|
|
167
|
+
|
|
168
|
+
problems = []
|
|
169
|
+
model_type = config.get("model_type", "?")
|
|
170
|
+
if model_type not in _IOS_SUPPORTED_MODEL_TYPES:
|
|
171
|
+
problems.append(
|
|
172
|
+
f"model_type `{model_type}` is not in mlx-swift-lm's registry — "
|
|
173
|
+
"it won't load on iOS without adding the architecture in Swift."
|
|
174
|
+
)
|
|
175
|
+
for name in ("tokenizer.json", "tokenizer_config.json"):
|
|
176
|
+
if not (save_path / name).exists():
|
|
177
|
+
problems.append(
|
|
178
|
+
f"{name} is missing — swift-transformers needs it to build "
|
|
179
|
+
"the tokenizer on device."
|
|
180
|
+
)
|
|
181
|
+
|
|
182
|
+
gb = _dir_weight_bytes(save_path) / 1e9
|
|
183
|
+
quantized = "quantization" in config
|
|
184
|
+
print(f"Weights: {gb:.2f} GB ({'quantized' if quantized else 'float'}), "
|
|
185
|
+
f"model_type: {model_type}")
|
|
186
|
+
if not quantized:
|
|
187
|
+
print("Tip: unquantized weights are heavy for iPhone — train from a "
|
|
188
|
+
"4-bit base (e.g. an mlx-community *-4bit model) for iOS.")
|
|
189
|
+
if gb <= _IOS_COMFORTABLE_GB:
|
|
190
|
+
print("iPhone fit: comfortable — loads on recent iPhones; the "
|
|
191
|
+
"Increased Memory Limit capability is still recommended.")
|
|
192
|
+
elif gb <= _IOS_MAX_GB:
|
|
193
|
+
print("iPhone fit: needs the Increased Memory Limit or Extended "
|
|
194
|
+
"Virtual Addressing capability; expect 8 GB-RAM devices only.")
|
|
195
|
+
else:
|
|
196
|
+
print(f"iPhone fit: unlikely — {gb:.1f} GB of weights exceeds what "
|
|
197
|
+
"an 8 GB iPhone can keep resident. Use a smaller or more "
|
|
198
|
+
"aggressively quantized base.")
|
|
199
|
+
|
|
200
|
+
(save_path / "README-iOS.md").write_text(_IOS_README)
|
|
201
|
+
print(f"Wrote {save_path / 'README-iOS.md'}")
|
|
202
|
+
|
|
203
|
+
if problems:
|
|
204
|
+
for p in problems:
|
|
205
|
+
print(f"WARNING: {p}")
|
|
206
|
+
raise SystemExit(1)
|
|
207
|
+
|
|
208
|
+
|
|
209
|
+
def run_export(
|
|
210
|
+
base: str,
|
|
211
|
+
adapter_path: str,
|
|
212
|
+
save_path: str,
|
|
213
|
+
fmt: str = "mlx",
|
|
214
|
+
dequantize: bool = False,
|
|
215
|
+
) -> None:
|
|
216
|
+
"""Fuse a LoRA adapter into the base model; optionally convert to GGUF.
|
|
217
|
+
|
|
218
|
+
fmt: "mlx" (fused MLX weights), "gguf" (also writes ggml-model-f16.gguf
|
|
219
|
+
for llama.cpp / Ollama / LM Studio; llama, mistral and mixtral
|
|
220
|
+
architectures, unquantized base), or "ios" (fused MLX weights validated
|
|
221
|
+
for mlx-swift-lm on iPhone/iPad, plus a README-iOS.md how-to).
|
|
222
|
+
"""
|
|
223
|
+
if fmt == "ios" and dequantize:
|
|
224
|
+
print("Note: --dequantize with -f ios makes weights ~4x larger; "
|
|
225
|
+
"quantized weights are what you want on iPhone.")
|
|
226
|
+
if not Path(base).exists():
|
|
227
|
+
# mlx-lm's training path downloads a partial snapshot; fuse checks for
|
|
228
|
+
# a complete one. Top it up before fusing.
|
|
229
|
+
from huggingface_hub import snapshot_download
|
|
230
|
+
|
|
231
|
+
snapshot_download(base)
|
|
232
|
+
|
|
233
|
+
if fmt == "ios":
|
|
234
|
+
_fuse_for_ios(base, adapter_path, save_path)
|
|
235
|
+
else:
|
|
236
|
+
_fuse(base, adapter_path, save_path, dequantize)
|
|
237
|
+
print(f"Fused model: {Path(save_path).resolve()}")
|
|
238
|
+
if fmt == "gguf":
|
|
239
|
+
out = _export_gguf(Path(save_path))
|
|
240
|
+
print(f"GGUF: {out.resolve()}")
|
|
241
|
+
elif fmt == "ios":
|
|
242
|
+
_export_ios(Path(save_path))
|