troy-cli 0.4.0__tar.gz → 0.5.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (30) hide show
  1. {troy_cli-0.4.0 → troy_cli-0.5.0}/PKG-INFO +20 -1
  2. {troy_cli-0.4.0 → troy_cli-0.5.0}/README.md +19 -0
  3. {troy_cli-0.4.0 → troy_cli-0.5.0}/pyproject.toml +1 -1
  4. {troy_cli-0.4.0 → troy_cli-0.5.0}/src/troy/__init__.py +1 -1
  5. {troy_cli-0.4.0 → troy_cli-0.5.0}/src/troy/cli.py +179 -6
  6. {troy_cli-0.4.0 → troy_cli-0.5.0}/src/troy/data.py +4 -0
  7. troy_cli-0.5.0/src/troy/export.py +242 -0
  8. troy_cli-0.5.0/src/troy/mesh.py +427 -0
  9. troy_cli-0.5.0/src/troy/synth.py +424 -0
  10. troy_cli-0.5.0/tests/test_export.py +54 -0
  11. troy_cli-0.5.0/tests/test_mesh.py +123 -0
  12. troy_cli-0.5.0/tests/test_synth_tools.py +60 -0
  13. troy_cli-0.4.0/src/troy/export.py +0 -81
  14. troy_cli-0.4.0/src/troy/synth.py +0 -247
  15. {troy_cli-0.4.0 → troy_cli-0.5.0}/.gitignore +0 -0
  16. {troy_cli-0.4.0 → troy_cli-0.5.0}/src/troy/chat.py +0 -0
  17. {troy_cli-0.4.0 → troy_cli-0.5.0}/src/troy/config.py +0 -0
  18. {troy_cli-0.4.0 → troy_cli-0.5.0}/src/troy/evaluate.py +0 -0
  19. {troy_cli-0.4.0 → troy_cli-0.5.0}/src/troy/hardware.py +0 -0
  20. {troy_cli-0.4.0 → troy_cli-0.5.0}/src/troy/push.py +0 -0
  21. {troy_cli-0.4.0 → troy_cli-0.5.0}/src/troy/serve.py +0 -0
  22. {troy_cli-0.4.0 → troy_cli-0.5.0}/src/troy/templates.py +0 -0
  23. {troy_cli-0.4.0 → troy_cli-0.5.0}/src/troy/train_dpo.py +0 -0
  24. {troy_cli-0.4.0 → troy_cli-0.5.0}/src/troy/train_orpo.py +0 -0
  25. {troy_cli-0.4.0 → troy_cli-0.5.0}/src/troy/train_sft.py +0 -0
  26. {troy_cli-0.4.0 → troy_cli-0.5.0}/src/troy/train_vision.py +0 -0
  27. {troy_cli-0.4.0 → troy_cli-0.5.0}/tests/test_config.py +0 -0
  28. {troy_cli-0.4.0 → troy_cli-0.5.0}/tests/test_data.py +0 -0
  29. {troy_cli-0.4.0 → troy_cli-0.5.0}/tests/test_hardware.py +0 -0
  30. {troy_cli-0.4.0 → troy_cli-0.5.0}/tests/test_synth.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: troy-cli
3
- Version: 0.4.0
3
+ Version: 0.5.0
4
4
  Summary: Fine-tune LLMs on your MacBook with one YAML file. Built for Apple Silicon.
5
5
  Author: Troy
6
6
  License: Apache-2.0
@@ -88,6 +88,25 @@ output: ./output
88
88
  | `troy export` | Fuse the adapter; export MLX or GGUF |
89
89
  | `troy push` | Upload adapter or fused model to the Hugging Face Hub |
90
90
  | `troy data inspect` | Dataset stats and format detection |
91
+ | `troy data synth` | Synthesize a dataset with a local teacher model |
92
+ | `troy mesh serve` | Coordinate a LAN mesh: iPhones and Macs generate the dataset for you |
93
+ | `troy mesh join` | Join a mesh as a worker from any Mac |
94
+
95
+ ## The mesh: your idle iPhones generate the dataset
96
+
97
+ `troy data synth` runs the teacher on one Mac. `troy mesh` farms the same job
98
+ out to every Apple device on your network — the coordinator mints prompts and
99
+ validates results (identical parsing to local synth), workers run the teacher:
100
+
101
+ ```bash
102
+ troy mesh serve --from ./docs --n 500 # Mac: prints URL + token
103
+ troy mesh join http://mac:8765 --token … # any other Mac
104
+ # iPhones: the TroyWorker app (examples/ios/TroyWorker)
105
+ ```
106
+
107
+ Workers can drop out at any time — leased work requeues automatically, and
108
+ duplicates are rejected centrally. The output is a normal `train.jsonl`:
109
+ validate it, then `troy train`.
91
110
 
92
111
  ## What Troy can train on your Mac
93
112
 
@@ -71,6 +71,25 @@ output: ./output
71
71
  | `troy export` | Fuse the adapter; export MLX or GGUF |
72
72
  | `troy push` | Upload adapter or fused model to the Hugging Face Hub |
73
73
  | `troy data inspect` | Dataset stats and format detection |
74
+ | `troy data synth` | Synthesize a dataset with a local teacher model |
75
+ | `troy mesh serve` | Coordinate a LAN mesh: iPhones and Macs generate the dataset for you |
76
+ | `troy mesh join` | Join a mesh as a worker from any Mac |
77
+
78
+ ## The mesh: your idle iPhones generate the dataset
79
+
80
+ `troy data synth` runs the teacher on one Mac. `troy mesh` farms the same job
81
+ out to every Apple device on your network — the coordinator mints prompts and
82
+ validates results (identical parsing to local synth), workers run the teacher:
83
+
84
+ ```bash
85
+ troy mesh serve --from ./docs --n 500 # Mac: prints URL + token
86
+ troy mesh join http://mac:8765 --token … # any other Mac
87
+ # iPhones: the TroyWorker app (examples/ios/TroyWorker)
88
+ ```
89
+
90
+ Workers can drop out at any time — leased work requeues automatically, and
91
+ duplicates are rejected centrally. The output is a normal `train.jsonl`:
92
+ validate it, then `troy train`.
74
93
 
75
94
  ## What Troy can train on your Mac
76
95
 
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "troy-cli"
3
- version = "0.4.0"
3
+ version = "0.5.0"
4
4
  description = "Fine-tune LLMs on your MacBook with one YAML file. Built for Apple Silicon."
5
5
  readme = "README.md"
6
6
  requires-python = ">=3.10"
@@ -1,3 +1,3 @@
1
1
  """Troy: fine-tune LLMs on your MacBook with one YAML file."""
2
2
 
3
- __version__ = "0.4.0"
3
+ __version__ = "0.5.0"
@@ -258,14 +258,14 @@ def serve(
258
258
  @app.command()
259
259
  def export(
260
260
  config: Path = typer.Option(Path("troy.yaml"), "--config", "-c", help="Config file."),
261
- fmt: str = typer.Option("mlx", "--format", "-f", help="Export format: mlx or gguf."),
261
+ fmt: str = typer.Option("mlx", "--format", "-f", help="Export format: mlx, gguf, or ios."),
262
262
  save_path: Optional[Path] = typer.Option(None, help="Output directory (default: <output>/fused)."),
263
263
  dequantize: bool = typer.Option(False, help="Dequantize when fusing a quantized base."),
264
264
  ) -> None:
265
265
  """Merge the trained adapter into the base model and export it."""
266
266
  _require_apple_silicon()
267
- if fmt not in ("mlx", "gguf"):
268
- console.print("[red]--format must be `mlx` or `gguf`.[/red]")
267
+ if fmt not in ("mlx", "gguf", "ios"):
268
+ console.print("[red]--format must be `mlx`, `gguf`, or `ios`.[/red]")
269
269
  raise typer.Exit(1)
270
270
  from .config import load_config
271
271
  from .export import run_export
@@ -378,7 +378,15 @@ def synth(
378
378
  n: int = typer.Option(100, "--n", help="Number of examples to generate."),
379
379
  fmt: str = typer.Option(
380
380
  "chat", "--format", "-f",
381
- help="Output format: chat (SFT) or preference (DPO/ORPO).",
381
+ help="Output format: chat (SFT), preference (DPO/ORPO), or tools (tool calling).",
382
+ ),
383
+ tools: Optional[Path] = typer.Option(
384
+ None, "--tools",
385
+ help="For -f tools: JSON file with the tool schemas (OpenAI function format).",
386
+ ),
387
+ no_think: bool = typer.Option(
388
+ False, "--no-think",
389
+ help="For -f tools: omit the <think> reasoning traces.",
382
390
  ),
383
391
  teacher: str = typer.Option(
384
392
  "auto", help="Teacher model (auto = sized to this Mac's memory)."
@@ -398,9 +406,22 @@ def synth(
398
406
  '--from ./docs and/or --seed "task description".'
399
407
  )
400
408
  raise typer.Exit(1)
401
- if fmt not in ("chat", "preference"):
402
- console.print("[red]--format must be `chat` or `preference`.[/red]")
409
+ if fmt not in ("chat", "preference", "tools"):
410
+ console.print("[red]--format must be `chat`, `preference`, or `tools`.[/red]")
403
411
  raise typer.Exit(1)
412
+ if fmt == "tools":
413
+ if tools is None or not tools.exists():
414
+ console.print(
415
+ "[red]-f tools needs --tools schemas.json[/red] — a JSON list of "
416
+ "OpenAI-style function specs the assistant can call."
417
+ )
418
+ raise typer.Exit(1)
419
+ if seed is None:
420
+ console.print(
421
+ '[red]-f tools needs --seed[/red] — it becomes the system prompt, '
422
+ 'e.g. "travel planning assistant that books nothing".'
423
+ )
424
+ raise typer.Exit(1)
404
425
 
405
426
  from .synth import pick_teacher, run_synth
406
427
 
@@ -416,6 +437,7 @@ def synth(
416
437
  n=n, out_path=out, fmt=fmt, teacher=teacher,
417
438
  seed_task=seed, source=source,
418
439
  max_tokens=max_tokens, temperature=temperature,
440
+ tools_path=tools, think=not no_think,
419
441
  )
420
442
  console.print(
421
443
  f"\nWrote [bold]{stats['records']}[/bold] examples to [bold]{stats['out']}[/bold] "
@@ -432,5 +454,156 @@ def synth(
432
454
  )
433
455
 
434
456
 
457
+ mesh_app = typer.Typer(
458
+ name="mesh",
459
+ help="Distribute data synthesis across devices on your LAN.",
460
+ no_args_is_help=True,
461
+ )
462
+ app.add_typer(mesh_app)
463
+
464
+
465
+ @mesh_app.command("serve")
466
+ def mesh_serve(
467
+ source: Optional[Path] = typer.Option(
468
+ None, "--from", help="Ground examples in a file or folder of docs/code."
469
+ ),
470
+ seed: Optional[str] = typer.Option(
471
+ None, "--seed", help='Task description, e.g. "customer support bot for Acme".'
472
+ ),
473
+ n: int = typer.Option(100, "--n", help="Number of examples to generate."),
474
+ fmt: str = typer.Option(
475
+ "chat", "--format", "-f",
476
+ help="Output format: chat (SFT), preference (DPO/ORPO), or tools (tool calling).",
477
+ ),
478
+ tools: Optional[Path] = typer.Option(
479
+ None, "--tools",
480
+ help="For -f tools: JSON file with the tool schemas (OpenAI function format).",
481
+ ),
482
+ no_think: bool = typer.Option(
483
+ False, "--no-think",
484
+ help="For -f tools: omit the <think> reasoning traces.",
485
+ ),
486
+ out: Optional[Path] = typer.Option(
487
+ None, "--out", "-o",
488
+ help="Output file (default: data/train.jsonl or data/preferences.jsonl).",
489
+ ),
490
+ max_tokens: int = typer.Option(2048, help="Max tokens per teacher call."),
491
+ temperature: float = typer.Option(0.8),
492
+ host: str = typer.Option("0.0.0.0", help="Interface to bind."),
493
+ port: int = typer.Option(8765),
494
+ token: Optional[str] = typer.Option(
495
+ None, help="Shared worker token (default: generated at startup)."
496
+ ),
497
+ lease_timeout: float = typer.Option(
498
+ 300.0, help="Seconds before an unanswered work item is requeued."
499
+ ),
500
+ linger: float = typer.Option(
501
+ 30.0, help="Seconds to wait for in-flight results after the target is hit."
502
+ ),
503
+ ) -> None:
504
+ """Coordinate a mesh: serve synth work to iPhones and Macs on your LAN.
505
+
506
+ Workers run the teacher model; this machine only mints prompts and
507
+ validates results, so it can be any Mac (the model never loads here).
508
+ """
509
+ if source is None and seed is None:
510
+ console.print(
511
+ '[red]Give the workers something to work from:[/red] '
512
+ '--from ./docs and/or --seed "task description".'
513
+ )
514
+ raise typer.Exit(1)
515
+ if fmt not in ("chat", "preference", "tools"):
516
+ console.print("[red]--format must be `chat`, `preference`, or `tools`.[/red]")
517
+ raise typer.Exit(1)
518
+ if fmt == "tools":
519
+ if tools is None or not tools.exists():
520
+ console.print(
521
+ "[red]-f tools needs --tools schemas.json[/red] — a JSON list of "
522
+ "OpenAI-style function specs the assistant can call."
523
+ )
524
+ raise typer.Exit(1)
525
+ if seed is None:
526
+ console.print(
527
+ '[red]-f tools needs --seed[/red] — it becomes the system prompt.'
528
+ )
529
+ raise typer.Exit(1)
530
+ out = out or Path("data") / ("preferences.jsonl" if fmt == "preference" else "train.jsonl")
531
+ if out.exists():
532
+ console.print(f"[red]{out} already exists[/red] — pass -o to write elsewhere.")
533
+ raise typer.Exit(1)
534
+
535
+ import secrets
536
+
537
+ from rich.panel import Panel
538
+
539
+ from .mesh import MeshState, lan_ip, run_mesh_serve
540
+ from .synth import prepare_synth
541
+
542
+ token = token or secrets.token_urlsafe(16)
543
+ state = MeshState(
544
+ prepare_synth(fmt, seed, source, tools), n, out,
545
+ max_tokens=max_tokens, temperature=temperature,
546
+ lease_timeout=lease_timeout, think=not no_think,
547
+ )
548
+ join_url = f"http://{lan_ip()}:{port}"
549
+ console.print(Panel.fit(
550
+ f"Join from a Mac: [bold]troy mesh join {join_url} --token {token}[/bold]\n"
551
+ f"Join from iPhone: TroyWorker app → {join_url} + token [bold]{token}[/bold]",
552
+ title="troy mesh coordinator",
553
+ ))
554
+
555
+ try:
556
+ stats = run_mesh_serve(state, host, port, token, linger=linger)
557
+ except OSError as e:
558
+ state.close()
559
+ if out.exists() and out.stat().st_size == 0:
560
+ out.unlink() # nothing was written; don't block a relaunch
561
+ console.print(f"[red]Can't bind {host}:{port}[/red] ({e.strerror}) — "
562
+ "pass --port to use a different one.")
563
+ raise typer.Exit(1)
564
+ console.print(
565
+ f"\nWrote [bold]{stats['records']}[/bold] examples to [bold]{stats['out']}[/bold] "
566
+ f"across {len(stats['workers'])} worker(s)"
567
+ )
568
+ if stats["records"] < stats["target"]:
569
+ console.print(
570
+ f"[yellow]Stopped at {stats['records']}/{stats['target']}.[/yellow]"
571
+ )
572
+ console.print(
573
+ "Review the data before training — spot-check a dozen examples, then: "
574
+ f"[bold]troy data validate {out}[/bold] and [bold]troy train[/bold]."
575
+ )
576
+
577
+
578
+ @mesh_app.command("join")
579
+ def mesh_join(
580
+ url: str = typer.Argument(..., help="Coordinator URL, e.g. http://192.168.1.5:8765"),
581
+ token: str = typer.Option(..., help="Token printed by `troy mesh serve`."),
582
+ model: str = typer.Option(
583
+ "auto", help="Teacher model to run here (auto = sized to this Mac's memory)."
584
+ ),
585
+ name: Optional[str] = typer.Option(
586
+ None, help="Worker name shown on the coordinator (default: hostname)."
587
+ ),
588
+ batch: int = typer.Option(2, help="Work items to lease per request."),
589
+ ) -> None:
590
+ """Join a mesh as a worker: run the teacher here, send results back."""
591
+ _require_apple_silicon()
592
+
593
+ import socket
594
+
595
+ from .mesh import run_mesh_join
596
+ from .synth import pick_teacher
597
+
598
+ if model == "auto":
599
+ model = pick_teacher()
600
+ console.print(f"Teacher: [bold]{model}[/bold] (picked for this Mac's memory)")
601
+ stats = run_mesh_join(url, token, model, name or socket.gethostname(), batch=batch)
602
+ console.print(
603
+ f"\nDone: completed [bold]{stats['completed']}[/bold] work item(s) "
604
+ f"({stats.get('records', '?')}/{stats.get('target', '?')} mesh total)."
605
+ )
606
+
607
+
435
608
  if __name__ == "__main__":
436
609
  app()
@@ -62,6 +62,10 @@ _SHAREGPT_ROLES = {
62
62
  def _to_messages(record: Dict[str, Any], fmt: str) -> Dict[str, Any]:
63
63
  """Normalize an SFT record to mlx-lm chat format ({"messages": [...]})."""
64
64
  if fmt == "chat":
65
+ # Preserve a per-record `tools` list — mlx-lm passes it to the chat
66
+ # template so tool schemas render exactly as they will at inference.
67
+ if "tools" in record:
68
+ return {"messages": record["messages"], "tools": record["tools"]}
65
69
  return {"messages": record["messages"]}
66
70
  if fmt == "sharegpt":
67
71
  messages = [
@@ -0,0 +1,242 @@
1
+ """Merge adapters and export to deployment formats (MLX, GGUF)."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import subprocess
6
+ import sys
7
+ from pathlib import Path
8
+
9
+
10
+ def _fuse(base: str, adapter_path: str, save_path: str, dequantize: bool) -> None:
11
+ cmd = [
12
+ sys.executable,
13
+ "-m",
14
+ "mlx_lm",
15
+ "fuse",
16
+ "--model",
17
+ base,
18
+ "--adapter-path",
19
+ adapter_path,
20
+ "--save-path",
21
+ save_path,
22
+ ]
23
+ if dequantize:
24
+ cmd.append("--dequantize")
25
+ print("Fusing adapter into base model ...")
26
+ result = subprocess.run(cmd)
27
+ if result.returncode != 0:
28
+ raise SystemExit(result.returncode)
29
+
30
+
31
+ def _export_gguf(save_path: Path) -> Path:
32
+ """Convert a fused MLX model directory to GGUF (llama/mistral/mixtral archs)."""
33
+ import json
34
+
35
+ import mlx.core as mx
36
+ from mlx_lm import gguf as gguf_mod
37
+
38
+ with open(save_path / "config.json") as f:
39
+ config = json.load(f)
40
+
41
+ weights = {}
42
+ for part in sorted(save_path.glob("*.safetensors")):
43
+ weights.update(mx.load(str(part)))
44
+
45
+ # mlx-lm's converter permutes attention weights into non-contiguous views,
46
+ # which save_gguf rejects — force contiguity on its output.
47
+ orig_permute = gguf_mod.permute_weights
48
+ gguf_mod.permute_weights = lambda *a, **k: mx.contiguous(orig_permute(*a, **k))
49
+ try:
50
+ out = save_path / "ggml-model-f16.gguf"
51
+ gguf_mod.convert_to_gguf(save_path, weights, config, str(out))
52
+ finally:
53
+ gguf_mod.permute_weights = orig_permute
54
+ return out
55
+
56
+
57
+ # model_type values registered in mlx-swift-lm's LLMTypeRegistry (MLXLLM).
58
+ # Fused models outside this set won't load in an iOS app using mlx-swift-lm.
59
+ _IOS_SUPPORTED_MODEL_TYPES = {
60
+ "mistral", "mixtral", "llama", "phi", "phi3", "phimoe",
61
+ "gemma", "gemma2", "gemma3", "gemma3_text", "gemma3n",
62
+ "gemma4", "gemma4_unified", "gemma4_text",
63
+ "qwen2", "qwen3", "qwen3_moe", "qwen3_next",
64
+ "qwen3_5", "qwen3_5_moe", "qwen3_5_text",
65
+ "minicpm", "starcoder2", "cohere", "openelm", "internlm2",
66
+ "deepseek_v2", "deepseek_v3", "granite", "helium", "granitemoehybrid",
67
+ "glm4", "glm4_moe", "glm4_moe_lite", "falcon_h1", "bitnet", "smollm3",
68
+ "ernie4_5", "lfm2", "lfm2_moe", "exaone4", "gpt_oss", "olmoe", "olmo2",
69
+ "olmo3", "nemotron_h", "jamba", "mamba2", "mistral3", "apertus",
70
+ }
71
+
72
+ # On an 8 GB iPhone, an app with the Increased Memory Limit entitlement gets
73
+ # roughly 6 GB resident; weights + KV cache + app overhead must fit inside it.
74
+ _IOS_COMFORTABLE_GB = 2.2 # loads without entitlements on recent iPhones
75
+ _IOS_MAX_GB = 4.0 # needs Increased Memory Limit / Extended Virtual Addressing
76
+
77
+ _IOS_README = """\
78
+ # Run this model on iPhone / iPad with MLX Swift
79
+
80
+ This directory is a fused MLX model in the layout `mlx-swift-lm` loads directly.
81
+
82
+ ## Load it
83
+
84
+ Add the Swift package: https://github.com/ml-explore/mlx-swift-lm
85
+
86
+ ```swift
87
+ import MLXLLM
88
+ import MLXLMCommon
89
+
90
+ // From a local directory bundled with (or downloaded by) your app:
91
+ let modelDirectory: URL = ... // this folder on device
92
+ let container = try await LLMModelFactory.shared.loadContainer(
93
+ configuration: ModelConfiguration(directory: modelDirectory))
94
+
95
+ let result = try await container.perform { context in
96
+ let input = try await context.processor.prepare(
97
+ input: .init(prompt: "Hello!"))
98
+ return try MLXLMCommon.generate(
99
+ input: input, parameters: .init(), context: context)
100
+ }
101
+ ```
102
+
103
+ Weights this large should not ship inside the app bundle — download them on
104
+ first launch (Background Assets, or a direct download from your server or the
105
+ Hugging Face Hub via `troy push`).
106
+
107
+ ## Memory entitlements
108
+
109
+ Models over ~2 GB need one of these capabilities in Xcode
110
+ (Signing & Capabilities → + Capability):
111
+
112
+ - **Increased Memory Limit** (`com.apple.developer.kernel.increased-memory-limit`)
113
+ - **Extended Virtual Addressing**
114
+
115
+ Either is usually sufficient; devices with 8 GB RAM run 4-bit models up to
116
+ roughly 4 GB of weights. Test on the oldest device you target.
117
+ """
118
+
119
+
120
+ def _base_is_quantized(base: str) -> bool:
121
+ import json
122
+
123
+ p = Path(base)
124
+ if not p.exists():
125
+ from huggingface_hub import snapshot_download
126
+
127
+ p = Path(snapshot_download(base, allow_patterns=["config.json"]))
128
+ return "quantization" in json.loads((p / "config.json").read_text())
129
+
130
+
131
+ def _fuse_for_ios(base: str, adapter_path: str, save_path: str) -> None:
132
+ """Fuse for iOS without destroying the adapter.
133
+
134
+ Fusing a LoRA into 4-bit weights loses the deltas to quantization noise —
135
+ the exported model silently behaves like the base. So for a quantized
136
+ base: dequantize-fuse to fp16, then requantize fresh at 8 bits (verified
137
+ to preserve tuned behavior; 6 bits already degrades it).
138
+ """
139
+ import shutil
140
+ import tempfile
141
+
142
+ if not _base_is_quantized(base):
143
+ _fuse(base, adapter_path, save_path, dequantize=False)
144
+ return
145
+ from mlx_lm import convert
146
+
147
+ print("Quantized base: dequantize-fusing, then requantizing at 8 bits")
148
+ print("(fusing straight into 4-bit silently erases the adapter).")
149
+ out = Path(save_path)
150
+ if out.exists():
151
+ shutil.rmtree(out)
152
+ with tempfile.TemporaryDirectory() as td:
153
+ _fuse(base, adapter_path, td, dequantize=True)
154
+ convert(td, mlx_path=str(out), quantize=True, q_bits=8)
155
+
156
+
157
+ def _dir_weight_bytes(save_path: Path) -> int:
158
+ return sum(p.stat().st_size for p in save_path.glob("*.safetensors"))
159
+
160
+
161
+ def _export_ios(save_path: Path) -> None:
162
+ """Validate a fused MLX directory for mlx-swift-lm on iOS and add a how-to."""
163
+ import json
164
+
165
+ with open(save_path / "config.json") as f:
166
+ config = json.load(f)
167
+
168
+ problems = []
169
+ model_type = config.get("model_type", "?")
170
+ if model_type not in _IOS_SUPPORTED_MODEL_TYPES:
171
+ problems.append(
172
+ f"model_type `{model_type}` is not in mlx-swift-lm's registry — "
173
+ "it won't load on iOS without adding the architecture in Swift."
174
+ )
175
+ for name in ("tokenizer.json", "tokenizer_config.json"):
176
+ if not (save_path / name).exists():
177
+ problems.append(
178
+ f"{name} is missing — swift-transformers needs it to build "
179
+ "the tokenizer on device."
180
+ )
181
+
182
+ gb = _dir_weight_bytes(save_path) / 1e9
183
+ quantized = "quantization" in config
184
+ print(f"Weights: {gb:.2f} GB ({'quantized' if quantized else 'float'}), "
185
+ f"model_type: {model_type}")
186
+ if not quantized:
187
+ print("Tip: unquantized weights are heavy for iPhone — train from a "
188
+ "4-bit base (e.g. an mlx-community *-4bit model) for iOS.")
189
+ if gb <= _IOS_COMFORTABLE_GB:
190
+ print("iPhone fit: comfortable — loads on recent iPhones; the "
191
+ "Increased Memory Limit capability is still recommended.")
192
+ elif gb <= _IOS_MAX_GB:
193
+ print("iPhone fit: needs the Increased Memory Limit or Extended "
194
+ "Virtual Addressing capability; expect 8 GB-RAM devices only.")
195
+ else:
196
+ print(f"iPhone fit: unlikely — {gb:.1f} GB of weights exceeds what "
197
+ "an 8 GB iPhone can keep resident. Use a smaller or more "
198
+ "aggressively quantized base.")
199
+
200
+ (save_path / "README-iOS.md").write_text(_IOS_README)
201
+ print(f"Wrote {save_path / 'README-iOS.md'}")
202
+
203
+ if problems:
204
+ for p in problems:
205
+ print(f"WARNING: {p}")
206
+ raise SystemExit(1)
207
+
208
+
209
+ def run_export(
210
+ base: str,
211
+ adapter_path: str,
212
+ save_path: str,
213
+ fmt: str = "mlx",
214
+ dequantize: bool = False,
215
+ ) -> None:
216
+ """Fuse a LoRA adapter into the base model; optionally convert to GGUF.
217
+
218
+ fmt: "mlx" (fused MLX weights), "gguf" (also writes ggml-model-f16.gguf
219
+ for llama.cpp / Ollama / LM Studio; llama, mistral and mixtral
220
+ architectures, unquantized base), or "ios" (fused MLX weights validated
221
+ for mlx-swift-lm on iPhone/iPad, plus a README-iOS.md how-to).
222
+ """
223
+ if fmt == "ios" and dequantize:
224
+ print("Note: --dequantize with -f ios makes weights ~4x larger; "
225
+ "quantized weights are what you want on iPhone.")
226
+ if not Path(base).exists():
227
+ # mlx-lm's training path downloads a partial snapshot; fuse checks for
228
+ # a complete one. Top it up before fusing.
229
+ from huggingface_hub import snapshot_download
230
+
231
+ snapshot_download(base)
232
+
233
+ if fmt == "ios":
234
+ _fuse_for_ios(base, adapter_path, save_path)
235
+ else:
236
+ _fuse(base, adapter_path, save_path, dequantize)
237
+ print(f"Fused model: {Path(save_path).resolve()}")
238
+ if fmt == "gguf":
239
+ out = _export_gguf(Path(save_path))
240
+ print(f"GGUF: {out.resolve()}")
241
+ elif fmt == "ios":
242
+ _export_ios(Path(save_path))