mtplx 2.0.2__tar.gz → 2.2.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (354) hide show
  1. {mtplx-2.0.2 → mtplx-2.2.0}/CHANGELOG.md +47 -0
  2. {mtplx-2.0.2 → mtplx-2.2.0}/PKG-INFO +3 -3
  3. mtplx-2.2.0/mtplx/backends/hy_v3_mtp.py +69 -0
  4. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/backends/registry.py +47 -3
  5. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/cache_bank/cold_tier.py +128 -101
  6. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/cache_state.py +20 -1
  7. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/cli.py +15 -6
  8. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/commands/forge.py +161 -4
  9. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/commands/public.py +224 -33
  10. mtplx-2.2.0/mtplx/context_copy.py +114 -0
  11. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/engine_session.py +88 -2
  12. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/generation.py +845 -30
  13. mtplx-2.2.0/mtplx/hy_v3_mtp_patch.py +165 -0
  14. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/model_catalog.py +31 -4
  15. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/model_scheduler.py +43 -0
  16. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/mtp_patch.py +63 -1
  17. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/profiles.py +11 -1
  18. mtplx-2.2.0/mtplx/qwen3_5_mtp_patch.py +360 -0
  19. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/runtime.py +99 -1
  20. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/server/dashboard_state.py +3 -0
  21. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/server/openai.py +864 -67
  22. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/session_bank.py +118 -2
  23. mtplx-2.2.0/mtplx/templates/qwen36_froggeric_v19/chat_template.jinja +267 -0
  24. mtplx-2.2.0/mtplx/templates/qwen36_froggeric_v21_3/chat_template.jinja +329 -0
  25. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/version.py +2 -2
  26. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/vision/splice.py +57 -0
  27. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx.egg-info/PKG-INFO +3 -3
  28. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx.egg-info/SOURCES.txt +43 -0
  29. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx.egg-info/requires.txt +2 -2
  30. {mtplx-2.0.2 → mtplx-2.2.0}/pyproject.toml +11 -5
  31. mtplx-2.2.0/scripts/fp16_turbo_exactness_20260707.py +303 -0
  32. mtplx-2.2.0/scripts/kvcache_exactness_audit_20260703.py +303 -0
  33. mtplx-2.2.0/scripts/kvcache_soak_20260703.py +227 -0
  34. mtplx-2.2.0/scripts/kvcache_warm_probe_20260703.py +357 -0
  35. mtplx-2.2.0/scripts/ocspeed-20260703/abba_final.sh +85 -0
  36. mtplx-2.2.0/scripts/ocspeed-20260703/accept_depth_probe.py +209 -0
  37. mtplx-2.2.0/scripts/ocspeed-20260703/arm_runner.sh +38 -0
  38. mtplx-2.2.0/scripts/ocspeed-20260703/forward_depth_bisect.py +132 -0
  39. mtplx-2.2.0/scripts/ocspeed-20260703/longgen_probe.py +85 -0
  40. mtplx-2.2.0/scripts/ocspeed-20260703/loop_degradation_probe.py +81 -0
  41. mtplx-2.2.0/scripts/ocspeed-20260703/paired_longgen.sh +61 -0
  42. mtplx-2.2.0/scripts/ocspeed-20260703/sdpa_microbench.py +66 -0
  43. mtplx-2.2.0/scripts/ocspeed-20260703/snapshot_tail.py +60 -0
  44. mtplx-2.2.0/scripts/pillar_gate_qa.py +317 -0
  45. mtplx-2.2.0/scripts/prodqa-20260703/candidate_truecold_20260703.py +101 -0
  46. mtplx-2.2.0/scripts/prodqa-20260703/cold_pair_recheck.py +88 -0
  47. mtplx-2.2.0/scripts/prodqa-20260703/competitors_20260703.sh +63 -0
  48. mtplx-2.2.0/scripts/prodqa-20260703/decode_gap_matrix.py +110 -0
  49. mtplx-2.2.0/scripts/prodqa-20260703/hermes_pty_driver.py +87 -0
  50. mtplx-2.2.0/scripts/prodqa-20260703/omlx_probe.py +63 -0
  51. mtplx-2.2.0/scripts/prodqa-20260703/pi_pty_driver.py +77 -0
  52. mtplx-2.2.0/scripts/prodqa-20260703/pillar_ab_20260703.py +169 -0
  53. mtplx-2.2.0/scripts/prodqa-20260703/tool_gauntlet_20260703.sh +40 -0
  54. mtplx-2.2.0/scripts/quality_fp16_parent_logitdiff_20260707.py +135 -0
  55. {mtplx-2.0.2 → mtplx-2.2.0}/scripts/release_macos_v1.sh +22 -0
  56. mtplx-2.2.0/tests/test_ar_batch_penalties.py +114 -0
  57. {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_artifacts.py +0 -1
  58. mtplx-2.2.0/tests/test_cold_tier_write_budget.py +123 -0
  59. mtplx-2.2.0/tests/test_context_copy_stats.py +405 -0
  60. mtplx-2.2.0/tests/test_device_draft_core.py +117 -0
  61. {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_engine_session_env.py +36 -2
  62. {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_forge_cli.py +129 -4
  63. mtplx-2.2.0/tests/test_gdn_boundary_retention.py +150 -0
  64. {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_generation_sustained.py +24 -0
  65. mtplx-2.2.0/tests/test_hy_v3_mtp_backend.py +72 -0
  66. mtplx-2.2.0/tests/test_memory_pressure_guard.py +169 -0
  67. {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_model_catalog.py +43 -9
  68. mtplx-2.2.0/tests/test_mtp_alias_load_path.py +73 -0
  69. {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_mtp_patch.py +61 -0
  70. {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_openai_bridge.py +6 -1
  71. mtplx-2.2.0/tests/test_orphan_tool_markup.py +85 -0
  72. {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_postcommit_prefix_reuse.py +3 -3
  73. mtplx-2.2.0/tests/test_postcommit_resolve_for_request.py +101 -0
  74. {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_postcommit_tools_plumbing.py +3 -3
  75. {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_postcommit_wait_integration.py +1 -1
  76. {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_profiles.py +1 -1
  77. {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_public_cli.py +338 -4
  78. mtplx-2.2.0/tests/test_qwen3_5_mtp_backend.py +66 -0
  79. {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_server_openai.py +107 -3
  80. {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_session_bank_env_caps.py +9 -6
  81. mtplx-2.2.0/tests/test_ssd_boundary_repersist.py +221 -0
  82. {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_tool_aware_stream_translator.py +29 -1
  83. mtplx-2.2.0/tests/test_vision_session_cache.py +195 -0
  84. {mtplx-2.0.2 → mtplx-2.2.0}/CITATION.cff +0 -0
  85. {mtplx-2.0.2 → mtplx-2.2.0}/LICENSE +0 -0
  86. {mtplx-2.0.2 → mtplx-2.2.0}/MANIFEST.in +0 -0
  87. {mtplx-2.0.2 → mtplx-2.2.0}/NOTICE +0 -0
  88. {mtplx-2.0.2 → mtplx-2.2.0}/README.md +0 -0
  89. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/__init__.py +0 -0
  90. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/adaptive.py +0 -0
  91. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/app_settings.py +0 -0
  92. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/artifacts.py +0 -0
  93. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/attention_context.py +0 -0
  94. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/attention_split.py +0 -0
  95. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/backends/__init__.py +0 -0
  96. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/backends/deepseek_mtp.py +0 -0
  97. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/backends/descriptors.py +0 -0
  98. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/backends/gemma4_assistant.py +0 -0
  99. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/backends/glm_mtp.py +0 -0
  100. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/backends/mimo_mtp.py +0 -0
  101. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/backends/nemotron_h_mtp.py +0 -0
  102. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/backends/qwen3_next.py +0 -0
  103. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/backends/step3p5_mtp.py +0 -0
  104. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/batching/__init__.py +0 -0
  105. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/batching/admission.py +0 -0
  106. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/batching/buckets.py +0 -0
  107. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/batching/scheduler.py +0 -0
  108. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/batching/state.py +0 -0
  109. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/benchmarks/__init__.py +0 -0
  110. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/benchmarks/prompts/aime_2026.jsonl +0 -0
  111. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/benchmarks/prompts/calibration_coding.jsonl +0 -0
  112. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/benchmarks/prompts/default.jsonl +0 -0
  113. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/benchmarks/prompts/flappy.jsonl +0 -0
  114. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/benchmarks/prompts/long_code.jsonl +0 -0
  115. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/benchmarks/prompts/long_code_uncapped.jsonl +0 -0
  116. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/benchmarks/prompts/python_modules_long.jsonl +0 -0
  117. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/benchmarks/runners/__init__.py +0 -0
  118. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/benchmarks/runners/aime.py +0 -0
  119. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/benchmarks/runners/batch_equivalence.py +0 -0
  120. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/benchmarks/runners/capture_commit_equivalence.py +0 -0
  121. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/benchmarks/runners/competitor_baselines.py +0 -0
  122. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/benchmarks/runners/contract_probe.py +0 -0
  123. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/benchmarks/runners/harness.py +0 -0
  124. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/benchmarks/runners/mtp1_gate.py +0 -0
  125. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/benchmarks/runners/mtp1_sampler_smoke.py +0 -0
  126. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/benchmarks/runners/mtp_adaptive.py +0 -0
  127. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/benchmarks/runners/mtp_chain_probe.py +0 -0
  128. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/benchmarks/runners/mtp_depth_grid.py +0 -0
  129. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/benchmarks/runners/mtp_depth_sweep.py +0 -0
  130. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/benchmarks/runners/mtp_tree_probe.py +0 -0
  131. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/benchmarks/runners/multi_qmv_probe.py +0 -0
  132. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/benchmarks/runners/preflight.py +0 -0
  133. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/benchmarks/runners/runtime_smoke.py +0 -0
  134. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/benchmarks/runners/session_bank.py +0 -0
  135. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/benchmarks/runners/truth.py +0 -0
  136. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/benchmarks/runners/verify_profile.py +0 -0
  137. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/benchmarks/runners/verify_qmm_probe.py +0 -0
  138. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/benchmarks/runners/verify_ratio.py +0 -0
  139. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/benchmarks/schema.py +0 -0
  140. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/benchmarks/validators/__init__.py +0 -0
  141. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/benchmarks/validators/aime.py +0 -0
  142. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/benchmarks/validators/basic.py +0 -0
  143. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/block_attention.py +0 -0
  144. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/cache_bank/__init__.py +0 -0
  145. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/cache_bank/codec.py +0 -0
  146. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/chat_encoding.py +0 -0
  147. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/commands/__init__.py +0 -0
  148. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/compressed_tensors.py +0 -0
  149. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/config.py +0 -0
  150. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/constants.py +0 -0
  151. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/correctors/__init__.py +0 -0
  152. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/correctors/diagonal_affine.py +0 -0
  153. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/correctors/low_rank.py +0 -0
  154. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/daemon_client.py +0 -0
  155. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/dashboard/__init__.py +0 -0
  156. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/dashboard/_static/assets/index-BYd4MFty.css +0 -0
  157. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/dashboard/_static/assets/index-COqTDxL-.js +0 -0
  158. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/dashboard/_static/index.html +0 -0
  159. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/deepseek_mtp_patch.py +0 -0
  160. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/default_models.py +0 -0
  161. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/diagnostics.py +0 -0
  162. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/draft_lm_head.py +0 -0
  163. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/draft_sampling.py +0 -0
  164. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/env.py +0 -0
  165. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/errors.py +0 -0
  166. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/expert_layout.py +0 -0
  167. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/fan_mode.py +0 -0
  168. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/fast_sampling.py +0 -0
  169. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/gdn_capture.py +0 -0
  170. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/gemma4_pair.py +0 -0
  171. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/glm_mtp_patch.py +0 -0
  172. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/graphbank.py +0 -0
  173. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/hardware.py +0 -0
  174. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/hf_loader.py +0 -0
  175. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/kernel_selfcheck.py +0 -0
  176. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/kernels/__init__.py +0 -0
  177. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/kernels/copy_leaf.py +0 -0
  178. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/kernels/fused_norm.py +0 -0
  179. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/kernels/lm_head_topk.py +0 -0
  180. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/kernels/logits_topk.py +0 -0
  181. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/kernels/native_gdn_tail.py +0 -0
  182. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/kernels/sdpa_2pass.py +0 -0
  183. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/kernels/sdpa_2pass_paged.py +0 -0
  184. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/kernels/sdpa_2pass_paged_q8.py +0 -0
  185. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/kernels/sdpa_gqa_packed.py +0 -0
  186. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/kernels/verify_mlp_fused.py +0 -0
  187. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/kpi/__init__.py +0 -0
  188. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/kpi/reference_vllm.py +0 -0
  189. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/kpi/runtime_kpis.py +0 -0
  190. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/kv_quant.py +0 -0
  191. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/loop_guard.py +0 -0
  192. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/mimo_mtp_patch.py +0 -0
  193. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/mtp_activation_stats.py +0 -0
  194. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/mtp_adapters.py +0 -0
  195. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/native_mlp.py +0 -0
  196. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/nax_verify.py +0 -0
  197. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/nemotron_h_mtp_patch.py +0 -0
  198. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/opencode.py +0 -0
  199. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/pi.py +0 -0
  200. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/prefill_bench.py +0 -0
  201. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/proposal_reranker.py +0 -0
  202. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/reasoning_codecs.py +0 -0
  203. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/runtime_options.py +0 -0
  204. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/sampling.py +0 -0
  205. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/server/__init__.py +0 -0
  206. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/server/omlx_bridge/__init__.py +0 -0
  207. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/server/omlx_bridge/adapter.py +0 -0
  208. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/server/omlx_bridge/thinking.py +0 -0
  209. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/server/omlx_bridge/tool_calling.py +0 -0
  210. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/server_urls.py +0 -0
  211. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/speculative.py +0 -0
  212. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/step3p5_mtp_patch.py +0 -0
  213. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/swival.py +0 -0
  214. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/thermal.py +0 -0
  215. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/thermal_sidecar.py +0 -0
  216. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/trace_parity.py +0 -0
  217. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/turboquant.py +0 -0
  218. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/ui/__init__.py +0 -0
  219. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/ui/banner.py +0 -0
  220. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/ui/chat_printer.py +0 -0
  221. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/ui/download_progress.py +0 -0
  222. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/ui/onboarding.py +0 -0
  223. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/ui/panels.py +0 -0
  224. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/ui/progress.py +0 -0
  225. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/verify_kernels.py +0 -0
  226. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/verify_qmv.py +0 -0
  227. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/vision/__init__.py +0 -0
  228. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/vision/processing.py +0 -0
  229. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/vision/qwen3_vl_tower.py +0 -0
  230. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx.egg-info/dependency_links.txt +0 -0
  231. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx.egg-info/entry_points.txt +0 -0
  232. {mtplx-2.0.2 → mtplx-2.2.0}/mtplx.egg-info/top_level.txt +0 -0
  233. {mtplx-2.0.2 → mtplx-2.2.0}/scripts/agent_user_path_qa.py +0 -0
  234. {mtplx-2.0.2 → mtplx-2.2.0}/scripts/aime_serve_gate.py +0 -0
  235. {mtplx-2.0.2 → mtplx-2.2.0}/scripts/aime_shape_memory_bench.py +0 -0
  236. {mtplx-2.0.2 → mtplx-2.2.0}/scripts/collect_mtp_activation_stats.py +0 -0
  237. {mtplx-2.0.2 → mtplx-2.2.0}/scripts/collect_mtp_hidden_calib.py +0 -0
  238. {mtplx-2.0.2 → mtplx-2.2.0}/scripts/combine_mtp_adapters.py +0 -0
  239. {mtplx-2.0.2 → mtplx-2.2.0}/scripts/compiled_verify_exactness.py +0 -0
  240. {mtplx-2.0.2 → mtplx-2.2.0}/scripts/convert_step37_to_mtplx_step3p5.py +0 -0
  241. {mtplx-2.0.2 → mtplx-2.2.0}/scripts/eval_mtp_corrector.py +0 -0
  242. {mtplx-2.0.2 → mtplx-2.2.0}/scripts/filter_mtp_adapter.py +0 -0
  243. {mtplx-2.0.2 → mtplx-2.2.0}/scripts/fresh_venv_smoke.sh +0 -0
  244. {mtplx-2.0.2 → mtplx-2.2.0}/scripts/hygiene_scan.sh +0 -0
  245. {mtplx-2.0.2 → mtplx-2.2.0}/scripts/install_macos.sh +0 -0
  246. {mtplx-2.0.2 → mtplx-2.2.0}/scripts/install_preview_global.sh +0 -0
  247. {mtplx-2.0.2 → mtplx-2.2.0}/scripts/make_fp16_precision_sibling.py +0 -0
  248. {mtplx-2.0.2 → mtplx-2.2.0}/scripts/midform_gate.py +0 -0
  249. {mtplx-2.0.2 → mtplx-2.2.0}/scripts/opencode_concurrency_qa.py +0 -0
  250. {mtplx-2.0.2 → mtplx-2.2.0}/scripts/phase0h_paged_verifier_exactness.py +0 -0
  251. {mtplx-2.0.2 → mtplx-2.2.0}/scripts/probe_mlx_pr3026_qsdpa.py +0 -0
  252. {mtplx-2.0.2 → mtplx-2.2.0}/scripts/probe_mx_compile_buckets.py +0 -0
  253. {mtplx-2.0.2 → mtplx-2.2.0}/scripts/probe_paged_gqa_sdpa_routes.py +0 -0
  254. {mtplx-2.0.2 → mtplx-2.2.0}/scripts/r1_chisquare_verifier_correctness.py +0 -0
  255. {mtplx-2.0.2 → mtplx-2.2.0}/scripts/run_context_degradation_diagnostics.py +0 -0
  256. {mtplx-2.0.2 → mtplx-2.2.0}/scripts/serve_openai_mtplx.py +0 -0
  257. {mtplx-2.0.2 → mtplx-2.2.0}/scripts/session_cache_followup_qa.py +0 -0
  258. {mtplx-2.0.2 → mtplx-2.2.0}/scripts/sparkle_rehearsal_kit.sh +0 -0
  259. {mtplx-2.0.2 → mtplx-2.2.0}/scripts/step_acceptance.py +0 -0
  260. {mtplx-2.0.2 → mtplx-2.2.0}/scripts/step_smoke.py +0 -0
  261. {mtplx-2.0.2 → mtplx-2.2.0}/scripts/train_mtp_adapter_c4.py +0 -0
  262. {mtplx-2.0.2 → mtplx-2.2.0}/scripts/validate_step_mtp_injector.py +0 -0
  263. {mtplx-2.0.2 → mtplx-2.2.0}/setup.cfg +0 -0
  264. {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_adaptive.py +0 -0
  265. {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_aime_serve_gate.py +0 -0
  266. {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_attention_split.py +0 -0
  267. {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_background_warmup.py +0 -0
  268. {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_batching_foundation.py +0 -0
  269. {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_cache_bank.py +0 -0
  270. {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_cache_state.py +0 -0
  271. {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_cli_parity_tools.py +0 -0
  272. {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_compressed_tensors.py +0 -0
  273. {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_config.py +0 -0
  274. {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_config_profile_precedence.py +0 -0
  275. {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_context_degradation_profiles.py +0 -0
  276. {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_correctors.py +0 -0
  277. {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_daemon_client.py +0 -0
  278. {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_dashboard_endpoints.py +0 -0
  279. {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_default_models.py +0 -0
  280. {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_diagnostics.py +0 -0
  281. {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_download_progress.py +0 -0
  282. {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_draft_lm_head.py +0 -0
  283. {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_engine_session_concurrency.py +0 -0
  284. {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_fast_sampling.py +0 -0
  285. {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_generation_store_on_prefill.py +0 -0
  286. {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_graphbank_compiled_verify.py +0 -0
  287. {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_hf_loader.py +0 -0
  288. {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_hygiene_scan.py +0 -0
  289. {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_idle_postcommit_subagent.py +0 -0
  290. {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_kernel_selfcheck.py +0 -0
  291. {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_lazy_snapshot_cow.py +0 -0
  292. {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_loop_guard.py +0 -0
  293. {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_max_idle_watchdog.py +0 -0
  294. {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_max_lifecycle.py +0 -0
  295. {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_metal_memory_caps.py +0 -0
  296. {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_midform_gate.py +0 -0
  297. {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_model_scheduler.py +0 -0
  298. {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_mtp_activation_stats.py +0 -0
  299. {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_mtp_adapters.py +0 -0
  300. {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_mtp_depth_sweep.py +0 -0
  301. {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_nax_verify.py +0 -0
  302. {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_no_mlx_imports.py +0 -0
  303. {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_omlx_bridge.py +0 -0
  304. {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_onboarding.py +0 -0
  305. {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_opencode.py +0 -0
  306. {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_penalties.py +0 -0
  307. {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_penalty_request_wiring.py +0 -0
  308. {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_persistent_replay.py +0 -0
  309. {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_phase0h_paged_verifier_exactness.py +0 -0
  310. {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_policy_fingerprint_stability.py +0 -0
  311. {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_postcommit_wait.py +0 -0
  312. {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_prefill_bench.py +0 -0
  313. {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_prefill_chunk_defaults.py +0 -0
  314. {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_prefill_tps_regression.py +0 -0
  315. {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_prompt_encoding.py +0 -0
  316. {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_reasoning_stream_split.py +0 -0
  317. {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_runtime_kpis.py +0 -0
  318. {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_sampling.py +0 -0
  319. {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_scoped_reasoning_history.py +0 -0
  320. {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_sdpa_gqa_packed.py +0 -0
  321. {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_session_bank.py +0 -0
  322. {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_step3p5_mtp_patch.py +0 -0
  323. {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_sustained_long_context_qa.py +0 -0
  324. {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_thermal.py +0 -0
  325. {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_thermal_sidecar.py +0 -0
  326. {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_trace_parity.py +0 -0
  327. {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_turboquant.py +0 -0
  328. {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_turboquant_fallback.py +0 -0
  329. {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_ui_progress.py +0 -0
  330. {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_validators.py +0 -0
  331. {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_vision_tower.py +0 -0
  332. {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_vllm_reference.py +0 -0
  333. {mtplx-2.0.2 → mtplx-2.2.0}/vllm_metal/__init__.py +0 -0
  334. {mtplx-2.0.2 → mtplx-2.2.0}/vllm_metal/metal/README.md +0 -0
  335. {mtplx-2.0.2 → mtplx-2.2.0}/vllm_metal/metal/__init__.py +0 -0
  336. {mtplx-2.0.2 → mtplx-2.2.0}/vllm_metal/metal/build.py +0 -0
  337. {mtplx-2.0.2 → mtplx-2.2.0}/vllm_metal/metal/constants.py +0 -0
  338. {mtplx-2.0.2 → mtplx-2.2.0}/vllm_metal/metal/kernels_v1/copy_blocks.metal +0 -0
  339. {mtplx-2.0.2 → mtplx-2.2.0}/vllm_metal/metal/kernels_v1/float8.metal +0 -0
  340. {mtplx-2.0.2 → mtplx-2.2.0}/vllm_metal/metal/kernels_v1/gather_kv_cache.metal +0 -0
  341. {mtplx-2.0.2 → mtplx-2.2.0}/vllm_metal/metal/kernels_v1/kv_scale_update.metal +0 -0
  342. {mtplx-2.0.2 → mtplx-2.2.0}/vllm_metal/metal/kernels_v1/pagedattention.metal +0 -0
  343. {mtplx-2.0.2 → mtplx-2.2.0}/vllm_metal/metal/kernels_v1/reshape_and_cache.metal +0 -0
  344. {mtplx-2.0.2 → mtplx-2.2.0}/vllm_metal/metal/kernels_v1/utils.metal +0 -0
  345. {mtplx-2.0.2 → mtplx-2.2.0}/vllm_metal/metal/kernels_v2/copy_blocks.metal +0 -0
  346. {mtplx-2.0.2 → mtplx-2.2.0}/vllm_metal/metal/kernels_v2/float8.metal +0 -0
  347. {mtplx-2.0.2 → mtplx-2.2.0}/vllm_metal/metal/kernels_v2/gather_kv_cache.metal +0 -0
  348. {mtplx-2.0.2 → mtplx-2.2.0}/vllm_metal/metal/kernels_v2/gdn_linear_attention.metal +0 -0
  349. {mtplx-2.0.2 → mtplx-2.2.0}/vllm_metal/metal/kernels_v2/kv_scale_update.metal +0 -0
  350. {mtplx-2.0.2 → mtplx-2.2.0}/vllm_metal/metal/kernels_v2/pagedattention.metal +0 -0
  351. {mtplx-2.0.2 → mtplx-2.2.0}/vllm_metal/metal/kernels_v2/reshape_and_cache.metal +0 -0
  352. {mtplx-2.0.2 → mtplx-2.2.0}/vllm_metal/metal/kernels_v2/turboquant.metal +0 -0
  353. {mtplx-2.0.2 → mtplx-2.2.0}/vllm_metal/metal/kernels_v2/utils.metal +0 -0
  354. {mtplx-2.0.2 → mtplx-2.2.0}/vllm_metal/metal/paged_ops.cpp +0 -0
@@ -4,6 +4,53 @@ All notable user-facing changes to MTPLX. The format is based on
4
4
  [Keep a Changelog](https://keepachangelog.com/en/1.1.0/); versions follow
5
5
  [Semantic Versioning](https://semver.org/).
6
6
 
7
+ ## [2.2.0] - 2026-07-19
8
+
9
+ The copy-drafting and small-Mac release. Decoding: context-copy
10
+ (prompt-lookup) drafting lands on by default (PR #151 by lBroth) with an
11
+ exact temperature path — copied blocks are accepted with the target's own
12
+ shaped probability, so the output distribution is unchanged at any
13
+ temperature; measured +53% on edit-heavy agent turns at temp 0.6, parity
14
+ on novel text (disable with MTPLX_CONTEXT_COPY=0). Models: the 4B
15
+ zero-acceptance defect (#176) is root-caused and fixed — the engine heals
16
+ raw delta-encoded MTP sidecars at load so existing downloads recover
17
+ without re-downloading, the 4B Speed artifact is rebuilt (227.8 tok/s D3
18
+ on M5 Max, 1.71x AR), and a new 4B Quality artifact ships (191.7 tok/s
19
+ D3, 2.19x — the largest MTP multiplier in the fleet); sub-16GB Macs get
20
+ first-class catalog recommendations. Tune (#177): a 0.0-acceptance depth
21
+ can never win, be saved, or be replayed, and poisoned records are
22
+ quarantined at load. Forge: FP16 precision option with M1/M2 auto-select
23
+ (#166); the fp16 cast can no longer corrupt a sidecar in place;
24
+ degenerate sidecars are quarantined on re-forge. Server: SSD session
25
+ writer crash fixed — encode at enqueue (#169); live requests preempt idle
26
+ cache maintenance. MoE: mtp_depth_max is a ceiling, not the default; the
27
+ 35B-A3B launches at its measured D2 (#174 part 1 by davidtai). Also: the
28
+ Qwen 3.6 27B AR decode-trace crash fix (#167 by davidtai), truthful
29
+ per-model profile display, hybrid-model boundary retention across append
30
+ churn, and an experimental --draft-core device (opt-in). Full details in
31
+ docs/releases/v2.2.0.md.
32
+
33
+ ## [2.1.0] - 2026-07-17
34
+
35
+ The community-fixes release. Memory: the v2.x reports are root-caused and
36
+ closed (MLX allocator cache bounded by default, per-session admission
37
+ re-clamped on sub-96GB machines, paged pool bounded by the context
38
+ window, pressure responder redesigned, q4 kv-quant crash fixed, new
39
+ `--memory-budget` knob). Agent sessions: warm prefix reuse survives every
40
+ tool turn (#121), hybrid-model near-prefix restores no longer collapse to
41
+ the oldest boundary (measured 0.4s instead of 33.8s on a 22k follow-up),
42
+ restart-warm sessions keep their boundary records across SSD generations
43
+ (#159, #144), and cache hits are reported in standard `usage` fields.
44
+ Sampling: presence and frequency penalties fixed in the batched AR lane
45
+ (#156). App: startup and update hang fixed plus a full subprocess
46
+ watchdog sweep (#158), the Hermes tile launches Hermes Desktop, raw
47
+ tool-call XML no longer leaks into no-tools chats (#160). CLI: `start
48
+ opencode` serves the same lane the app serves. Backends: qwen3_5_mtp and
49
+ hy_v3 land (#142, #147). Performance: the model-owner thread is
50
+ QoS-pinned for 8 to 10% faster decode under real multitasking load.
51
+ Operators: `MTPLX_COMPILED_VERIFY_MAX_CONTEXT` is env-overridable. Full
52
+ details in docs/releases/v2.1.0.md.
53
+
7
54
  ## [2.0.2] - 2026-07-09
8
55
 
9
56
  The agent-reliability release: multi-turn agent sessions now render
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: mtplx
3
- Version: 2.0.2
3
+ Version: 2.2.0
4
4
  Summary: Native MTP speculative decoding for Qwen3-Next on Apple Silicon.
5
5
  Author-email: Youssof Altoukhi <business@youssofal.com>
6
6
  License-Expression: Apache-2.0
@@ -25,9 +25,9 @@ License-File: LICENSE
25
25
  License-File: NOTICE
26
26
  Requires-Dist: fastapi>=0.136
27
27
  Requires-Dist: huggingface-hub>=0.36
28
- Requires-Dist: mlx<0.32,>=0.31; sys_platform == "darwin" and platform_machine == "arm64"
28
+ Requires-Dist: mlx<0.33,>=0.31; sys_platform == "darwin" and platform_machine == "arm64"
29
29
  Requires-Dist: mlx-lm<0.32,>=0.31; sys_platform == "darwin" and platform_machine == "arm64"
30
- Requires-Dist: transformers<5.13; sys_platform == "darwin" and platform_machine == "arm64"
30
+ Requires-Dist: transformers!=5.13.0,<5.14; sys_platform == "darwin" and platform_machine == "arm64"
31
31
  Requires-Dist: nanobind>=2; sys_platform == "darwin" and platform_machine == "arm64"
32
32
  Requires-Dist: numpy>=2
33
33
  Requires-Dist: pydantic>=2
@@ -0,0 +1,69 @@
1
+ """Tencent Hy3 (hy_v3) native MTP backend facade.
2
+
3
+ Hy3 ships one appended MTP layer (num_nextn_predict_layers=1): a full MoE
4
+ decoder layer fed concat[RMSNorm(next-token embedding), RMSNorm(trunk
5
+ pre-final-norm hidden state)] through an eh_proj down-projection, sharing the
6
+ trunk's embeddings and lm_head. The MLX reference implementation exposes it as
7
+ ``Model.predict_next_tokens(hidden, token_ids, cache)`` with
8
+ ``return_hidden_states=True`` on the trunk forward (see
9
+ mlx-lm ``models/hy_v3.py``, MTP revision).
10
+
11
+ Like the GLM/DeepSeek facades, drafting and verification are wired through the
12
+ shared speculative sampler in ``generation.py``; this facade gates execution
13
+ behind the verified runtime contract.
14
+ """
15
+
16
+ from __future__ import annotations
17
+
18
+ from pathlib import Path
19
+ from typing import Any
20
+
21
+ from . import DraftTokens, ModelState, MTPBackend, VerifyOutput
22
+ from mtplx.profiles import DEFAULT_PROFILE_NAME
23
+
24
+
25
+ class HyV3MTPBackend(MTPBackend):
26
+ arch_id = "hy-v3-mtp"
27
+
28
+ def load(self, model_path: Path) -> ModelState:
29
+ from mtplx.mtp_patch import MTPContract
30
+ from mtplx.runtime import load
31
+
32
+ runtime = load(model_path, mtp=True, contract=MTPContract())
33
+ return ModelState(
34
+ model_path=Path(model_path),
35
+ runtime=runtime,
36
+ metadata={"arch_id": self.arch_id, "contract_gated": True},
37
+ )
38
+
39
+ def verify(self, state: ModelState, draft_tokens: DraftTokens, hidden: Any) -> VerifyOutput:
40
+ raise NotImplementedError("HyV3MTPBackend.verify is wired through generation.py")
41
+
42
+ def propose(self, state: ModelState, hidden: Any) -> DraftTokens:
43
+ raise NotImplementedError("HyV3MTPBackend.propose is wired through generation.py")
44
+
45
+ def recommended_profile(self) -> str:
46
+ return DEFAULT_PROFILE_NAME
47
+
48
+ def health(self) -> dict[str, Any]:
49
+ return {
50
+ "arch_id": self.arch_id,
51
+ "runtime_path": "mtplx.runtime + mtplx.hy_v3_mtp_patch + mtplx.generation",
52
+ "support_level": "experimental-native-contract-gated",
53
+ "contract_required": True,
54
+ "supported_model_types": ["hy_v3"],
55
+ "mtp_depth_max": 1,
56
+ "notes": (
57
+ "Single appended NextN layer with its own 192-expert MoE MLP, "
58
+ "sigmoid top-8 routing with expert bias, eh_proj over "
59
+ "concat[enorm(embedding), hnorm(hidden)], shared embeddings "
60
+ "and head. Draft layer consumes the trunk pre-final-norm "
61
+ "hidden state. Verification is exact rejection sampling in "
62
+ "generation.py; the MLX reference (mlx-lm hy_v3 MTP revision) "
63
+ "verifies greedily and is temp-0 exact."
64
+ ),
65
+ "references": [
66
+ "REFERENCES:TOOLS/vllm-official-main/vllm/model_executor/models/hy_v3_mtp.py",
67
+ "REFERENCES:TOOLS/mlx-lm/mlx_lm/models/hy_v3.py",
68
+ ],
69
+ }
@@ -21,6 +21,7 @@ SUPPORTED_ARCH_IDS = {
21
21
  "nemotron-h-mtp",
22
22
  "gemma4-assistant-mtp",
23
23
  "step3p5-mtp",
24
+ "hy-v3-mtp",
24
25
  }
25
26
 
26
27
  TIER_VERIFIED = "verified"
@@ -411,10 +412,22 @@ ARCHITECTURE_CATALOG: dict[str, ArchitectureSupport] = {
411
412
  display_name="HY V3 MTP",
412
413
  family="hy",
413
414
  backend="hy_v3_mtp",
414
- support_level="recognized-backend-pending",
415
- runtime_compatibility="recognized-backend-pending",
415
+ support_level="experimental-native-contract-gated",
416
+ runtime_compatibility="native-contract-gated",
417
+ can_run_verified=True,
416
418
  aliases=("hy_v3_mtp", "hy_v3"),
417
- references=("REFERENCES:TOOLS/vllm-official-main/vllm/model_executor/models/hy_v3_mtp.py",),
419
+ family_gate="appended-layer-mtp-markers",
420
+ references=(
421
+ "REFERENCES:TOOLS/vllm-official-main/vllm/model_executor/models/hy_v3_mtp.py",
422
+ "REFERENCES:TOOLS/mlx-lm/mlx_lm/models/hy_v3.py",
423
+ ),
424
+ notes=(
425
+ "Hy3 ships one appended NextN layer with its own 192-expert MoE, "
426
+ "eh_proj over concat[enorm(embedding), hnorm(hidden)], and shared "
427
+ "embeddings/head. The mlx-lm hy_v3 MTP revision exposes the head "
428
+ "natively (predict_next_tokens), so injection binds the existing "
429
+ "surface rather than grafting weights."
430
+ ),
418
431
  ),
419
432
  "generic-mtp": ArchitectureSupport(
420
433
  arch_id="generic-mtp",
@@ -794,6 +807,35 @@ def _has_all_suffixes_under_prefixes(
794
807
  return True
795
808
 
796
809
 
810
+ _HY_V3_MTP_MARKER_SUFFIXES = (
811
+ "enorm.weight",
812
+ "hnorm.weight",
813
+ "eh_proj.weight",
814
+ "final_layernorm.weight",
815
+ )
816
+
817
+
818
+ def _passes_hy_v3_gate(inspection: Any) -> bool:
819
+ """Hy3's appended MTP block lives directly under an ``mtp.`` prefix
820
+ (``mtp.enorm.weight``, ``mtp.hnorm.weight``, ``mtp.eh_proj.weight``,
821
+ ``mtp.final_layernorm.weight``, ``mtp.layer.*``) rather than the
822
+ ``mtp.layers.{idx}.`` nesting DeepSeek/GLM/Step use, so it needs its own
823
+ gate instead of `_passes_appended_layer_gate` (verified against the
824
+ shipped `hy3-demolition-mlx-*-mtp` checkpoints' safetensors index)."""
825
+ keys = _weight_keys(inspection)
826
+ if not keys:
827
+ return False
828
+ count = int(getattr(inspection, "mtp_num_hidden_layers", 0) or 0)
829
+ if count <= 0:
830
+ return False
831
+ return _has_marker_under_prefixes(
832
+ keys,
833
+ ("mtp.",),
834
+ _HY_V3_MTP_MARKER_SUFFIXES,
835
+ ("mtp.layer.",),
836
+ )
837
+
838
+
797
839
  def _passes_appended_layer_gate(inspection: Any) -> bool:
798
840
  keys = _weight_keys(inspection)
799
841
  if not keys:
@@ -910,6 +952,8 @@ def _passes_family_runtime_gate(arch_id: str, inspection: Any, tensor_gate: bool
910
952
  tensor_gate
911
953
  and int(getattr(inspection, "mtp_num_hidden_layers", 0) or 0) > 0
912
954
  )
955
+ if arch_id == "hy-v3-mtp":
956
+ return _passes_hy_v3_gate(inspection)
913
957
  if arch_id in {
914
958
  "deepseek-v3-mtp",
915
959
  "glm-moe-dsa-mtp",
@@ -3,6 +3,7 @@
3
3
  from __future__ import annotations
4
4
 
5
5
  import hashlib
6
+ from collections import deque
6
7
  import json
7
8
  import logging
8
9
  import os
@@ -70,14 +71,15 @@ _COMMITTED_CACHE_POLICIES = frozenset({"committed", "last_window"})
70
71
 
71
72
 
72
73
  def _deferred_encode_enabled() -> bool:
73
- """Writer-thread payload encode (kvcache-v2). Off-switch only.
74
+ """RETIRED (#169, 2026-07-17) — writer-side encode is gone; always False.
74
75
 
75
- The foreground evaluates payload arrays (ms-scale GPU slice kernels) so
76
- the writer thread never evaluates foreign lazy graphs — it only reads
77
- settled buffers into bytes (the GB-scale memcpy that used to run on the
78
- request thread)."""
79
- raw = str(os.environ.get("MTPLX_SSD_DEFERRED_ENCODE", "1")).strip().lower()
80
- return raw not in {"0", "false", "off", "no"}
76
+ The kvcache-v2 writer-thread encode block-sliced tensors at write time
77
+ (TreeCodec builds lazy slice arrays and mx.eval()s them), which crashed
78
+ on restore-derived arrays ("There is no Stream(gpu, 1) in current
79
+ thread") and could serialize donation-mutated KV pages from the writer
80
+ backlog. put_entry now always encodes at enqueue on the owner thread.
81
+ MTPLX_SSD_DEFERRED_ENCODE is parsed nowhere and ignored."""
82
+ return False
81
83
 
82
84
 
83
85
  @dataclass(frozen=True)
@@ -100,6 +102,9 @@ class PendingWrite:
100
102
  tensors: dict[str, bytes]
101
103
  deferred: DeferredPayload | None = None
102
104
  created_at_s: float = field(default_factory=time.time)
105
+ # Estimated bytes this write pins in memory until the writer drains it
106
+ # (deferred payloads hold live KV arrays; encoded ones hold the buffers).
107
+ pinned_nbytes: int = 0
103
108
 
104
109
 
105
110
  @dataclass(frozen=True)
@@ -169,6 +174,10 @@ def default_cold_tier_max_bytes() -> int:
169
174
  return DEFAULT_COLD_TIER_MAX_BYTES
170
175
 
171
176
 
177
+ def _env_size_bytes(name: str, default: int) -> int:
178
+ return parse_size_bytes(os.environ.get(name), default)
179
+
180
+
172
181
  def parse_size_bytes(value: str | int | None, default: int) -> int:
173
182
  if value is None:
174
183
  return int(default)
@@ -270,6 +279,24 @@ class SessionBankColdTier:
270
279
  self._queue: queue.Queue[PendingWrite | None] = queue.Queue(
271
280
  maxsize=max(1, int(writer_queue_depth))
272
281
  )
282
+ # Backlog byte cap (issue #145): every queued write pins its payload
283
+ # (deferred ones pin LIVE KV arrays) until the writer drains it. A
284
+ # count-bounded queue of 32 multi-GB snapshots can pin ~50 GB under
285
+ # distinct-prefix churn — measured live 2026-07-09 (active memory
286
+ # climbed 35 -> 66 GB while the bank ledger stayed flat). Cap the
287
+ # pinned bytes, drop new writes beyond it.
288
+ self._pending_bytes = 0
289
+ self._backlog_budget_bytes = _env_size_bytes(
290
+ "MTPLX_SSD_WRITER_BACKLOG_BYTES", 4 * 1024**3
291
+ )
292
+ # Hourly write budget (issue #144: 7 TB written / SSD wear): the
293
+ # different-repos pattern writes GBs per task and never restores
294
+ # them (measured 58 GB in 45 min with restore_hits=0). Rolling
295
+ # one-hour byte budget; beyond it new writes are skipped.
296
+ self._write_budget_per_hour_bytes = _env_size_bytes(
297
+ "MTPLX_SSD_WRITE_BUDGET_PER_HOUR", 64 * 1024**3
298
+ )
299
+ self._written_window: deque[tuple[float, int]] = deque()
273
300
  self._stop = threading.Event()
274
301
  self._base_lock = threading.RLock()
275
302
  self._disk_usage_lock = threading.Lock()
@@ -336,82 +363,55 @@ class SessionBankColdTier:
336
363
  if len(token_ids) < self.min_prefix_tokens:
337
364
  self._inc("skipped_too_short")
338
365
  return False
366
+ estimated_nbytes = int(getattr(entry, "nbytes", 0) or 0)
367
+ if not self._admit_write(estimated_nbytes):
368
+ return False
339
369
  boundaries = tuple(
340
370
  (int(r[0]), r[1], r[2] if len(r) > 2 else None)
341
371
  for r in (getattr(entry, "gdn_boundaries", None) or [])
342
372
  )
343
- if _deferred_encode_enabled():
344
- try:
345
- deferred = DeferredPayload(
346
- cache_snapshot=getattr(entry, "cache_snapshot"),
347
- logits=getattr(entry, "logits"),
348
- hidden=getattr(entry, "hidden"),
349
- mtp_history_snapshot=getattr(entry, "mtp_history_snapshot", None),
350
- gdn_boundaries=boundaries,
351
- has_recurrent=bool(getattr(entry, "has_recurrent", False)),
352
- block_size=self.block_size,
353
- )
354
- # Settle every payload array on the request thread so the
355
- # writer only reads buffers (MLX thread discipline: never
356
- # evaluate another thread's lazy graph).
357
- _eval_payload_trees(
358
- deferred.cache_snapshot,
359
- deferred.logits,
360
- deferred.hidden,
361
- deferred.mtp_history_snapshot,
362
- deferred.gdn_boundaries,
363
- )
364
- except Exception as exc:
365
- self._inc("skipped_serialize_error")
366
- logger.warning(
367
- "SessionBank SSD payload prep skipped: %s: %s",
368
- type(exc).__name__,
369
- exc,
370
- )
371
- return False
372
- metadata = self._metadata_for_entry(
373
- entry,
374
- capabilities=capabilities or (),
375
- payload_nbytes=0,
376
- )
377
- pending = PendingWrite(
378
- entry_id=str(metadata["entry_id"]),
379
- token_ids=token_ids,
380
- metadata=metadata,
381
- payload_spec=None,
382
- tensors={},
383
- deferred=deferred,
384
- )
385
- else:
386
- try:
387
- encoded = encode_payload(
388
- cache_snapshot=getattr(entry, "cache_snapshot"),
389
- logits=getattr(entry, "logits"),
390
- hidden=getattr(entry, "hidden"),
391
- mtp_history_snapshot=getattr(entry, "mtp_history_snapshot", None),
392
- gdn_boundaries=boundaries,
393
- has_recurrent=bool(getattr(entry, "has_recurrent", False)),
394
- block_size=self.block_size,
395
- )
396
- except Exception as exc:
397
- self._inc("skipped_serialize_error")
398
- logger.warning("SessionBank SSD serialize skipped: %s: %s", type(exc).__name__, exc)
399
- return False
400
- metadata = self._metadata_for_entry(
401
- entry,
402
- capabilities=capabilities or (),
403
- payload_nbytes=encoded.nbytes,
404
- )
405
- pending = PendingWrite(
406
- entry_id=str(metadata["entry_id"]),
407
- token_ids=token_ids,
408
- metadata=metadata,
409
- payload_spec=encoded.spec,
410
- tensors=encoded.tensors,
373
+ # Encode ALWAYS happens here, on the enqueueing (owner) thread, never
374
+ # on the writer thread (#169, 2026-07-17). The retired writer-side
375
+ # "deferred encode" (kvcache-v2) block-sliced tensors at write time:
376
+ # TreeCodec._encode_tensor_blocks builds new lazy slice arrays and
377
+ # mx.eval()s them, which (a) crashed the process on restore-derived
378
+ # arrays whose graphs referenced the restore stream ("There is no
379
+ # Stream(gpu, 1) in current thread"), and (b) held live KV references
380
+ # for seconds in the writer backlog, so under buffer donation the
381
+ # eventual serialization could capture mutated pages — silently
382
+ # corrupt persisted sessions that degrade on every restore. Bytes are
383
+ # captured at snapshot time; the writer thread is pure file IO.
384
+ try:
385
+ encoded = encode_payload(
386
+ cache_snapshot=getattr(entry, "cache_snapshot"),
387
+ logits=getattr(entry, "logits"),
388
+ hidden=getattr(entry, "hidden"),
389
+ mtp_history_snapshot=getattr(entry, "mtp_history_snapshot", None),
390
+ gdn_boundaries=boundaries,
391
+ has_recurrent=bool(getattr(entry, "has_recurrent", False)),
392
+ block_size=self.block_size,
411
393
  )
394
+ except Exception as exc:
395
+ self._inc("skipped_serialize_error")
396
+ logger.warning("SessionBank SSD serialize skipped: %s: %s", type(exc).__name__, exc)
397
+ return False
398
+ metadata = self._metadata_for_entry(
399
+ entry,
400
+ capabilities=capabilities or (),
401
+ payload_nbytes=encoded.nbytes,
402
+ )
403
+ pending = PendingWrite(
404
+ entry_id=str(metadata["entry_id"]),
405
+ token_ids=token_ids,
406
+ metadata=metadata,
407
+ payload_spec=encoded.spec,
408
+ tensors=encoded.tensors,
409
+ pinned_nbytes=max(estimated_nbytes, int(encoded.nbytes)),
410
+ )
412
411
  try:
413
412
  self._queue.put_nowait(pending)
414
413
  except queue.Full:
414
+ self._release_pending(pending.pinned_nbytes)
415
415
  self._inc("skipped_queue_full")
416
416
  logger.warning(
417
417
  "SessionBank SSD writer queue full; skipping prefix_len=%d token_hash=%s",
@@ -573,11 +573,18 @@ class SessionBankColdTier:
573
573
  def stats(self) -> dict[str, Any]:
574
574
  with self._stats_lock:
575
575
  stats = dict(self._stats)
576
+ with self._stats_lock:
577
+ pending_bytes = int(self._pending_bytes)
578
+ written_last_hour = sum(nbytes for _, nbytes in self._written_window)
576
579
  stats.update(
577
580
  {
578
581
  "enabled": self.enabled,
579
582
  "restorable": self.restorable,
580
583
  "writer_queue_depth": int(self._queue.qsize()),
584
+ "writer_backlog_bytes": pending_bytes,
585
+ "writer_backlog_budget_bytes": int(self._backlog_budget_bytes),
586
+ "written_bytes_last_hour": int(written_last_hour),
587
+ "write_budget_per_hour_bytes": int(self._write_budget_per_hour_bytes),
581
588
  "dir": str(self.base_dir),
582
589
  "manifest_path": str(self._manifest_path),
583
590
  }
@@ -786,6 +793,9 @@ class SessionBankColdTier:
786
793
  self._inc("writes_completed")
787
794
  with self._stats_lock:
788
795
  self._stats["last_write_s"] = time.time()
796
+ self._written_window.append(
797
+ (time.time(), int(pending.pinned_nbytes))
798
+ )
789
799
  logger.info(
790
800
  "SessionBank SSD wrote entry_id=%s prefix_len=%d nbytes=%d",
791
801
  pending.entry_id,
@@ -801,37 +811,54 @@ class SessionBankColdTier:
801
811
  exc,
802
812
  )
803
813
  finally:
814
+ self._release_pending(pending.pinned_nbytes)
804
815
  self._queue.task_done()
805
816
 
817
+ def _admit_write(self, estimated_nbytes: int) -> bool:
818
+ """Backlog + hourly-budget admission for a new SSD write."""
819
+
820
+ now = time.time()
821
+ with self._stats_lock:
822
+ if (
823
+ self._pending_bytes + max(0, estimated_nbytes)
824
+ > self._backlog_budget_bytes
825
+ ):
826
+ self._stats["skipped_backlog_bytes"] = (
827
+ int(self._stats.get("skipped_backlog_bytes", 0) or 0) + 1
828
+ )
829
+ return False
830
+ while self._written_window and self._written_window[0][0] < now - 3600:
831
+ self._written_window.popleft()
832
+ written_last_hour = sum(nbytes for _, nbytes in self._written_window)
833
+ if (
834
+ written_last_hour + max(0, estimated_nbytes)
835
+ > self._write_budget_per_hour_bytes
836
+ ):
837
+ self._stats["skipped_write_budget"] = (
838
+ int(self._stats.get("skipped_write_budget", 0) or 0) + 1
839
+ )
840
+ return False
841
+ self._pending_bytes += max(0, estimated_nbytes)
842
+ return True
843
+
844
+ def _release_pending(self, nbytes: int) -> None:
845
+ with self._stats_lock:
846
+ self._pending_bytes = max(0, self._pending_bytes - max(0, int(nbytes)))
847
+
806
848
  def _write_pending(self, pending: PendingWrite) -> bool:
807
849
  if pending.deferred is not None:
808
- # Writer-side encode (kvcache-v2): arrays were settled by the
809
- # request thread; this is pure buffer->bytes work off the
810
- # foreground. Failures count as write failures, not serialize
811
- # skips, so the stats distinguish the two eras.
812
- encoded = encode_payload(
813
- cache_snapshot=pending.deferred.cache_snapshot,
814
- logits=pending.deferred.logits,
815
- hidden=pending.deferred.hidden,
816
- mtp_history_snapshot=pending.deferred.mtp_history_snapshot,
817
- gdn_boundaries=pending.deferred.gdn_boundaries,
818
- has_recurrent=pending.deferred.has_recurrent,
819
- block_size=pending.deferred.block_size,
820
- )
821
- metadata = dict(pending.metadata)
822
- metadata["nbytes"] = int(
823
- max(int(metadata.get("nbytes", 0) or 0), int(encoded.nbytes))
824
- )
825
- metadata["logical_nbytes"] = int(encoded.nbytes)
826
- metadata["physical_nbytes"] = int(encoded.nbytes)
827
- pending = PendingWrite(
828
- entry_id=pending.entry_id,
829
- token_ids=pending.token_ids,
830
- metadata=metadata,
831
- payload_spec=encoded.spec,
832
- tensors=encoded.tensors,
833
- created_at_s=pending.created_at_s,
850
+ # Retired path (#169, 2026-07-17): writer-side encode ran MLX
851
+ # slice/eval graph work on the writer thread (crash on
852
+ # restore-stream arrays, donation-corruption window). put_entry
853
+ # now always encodes at enqueue; a deferred payload reaching the
854
+ # writer is a programming error, never silently encoded here.
855
+ self._inc("skipped_deferred_retired")
856
+ logger.error(
857
+ "SessionBank SSD writer received a deferred payload "
858
+ "entry_id=%s; writer-side encode is retired (#169), skipping",
859
+ pending.entry_id,
834
860
  )
861
+ return False
835
862
  with self._base_lock:
836
863
  self._ensure_store()
837
864
  entry_hash_prefix = pending.entry_id[:2]
@@ -889,6 +889,17 @@ class VllmMetalPagedKVCache:
889
889
  int((self.num_blocks * 3 + 1) // 2),
890
890
  int(self.num_blocks) + 1,
891
891
  )
892
+ window_tokens = _env_int("MTPLX_CONTEXT_WINDOW_TOKENS", 0)
893
+ if window_tokens > 0:
894
+ # Geometric growth must not overshoot the serving context window
895
+ # (#150: the 1.5x step at 100k+ ctx allocates GiBs of blocks no
896
+ # request can ever address). A genuinely larger requirement still
897
+ # wins — correctness over the clamp.
898
+ window_blocks = (int(window_tokens) + self.block_size - 1) // self.block_size
899
+ if window_blocks >= required_blocks:
900
+ grown_blocks = min(
901
+ grown_blocks, max(window_blocks, int(self.num_blocks))
902
+ )
892
903
  if grown_blocks <= self.num_blocks:
893
904
  return True
894
905
  if self.key_cache is None or self.value_cache is None:
@@ -1437,6 +1448,14 @@ class VllmMetalPagedKVCache:
1437
1448
  outputs: list[Any] = []
1438
1449
  very_negative = mx.array(-1.0e30, dtype=mx.float32)
1439
1450
  eps = mx.array(1.0e-20, dtype=mx.float32)
1451
+ # kv-quant stores values packed; _paged_range dequantizes them back to
1452
+ # the logical head dim recorded in _shape, so the accumulator must be
1453
+ # sized to the dequantized width, not the packed storage width (#150,
1454
+ # q4 crash on the paged split-SDPA path).
1455
+ if self.kv_quant and self._shape is not None:
1456
+ value_dim = int(self._shape[2])
1457
+ else:
1458
+ value_dim = int(self.value_cache.shape[3])
1440
1459
 
1441
1460
  for q_start in range(0, q_len, q_chunk_size):
1442
1461
  q_end = min(q_len, q_start + q_chunk_size)
@@ -1454,7 +1473,7 @@ class VllmMetalPagedKVCache:
1454
1473
  int(q.shape[0]),
1455
1474
  int(q.shape[1]),
1456
1475
  int(q.shape[2]),
1457
- int(self.value_cache.shape[3]),
1476
+ value_dim,
1458
1477
  ),
1459
1478
  dtype=mx.float32,
1460
1479
  )
@@ -585,7 +585,7 @@ def _add_bridge_prompt_args(parser: argparse.ArgumentParser) -> None:
585
585
  )
586
586
  parser.add_argument(
587
587
  "--chat-template-profile",
588
- choices=["local_qwen36", "froggeric_v19", "tokenizer"],
588
+ choices=["local_qwen36", "froggeric_v19", "froggeric_v21_3", "tokenizer"],
589
589
  default="local_qwen36",
590
590
  help="Chat template profile for server/OpenCode paths.",
591
591
  )
@@ -1924,7 +1924,7 @@ def build_parser() -> argparse.ArgumentParser:
1924
1924
  "--profile",
1925
1925
  type=_profile_arg, metavar=_PROFILE_METAVAR,
1926
1926
  default=DEFAULT_PROFILE_NAME,
1927
- help="Runtime profile; start defaults to Sustained. Use --profile turbo for the verify-kernel fast path (4/8-bit affine), or --profile performance-cold --max for Burst.",
1927
+ help="Runtime profile. Default resolves per model: Turbo for the quantized 27B and 9B flagships (the app's launch rule), Sustained otherwise. An explicit value always wins. Use --profile performance-cold --max for Burst.",
1928
1928
  )
1929
1929
  start_flow_p.add_argument("--download", action="store_true", help="Download the selected/default model if it is missing")
1930
1930
  start_flow_p.add_argument("--yes", action="store_true", help="Use defaults without interactive model prompts")
@@ -2106,7 +2106,7 @@ def build_parser() -> argparse.ArgumentParser:
2106
2106
  "--profile",
2107
2107
  type=_profile_arg, metavar=_PROFILE_METAVAR,
2108
2108
  default=DEFAULT_PROFILE_NAME,
2109
- help="Runtime profile. Direct server quickstart defaults to Sustained; use --profile performance-cold --max for Burst.",
2109
+ help="Runtime profile. Default resolves per model (Turbo for the quantized 27B and 9B flagships, Sustained otherwise); use --profile performance-cold --max for Burst.",
2110
2110
  )
2111
2111
  quickstart_server_p.add_argument("--unsafe-force-unverified", action="store_true")
2112
2112
  quickstart_server_p.add_argument("--yes", action="store_true", help="Confirm unsafe non-interactive actions")
@@ -2265,6 +2265,7 @@ def build_parser() -> argparse.ArgumentParser:
2265
2265
  tune_p.add_argument("--mtp-cache-policy", choices=["persistent", "fresh"], default="persistent", help=argparse.SUPPRESS)
2266
2266
  tune_p.add_argument("--mtp-history-policy", choices=["auto", "committed", "full", "last-window", "last_window", "cycle", "none"], default="committed", help=argparse.SUPPRESS)
2267
2267
  tune_p.add_argument("--draft-temperature", type=float, help=argparse.SUPPRESS)
2268
+ tune_p.add_argument("--draft-core", choices=["stock", "device-d2", "device"], default="stock", help=argparse.SUPPRESS)
2268
2269
  tune_p.add_argument("--draft-top-p", type=float, help=argparse.SUPPRESS)
2269
2270
  tune_p.add_argument("--draft-top-k", type=int, help=argparse.SUPPRESS)
2270
2271
  tune_p.add_argument("--prompt-suite", help=argparse.SUPPRESS)
@@ -2330,6 +2331,13 @@ def build_parser() -> argparse.ArgumentParser:
2330
2331
  forge_build_p.add_argument("--max", action="store_true", help="Opt into max-fan verification")
2331
2332
  forge_build_p.add_argument("--max-tokens", type=int, default=2048, help="Verification response budget")
2332
2333
  forge_build_p.add_argument("--suite", help="Verification prompt suite")
2334
+ forge_build_p.add_argument(
2335
+ "--dtype",
2336
+ choices=["auto", "bf16", "fp16"],
2337
+ help="Dtype for non-quantized parameters (overrides recipe body_dtype). "
2338
+ "fp16 prompt-processes faster on M1/M2 Macs, which have no native "
2339
+ "BF16; auto picks fp16 on those chips",
2340
+ )
2333
2341
  forge_build_p.add_argument(
2334
2342
  "--allow-degraded-mtp",
2335
2343
  action="store_true",
@@ -2503,8 +2511,9 @@ def build_parser() -> argparse.ArgumentParser:
2503
2511
  type=_profile_arg, metavar=_PROFILE_METAVAR,
2504
2512
  default=DEFAULT_PROFILE_NAME,
2505
2513
  help=(
2506
- "Runtime profile. Server defaults to Sustained so long-context "
2507
- "prefill uses the v0.1.7 fast path; use --profile performance-cold "
2514
+ "Runtime profile. Default resolves per model: Turbo for the "
2515
+ "quantized 27B and 9B flagships (the app's launch rule), "
2516
+ "Sustained otherwise; use --profile performance-cold "
2508
2517
  "--max for Burst."
2509
2518
  ),
2510
2519
  )
@@ -3461,7 +3470,7 @@ def build_parser() -> argparse.ArgumentParser:
3461
3470
  )
3462
3471
  depth_p.add_argument(
3463
3472
  "--draft-core",
3464
- choices=["stock", "device-d2"],
3473
+ choices=["stock", "device-d2", "device"],
3465
3474
  default="stock",
3466
3475
  help=(
3467
3476
  "Experimental DraftCore backend. device-d2 compiles the greedy D2 "