mtplx 2.4.0__tar.gz → 2.4.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (450) hide show
  1. {mtplx-2.4.0 → mtplx-2.4.2}/CHANGELOG.md +165 -1
  2. {mtplx-2.4.0 → mtplx-2.4.2}/CITATION.cff +1 -1
  3. {mtplx-2.4.0 → mtplx-2.4.2}/PKG-INFO +15 -10
  4. {mtplx-2.4.0 → mtplx-2.4.2}/README.md +14 -9
  5. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/adaptive.py +251 -0
  6. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/artifacts.py +56 -0
  7. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/backends/registry.py +67 -6
  8. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/cli.py +24 -11
  9. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/commands/public.py +73 -8
  10. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/default_models.py +72 -30
  11. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/diagnostics.py +12 -9
  12. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/engine_session.py +7 -0
  13. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/generation.py +20 -2
  14. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/graphbank.py +26 -3
  15. mtplx-2.4.2/mtplx/kernels/gdn_blocked_prefill.py +468 -0
  16. mtplx-2.4.2/mtplx/models/deepseek_v4.py +3440 -0
  17. mtplx-2.4.2/mtplx/request_capture.py +130 -0
  18. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/runtime.py +45 -3
  19. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/server/openai.py +222 -16
  20. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/session_bank.py +134 -1
  21. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/version.py +2 -2
  22. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx.egg-info/PKG-INFO +15 -10
  23. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx.egg-info/SOURCES.txt +38 -0
  24. {mtplx-2.4.0 → mtplx-2.4.2}/pyproject.toml +1 -1
  25. mtplx-2.4.2/scripts/deepseek_v4_build_mtp_model.py +673 -0
  26. mtplx-2.4.2/scripts/deepseek_v4_decode_verify.py +198 -0
  27. mtplx-2.4.2/scripts/deepseek_v4_dispatch_census.py +248 -0
  28. mtplx-2.4.2/scripts/deepseek_v4_guard_window.py +316 -0
  29. mtplx-2.4.2/scripts/deepseek_v4_logits_gate.py +93 -0
  30. mtplx-2.4.2/scripts/deepseek_v4_moe_tail_arms.sh +150 -0
  31. mtplx-2.4.2/scripts/deepseek_v4_moe_tail_gate.py +508 -0
  32. mtplx-2.4.2/scripts/deepseek_v4_moe_tail_guarded_bracket.py +301 -0
  33. mtplx-2.4.2/scripts/deepseek_v4_mtp_bind_check.py +162 -0
  34. mtplx-2.4.2/scripts/deepseek_v4_mtpk_bench.py +1104 -0
  35. mtplx-2.4.2/scripts/deepseek_v4_op_census.py +277 -0
  36. mtplx-2.4.2/scripts/deepseek_v4_smoke_generate.py +535 -0
  37. mtplx-2.4.2/scripts/deepseek_v4_sparse_gate_probe.py +311 -0
  38. mtplx-2.4.2/scripts/deepseek_v4_validate_moe_tail_k3_bracket.py +763 -0
  39. mtplx-2.4.2/scripts/gauntlet_scoreboard.py +85 -0
  40. {mtplx-2.4.0 → mtplx-2.4.2}/scripts/hygiene_scan.sh +7 -0
  41. mtplx-2.4.2/scripts/oc_tap.py +236 -0
  42. mtplx-2.4.2/scripts/oc_tap_diff.py +124 -0
  43. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_artifacts.py +53 -1
  44. mtplx-2.4.2/tests/test_cost_depth_policy.py +108 -0
  45. mtplx-2.4.2/tests/test_deepseek_v4_decode.py +263 -0
  46. mtplx-2.4.2/tests/test_deepseek_v4_dtypes.py +465 -0
  47. mtplx-2.4.2/tests/test_deepseek_v4_indexer.py +623 -0
  48. mtplx-2.4.2/tests/test_deepseek_v4_kernel_paths.py +495 -0
  49. mtplx-2.4.2/tests/test_deepseek_v4_loader.py +225 -0
  50. mtplx-2.4.2/tests/test_deepseek_v4_moe_tail.py +796 -0
  51. mtplx-2.4.2/tests/test_deepseek_v4_moe_tail_bracket.py +1303 -0
  52. mtplx-2.4.2/tests/test_deepseek_v4_mtp.py +753 -0
  53. mtplx-2.4.2/tests/test_deepseek_v4_new_math.py +282 -0
  54. mtplx-2.4.2/tests/test_deepseek_v4_o_lora.py +510 -0
  55. mtplx-2.4.2/tests/test_deepseek_v4_parity.py +106 -0
  56. mtplx-2.4.2/tests/test_deepseek_v4_sinkhorn_kernel.py +474 -0
  57. mtplx-2.4.2/tests/test_deepseek_v4_spec.py +915 -0
  58. mtplx-2.4.2/tests/test_deepseek_v4_swiglu_clamp.py +433 -0
  59. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_default_models.py +41 -0
  60. mtplx-2.4.2/tests/test_gdn_blocked_prefill.py +115 -0
  61. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_generation_sustained.py +34 -0
  62. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_graphbank_compiled_verify.py +92 -25
  63. mtplx-2.4.2/tests/test_mtp_weightless_degrade.py +118 -0
  64. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_postcommit_wait_integration.py +62 -0
  65. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_public_cli.py +32 -8
  66. mtplx-2.4.2/tests/test_request_capture.py +76 -0
  67. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_session_bank.py +73 -0
  68. {mtplx-2.4.0 → mtplx-2.4.2}/LICENSE +0 -0
  69. {mtplx-2.4.0 → mtplx-2.4.2}/MANIFEST.in +0 -0
  70. {mtplx-2.4.0 → mtplx-2.4.2}/NOTICE +0 -0
  71. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/__init__.py +0 -0
  72. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/a3b_compiled_target_prefix.py +0 -0
  73. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/a3b_whole_moe.py +0 -0
  74. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/app_settings.py +0 -0
  75. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/attention_context.py +0 -0
  76. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/attention_split.py +0 -0
  77. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/backends/__init__.py +0 -0
  78. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/backends/deepseek_mtp.py +0 -0
  79. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/backends/descriptors.py +0 -0
  80. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/backends/gemma4_assistant.py +0 -0
  81. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/backends/glm_mtp.py +0 -0
  82. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/backends/hy_v3_mtp.py +0 -0
  83. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/backends/mimo_mtp.py +0 -0
  84. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/backends/nemotron_h_mtp.py +0 -0
  85. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/backends/qwen3_next.py +0 -0
  86. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/backends/step3p5_mtp.py +0 -0
  87. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/batched_decode.py +0 -0
  88. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/batching/__init__.py +0 -0
  89. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/batching/admission.py +0 -0
  90. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/batching/buckets.py +0 -0
  91. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/batching/scheduler.py +0 -0
  92. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/batching/state.py +0 -0
  93. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/benchmarks/__init__.py +0 -0
  94. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/benchmarks/code_eval.py +0 -0
  95. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/benchmarks/programming_prompts.py +0 -0
  96. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/benchmarks/prompts/aime_2026.jsonl +0 -0
  97. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/benchmarks/prompts/calibration_coding.jsonl +0 -0
  98. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/benchmarks/prompts/default.jsonl +0 -0
  99. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/benchmarks/prompts/flappy.jsonl +0 -0
  100. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/benchmarks/prompts/long_code.jsonl +0 -0
  101. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/benchmarks/prompts/long_code_uncapped.jsonl +0 -0
  102. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/benchmarks/prompts/python_modules_long.jsonl +0 -0
  103. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/benchmarks/runners/__init__.py +0 -0
  104. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/benchmarks/runners/aime.py +0 -0
  105. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/benchmarks/runners/batch_equivalence.py +0 -0
  106. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/benchmarks/runners/capture_commit_equivalence.py +0 -0
  107. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/benchmarks/runners/competitor_baselines.py +0 -0
  108. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/benchmarks/runners/contract_probe.py +0 -0
  109. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/benchmarks/runners/harness.py +0 -0
  110. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/benchmarks/runners/mtp1_gate.py +0 -0
  111. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/benchmarks/runners/mtp1_sampler_smoke.py +0 -0
  112. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/benchmarks/runners/mtp_adaptive.py +0 -0
  113. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/benchmarks/runners/mtp_chain_probe.py +0 -0
  114. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/benchmarks/runners/mtp_depth_grid.py +0 -0
  115. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/benchmarks/runners/mtp_depth_sweep.py +0 -0
  116. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/benchmarks/runners/mtp_tree_probe.py +0 -0
  117. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/benchmarks/runners/multi_qmv_probe.py +0 -0
  118. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/benchmarks/runners/preflight.py +0 -0
  119. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/benchmarks/runners/runtime_smoke.py +0 -0
  120. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/benchmarks/runners/session_bank.py +0 -0
  121. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/benchmarks/runners/truth.py +0 -0
  122. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/benchmarks/runners/verify_profile.py +0 -0
  123. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/benchmarks/runners/verify_qmm_probe.py +0 -0
  124. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/benchmarks/runners/verify_ratio.py +0 -0
  125. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/benchmarks/schema.py +0 -0
  126. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/benchmarks/validators/__init__.py +0 -0
  127. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/benchmarks/validators/aime.py +0 -0
  128. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/benchmarks/validators/basic.py +0 -0
  129. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/block_attention.py +0 -0
  130. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/cache_bank/__init__.py +0 -0
  131. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/cache_bank/codec.py +0 -0
  132. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/cache_bank/cold_tier.py +0 -0
  133. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/cache_state.py +0 -0
  134. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/chat_encoding.py +0 -0
  135. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/commands/__init__.py +0 -0
  136. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/commands/forge.py +0 -0
  137. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/compile_state.py +0 -0
  138. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/compiled_forward.py +0 -0
  139. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/compressed_tensors.py +0 -0
  140. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/config.py +0 -0
  141. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/constants.py +0 -0
  142. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/constrained.py +0 -0
  143. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/context_copy.py +0 -0
  144. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/correctors/__init__.py +0 -0
  145. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/correctors/diagonal_affine.py +0 -0
  146. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/correctors/low_rank.py +0 -0
  147. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/daemon_client.py +0 -0
  148. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/dashboard/__init__.py +0 -0
  149. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/dashboard/_static/assets/index-CRP90ECi.js +0 -0
  150. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/dashboard/_static/assets/index-DYvLRZ33.css +0 -0
  151. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/dashboard/_static/index.html +0 -0
  152. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/deepseek_mtp_patch.py +0 -0
  153. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/draft_lm_head.py +0 -0
  154. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/draft_sampling.py +0 -0
  155. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/env.py +0 -0
  156. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/errors.py +0 -0
  157. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/expert_layout.py +0 -0
  158. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/fan_mode.py +0 -0
  159. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/fast_sampling.py +0 -0
  160. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/gdn_capture.py +0 -0
  161. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/gemma4_pair.py +0 -0
  162. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/glm_mtp_patch.py +0 -0
  163. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/hardware.py +0 -0
  164. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/hf_loader.py +0 -0
  165. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/hy_v3_mtp_patch.py +0 -0
  166. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/kernel_selfcheck.py +0 -0
  167. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/kernels/__init__.py +0 -0
  168. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/kernels/a3b_whole_moe.py +0 -0
  169. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/kernels/copy_leaf.py +0 -0
  170. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/kernels/fused_norm.py +0 -0
  171. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/kernels/laguna_decode.py +0 -0
  172. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/kernels/lm_head_topk.py +0 -0
  173. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/kernels/logits_topk.py +0 -0
  174. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/kernels/native_gdn_tail.py +0 -0
  175. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/kernels/sdpa_2pass.py +0 -0
  176. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/kernels/sdpa_2pass_paged.py +0 -0
  177. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/kernels/sdpa_2pass_paged_q8.py +0 -0
  178. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/kernels/sdpa_gqa_packed.py +0 -0
  179. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/kernels/verify_mlp_fused.py +0 -0
  180. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/kpi/__init__.py +0 -0
  181. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/kpi/reference_vllm.py +0 -0
  182. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/kpi/runtime_kpis.py +0 -0
  183. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/kv_quant.py +0 -0
  184. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/laguna_compiled_step.py +0 -0
  185. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/loop_guard.py +0 -0
  186. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/metadata_scrub.py +0 -0
  187. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/mimo_mtp_patch.py +0 -0
  188. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/model_catalog.py +0 -0
  189. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/model_scheduler.py +0 -0
  190. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/models/__init__.py +0 -0
  191. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/models/laguna.py +0 -0
  192. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/models/laguna_config.py +0 -0
  193. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/models/laguna_fused.py +0 -0
  194. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/moe_packed_projections.py +0 -0
  195. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/mtp_activation_stats.py +0 -0
  196. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/mtp_adapters.py +0 -0
  197. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/mtp_patch.py +0 -0
  198. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/native_mlp.py +0 -0
  199. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/nax_verify.py +0 -0
  200. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/nemotron_h_mtp_patch.py +0 -0
  201. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/opencode.py +0 -0
  202. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/optimization_profiles.py +0 -0
  203. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/pi.py +0 -0
  204. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/prefill_bench.py +0 -0
  205. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/profiles.py +0 -0
  206. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/progress_heartbeat.py +0 -0
  207. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/proj_quant.py +0 -0
  208. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/proposal_reranker.py +0 -0
  209. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/qwen3_5_mtp_patch.py +0 -0
  210. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/qwen_row_owned_router.py +0 -0
  211. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/ragged_attention.py +0 -0
  212. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/ragged_kv_cache.py +0 -0
  213. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/reasoning_codecs.py +0 -0
  214. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/roofline_profile.py +0 -0
  215. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/runtime_options.py +0 -0
  216. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/sampling.py +0 -0
  217. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/server/__init__.py +0 -0
  218. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/server/dashboard_state.py +0 -0
  219. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/server/omlx_bridge/__init__.py +0 -0
  220. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/server/omlx_bridge/adapter.py +0 -0
  221. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/server/omlx_bridge/thinking.py +0 -0
  222. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/server/omlx_bridge/tool_calling.py +0 -0
  223. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/server_urls.py +0 -0
  224. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/speculative.py +0 -0
  225. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/step3p5_mtp_patch.py +0 -0
  226. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/swival.py +0 -0
  227. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/templates/qwen36_froggeric_v19/chat_template.jinja +0 -0
  228. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/templates/qwen36_froggeric_v21_3/chat_template.jinja +0 -0
  229. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/thermal.py +0 -0
  230. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/thermal_sidecar.py +0 -0
  231. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/thinking_guard.py +0 -0
  232. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/trace_parity.py +0 -0
  233. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/turboquant.py +0 -0
  234. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/ui/__init__.py +0 -0
  235. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/ui/banner.py +0 -0
  236. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/ui/chat_printer.py +0 -0
  237. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/ui/download_progress.py +0 -0
  238. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/ui/onboarding.py +0 -0
  239. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/ui/panels.py +0 -0
  240. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/ui/progress.py +0 -0
  241. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/verify_kernels.py +0 -0
  242. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/verify_qmv.py +0 -0
  243. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/vision/__init__.py +0 -0
  244. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/vision/processing.py +0 -0
  245. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/vision/qwen3_vl_tower.py +0 -0
  246. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/vision/splice.py +0 -0
  247. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx.egg-info/dependency_links.txt +0 -0
  248. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx.egg-info/entry_points.txt +0 -0
  249. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx.egg-info/requires.txt +0 -0
  250. {mtplx-2.4.0 → mtplx-2.4.2}/mtplx.egg-info/top_level.txt +0 -0
  251. {mtplx-2.4.0 → mtplx-2.4.2}/scripts/agent_user_path_qa.py +0 -0
  252. {mtplx-2.4.0 → mtplx-2.4.2}/scripts/aime_serve_gate.py +0 -0
  253. {mtplx-2.4.0 → mtplx-2.4.2}/scripts/aime_shape_memory_bench.py +0 -0
  254. {mtplx-2.4.0 → mtplx-2.4.2}/scripts/code_eval_gate.py +0 -0
  255. {mtplx-2.4.0 → mtplx-2.4.2}/scripts/collect_mtp_activation_stats.py +0 -0
  256. {mtplx-2.4.0 → mtplx-2.4.2}/scripts/collect_mtp_hidden_calib.py +0 -0
  257. {mtplx-2.4.0 → mtplx-2.4.2}/scripts/combine_mtp_adapters.py +0 -0
  258. {mtplx-2.4.0 → mtplx-2.4.2}/scripts/compiled_verify_exactness.py +0 -0
  259. {mtplx-2.4.0 → mtplx-2.4.2}/scripts/convert_step37_to_mtplx_step3p5.py +0 -0
  260. {mtplx-2.4.0 → mtplx-2.4.2}/scripts/eval_mtp_corrector.py +0 -0
  261. {mtplx-2.4.0 → mtplx-2.4.2}/scripts/filter_mtp_adapter.py +0 -0
  262. {mtplx-2.4.0 → mtplx-2.4.2}/scripts/fp16_turbo_exactness_20260707.py +0 -0
  263. {mtplx-2.4.0 → mtplx-2.4.2}/scripts/fresh_venv_smoke.sh +0 -0
  264. {mtplx-2.4.0 → mtplx-2.4.2}/scripts/install_macos.sh +0 -0
  265. {mtplx-2.4.0 → mtplx-2.4.2}/scripts/install_preview_global.sh +0 -0
  266. {mtplx-2.4.0 → mtplx-2.4.2}/scripts/kvcache_exactness_audit_20260703.py +0 -0
  267. {mtplx-2.4.0 → mtplx-2.4.2}/scripts/kvcache_soak_20260703.py +0 -0
  268. {mtplx-2.4.0 → mtplx-2.4.2}/scripts/kvcache_warm_probe_20260703.py +0 -0
  269. {mtplx-2.4.0 → mtplx-2.4.2}/scripts/make_fp16_precision_sibling.py +0 -0
  270. {mtplx-2.4.0 → mtplx-2.4.2}/scripts/midform_gate.py +0 -0
  271. {mtplx-2.4.0 → mtplx-2.4.2}/scripts/ocspeed-20260703/abba_final.sh +0 -0
  272. {mtplx-2.4.0 → mtplx-2.4.2}/scripts/ocspeed-20260703/accept_depth_probe.py +0 -0
  273. {mtplx-2.4.0 → mtplx-2.4.2}/scripts/ocspeed-20260703/arm_runner.sh +0 -0
  274. {mtplx-2.4.0 → mtplx-2.4.2}/scripts/ocspeed-20260703/forward_depth_bisect.py +0 -0
  275. {mtplx-2.4.0 → mtplx-2.4.2}/scripts/ocspeed-20260703/longgen_probe.py +0 -0
  276. {mtplx-2.4.0 → mtplx-2.4.2}/scripts/ocspeed-20260703/loop_degradation_probe.py +0 -0
  277. {mtplx-2.4.0 → mtplx-2.4.2}/scripts/ocspeed-20260703/paired_longgen.sh +0 -0
  278. {mtplx-2.4.0 → mtplx-2.4.2}/scripts/ocspeed-20260703/sdpa_microbench.py +0 -0
  279. {mtplx-2.4.0 → mtplx-2.4.2}/scripts/ocspeed-20260703/snapshot_tail.py +0 -0
  280. {mtplx-2.4.0 → mtplx-2.4.2}/scripts/opencode_concurrency_qa.py +0 -0
  281. {mtplx-2.4.0 → mtplx-2.4.2}/scripts/phase0h_paged_verifier_exactness.py +0 -0
  282. {mtplx-2.4.0 → mtplx-2.4.2}/scripts/pillar_gate_qa.py +0 -0
  283. {mtplx-2.4.0 → mtplx-2.4.2}/scripts/probe_mlx_pr3026_qsdpa.py +0 -0
  284. {mtplx-2.4.0 → mtplx-2.4.2}/scripts/probe_mx_compile_buckets.py +0 -0
  285. {mtplx-2.4.0 → mtplx-2.4.2}/scripts/probe_paged_gqa_sdpa_routes.py +0 -0
  286. {mtplx-2.4.0 → mtplx-2.4.2}/scripts/prodqa-20260703/candidate_truecold_20260703.py +0 -0
  287. {mtplx-2.4.0 → mtplx-2.4.2}/scripts/prodqa-20260703/cold_pair_recheck.py +0 -0
  288. {mtplx-2.4.0 → mtplx-2.4.2}/scripts/prodqa-20260703/competitors_20260703.sh +0 -0
  289. {mtplx-2.4.0 → mtplx-2.4.2}/scripts/prodqa-20260703/decode_gap_matrix.py +0 -0
  290. {mtplx-2.4.0 → mtplx-2.4.2}/scripts/prodqa-20260703/hermes_pty_driver.py +0 -0
  291. {mtplx-2.4.0 → mtplx-2.4.2}/scripts/prodqa-20260703/omlx_probe.py +0 -0
  292. {mtplx-2.4.0 → mtplx-2.4.2}/scripts/prodqa-20260703/pi_pty_driver.py +0 -0
  293. {mtplx-2.4.0 → mtplx-2.4.2}/scripts/prodqa-20260703/pillar_ab_20260703.py +0 -0
  294. {mtplx-2.4.0 → mtplx-2.4.2}/scripts/prodqa-20260703/tool_gauntlet_20260703.sh +0 -0
  295. {mtplx-2.4.0 → mtplx-2.4.2}/scripts/quality_fp16_parent_logitdiff_20260707.py +0 -0
  296. {mtplx-2.4.0 → mtplx-2.4.2}/scripts/r1_chisquare_verifier_correctness.py +0 -0
  297. {mtplx-2.4.0 → mtplx-2.4.2}/scripts/release_macos_v1.sh +0 -0
  298. {mtplx-2.4.0 → mtplx-2.4.2}/scripts/run_context_degradation_diagnostics.py +0 -0
  299. {mtplx-2.4.0 → mtplx-2.4.2}/scripts/serve_openai_mtplx.py +0 -0
  300. {mtplx-2.4.0 → mtplx-2.4.2}/scripts/session_cache_followup_qa.py +0 -0
  301. {mtplx-2.4.0 → mtplx-2.4.2}/scripts/session_forensics.py +0 -0
  302. {mtplx-2.4.0 → mtplx-2.4.2}/scripts/sparkle_rehearsal_kit.sh +0 -0
  303. {mtplx-2.4.0 → mtplx-2.4.2}/scripts/step_acceptance.py +0 -0
  304. {mtplx-2.4.0 → mtplx-2.4.2}/scripts/step_smoke.py +0 -0
  305. {mtplx-2.4.0 → mtplx-2.4.2}/scripts/train_mtp_adapter_c4.py +0 -0
  306. {mtplx-2.4.0 → mtplx-2.4.2}/scripts/validate_step_mtp_injector.py +0 -0
  307. {mtplx-2.4.0 → mtplx-2.4.2}/scripts/walltime-lab/run_project.py +0 -0
  308. {mtplx-2.4.0 → mtplx-2.4.2}/setup.cfg +0 -0
  309. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_a3b_compiled_target_prefix.py +0 -0
  310. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_a3b_whole_moe.py +0 -0
  311. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_adaptive.py +0 -0
  312. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_aime_serve_gate.py +0 -0
  313. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_ar_batch_penalties.py +0 -0
  314. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_attention_split.py +0 -0
  315. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_background_warmup.py +0 -0
  316. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_batched_decode.py +0 -0
  317. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_batching_foundation.py +0 -0
  318. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_bracket_tool_rescue.py +0 -0
  319. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_cache_bank.py +0 -0
  320. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_cache_state.py +0 -0
  321. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_cli_parity_tools.py +0 -0
  322. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_code_eval.py +0 -0
  323. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_code_eval_gate.py +0 -0
  324. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_cold_tier_write_budget.py +0 -0
  325. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_compile_state.py +0 -0
  326. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_compiled_ar_wiring.py +0 -0
  327. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_compiled_forward.py +0 -0
  328. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_compressed_tensors.py +0 -0
  329. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_config.py +0 -0
  330. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_config_profile_precedence.py +0 -0
  331. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_constrained.py +0 -0
  332. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_context_copy_stats.py +0 -0
  333. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_context_degradation_profiles.py +0 -0
  334. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_correctors.py +0 -0
  335. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_daemon_client.py +0 -0
  336. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_dashboard_endpoints.py +0 -0
  337. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_device_draft_core.py +0 -0
  338. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_diagnostics.py +0 -0
  339. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_download_progress.py +0 -0
  340. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_draft_lm_head.py +0 -0
  341. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_engine_session_concurrency.py +0 -0
  342. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_engine_session_env.py +0 -0
  343. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_env_flag_parsing.py +0 -0
  344. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_fast_sampling.py +0 -0
  345. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_forge_cli.py +0 -0
  346. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_gdn_boundary_retention.py +0 -0
  347. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_gdn_postconv_fusion.py +0 -0
  348. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_gdn_postconv_impl_gate.py +0 -0
  349. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_generation_store_on_prefill.py +0 -0
  350. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_hf_loader.py +0 -0
  351. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_hy_v3_mtp_backend.py +0 -0
  352. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_hy_v3_mtp_graft.py +0 -0
  353. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_hygiene_scan.py +0 -0
  354. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_idle_postcommit_subagent.py +0 -0
  355. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_kernel_selfcheck.py +0 -0
  356. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_laguna_compiled_step.py +0 -0
  357. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_laguna_fused.py +0 -0
  358. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_laguna_model.py +0 -0
  359. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_lazy_snapshot_cow.py +0 -0
  360. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_loop_guard.py +0 -0
  361. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_max_idle_watchdog.py +0 -0
  362. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_max_lifecycle.py +0 -0
  363. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_memory_pressure_guard.py +0 -0
  364. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_metadata_scrub.py +0 -0
  365. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_metal_memory_caps.py +0 -0
  366. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_midform_gate.py +0 -0
  367. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_model_catalog.py +0 -0
  368. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_model_scheduler.py +0 -0
  369. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_moe_force_unsorted.py +0 -0
  370. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_moe_packed_projections.py +0 -0
  371. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_mtp_activation_stats.py +0 -0
  372. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_mtp_adapters.py +0 -0
  373. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_mtp_alias_load_path.py +0 -0
  374. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_mtp_depth_sweep.py +0 -0
  375. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_mtp_patch.py +0 -0
  376. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_mtp_payload_guards.py +0 -0
  377. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_nax_verify.py +0 -0
  378. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_no_mlx_imports.py +0 -0
  379. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_omlx_bridge.py +0 -0
  380. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_onboarding.py +0 -0
  381. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_openai_bridge.py +0 -0
  382. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_opencode.py +0 -0
  383. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_optimization_profiles.py +0 -0
  384. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_orphan_tool_markup.py +0 -0
  385. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_penalties.py +0 -0
  386. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_penalty_request_wiring.py +0 -0
  387. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_persistent_replay.py +0 -0
  388. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_phase0h_paged_verifier_exactness.py +0 -0
  389. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_policy_fingerprint_stability.py +0 -0
  390. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_postcommit_prefix_reuse.py +0 -0
  391. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_postcommit_resolve_for_request.py +0 -0
  392. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_postcommit_tools_plumbing.py +0 -0
  393. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_postcommit_wait.py +0 -0
  394. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_prefill_bench.py +0 -0
  395. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_prefill_chunk_defaults.py +0 -0
  396. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_prefill_tps_regression.py +0 -0
  397. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_profiles.py +0 -0
  398. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_proj_quant.py +0 -0
  399. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_prompt_encoding.py +0 -0
  400. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_qwen3_5_mtp_backend.py +0 -0
  401. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_qwen_row_owned_router.py +0 -0
  402. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_ragged_kv_cache.py +0 -0
  403. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_reasoning_stream_split.py +0 -0
  404. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_runtime_kpis.py +0 -0
  405. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_sampling.py +0 -0
  406. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_scoped_reasoning_history.py +0 -0
  407. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_sdpa_gqa_packed.py +0 -0
  408. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_server_openai.py +0 -0
  409. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_session_bank_env_caps.py +0 -0
  410. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_snapshot_free_rejection_repair.py +0 -0
  411. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_ssd_boundary_repersist.py +0 -0
  412. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_step3p5_mtp_patch.py +0 -0
  413. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_stream_guard_env.py +0 -0
  414. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_stream_stall_watchdog.py +0 -0
  415. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_sustained_long_context_qa.py +0 -0
  416. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_thermal.py +0 -0
  417. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_thermal_sidecar.py +0 -0
  418. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_thinking_guard.py +0 -0
  419. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_tool_aware_stream_translator.py +0 -0
  420. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_tool_nested_args_streaming.py +0 -0
  421. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_trace_parity.py +0 -0
  422. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_turboquant.py +0 -0
  423. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_turboquant_fallback.py +0 -0
  424. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_ui_progress.py +0 -0
  425. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_validators.py +0 -0
  426. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_vision_session_cache.py +0 -0
  427. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_vision_tower.py +0 -0
  428. {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_vllm_reference.py +0 -0
  429. {mtplx-2.4.0 → mtplx-2.4.2}/vllm_metal/__init__.py +0 -0
  430. {mtplx-2.4.0 → mtplx-2.4.2}/vllm_metal/metal/README.md +0 -0
  431. {mtplx-2.4.0 → mtplx-2.4.2}/vllm_metal/metal/__init__.py +0 -0
  432. {mtplx-2.4.0 → mtplx-2.4.2}/vllm_metal/metal/build.py +0 -0
  433. {mtplx-2.4.0 → mtplx-2.4.2}/vllm_metal/metal/constants.py +0 -0
  434. {mtplx-2.4.0 → mtplx-2.4.2}/vllm_metal/metal/kernels_v1/copy_blocks.metal +0 -0
  435. {mtplx-2.4.0 → mtplx-2.4.2}/vllm_metal/metal/kernels_v1/float8.metal +0 -0
  436. {mtplx-2.4.0 → mtplx-2.4.2}/vllm_metal/metal/kernels_v1/gather_kv_cache.metal +0 -0
  437. {mtplx-2.4.0 → mtplx-2.4.2}/vllm_metal/metal/kernels_v1/kv_scale_update.metal +0 -0
  438. {mtplx-2.4.0 → mtplx-2.4.2}/vllm_metal/metal/kernels_v1/pagedattention.metal +0 -0
  439. {mtplx-2.4.0 → mtplx-2.4.2}/vllm_metal/metal/kernels_v1/reshape_and_cache.metal +0 -0
  440. {mtplx-2.4.0 → mtplx-2.4.2}/vllm_metal/metal/kernels_v1/utils.metal +0 -0
  441. {mtplx-2.4.0 → mtplx-2.4.2}/vllm_metal/metal/kernels_v2/copy_blocks.metal +0 -0
  442. {mtplx-2.4.0 → mtplx-2.4.2}/vllm_metal/metal/kernels_v2/float8.metal +0 -0
  443. {mtplx-2.4.0 → mtplx-2.4.2}/vllm_metal/metal/kernels_v2/gather_kv_cache.metal +0 -0
  444. {mtplx-2.4.0 → mtplx-2.4.2}/vllm_metal/metal/kernels_v2/gdn_linear_attention.metal +0 -0
  445. {mtplx-2.4.0 → mtplx-2.4.2}/vllm_metal/metal/kernels_v2/kv_scale_update.metal +0 -0
  446. {mtplx-2.4.0 → mtplx-2.4.2}/vllm_metal/metal/kernels_v2/pagedattention.metal +0 -0
  447. {mtplx-2.4.0 → mtplx-2.4.2}/vllm_metal/metal/kernels_v2/reshape_and_cache.metal +0 -0
  448. {mtplx-2.4.0 → mtplx-2.4.2}/vllm_metal/metal/kernels_v2/turboquant.metal +0 -0
  449. {mtplx-2.4.0 → mtplx-2.4.2}/vllm_metal/metal/kernels_v2/utils.metal +0 -0
  450. {mtplx-2.4.0 → mtplx-2.4.2}/vllm_metal/metal/paged_ops.cpp +0 -0
@@ -4,6 +4,157 @@ All notable user-facing changes to MTPLX. The format is based on
4
4
  [Keep a Changelog](https://keepachangelog.com/en/1.1.0/); versions follow
5
5
  [Semantic Versioning](https://semver.org/).
6
6
 
7
+ ## [2.4.2] - 2026-08-02
8
+
9
+ The agentic-cache release: the session cache stops losing warm state
10
+ mid-run, tool-turn commits stop being ghosts, every serve keeps a durable
11
+ per-request trail by default, an experimental DeepSeek-V4-Flash backend
12
+ lands, and the documentation now matches the code everywhere it was
13
+ audited.
14
+
15
+ ### Added
16
+
17
+ - DeepSeek-V4-Flash: experimental native AR backend
18
+ (`model_type: deepseek_v4`) — Hyper-Connections, compressed sparse
19
+ attention, hash-routed MoE, grouped output-LoRA — loading the
20
+ mlx-community checkpoints directly, with an optional single-block MTP
21
+ speculative lane when the checkpoint carries `mtp.0.*` weights
22
+ (spec == AR gated; K=1-3 measured up to 2.28x on the 2bit-DQ build).
23
+ MTP-declaring checkpoints that ship no draft weights degrade to AR
24
+ with a clear message instead of failing at bind. Thanks @davidtai
25
+ (#216).
26
+ - Request log, default on: every serve writes numeric/hash per-request
27
+ telemetry to `~/.mtplx/logs/request-log-<port>.jsonl` (64 MB x 4
28
+ rotation; no prompt or completion content; disable with
29
+ `MTPLX_REQUEST_LOG_JSONL=off`). Pairs with 2.4.1's opt-in bit-exact
30
+ request capture to make agent-session incidents diagnosable after the
31
+ fact (#196/#197). New helpers: `scripts/gauntlet_scoreboard.py`
32
+ per-session summarizer, `scripts/oc_tap.py` recording proxy and
33
+ `scripts/oc_tap_diff.py` request-mutation analyzer for content-level
34
+ wire truth.
35
+ - Session bank, active-session eviction protection: sessions that
36
+ touched the bank within `MTPLX_SESSION_BANK_ACTIVE_PIN_TTL_S`
37
+ (default 600 s) are eviction-last, so cross-session pressure evicts
38
+ idle victims instead of the session that is mid-run.
39
+ - Session bank, newest-K per-session snapshot retention
40
+ (`MTPLX_SESSION_BANK_PER_SESSION_MAX_ENTRIES`, default 3): divergent
41
+ per-turn sibling snapshots no longer accumulate unreclaimed.
42
+ `/health` now reports active sessions, the pin TTL, and recent
43
+ evictions.
44
+ - Postcommit foreground grace (`MTPLX_POSTCOMMIT_FOREGROUND_GRACE_S`,
45
+ default 2 s): a nearly-finished background cache commit lands instead
46
+ of being preempted by the next fast agent-loop request.
47
+ - Session identity honors `x-session-affinity` / `x-session-id` request
48
+ headers (OpenCode sends these per request), ending cross-request
49
+ identity churn on that client.
50
+
51
+ ### Fixed
52
+
53
+ - Tool-turn "ghost re-prefills": the tool-rewrite async commit rendered
54
+ a canonical history that matched neither the generation nor the next
55
+ prompt, burning full-history re-forwards (26.8 s observed) without
56
+ ever storing. It is disabled pending a byte-proven canonical render
57
+ (`MTPLX_IDLE_POSTCOMMIT_TOOL_REWRITE` re-enables);
58
+ store-on-prefill and block salvage cover the lane.
59
+ - The bridge's convergence guard now states explicitly that editing and
60
+ verification tools remain allowed and that its restriction covers
61
+ only the current reply — a model read the old wording as a
62
+ session-wide tool ban and stalled an entire session.
63
+ - `mtplx profile thermal`, `profile eval-attribution`,
64
+ `profile dispatch --trace`, and `thermal fanmax-run` invoked
65
+ research-workspace scripts that are not part of the distribution, and
66
+ `--dry-run` printed those phantom paths as runnable commands. They
67
+ now report availability honestly (exit 2, machine-readable
68
+ `available: false`) and run the real script when present.
69
+ - `mtplx doctor`: Python floor corrected to 3.11 (matching
70
+ `requires-python`); remediation texts no longer tell end users to
71
+ edit source constants or to move a healthy server off its port;
72
+ `--port` is documented and, when passed explicitly, aims the server
73
+ connectivity checks.
74
+ - Session-bank near-prefix restores on backends with bounded rollback
75
+ (DeepSeek-V4) pre-check `max_rollback` and fall back to a cold
76
+ prefill instead of raising (#216).
77
+ - Help surfaces match their own parsers: the onboarding help no longer
78
+ promises a Turbo wizard choice that does not exist (Turbo
79
+ auto-selects for the quantized flagships), `--strict-cold` names the
80
+ enforced 59 tok/s gate, `--open-dashboard` opens alongside the chosen
81
+ client (as it always did), and the command reference teaches
82
+ `mtplx <command> --help`, which also works for multi-word commands.
83
+
84
+ ### Documentation
85
+
86
+ - Full truth sweep: ~450 documentation claims reconciled against the
87
+ code across 27 files. Highlights: INSTALL.md no longer references an
88
+ MLX fork removed in 2.0.0; turbo-verify.md no longer calls the
89
+ shipped default "experimental, off by default" nor excludes the
90
+ 6-bit lane that ships; the Anthropic base-URL instruction (docs and
91
+ the canonical example) no longer 404s; `/metrics` no longer claims a
92
+ Prometheus mode that never existed; the README modes table shows
93
+ Turbo as the default for the quantized 27B/9B flagships; the Laguna
94
+ memory requirement states the real ~85.3 GiB preflight gate;
95
+ version-era staleness ("v0.1", "preview", v0.3.x runbook pins) is
96
+ cleared; historical release notes gain bracketed corrections where
97
+ they documented commands that never worked. Thanks
98
+ @PhilipJohnBasile for #218 (removed the unsupported MTP-sidecar
99
+ graft guidance; seeded by #215).
100
+ - Dependency-record correction: the transformers pin has been
101
+ `<5.14,!=5.13.0` since shortly after 2.0.0; the changelog never
102
+ recorded the relaxation from `<5.13`.
103
+
104
+ ### Dependencies
105
+
106
+ - pypa/gh-action-pypi-publish 1.14.1 -> 1.14.2 (#217).
107
+
108
+ ## [2.4.1] - 2026-08-01
109
+
110
+ The smooth-streaming release: the app's chat render path is overhauled
111
+ (no more freeze-then-catch-up stutter, scroll bounce, or plain-text code
112
+ blocks — real syntax coloring, live code cards, tables, and actual math
113
+ notation), and the 2.4.0 short-turn regression is fixed.
114
+
115
+ ### Added
116
+
117
+ - Live syntax coloring for code blocks (12 languages + generic) from a
118
+ freeze-time lexer that colors each line exactly once; streaming cost
119
+ is O(new text), never O(document).
120
+ - Streaming code card: an open fence renders as a live card with
121
+ colored lines and flips once to its settled form at close.
122
+ - Pipe tables render as real tables; math renders as real notation
123
+ (Unicode super/subscripts, stacked matrices and fractions, inline
124
+ conversion instead of dollar-sign leaks).
125
+ - Typewriter pacing for streamed text with geometric catch-up and a
126
+ hard drain bound (`MTPLX_STREAM_TYPEWRITER=0` to disable), and a live
127
+ tok/s chip computed over a sliding ~5 s window.
128
+ - Performance mode is a true kill switch: plain text only, through both
129
+ the streaming and settled render paths.
130
+ - Opt-in per-request capture for bit-exact failure replay:
131
+ `MTPLX_REQUEST_CAPTURE_DIR=<dir>` persists each request's
132
+ reproduction envelope at dispatch time (#196/#197, third layer).
133
+ - Opt-in frontend stream-performance probe (`MTPLX_UI_PERF=1`, HUD via
134
+ `MTPLX_UI_PERF_HUD=1`) with a per-turn JSONL trace joinable to engine
135
+ stats by request id.
136
+ - Experimental: cost-model speculative-depth policy
137
+ (`--adaptive-policy cost`) and blocked-sequential GDN prefill
138
+ (`MTPLX_GDN_BLOCKED_PREFILL=1`). Defaults unchanged.
139
+
140
+ ### Fixed
141
+
142
+ - 2.4.0 short-turn regression: the compiled-verify path could reserve
143
+ KV budget above the configured ceiling, taxing short requests with
144
+ setup work they never used; the reserve is now clamped.
145
+ - Warming prefills yield to real traffic within one small chunk instead
146
+ of delaying a freshly arrived request.
147
+ - Derivative model artifacts whose names extend a first-party model
148
+ name are served under their own id, not the flagship's — the health
149
+ payload, OpenAI `model` field, and app model chip now report the
150
+ artifact actually loaded.
151
+ - Streaming render: line-segment coalescing keeps realized view count
152
+ bounded on long answers; the bottom-pin scroll correction runs in the
153
+ same display cycle as layout so the streaming bubble can no longer
154
+ visibly bounce; a per-display-cycle window-sizing walk that floored
155
+ every update at ~50 ms is removed (`MTPLX_APP_SIZING_TUNER=0`
156
+ restores it).
157
+
7
158
  ## [2.4.0] - 2026-07-31
8
159
 
9
160
  The 35B speed release: the 35B-A3B MoE gets a compiled decode stack and
@@ -676,4 +827,17 @@ working as one product. Full notes:
676
827
  completions, and Anthropic `stop_sequences`) and `/v1/completions`
677
828
  streams tokens as they are generated with real finish reasons.
678
829
 
679
- [1.0.0]: https://github.com/youssofal/mtplx/releases/tag/v1.0.0
830
+ [2.4.2]: https://github.com/youssofal/MTPLX/releases/tag/v2.4.2
831
+ [2.4.1]: https://github.com/youssofal/MTPLX/releases/tag/v2.4.1
832
+ [2.4.0]: https://github.com/youssofal/MTPLX/releases/tag/v2.4.0
833
+ [2.3.0]: https://github.com/youssofal/MTPLX/releases/tag/v2.3.0
834
+ [2.2.0]: https://github.com/youssofal/MTPLX/releases/tag/v2.2.0
835
+ [2.1.0]: https://github.com/youssofal/MTPLX/releases/tag/v2.1.0
836
+ [2.0.2]: https://github.com/youssofal/MTPLX/releases/tag/v2.0.2
837
+ [2.0.1]: https://github.com/youssofal/MTPLX/releases/tag/v2.0.1
838
+ [2.0.0]: https://github.com/youssofal/MTPLX/releases/tag/v2.0.0
839
+ [1.0.4]: https://github.com/youssofal/MTPLX/releases/tag/v1.0.4
840
+ [1.0.3]: https://github.com/youssofal/MTPLX/releases/tag/v1.0.3
841
+ [1.0.2]: https://github.com/youssofal/MTPLX/releases/tag/v1.0.2
842
+ [1.0.1]: https://github.com/youssofal/MTPLX/releases/tag/v1.0.1
843
+ [1.0.0]: https://github.com/youssofal/MTPLX/releases/tag/v1.0.0
@@ -8,7 +8,7 @@ authors:
8
8
  repository-code: "https://github.com/youssofal/mtplx"
9
9
  url: "https://github.com/youssofal/mtplx"
10
10
  license: Apache-2.0
11
- version: 0.1.0rc1
11
+ version: 2.4.2
12
12
  abstract: "Native MTP speculative decoding for Qwen3-Next on Apple Silicon, using built-in MTP heads with math-correct rejection sampling and an OpenAI/Anthropic-compatible serving surface."
13
13
  keywords:
14
14
  - speculative decoding
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: mtplx
3
- Version: 2.4.0
3
+ Version: 2.4.2
4
4
  Summary: Native MTP speculative decoding for Qwen3-Next on Apple Silicon.
5
5
  Author-email: Youssof Altoukhi <business@youssofal.com>
6
6
  License-Expression: Apache-2.0
@@ -105,9 +105,11 @@ On a 16 GB M4 Mac mini, tuning the 9B model lands on depth 1: 14.4 tok/s baselin
105
105
 
106
106
  <img src="docs/assets/readme/app-forge.jpg" alt="Forge verifying a freshly built MTP model" width="100%" />
107
107
 
108
- Forge takes a Hugging Face repo and turns it into an MTPLX-ready MTP model: convert to MLX, train the MTP adapter, verify that the result is actually faster and still exact, and publish back to the Hub if you want to share it. The honest part matters: Forge measures before and after on your hardware and shows you the verdict ("Depth 1 is fastest: 227.1 to 296.1, 1.30x") rather than assuming the adapter helped. Available in the app and as `mtplx forge`.
108
+ Forge takes a Hugging Face repo and turns it into an MTPLX-ready MTP model: convert to MLX, train the MTP adapter, verify that the result is actually faster and still exact, and publish back to the Hub if you want to share it. The honest part matters: Forge measures before and after on your hardware and shows you the verdict ("Depth 1 is fastest: 227.1 to 296.1, 1.30x") rather than assuming the adapter helped. Available in the app and as `mtplx forge` subcommands.
109
109
 
110
- The official catalog lives on Hugging Face under [Youssofal](https://huggingface.co/Youssofal): Qwen 3.5 (4B, 9B), Qwen 3.6 (27B, 35B MoE) in speed, balance, and quality builds, plus Gemma 4. The app recommends from these based on your hardware.
110
+ MTPLX does not support attaching a separately supplied MTP sidecar to an arbitrary MLX trunk. Matching architecture fields, tensor shapes, or provenance labels cannot prove that the head was trained against those exact trunk weights. Use a complete model that already includes its matching MTP weights, or use Forge to build and verify an artifact from its original source checkpoint.
111
+
112
+ The official catalog lives on Hugging Face under [Youssofal](https://huggingface.co/Youssofal): Qwen 3.5 (4B, 9B), Qwen 3.6 (27B, 35B MoE) in speed and quality builds (the 35B MoE adds a balance build), plus Gemma 4. The app recommends from these based on your hardware.
111
113
 
112
114
  ## The server
113
115
 
@@ -119,7 +121,7 @@ curl http://127.0.0.1:8000/v1/chat/completions \
119
121
  -d '{"model":"mtplx","messages":[{"role":"user","content":"hi"}],"stream":true}'
120
122
  ```
121
123
 
122
- Sessions survive: a warm-prefix session bank keeps multi-turn chats fast, and an optional SSD cache restores sessions near-instantly across restarts.
124
+ Sessions survive: a warm-prefix session bank keeps multi-turn chats fast, and a default-on SSD session cache restores sessions near-instantly across restarts (disable with `--ssd-session-cache off`).
123
125
 
124
126
  Sampler controls cover `temperature`, `top_p`, `top_k`, and the OpenAI penalty pair `presence_penalty` / `frequency_penalty` — per request, as server defaults (`--default-presence-penalty` / `--default-frequency-penalty` on `start`/`serve`/`quickstart`), or live via `mtplx settings set` and the app's Presence Penalty dial. Penalties default to 0, which is an exact no-op that preserves MTP exactness. Qwen's guidance: leave them at 0 for coding and agent work; ~0.5–1.5 presence penalty helps creative writing or when a model loops on itself.
125
127
 
@@ -133,20 +135,21 @@ mtplx pull <hf-repo> # download a model safely
133
135
  mtplx models # what is cached, sizes, validation
134
136
  mtplx inspect <model> # compatibility report before anything runs
135
137
  mtplx tune --retune # measure AR vs D1/D2/D3 on your Mac
136
- mtplx forge # build, verify, and publish MTP models
138
+ mtplx forge --help # build, verify, and publish MTP models (probe/build/publish/verify subcommands)
137
139
  mtplx bench aime --quick # run the AIME benchmark from the terminal
138
140
  mtplx doctor # install and integration health
139
141
  mtplx max --install # fan control (one sudo prompt, crash-safe)
140
142
  mtplx settings get/set # read or change live server settings
141
143
  ```
142
144
 
143
- Every command takes `--json` and `--help`. The CLI works without MLX installed for everything that does not need a model, so `doctor` and `inspect` run on any machine.
145
+ Every command takes `--help`, and most inspection/diagnostic commands take `--json`. The CLI works without MLX installed for everything that does not need a model, so `doctor` and `inspect` run on any machine.
144
146
 
145
147
  ## Modes
146
148
 
147
149
  | Mode | What it does | When |
148
150
  |---|---|---|
149
- | **Sustained** | Default. Long-context MTP path with chunked prefill and request-sized KV | Everyday use, big files, 16K-200K prompts |
151
+ | **Turbo** | NAX verify kernels + compiled verify; the default for the quantized 27B and 9B flagship models | Picked automatically for those models |
152
+ | **Sustained** | Default for all other models. Long-context MTP path with chunked prefill and request-sized KV | Everyday use, big files, 16K-200K prompts |
150
153
  | **Sustained Max** | Sustained with fans pinned at 100% | Long work where you want maximum cooling |
151
154
  | **Burst** | Legacy short-context benchmark lane, loud | Short prompts and benchmarks only |
152
155
 
@@ -154,7 +157,7 @@ Fan-backed modes restore your fans to automatic if MTPLX dies for any reason, in
154
157
 
155
158
  ## Compatibility, honestly
156
159
 
157
- `mtplx inspect` classifies models before anything runs: verified, architecture-compatible but unverified, AR-only, incompatible architecture, or no MTP heads at all. Unverified models refuse to run unless you explicitly force them. There are no silent fallbacks: if MTPLX cannot run a model correctly, it tells you instead of running it badly.
160
+ `mtplx inspect` classifies models before anything runs: verified, family-compatible but unverified, architecture-compatible but unverified, AR-only, incompatible architecture, or no MTP heads at all. Unverified models load with an explicit unverified label. There are no silent fallbacks: if MTPLX cannot run a model correctly, it tells you instead of running it badly.
158
161
 
159
162
  [Laguna-S-2.1 oQ4e](https://huggingface.co/mlx-community/Laguna-S-2.1-oQ4e) is supported through its exact MLX architecture in target-only AR mode:
160
163
 
@@ -170,8 +173,10 @@ MTPLX pins that model to revision
170
173
  tokenizer, generation config, special tokens map, and Poolside chat template
171
174
  before admitting it. The checkpoint has no native MTP head, so an MTP launch is
172
175
  rejected before weights load instead of falling back during execution. The
173
- weights occupy 59.72 GiB (64.13 GB); use a Mac with at least 96 GiB unified
174
- memory (128 GiB is recommended). MTPLX defaults Laguna to a 32,768-token context
176
+ weights occupy 59.72 GiB, a 64.13 GB snapshot on disk. The launch preflight
177
+ requires about 85 GiB of unified memory (weights, runtime headroom, and a
178
+ 16 GiB system reserve) — in practice a 96 GB Mac; 128 GB is
179
+ comfortable. MTPLX defaults Laguna to a 32,768-token context
175
180
  and response cap, and checks larger explicit server contexts against the active
176
181
  Metal memory cap.
177
182
 
@@ -55,9 +55,11 @@ On a 16 GB M4 Mac mini, tuning the 9B model lands on depth 1: 14.4 tok/s baselin
55
55
 
56
56
  <img src="docs/assets/readme/app-forge.jpg" alt="Forge verifying a freshly built MTP model" width="100%" />
57
57
 
58
- Forge takes a Hugging Face repo and turns it into an MTPLX-ready MTP model: convert to MLX, train the MTP adapter, verify that the result is actually faster and still exact, and publish back to the Hub if you want to share it. The honest part matters: Forge measures before and after on your hardware and shows you the verdict ("Depth 1 is fastest: 227.1 to 296.1, 1.30x") rather than assuming the adapter helped. Available in the app and as `mtplx forge`.
58
+ Forge takes a Hugging Face repo and turns it into an MTPLX-ready MTP model: convert to MLX, train the MTP adapter, verify that the result is actually faster and still exact, and publish back to the Hub if you want to share it. The honest part matters: Forge measures before and after on your hardware and shows you the verdict ("Depth 1 is fastest: 227.1 to 296.1, 1.30x") rather than assuming the adapter helped. Available in the app and as `mtplx forge` subcommands.
59
59
 
60
- The official catalog lives on Hugging Face under [Youssofal](https://huggingface.co/Youssofal): Qwen 3.5 (4B, 9B), Qwen 3.6 (27B, 35B MoE) in speed, balance, and quality builds, plus Gemma 4. The app recommends from these based on your hardware.
60
+ MTPLX does not support attaching a separately supplied MTP sidecar to an arbitrary MLX trunk. Matching architecture fields, tensor shapes, or provenance labels cannot prove that the head was trained against those exact trunk weights. Use a complete model that already includes its matching MTP weights, or use Forge to build and verify an artifact from its original source checkpoint.
61
+
62
+ The official catalog lives on Hugging Face under [Youssofal](https://huggingface.co/Youssofal): Qwen 3.5 (4B, 9B), Qwen 3.6 (27B, 35B MoE) in speed and quality builds (the 35B MoE adds a balance build), plus Gemma 4. The app recommends from these based on your hardware.
61
63
 
62
64
  ## The server
63
65
 
@@ -69,7 +71,7 @@ curl http://127.0.0.1:8000/v1/chat/completions \
69
71
  -d '{"model":"mtplx","messages":[{"role":"user","content":"hi"}],"stream":true}'
70
72
  ```
71
73
 
72
- Sessions survive: a warm-prefix session bank keeps multi-turn chats fast, and an optional SSD cache restores sessions near-instantly across restarts.
74
+ Sessions survive: a warm-prefix session bank keeps multi-turn chats fast, and a default-on SSD session cache restores sessions near-instantly across restarts (disable with `--ssd-session-cache off`).
73
75
 
74
76
  Sampler controls cover `temperature`, `top_p`, `top_k`, and the OpenAI penalty pair `presence_penalty` / `frequency_penalty` — per request, as server defaults (`--default-presence-penalty` / `--default-frequency-penalty` on `start`/`serve`/`quickstart`), or live via `mtplx settings set` and the app's Presence Penalty dial. Penalties default to 0, which is an exact no-op that preserves MTP exactness. Qwen's guidance: leave them at 0 for coding and agent work; ~0.5–1.5 presence penalty helps creative writing or when a model loops on itself.
75
77
 
@@ -83,20 +85,21 @@ mtplx pull <hf-repo> # download a model safely
83
85
  mtplx models # what is cached, sizes, validation
84
86
  mtplx inspect <model> # compatibility report before anything runs
85
87
  mtplx tune --retune # measure AR vs D1/D2/D3 on your Mac
86
- mtplx forge # build, verify, and publish MTP models
88
+ mtplx forge --help # build, verify, and publish MTP models (probe/build/publish/verify subcommands)
87
89
  mtplx bench aime --quick # run the AIME benchmark from the terminal
88
90
  mtplx doctor # install and integration health
89
91
  mtplx max --install # fan control (one sudo prompt, crash-safe)
90
92
  mtplx settings get/set # read or change live server settings
91
93
  ```
92
94
 
93
- Every command takes `--json` and `--help`. The CLI works without MLX installed for everything that does not need a model, so `doctor` and `inspect` run on any machine.
95
+ Every command takes `--help`, and most inspection/diagnostic commands take `--json`. The CLI works without MLX installed for everything that does not need a model, so `doctor` and `inspect` run on any machine.
94
96
 
95
97
  ## Modes
96
98
 
97
99
  | Mode | What it does | When |
98
100
  |---|---|---|
99
- | **Sustained** | Default. Long-context MTP path with chunked prefill and request-sized KV | Everyday use, big files, 16K-200K prompts |
101
+ | **Turbo** | NAX verify kernels + compiled verify; the default for the quantized 27B and 9B flagship models | Picked automatically for those models |
102
+ | **Sustained** | Default for all other models. Long-context MTP path with chunked prefill and request-sized KV | Everyday use, big files, 16K-200K prompts |
100
103
  | **Sustained Max** | Sustained with fans pinned at 100% | Long work where you want maximum cooling |
101
104
  | **Burst** | Legacy short-context benchmark lane, loud | Short prompts and benchmarks only |
102
105
 
@@ -104,7 +107,7 @@ Fan-backed modes restore your fans to automatic if MTPLX dies for any reason, in
104
107
 
105
108
  ## Compatibility, honestly
106
109
 
107
- `mtplx inspect` classifies models before anything runs: verified, architecture-compatible but unverified, AR-only, incompatible architecture, or no MTP heads at all. Unverified models refuse to run unless you explicitly force them. There are no silent fallbacks: if MTPLX cannot run a model correctly, it tells you instead of running it badly.
110
+ `mtplx inspect` classifies models before anything runs: verified, family-compatible but unverified, architecture-compatible but unverified, AR-only, incompatible architecture, or no MTP heads at all. Unverified models load with an explicit unverified label. There are no silent fallbacks: if MTPLX cannot run a model correctly, it tells you instead of running it badly.
108
111
 
109
112
  [Laguna-S-2.1 oQ4e](https://huggingface.co/mlx-community/Laguna-S-2.1-oQ4e) is supported through its exact MLX architecture in target-only AR mode:
110
113
 
@@ -120,8 +123,10 @@ MTPLX pins that model to revision
120
123
  tokenizer, generation config, special tokens map, and Poolside chat template
121
124
  before admitting it. The checkpoint has no native MTP head, so an MTP launch is
122
125
  rejected before weights load instead of falling back during execution. The
123
- weights occupy 59.72 GiB (64.13 GB); use a Mac with at least 96 GiB unified
124
- memory (128 GiB is recommended). MTPLX defaults Laguna to a 32,768-token context
126
+ weights occupy 59.72 GiB, a 64.13 GB snapshot on disk. The launch preflight
127
+ requires about 85 GiB of unified memory (weights, runtime headroom, and a
128
+ 16 GiB system reserve) — in practice a 96 GB Mac; 128 GB is
129
+ comfortable. MTPLX defaults Laguna to a 32,768-token context
125
130
  and response cap, and checks larger explicit server contexts against the active
126
131
  Metal memory cap.
127
132
 
@@ -250,3 +250,254 @@ class ExpectedValueDepthPolicy:
250
250
  prob_term = 2.0 * _clamp(float(top1_prob), 0.0, 1.0) - 1.0
251
251
  raw = 1.0 + self.confidence_weight * (0.75 * margin_term + 0.25 * prob_term)
252
252
  return _clamp(raw, 0.25, 1.75)
253
+
254
+
255
+ class CostModelDepthPolicy:
256
+ """Cost-model adaptive depth: maximize expected committed tokens per
257
+ wall-clock cycle, not acceptance streaks.
258
+
259
+ Ported from omlx's ``_DepthController`` (jundot/omlx v0.5.4rc1,
260
+ omlx/patches/mlx_lm_mtp/batch_generator.py) with the depth-0
261
+ park/exit machinery deliberately left out for now (our 27B lanes
262
+ always profit from speculation; the escape hatch matters for
263
+ head_dim-512/MoE models and can come later). Their measured design
264
+ decisions preserved verbatim:
265
+
266
+ - ``score(d) = (1 + p1 + p1 p2 + ...) / t_est(d)``.
267
+ - Acceptance is a token-domain EMA (a property of model/content);
268
+ cost is a wall-clock-horizon EMA (tracks context growth, thermal
269
+ state, and external GPU load at constant real-time responsiveness)
270
+ with a one-off-spike damp.
271
+ - The marginal cost of an extra verify row is the measured slope
272
+ between the cheapest and priciest measured depths, not a constant.
273
+ - Probes are bidirectional and staleness-directed, duty-bounded to
274
+ ~15% of cycles: re-measuring a SHALLOWER rival is what breaks the
275
+ depth-2 lock omlx measured (stale-high t[1] hides depth 1 forever).
276
+ - The cost EMA is per-cycle, NOT staleness-age-weighted: omlx
277
+ measured age-weighting worse (probe-burst noise injected straight
278
+ into the decision; prose re-over-drafted 1.6%).
279
+
280
+ Drop-in for the ``AdaptiveDepthPolicy`` interface: ``current_depth``
281
+ plus ``observe(attempted_depth=, accepted_depths=)``. Cycle cost is
282
+ self-timed as the wall interval between observe calls (one observe
283
+ per verify cycle), which deliberately includes the loop's host
284
+ bookkeeping — that tax is part of the real cost of running a cycle.
285
+ """
286
+
287
+ ALPHA = 0.08
288
+ TAU_MS = 400.0
289
+ PROBE_PERIOD_MS = 1000.0
290
+ PROBE_PERIOD_MAX_MS = 5000.0
291
+ PROBE_LEN = 4
292
+ PROBE_DUTY = 0.15
293
+ PROBE_MARGIN = 1.15
294
+ SPIKE_RATIO = 2.0
295
+ SPIKE_DAMP = 0.25
296
+ MARGINAL_MS = 7.0
297
+ HYSTERESIS = 1.03
298
+ # Ignore absurd inter-observe gaps (queue waits, tool round-trips in
299
+ # agent serving): a "cycle" above this is not a cycle measurement.
300
+ MAX_CYCLE_MS = 5000.0
301
+ # The generation loop passes its own measured cycle wall-time when it
302
+ # sees this flag; the self-timed inter-observe fallback stays for
303
+ # callers that do not.
304
+ accepts_cycle_ms = True
305
+
306
+ def __init__(
307
+ self,
308
+ max_depth: int,
309
+ min_depth: int = 1,
310
+ marginal_ms: float | None = None,
311
+ ) -> None:
312
+ if max_depth < 1:
313
+ raise ValueError("max_depth must be >= 1")
314
+ self.max_depth = int(max_depth)
315
+ self.min_depth = max(1, min(int(min_depth), self.max_depth))
316
+ if marginal_ms:
317
+ self.MARGINAL_MS = float(marginal_ms)
318
+ self.current_depth = self.max_depth
319
+ self.p = [0.6] * self.max_depth
320
+ self.t: dict[int, float] = {}
321
+ self.t_age: dict[int, float] = {}
322
+ self.cycles = 0
323
+ self.probe_left = 0
324
+ self._ms_probe = 0.0
325
+ self._ms_explore = 0.0
326
+ self._warmup = list(range(self.max_depth, self.min_depth - 1, -1))
327
+ self._last_observe_s: float | None = None
328
+
329
+ # -- cost bookkeeping --------------------------------------------------
330
+
331
+ def _time_alpha(self, cycle_ms: float) -> float:
332
+ return 1.0 - math.exp(-max(0.0, float(cycle_ms)) / self.TAU_MS)
333
+
334
+ def _update_time(self, used: int, cycle_ms: float) -> None:
335
+ prev = self.t.get(used)
336
+ if prev is None:
337
+ self.t[used] = cycle_ms
338
+ return
339
+ if self._warmup:
340
+ self.t[used] = min(prev, cycle_ms)
341
+ return
342
+ a = self._time_alpha(cycle_ms)
343
+ if cycle_ms > self.SPIKE_RATIO * prev:
344
+ a *= self.SPIKE_DAMP
345
+ self.t[used] = (1.0 - a) * prev + a * cycle_ms
346
+
347
+ def _marginal_est(self) -> float:
348
+ if len(self.t) >= 2:
349
+ depths = sorted(self.t)
350
+ lo, hi = depths[0], depths[-1]
351
+ if hi > lo:
352
+ slope = (self.t[hi] - self.t[lo]) / (hi - lo)
353
+ if slope > 0.0:
354
+ return slope
355
+ return self.MARGINAL_MS
356
+
357
+ def _t_est(self, d: int) -> float:
358
+ if d in self.t:
359
+ return self.t[d]
360
+ if not self.t:
361
+ return 30.0 + self.MARGINAL_MS * d
362
+ ref = min(self.t, key=lambda x: abs(x - d))
363
+ return max(1e-3, self.t[ref] + self._marginal_est() * (d - ref))
364
+
365
+ def _score(self, d: int) -> float:
366
+ expected = 1.0
367
+ run = 1.0
368
+ for j in range(d):
369
+ run *= self.p[j]
370
+ expected += run
371
+ return expected / max(1e-6, self._t_est(d))
372
+
373
+ # -- selection ---------------------------------------------------------
374
+
375
+ def _depths(self) -> list[int]:
376
+ return list(range(self.min_depth, self.max_depth + 1))
377
+
378
+ def _best(self) -> int:
379
+ cur_score = self._score(self.current_depth)
380
+ best_d, best_score = self.current_depth, cur_score
381
+ for d in self._depths():
382
+ s = self._score(d)
383
+ if s > best_score:
384
+ best_d, best_score = d, s
385
+ if best_d != self.current_depth and best_score < cur_score * self.HYSTERESIS:
386
+ return self.current_depth
387
+ return best_d
388
+
389
+ def _best_rival(self) -> int | None:
390
+ best = self._score(self.current_depth)
391
+ if best <= 0.0:
392
+ return self._most_stale()
393
+ rival, rival_score = None, 0.0
394
+ for d in self._depths():
395
+ if d == self.current_depth:
396
+ continue
397
+ s = self._score(d)
398
+ if s > rival_score:
399
+ rival, rival_score = d, s
400
+ if rival is not None and rival_score * self.PROBE_MARGIN >= best:
401
+ return rival
402
+ return None
403
+
404
+ def _most_stale(self) -> int | None:
405
+ candidates = [d for d in self._depths() if d != self.current_depth]
406
+ if not candidates:
407
+ return None
408
+ never = [d for d in candidates if d not in self.t]
409
+ if never:
410
+ return never[0]
411
+ return max(candidates, key=lambda d: self.t_age.get(d, 0.0))
412
+
413
+ # -- the drop-in interface ---------------------------------------------
414
+
415
+ def observe(
416
+ self,
417
+ *,
418
+ attempted_depth: int,
419
+ accepted_depths: int,
420
+ cycle_ms: float | None = None,
421
+ ) -> dict:
422
+ import time as _time
423
+
424
+ now = _time.perf_counter()
425
+ if cycle_ms is None and self._last_observe_s is not None:
426
+ cycle_ms = (now - self._last_observe_s) * 1000.0
427
+ if cycle_ms is not None and (cycle_ms > self.MAX_CYCLE_MS or cycle_ms <= 0.0):
428
+ cycle_ms = None
429
+ self._last_observe_s = now
430
+
431
+ self.cycles += 1
432
+ used = max(1, min(int(attempted_depth), self.max_depth))
433
+ accepted = max(0, min(int(accepted_depths), used))
434
+ previous_depth = self.current_depth
435
+
436
+ a = self.ALPHA
437
+ for j in range(used):
438
+ hit = 1.0 if j < accepted else 0.0
439
+ self.p[j] = (1.0 - a) * self.p[j] + a * hit
440
+ if j >= accepted:
441
+ break
442
+
443
+ if cycle_ms is not None:
444
+ self._update_time(used, cycle_ms)
445
+ for d in list(self.t_age):
446
+ self.t_age[d] += cycle_ms
447
+ self.t_age[used] = 0.0
448
+ self._ms_probe += cycle_ms
449
+ self._ms_explore += cycle_ms
450
+
451
+ action = "hold"
452
+ if self._warmup:
453
+ # A warmup slot is consumed only once its depth has a real cost
454
+ # sample; the very first observe has no prior timestamp to diff
455
+ # against, so that cycle repeats its depth instead of advancing.
456
+ if cycle_ms is not None and used == self._warmup[0]:
457
+ self._warmup.pop(0)
458
+ if self._warmup:
459
+ self.current_depth = self._warmup[0]
460
+ action = "warmup"
461
+ else:
462
+ self.current_depth = self._best()
463
+ self._ms_probe = 0.0
464
+ action = "warmup_done"
465
+ elif self.probe_left > 0:
466
+ self.probe_left -= 1
467
+ if self.probe_left == 0:
468
+ self.current_depth = self._best()
469
+ self._ms_probe = 0.0
470
+ action = "probe_done"
471
+ else:
472
+ action = "probing"
473
+ else:
474
+ self.current_depth = self._best()
475
+ if self.current_depth != previous_depth:
476
+ action = (
477
+ "increase" if self.current_depth > previous_depth else "decrease"
478
+ )
479
+ if self.max_depth > self.min_depth and cycle_ms is not None:
480
+ period = max(
481
+ self.PROBE_PERIOD_MS,
482
+ self.PROBE_LEN * cycle_ms / self.PROBE_DUTY,
483
+ )
484
+ if self._ms_probe >= period:
485
+ explore_due = self._ms_explore >= max(
486
+ self.PROBE_PERIOD_MAX_MS, 2.0 * period
487
+ )
488
+ target = self._most_stale() if explore_due else self._best_rival()
489
+ if target is not None:
490
+ self.current_depth = target
491
+ self.probe_left = self.PROBE_LEN
492
+ self._ms_probe = 0.0
493
+ if explore_due:
494
+ self._ms_explore = 0.0
495
+ action = "probe"
496
+
497
+ return {
498
+ "previous_depth": previous_depth,
499
+ "attempted_depth": used,
500
+ "accepted_depths": accepted,
501
+ "next_depth": self.current_depth,
502
+ "action": action,
503
+ }
@@ -313,6 +313,62 @@ def expected_mtp_file(model_dir: Path | str, config: dict[str, Any] | None = Non
313
313
  return model_path / "mtp.safetensors"
314
314
 
315
315
 
316
+ def mtp_weights_present_on_disk(
317
+ model_dir: Path | str, config: dict[str, Any] | None = None
318
+ ) -> bool:
319
+ """Whether a model that declares MTP layers actually ships MTP weights.
320
+
321
+ A conversion can declare ``num_nextn_predict_layers`` in the config while
322
+ dropping the MTP weights themselves (e.g. the DeepSeek-V4-Flash 2bit-DQ
323
+ build). The runtime uses this probe to tell that benign case (config field
324
+ only -> degrade to autoregressive) apart from a genuine injection failure
325
+ (weights present but unusable -> raise).
326
+
327
+ Conservative by design: it only returns ``False`` when it can *positively*
328
+ confirm absence via a shard index that carries no MTP-shaped keys under any
329
+ known naming convention. A sidecar file, a missing/unreadable index, or any
330
+ ambiguity returns ``True`` so the existing injection + validation path runs
331
+ unchanged and a real detection bug on an MTP-bearing model still surfaces.
332
+ """
333
+ model_path = Path(model_dir)
334
+ config = config if config is not None else load_config(model_path)
335
+
336
+ # 1. Explicit MTP sidecar file (Qwen/GLM/hy3 external draft head).
337
+ if expected_mtp_file(model_path, config).exists():
338
+ return True
339
+
340
+ index_path = model_path / "model.safetensors.index.json"
341
+ if not index_path.exists():
342
+ # No index to inspect: cannot prove absence, preserve legacy behavior.
343
+ return True
344
+ try:
345
+ weight_map = json.loads(index_path.read_text(encoding="utf-8")).get(
346
+ "weight_map", {}
347
+ )
348
+ except Exception:
349
+ return True
350
+ keys = [str(k) for k in weight_map]
351
+
352
+ # 2. Namespaced embedded MTP weights ("mtp.*" / "language_model.mtp.*").
353
+ if any(is_mtp_key(k) for k in keys):
354
+ return True
355
+
356
+ # 3. DeepSeek-style trailing MTP decoder layer(s) appended after the trunk:
357
+ # model.layers.{num_hidden_layers + i}.*
358
+ start = int(
359
+ text_config(config).get("num_hidden_layers")
360
+ or config.get("num_hidden_layers")
361
+ or 0
362
+ )
363
+ count = _num_mtp_layers(config)
364
+ if start and count:
365
+ wanted = tuple(f"model.layers.{start + i}." for i in range(count))
366
+ if any(k.startswith(wanted) for k in keys):
367
+ return True
368
+
369
+ return False
370
+
371
+
316
372
  @dataclass(frozen=True)
317
373
  class TensorInfo:
318
374
  key: str