mtplx 2.4.0__tar.gz → 2.4.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {mtplx-2.4.0 → mtplx-2.4.2}/CHANGELOG.md +165 -1
- {mtplx-2.4.0 → mtplx-2.4.2}/CITATION.cff +1 -1
- {mtplx-2.4.0 → mtplx-2.4.2}/PKG-INFO +15 -10
- {mtplx-2.4.0 → mtplx-2.4.2}/README.md +14 -9
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/adaptive.py +251 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/artifacts.py +56 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/backends/registry.py +67 -6
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/cli.py +24 -11
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/commands/public.py +73 -8
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/default_models.py +72 -30
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/diagnostics.py +12 -9
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/engine_session.py +7 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/generation.py +20 -2
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/graphbank.py +26 -3
- mtplx-2.4.2/mtplx/kernels/gdn_blocked_prefill.py +468 -0
- mtplx-2.4.2/mtplx/models/deepseek_v4.py +3440 -0
- mtplx-2.4.2/mtplx/request_capture.py +130 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/runtime.py +45 -3
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/server/openai.py +222 -16
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/session_bank.py +134 -1
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/version.py +2 -2
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx.egg-info/PKG-INFO +15 -10
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx.egg-info/SOURCES.txt +38 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/pyproject.toml +1 -1
- mtplx-2.4.2/scripts/deepseek_v4_build_mtp_model.py +673 -0
- mtplx-2.4.2/scripts/deepseek_v4_decode_verify.py +198 -0
- mtplx-2.4.2/scripts/deepseek_v4_dispatch_census.py +248 -0
- mtplx-2.4.2/scripts/deepseek_v4_guard_window.py +316 -0
- mtplx-2.4.2/scripts/deepseek_v4_logits_gate.py +93 -0
- mtplx-2.4.2/scripts/deepseek_v4_moe_tail_arms.sh +150 -0
- mtplx-2.4.2/scripts/deepseek_v4_moe_tail_gate.py +508 -0
- mtplx-2.4.2/scripts/deepseek_v4_moe_tail_guarded_bracket.py +301 -0
- mtplx-2.4.2/scripts/deepseek_v4_mtp_bind_check.py +162 -0
- mtplx-2.4.2/scripts/deepseek_v4_mtpk_bench.py +1104 -0
- mtplx-2.4.2/scripts/deepseek_v4_op_census.py +277 -0
- mtplx-2.4.2/scripts/deepseek_v4_smoke_generate.py +535 -0
- mtplx-2.4.2/scripts/deepseek_v4_sparse_gate_probe.py +311 -0
- mtplx-2.4.2/scripts/deepseek_v4_validate_moe_tail_k3_bracket.py +763 -0
- mtplx-2.4.2/scripts/gauntlet_scoreboard.py +85 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/scripts/hygiene_scan.sh +7 -0
- mtplx-2.4.2/scripts/oc_tap.py +236 -0
- mtplx-2.4.2/scripts/oc_tap_diff.py +124 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_artifacts.py +53 -1
- mtplx-2.4.2/tests/test_cost_depth_policy.py +108 -0
- mtplx-2.4.2/tests/test_deepseek_v4_decode.py +263 -0
- mtplx-2.4.2/tests/test_deepseek_v4_dtypes.py +465 -0
- mtplx-2.4.2/tests/test_deepseek_v4_indexer.py +623 -0
- mtplx-2.4.2/tests/test_deepseek_v4_kernel_paths.py +495 -0
- mtplx-2.4.2/tests/test_deepseek_v4_loader.py +225 -0
- mtplx-2.4.2/tests/test_deepseek_v4_moe_tail.py +796 -0
- mtplx-2.4.2/tests/test_deepseek_v4_moe_tail_bracket.py +1303 -0
- mtplx-2.4.2/tests/test_deepseek_v4_mtp.py +753 -0
- mtplx-2.4.2/tests/test_deepseek_v4_new_math.py +282 -0
- mtplx-2.4.2/tests/test_deepseek_v4_o_lora.py +510 -0
- mtplx-2.4.2/tests/test_deepseek_v4_parity.py +106 -0
- mtplx-2.4.2/tests/test_deepseek_v4_sinkhorn_kernel.py +474 -0
- mtplx-2.4.2/tests/test_deepseek_v4_spec.py +915 -0
- mtplx-2.4.2/tests/test_deepseek_v4_swiglu_clamp.py +433 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_default_models.py +41 -0
- mtplx-2.4.2/tests/test_gdn_blocked_prefill.py +115 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_generation_sustained.py +34 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_graphbank_compiled_verify.py +92 -25
- mtplx-2.4.2/tests/test_mtp_weightless_degrade.py +118 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_postcommit_wait_integration.py +62 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_public_cli.py +32 -8
- mtplx-2.4.2/tests/test_request_capture.py +76 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_session_bank.py +73 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/LICENSE +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/MANIFEST.in +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/NOTICE +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/__init__.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/a3b_compiled_target_prefix.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/a3b_whole_moe.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/app_settings.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/attention_context.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/attention_split.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/backends/__init__.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/backends/deepseek_mtp.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/backends/descriptors.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/backends/gemma4_assistant.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/backends/glm_mtp.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/backends/hy_v3_mtp.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/backends/mimo_mtp.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/backends/nemotron_h_mtp.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/backends/qwen3_next.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/backends/step3p5_mtp.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/batched_decode.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/batching/__init__.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/batching/admission.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/batching/buckets.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/batching/scheduler.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/batching/state.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/benchmarks/__init__.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/benchmarks/code_eval.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/benchmarks/programming_prompts.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/benchmarks/prompts/aime_2026.jsonl +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/benchmarks/prompts/calibration_coding.jsonl +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/benchmarks/prompts/default.jsonl +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/benchmarks/prompts/flappy.jsonl +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/benchmarks/prompts/long_code.jsonl +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/benchmarks/prompts/long_code_uncapped.jsonl +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/benchmarks/prompts/python_modules_long.jsonl +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/benchmarks/runners/__init__.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/benchmarks/runners/aime.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/benchmarks/runners/batch_equivalence.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/benchmarks/runners/capture_commit_equivalence.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/benchmarks/runners/competitor_baselines.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/benchmarks/runners/contract_probe.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/benchmarks/runners/harness.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/benchmarks/runners/mtp1_gate.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/benchmarks/runners/mtp1_sampler_smoke.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/benchmarks/runners/mtp_adaptive.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/benchmarks/runners/mtp_chain_probe.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/benchmarks/runners/mtp_depth_grid.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/benchmarks/runners/mtp_depth_sweep.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/benchmarks/runners/mtp_tree_probe.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/benchmarks/runners/multi_qmv_probe.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/benchmarks/runners/preflight.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/benchmarks/runners/runtime_smoke.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/benchmarks/runners/session_bank.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/benchmarks/runners/truth.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/benchmarks/runners/verify_profile.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/benchmarks/runners/verify_qmm_probe.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/benchmarks/runners/verify_ratio.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/benchmarks/schema.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/benchmarks/validators/__init__.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/benchmarks/validators/aime.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/benchmarks/validators/basic.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/block_attention.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/cache_bank/__init__.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/cache_bank/codec.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/cache_bank/cold_tier.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/cache_state.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/chat_encoding.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/commands/__init__.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/commands/forge.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/compile_state.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/compiled_forward.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/compressed_tensors.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/config.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/constants.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/constrained.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/context_copy.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/correctors/__init__.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/correctors/diagonal_affine.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/correctors/low_rank.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/daemon_client.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/dashboard/__init__.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/dashboard/_static/assets/index-CRP90ECi.js +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/dashboard/_static/assets/index-DYvLRZ33.css +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/dashboard/_static/index.html +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/deepseek_mtp_patch.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/draft_lm_head.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/draft_sampling.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/env.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/errors.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/expert_layout.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/fan_mode.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/fast_sampling.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/gdn_capture.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/gemma4_pair.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/glm_mtp_patch.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/hardware.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/hf_loader.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/hy_v3_mtp_patch.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/kernel_selfcheck.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/kernels/__init__.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/kernels/a3b_whole_moe.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/kernels/copy_leaf.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/kernels/fused_norm.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/kernels/laguna_decode.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/kernels/lm_head_topk.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/kernels/logits_topk.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/kernels/native_gdn_tail.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/kernels/sdpa_2pass.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/kernels/sdpa_2pass_paged.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/kernels/sdpa_2pass_paged_q8.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/kernels/sdpa_gqa_packed.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/kernels/verify_mlp_fused.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/kpi/__init__.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/kpi/reference_vllm.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/kpi/runtime_kpis.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/kv_quant.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/laguna_compiled_step.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/loop_guard.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/metadata_scrub.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/mimo_mtp_patch.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/model_catalog.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/model_scheduler.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/models/__init__.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/models/laguna.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/models/laguna_config.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/models/laguna_fused.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/moe_packed_projections.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/mtp_activation_stats.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/mtp_adapters.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/mtp_patch.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/native_mlp.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/nax_verify.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/nemotron_h_mtp_patch.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/opencode.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/optimization_profiles.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/pi.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/prefill_bench.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/profiles.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/progress_heartbeat.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/proj_quant.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/proposal_reranker.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/qwen3_5_mtp_patch.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/qwen_row_owned_router.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/ragged_attention.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/ragged_kv_cache.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/reasoning_codecs.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/roofline_profile.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/runtime_options.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/sampling.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/server/__init__.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/server/dashboard_state.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/server/omlx_bridge/__init__.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/server/omlx_bridge/adapter.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/server/omlx_bridge/thinking.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/server/omlx_bridge/tool_calling.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/server_urls.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/speculative.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/step3p5_mtp_patch.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/swival.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/templates/qwen36_froggeric_v19/chat_template.jinja +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/templates/qwen36_froggeric_v21_3/chat_template.jinja +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/thermal.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/thermal_sidecar.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/thinking_guard.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/trace_parity.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/turboquant.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/ui/__init__.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/ui/banner.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/ui/chat_printer.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/ui/download_progress.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/ui/onboarding.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/ui/panels.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/ui/progress.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/verify_kernels.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/verify_qmv.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/vision/__init__.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/vision/processing.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/vision/qwen3_vl_tower.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx/vision/splice.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx.egg-info/dependency_links.txt +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx.egg-info/entry_points.txt +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx.egg-info/requires.txt +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/mtplx.egg-info/top_level.txt +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/scripts/agent_user_path_qa.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/scripts/aime_serve_gate.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/scripts/aime_shape_memory_bench.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/scripts/code_eval_gate.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/scripts/collect_mtp_activation_stats.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/scripts/collect_mtp_hidden_calib.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/scripts/combine_mtp_adapters.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/scripts/compiled_verify_exactness.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/scripts/convert_step37_to_mtplx_step3p5.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/scripts/eval_mtp_corrector.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/scripts/filter_mtp_adapter.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/scripts/fp16_turbo_exactness_20260707.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/scripts/fresh_venv_smoke.sh +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/scripts/install_macos.sh +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/scripts/install_preview_global.sh +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/scripts/kvcache_exactness_audit_20260703.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/scripts/kvcache_soak_20260703.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/scripts/kvcache_warm_probe_20260703.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/scripts/make_fp16_precision_sibling.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/scripts/midform_gate.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/scripts/ocspeed-20260703/abba_final.sh +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/scripts/ocspeed-20260703/accept_depth_probe.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/scripts/ocspeed-20260703/arm_runner.sh +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/scripts/ocspeed-20260703/forward_depth_bisect.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/scripts/ocspeed-20260703/longgen_probe.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/scripts/ocspeed-20260703/loop_degradation_probe.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/scripts/ocspeed-20260703/paired_longgen.sh +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/scripts/ocspeed-20260703/sdpa_microbench.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/scripts/ocspeed-20260703/snapshot_tail.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/scripts/opencode_concurrency_qa.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/scripts/phase0h_paged_verifier_exactness.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/scripts/pillar_gate_qa.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/scripts/probe_mlx_pr3026_qsdpa.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/scripts/probe_mx_compile_buckets.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/scripts/probe_paged_gqa_sdpa_routes.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/scripts/prodqa-20260703/candidate_truecold_20260703.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/scripts/prodqa-20260703/cold_pair_recheck.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/scripts/prodqa-20260703/competitors_20260703.sh +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/scripts/prodqa-20260703/decode_gap_matrix.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/scripts/prodqa-20260703/hermes_pty_driver.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/scripts/prodqa-20260703/omlx_probe.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/scripts/prodqa-20260703/pi_pty_driver.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/scripts/prodqa-20260703/pillar_ab_20260703.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/scripts/prodqa-20260703/tool_gauntlet_20260703.sh +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/scripts/quality_fp16_parent_logitdiff_20260707.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/scripts/r1_chisquare_verifier_correctness.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/scripts/release_macos_v1.sh +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/scripts/run_context_degradation_diagnostics.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/scripts/serve_openai_mtplx.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/scripts/session_cache_followup_qa.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/scripts/session_forensics.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/scripts/sparkle_rehearsal_kit.sh +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/scripts/step_acceptance.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/scripts/step_smoke.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/scripts/train_mtp_adapter_c4.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/scripts/validate_step_mtp_injector.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/scripts/walltime-lab/run_project.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/setup.cfg +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_a3b_compiled_target_prefix.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_a3b_whole_moe.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_adaptive.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_aime_serve_gate.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_ar_batch_penalties.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_attention_split.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_background_warmup.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_batched_decode.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_batching_foundation.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_bracket_tool_rescue.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_cache_bank.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_cache_state.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_cli_parity_tools.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_code_eval.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_code_eval_gate.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_cold_tier_write_budget.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_compile_state.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_compiled_ar_wiring.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_compiled_forward.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_compressed_tensors.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_config.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_config_profile_precedence.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_constrained.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_context_copy_stats.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_context_degradation_profiles.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_correctors.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_daemon_client.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_dashboard_endpoints.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_device_draft_core.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_diagnostics.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_download_progress.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_draft_lm_head.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_engine_session_concurrency.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_engine_session_env.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_env_flag_parsing.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_fast_sampling.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_forge_cli.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_gdn_boundary_retention.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_gdn_postconv_fusion.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_gdn_postconv_impl_gate.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_generation_store_on_prefill.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_hf_loader.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_hy_v3_mtp_backend.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_hy_v3_mtp_graft.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_hygiene_scan.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_idle_postcommit_subagent.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_kernel_selfcheck.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_laguna_compiled_step.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_laguna_fused.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_laguna_model.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_lazy_snapshot_cow.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_loop_guard.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_max_idle_watchdog.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_max_lifecycle.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_memory_pressure_guard.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_metadata_scrub.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_metal_memory_caps.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_midform_gate.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_model_catalog.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_model_scheduler.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_moe_force_unsorted.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_moe_packed_projections.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_mtp_activation_stats.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_mtp_adapters.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_mtp_alias_load_path.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_mtp_depth_sweep.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_mtp_patch.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_mtp_payload_guards.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_nax_verify.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_no_mlx_imports.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_omlx_bridge.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_onboarding.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_openai_bridge.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_opencode.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_optimization_profiles.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_orphan_tool_markup.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_penalties.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_penalty_request_wiring.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_persistent_replay.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_phase0h_paged_verifier_exactness.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_policy_fingerprint_stability.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_postcommit_prefix_reuse.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_postcommit_resolve_for_request.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_postcommit_tools_plumbing.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_postcommit_wait.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_prefill_bench.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_prefill_chunk_defaults.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_prefill_tps_regression.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_profiles.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_proj_quant.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_prompt_encoding.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_qwen3_5_mtp_backend.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_qwen_row_owned_router.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_ragged_kv_cache.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_reasoning_stream_split.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_runtime_kpis.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_sampling.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_scoped_reasoning_history.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_sdpa_gqa_packed.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_server_openai.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_session_bank_env_caps.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_snapshot_free_rejection_repair.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_ssd_boundary_repersist.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_step3p5_mtp_patch.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_stream_guard_env.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_stream_stall_watchdog.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_sustained_long_context_qa.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_thermal.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_thermal_sidecar.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_thinking_guard.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_tool_aware_stream_translator.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_tool_nested_args_streaming.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_trace_parity.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_turboquant.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_turboquant_fallback.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_ui_progress.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_validators.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_vision_session_cache.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_vision_tower.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/tests/test_vllm_reference.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/vllm_metal/__init__.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/vllm_metal/metal/README.md +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/vllm_metal/metal/__init__.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/vllm_metal/metal/build.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/vllm_metal/metal/constants.py +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/vllm_metal/metal/kernels_v1/copy_blocks.metal +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/vllm_metal/metal/kernels_v1/float8.metal +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/vllm_metal/metal/kernels_v1/gather_kv_cache.metal +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/vllm_metal/metal/kernels_v1/kv_scale_update.metal +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/vllm_metal/metal/kernels_v1/pagedattention.metal +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/vllm_metal/metal/kernels_v1/reshape_and_cache.metal +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/vllm_metal/metal/kernels_v1/utils.metal +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/vllm_metal/metal/kernels_v2/copy_blocks.metal +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/vllm_metal/metal/kernels_v2/float8.metal +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/vllm_metal/metal/kernels_v2/gather_kv_cache.metal +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/vllm_metal/metal/kernels_v2/gdn_linear_attention.metal +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/vllm_metal/metal/kernels_v2/kv_scale_update.metal +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/vllm_metal/metal/kernels_v2/pagedattention.metal +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/vllm_metal/metal/kernels_v2/reshape_and_cache.metal +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/vllm_metal/metal/kernels_v2/turboquant.metal +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/vllm_metal/metal/kernels_v2/utils.metal +0 -0
- {mtplx-2.4.0 → mtplx-2.4.2}/vllm_metal/metal/paged_ops.cpp +0 -0
|
@@ -4,6 +4,157 @@ All notable user-facing changes to MTPLX. The format is based on
|
|
|
4
4
|
[Keep a Changelog](https://keepachangelog.com/en/1.1.0/); versions follow
|
|
5
5
|
[Semantic Versioning](https://semver.org/).
|
|
6
6
|
|
|
7
|
+
## [2.4.2] - 2026-08-02
|
|
8
|
+
|
|
9
|
+
The agentic-cache release: the session cache stops losing warm state
|
|
10
|
+
mid-run, tool-turn commits stop being ghosts, every serve keeps a durable
|
|
11
|
+
per-request trail by default, an experimental DeepSeek-V4-Flash backend
|
|
12
|
+
lands, and the documentation now matches the code everywhere it was
|
|
13
|
+
audited.
|
|
14
|
+
|
|
15
|
+
### Added
|
|
16
|
+
|
|
17
|
+
- DeepSeek-V4-Flash: experimental native AR backend
|
|
18
|
+
(`model_type: deepseek_v4`) — Hyper-Connections, compressed sparse
|
|
19
|
+
attention, hash-routed MoE, grouped output-LoRA — loading the
|
|
20
|
+
mlx-community checkpoints directly, with an optional single-block MTP
|
|
21
|
+
speculative lane when the checkpoint carries `mtp.0.*` weights
|
|
22
|
+
(spec == AR gated; K=1-3 measured up to 2.28x on the 2bit-DQ build).
|
|
23
|
+
MTP-declaring checkpoints that ship no draft weights degrade to AR
|
|
24
|
+
with a clear message instead of failing at bind. Thanks @davidtai
|
|
25
|
+
(#216).
|
|
26
|
+
- Request log, default on: every serve writes numeric/hash per-request
|
|
27
|
+
telemetry to `~/.mtplx/logs/request-log-<port>.jsonl` (64 MB x 4
|
|
28
|
+
rotation; no prompt or completion content; disable with
|
|
29
|
+
`MTPLX_REQUEST_LOG_JSONL=off`). Pairs with 2.4.1's opt-in bit-exact
|
|
30
|
+
request capture to make agent-session incidents diagnosable after the
|
|
31
|
+
fact (#196/#197). New helpers: `scripts/gauntlet_scoreboard.py`
|
|
32
|
+
per-session summarizer, `scripts/oc_tap.py` recording proxy and
|
|
33
|
+
`scripts/oc_tap_diff.py` request-mutation analyzer for content-level
|
|
34
|
+
wire truth.
|
|
35
|
+
- Session bank, active-session eviction protection: sessions that
|
|
36
|
+
touched the bank within `MTPLX_SESSION_BANK_ACTIVE_PIN_TTL_S`
|
|
37
|
+
(default 600 s) are eviction-last, so cross-session pressure evicts
|
|
38
|
+
idle victims instead of the session that is mid-run.
|
|
39
|
+
- Session bank, newest-K per-session snapshot retention
|
|
40
|
+
(`MTPLX_SESSION_BANK_PER_SESSION_MAX_ENTRIES`, default 3): divergent
|
|
41
|
+
per-turn sibling snapshots no longer accumulate unreclaimed.
|
|
42
|
+
`/health` now reports active sessions, the pin TTL, and recent
|
|
43
|
+
evictions.
|
|
44
|
+
- Postcommit foreground grace (`MTPLX_POSTCOMMIT_FOREGROUND_GRACE_S`,
|
|
45
|
+
default 2 s): a nearly-finished background cache commit lands instead
|
|
46
|
+
of being preempted by the next fast agent-loop request.
|
|
47
|
+
- Session identity honors `x-session-affinity` / `x-session-id` request
|
|
48
|
+
headers (OpenCode sends these per request), ending cross-request
|
|
49
|
+
identity churn on that client.
|
|
50
|
+
|
|
51
|
+
### Fixed
|
|
52
|
+
|
|
53
|
+
- Tool-turn "ghost re-prefills": the tool-rewrite async commit rendered
|
|
54
|
+
a canonical history that matched neither the generation nor the next
|
|
55
|
+
prompt, burning full-history re-forwards (26.8 s observed) without
|
|
56
|
+
ever storing. It is disabled pending a byte-proven canonical render
|
|
57
|
+
(`MTPLX_IDLE_POSTCOMMIT_TOOL_REWRITE` re-enables);
|
|
58
|
+
store-on-prefill and block salvage cover the lane.
|
|
59
|
+
- The bridge's convergence guard now states explicitly that editing and
|
|
60
|
+
verification tools remain allowed and that its restriction covers
|
|
61
|
+
only the current reply — a model read the old wording as a
|
|
62
|
+
session-wide tool ban and stalled an entire session.
|
|
63
|
+
- `mtplx profile thermal`, `profile eval-attribution`,
|
|
64
|
+
`profile dispatch --trace`, and `thermal fanmax-run` invoked
|
|
65
|
+
research-workspace scripts that are not part of the distribution, and
|
|
66
|
+
`--dry-run` printed those phantom paths as runnable commands. They
|
|
67
|
+
now report availability honestly (exit 2, machine-readable
|
|
68
|
+
`available: false`) and run the real script when present.
|
|
69
|
+
- `mtplx doctor`: Python floor corrected to 3.11 (matching
|
|
70
|
+
`requires-python`); remediation texts no longer tell end users to
|
|
71
|
+
edit source constants or to move a healthy server off its port;
|
|
72
|
+
`--port` is documented and, when passed explicitly, aims the server
|
|
73
|
+
connectivity checks.
|
|
74
|
+
- Session-bank near-prefix restores on backends with bounded rollback
|
|
75
|
+
(DeepSeek-V4) pre-check `max_rollback` and fall back to a cold
|
|
76
|
+
prefill instead of raising (#216).
|
|
77
|
+
- Help surfaces match their own parsers: the onboarding help no longer
|
|
78
|
+
promises a Turbo wizard choice that does not exist (Turbo
|
|
79
|
+
auto-selects for the quantized flagships), `--strict-cold` names the
|
|
80
|
+
enforced 59 tok/s gate, `--open-dashboard` opens alongside the chosen
|
|
81
|
+
client (as it always did), and the command reference teaches
|
|
82
|
+
`mtplx <command> --help`, which also works for multi-word commands.
|
|
83
|
+
|
|
84
|
+
### Documentation
|
|
85
|
+
|
|
86
|
+
- Full truth sweep: ~450 documentation claims reconciled against the
|
|
87
|
+
code across 27 files. Highlights: INSTALL.md no longer references an
|
|
88
|
+
MLX fork removed in 2.0.0; turbo-verify.md no longer calls the
|
|
89
|
+
shipped default "experimental, off by default" nor excludes the
|
|
90
|
+
6-bit lane that ships; the Anthropic base-URL instruction (docs and
|
|
91
|
+
the canonical example) no longer 404s; `/metrics` no longer claims a
|
|
92
|
+
Prometheus mode that never existed; the README modes table shows
|
|
93
|
+
Turbo as the default for the quantized 27B/9B flagships; the Laguna
|
|
94
|
+
memory requirement states the real ~85.3 GiB preflight gate;
|
|
95
|
+
version-era staleness ("v0.1", "preview", v0.3.x runbook pins) is
|
|
96
|
+
cleared; historical release notes gain bracketed corrections where
|
|
97
|
+
they documented commands that never worked. Thanks
|
|
98
|
+
@PhilipJohnBasile for #218 (removed the unsupported MTP-sidecar
|
|
99
|
+
graft guidance; seeded by #215).
|
|
100
|
+
- Dependency-record correction: the transformers pin has been
|
|
101
|
+
`<5.14,!=5.13.0` since shortly after 2.0.0; the changelog never
|
|
102
|
+
recorded the relaxation from `<5.13`.
|
|
103
|
+
|
|
104
|
+
### Dependencies
|
|
105
|
+
|
|
106
|
+
- pypa/gh-action-pypi-publish 1.14.1 -> 1.14.2 (#217).
|
|
107
|
+
|
|
108
|
+
## [2.4.1] - 2026-08-01
|
|
109
|
+
|
|
110
|
+
The smooth-streaming release: the app's chat render path is overhauled
|
|
111
|
+
(no more freeze-then-catch-up stutter, scroll bounce, or plain-text code
|
|
112
|
+
blocks — real syntax coloring, live code cards, tables, and actual math
|
|
113
|
+
notation), and the 2.4.0 short-turn regression is fixed.
|
|
114
|
+
|
|
115
|
+
### Added
|
|
116
|
+
|
|
117
|
+
- Live syntax coloring for code blocks (12 languages + generic) from a
|
|
118
|
+
freeze-time lexer that colors each line exactly once; streaming cost
|
|
119
|
+
is O(new text), never O(document).
|
|
120
|
+
- Streaming code card: an open fence renders as a live card with
|
|
121
|
+
colored lines and flips once to its settled form at close.
|
|
122
|
+
- Pipe tables render as real tables; math renders as real notation
|
|
123
|
+
(Unicode super/subscripts, stacked matrices and fractions, inline
|
|
124
|
+
conversion instead of dollar-sign leaks).
|
|
125
|
+
- Typewriter pacing for streamed text with geometric catch-up and a
|
|
126
|
+
hard drain bound (`MTPLX_STREAM_TYPEWRITER=0` to disable), and a live
|
|
127
|
+
tok/s chip computed over a sliding ~5 s window.
|
|
128
|
+
- Performance mode is a true kill switch: plain text only, through both
|
|
129
|
+
the streaming and settled render paths.
|
|
130
|
+
- Opt-in per-request capture for bit-exact failure replay:
|
|
131
|
+
`MTPLX_REQUEST_CAPTURE_DIR=<dir>` persists each request's
|
|
132
|
+
reproduction envelope at dispatch time (#196/#197, third layer).
|
|
133
|
+
- Opt-in frontend stream-performance probe (`MTPLX_UI_PERF=1`, HUD via
|
|
134
|
+
`MTPLX_UI_PERF_HUD=1`) with a per-turn JSONL trace joinable to engine
|
|
135
|
+
stats by request id.
|
|
136
|
+
- Experimental: cost-model speculative-depth policy
|
|
137
|
+
(`--adaptive-policy cost`) and blocked-sequential GDN prefill
|
|
138
|
+
(`MTPLX_GDN_BLOCKED_PREFILL=1`). Defaults unchanged.
|
|
139
|
+
|
|
140
|
+
### Fixed
|
|
141
|
+
|
|
142
|
+
- 2.4.0 short-turn regression: the compiled-verify path could reserve
|
|
143
|
+
KV budget above the configured ceiling, taxing short requests with
|
|
144
|
+
setup work they never used; the reserve is now clamped.
|
|
145
|
+
- Warming prefills yield to real traffic within one small chunk instead
|
|
146
|
+
of delaying a freshly arrived request.
|
|
147
|
+
- Derivative model artifacts whose names extend a first-party model
|
|
148
|
+
name are served under their own id, not the flagship's — the health
|
|
149
|
+
payload, OpenAI `model` field, and app model chip now report the
|
|
150
|
+
artifact actually loaded.
|
|
151
|
+
- Streaming render: line-segment coalescing keeps realized view count
|
|
152
|
+
bounded on long answers; the bottom-pin scroll correction runs in the
|
|
153
|
+
same display cycle as layout so the streaming bubble can no longer
|
|
154
|
+
visibly bounce; a per-display-cycle window-sizing walk that floored
|
|
155
|
+
every update at ~50 ms is removed (`MTPLX_APP_SIZING_TUNER=0`
|
|
156
|
+
restores it).
|
|
157
|
+
|
|
7
158
|
## [2.4.0] - 2026-07-31
|
|
8
159
|
|
|
9
160
|
The 35B speed release: the 35B-A3B MoE gets a compiled decode stack and
|
|
@@ -676,4 +827,17 @@ working as one product. Full notes:
|
|
|
676
827
|
completions, and Anthropic `stop_sequences`) and `/v1/completions`
|
|
677
828
|
streams tokens as they are generated with real finish reasons.
|
|
678
829
|
|
|
679
|
-
[
|
|
830
|
+
[2.4.2]: https://github.com/youssofal/MTPLX/releases/tag/v2.4.2
|
|
831
|
+
[2.4.1]: https://github.com/youssofal/MTPLX/releases/tag/v2.4.1
|
|
832
|
+
[2.4.0]: https://github.com/youssofal/MTPLX/releases/tag/v2.4.0
|
|
833
|
+
[2.3.0]: https://github.com/youssofal/MTPLX/releases/tag/v2.3.0
|
|
834
|
+
[2.2.0]: https://github.com/youssofal/MTPLX/releases/tag/v2.2.0
|
|
835
|
+
[2.1.0]: https://github.com/youssofal/MTPLX/releases/tag/v2.1.0
|
|
836
|
+
[2.0.2]: https://github.com/youssofal/MTPLX/releases/tag/v2.0.2
|
|
837
|
+
[2.0.1]: https://github.com/youssofal/MTPLX/releases/tag/v2.0.1
|
|
838
|
+
[2.0.0]: https://github.com/youssofal/MTPLX/releases/tag/v2.0.0
|
|
839
|
+
[1.0.4]: https://github.com/youssofal/MTPLX/releases/tag/v1.0.4
|
|
840
|
+
[1.0.3]: https://github.com/youssofal/MTPLX/releases/tag/v1.0.3
|
|
841
|
+
[1.0.2]: https://github.com/youssofal/MTPLX/releases/tag/v1.0.2
|
|
842
|
+
[1.0.1]: https://github.com/youssofal/MTPLX/releases/tag/v1.0.1
|
|
843
|
+
[1.0.0]: https://github.com/youssofal/MTPLX/releases/tag/v1.0.0
|
|
@@ -8,7 +8,7 @@ authors:
|
|
|
8
8
|
repository-code: "https://github.com/youssofal/mtplx"
|
|
9
9
|
url: "https://github.com/youssofal/mtplx"
|
|
10
10
|
license: Apache-2.0
|
|
11
|
-
version:
|
|
11
|
+
version: 2.4.2
|
|
12
12
|
abstract: "Native MTP speculative decoding for Qwen3-Next on Apple Silicon, using built-in MTP heads with math-correct rejection sampling and an OpenAI/Anthropic-compatible serving surface."
|
|
13
13
|
keywords:
|
|
14
14
|
- speculative decoding
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: mtplx
|
|
3
|
-
Version: 2.4.
|
|
3
|
+
Version: 2.4.2
|
|
4
4
|
Summary: Native MTP speculative decoding for Qwen3-Next on Apple Silicon.
|
|
5
5
|
Author-email: Youssof Altoukhi <business@youssofal.com>
|
|
6
6
|
License-Expression: Apache-2.0
|
|
@@ -105,9 +105,11 @@ On a 16 GB M4 Mac mini, tuning the 9B model lands on depth 1: 14.4 tok/s baselin
|
|
|
105
105
|
|
|
106
106
|
<img src="docs/assets/readme/app-forge.jpg" alt="Forge verifying a freshly built MTP model" width="100%" />
|
|
107
107
|
|
|
108
|
-
Forge takes a Hugging Face repo and turns it into an MTPLX-ready MTP model: convert to MLX, train the MTP adapter, verify that the result is actually faster and still exact, and publish back to the Hub if you want to share it. The honest part matters: Forge measures before and after on your hardware and shows you the verdict ("Depth 1 is fastest: 227.1 to 296.1, 1.30x") rather than assuming the adapter helped. Available in the app and as `mtplx forge
|
|
108
|
+
Forge takes a Hugging Face repo and turns it into an MTPLX-ready MTP model: convert to MLX, train the MTP adapter, verify that the result is actually faster and still exact, and publish back to the Hub if you want to share it. The honest part matters: Forge measures before and after on your hardware and shows you the verdict ("Depth 1 is fastest: 227.1 to 296.1, 1.30x") rather than assuming the adapter helped. Available in the app and as `mtplx forge` subcommands.
|
|
109
109
|
|
|
110
|
-
|
|
110
|
+
MTPLX does not support attaching a separately supplied MTP sidecar to an arbitrary MLX trunk. Matching architecture fields, tensor shapes, or provenance labels cannot prove that the head was trained against those exact trunk weights. Use a complete model that already includes its matching MTP weights, or use Forge to build and verify an artifact from its original source checkpoint.
|
|
111
|
+
|
|
112
|
+
The official catalog lives on Hugging Face under [Youssofal](https://huggingface.co/Youssofal): Qwen 3.5 (4B, 9B), Qwen 3.6 (27B, 35B MoE) in speed and quality builds (the 35B MoE adds a balance build), plus Gemma 4. The app recommends from these based on your hardware.
|
|
111
113
|
|
|
112
114
|
## The server
|
|
113
115
|
|
|
@@ -119,7 +121,7 @@ curl http://127.0.0.1:8000/v1/chat/completions \
|
|
|
119
121
|
-d '{"model":"mtplx","messages":[{"role":"user","content":"hi"}],"stream":true}'
|
|
120
122
|
```
|
|
121
123
|
|
|
122
|
-
Sessions survive: a warm-prefix session bank keeps multi-turn chats fast, and
|
|
124
|
+
Sessions survive: a warm-prefix session bank keeps multi-turn chats fast, and a default-on SSD session cache restores sessions near-instantly across restarts (disable with `--ssd-session-cache off`).
|
|
123
125
|
|
|
124
126
|
Sampler controls cover `temperature`, `top_p`, `top_k`, and the OpenAI penalty pair `presence_penalty` / `frequency_penalty` — per request, as server defaults (`--default-presence-penalty` / `--default-frequency-penalty` on `start`/`serve`/`quickstart`), or live via `mtplx settings set` and the app's Presence Penalty dial. Penalties default to 0, which is an exact no-op that preserves MTP exactness. Qwen's guidance: leave them at 0 for coding and agent work; ~0.5–1.5 presence penalty helps creative writing or when a model loops on itself.
|
|
125
127
|
|
|
@@ -133,20 +135,21 @@ mtplx pull <hf-repo> # download a model safely
|
|
|
133
135
|
mtplx models # what is cached, sizes, validation
|
|
134
136
|
mtplx inspect <model> # compatibility report before anything runs
|
|
135
137
|
mtplx tune --retune # measure AR vs D1/D2/D3 on your Mac
|
|
136
|
-
mtplx forge
|
|
138
|
+
mtplx forge --help # build, verify, and publish MTP models (probe/build/publish/verify subcommands)
|
|
137
139
|
mtplx bench aime --quick # run the AIME benchmark from the terminal
|
|
138
140
|
mtplx doctor # install and integration health
|
|
139
141
|
mtplx max --install # fan control (one sudo prompt, crash-safe)
|
|
140
142
|
mtplx settings get/set # read or change live server settings
|
|
141
143
|
```
|
|
142
144
|
|
|
143
|
-
Every command takes `--
|
|
145
|
+
Every command takes `--help`, and most inspection/diagnostic commands take `--json`. The CLI works without MLX installed for everything that does not need a model, so `doctor` and `inspect` run on any machine.
|
|
144
146
|
|
|
145
147
|
## Modes
|
|
146
148
|
|
|
147
149
|
| Mode | What it does | When |
|
|
148
150
|
|---|---|---|
|
|
149
|
-
| **
|
|
151
|
+
| **Turbo** | NAX verify kernels + compiled verify; the default for the quantized 27B and 9B flagship models | Picked automatically for those models |
|
|
152
|
+
| **Sustained** | Default for all other models. Long-context MTP path with chunked prefill and request-sized KV | Everyday use, big files, 16K-200K prompts |
|
|
150
153
|
| **Sustained Max** | Sustained with fans pinned at 100% | Long work where you want maximum cooling |
|
|
151
154
|
| **Burst** | Legacy short-context benchmark lane, loud | Short prompts and benchmarks only |
|
|
152
155
|
|
|
@@ -154,7 +157,7 @@ Fan-backed modes restore your fans to automatic if MTPLX dies for any reason, in
|
|
|
154
157
|
|
|
155
158
|
## Compatibility, honestly
|
|
156
159
|
|
|
157
|
-
`mtplx inspect` classifies models before anything runs: verified, architecture-compatible but unverified, AR-only, incompatible architecture, or no MTP heads at all. Unverified models
|
|
160
|
+
`mtplx inspect` classifies models before anything runs: verified, family-compatible but unverified, architecture-compatible but unverified, AR-only, incompatible architecture, or no MTP heads at all. Unverified models load with an explicit unverified label. There are no silent fallbacks: if MTPLX cannot run a model correctly, it tells you instead of running it badly.
|
|
158
161
|
|
|
159
162
|
[Laguna-S-2.1 oQ4e](https://huggingface.co/mlx-community/Laguna-S-2.1-oQ4e) is supported through its exact MLX architecture in target-only AR mode:
|
|
160
163
|
|
|
@@ -170,8 +173,10 @@ MTPLX pins that model to revision
|
|
|
170
173
|
tokenizer, generation config, special tokens map, and Poolside chat template
|
|
171
174
|
before admitting it. The checkpoint has no native MTP head, so an MTP launch is
|
|
172
175
|
rejected before weights load instead of falling back during execution. The
|
|
173
|
-
weights occupy 59.72 GiB
|
|
174
|
-
|
|
176
|
+
weights occupy 59.72 GiB, a 64.13 GB snapshot on disk. The launch preflight
|
|
177
|
+
requires about 85 GiB of unified memory (weights, runtime headroom, and a
|
|
178
|
+
16 GiB system reserve) — in practice a 96 GB Mac; 128 GB is
|
|
179
|
+
comfortable. MTPLX defaults Laguna to a 32,768-token context
|
|
175
180
|
and response cap, and checks larger explicit server contexts against the active
|
|
176
181
|
Metal memory cap.
|
|
177
182
|
|
|
@@ -55,9 +55,11 @@ On a 16 GB M4 Mac mini, tuning the 9B model lands on depth 1: 14.4 tok/s baselin
|
|
|
55
55
|
|
|
56
56
|
<img src="docs/assets/readme/app-forge.jpg" alt="Forge verifying a freshly built MTP model" width="100%" />
|
|
57
57
|
|
|
58
|
-
Forge takes a Hugging Face repo and turns it into an MTPLX-ready MTP model: convert to MLX, train the MTP adapter, verify that the result is actually faster and still exact, and publish back to the Hub if you want to share it. The honest part matters: Forge measures before and after on your hardware and shows you the verdict ("Depth 1 is fastest: 227.1 to 296.1, 1.30x") rather than assuming the adapter helped. Available in the app and as `mtplx forge
|
|
58
|
+
Forge takes a Hugging Face repo and turns it into an MTPLX-ready MTP model: convert to MLX, train the MTP adapter, verify that the result is actually faster and still exact, and publish back to the Hub if you want to share it. The honest part matters: Forge measures before and after on your hardware and shows you the verdict ("Depth 1 is fastest: 227.1 to 296.1, 1.30x") rather than assuming the adapter helped. Available in the app and as `mtplx forge` subcommands.
|
|
59
59
|
|
|
60
|
-
|
|
60
|
+
MTPLX does not support attaching a separately supplied MTP sidecar to an arbitrary MLX trunk. Matching architecture fields, tensor shapes, or provenance labels cannot prove that the head was trained against those exact trunk weights. Use a complete model that already includes its matching MTP weights, or use Forge to build and verify an artifact from its original source checkpoint.
|
|
61
|
+
|
|
62
|
+
The official catalog lives on Hugging Face under [Youssofal](https://huggingface.co/Youssofal): Qwen 3.5 (4B, 9B), Qwen 3.6 (27B, 35B MoE) in speed and quality builds (the 35B MoE adds a balance build), plus Gemma 4. The app recommends from these based on your hardware.
|
|
61
63
|
|
|
62
64
|
## The server
|
|
63
65
|
|
|
@@ -69,7 +71,7 @@ curl http://127.0.0.1:8000/v1/chat/completions \
|
|
|
69
71
|
-d '{"model":"mtplx","messages":[{"role":"user","content":"hi"}],"stream":true}'
|
|
70
72
|
```
|
|
71
73
|
|
|
72
|
-
Sessions survive: a warm-prefix session bank keeps multi-turn chats fast, and
|
|
74
|
+
Sessions survive: a warm-prefix session bank keeps multi-turn chats fast, and a default-on SSD session cache restores sessions near-instantly across restarts (disable with `--ssd-session-cache off`).
|
|
73
75
|
|
|
74
76
|
Sampler controls cover `temperature`, `top_p`, `top_k`, and the OpenAI penalty pair `presence_penalty` / `frequency_penalty` — per request, as server defaults (`--default-presence-penalty` / `--default-frequency-penalty` on `start`/`serve`/`quickstart`), or live via `mtplx settings set` and the app's Presence Penalty dial. Penalties default to 0, which is an exact no-op that preserves MTP exactness. Qwen's guidance: leave them at 0 for coding and agent work; ~0.5–1.5 presence penalty helps creative writing or when a model loops on itself.
|
|
75
77
|
|
|
@@ -83,20 +85,21 @@ mtplx pull <hf-repo> # download a model safely
|
|
|
83
85
|
mtplx models # what is cached, sizes, validation
|
|
84
86
|
mtplx inspect <model> # compatibility report before anything runs
|
|
85
87
|
mtplx tune --retune # measure AR vs D1/D2/D3 on your Mac
|
|
86
|
-
mtplx forge
|
|
88
|
+
mtplx forge --help # build, verify, and publish MTP models (probe/build/publish/verify subcommands)
|
|
87
89
|
mtplx bench aime --quick # run the AIME benchmark from the terminal
|
|
88
90
|
mtplx doctor # install and integration health
|
|
89
91
|
mtplx max --install # fan control (one sudo prompt, crash-safe)
|
|
90
92
|
mtplx settings get/set # read or change live server settings
|
|
91
93
|
```
|
|
92
94
|
|
|
93
|
-
Every command takes `--
|
|
95
|
+
Every command takes `--help`, and most inspection/diagnostic commands take `--json`. The CLI works without MLX installed for everything that does not need a model, so `doctor` and `inspect` run on any machine.
|
|
94
96
|
|
|
95
97
|
## Modes
|
|
96
98
|
|
|
97
99
|
| Mode | What it does | When |
|
|
98
100
|
|---|---|---|
|
|
99
|
-
| **
|
|
101
|
+
| **Turbo** | NAX verify kernels + compiled verify; the default for the quantized 27B and 9B flagship models | Picked automatically for those models |
|
|
102
|
+
| **Sustained** | Default for all other models. Long-context MTP path with chunked prefill and request-sized KV | Everyday use, big files, 16K-200K prompts |
|
|
100
103
|
| **Sustained Max** | Sustained with fans pinned at 100% | Long work where you want maximum cooling |
|
|
101
104
|
| **Burst** | Legacy short-context benchmark lane, loud | Short prompts and benchmarks only |
|
|
102
105
|
|
|
@@ -104,7 +107,7 @@ Fan-backed modes restore your fans to automatic if MTPLX dies for any reason, in
|
|
|
104
107
|
|
|
105
108
|
## Compatibility, honestly
|
|
106
109
|
|
|
107
|
-
`mtplx inspect` classifies models before anything runs: verified, architecture-compatible but unverified, AR-only, incompatible architecture, or no MTP heads at all. Unverified models
|
|
110
|
+
`mtplx inspect` classifies models before anything runs: verified, family-compatible but unverified, architecture-compatible but unverified, AR-only, incompatible architecture, or no MTP heads at all. Unverified models load with an explicit unverified label. There are no silent fallbacks: if MTPLX cannot run a model correctly, it tells you instead of running it badly.
|
|
108
111
|
|
|
109
112
|
[Laguna-S-2.1 oQ4e](https://huggingface.co/mlx-community/Laguna-S-2.1-oQ4e) is supported through its exact MLX architecture in target-only AR mode:
|
|
110
113
|
|
|
@@ -120,8 +123,10 @@ MTPLX pins that model to revision
|
|
|
120
123
|
tokenizer, generation config, special tokens map, and Poolside chat template
|
|
121
124
|
before admitting it. The checkpoint has no native MTP head, so an MTP launch is
|
|
122
125
|
rejected before weights load instead of falling back during execution. The
|
|
123
|
-
weights occupy 59.72 GiB
|
|
124
|
-
|
|
126
|
+
weights occupy 59.72 GiB, a 64.13 GB snapshot on disk. The launch preflight
|
|
127
|
+
requires about 85 GiB of unified memory (weights, runtime headroom, and a
|
|
128
|
+
16 GiB system reserve) — in practice a 96 GB Mac; 128 GB is
|
|
129
|
+
comfortable. MTPLX defaults Laguna to a 32,768-token context
|
|
125
130
|
and response cap, and checks larger explicit server contexts against the active
|
|
126
131
|
Metal memory cap.
|
|
127
132
|
|
|
@@ -250,3 +250,254 @@ class ExpectedValueDepthPolicy:
|
|
|
250
250
|
prob_term = 2.0 * _clamp(float(top1_prob), 0.0, 1.0) - 1.0
|
|
251
251
|
raw = 1.0 + self.confidence_weight * (0.75 * margin_term + 0.25 * prob_term)
|
|
252
252
|
return _clamp(raw, 0.25, 1.75)
|
|
253
|
+
|
|
254
|
+
|
|
255
|
+
class CostModelDepthPolicy:
|
|
256
|
+
"""Cost-model adaptive depth: maximize expected committed tokens per
|
|
257
|
+
wall-clock cycle, not acceptance streaks.
|
|
258
|
+
|
|
259
|
+
Ported from omlx's ``_DepthController`` (jundot/omlx v0.5.4rc1,
|
|
260
|
+
omlx/patches/mlx_lm_mtp/batch_generator.py) with the depth-0
|
|
261
|
+
park/exit machinery deliberately left out for now (our 27B lanes
|
|
262
|
+
always profit from speculation; the escape hatch matters for
|
|
263
|
+
head_dim-512/MoE models and can come later). Their measured design
|
|
264
|
+
decisions preserved verbatim:
|
|
265
|
+
|
|
266
|
+
- ``score(d) = (1 + p1 + p1 p2 + ...) / t_est(d)``.
|
|
267
|
+
- Acceptance is a token-domain EMA (a property of model/content);
|
|
268
|
+
cost is a wall-clock-horizon EMA (tracks context growth, thermal
|
|
269
|
+
state, and external GPU load at constant real-time responsiveness)
|
|
270
|
+
with a one-off-spike damp.
|
|
271
|
+
- The marginal cost of an extra verify row is the measured slope
|
|
272
|
+
between the cheapest and priciest measured depths, not a constant.
|
|
273
|
+
- Probes are bidirectional and staleness-directed, duty-bounded to
|
|
274
|
+
~15% of cycles: re-measuring a SHALLOWER rival is what breaks the
|
|
275
|
+
depth-2 lock omlx measured (stale-high t[1] hides depth 1 forever).
|
|
276
|
+
- The cost EMA is per-cycle, NOT staleness-age-weighted: omlx
|
|
277
|
+
measured age-weighting worse (probe-burst noise injected straight
|
|
278
|
+
into the decision; prose re-over-drafted 1.6%).
|
|
279
|
+
|
|
280
|
+
Drop-in for the ``AdaptiveDepthPolicy`` interface: ``current_depth``
|
|
281
|
+
plus ``observe(attempted_depth=, accepted_depths=)``. Cycle cost is
|
|
282
|
+
self-timed as the wall interval between observe calls (one observe
|
|
283
|
+
per verify cycle), which deliberately includes the loop's host
|
|
284
|
+
bookkeeping — that tax is part of the real cost of running a cycle.
|
|
285
|
+
"""
|
|
286
|
+
|
|
287
|
+
ALPHA = 0.08
|
|
288
|
+
TAU_MS = 400.0
|
|
289
|
+
PROBE_PERIOD_MS = 1000.0
|
|
290
|
+
PROBE_PERIOD_MAX_MS = 5000.0
|
|
291
|
+
PROBE_LEN = 4
|
|
292
|
+
PROBE_DUTY = 0.15
|
|
293
|
+
PROBE_MARGIN = 1.15
|
|
294
|
+
SPIKE_RATIO = 2.0
|
|
295
|
+
SPIKE_DAMP = 0.25
|
|
296
|
+
MARGINAL_MS = 7.0
|
|
297
|
+
HYSTERESIS = 1.03
|
|
298
|
+
# Ignore absurd inter-observe gaps (queue waits, tool round-trips in
|
|
299
|
+
# agent serving): a "cycle" above this is not a cycle measurement.
|
|
300
|
+
MAX_CYCLE_MS = 5000.0
|
|
301
|
+
# The generation loop passes its own measured cycle wall-time when it
|
|
302
|
+
# sees this flag; the self-timed inter-observe fallback stays for
|
|
303
|
+
# callers that do not.
|
|
304
|
+
accepts_cycle_ms = True
|
|
305
|
+
|
|
306
|
+
def __init__(
|
|
307
|
+
self,
|
|
308
|
+
max_depth: int,
|
|
309
|
+
min_depth: int = 1,
|
|
310
|
+
marginal_ms: float | None = None,
|
|
311
|
+
) -> None:
|
|
312
|
+
if max_depth < 1:
|
|
313
|
+
raise ValueError("max_depth must be >= 1")
|
|
314
|
+
self.max_depth = int(max_depth)
|
|
315
|
+
self.min_depth = max(1, min(int(min_depth), self.max_depth))
|
|
316
|
+
if marginal_ms:
|
|
317
|
+
self.MARGINAL_MS = float(marginal_ms)
|
|
318
|
+
self.current_depth = self.max_depth
|
|
319
|
+
self.p = [0.6] * self.max_depth
|
|
320
|
+
self.t: dict[int, float] = {}
|
|
321
|
+
self.t_age: dict[int, float] = {}
|
|
322
|
+
self.cycles = 0
|
|
323
|
+
self.probe_left = 0
|
|
324
|
+
self._ms_probe = 0.0
|
|
325
|
+
self._ms_explore = 0.0
|
|
326
|
+
self._warmup = list(range(self.max_depth, self.min_depth - 1, -1))
|
|
327
|
+
self._last_observe_s: float | None = None
|
|
328
|
+
|
|
329
|
+
# -- cost bookkeeping --------------------------------------------------
|
|
330
|
+
|
|
331
|
+
def _time_alpha(self, cycle_ms: float) -> float:
|
|
332
|
+
return 1.0 - math.exp(-max(0.0, float(cycle_ms)) / self.TAU_MS)
|
|
333
|
+
|
|
334
|
+
def _update_time(self, used: int, cycle_ms: float) -> None:
|
|
335
|
+
prev = self.t.get(used)
|
|
336
|
+
if prev is None:
|
|
337
|
+
self.t[used] = cycle_ms
|
|
338
|
+
return
|
|
339
|
+
if self._warmup:
|
|
340
|
+
self.t[used] = min(prev, cycle_ms)
|
|
341
|
+
return
|
|
342
|
+
a = self._time_alpha(cycle_ms)
|
|
343
|
+
if cycle_ms > self.SPIKE_RATIO * prev:
|
|
344
|
+
a *= self.SPIKE_DAMP
|
|
345
|
+
self.t[used] = (1.0 - a) * prev + a * cycle_ms
|
|
346
|
+
|
|
347
|
+
def _marginal_est(self) -> float:
|
|
348
|
+
if len(self.t) >= 2:
|
|
349
|
+
depths = sorted(self.t)
|
|
350
|
+
lo, hi = depths[0], depths[-1]
|
|
351
|
+
if hi > lo:
|
|
352
|
+
slope = (self.t[hi] - self.t[lo]) / (hi - lo)
|
|
353
|
+
if slope > 0.0:
|
|
354
|
+
return slope
|
|
355
|
+
return self.MARGINAL_MS
|
|
356
|
+
|
|
357
|
+
def _t_est(self, d: int) -> float:
|
|
358
|
+
if d in self.t:
|
|
359
|
+
return self.t[d]
|
|
360
|
+
if not self.t:
|
|
361
|
+
return 30.0 + self.MARGINAL_MS * d
|
|
362
|
+
ref = min(self.t, key=lambda x: abs(x - d))
|
|
363
|
+
return max(1e-3, self.t[ref] + self._marginal_est() * (d - ref))
|
|
364
|
+
|
|
365
|
+
def _score(self, d: int) -> float:
|
|
366
|
+
expected = 1.0
|
|
367
|
+
run = 1.0
|
|
368
|
+
for j in range(d):
|
|
369
|
+
run *= self.p[j]
|
|
370
|
+
expected += run
|
|
371
|
+
return expected / max(1e-6, self._t_est(d))
|
|
372
|
+
|
|
373
|
+
# -- selection ---------------------------------------------------------
|
|
374
|
+
|
|
375
|
+
def _depths(self) -> list[int]:
|
|
376
|
+
return list(range(self.min_depth, self.max_depth + 1))
|
|
377
|
+
|
|
378
|
+
def _best(self) -> int:
|
|
379
|
+
cur_score = self._score(self.current_depth)
|
|
380
|
+
best_d, best_score = self.current_depth, cur_score
|
|
381
|
+
for d in self._depths():
|
|
382
|
+
s = self._score(d)
|
|
383
|
+
if s > best_score:
|
|
384
|
+
best_d, best_score = d, s
|
|
385
|
+
if best_d != self.current_depth and best_score < cur_score * self.HYSTERESIS:
|
|
386
|
+
return self.current_depth
|
|
387
|
+
return best_d
|
|
388
|
+
|
|
389
|
+
def _best_rival(self) -> int | None:
|
|
390
|
+
best = self._score(self.current_depth)
|
|
391
|
+
if best <= 0.0:
|
|
392
|
+
return self._most_stale()
|
|
393
|
+
rival, rival_score = None, 0.0
|
|
394
|
+
for d in self._depths():
|
|
395
|
+
if d == self.current_depth:
|
|
396
|
+
continue
|
|
397
|
+
s = self._score(d)
|
|
398
|
+
if s > rival_score:
|
|
399
|
+
rival, rival_score = d, s
|
|
400
|
+
if rival is not None and rival_score * self.PROBE_MARGIN >= best:
|
|
401
|
+
return rival
|
|
402
|
+
return None
|
|
403
|
+
|
|
404
|
+
def _most_stale(self) -> int | None:
|
|
405
|
+
candidates = [d for d in self._depths() if d != self.current_depth]
|
|
406
|
+
if not candidates:
|
|
407
|
+
return None
|
|
408
|
+
never = [d for d in candidates if d not in self.t]
|
|
409
|
+
if never:
|
|
410
|
+
return never[0]
|
|
411
|
+
return max(candidates, key=lambda d: self.t_age.get(d, 0.0))
|
|
412
|
+
|
|
413
|
+
# -- the drop-in interface ---------------------------------------------
|
|
414
|
+
|
|
415
|
+
def observe(
|
|
416
|
+
self,
|
|
417
|
+
*,
|
|
418
|
+
attempted_depth: int,
|
|
419
|
+
accepted_depths: int,
|
|
420
|
+
cycle_ms: float | None = None,
|
|
421
|
+
) -> dict:
|
|
422
|
+
import time as _time
|
|
423
|
+
|
|
424
|
+
now = _time.perf_counter()
|
|
425
|
+
if cycle_ms is None and self._last_observe_s is not None:
|
|
426
|
+
cycle_ms = (now - self._last_observe_s) * 1000.0
|
|
427
|
+
if cycle_ms is not None and (cycle_ms > self.MAX_CYCLE_MS or cycle_ms <= 0.0):
|
|
428
|
+
cycle_ms = None
|
|
429
|
+
self._last_observe_s = now
|
|
430
|
+
|
|
431
|
+
self.cycles += 1
|
|
432
|
+
used = max(1, min(int(attempted_depth), self.max_depth))
|
|
433
|
+
accepted = max(0, min(int(accepted_depths), used))
|
|
434
|
+
previous_depth = self.current_depth
|
|
435
|
+
|
|
436
|
+
a = self.ALPHA
|
|
437
|
+
for j in range(used):
|
|
438
|
+
hit = 1.0 if j < accepted else 0.0
|
|
439
|
+
self.p[j] = (1.0 - a) * self.p[j] + a * hit
|
|
440
|
+
if j >= accepted:
|
|
441
|
+
break
|
|
442
|
+
|
|
443
|
+
if cycle_ms is not None:
|
|
444
|
+
self._update_time(used, cycle_ms)
|
|
445
|
+
for d in list(self.t_age):
|
|
446
|
+
self.t_age[d] += cycle_ms
|
|
447
|
+
self.t_age[used] = 0.0
|
|
448
|
+
self._ms_probe += cycle_ms
|
|
449
|
+
self._ms_explore += cycle_ms
|
|
450
|
+
|
|
451
|
+
action = "hold"
|
|
452
|
+
if self._warmup:
|
|
453
|
+
# A warmup slot is consumed only once its depth has a real cost
|
|
454
|
+
# sample; the very first observe has no prior timestamp to diff
|
|
455
|
+
# against, so that cycle repeats its depth instead of advancing.
|
|
456
|
+
if cycle_ms is not None and used == self._warmup[0]:
|
|
457
|
+
self._warmup.pop(0)
|
|
458
|
+
if self._warmup:
|
|
459
|
+
self.current_depth = self._warmup[0]
|
|
460
|
+
action = "warmup"
|
|
461
|
+
else:
|
|
462
|
+
self.current_depth = self._best()
|
|
463
|
+
self._ms_probe = 0.0
|
|
464
|
+
action = "warmup_done"
|
|
465
|
+
elif self.probe_left > 0:
|
|
466
|
+
self.probe_left -= 1
|
|
467
|
+
if self.probe_left == 0:
|
|
468
|
+
self.current_depth = self._best()
|
|
469
|
+
self._ms_probe = 0.0
|
|
470
|
+
action = "probe_done"
|
|
471
|
+
else:
|
|
472
|
+
action = "probing"
|
|
473
|
+
else:
|
|
474
|
+
self.current_depth = self._best()
|
|
475
|
+
if self.current_depth != previous_depth:
|
|
476
|
+
action = (
|
|
477
|
+
"increase" if self.current_depth > previous_depth else "decrease"
|
|
478
|
+
)
|
|
479
|
+
if self.max_depth > self.min_depth and cycle_ms is not None:
|
|
480
|
+
period = max(
|
|
481
|
+
self.PROBE_PERIOD_MS,
|
|
482
|
+
self.PROBE_LEN * cycle_ms / self.PROBE_DUTY,
|
|
483
|
+
)
|
|
484
|
+
if self._ms_probe >= period:
|
|
485
|
+
explore_due = self._ms_explore >= max(
|
|
486
|
+
self.PROBE_PERIOD_MAX_MS, 2.0 * period
|
|
487
|
+
)
|
|
488
|
+
target = self._most_stale() if explore_due else self._best_rival()
|
|
489
|
+
if target is not None:
|
|
490
|
+
self.current_depth = target
|
|
491
|
+
self.probe_left = self.PROBE_LEN
|
|
492
|
+
self._ms_probe = 0.0
|
|
493
|
+
if explore_due:
|
|
494
|
+
self._ms_explore = 0.0
|
|
495
|
+
action = "probe"
|
|
496
|
+
|
|
497
|
+
return {
|
|
498
|
+
"previous_depth": previous_depth,
|
|
499
|
+
"attempted_depth": used,
|
|
500
|
+
"accepted_depths": accepted,
|
|
501
|
+
"next_depth": self.current_depth,
|
|
502
|
+
"action": action,
|
|
503
|
+
}
|
|
@@ -313,6 +313,62 @@ def expected_mtp_file(model_dir: Path | str, config: dict[str, Any] | None = Non
|
|
|
313
313
|
return model_path / "mtp.safetensors"
|
|
314
314
|
|
|
315
315
|
|
|
316
|
+
def mtp_weights_present_on_disk(
|
|
317
|
+
model_dir: Path | str, config: dict[str, Any] | None = None
|
|
318
|
+
) -> bool:
|
|
319
|
+
"""Whether a model that declares MTP layers actually ships MTP weights.
|
|
320
|
+
|
|
321
|
+
A conversion can declare ``num_nextn_predict_layers`` in the config while
|
|
322
|
+
dropping the MTP weights themselves (e.g. the DeepSeek-V4-Flash 2bit-DQ
|
|
323
|
+
build). The runtime uses this probe to tell that benign case (config field
|
|
324
|
+
only -> degrade to autoregressive) apart from a genuine injection failure
|
|
325
|
+
(weights present but unusable -> raise).
|
|
326
|
+
|
|
327
|
+
Conservative by design: it only returns ``False`` when it can *positively*
|
|
328
|
+
confirm absence via a shard index that carries no MTP-shaped keys under any
|
|
329
|
+
known naming convention. A sidecar file, a missing/unreadable index, or any
|
|
330
|
+
ambiguity returns ``True`` so the existing injection + validation path runs
|
|
331
|
+
unchanged and a real detection bug on an MTP-bearing model still surfaces.
|
|
332
|
+
"""
|
|
333
|
+
model_path = Path(model_dir)
|
|
334
|
+
config = config if config is not None else load_config(model_path)
|
|
335
|
+
|
|
336
|
+
# 1. Explicit MTP sidecar file (Qwen/GLM/hy3 external draft head).
|
|
337
|
+
if expected_mtp_file(model_path, config).exists():
|
|
338
|
+
return True
|
|
339
|
+
|
|
340
|
+
index_path = model_path / "model.safetensors.index.json"
|
|
341
|
+
if not index_path.exists():
|
|
342
|
+
# No index to inspect: cannot prove absence, preserve legacy behavior.
|
|
343
|
+
return True
|
|
344
|
+
try:
|
|
345
|
+
weight_map = json.loads(index_path.read_text(encoding="utf-8")).get(
|
|
346
|
+
"weight_map", {}
|
|
347
|
+
)
|
|
348
|
+
except Exception:
|
|
349
|
+
return True
|
|
350
|
+
keys = [str(k) for k in weight_map]
|
|
351
|
+
|
|
352
|
+
# 2. Namespaced embedded MTP weights ("mtp.*" / "language_model.mtp.*").
|
|
353
|
+
if any(is_mtp_key(k) for k in keys):
|
|
354
|
+
return True
|
|
355
|
+
|
|
356
|
+
# 3. DeepSeek-style trailing MTP decoder layer(s) appended after the trunk:
|
|
357
|
+
# model.layers.{num_hidden_layers + i}.*
|
|
358
|
+
start = int(
|
|
359
|
+
text_config(config).get("num_hidden_layers")
|
|
360
|
+
or config.get("num_hidden_layers")
|
|
361
|
+
or 0
|
|
362
|
+
)
|
|
363
|
+
count = _num_mtp_layers(config)
|
|
364
|
+
if start and count:
|
|
365
|
+
wanted = tuple(f"model.layers.{start + i}." for i in range(count))
|
|
366
|
+
if any(k.startswith(wanted) for k in keys):
|
|
367
|
+
return True
|
|
368
|
+
|
|
369
|
+
return False
|
|
370
|
+
|
|
371
|
+
|
|
316
372
|
@dataclass(frozen=True)
|
|
317
373
|
class TensorInfo:
|
|
318
374
|
key: str
|