mtplx 2.0.2__tar.gz → 2.2.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {mtplx-2.0.2 → mtplx-2.2.0}/CHANGELOG.md +47 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/PKG-INFO +3 -3
- mtplx-2.2.0/mtplx/backends/hy_v3_mtp.py +69 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/backends/registry.py +47 -3
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/cache_bank/cold_tier.py +128 -101
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/cache_state.py +20 -1
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/cli.py +15 -6
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/commands/forge.py +161 -4
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/commands/public.py +224 -33
- mtplx-2.2.0/mtplx/context_copy.py +114 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/engine_session.py +88 -2
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/generation.py +845 -30
- mtplx-2.2.0/mtplx/hy_v3_mtp_patch.py +165 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/model_catalog.py +31 -4
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/model_scheduler.py +43 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/mtp_patch.py +63 -1
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/profiles.py +11 -1
- mtplx-2.2.0/mtplx/qwen3_5_mtp_patch.py +360 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/runtime.py +99 -1
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/server/dashboard_state.py +3 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/server/openai.py +864 -67
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/session_bank.py +118 -2
- mtplx-2.2.0/mtplx/templates/qwen36_froggeric_v19/chat_template.jinja +267 -0
- mtplx-2.2.0/mtplx/templates/qwen36_froggeric_v21_3/chat_template.jinja +329 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/version.py +2 -2
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/vision/splice.py +57 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx.egg-info/PKG-INFO +3 -3
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx.egg-info/SOURCES.txt +43 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx.egg-info/requires.txt +2 -2
- {mtplx-2.0.2 → mtplx-2.2.0}/pyproject.toml +11 -5
- mtplx-2.2.0/scripts/fp16_turbo_exactness_20260707.py +303 -0
- mtplx-2.2.0/scripts/kvcache_exactness_audit_20260703.py +303 -0
- mtplx-2.2.0/scripts/kvcache_soak_20260703.py +227 -0
- mtplx-2.2.0/scripts/kvcache_warm_probe_20260703.py +357 -0
- mtplx-2.2.0/scripts/ocspeed-20260703/abba_final.sh +85 -0
- mtplx-2.2.0/scripts/ocspeed-20260703/accept_depth_probe.py +209 -0
- mtplx-2.2.0/scripts/ocspeed-20260703/arm_runner.sh +38 -0
- mtplx-2.2.0/scripts/ocspeed-20260703/forward_depth_bisect.py +132 -0
- mtplx-2.2.0/scripts/ocspeed-20260703/longgen_probe.py +85 -0
- mtplx-2.2.0/scripts/ocspeed-20260703/loop_degradation_probe.py +81 -0
- mtplx-2.2.0/scripts/ocspeed-20260703/paired_longgen.sh +61 -0
- mtplx-2.2.0/scripts/ocspeed-20260703/sdpa_microbench.py +66 -0
- mtplx-2.2.0/scripts/ocspeed-20260703/snapshot_tail.py +60 -0
- mtplx-2.2.0/scripts/pillar_gate_qa.py +317 -0
- mtplx-2.2.0/scripts/prodqa-20260703/candidate_truecold_20260703.py +101 -0
- mtplx-2.2.0/scripts/prodqa-20260703/cold_pair_recheck.py +88 -0
- mtplx-2.2.0/scripts/prodqa-20260703/competitors_20260703.sh +63 -0
- mtplx-2.2.0/scripts/prodqa-20260703/decode_gap_matrix.py +110 -0
- mtplx-2.2.0/scripts/prodqa-20260703/hermes_pty_driver.py +87 -0
- mtplx-2.2.0/scripts/prodqa-20260703/omlx_probe.py +63 -0
- mtplx-2.2.0/scripts/prodqa-20260703/pi_pty_driver.py +77 -0
- mtplx-2.2.0/scripts/prodqa-20260703/pillar_ab_20260703.py +169 -0
- mtplx-2.2.0/scripts/prodqa-20260703/tool_gauntlet_20260703.sh +40 -0
- mtplx-2.2.0/scripts/quality_fp16_parent_logitdiff_20260707.py +135 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/scripts/release_macos_v1.sh +22 -0
- mtplx-2.2.0/tests/test_ar_batch_penalties.py +114 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_artifacts.py +0 -1
- mtplx-2.2.0/tests/test_cold_tier_write_budget.py +123 -0
- mtplx-2.2.0/tests/test_context_copy_stats.py +405 -0
- mtplx-2.2.0/tests/test_device_draft_core.py +117 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_engine_session_env.py +36 -2
- {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_forge_cli.py +129 -4
- mtplx-2.2.0/tests/test_gdn_boundary_retention.py +150 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_generation_sustained.py +24 -0
- mtplx-2.2.0/tests/test_hy_v3_mtp_backend.py +72 -0
- mtplx-2.2.0/tests/test_memory_pressure_guard.py +169 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_model_catalog.py +43 -9
- mtplx-2.2.0/tests/test_mtp_alias_load_path.py +73 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_mtp_patch.py +61 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_openai_bridge.py +6 -1
- mtplx-2.2.0/tests/test_orphan_tool_markup.py +85 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_postcommit_prefix_reuse.py +3 -3
- mtplx-2.2.0/tests/test_postcommit_resolve_for_request.py +101 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_postcommit_tools_plumbing.py +3 -3
- {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_postcommit_wait_integration.py +1 -1
- {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_profiles.py +1 -1
- {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_public_cli.py +338 -4
- mtplx-2.2.0/tests/test_qwen3_5_mtp_backend.py +66 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_server_openai.py +107 -3
- {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_session_bank_env_caps.py +9 -6
- mtplx-2.2.0/tests/test_ssd_boundary_repersist.py +221 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_tool_aware_stream_translator.py +29 -1
- mtplx-2.2.0/tests/test_vision_session_cache.py +195 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/CITATION.cff +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/LICENSE +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/MANIFEST.in +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/NOTICE +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/README.md +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/__init__.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/adaptive.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/app_settings.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/artifacts.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/attention_context.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/attention_split.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/backends/__init__.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/backends/deepseek_mtp.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/backends/descriptors.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/backends/gemma4_assistant.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/backends/glm_mtp.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/backends/mimo_mtp.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/backends/nemotron_h_mtp.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/backends/qwen3_next.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/backends/step3p5_mtp.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/batching/__init__.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/batching/admission.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/batching/buckets.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/batching/scheduler.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/batching/state.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/benchmarks/__init__.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/benchmarks/prompts/aime_2026.jsonl +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/benchmarks/prompts/calibration_coding.jsonl +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/benchmarks/prompts/default.jsonl +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/benchmarks/prompts/flappy.jsonl +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/benchmarks/prompts/long_code.jsonl +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/benchmarks/prompts/long_code_uncapped.jsonl +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/benchmarks/prompts/python_modules_long.jsonl +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/benchmarks/runners/__init__.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/benchmarks/runners/aime.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/benchmarks/runners/batch_equivalence.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/benchmarks/runners/capture_commit_equivalence.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/benchmarks/runners/competitor_baselines.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/benchmarks/runners/contract_probe.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/benchmarks/runners/harness.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/benchmarks/runners/mtp1_gate.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/benchmarks/runners/mtp1_sampler_smoke.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/benchmarks/runners/mtp_adaptive.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/benchmarks/runners/mtp_chain_probe.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/benchmarks/runners/mtp_depth_grid.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/benchmarks/runners/mtp_depth_sweep.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/benchmarks/runners/mtp_tree_probe.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/benchmarks/runners/multi_qmv_probe.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/benchmarks/runners/preflight.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/benchmarks/runners/runtime_smoke.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/benchmarks/runners/session_bank.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/benchmarks/runners/truth.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/benchmarks/runners/verify_profile.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/benchmarks/runners/verify_qmm_probe.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/benchmarks/runners/verify_ratio.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/benchmarks/schema.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/benchmarks/validators/__init__.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/benchmarks/validators/aime.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/benchmarks/validators/basic.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/block_attention.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/cache_bank/__init__.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/cache_bank/codec.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/chat_encoding.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/commands/__init__.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/compressed_tensors.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/config.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/constants.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/correctors/__init__.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/correctors/diagonal_affine.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/correctors/low_rank.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/daemon_client.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/dashboard/__init__.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/dashboard/_static/assets/index-BYd4MFty.css +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/dashboard/_static/assets/index-COqTDxL-.js +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/dashboard/_static/index.html +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/deepseek_mtp_patch.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/default_models.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/diagnostics.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/draft_lm_head.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/draft_sampling.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/env.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/errors.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/expert_layout.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/fan_mode.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/fast_sampling.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/gdn_capture.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/gemma4_pair.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/glm_mtp_patch.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/graphbank.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/hardware.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/hf_loader.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/kernel_selfcheck.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/kernels/__init__.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/kernels/copy_leaf.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/kernels/fused_norm.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/kernels/lm_head_topk.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/kernels/logits_topk.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/kernels/native_gdn_tail.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/kernels/sdpa_2pass.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/kernels/sdpa_2pass_paged.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/kernels/sdpa_2pass_paged_q8.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/kernels/sdpa_gqa_packed.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/kernels/verify_mlp_fused.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/kpi/__init__.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/kpi/reference_vllm.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/kpi/runtime_kpis.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/kv_quant.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/loop_guard.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/mimo_mtp_patch.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/mtp_activation_stats.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/mtp_adapters.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/native_mlp.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/nax_verify.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/nemotron_h_mtp_patch.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/opencode.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/pi.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/prefill_bench.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/proposal_reranker.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/reasoning_codecs.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/runtime_options.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/sampling.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/server/__init__.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/server/omlx_bridge/__init__.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/server/omlx_bridge/adapter.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/server/omlx_bridge/thinking.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/server/omlx_bridge/tool_calling.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/server_urls.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/speculative.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/step3p5_mtp_patch.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/swival.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/thermal.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/thermal_sidecar.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/trace_parity.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/turboquant.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/ui/__init__.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/ui/banner.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/ui/chat_printer.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/ui/download_progress.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/ui/onboarding.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/ui/panels.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/ui/progress.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/verify_kernels.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/verify_qmv.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/vision/__init__.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/vision/processing.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx/vision/qwen3_vl_tower.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx.egg-info/dependency_links.txt +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx.egg-info/entry_points.txt +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/mtplx.egg-info/top_level.txt +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/scripts/agent_user_path_qa.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/scripts/aime_serve_gate.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/scripts/aime_shape_memory_bench.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/scripts/collect_mtp_activation_stats.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/scripts/collect_mtp_hidden_calib.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/scripts/combine_mtp_adapters.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/scripts/compiled_verify_exactness.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/scripts/convert_step37_to_mtplx_step3p5.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/scripts/eval_mtp_corrector.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/scripts/filter_mtp_adapter.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/scripts/fresh_venv_smoke.sh +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/scripts/hygiene_scan.sh +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/scripts/install_macos.sh +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/scripts/install_preview_global.sh +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/scripts/make_fp16_precision_sibling.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/scripts/midform_gate.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/scripts/opencode_concurrency_qa.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/scripts/phase0h_paged_verifier_exactness.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/scripts/probe_mlx_pr3026_qsdpa.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/scripts/probe_mx_compile_buckets.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/scripts/probe_paged_gqa_sdpa_routes.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/scripts/r1_chisquare_verifier_correctness.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/scripts/run_context_degradation_diagnostics.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/scripts/serve_openai_mtplx.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/scripts/session_cache_followup_qa.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/scripts/sparkle_rehearsal_kit.sh +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/scripts/step_acceptance.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/scripts/step_smoke.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/scripts/train_mtp_adapter_c4.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/scripts/validate_step_mtp_injector.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/setup.cfg +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_adaptive.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_aime_serve_gate.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_attention_split.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_background_warmup.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_batching_foundation.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_cache_bank.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_cache_state.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_cli_parity_tools.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_compressed_tensors.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_config.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_config_profile_precedence.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_context_degradation_profiles.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_correctors.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_daemon_client.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_dashboard_endpoints.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_default_models.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_diagnostics.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_download_progress.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_draft_lm_head.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_engine_session_concurrency.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_fast_sampling.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_generation_store_on_prefill.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_graphbank_compiled_verify.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_hf_loader.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_hygiene_scan.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_idle_postcommit_subagent.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_kernel_selfcheck.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_lazy_snapshot_cow.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_loop_guard.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_max_idle_watchdog.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_max_lifecycle.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_metal_memory_caps.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_midform_gate.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_model_scheduler.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_mtp_activation_stats.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_mtp_adapters.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_mtp_depth_sweep.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_nax_verify.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_no_mlx_imports.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_omlx_bridge.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_onboarding.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_opencode.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_penalties.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_penalty_request_wiring.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_persistent_replay.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_phase0h_paged_verifier_exactness.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_policy_fingerprint_stability.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_postcommit_wait.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_prefill_bench.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_prefill_chunk_defaults.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_prefill_tps_regression.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_prompt_encoding.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_reasoning_stream_split.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_runtime_kpis.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_sampling.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_scoped_reasoning_history.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_sdpa_gqa_packed.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_session_bank.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_step3p5_mtp_patch.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_sustained_long_context_qa.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_thermal.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_thermal_sidecar.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_trace_parity.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_turboquant.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_turboquant_fallback.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_ui_progress.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_validators.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_vision_tower.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/tests/test_vllm_reference.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/vllm_metal/__init__.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/vllm_metal/metal/README.md +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/vllm_metal/metal/__init__.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/vllm_metal/metal/build.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/vllm_metal/metal/constants.py +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/vllm_metal/metal/kernels_v1/copy_blocks.metal +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/vllm_metal/metal/kernels_v1/float8.metal +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/vllm_metal/metal/kernels_v1/gather_kv_cache.metal +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/vllm_metal/metal/kernels_v1/kv_scale_update.metal +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/vllm_metal/metal/kernels_v1/pagedattention.metal +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/vllm_metal/metal/kernels_v1/reshape_and_cache.metal +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/vllm_metal/metal/kernels_v1/utils.metal +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/vllm_metal/metal/kernels_v2/copy_blocks.metal +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/vllm_metal/metal/kernels_v2/float8.metal +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/vllm_metal/metal/kernels_v2/gather_kv_cache.metal +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/vllm_metal/metal/kernels_v2/gdn_linear_attention.metal +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/vllm_metal/metal/kernels_v2/kv_scale_update.metal +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/vllm_metal/metal/kernels_v2/pagedattention.metal +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/vllm_metal/metal/kernels_v2/reshape_and_cache.metal +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/vllm_metal/metal/kernels_v2/turboquant.metal +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/vllm_metal/metal/kernels_v2/utils.metal +0 -0
- {mtplx-2.0.2 → mtplx-2.2.0}/vllm_metal/metal/paged_ops.cpp +0 -0
|
@@ -4,6 +4,53 @@ All notable user-facing changes to MTPLX. The format is based on
|
|
|
4
4
|
[Keep a Changelog](https://keepachangelog.com/en/1.1.0/); versions follow
|
|
5
5
|
[Semantic Versioning](https://semver.org/).
|
|
6
6
|
|
|
7
|
+
## [2.2.0] - 2026-07-19
|
|
8
|
+
|
|
9
|
+
The copy-drafting and small-Mac release. Decoding: context-copy
|
|
10
|
+
(prompt-lookup) drafting lands on by default (PR #151 by lBroth) with an
|
|
11
|
+
exact temperature path — copied blocks are accepted with the target's own
|
|
12
|
+
shaped probability, so the output distribution is unchanged at any
|
|
13
|
+
temperature; measured +53% on edit-heavy agent turns at temp 0.6, parity
|
|
14
|
+
on novel text (disable with MTPLX_CONTEXT_COPY=0). Models: the 4B
|
|
15
|
+
zero-acceptance defect (#176) is root-caused and fixed — the engine heals
|
|
16
|
+
raw delta-encoded MTP sidecars at load so existing downloads recover
|
|
17
|
+
without re-downloading, the 4B Speed artifact is rebuilt (227.8 tok/s D3
|
|
18
|
+
on M5 Max, 1.71x AR), and a new 4B Quality artifact ships (191.7 tok/s
|
|
19
|
+
D3, 2.19x — the largest MTP multiplier in the fleet); sub-16GB Macs get
|
|
20
|
+
first-class catalog recommendations. Tune (#177): a 0.0-acceptance depth
|
|
21
|
+
can never win, be saved, or be replayed, and poisoned records are
|
|
22
|
+
quarantined at load. Forge: FP16 precision option with M1/M2 auto-select
|
|
23
|
+
(#166); the fp16 cast can no longer corrupt a sidecar in place;
|
|
24
|
+
degenerate sidecars are quarantined on re-forge. Server: SSD session
|
|
25
|
+
writer crash fixed — encode at enqueue (#169); live requests preempt idle
|
|
26
|
+
cache maintenance. MoE: mtp_depth_max is a ceiling, not the default; the
|
|
27
|
+
35B-A3B launches at its measured D2 (#174 part 1 by davidtai). Also: the
|
|
28
|
+
Qwen 3.6 27B AR decode-trace crash fix (#167 by davidtai), truthful
|
|
29
|
+
per-model profile display, hybrid-model boundary retention across append
|
|
30
|
+
churn, and an experimental --draft-core device (opt-in). Full details in
|
|
31
|
+
docs/releases/v2.2.0.md.
|
|
32
|
+
|
|
33
|
+
## [2.1.0] - 2026-07-17
|
|
34
|
+
|
|
35
|
+
The community-fixes release. Memory: the v2.x reports are root-caused and
|
|
36
|
+
closed (MLX allocator cache bounded by default, per-session admission
|
|
37
|
+
re-clamped on sub-96GB machines, paged pool bounded by the context
|
|
38
|
+
window, pressure responder redesigned, q4 kv-quant crash fixed, new
|
|
39
|
+
`--memory-budget` knob). Agent sessions: warm prefix reuse survives every
|
|
40
|
+
tool turn (#121), hybrid-model near-prefix restores no longer collapse to
|
|
41
|
+
the oldest boundary (measured 0.4s instead of 33.8s on a 22k follow-up),
|
|
42
|
+
restart-warm sessions keep their boundary records across SSD generations
|
|
43
|
+
(#159, #144), and cache hits are reported in standard `usage` fields.
|
|
44
|
+
Sampling: presence and frequency penalties fixed in the batched AR lane
|
|
45
|
+
(#156). App: startup and update hang fixed plus a full subprocess
|
|
46
|
+
watchdog sweep (#158), the Hermes tile launches Hermes Desktop, raw
|
|
47
|
+
tool-call XML no longer leaks into no-tools chats (#160). CLI: `start
|
|
48
|
+
opencode` serves the same lane the app serves. Backends: qwen3_5_mtp and
|
|
49
|
+
hy_v3 land (#142, #147). Performance: the model-owner thread is
|
|
50
|
+
QoS-pinned for 8 to 10% faster decode under real multitasking load.
|
|
51
|
+
Operators: `MTPLX_COMPILED_VERIFY_MAX_CONTEXT` is env-overridable. Full
|
|
52
|
+
details in docs/releases/v2.1.0.md.
|
|
53
|
+
|
|
7
54
|
## [2.0.2] - 2026-07-09
|
|
8
55
|
|
|
9
56
|
The agent-reliability release: multi-turn agent sessions now render
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: mtplx
|
|
3
|
-
Version: 2.0
|
|
3
|
+
Version: 2.2.0
|
|
4
4
|
Summary: Native MTP speculative decoding for Qwen3-Next on Apple Silicon.
|
|
5
5
|
Author-email: Youssof Altoukhi <business@youssofal.com>
|
|
6
6
|
License-Expression: Apache-2.0
|
|
@@ -25,9 +25,9 @@ License-File: LICENSE
|
|
|
25
25
|
License-File: NOTICE
|
|
26
26
|
Requires-Dist: fastapi>=0.136
|
|
27
27
|
Requires-Dist: huggingface-hub>=0.36
|
|
28
|
-
Requires-Dist: mlx<0.
|
|
28
|
+
Requires-Dist: mlx<0.33,>=0.31; sys_platform == "darwin" and platform_machine == "arm64"
|
|
29
29
|
Requires-Dist: mlx-lm<0.32,>=0.31; sys_platform == "darwin" and platform_machine == "arm64"
|
|
30
|
-
Requires-Dist: transformers
|
|
30
|
+
Requires-Dist: transformers!=5.13.0,<5.14; sys_platform == "darwin" and platform_machine == "arm64"
|
|
31
31
|
Requires-Dist: nanobind>=2; sys_platform == "darwin" and platform_machine == "arm64"
|
|
32
32
|
Requires-Dist: numpy>=2
|
|
33
33
|
Requires-Dist: pydantic>=2
|
|
@@ -0,0 +1,69 @@
|
|
|
1
|
+
"""Tencent Hy3 (hy_v3) native MTP backend facade.
|
|
2
|
+
|
|
3
|
+
Hy3 ships one appended MTP layer (num_nextn_predict_layers=1): a full MoE
|
|
4
|
+
decoder layer fed concat[RMSNorm(next-token embedding), RMSNorm(trunk
|
|
5
|
+
pre-final-norm hidden state)] through an eh_proj down-projection, sharing the
|
|
6
|
+
trunk's embeddings and lm_head. The MLX reference implementation exposes it as
|
|
7
|
+
``Model.predict_next_tokens(hidden, token_ids, cache)`` with
|
|
8
|
+
``return_hidden_states=True`` on the trunk forward (see
|
|
9
|
+
mlx-lm ``models/hy_v3.py``, MTP revision).
|
|
10
|
+
|
|
11
|
+
Like the GLM/DeepSeek facades, drafting and verification are wired through the
|
|
12
|
+
shared speculative sampler in ``generation.py``; this facade gates execution
|
|
13
|
+
behind the verified runtime contract.
|
|
14
|
+
"""
|
|
15
|
+
|
|
16
|
+
from __future__ import annotations
|
|
17
|
+
|
|
18
|
+
from pathlib import Path
|
|
19
|
+
from typing import Any
|
|
20
|
+
|
|
21
|
+
from . import DraftTokens, ModelState, MTPBackend, VerifyOutput
|
|
22
|
+
from mtplx.profiles import DEFAULT_PROFILE_NAME
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
class HyV3MTPBackend(MTPBackend):
|
|
26
|
+
arch_id = "hy-v3-mtp"
|
|
27
|
+
|
|
28
|
+
def load(self, model_path: Path) -> ModelState:
|
|
29
|
+
from mtplx.mtp_patch import MTPContract
|
|
30
|
+
from mtplx.runtime import load
|
|
31
|
+
|
|
32
|
+
runtime = load(model_path, mtp=True, contract=MTPContract())
|
|
33
|
+
return ModelState(
|
|
34
|
+
model_path=Path(model_path),
|
|
35
|
+
runtime=runtime,
|
|
36
|
+
metadata={"arch_id": self.arch_id, "contract_gated": True},
|
|
37
|
+
)
|
|
38
|
+
|
|
39
|
+
def verify(self, state: ModelState, draft_tokens: DraftTokens, hidden: Any) -> VerifyOutput:
|
|
40
|
+
raise NotImplementedError("HyV3MTPBackend.verify is wired through generation.py")
|
|
41
|
+
|
|
42
|
+
def propose(self, state: ModelState, hidden: Any) -> DraftTokens:
|
|
43
|
+
raise NotImplementedError("HyV3MTPBackend.propose is wired through generation.py")
|
|
44
|
+
|
|
45
|
+
def recommended_profile(self) -> str:
|
|
46
|
+
return DEFAULT_PROFILE_NAME
|
|
47
|
+
|
|
48
|
+
def health(self) -> dict[str, Any]:
|
|
49
|
+
return {
|
|
50
|
+
"arch_id": self.arch_id,
|
|
51
|
+
"runtime_path": "mtplx.runtime + mtplx.hy_v3_mtp_patch + mtplx.generation",
|
|
52
|
+
"support_level": "experimental-native-contract-gated",
|
|
53
|
+
"contract_required": True,
|
|
54
|
+
"supported_model_types": ["hy_v3"],
|
|
55
|
+
"mtp_depth_max": 1,
|
|
56
|
+
"notes": (
|
|
57
|
+
"Single appended NextN layer with its own 192-expert MoE MLP, "
|
|
58
|
+
"sigmoid top-8 routing with expert bias, eh_proj over "
|
|
59
|
+
"concat[enorm(embedding), hnorm(hidden)], shared embeddings "
|
|
60
|
+
"and head. Draft layer consumes the trunk pre-final-norm "
|
|
61
|
+
"hidden state. Verification is exact rejection sampling in "
|
|
62
|
+
"generation.py; the MLX reference (mlx-lm hy_v3 MTP revision) "
|
|
63
|
+
"verifies greedily and is temp-0 exact."
|
|
64
|
+
),
|
|
65
|
+
"references": [
|
|
66
|
+
"REFERENCES:TOOLS/vllm-official-main/vllm/model_executor/models/hy_v3_mtp.py",
|
|
67
|
+
"REFERENCES:TOOLS/mlx-lm/mlx_lm/models/hy_v3.py",
|
|
68
|
+
],
|
|
69
|
+
}
|
|
@@ -21,6 +21,7 @@ SUPPORTED_ARCH_IDS = {
|
|
|
21
21
|
"nemotron-h-mtp",
|
|
22
22
|
"gemma4-assistant-mtp",
|
|
23
23
|
"step3p5-mtp",
|
|
24
|
+
"hy-v3-mtp",
|
|
24
25
|
}
|
|
25
26
|
|
|
26
27
|
TIER_VERIFIED = "verified"
|
|
@@ -411,10 +412,22 @@ ARCHITECTURE_CATALOG: dict[str, ArchitectureSupport] = {
|
|
|
411
412
|
display_name="HY V3 MTP",
|
|
412
413
|
family="hy",
|
|
413
414
|
backend="hy_v3_mtp",
|
|
414
|
-
support_level="
|
|
415
|
-
runtime_compatibility="
|
|
415
|
+
support_level="experimental-native-contract-gated",
|
|
416
|
+
runtime_compatibility="native-contract-gated",
|
|
417
|
+
can_run_verified=True,
|
|
416
418
|
aliases=("hy_v3_mtp", "hy_v3"),
|
|
417
|
-
|
|
419
|
+
family_gate="appended-layer-mtp-markers",
|
|
420
|
+
references=(
|
|
421
|
+
"REFERENCES:TOOLS/vllm-official-main/vllm/model_executor/models/hy_v3_mtp.py",
|
|
422
|
+
"REFERENCES:TOOLS/mlx-lm/mlx_lm/models/hy_v3.py",
|
|
423
|
+
),
|
|
424
|
+
notes=(
|
|
425
|
+
"Hy3 ships one appended NextN layer with its own 192-expert MoE, "
|
|
426
|
+
"eh_proj over concat[enorm(embedding), hnorm(hidden)], and shared "
|
|
427
|
+
"embeddings/head. The mlx-lm hy_v3 MTP revision exposes the head "
|
|
428
|
+
"natively (predict_next_tokens), so injection binds the existing "
|
|
429
|
+
"surface rather than grafting weights."
|
|
430
|
+
),
|
|
418
431
|
),
|
|
419
432
|
"generic-mtp": ArchitectureSupport(
|
|
420
433
|
arch_id="generic-mtp",
|
|
@@ -794,6 +807,35 @@ def _has_all_suffixes_under_prefixes(
|
|
|
794
807
|
return True
|
|
795
808
|
|
|
796
809
|
|
|
810
|
+
_HY_V3_MTP_MARKER_SUFFIXES = (
|
|
811
|
+
"enorm.weight",
|
|
812
|
+
"hnorm.weight",
|
|
813
|
+
"eh_proj.weight",
|
|
814
|
+
"final_layernorm.weight",
|
|
815
|
+
)
|
|
816
|
+
|
|
817
|
+
|
|
818
|
+
def _passes_hy_v3_gate(inspection: Any) -> bool:
|
|
819
|
+
"""Hy3's appended MTP block lives directly under an ``mtp.`` prefix
|
|
820
|
+
(``mtp.enorm.weight``, ``mtp.hnorm.weight``, ``mtp.eh_proj.weight``,
|
|
821
|
+
``mtp.final_layernorm.weight``, ``mtp.layer.*``) rather than the
|
|
822
|
+
``mtp.layers.{idx}.`` nesting DeepSeek/GLM/Step use, so it needs its own
|
|
823
|
+
gate instead of `_passes_appended_layer_gate` (verified against the
|
|
824
|
+
shipped `hy3-demolition-mlx-*-mtp` checkpoints' safetensors index)."""
|
|
825
|
+
keys = _weight_keys(inspection)
|
|
826
|
+
if not keys:
|
|
827
|
+
return False
|
|
828
|
+
count = int(getattr(inspection, "mtp_num_hidden_layers", 0) or 0)
|
|
829
|
+
if count <= 0:
|
|
830
|
+
return False
|
|
831
|
+
return _has_marker_under_prefixes(
|
|
832
|
+
keys,
|
|
833
|
+
("mtp.",),
|
|
834
|
+
_HY_V3_MTP_MARKER_SUFFIXES,
|
|
835
|
+
("mtp.layer.",),
|
|
836
|
+
)
|
|
837
|
+
|
|
838
|
+
|
|
797
839
|
def _passes_appended_layer_gate(inspection: Any) -> bool:
|
|
798
840
|
keys = _weight_keys(inspection)
|
|
799
841
|
if not keys:
|
|
@@ -910,6 +952,8 @@ def _passes_family_runtime_gate(arch_id: str, inspection: Any, tensor_gate: bool
|
|
|
910
952
|
tensor_gate
|
|
911
953
|
and int(getattr(inspection, "mtp_num_hidden_layers", 0) or 0) > 0
|
|
912
954
|
)
|
|
955
|
+
if arch_id == "hy-v3-mtp":
|
|
956
|
+
return _passes_hy_v3_gate(inspection)
|
|
913
957
|
if arch_id in {
|
|
914
958
|
"deepseek-v3-mtp",
|
|
915
959
|
"glm-moe-dsa-mtp",
|
|
@@ -3,6 +3,7 @@
|
|
|
3
3
|
from __future__ import annotations
|
|
4
4
|
|
|
5
5
|
import hashlib
|
|
6
|
+
from collections import deque
|
|
6
7
|
import json
|
|
7
8
|
import logging
|
|
8
9
|
import os
|
|
@@ -70,14 +71,15 @@ _COMMITTED_CACHE_POLICIES = frozenset({"committed", "last_window"})
|
|
|
70
71
|
|
|
71
72
|
|
|
72
73
|
def _deferred_encode_enabled() -> bool:
|
|
73
|
-
"""
|
|
74
|
+
"""RETIRED (#169, 2026-07-17) — writer-side encode is gone; always False.
|
|
74
75
|
|
|
75
|
-
The
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
|
|
80
|
-
|
|
76
|
+
The kvcache-v2 writer-thread encode block-sliced tensors at write time
|
|
77
|
+
(TreeCodec builds lazy slice arrays and mx.eval()s them), which crashed
|
|
78
|
+
on restore-derived arrays ("There is no Stream(gpu, 1) in current
|
|
79
|
+
thread") and could serialize donation-mutated KV pages from the writer
|
|
80
|
+
backlog. put_entry now always encodes at enqueue on the owner thread.
|
|
81
|
+
MTPLX_SSD_DEFERRED_ENCODE is parsed nowhere and ignored."""
|
|
82
|
+
return False
|
|
81
83
|
|
|
82
84
|
|
|
83
85
|
@dataclass(frozen=True)
|
|
@@ -100,6 +102,9 @@ class PendingWrite:
|
|
|
100
102
|
tensors: dict[str, bytes]
|
|
101
103
|
deferred: DeferredPayload | None = None
|
|
102
104
|
created_at_s: float = field(default_factory=time.time)
|
|
105
|
+
# Estimated bytes this write pins in memory until the writer drains it
|
|
106
|
+
# (deferred payloads hold live KV arrays; encoded ones hold the buffers).
|
|
107
|
+
pinned_nbytes: int = 0
|
|
103
108
|
|
|
104
109
|
|
|
105
110
|
@dataclass(frozen=True)
|
|
@@ -169,6 +174,10 @@ def default_cold_tier_max_bytes() -> int:
|
|
|
169
174
|
return DEFAULT_COLD_TIER_MAX_BYTES
|
|
170
175
|
|
|
171
176
|
|
|
177
|
+
def _env_size_bytes(name: str, default: int) -> int:
|
|
178
|
+
return parse_size_bytes(os.environ.get(name), default)
|
|
179
|
+
|
|
180
|
+
|
|
172
181
|
def parse_size_bytes(value: str | int | None, default: int) -> int:
|
|
173
182
|
if value is None:
|
|
174
183
|
return int(default)
|
|
@@ -270,6 +279,24 @@ class SessionBankColdTier:
|
|
|
270
279
|
self._queue: queue.Queue[PendingWrite | None] = queue.Queue(
|
|
271
280
|
maxsize=max(1, int(writer_queue_depth))
|
|
272
281
|
)
|
|
282
|
+
# Backlog byte cap (issue #145): every queued write pins its payload
|
|
283
|
+
# (deferred ones pin LIVE KV arrays) until the writer drains it. A
|
|
284
|
+
# count-bounded queue of 32 multi-GB snapshots can pin ~50 GB under
|
|
285
|
+
# distinct-prefix churn — measured live 2026-07-09 (active memory
|
|
286
|
+
# climbed 35 -> 66 GB while the bank ledger stayed flat). Cap the
|
|
287
|
+
# pinned bytes, drop new writes beyond it.
|
|
288
|
+
self._pending_bytes = 0
|
|
289
|
+
self._backlog_budget_bytes = _env_size_bytes(
|
|
290
|
+
"MTPLX_SSD_WRITER_BACKLOG_BYTES", 4 * 1024**3
|
|
291
|
+
)
|
|
292
|
+
# Hourly write budget (issue #144: 7 TB written / SSD wear): the
|
|
293
|
+
# different-repos pattern writes GBs per task and never restores
|
|
294
|
+
# them (measured 58 GB in 45 min with restore_hits=0). Rolling
|
|
295
|
+
# one-hour byte budget; beyond it new writes are skipped.
|
|
296
|
+
self._write_budget_per_hour_bytes = _env_size_bytes(
|
|
297
|
+
"MTPLX_SSD_WRITE_BUDGET_PER_HOUR", 64 * 1024**3
|
|
298
|
+
)
|
|
299
|
+
self._written_window: deque[tuple[float, int]] = deque()
|
|
273
300
|
self._stop = threading.Event()
|
|
274
301
|
self._base_lock = threading.RLock()
|
|
275
302
|
self._disk_usage_lock = threading.Lock()
|
|
@@ -336,82 +363,55 @@ class SessionBankColdTier:
|
|
|
336
363
|
if len(token_ids) < self.min_prefix_tokens:
|
|
337
364
|
self._inc("skipped_too_short")
|
|
338
365
|
return False
|
|
366
|
+
estimated_nbytes = int(getattr(entry, "nbytes", 0) or 0)
|
|
367
|
+
if not self._admit_write(estimated_nbytes):
|
|
368
|
+
return False
|
|
339
369
|
boundaries = tuple(
|
|
340
370
|
(int(r[0]), r[1], r[2] if len(r) > 2 else None)
|
|
341
371
|
for r in (getattr(entry, "gdn_boundaries", None) or [])
|
|
342
372
|
)
|
|
343
|
-
|
|
344
|
-
|
|
345
|
-
|
|
346
|
-
|
|
347
|
-
|
|
348
|
-
|
|
349
|
-
|
|
350
|
-
|
|
351
|
-
|
|
352
|
-
|
|
353
|
-
|
|
354
|
-
|
|
355
|
-
|
|
356
|
-
|
|
357
|
-
|
|
358
|
-
|
|
359
|
-
|
|
360
|
-
|
|
361
|
-
|
|
362
|
-
|
|
363
|
-
)
|
|
364
|
-
except Exception as exc:
|
|
365
|
-
self._inc("skipped_serialize_error")
|
|
366
|
-
logger.warning(
|
|
367
|
-
"SessionBank SSD payload prep skipped: %s: %s",
|
|
368
|
-
type(exc).__name__,
|
|
369
|
-
exc,
|
|
370
|
-
)
|
|
371
|
-
return False
|
|
372
|
-
metadata = self._metadata_for_entry(
|
|
373
|
-
entry,
|
|
374
|
-
capabilities=capabilities or (),
|
|
375
|
-
payload_nbytes=0,
|
|
376
|
-
)
|
|
377
|
-
pending = PendingWrite(
|
|
378
|
-
entry_id=str(metadata["entry_id"]),
|
|
379
|
-
token_ids=token_ids,
|
|
380
|
-
metadata=metadata,
|
|
381
|
-
payload_spec=None,
|
|
382
|
-
tensors={},
|
|
383
|
-
deferred=deferred,
|
|
384
|
-
)
|
|
385
|
-
else:
|
|
386
|
-
try:
|
|
387
|
-
encoded = encode_payload(
|
|
388
|
-
cache_snapshot=getattr(entry, "cache_snapshot"),
|
|
389
|
-
logits=getattr(entry, "logits"),
|
|
390
|
-
hidden=getattr(entry, "hidden"),
|
|
391
|
-
mtp_history_snapshot=getattr(entry, "mtp_history_snapshot", None),
|
|
392
|
-
gdn_boundaries=boundaries,
|
|
393
|
-
has_recurrent=bool(getattr(entry, "has_recurrent", False)),
|
|
394
|
-
block_size=self.block_size,
|
|
395
|
-
)
|
|
396
|
-
except Exception as exc:
|
|
397
|
-
self._inc("skipped_serialize_error")
|
|
398
|
-
logger.warning("SessionBank SSD serialize skipped: %s: %s", type(exc).__name__, exc)
|
|
399
|
-
return False
|
|
400
|
-
metadata = self._metadata_for_entry(
|
|
401
|
-
entry,
|
|
402
|
-
capabilities=capabilities or (),
|
|
403
|
-
payload_nbytes=encoded.nbytes,
|
|
404
|
-
)
|
|
405
|
-
pending = PendingWrite(
|
|
406
|
-
entry_id=str(metadata["entry_id"]),
|
|
407
|
-
token_ids=token_ids,
|
|
408
|
-
metadata=metadata,
|
|
409
|
-
payload_spec=encoded.spec,
|
|
410
|
-
tensors=encoded.tensors,
|
|
373
|
+
# Encode ALWAYS happens here, on the enqueueing (owner) thread, never
|
|
374
|
+
# on the writer thread (#169, 2026-07-17). The retired writer-side
|
|
375
|
+
# "deferred encode" (kvcache-v2) block-sliced tensors at write time:
|
|
376
|
+
# TreeCodec._encode_tensor_blocks builds new lazy slice arrays and
|
|
377
|
+
# mx.eval()s them, which (a) crashed the process on restore-derived
|
|
378
|
+
# arrays whose graphs referenced the restore stream ("There is no
|
|
379
|
+
# Stream(gpu, 1) in current thread"), and (b) held live KV references
|
|
380
|
+
# for seconds in the writer backlog, so under buffer donation the
|
|
381
|
+
# eventual serialization could capture mutated pages — silently
|
|
382
|
+
# corrupt persisted sessions that degrade on every restore. Bytes are
|
|
383
|
+
# captured at snapshot time; the writer thread is pure file IO.
|
|
384
|
+
try:
|
|
385
|
+
encoded = encode_payload(
|
|
386
|
+
cache_snapshot=getattr(entry, "cache_snapshot"),
|
|
387
|
+
logits=getattr(entry, "logits"),
|
|
388
|
+
hidden=getattr(entry, "hidden"),
|
|
389
|
+
mtp_history_snapshot=getattr(entry, "mtp_history_snapshot", None),
|
|
390
|
+
gdn_boundaries=boundaries,
|
|
391
|
+
has_recurrent=bool(getattr(entry, "has_recurrent", False)),
|
|
392
|
+
block_size=self.block_size,
|
|
411
393
|
)
|
|
394
|
+
except Exception as exc:
|
|
395
|
+
self._inc("skipped_serialize_error")
|
|
396
|
+
logger.warning("SessionBank SSD serialize skipped: %s: %s", type(exc).__name__, exc)
|
|
397
|
+
return False
|
|
398
|
+
metadata = self._metadata_for_entry(
|
|
399
|
+
entry,
|
|
400
|
+
capabilities=capabilities or (),
|
|
401
|
+
payload_nbytes=encoded.nbytes,
|
|
402
|
+
)
|
|
403
|
+
pending = PendingWrite(
|
|
404
|
+
entry_id=str(metadata["entry_id"]),
|
|
405
|
+
token_ids=token_ids,
|
|
406
|
+
metadata=metadata,
|
|
407
|
+
payload_spec=encoded.spec,
|
|
408
|
+
tensors=encoded.tensors,
|
|
409
|
+
pinned_nbytes=max(estimated_nbytes, int(encoded.nbytes)),
|
|
410
|
+
)
|
|
412
411
|
try:
|
|
413
412
|
self._queue.put_nowait(pending)
|
|
414
413
|
except queue.Full:
|
|
414
|
+
self._release_pending(pending.pinned_nbytes)
|
|
415
415
|
self._inc("skipped_queue_full")
|
|
416
416
|
logger.warning(
|
|
417
417
|
"SessionBank SSD writer queue full; skipping prefix_len=%d token_hash=%s",
|
|
@@ -573,11 +573,18 @@ class SessionBankColdTier:
|
|
|
573
573
|
def stats(self) -> dict[str, Any]:
|
|
574
574
|
with self._stats_lock:
|
|
575
575
|
stats = dict(self._stats)
|
|
576
|
+
with self._stats_lock:
|
|
577
|
+
pending_bytes = int(self._pending_bytes)
|
|
578
|
+
written_last_hour = sum(nbytes for _, nbytes in self._written_window)
|
|
576
579
|
stats.update(
|
|
577
580
|
{
|
|
578
581
|
"enabled": self.enabled,
|
|
579
582
|
"restorable": self.restorable,
|
|
580
583
|
"writer_queue_depth": int(self._queue.qsize()),
|
|
584
|
+
"writer_backlog_bytes": pending_bytes,
|
|
585
|
+
"writer_backlog_budget_bytes": int(self._backlog_budget_bytes),
|
|
586
|
+
"written_bytes_last_hour": int(written_last_hour),
|
|
587
|
+
"write_budget_per_hour_bytes": int(self._write_budget_per_hour_bytes),
|
|
581
588
|
"dir": str(self.base_dir),
|
|
582
589
|
"manifest_path": str(self._manifest_path),
|
|
583
590
|
}
|
|
@@ -786,6 +793,9 @@ class SessionBankColdTier:
|
|
|
786
793
|
self._inc("writes_completed")
|
|
787
794
|
with self._stats_lock:
|
|
788
795
|
self._stats["last_write_s"] = time.time()
|
|
796
|
+
self._written_window.append(
|
|
797
|
+
(time.time(), int(pending.pinned_nbytes))
|
|
798
|
+
)
|
|
789
799
|
logger.info(
|
|
790
800
|
"SessionBank SSD wrote entry_id=%s prefix_len=%d nbytes=%d",
|
|
791
801
|
pending.entry_id,
|
|
@@ -801,37 +811,54 @@ class SessionBankColdTier:
|
|
|
801
811
|
exc,
|
|
802
812
|
)
|
|
803
813
|
finally:
|
|
814
|
+
self._release_pending(pending.pinned_nbytes)
|
|
804
815
|
self._queue.task_done()
|
|
805
816
|
|
|
817
|
+
def _admit_write(self, estimated_nbytes: int) -> bool:
|
|
818
|
+
"""Backlog + hourly-budget admission for a new SSD write."""
|
|
819
|
+
|
|
820
|
+
now = time.time()
|
|
821
|
+
with self._stats_lock:
|
|
822
|
+
if (
|
|
823
|
+
self._pending_bytes + max(0, estimated_nbytes)
|
|
824
|
+
> self._backlog_budget_bytes
|
|
825
|
+
):
|
|
826
|
+
self._stats["skipped_backlog_bytes"] = (
|
|
827
|
+
int(self._stats.get("skipped_backlog_bytes", 0) or 0) + 1
|
|
828
|
+
)
|
|
829
|
+
return False
|
|
830
|
+
while self._written_window and self._written_window[0][0] < now - 3600:
|
|
831
|
+
self._written_window.popleft()
|
|
832
|
+
written_last_hour = sum(nbytes for _, nbytes in self._written_window)
|
|
833
|
+
if (
|
|
834
|
+
written_last_hour + max(0, estimated_nbytes)
|
|
835
|
+
> self._write_budget_per_hour_bytes
|
|
836
|
+
):
|
|
837
|
+
self._stats["skipped_write_budget"] = (
|
|
838
|
+
int(self._stats.get("skipped_write_budget", 0) or 0) + 1
|
|
839
|
+
)
|
|
840
|
+
return False
|
|
841
|
+
self._pending_bytes += max(0, estimated_nbytes)
|
|
842
|
+
return True
|
|
843
|
+
|
|
844
|
+
def _release_pending(self, nbytes: int) -> None:
|
|
845
|
+
with self._stats_lock:
|
|
846
|
+
self._pending_bytes = max(0, self._pending_bytes - max(0, int(nbytes)))
|
|
847
|
+
|
|
806
848
|
def _write_pending(self, pending: PendingWrite) -> bool:
|
|
807
849
|
if pending.deferred is not None:
|
|
808
|
-
#
|
|
809
|
-
#
|
|
810
|
-
#
|
|
811
|
-
#
|
|
812
|
-
encoded
|
|
813
|
-
|
|
814
|
-
|
|
815
|
-
|
|
816
|
-
|
|
817
|
-
|
|
818
|
-
has_recurrent=pending.deferred.has_recurrent,
|
|
819
|
-
block_size=pending.deferred.block_size,
|
|
820
|
-
)
|
|
821
|
-
metadata = dict(pending.metadata)
|
|
822
|
-
metadata["nbytes"] = int(
|
|
823
|
-
max(int(metadata.get("nbytes", 0) or 0), int(encoded.nbytes))
|
|
824
|
-
)
|
|
825
|
-
metadata["logical_nbytes"] = int(encoded.nbytes)
|
|
826
|
-
metadata["physical_nbytes"] = int(encoded.nbytes)
|
|
827
|
-
pending = PendingWrite(
|
|
828
|
-
entry_id=pending.entry_id,
|
|
829
|
-
token_ids=pending.token_ids,
|
|
830
|
-
metadata=metadata,
|
|
831
|
-
payload_spec=encoded.spec,
|
|
832
|
-
tensors=encoded.tensors,
|
|
833
|
-
created_at_s=pending.created_at_s,
|
|
850
|
+
# Retired path (#169, 2026-07-17): writer-side encode ran MLX
|
|
851
|
+
# slice/eval graph work on the writer thread (crash on
|
|
852
|
+
# restore-stream arrays, donation-corruption window). put_entry
|
|
853
|
+
# now always encodes at enqueue; a deferred payload reaching the
|
|
854
|
+
# writer is a programming error, never silently encoded here.
|
|
855
|
+
self._inc("skipped_deferred_retired")
|
|
856
|
+
logger.error(
|
|
857
|
+
"SessionBank SSD writer received a deferred payload "
|
|
858
|
+
"entry_id=%s; writer-side encode is retired (#169), skipping",
|
|
859
|
+
pending.entry_id,
|
|
834
860
|
)
|
|
861
|
+
return False
|
|
835
862
|
with self._base_lock:
|
|
836
863
|
self._ensure_store()
|
|
837
864
|
entry_hash_prefix = pending.entry_id[:2]
|
|
@@ -889,6 +889,17 @@ class VllmMetalPagedKVCache:
|
|
|
889
889
|
int((self.num_blocks * 3 + 1) // 2),
|
|
890
890
|
int(self.num_blocks) + 1,
|
|
891
891
|
)
|
|
892
|
+
window_tokens = _env_int("MTPLX_CONTEXT_WINDOW_TOKENS", 0)
|
|
893
|
+
if window_tokens > 0:
|
|
894
|
+
# Geometric growth must not overshoot the serving context window
|
|
895
|
+
# (#150: the 1.5x step at 100k+ ctx allocates GiBs of blocks no
|
|
896
|
+
# request can ever address). A genuinely larger requirement still
|
|
897
|
+
# wins — correctness over the clamp.
|
|
898
|
+
window_blocks = (int(window_tokens) + self.block_size - 1) // self.block_size
|
|
899
|
+
if window_blocks >= required_blocks:
|
|
900
|
+
grown_blocks = min(
|
|
901
|
+
grown_blocks, max(window_blocks, int(self.num_blocks))
|
|
902
|
+
)
|
|
892
903
|
if grown_blocks <= self.num_blocks:
|
|
893
904
|
return True
|
|
894
905
|
if self.key_cache is None or self.value_cache is None:
|
|
@@ -1437,6 +1448,14 @@ class VllmMetalPagedKVCache:
|
|
|
1437
1448
|
outputs: list[Any] = []
|
|
1438
1449
|
very_negative = mx.array(-1.0e30, dtype=mx.float32)
|
|
1439
1450
|
eps = mx.array(1.0e-20, dtype=mx.float32)
|
|
1451
|
+
# kv-quant stores values packed; _paged_range dequantizes them back to
|
|
1452
|
+
# the logical head dim recorded in _shape, so the accumulator must be
|
|
1453
|
+
# sized to the dequantized width, not the packed storage width (#150,
|
|
1454
|
+
# q4 crash on the paged split-SDPA path).
|
|
1455
|
+
if self.kv_quant and self._shape is not None:
|
|
1456
|
+
value_dim = int(self._shape[2])
|
|
1457
|
+
else:
|
|
1458
|
+
value_dim = int(self.value_cache.shape[3])
|
|
1440
1459
|
|
|
1441
1460
|
for q_start in range(0, q_len, q_chunk_size):
|
|
1442
1461
|
q_end = min(q_len, q_start + q_chunk_size)
|
|
@@ -1454,7 +1473,7 @@ class VllmMetalPagedKVCache:
|
|
|
1454
1473
|
int(q.shape[0]),
|
|
1455
1474
|
int(q.shape[1]),
|
|
1456
1475
|
int(q.shape[2]),
|
|
1457
|
-
|
|
1476
|
+
value_dim,
|
|
1458
1477
|
),
|
|
1459
1478
|
dtype=mx.float32,
|
|
1460
1479
|
)
|
|
@@ -585,7 +585,7 @@ def _add_bridge_prompt_args(parser: argparse.ArgumentParser) -> None:
|
|
|
585
585
|
)
|
|
586
586
|
parser.add_argument(
|
|
587
587
|
"--chat-template-profile",
|
|
588
|
-
choices=["local_qwen36", "froggeric_v19", "tokenizer"],
|
|
588
|
+
choices=["local_qwen36", "froggeric_v19", "froggeric_v21_3", "tokenizer"],
|
|
589
589
|
default="local_qwen36",
|
|
590
590
|
help="Chat template profile for server/OpenCode paths.",
|
|
591
591
|
)
|
|
@@ -1924,7 +1924,7 @@ def build_parser() -> argparse.ArgumentParser:
|
|
|
1924
1924
|
"--profile",
|
|
1925
1925
|
type=_profile_arg, metavar=_PROFILE_METAVAR,
|
|
1926
1926
|
default=DEFAULT_PROFILE_NAME,
|
|
1927
|
-
help="Runtime profile
|
|
1927
|
+
help="Runtime profile. Default resolves per model: Turbo for the quantized 27B and 9B flagships (the app's launch rule), Sustained otherwise. An explicit value always wins. Use --profile performance-cold --max for Burst.",
|
|
1928
1928
|
)
|
|
1929
1929
|
start_flow_p.add_argument("--download", action="store_true", help="Download the selected/default model if it is missing")
|
|
1930
1930
|
start_flow_p.add_argument("--yes", action="store_true", help="Use defaults without interactive model prompts")
|
|
@@ -2106,7 +2106,7 @@ def build_parser() -> argparse.ArgumentParser:
|
|
|
2106
2106
|
"--profile",
|
|
2107
2107
|
type=_profile_arg, metavar=_PROFILE_METAVAR,
|
|
2108
2108
|
default=DEFAULT_PROFILE_NAME,
|
|
2109
|
-
help="Runtime profile.
|
|
2109
|
+
help="Runtime profile. Default resolves per model (Turbo for the quantized 27B and 9B flagships, Sustained otherwise); use --profile performance-cold --max for Burst.",
|
|
2110
2110
|
)
|
|
2111
2111
|
quickstart_server_p.add_argument("--unsafe-force-unverified", action="store_true")
|
|
2112
2112
|
quickstart_server_p.add_argument("--yes", action="store_true", help="Confirm unsafe non-interactive actions")
|
|
@@ -2265,6 +2265,7 @@ def build_parser() -> argparse.ArgumentParser:
|
|
|
2265
2265
|
tune_p.add_argument("--mtp-cache-policy", choices=["persistent", "fresh"], default="persistent", help=argparse.SUPPRESS)
|
|
2266
2266
|
tune_p.add_argument("--mtp-history-policy", choices=["auto", "committed", "full", "last-window", "last_window", "cycle", "none"], default="committed", help=argparse.SUPPRESS)
|
|
2267
2267
|
tune_p.add_argument("--draft-temperature", type=float, help=argparse.SUPPRESS)
|
|
2268
|
+
tune_p.add_argument("--draft-core", choices=["stock", "device-d2", "device"], default="stock", help=argparse.SUPPRESS)
|
|
2268
2269
|
tune_p.add_argument("--draft-top-p", type=float, help=argparse.SUPPRESS)
|
|
2269
2270
|
tune_p.add_argument("--draft-top-k", type=int, help=argparse.SUPPRESS)
|
|
2270
2271
|
tune_p.add_argument("--prompt-suite", help=argparse.SUPPRESS)
|
|
@@ -2330,6 +2331,13 @@ def build_parser() -> argparse.ArgumentParser:
|
|
|
2330
2331
|
forge_build_p.add_argument("--max", action="store_true", help="Opt into max-fan verification")
|
|
2331
2332
|
forge_build_p.add_argument("--max-tokens", type=int, default=2048, help="Verification response budget")
|
|
2332
2333
|
forge_build_p.add_argument("--suite", help="Verification prompt suite")
|
|
2334
|
+
forge_build_p.add_argument(
|
|
2335
|
+
"--dtype",
|
|
2336
|
+
choices=["auto", "bf16", "fp16"],
|
|
2337
|
+
help="Dtype for non-quantized parameters (overrides recipe body_dtype). "
|
|
2338
|
+
"fp16 prompt-processes faster on M1/M2 Macs, which have no native "
|
|
2339
|
+
"BF16; auto picks fp16 on those chips",
|
|
2340
|
+
)
|
|
2333
2341
|
forge_build_p.add_argument(
|
|
2334
2342
|
"--allow-degraded-mtp",
|
|
2335
2343
|
action="store_true",
|
|
@@ -2503,8 +2511,9 @@ def build_parser() -> argparse.ArgumentParser:
|
|
|
2503
2511
|
type=_profile_arg, metavar=_PROFILE_METAVAR,
|
|
2504
2512
|
default=DEFAULT_PROFILE_NAME,
|
|
2505
2513
|
help=(
|
|
2506
|
-
"Runtime profile.
|
|
2507
|
-
"
|
|
2514
|
+
"Runtime profile. Default resolves per model: Turbo for the "
|
|
2515
|
+
"quantized 27B and 9B flagships (the app's launch rule), "
|
|
2516
|
+
"Sustained otherwise; use --profile performance-cold "
|
|
2508
2517
|
"--max for Burst."
|
|
2509
2518
|
),
|
|
2510
2519
|
)
|
|
@@ -3461,7 +3470,7 @@ def build_parser() -> argparse.ArgumentParser:
|
|
|
3461
3470
|
)
|
|
3462
3471
|
depth_p.add_argument(
|
|
3463
3472
|
"--draft-core",
|
|
3464
|
-
choices=["stock", "device-d2"],
|
|
3473
|
+
choices=["stock", "device-d2", "device"],
|
|
3465
3474
|
default="stock",
|
|
3466
3475
|
help=(
|
|
3467
3476
|
"Experimental DraftCore backend. device-d2 compiles the greedy D2 "
|