onecomp 1.2.2__tar.gz → 1.3.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {onecomp-1.2.2/onecomp.egg-info → onecomp-1.3.0}/PKG-INFO +64 -2
- {onecomp-1.2.2 → onecomp-1.3.0}/README.md +55 -1
- onecomp-1.3.0/envs/vllm/v0_24_0_rocm/src/onecomp_vllm_v0_24_0_rocm/__init__.py +12 -0
- onecomp-1.3.0/envs/vllm/v0_24_0_rocm/src/onecomp_vllm_v0_24_0_rocm/patch.py +238 -0
- onecomp-1.3.0/example/cpu_inference/example_gptq_gguf_cpu.py +57 -0
- onecomp-1.3.0/example/cpu_inference/example_mixed_gptq_gguf_cpu.py +63 -0
- onecomp-1.3.0/example/cpu_inference/example_serve_cpu.py +43 -0
- onecomp-1.3.0/example/example_mdbf.py +57 -0
- onecomp-1.3.0/example/post_process/example_blockwise_global_ptq.py +167 -0
- onecomp-1.3.0/example/post_process/example_blockwise_global_ptq_staged.py +212 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/example/post_process/example_blockwise_ptq.py +58 -49
- onecomp-1.3.0/example/post_process/example_global_ptq.py +102 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/example/post_process/example_global_ptq_dbf.py +22 -5
- {onecomp-1.2.2 → onecomp-1.3.0}/example/post_process/example_global_ptq_distributed.py +20 -4
- onecomp-1.3.0/example/post_process/example_lora_gptq_vllm_inference.py +146 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/example/post_process/example_lora_sft.py +15 -23
- {onecomp-1.2.2 → onecomp-1.3.0}/example/post_process/example_lora_sft_knowledge.py +52 -11
- onecomp-1.3.0/example/post_process/example_lora_sft_knowledge_jointq.py +184 -0
- onecomp-1.3.0/example/post_process/example_reload_post_process_resave.py +130 -0
- onecomp-1.3.0/example/vllm_inference/example_dbf_vllm_inference.py +124 -0
- onecomp-1.3.0/example/vllm_inference/example_gptq_vllm_gptoss_inference.py +100 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/example/vllm_inference/example_gptq_vllm_inference.py +3 -1
- onecomp-1.3.0/example/vllm_inference/example_gptq_vllm_qwen36_inference.py +114 -0
- onecomp-1.3.0/llamacpp_plugins/gptq/__init__.py +27 -0
- onecomp-1.3.0/llamacpp_plugins/gptq/constants.py +66 -0
- onecomp-1.3.0/llamacpp_plugins/gptq/llamacpp_plugin.py +280 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/__init__.py +1 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/__version__.py +1 -1
- onecomp-1.3.0/onecomp/cli.py +148 -0
- onecomp-1.3.0/onecomp/cpu/__init__.py +48 -0
- onecomp-1.3.0/onecomp/cpu/cli.py +274 -0
- onecomp-1.3.0/onecomp/cpu/eval/__init__.py +42 -0
- onecomp-1.3.0/onecomp/cpu/eval/benchmark.py +118 -0
- onecomp-1.3.0/onecomp/cpu/eval/inspect_gguf.py +160 -0
- onecomp-1.3.0/onecomp/cpu/eval/parity.py +162 -0
- onecomp-1.3.0/onecomp/cpu/eval/perplexity.py +141 -0
- onecomp-1.3.0/onecomp/cpu/export/__init__.py +62 -0
- onecomp-1.3.0/onecomp/cpu/export/auto.py +162 -0
- onecomp-1.3.0/onecomp/cpu/export/blocks.py +210 -0
- onecomp-1.3.0/onecomp/cpu/export/checkpoint.py +254 -0
- onecomp-1.3.0/onecomp/cpu/export/dequantize.py +166 -0
- onecomp-1.3.0/onecomp/cpu/export/direct.py +150 -0
- onecomp-1.3.0/onecomp/cpu/export/fallback.py +63 -0
- onecomp-1.3.0/onecomp/cpu/export/rotation.py +65 -0
- onecomp-1.3.0/onecomp/cpu/export/skeleton.py +260 -0
- onecomp-1.3.0/onecomp/cpu/inference.py +133 -0
- onecomp-1.3.0/onecomp/cpu/llama_tooling.py +211 -0
- onecomp-1.3.0/onecomp/cpu/serve.py +435 -0
- onecomp-1.3.0/onecomp/export/__init__.py +37 -0
- onecomp-1.3.0/onecomp/export/gguf_export.py +665 -0
- onecomp-1.3.0/onecomp/export/gguf_reader.py +213 -0
- onecomp-1.3.0/onecomp/export/gguf_writer.py +284 -0
- onecomp-1.3.0/onecomp/export/hub.py +86 -0
- onecomp-1.3.0/onecomp/export/model_card.py +107 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/lpcd/_lpcd_runner.py +3 -1
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/lpcd/_refiner.py +8 -2
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/model_config.py +24 -0
- onecomp-1.3.0/onecomp/post_process/_base.py +192 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/post_process/_blockwise/dbf_block_optimizer.py +5 -5
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/post_process/_blockwise/dbf_cbq_optimizer.py +3 -3
- onecomp-1.3.0/onecomp/post_process/_runtime.py +137 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/post_process/blockwise_ptq.py +33 -1
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/post_process/global_ptq.py +37 -2
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/post_process/global_ptq_distributed.py +8 -5
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/post_process/post_process_lora_sft.py +78 -4
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/pre_process/rotation_utils.py +87 -3
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/qep/_quantize_with_qep_arch.py +121 -28
- onecomp-1.3.0/onecomp/quantized_model_loader.py +1396 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/__init__.py +1 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/_quantizer.py +51 -9
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/autobit/_autobit.py +53 -6
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/autobit/dbf_fallback.py +9 -1
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/dbf/_dbf.py +139 -7
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/dbf/dbf_layer.py +100 -9
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/gptq/_gptq.py +139 -12
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/gptq/config.py +6 -5
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/gptq/gptq_layer.py +219 -23
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/jointq/_jointq.py +15 -3
- onecomp-1.3.0/onecomp/quantizer/mdbf/__init__.py +12 -0
- onecomp-1.3.0/onecomp/quantizer/mdbf/_mdbf.py +607 -0
- onecomp-1.3.0/onecomp/quantizer/mdbf/admm.py +1081 -0
- onecomp-1.3.0/onecomp/quantizer/mdbf/config.py +108 -0
- onecomp-1.3.0/onecomp/quantizer/mdbf/gradient_refine.py +324 -0
- onecomp-1.3.0/onecomp/quantizer/mdbf/initialize.py +428 -0
- onecomp-1.3.0/onecomp/quantizer/mdbf/mdbf_impl.py +295 -0
- onecomp-1.3.0/onecomp/quantizer/mdbf/mdbf_layer.py +601 -0
- onecomp-1.3.0/onecomp/quantizer/mdbf/utils.py +251 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/onebit/onebit_impl.py +6 -12
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/quip/_quip.py +1 -1
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/runner.py +990 -68
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/runner_methods/chunked_quantization.py +4 -1
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/runner_methods/multi_gpu_quantization.py +8 -3
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/utils/__init__.py +1 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/utils/blockwise.py +70 -13
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/utils/device.py +15 -0
- onecomp-1.3.0/onecomp/utils/lora.py +9 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/utils/model_inputs.py +2 -1
- onecomp-1.3.0/onecomp/utils/mxfp4_compat.py +73 -0
- onecomp-1.3.0/onecomp/utils/quant_config.py +85 -0
- onecomp-1.3.0/onecomp/utils/unfuse_moe.py +724 -0
- {onecomp-1.2.2 → onecomp-1.3.0/onecomp.egg-info}/PKG-INFO +64 -2
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp.egg-info/SOURCES.txt +63 -1
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp.egg-info/entry_points.txt +1 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp.egg-info/requires.txt +10 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp.egg-info/top_level.txt +3 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/pyproject.toml +14 -1
- {onecomp-1.2.2 → onecomp-1.3.0}/vllm_plugins/dbf/vllm_plugin.py +105 -25
- onecomp-1.3.0/vllm_plugins/gptq/gptoss_wna16_moe.py +198 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/vllm_plugins/gptq/vllm_plugin.py +151 -18
- onecomp-1.3.0/vllm_plugins/patches/__init__.py +5 -0
- onecomp-1.3.0/vllm_plugins/patches/_paths.py +19 -0
- onecomp-1.3.0/vllm_plugins/patches/apply_all.py +87 -0
- onecomp-1.3.0/vllm_plugins/patches/gpt_oss_gptq_moe.py +184 -0
- onecomp-1.3.0/vllm_plugins/patches/gpt_oss_wna16_bias.py +313 -0
- onecomp-1.3.0/vllm_plugins/utils/__init__.py +5 -0
- onecomp-1.3.0/vllm_plugins/utils/module.py +177 -0
- onecomp-1.3.0/vllm_plugins/utils/rotation.py +261 -0
- onecomp-1.2.2/example/post_process/example_global_ptq.py +0 -73
- onecomp-1.2.2/onecomp/cli.py +0 -89
- onecomp-1.2.2/onecomp/post_process/_base.py +0 -79
- onecomp-1.2.2/onecomp/quantized_model_loader.py +0 -624
- onecomp-1.2.2/onecomp/utils/quant_config.py +0 -28
- onecomp-1.2.2/onecomp/utils/unfuse_moe.py +0 -160
- onecomp-1.2.2/vllm_plugins/utils/module.py +0 -94
- {onecomp-1.2.2 → onecomp-1.3.0}/LICENSE +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/benchmark/llama3-8b-gptq/quant_benchmark.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/benchmark/llama3-8b-jointq/quant_benchmark.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/benchmark/llama3-8b-lpcd-gptq/quant_benchmark.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/benchmark/llama3-8b-qep-gptq/quant_benchmark.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/benchmark/llama3-8b-various/quant_benchmark.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/benchmark/qwen3-14b-gptq/quant_benchmark.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/benchmark/qwen3-14b-jointq/quant_benchmark.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/benchmark/qwen3-8b-gptq/quant_benchmark.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/benchmark/qwen3-8b-jointq/quant_benchmark.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/dashboard/backend/app/__init__.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/dashboard/backend/app/api/__init__.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/dashboard/backend/app/api/jobs.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/dashboard/backend/app/constants.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/dashboard/backend/app/core/__init__.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/dashboard/backend/app/core/config.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/dashboard/backend/app/core/database.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/dashboard/backend/app/main.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/dashboard/backend/app/models/__init__.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/dashboard/backend/app/models/job.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/dashboard/backend/app/schemas/__init__.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/dashboard/backend/app/schemas/job.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/dashboard/backend/app/services/__init__.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/dashboard/backend/app/services/huggingface.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/dashboard/backend/app/services/inference.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/dashboard/backend/app/services/job_store.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/dashboard/backend/app/worker/__init__.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/dashboard/backend/app/worker/celery_app.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/dashboard/backend/app/worker/tasks.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/dashboard/backend/cpu_patch.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/dashboard/backend/start_backend.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/dashboard/backend/start_worker.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/example/example_auto_run.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/example/example_autobit.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/example/example_custom_calibration.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/example/example_gptq.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/example/example_jointq.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/example/example_lpcd_gptq.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/example/example_qep_gptq.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/example/example_save_load.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/example/pre_process/example_llama_preprocess_rtn.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/example/pre_process/example_preprocess_save_load.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/example/vllm_inference/example_autobit_vllm_inference.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/example/vllm_inference/example_jointq_vllm_inference.py +0 -0
- {onecomp-1.2.2/vllm_plugins → onecomp-1.3.0/llamacpp_plugins}/__init__.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/model_validation/autobit/validate_autobit.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/model_validation/autobit_qep/validate_autobit.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/model_validation/gptq/validate_gptq.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/model_validation/gptq/validate_load.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/model_validation/gptq/validate_vllm.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/model_validation/jointq/validate_jointq.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/model_validation/qep_gptq/validate_gptq.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/__main__.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/analyzer/__init__.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/analyzer/cumulative_error.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/analyzer/quantization_error.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/analyzer/weight_outlier.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/calibration/__init__.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/calibration/_cache.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/calibration/c4.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/calibration/calibration_config.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/calibration/calibration_data_loader.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/calibration/chunking.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/calibration/custom.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/calibration/wikitext.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/eval/__init__.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/eval/__main__.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/eval/conf/__init__.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/eval/conf/eval_config.yaml +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/eval/evals/__init__.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/eval/evals/base.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/eval/evals/mt_bench/__init__.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/eval/evals/mt_bench/adapter.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/eval/evals/mt_bench/data.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/eval/evals/mt_bench/gen_answer.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/eval/evals/mt_bench/judge.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/eval/evals/mt_bench/radar_chart.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/eval/evals/mt_bench/run.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/eval/evals/mt_bench/show_result.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/eval/evals/throughput/__init__.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/eval/evals/throughput/adapter.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/eval/evals/throughput/bench.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/eval/evals/throughput/run.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/eval/orchestrator/__init__.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/eval/orchestrator/aggregator.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/eval/orchestrator/runner.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/eval/orchestrator/server.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/eval/orchestrator/subprocess_runner.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/eval/run_evaluate.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/eval/schema.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/eval/utils/__init__.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/eval/utils/model_utils.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/eval/utils/ports.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/eval/utils/resources.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/eval/utils/secrets.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/log.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/lpcd/__init__.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/lpcd/_gradient_solver.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/lpcd/_lpcd_config.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/lpcd/_metric.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/lpcd/arch/_llama.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/lpcd/arch/_llama_cf.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/lpcd/arch/_qwen3.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/post_process/__init__.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/post_process/_blockwise/__init__.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/post_process/_blockwise/generic_block_optimizer.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/post_process/_blockwise/gptq_block_optimizer.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/post_process/_blockwise/gptq_cbq_optimizer.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/post_process/_blockwise/helpers.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/post_process/_blockwise/onebit_block_optimizer.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/post_process/_blockwise/onebit_cbq_optimizer.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/post_process/_global_ptq/__init__.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/post_process/_global_ptq/core.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/post_process/_global_ptq/dbf_adapter.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/post_process/_global_ptq/gptq_adapter.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/post_process/_global_ptq/helpers.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/post_process/_global_ptq/losses.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/post_process/_global_ptq/trainer.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/pre_process/__init__.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/pre_process/hadamard_utils.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/pre_process/modeling_llama.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/pre_process/modeling_qwen3.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/pre_process/optimizer.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/pre_process/prepare_rotated_model.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/pre_process/preprocess_args.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/pre_process/quant_models.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/pre_process/train_rotation.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/qep/__init__.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/qep/_qep_config.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/qep/_quantize_with_qep.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/arb/__init__.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/arb/_arb.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/arb/arb_impl.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/autobit/__init__.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/autobit/activation_stats.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/autobit/ilp.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/autobit/manual.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/autobit/visualize.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/cq/__init__.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/cq/_cq.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/cq/cq_impl.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/dbf/__init__.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/dbf/admm_extended.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/dbf/balance.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/dbf/config.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/dbf/dbf_impl.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/dbf/dbf_original.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/dbf/fine_tune.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/dbf/middle.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/gemlite.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/gptq/__init__.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/jointq/__init__.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/jointq/core/__init__.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/jointq/core/__version__.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/jointq/core/clip.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/jointq/core/error_propagation/__init__.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/jointq/core/error_propagation/local_search_advanced.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/jointq/core/error_propagation/quantize_advanced.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/jointq/core/error_propagation/quantizer_advanced.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/jointq/core/local_search.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/jointq/core/quantize.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/jointq/core/quantize_multi_gpu.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/jointq/core/quantizer.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/jointq/core/solution.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/onebit/__init__.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/onebit/_onebit.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/onebit/onebit_layer.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/qbb/__init__.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/qbb/_qbb.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/qbb/qbb_impl.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/quip/__init__.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/quip/quant_quip.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/quip/quip_impl.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/quip/utils.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/quip/utils_had.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/quip/vector_balance.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/rtn/__init__.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/rtn/_rtn.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/rtn/quantizer.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/rtn/rtn_impl.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/rotated_model_config.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/runner_methods/__init__.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/runner_methods/jointq_error_propagation.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/utils/accuracy.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/utils/activation_capture.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/utils/activation_check.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/utils/dtype.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/utils/perplexity.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/utils/quantization_progress.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/utils/vram_estimator.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/onecomp.egg-info/dependency_links.txt +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/scripts/check_copyright_header.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/scripts/check_no_japanese.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/setup.cfg +0 -0
- {onecomp-1.2.2/vllm_plugins/dbf/modules → onecomp-1.3.0/vllm_plugins}/__init__.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/vllm_plugins/dbf/__init__.py +0 -0
- {onecomp-1.2.2/vllm_plugins/utils → onecomp-1.3.0/vllm_plugins/dbf/modules}/__init__.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/vllm_plugins/dbf/modules/gemlite_linear.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/vllm_plugins/dbf/modules/naive.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/vllm_plugins/gptq/__init__.py +0 -0
- {onecomp-1.2.2 → onecomp-1.3.0}/vllm_plugins/gptq/constants.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: onecomp
|
|
3
|
-
Version: 1.
|
|
3
|
+
Version: 1.3.0
|
|
4
4
|
Summary: Python package for LLM compression
|
|
5
5
|
Author: Keiji Kimura
|
|
6
6
|
License: MIT License
|
|
@@ -89,6 +89,14 @@ Provides-Extra: hydra
|
|
|
89
89
|
Requires-Dist: hydra-core; extra == "hydra"
|
|
90
90
|
Provides-Extra: vllm
|
|
91
91
|
Requires-Dist: vllm<0.22,>=0.10; extra == "vllm"
|
|
92
|
+
Requires-Dist: conch-triton-kernels; extra == "vllm"
|
|
93
|
+
Provides-Extra: gguf
|
|
94
|
+
Requires-Dist: gguf>=0.10; extra == "gguf"
|
|
95
|
+
Provides-Extra: llamacpp
|
|
96
|
+
Requires-Dist: gguf>=0.10; extra == "llamacpp"
|
|
97
|
+
Requires-Dist: llama-cpp-python>=0.3.0; extra == "llamacpp"
|
|
98
|
+
Requires-Dist: sentencepiece; extra == "llamacpp"
|
|
99
|
+
Requires-Dist: protobuf; extra == "llamacpp"
|
|
92
100
|
Provides-Extra: eval
|
|
93
101
|
Requires-Dist: hydra-core>=1.3; extra == "eval"
|
|
94
102
|
Requires-Dist: omegaconf>=2.3; extra == "eval"
|
|
@@ -143,10 +151,12 @@ Full documentation is available at **[https://FujitsuResearch.github.io/OneCompr
|
|
|
143
151
|
- **vLLM Plugin Integration**: Serve OneComp-quantized models with [vLLM](https://docs.vllm.ai/) via built-in plugins for DBF and Mixed-GPTQ quantization methods. Pair with [Open WebUI](https://github.com/open-webui/open-webui) for a ChatGPT-like chat experience on your local machine.
|
|
144
152
|
- **AutoBit**: Mixed-precision quantization with ILP-based bitwidth assignment. Automatically estimates the target bitwidth from available VRAM and assigns per-layer bitwidths to minimize quantization error under the memory budget.
|
|
145
153
|
- **JointQ**: Joint quantization method that optimizes weight assignments and scale parameters simultaneously for improved quantization accuracy. Supports group-wise quantization (e.g., 4-bit, groupsize=128).
|
|
154
|
+
- **MDBF (Multi-Envelope Double Binary Factorization)**: A binary factorization quantizer that approximates each weight matrix as a sum of multi-path sign matrices with multi-scale FP16 envelopes, generalizing DBF and LittleBit for aggressive low-bit (sub-1-bit) compression. Supports ADMM/gradient refinement, activation-aware initialization, and a GemLite-accelerated 1-bit inference path. See the [MDBF guide](https://FujitsuResearch.github.io/OneCompression/algorithms/mdbf/) for details.
|
|
146
155
|
- **Block-wise PTQ**: Post-quantization block-wise distillation that minimises intermediate-representation MSE against an FP16 teacher model at Transformer-block granularity. Includes Phase 1 (greedy per-block optimisation) and Phase 2 CBQ (cross-block sliding-window optimisation). Supports GPTQ, DBF, and OneBit quantizers.
|
|
147
156
|
- **LoRA SFT Post-Process**: Fine-tune quantized models with LoRA adapters for accuracy recovery or domain-specific knowledge injection. Supports SFT loss, teacher distillation, and intermediate block alignment.
|
|
148
157
|
- **Rotation Preprocessing**: SpinQuant/OstQuant-based rotation preprocessing that reduces quantization error by learning optimal rotation matrices before quantization. Rotation/scaling matrices are absorbed into model weights, with online Hadamard hooks automatically registered at load time. Supports Llama and Qwen3 architectures.
|
|
149
158
|
- **Web Dashboard (HPC)**: A browser-based dashboard for launching quantization jobs, deploying models, and validating chat-based inference in HPC environments. See [dashboard/README.md](dashboard/README.md) for details.
|
|
159
|
+
- **GGUF Export & Hugging Face Hub Integration**: Convert models to the GGUF v3 format (F16) for llama.cpp/Ollama with a dependency-free built-in writer, generate model cards with quantization recipes and evaluation results, and push save directories to the Hugging Face Hub. Supports Llama (SentencePiece or Llama-3-style BPE) and Qwen2 (BPE) architectures, including multi-EOS stop-token mapping. See the [GGUF Export guide](https://FujitsuResearch.github.io/OneCompression/user-guide/gguf-export/).
|
|
150
160
|
- (TBD)
|
|
151
161
|
|
|
152
162
|
## 🤖 Supported Models
|
|
@@ -159,6 +169,8 @@ Other Hugging Face-compatible models may work but are currently untested.
|
|
|
159
169
|
| 1 | Llama | TinyLlama, Llama-2, Llama-3 | ✅ Verified |
|
|
160
170
|
| 2 | Qwen3 | Qwen3-0.6B ~ 32B | ✅ Verified |
|
|
161
171
|
| 3 | Gemma | Gemma 2, Gemma 3, Gemma 4 | ✅ Verified |
|
|
172
|
+
| 4 | [GPT-OSS](docs/user-guide/gptoss.md) | openai/gpt-oss-20b, gpt-oss-120b | ✅ Verified |
|
|
173
|
+
| 5 | Qwen3.6 | Qwen3.6-27B, Qwen3.6-35B-A3B | ✅ Verified |
|
|
162
174
|
|
|
163
175
|
|
|
164
176
|
> **Note:** Support for additional architectures is planned. Contributions and test reports are welcome.
|
|
@@ -379,10 +391,19 @@ uv run mkdocs serve
|
|
|
379
391
|
|
|
380
392
|
Then open [http://127.0.0.1:8000](http://127.0.0.1:8000) in your browser.
|
|
381
393
|
|
|
394
|
+
## 📓 Tutorial Notebook
|
|
395
|
+
|
|
396
|
+
Interactive walkthrough in Jupyter or [Google Colab](https://colab.research.google.com/github/FujitsuResearch/OneCompression/blob/main/notebook/01_tutorial.ipynb)
|
|
397
|
+
— RTN visualization, `Runner.auto_run`, and vLLM chat inference.
|
|
398
|
+
|
|
399
|
+
See [`notebook/README.md`](./notebook/README.md) for local setup, or the
|
|
400
|
+
[Tutorial Notebook guide](https://FujitsuResearch.github.io/OneCompression/getting-started/tutorial-notebook/) in the docs.
|
|
401
|
+
|
|
382
402
|
## 🚀 Examples
|
|
383
403
|
|
|
384
404
|
| Category | Script | Description |
|
|
385
405
|
|----------|--------|-------------|
|
|
406
|
+
| Tutorial | [01_tutorial.ipynb](./notebook/01_tutorial.ipynb) | Interactive notebook (Jupyter / Colab) |
|
|
386
407
|
| Quantization | [example_gptq.py](./example/example_gptq.py) | GPTQ quantization |
|
|
387
408
|
| | [example_qep_gptq.py](./example/example_qep_gptq.py) | GPTQ + QEP (error propagation) |
|
|
388
409
|
| | [example_lpcd_gptq.py](./example/example_lpcd_gptq.py) | GPTQ + QEP + LPCD quantization |
|
|
@@ -391,14 +412,25 @@ Then open [http://127.0.0.1:8000](http://127.0.0.1:8000) in your browser.
|
|
|
391
412
|
| | [example_auto_run.py](./example/example_auto_run.py) | AutoBit with automatic VRAM estimation |
|
|
392
413
|
| Calibration | [example_custom_calibration.py](./example/example_custom_calibration.py) | Custom calibration dataset with CalibrationConfig |
|
|
393
414
|
| Save / Load | [example_save_load.py](./example/example_save_load.py) | Save and load quantized models |
|
|
415
|
+
| | [example_reload_post_process_resave.py](./example/post_process/example_reload_post_process_resave.py) | Reload, post-process, and re-save a quantized checkpoint |
|
|
394
416
|
| Rotation Preprocessing | [example_llama_preprocess_rtn.py](./example/pre_process/example_llama_preprocess_rtn.py) | Rotation preprocessing + RTN (TinyLlama) |
|
|
395
417
|
| | [example_preprocess_save_load.py](./example/pre_process/example_preprocess_save_load.py) | Save and load rotation-preprocessed quantized models |
|
|
396
|
-
| Post-Process | [example_blockwise_ptq.py](./example/post_process/example_blockwise_ptq.py) | Block-wise PTQ (
|
|
418
|
+
| Post-Process | [example_blockwise_ptq.py](./example/post_process/example_blockwise_ptq.py) | Block-wise PTQ via `Runner.run()` with packed buffers by default |
|
|
419
|
+
| | [example_blockwise_global_ptq.py](./example/post_process/example_blockwise_global_ptq.py) | BlockWisePTQ → GlobalPTQ in a single Runner, followed by safetensors save/load |
|
|
420
|
+
| | [example_blockwise_global_ptq_staged.py](./example/post_process/example_blockwise_global_ptq_staged.py) | Staged BlockWisePTQ → GlobalPTQ across save/load boundaries with accumulated post-process metadata |
|
|
421
|
+
| | [example_global_ptq.py](./example/post_process/example_global_ptq.py) | Global PTQ with packed buffers by default and HF-compatible safetensors output |
|
|
422
|
+
| | [example_global_ptq_dbf.py](./example/post_process/example_global_ptq_dbf.py) | Global PTQ with the DBF backend and HF-compatible safetensors output |
|
|
423
|
+
| | [example_global_ptq_distributed.py](./example/post_process/example_global_ptq_distributed.py) | Multi-GPU Global PTQ with DeepSpeed / torchrun and safetensors output |
|
|
397
424
|
| | [example_lora_sft.py](./example/post_process/example_lora_sft.py) | LoRA SFT post-quantization fine-tuning |
|
|
398
425
|
| | [example_lora_sft_knowledge.py](./example/post_process/example_lora_sft_knowledge.py) | LoRA SFT knowledge injection |
|
|
426
|
+
| | [example_lora_sft_knowledge_jointq.py](./example/post_process/example_lora_sft_knowledge_jointq.py) | LoRA SFT knowledge injection on a JointQ-quantized model |
|
|
427
|
+
| | [example_lora_gptq_vllm_inference.py](./example/post_process/example_lora_gptq_vllm_inference.py) | GPTQ + LoRA SFT, HF-compatible safetensors + PEFT sidecar, and vLLM inference |
|
|
399
428
|
| vLLM | [example_gptq_vllm_inference.py](./example/vllm_inference/example_gptq_vllm_inference.py) | GPTQ + QEP quantization and vLLM inference |
|
|
429
|
+
| | [example_gptq_vllm_qwen36_inference.py](./example/vllm_inference/example_gptq_vllm_qwen36_inference.py) | GPTQ quantization and vLLM inference for Qwen3.6 (full-wrapper save) |
|
|
430
|
+
| | [example_gptq_vllm_gptoss_inference.py](./example/vllm_inference/example_gptq_vllm_gptoss_inference.py) | GPTQ quantization and vLLM inference for gpt-oss |
|
|
400
431
|
| | [example_jointq_vllm_inference.py](./example/vllm_inference/example_jointq_vllm_inference.py) | JointQ quantization and vLLM inference |
|
|
401
432
|
| | [example_autobit_vllm_inference.py](./example/vllm_inference/example_autobit_vllm_inference.py) | AutoBit quantization and vLLM inference |
|
|
433
|
+
| | [example_dbf_vllm_inference.py](./example/vllm_inference/example_dbf_vllm_inference.py) | DBF quantization and vLLM inference |
|
|
402
434
|
|
|
403
435
|
## 🔌 vLLM Inference
|
|
404
436
|
|
|
@@ -415,6 +447,22 @@ pip install vllm
|
|
|
415
447
|
|
|
416
448
|
See the [vLLM Inference guide](https://FujitsuResearch.github.io/OneCompression/user-guide/vllm-inference/) for details, including Open WebUI setup instructions.
|
|
417
449
|
|
|
450
|
+
### GPT-OSS (mixed_gptq)
|
|
451
|
+
|
|
452
|
+
[GPT-OSS models](docs/user-guide/gptoss.md) (`openai/gpt-oss-20b`, `openai/gpt-oss-120b`) need extra steps beyond the
|
|
453
|
+
standard vLLM plugin flow:
|
|
454
|
+
|
|
455
|
+
1. Quantize with **`GPTQ(wbits=4)`** and keep the MoE experts 4-bit via `Runner(..., moe_quant_experts=True)` with `groupsize=64`. GPT-OSS `hidden_size` (2880) is not divisible by 128, so the experts must use **`group_size=64`** for vLLM's WNA16 MoE kernel.
|
|
456
|
+
2. Before `LLM(...)`, apply vLLM runtime patches: `python -m vllm_plugins.patches.apply_all`
|
|
457
|
+
3. Set `export VLLM_USE_DEEP_GEMM=0`
|
|
458
|
+
|
|
459
|
+
The `--extra vllm` dependency set includes **`conch-triton-kernels`**. On NVIDIA Blackwell (B200, sm100),
|
|
460
|
+
vLLM 0.20 needs Conch for `mixed_gptq` linear layers when the Marlin kernel cannot handle GPT-OSS weight shapes.
|
|
461
|
+
If you install vLLM manually with `pip install vllm`, also run `pip install conch-triton-kernels`.
|
|
462
|
+
|
|
463
|
+
See the [GPT-OSS guide](docs/user-guide/gptoss.md) for HF save/load, patch details, and verification scripts.
|
|
464
|
+
|
|
465
|
+
|
|
418
466
|
|
|
419
467
|
## 📬 Contact Us
|
|
420
468
|
|
|
@@ -466,3 +514,17 @@ year={2025},
|
|
|
466
514
|
url={https://arxiv.org/abs/2512.01546}
|
|
467
515
|
}
|
|
468
516
|
```
|
|
517
|
+
|
|
518
|
+
MDBF (Multi-Envelope Double Binary Factorization):
|
|
519
|
+
|
|
520
|
+
```
|
|
521
|
+
@misc{ichikawa2025bitsmultienvelopedoublebinary,
|
|
522
|
+
title={More Than Bits: Multi-Envelope Double Binary Factorization for Extreme Quantization},
|
|
523
|
+
author={Yuma Ichikawa and Yoshihiko Fujisawa and Yudai Fujimoto and Akira Sakai and Katsuki Fujisawa},
|
|
524
|
+
year={2025},
|
|
525
|
+
eprint={2512.24545},
|
|
526
|
+
archivePrefix={arXiv},
|
|
527
|
+
primaryClass={cs.LG},
|
|
528
|
+
url={https://arxiv.org/abs/2512.24545},
|
|
529
|
+
}
|
|
530
|
+
```
|
|
@@ -38,10 +38,12 @@ Full documentation is available at **[https://FujitsuResearch.github.io/OneCompr
|
|
|
38
38
|
- **vLLM Plugin Integration**: Serve OneComp-quantized models with [vLLM](https://docs.vllm.ai/) via built-in plugins for DBF and Mixed-GPTQ quantization methods. Pair with [Open WebUI](https://github.com/open-webui/open-webui) for a ChatGPT-like chat experience on your local machine.
|
|
39
39
|
- **AutoBit**: Mixed-precision quantization with ILP-based bitwidth assignment. Automatically estimates the target bitwidth from available VRAM and assigns per-layer bitwidths to minimize quantization error under the memory budget.
|
|
40
40
|
- **JointQ**: Joint quantization method that optimizes weight assignments and scale parameters simultaneously for improved quantization accuracy. Supports group-wise quantization (e.g., 4-bit, groupsize=128).
|
|
41
|
+
- **MDBF (Multi-Envelope Double Binary Factorization)**: A binary factorization quantizer that approximates each weight matrix as a sum of multi-path sign matrices with multi-scale FP16 envelopes, generalizing DBF and LittleBit for aggressive low-bit (sub-1-bit) compression. Supports ADMM/gradient refinement, activation-aware initialization, and a GemLite-accelerated 1-bit inference path. See the [MDBF guide](https://FujitsuResearch.github.io/OneCompression/algorithms/mdbf/) for details.
|
|
41
42
|
- **Block-wise PTQ**: Post-quantization block-wise distillation that minimises intermediate-representation MSE against an FP16 teacher model at Transformer-block granularity. Includes Phase 1 (greedy per-block optimisation) and Phase 2 CBQ (cross-block sliding-window optimisation). Supports GPTQ, DBF, and OneBit quantizers.
|
|
42
43
|
- **LoRA SFT Post-Process**: Fine-tune quantized models with LoRA adapters for accuracy recovery or domain-specific knowledge injection. Supports SFT loss, teacher distillation, and intermediate block alignment.
|
|
43
44
|
- **Rotation Preprocessing**: SpinQuant/OstQuant-based rotation preprocessing that reduces quantization error by learning optimal rotation matrices before quantization. Rotation/scaling matrices are absorbed into model weights, with online Hadamard hooks automatically registered at load time. Supports Llama and Qwen3 architectures.
|
|
44
45
|
- **Web Dashboard (HPC)**: A browser-based dashboard for launching quantization jobs, deploying models, and validating chat-based inference in HPC environments. See [dashboard/README.md](dashboard/README.md) for details.
|
|
46
|
+
- **GGUF Export & Hugging Face Hub Integration**: Convert models to the GGUF v3 format (F16) for llama.cpp/Ollama with a dependency-free built-in writer, generate model cards with quantization recipes and evaluation results, and push save directories to the Hugging Face Hub. Supports Llama (SentencePiece or Llama-3-style BPE) and Qwen2 (BPE) architectures, including multi-EOS stop-token mapping. See the [GGUF Export guide](https://FujitsuResearch.github.io/OneCompression/user-guide/gguf-export/).
|
|
45
47
|
- (TBD)
|
|
46
48
|
|
|
47
49
|
## 🤖 Supported Models
|
|
@@ -54,6 +56,8 @@ Other Hugging Face-compatible models may work but are currently untested.
|
|
|
54
56
|
| 1 | Llama | TinyLlama, Llama-2, Llama-3 | ✅ Verified |
|
|
55
57
|
| 2 | Qwen3 | Qwen3-0.6B ~ 32B | ✅ Verified |
|
|
56
58
|
| 3 | Gemma | Gemma 2, Gemma 3, Gemma 4 | ✅ Verified |
|
|
59
|
+
| 4 | [GPT-OSS](docs/user-guide/gptoss.md) | openai/gpt-oss-20b, gpt-oss-120b | ✅ Verified |
|
|
60
|
+
| 5 | Qwen3.6 | Qwen3.6-27B, Qwen3.6-35B-A3B | ✅ Verified |
|
|
57
61
|
|
|
58
62
|
|
|
59
63
|
> **Note:** Support for additional architectures is planned. Contributions and test reports are welcome.
|
|
@@ -274,10 +278,19 @@ uv run mkdocs serve
|
|
|
274
278
|
|
|
275
279
|
Then open [http://127.0.0.1:8000](http://127.0.0.1:8000) in your browser.
|
|
276
280
|
|
|
281
|
+
## 📓 Tutorial Notebook
|
|
282
|
+
|
|
283
|
+
Interactive walkthrough in Jupyter or [Google Colab](https://colab.research.google.com/github/FujitsuResearch/OneCompression/blob/main/notebook/01_tutorial.ipynb)
|
|
284
|
+
— RTN visualization, `Runner.auto_run`, and vLLM chat inference.
|
|
285
|
+
|
|
286
|
+
See [`notebook/README.md`](./notebook/README.md) for local setup, or the
|
|
287
|
+
[Tutorial Notebook guide](https://FujitsuResearch.github.io/OneCompression/getting-started/tutorial-notebook/) in the docs.
|
|
288
|
+
|
|
277
289
|
## 🚀 Examples
|
|
278
290
|
|
|
279
291
|
| Category | Script | Description |
|
|
280
292
|
|----------|--------|-------------|
|
|
293
|
+
| Tutorial | [01_tutorial.ipynb](./notebook/01_tutorial.ipynb) | Interactive notebook (Jupyter / Colab) |
|
|
281
294
|
| Quantization | [example_gptq.py](./example/example_gptq.py) | GPTQ quantization |
|
|
282
295
|
| | [example_qep_gptq.py](./example/example_qep_gptq.py) | GPTQ + QEP (error propagation) |
|
|
283
296
|
| | [example_lpcd_gptq.py](./example/example_lpcd_gptq.py) | GPTQ + QEP + LPCD quantization |
|
|
@@ -286,14 +299,25 @@ Then open [http://127.0.0.1:8000](http://127.0.0.1:8000) in your browser.
|
|
|
286
299
|
| | [example_auto_run.py](./example/example_auto_run.py) | AutoBit with automatic VRAM estimation |
|
|
287
300
|
| Calibration | [example_custom_calibration.py](./example/example_custom_calibration.py) | Custom calibration dataset with CalibrationConfig |
|
|
288
301
|
| Save / Load | [example_save_load.py](./example/example_save_load.py) | Save and load quantized models |
|
|
302
|
+
| | [example_reload_post_process_resave.py](./example/post_process/example_reload_post_process_resave.py) | Reload, post-process, and re-save a quantized checkpoint |
|
|
289
303
|
| Rotation Preprocessing | [example_llama_preprocess_rtn.py](./example/pre_process/example_llama_preprocess_rtn.py) | Rotation preprocessing + RTN (TinyLlama) |
|
|
290
304
|
| | [example_preprocess_save_load.py](./example/pre_process/example_preprocess_save_load.py) | Save and load rotation-preprocessed quantized models |
|
|
291
|
-
| Post-Process | [example_blockwise_ptq.py](./example/post_process/example_blockwise_ptq.py) | Block-wise PTQ (
|
|
305
|
+
| Post-Process | [example_blockwise_ptq.py](./example/post_process/example_blockwise_ptq.py) | Block-wise PTQ via `Runner.run()` with packed buffers by default |
|
|
306
|
+
| | [example_blockwise_global_ptq.py](./example/post_process/example_blockwise_global_ptq.py) | BlockWisePTQ → GlobalPTQ in a single Runner, followed by safetensors save/load |
|
|
307
|
+
| | [example_blockwise_global_ptq_staged.py](./example/post_process/example_blockwise_global_ptq_staged.py) | Staged BlockWisePTQ → GlobalPTQ across save/load boundaries with accumulated post-process metadata |
|
|
308
|
+
| | [example_global_ptq.py](./example/post_process/example_global_ptq.py) | Global PTQ with packed buffers by default and HF-compatible safetensors output |
|
|
309
|
+
| | [example_global_ptq_dbf.py](./example/post_process/example_global_ptq_dbf.py) | Global PTQ with the DBF backend and HF-compatible safetensors output |
|
|
310
|
+
| | [example_global_ptq_distributed.py](./example/post_process/example_global_ptq_distributed.py) | Multi-GPU Global PTQ with DeepSpeed / torchrun and safetensors output |
|
|
292
311
|
| | [example_lora_sft.py](./example/post_process/example_lora_sft.py) | LoRA SFT post-quantization fine-tuning |
|
|
293
312
|
| | [example_lora_sft_knowledge.py](./example/post_process/example_lora_sft_knowledge.py) | LoRA SFT knowledge injection |
|
|
313
|
+
| | [example_lora_sft_knowledge_jointq.py](./example/post_process/example_lora_sft_knowledge_jointq.py) | LoRA SFT knowledge injection on a JointQ-quantized model |
|
|
314
|
+
| | [example_lora_gptq_vllm_inference.py](./example/post_process/example_lora_gptq_vllm_inference.py) | GPTQ + LoRA SFT, HF-compatible safetensors + PEFT sidecar, and vLLM inference |
|
|
294
315
|
| vLLM | [example_gptq_vllm_inference.py](./example/vllm_inference/example_gptq_vllm_inference.py) | GPTQ + QEP quantization and vLLM inference |
|
|
316
|
+
| | [example_gptq_vllm_qwen36_inference.py](./example/vllm_inference/example_gptq_vllm_qwen36_inference.py) | GPTQ quantization and vLLM inference for Qwen3.6 (full-wrapper save) |
|
|
317
|
+
| | [example_gptq_vllm_gptoss_inference.py](./example/vllm_inference/example_gptq_vllm_gptoss_inference.py) | GPTQ quantization and vLLM inference for gpt-oss |
|
|
295
318
|
| | [example_jointq_vllm_inference.py](./example/vllm_inference/example_jointq_vllm_inference.py) | JointQ quantization and vLLM inference |
|
|
296
319
|
| | [example_autobit_vllm_inference.py](./example/vllm_inference/example_autobit_vllm_inference.py) | AutoBit quantization and vLLM inference |
|
|
320
|
+
| | [example_dbf_vllm_inference.py](./example/vllm_inference/example_dbf_vllm_inference.py) | DBF quantization and vLLM inference |
|
|
297
321
|
|
|
298
322
|
## 🔌 vLLM Inference
|
|
299
323
|
|
|
@@ -310,6 +334,22 @@ pip install vllm
|
|
|
310
334
|
|
|
311
335
|
See the [vLLM Inference guide](https://FujitsuResearch.github.io/OneCompression/user-guide/vllm-inference/) for details, including Open WebUI setup instructions.
|
|
312
336
|
|
|
337
|
+
### GPT-OSS (mixed_gptq)
|
|
338
|
+
|
|
339
|
+
[GPT-OSS models](docs/user-guide/gptoss.md) (`openai/gpt-oss-20b`, `openai/gpt-oss-120b`) need extra steps beyond the
|
|
340
|
+
standard vLLM plugin flow:
|
|
341
|
+
|
|
342
|
+
1. Quantize with **`GPTQ(wbits=4)`** and keep the MoE experts 4-bit via `Runner(..., moe_quant_experts=True)` with `groupsize=64`. GPT-OSS `hidden_size` (2880) is not divisible by 128, so the experts must use **`group_size=64`** for vLLM's WNA16 MoE kernel.
|
|
343
|
+
2. Before `LLM(...)`, apply vLLM runtime patches: `python -m vllm_plugins.patches.apply_all`
|
|
344
|
+
3. Set `export VLLM_USE_DEEP_GEMM=0`
|
|
345
|
+
|
|
346
|
+
The `--extra vllm` dependency set includes **`conch-triton-kernels`**. On NVIDIA Blackwell (B200, sm100),
|
|
347
|
+
vLLM 0.20 needs Conch for `mixed_gptq` linear layers when the Marlin kernel cannot handle GPT-OSS weight shapes.
|
|
348
|
+
If you install vLLM manually with `pip install vllm`, also run `pip install conch-triton-kernels`.
|
|
349
|
+
|
|
350
|
+
See the [GPT-OSS guide](docs/user-guide/gptoss.md) for HF save/load, patch details, and verification scripts.
|
|
351
|
+
|
|
352
|
+
|
|
313
353
|
|
|
314
354
|
## 📬 Contact Us
|
|
315
355
|
|
|
@@ -361,3 +401,17 @@ year={2025},
|
|
|
361
401
|
url={https://arxiv.org/abs/2512.01546}
|
|
362
402
|
}
|
|
363
403
|
```
|
|
404
|
+
|
|
405
|
+
MDBF (Multi-Envelope Double Binary Factorization):
|
|
406
|
+
|
|
407
|
+
```
|
|
408
|
+
@misc{ichikawa2025bitsmultienvelopedoublebinary,
|
|
409
|
+
title={More Than Bits: Multi-Envelope Double Binary Factorization for Extreme Quantization},
|
|
410
|
+
author={Yuma Ichikawa and Yoshihiko Fujisawa and Yudai Fujimoto and Akira Sakai and Katsuki Fujisawa},
|
|
411
|
+
year={2025},
|
|
412
|
+
eprint={2512.24545},
|
|
413
|
+
archivePrefix={arXiv},
|
|
414
|
+
primaryClass={cs.LG},
|
|
415
|
+
url={https://arxiv.org/abs/2512.24545},
|
|
416
|
+
}
|
|
417
|
+
```
|
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
"""ROCm-only workaround for vLLM 0.24.0 TritonW4A16LinearKernel.
|
|
2
|
+
|
|
3
|
+
Copyright 2025-2026 Fujitsu Ltd.
|
|
4
|
+
|
|
5
|
+
The entry point installed via ``pip install``
|
|
6
|
+
(declared in this package's ``pyproject.toml``) is
|
|
7
|
+
:func:`onecomp_vllm_v0_24_0_rocm.patch.apply`.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from .patch import apply
|
|
11
|
+
|
|
12
|
+
__all__ = ["apply"]
|
|
@@ -0,0 +1,238 @@
|
|
|
1
|
+
"""ROCm-only workaround for vLLM 0.24.x TritonW4A16LinearKernel.
|
|
2
|
+
|
|
3
|
+
Copyright 2025-2026 Fujitsu Ltd.
|
|
4
|
+
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
_APPLIED_MARKER_AUTO_GPTQ = "_onecomp_vllm_0_24_0_rocm_applied_auto_gptq"
|
|
10
|
+
_APPLIED_MARKER_KERNEL = "_onecomp_vllm_0_24_0_rocm_applied_kernel"
|
|
11
|
+
|
|
12
|
+
# Set after the first inference-time fixup so logs stay readable (once per process).
|
|
13
|
+
_LOGGED_AUTO_GPTQ_QZEROS_FIXUP = False
|
|
14
|
+
_LOGGED_TRITON_QZEROS_PERMUTE = False
|
|
15
|
+
|
|
16
|
+
# vLLM only attaches handlers to the ``vllm`` logger tree; ``__name__`` logs are silently dropped.
|
|
17
|
+
# Use a child of ``vllm`` so INFO lines appear in engine output (including EngineCore subprocesses).
|
|
18
|
+
_VLLM_LOGGER_NAME = "vllm.onecomp_v0_24_0_rocm"
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def _get_logger():
|
|
22
|
+
from vllm.logger import init_logger
|
|
23
|
+
|
|
24
|
+
return init_logger(_VLLM_LOGGER_NAME)
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def _is_target_env() -> tuple[bool, str]:
|
|
28
|
+
"""Return ``(is_target, reason)`` for the current Python / vLLM env."""
|
|
29
|
+
try:
|
|
30
|
+
import vllm # noqa: F401
|
|
31
|
+
except ImportError as exc:
|
|
32
|
+
return False, f"vllm not importable ({exc})"
|
|
33
|
+
try:
|
|
34
|
+
from vllm.platforms import current_platform
|
|
35
|
+
except ImportError as exc:
|
|
36
|
+
return False, f"vllm.platforms not importable ({exc})"
|
|
37
|
+
|
|
38
|
+
if not current_platform.is_rocm():
|
|
39
|
+
return False, "not ROCm"
|
|
40
|
+
|
|
41
|
+
# Accept "0.24.0", "0.24.0+rocm723", "0.24.1", "0.24.0rc3.dev3", etc.
|
|
42
|
+
base_version = vllm.__version__.split("+", 1)[0]
|
|
43
|
+
if not base_version.startswith("0.24."):
|
|
44
|
+
return False, f"vllm base version {base_version!r} not in 0.24.x"
|
|
45
|
+
|
|
46
|
+
return True, "vllm 0.24.x on ROCm"
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def _unbias_v1_zeros(zp_packed):
|
|
50
|
+
"""Add 1 (mod 16) to every 4-bit nibble of a GPTQ-packed int32 tensor.
|
|
51
|
+
|
|
52
|
+
Parameters
|
|
53
|
+
----------
|
|
54
|
+
zp_packed
|
|
55
|
+
``[K // G, N // 8]`` int32 tensor. GPTQ sequential packing: each
|
|
56
|
+
int32 holds 8 consecutive N-values at bit offsets
|
|
57
|
+
``[0, 4, 8, ..., 28]``.
|
|
58
|
+
|
|
59
|
+
Returns
|
|
60
|
+
-------
|
|
61
|
+
Same shape / dtype / device, with every nibble incremented by 1 mod 16.
|
|
62
|
+
"""
|
|
63
|
+
import torch
|
|
64
|
+
|
|
65
|
+
shifts = torch.arange(8, device=zp_packed.device, dtype=torch.int32) * 4
|
|
66
|
+
nibbles = ((zp_packed.unsqueeze(-1) >> shifts) & 0xF) + 1
|
|
67
|
+
nibbles &= 0xF
|
|
68
|
+
return torch.sum(nibbles << shifts, dim=-1, dtype=torch.int32)
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def _patch_auto_gptq_process_weights(logger) -> None:
|
|
72
|
+
"""Wrap ``AutoGPTQLinearMethod.process_weights_after_loading`` with a +1
|
|
73
|
+
fixup on ``qzeros`` for the GPTQv1 convention.
|
|
74
|
+
|
|
75
|
+
Rationale
|
|
76
|
+
---------
|
|
77
|
+
GPTQv1 checkpoints store ``stored_zero = real_zero - 1``. Marlin/Machete
|
|
78
|
+
kernels re-add the ``-1`` bias internally, but ``TritonW4A16LinearKernel``
|
|
79
|
+
consumes ``qzeros`` verbatim, so we need to add 1 (mod 16) to every
|
|
80
|
+
packed nibble before the kernel sees the parameter. This is scoped to
|
|
81
|
+
the AutoGPTQ path (i.e. GPTQv1 checkpoints); other quantization methods
|
|
82
|
+
that reuse ``TritonW4A16LinearKernel`` (e.g. compressed-tensors) do not
|
|
83
|
+
have the ``-1`` convention and are left untouched.
|
|
84
|
+
"""
|
|
85
|
+
from vllm.model_executor.kernels.linear.mixed_precision.triton_w4a16 import (
|
|
86
|
+
TritonW4A16LinearKernel,
|
|
87
|
+
)
|
|
88
|
+
from vllm.model_executor.layers.quantization.auto_gptq import (
|
|
89
|
+
AutoGPTQConfig,
|
|
90
|
+
AutoGPTQLinearMethod,
|
|
91
|
+
)
|
|
92
|
+
from vllm.scalar_type import scalar_types
|
|
93
|
+
|
|
94
|
+
# Additive: unblock asymmetric GPTQ checkpoints. Stock vLLM 0.24 still
|
|
95
|
+
# only registers (n_bits, True) in TYPE_MAP, which makes AutoGPTQConfig
|
|
96
|
+
# .from_config raise ValueError for sym=False models. TritonW4A16
|
|
97
|
+
# already advertises uint4 (asymmetric) in SUPPORTED_QUANT_TYPES, so
|
|
98
|
+
# this extension does not change kernel selection for sym=True.
|
|
99
|
+
AutoGPTQConfig.TYPE_MAP.setdefault((4, False), scalar_types.uint4)
|
|
100
|
+
AutoGPTQConfig.TYPE_MAP.setdefault((8, False), scalar_types.uint8)
|
|
101
|
+
|
|
102
|
+
if getattr(
|
|
103
|
+
AutoGPTQLinearMethod.process_weights_after_loading,
|
|
104
|
+
_APPLIED_MARKER_AUTO_GPTQ,
|
|
105
|
+
False,
|
|
106
|
+
):
|
|
107
|
+
logger.debug("onecomp env-patch already installed on AutoGPTQLinearMethod; skip")
|
|
108
|
+
return
|
|
109
|
+
|
|
110
|
+
_orig = AutoGPTQLinearMethod.process_weights_after_loading
|
|
111
|
+
|
|
112
|
+
def _patched(self, layer):
|
|
113
|
+
global _LOGGED_AUTO_GPTQ_QZEROS_FIXUP
|
|
114
|
+
# +1 fixup runs BEFORE the kernel's process_weights_after_loading so
|
|
115
|
+
# that we operate on the AutoGPTQ-native layout [K//G, N//8] where
|
|
116
|
+
# the packed axis is dim -1 (which is what ``_unbias_v1_zeros``
|
|
117
|
+
# assumes). The subsequent shape/permute juggling in the kernel
|
|
118
|
+
# PWAL preserves per-nibble values.
|
|
119
|
+
if isinstance(self.kernel, TritonW4A16LinearKernel):
|
|
120
|
+
zp = getattr(layer, "qzeros", None)
|
|
121
|
+
if zp is not None and getattr(zp, "data", None) is not None:
|
|
122
|
+
zp.data = _unbias_v1_zeros(zp.data)
|
|
123
|
+
if not _LOGGED_AUTO_GPTQ_QZEROS_FIXUP:
|
|
124
|
+
_LOGGED_AUTO_GPTQ_QZEROS_FIXUP = True
|
|
125
|
+
logger.info(
|
|
126
|
+
"onecomp env-patch active at inference: "
|
|
127
|
+
"AutoGPTQLinearMethod qzeros +1 fixup "
|
|
128
|
+
"(TritonW4A16LinearKernel, qzeros shape=%s)",
|
|
129
|
+
tuple(zp.data.shape),
|
|
130
|
+
)
|
|
131
|
+
return _orig(self, layer)
|
|
132
|
+
|
|
133
|
+
setattr(_patched, _APPLIED_MARKER_AUTO_GPTQ, True)
|
|
134
|
+
AutoGPTQLinearMethod.process_weights_after_loading = _patched
|
|
135
|
+
logger.info("onecomp env-patch: wrapped AutoGPTQLinearMethod.process_weights_after_loading")
|
|
136
|
+
|
|
137
|
+
|
|
138
|
+
def _patch_triton_w4a16_kernel_process_weights(logger) -> None:
|
|
139
|
+
"""Wrap ``TritonW4A16LinearKernel.process_weights_after_loading`` to fix
|
|
140
|
+
a missing ``permute_param_layout_`` call on ``qzeros``.
|
|
141
|
+
|
|
142
|
+
Rationale
|
|
143
|
+
---------
|
|
144
|
+
In vLLM 0.24.x the stock ``TritonW4A16LinearKernel.process_weights_after_loading``
|
|
145
|
+
normalizes ``qweight`` and ``scales`` via ``permute_param_layout_`` so
|
|
146
|
+
that both compressed-tensors (``output_dim=0, packed_dim=0``) and
|
|
147
|
+
AutoGPTQ (``input_dim=0, output_dim=1, packed_dim=1``) checkpoint
|
|
148
|
+
layouts land in the same physical form before the ``.t().contiguous()``
|
|
149
|
+
that the kernel expects. The block for ``qzeros`` skips this step and
|
|
150
|
+
calls ``.t()`` unconditionally::
|
|
151
|
+
|
|
152
|
+
# Checkpoint: [N//8, K//G] int32 (N packed at dim 0, K//G at dim 1)
|
|
153
|
+
# Kernel needs: [K//G, N//8] -- just transpose
|
|
154
|
+
replace_parameter(
|
|
155
|
+
layer, self.w_zp_name,
|
|
156
|
+
torch.nn.Parameter(zp.data.t().contiguous(), requires_grad=False),
|
|
157
|
+
)
|
|
158
|
+
|
|
159
|
+
That works only for the compressed-tensors layout. For an AutoGPTQ
|
|
160
|
+
checkpoint the parameter is already ``[K//G, N//8]``, so the ``.t()``
|
|
161
|
+
flips it to ``[N//8, K//G]`` and the downstream shape assertion in
|
|
162
|
+
``triton_w4a16_gemm`` fires (observed as
|
|
163
|
+
``AssertionError: qzeros shape mismatch: torch.Size([N//8, K//G])``).
|
|
164
|
+
|
|
165
|
+
The one-liner upstream is missing is a ``permute_param_layout_(zp,
|
|
166
|
+
input_dim=1, output_dim=0, packed_dim=0)`` call before the ``.t()``.
|
|
167
|
+
Injecting it via a wrapper here fixes AutoGPTQ without regressing the
|
|
168
|
+
compressed-tensors path (that path already satisfies the target
|
|
169
|
+
layout, so the permute is a no-op).
|
|
170
|
+
"""
|
|
171
|
+
from vllm.model_executor.kernels.linear.mixed_precision.triton_w4a16 import (
|
|
172
|
+
TritonW4A16LinearKernel,
|
|
173
|
+
)
|
|
174
|
+
from vllm.model_executor.parameter import permute_param_layout_
|
|
175
|
+
|
|
176
|
+
if getattr(
|
|
177
|
+
TritonW4A16LinearKernel.process_weights_after_loading,
|
|
178
|
+
_APPLIED_MARKER_KERNEL,
|
|
179
|
+
False,
|
|
180
|
+
):
|
|
181
|
+
logger.debug("onecomp env-patch already installed on TritonW4A16LinearKernel; skip")
|
|
182
|
+
return
|
|
183
|
+
|
|
184
|
+
_orig = TritonW4A16LinearKernel.process_weights_after_loading
|
|
185
|
+
|
|
186
|
+
def _patched(self, layer):
|
|
187
|
+
global _LOGGED_TRITON_QZEROS_PERMUTE
|
|
188
|
+
if self.w_zp_name is not None:
|
|
189
|
+
zp = getattr(layer, self.w_zp_name, None)
|
|
190
|
+
if zp is not None:
|
|
191
|
+
# Normalise both compressed-tensors and AutoGPTQ layouts to
|
|
192
|
+
# [N//8, K//G] with (input_dim=1, output_dim=0, packed_dim=0)
|
|
193
|
+
# so the stock code's subsequent zp.data.t().contiguous()
|
|
194
|
+
# ends up at the kernel's expected [K//G, N//8].
|
|
195
|
+
permute_param_layout_(zp, input_dim=1, output_dim=0, packed_dim=0)
|
|
196
|
+
if not _LOGGED_TRITON_QZEROS_PERMUTE:
|
|
197
|
+
_LOGGED_TRITON_QZEROS_PERMUTE = True
|
|
198
|
+
logger.info(
|
|
199
|
+
"onecomp env-patch active at inference: "
|
|
200
|
+
"TritonW4A16LinearKernel qzeros permute_param_layout_ "
|
|
201
|
+
"(param=%s, shape=%s)",
|
|
202
|
+
self.w_zp_name,
|
|
203
|
+
tuple(zp.data.shape),
|
|
204
|
+
)
|
|
205
|
+
return _orig(self, layer)
|
|
206
|
+
|
|
207
|
+
setattr(_patched, _APPLIED_MARKER_KERNEL, True)
|
|
208
|
+
TritonW4A16LinearKernel.process_weights_after_loading = _patched
|
|
209
|
+
logger.info("onecomp env-patch: wrapped TritonW4A16LinearKernel.process_weights_after_loading")
|
|
210
|
+
|
|
211
|
+
|
|
212
|
+
def apply() -> None:
|
|
213
|
+
"""vLLM ``vllm.general_plugins`` entry point.
|
|
214
|
+
|
|
215
|
+
No-op on every platform / version other than ROCm + vLLM 0.24.x.
|
|
216
|
+
"""
|
|
217
|
+
logger = _get_logger()
|
|
218
|
+
|
|
219
|
+
is_target, reason = _is_target_env()
|
|
220
|
+
if not is_target:
|
|
221
|
+
logger.debug("onecomp env-patch vllm_v0_24_0_rocm skipped: %s", reason)
|
|
222
|
+
return
|
|
223
|
+
|
|
224
|
+
import vllm
|
|
225
|
+
|
|
226
|
+
logger.info(
|
|
227
|
+
"onecomp env-patch: loading for %s (vllm %s)",
|
|
228
|
+
reason,
|
|
229
|
+
vllm.__version__,
|
|
230
|
+
)
|
|
231
|
+
_patch_auto_gptq_process_weights(logger)
|
|
232
|
+
_patch_triton_w4a16_kernel_process_weights(logger)
|
|
233
|
+
logger.info(
|
|
234
|
+
"onecomp env-patch installed: "
|
|
235
|
+
"AutoGPTQLinearMethod qzeros +1 fixup for TritonW4A16LinearKernel; "
|
|
236
|
+
"TritonW4A16LinearKernel qzeros layout permute injected; "
|
|
237
|
+
"AutoGPTQConfig.TYPE_MAP extended with asymmetric uint4/uint8.",
|
|
238
|
+
)
|
|
@@ -0,0 +1,57 @@
|
|
|
1
|
+
"""GPTQ quantization -> direct GGUF export -> CPU inference with llama.cpp.
|
|
2
|
+
|
|
3
|
+
This example shows the OneComp CPU inference path:
|
|
4
|
+
1. Quantize a model with GPTQ (4-bit symmetric, group size 128).
|
|
5
|
+
2. Export it to GGUF without re-quantization (preserving the GPTQ codes).
|
|
6
|
+
3. Run CPU text generation via llama-cpp-python.
|
|
7
|
+
|
|
8
|
+
Only the GGUF export and inference are CPU-bound; quantization runs on the GPU
|
|
9
|
+
when one is available and falls back to CPU (ModelConfig default
|
|
10
|
+
``device="auto"``; both are supported).
|
|
11
|
+
|
|
12
|
+
Run:
|
|
13
|
+
python example/cpu_inference/example_gptq_gguf_cpu.py
|
|
14
|
+
|
|
15
|
+
Requires: pip install 'onecomp[llamacpp]' (gguf + llama-cpp-python)
|
|
16
|
+
|
|
17
|
+
Copyright 2025-2026 Fujitsu Ltd.
|
|
18
|
+
|
|
19
|
+
Author: Yuma Ichikawa
|
|
20
|
+
|
|
21
|
+
"""
|
|
22
|
+
|
|
23
|
+
import os
|
|
24
|
+
|
|
25
|
+
from onecomp import GPTQ, CalibrationConfig, ModelConfig, Runner
|
|
26
|
+
from onecomp.cpu import LlamaCppModel, convert_gptq_to_gguf
|
|
27
|
+
from onecomp.log import setup_logger
|
|
28
|
+
|
|
29
|
+
MODEL_ID = "TinyLlama/TinyLlama-1.1B-intermediate-step-1431k-3T"
|
|
30
|
+
SAVE_DIR = "./TinyLlama-1.1B-gptq-4bit"
|
|
31
|
+
GGUF_PATH = "./TinyLlama-1.1B-gptq-4bit.gguf"
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def main():
|
|
35
|
+
setup_logger()
|
|
36
|
+
|
|
37
|
+
# 1. Quantize (4-bit symmetric, group size 128, no actorder => lossless to Q4_0).
|
|
38
|
+
runner = Runner(
|
|
39
|
+
model_config=ModelConfig(model_id=MODEL_ID),
|
|
40
|
+
quantizer=GPTQ(wbits=4, groupsize=128, sym=True, actorder=False),
|
|
41
|
+
calibration_config=CalibrationConfig(num_calibration_samples=64, max_length=512),
|
|
42
|
+
qep=True,
|
|
43
|
+
)
|
|
44
|
+
runner.run()
|
|
45
|
+
runner.save_quantized_model(SAVE_DIR)
|
|
46
|
+
|
|
47
|
+
# 2. Direct GGUF export (no re-quantization; QEP-corrected codes preserved).
|
|
48
|
+
summary = convert_gptq_to_gguf(quantized_dir=SAVE_DIR, out_gguf=GGUF_PATH)
|
|
49
|
+
print("Export summary:", summary)
|
|
50
|
+
|
|
51
|
+
# 3. CPU inference.
|
|
52
|
+
model = LlamaCppModel(GGUF_PATH, n_ctx=1024)
|
|
53
|
+
print(model.generate("Fujitsu is", max_tokens=64, temperature=0.0))
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
if __name__ == "__main__":
|
|
57
|
+
main()
|
|
@@ -0,0 +1,63 @@
|
|
|
1
|
+
"""Example: mixed-bit GPTQ -> mixed-precision GGUF -> llama.cpp CPU inference.
|
|
2
|
+
|
|
3
|
+
Quantizes a model with per-module bit-widths (4-bit attention, 8-bit MLP gate,
|
|
4
|
+
3-bit MLP up, 2-bit MLP down) in a single GPTQ run, then exports it to ONE GGUF
|
|
5
|
+
where each tensor carries its own quantization type:
|
|
6
|
+
|
|
7
|
+
4-bit sym -> Q4_0 8-bit sym -> Q8_0 3-bit -> Q3_K 2-bit -> Q2_K
|
|
8
|
+
|
|
9
|
+
The 4/8-bit layers keep the exact GPTQ codes (lossless); the 2/3-bit layers are
|
|
10
|
+
re-quantized to K-quants via llama-quantize (requires the llama-quantize binary).
|
|
11
|
+
|
|
12
|
+
Copyright 2025-2026 Fujitsu Ltd.
|
|
13
|
+
|
|
14
|
+
Author: Yuma Ichikawa
|
|
15
|
+
"""
|
|
16
|
+
|
|
17
|
+
import json
|
|
18
|
+
import os
|
|
19
|
+
|
|
20
|
+
from transformers import AutoConfig
|
|
21
|
+
|
|
22
|
+
from llamacpp_plugins.gptq import export_mixed_gptq_gguf, plan_mixed_export
|
|
23
|
+
from onecomp import GPTQ, CalibrationConfig, ModelConfig, Runner, setup_logger
|
|
24
|
+
from onecomp.cpu import LlamaCppModel
|
|
25
|
+
|
|
26
|
+
MODEL_ID = "TinyLlama/TinyLlama-1.1B-intermediate-step-1431k-3T"
|
|
27
|
+
OUT_DIR = "./results/mixed-gptq"
|
|
28
|
+
OUT_GGUF = "./results/mixed-gptq.gguf"
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def main():
|
|
32
|
+
setup_logger()
|
|
33
|
+
|
|
34
|
+
num_layers = AutoConfig.from_pretrained(MODEL_ID).num_hidden_layers
|
|
35
|
+
module_wbits = {}
|
|
36
|
+
for i in range(num_layers):
|
|
37
|
+
module_wbits[f"model.layers.{i}.mlp.gate_proj"] = 8
|
|
38
|
+
module_wbits[f"model.layers.{i}.mlp.up_proj"] = 3
|
|
39
|
+
module_wbits[f"model.layers.{i}.mlp.down_proj"] = 2
|
|
40
|
+
|
|
41
|
+
quantizer = GPTQ(wbits=4, groupsize=128, sym=True, module_wbits=module_wbits)
|
|
42
|
+
runner = Runner(
|
|
43
|
+
model_config=ModelConfig(model_id=MODEL_ID, device="cpu"),
|
|
44
|
+
quantizer=quantizer,
|
|
45
|
+
calibration_config=CalibrationConfig(num_calibration_samples=32, max_length=512),
|
|
46
|
+
qep=False,
|
|
47
|
+
)
|
|
48
|
+
runner.run()
|
|
49
|
+
runner.save_quantized_model(OUT_DIR)
|
|
50
|
+
|
|
51
|
+
# Preview the per-module GGUF routing.
|
|
52
|
+
for p in plan_mixed_export(OUT_DIR)[:6]:
|
|
53
|
+
print(f"{p.name:45s} bits={p.bits} -> {p.route}/{p.ggml_type}")
|
|
54
|
+
|
|
55
|
+
summary = export_mixed_gptq_gguf(OUT_DIR, OUT_GGUF)
|
|
56
|
+
print(json.dumps(summary["plan"], indent=2))
|
|
57
|
+
|
|
58
|
+
model = LlamaCppModel(OUT_GGUF, n_ctx=512)
|
|
59
|
+
print(model.generate("The capital of Japan is", max_tokens=32, temperature=0.0))
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
if __name__ == "__main__":
|
|
63
|
+
main()
|
|
@@ -0,0 +1,43 @@
|
|
|
1
|
+
"""Serve a packed OneComp checkpoint (or GGUF) on CPU with one command.
|
|
2
|
+
|
|
3
|
+
The barrier-free path: drop in a *packed quantized* model directory and anyone
|
|
4
|
+
can run it on CPU behind an OpenAI-compatible API. If given a packed OneComp
|
|
5
|
+
checkpoint, the GGUF is auto-exported (losslessly, cached) on first launch.
|
|
6
|
+
|
|
7
|
+
Run:
|
|
8
|
+
# Packed OneComp GPTQ checkpoint (auto-exports to GGUF once):
|
|
9
|
+
python example/cpu_inference/example_serve_cpu.py ./TinyLlama-1.1B-gptq-4bit
|
|
10
|
+
# ...or an existing GGUF:
|
|
11
|
+
python example/cpu_inference/example_serve_cpu.py ./model.gguf
|
|
12
|
+
|
|
13
|
+
Then, from another shell:
|
|
14
|
+
curl http://localhost:8080/v1/chat/completions \
|
|
15
|
+
-d '{"messages":[{"role":"user","content":"Hello!"}],"max_tokens":64}'
|
|
16
|
+
|
|
17
|
+
Equivalent CLI: onecomp-gguf serve --model <path> --port 8080
|
|
18
|
+
|
|
19
|
+
Requires: pip install 'onecomp[llamacpp]' (no FastAPI/uvicorn needed)
|
|
20
|
+
|
|
21
|
+
Copyright 2025-2026 Fujitsu Ltd.
|
|
22
|
+
|
|
23
|
+
Author: Yuma Ichikawa
|
|
24
|
+
|
|
25
|
+
"""
|
|
26
|
+
|
|
27
|
+
import sys
|
|
28
|
+
|
|
29
|
+
from onecomp.cpu import serve
|
|
30
|
+
from onecomp.log import setup_logger
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def main():
|
|
34
|
+
setup_logger()
|
|
35
|
+
model_path = sys.argv[1] if len(sys.argv) > 1 else "./TinyLlama-1.1B-gptq-4bit"
|
|
36
|
+
port = int(sys.argv[2]) if len(sys.argv) > 2 else 8080
|
|
37
|
+
# host=0.0.0.0 to expose on the network; chat_format is auto-detected from
|
|
38
|
+
# the GGUF architecture when no chat template is embedded.
|
|
39
|
+
serve(model_path, host="0.0.0.0", port=port, n_ctx=2048)
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
if __name__ == "__main__":
|
|
43
|
+
main()
|