onecomp 1.2.2__tar.gz → 1.3.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (325) hide show
  1. {onecomp-1.2.2/onecomp.egg-info → onecomp-1.3.0}/PKG-INFO +64 -2
  2. {onecomp-1.2.2 → onecomp-1.3.0}/README.md +55 -1
  3. onecomp-1.3.0/envs/vllm/v0_24_0_rocm/src/onecomp_vllm_v0_24_0_rocm/__init__.py +12 -0
  4. onecomp-1.3.0/envs/vllm/v0_24_0_rocm/src/onecomp_vllm_v0_24_0_rocm/patch.py +238 -0
  5. onecomp-1.3.0/example/cpu_inference/example_gptq_gguf_cpu.py +57 -0
  6. onecomp-1.3.0/example/cpu_inference/example_mixed_gptq_gguf_cpu.py +63 -0
  7. onecomp-1.3.0/example/cpu_inference/example_serve_cpu.py +43 -0
  8. onecomp-1.3.0/example/example_mdbf.py +57 -0
  9. onecomp-1.3.0/example/post_process/example_blockwise_global_ptq.py +167 -0
  10. onecomp-1.3.0/example/post_process/example_blockwise_global_ptq_staged.py +212 -0
  11. {onecomp-1.2.2 → onecomp-1.3.0}/example/post_process/example_blockwise_ptq.py +58 -49
  12. onecomp-1.3.0/example/post_process/example_global_ptq.py +102 -0
  13. {onecomp-1.2.2 → onecomp-1.3.0}/example/post_process/example_global_ptq_dbf.py +22 -5
  14. {onecomp-1.2.2 → onecomp-1.3.0}/example/post_process/example_global_ptq_distributed.py +20 -4
  15. onecomp-1.3.0/example/post_process/example_lora_gptq_vllm_inference.py +146 -0
  16. {onecomp-1.2.2 → onecomp-1.3.0}/example/post_process/example_lora_sft.py +15 -23
  17. {onecomp-1.2.2 → onecomp-1.3.0}/example/post_process/example_lora_sft_knowledge.py +52 -11
  18. onecomp-1.3.0/example/post_process/example_lora_sft_knowledge_jointq.py +184 -0
  19. onecomp-1.3.0/example/post_process/example_reload_post_process_resave.py +130 -0
  20. onecomp-1.3.0/example/vllm_inference/example_dbf_vllm_inference.py +124 -0
  21. onecomp-1.3.0/example/vllm_inference/example_gptq_vllm_gptoss_inference.py +100 -0
  22. {onecomp-1.2.2 → onecomp-1.3.0}/example/vllm_inference/example_gptq_vllm_inference.py +3 -1
  23. onecomp-1.3.0/example/vllm_inference/example_gptq_vllm_qwen36_inference.py +114 -0
  24. onecomp-1.3.0/llamacpp_plugins/gptq/__init__.py +27 -0
  25. onecomp-1.3.0/llamacpp_plugins/gptq/constants.py +66 -0
  26. onecomp-1.3.0/llamacpp_plugins/gptq/llamacpp_plugin.py +280 -0
  27. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/__init__.py +1 -0
  28. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/__version__.py +1 -1
  29. onecomp-1.3.0/onecomp/cli.py +148 -0
  30. onecomp-1.3.0/onecomp/cpu/__init__.py +48 -0
  31. onecomp-1.3.0/onecomp/cpu/cli.py +274 -0
  32. onecomp-1.3.0/onecomp/cpu/eval/__init__.py +42 -0
  33. onecomp-1.3.0/onecomp/cpu/eval/benchmark.py +118 -0
  34. onecomp-1.3.0/onecomp/cpu/eval/inspect_gguf.py +160 -0
  35. onecomp-1.3.0/onecomp/cpu/eval/parity.py +162 -0
  36. onecomp-1.3.0/onecomp/cpu/eval/perplexity.py +141 -0
  37. onecomp-1.3.0/onecomp/cpu/export/__init__.py +62 -0
  38. onecomp-1.3.0/onecomp/cpu/export/auto.py +162 -0
  39. onecomp-1.3.0/onecomp/cpu/export/blocks.py +210 -0
  40. onecomp-1.3.0/onecomp/cpu/export/checkpoint.py +254 -0
  41. onecomp-1.3.0/onecomp/cpu/export/dequantize.py +166 -0
  42. onecomp-1.3.0/onecomp/cpu/export/direct.py +150 -0
  43. onecomp-1.3.0/onecomp/cpu/export/fallback.py +63 -0
  44. onecomp-1.3.0/onecomp/cpu/export/rotation.py +65 -0
  45. onecomp-1.3.0/onecomp/cpu/export/skeleton.py +260 -0
  46. onecomp-1.3.0/onecomp/cpu/inference.py +133 -0
  47. onecomp-1.3.0/onecomp/cpu/llama_tooling.py +211 -0
  48. onecomp-1.3.0/onecomp/cpu/serve.py +435 -0
  49. onecomp-1.3.0/onecomp/export/__init__.py +37 -0
  50. onecomp-1.3.0/onecomp/export/gguf_export.py +665 -0
  51. onecomp-1.3.0/onecomp/export/gguf_reader.py +213 -0
  52. onecomp-1.3.0/onecomp/export/gguf_writer.py +284 -0
  53. onecomp-1.3.0/onecomp/export/hub.py +86 -0
  54. onecomp-1.3.0/onecomp/export/model_card.py +107 -0
  55. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/lpcd/_lpcd_runner.py +3 -1
  56. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/lpcd/_refiner.py +8 -2
  57. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/model_config.py +24 -0
  58. onecomp-1.3.0/onecomp/post_process/_base.py +192 -0
  59. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/post_process/_blockwise/dbf_block_optimizer.py +5 -5
  60. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/post_process/_blockwise/dbf_cbq_optimizer.py +3 -3
  61. onecomp-1.3.0/onecomp/post_process/_runtime.py +137 -0
  62. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/post_process/blockwise_ptq.py +33 -1
  63. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/post_process/global_ptq.py +37 -2
  64. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/post_process/global_ptq_distributed.py +8 -5
  65. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/post_process/post_process_lora_sft.py +78 -4
  66. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/pre_process/rotation_utils.py +87 -3
  67. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/qep/_quantize_with_qep_arch.py +121 -28
  68. onecomp-1.3.0/onecomp/quantized_model_loader.py +1396 -0
  69. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/__init__.py +1 -0
  70. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/_quantizer.py +51 -9
  71. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/autobit/_autobit.py +53 -6
  72. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/autobit/dbf_fallback.py +9 -1
  73. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/dbf/_dbf.py +139 -7
  74. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/dbf/dbf_layer.py +100 -9
  75. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/gptq/_gptq.py +139 -12
  76. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/gptq/config.py +6 -5
  77. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/gptq/gptq_layer.py +219 -23
  78. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/jointq/_jointq.py +15 -3
  79. onecomp-1.3.0/onecomp/quantizer/mdbf/__init__.py +12 -0
  80. onecomp-1.3.0/onecomp/quantizer/mdbf/_mdbf.py +607 -0
  81. onecomp-1.3.0/onecomp/quantizer/mdbf/admm.py +1081 -0
  82. onecomp-1.3.0/onecomp/quantizer/mdbf/config.py +108 -0
  83. onecomp-1.3.0/onecomp/quantizer/mdbf/gradient_refine.py +324 -0
  84. onecomp-1.3.0/onecomp/quantizer/mdbf/initialize.py +428 -0
  85. onecomp-1.3.0/onecomp/quantizer/mdbf/mdbf_impl.py +295 -0
  86. onecomp-1.3.0/onecomp/quantizer/mdbf/mdbf_layer.py +601 -0
  87. onecomp-1.3.0/onecomp/quantizer/mdbf/utils.py +251 -0
  88. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/onebit/onebit_impl.py +6 -12
  89. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/quip/_quip.py +1 -1
  90. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/runner.py +990 -68
  91. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/runner_methods/chunked_quantization.py +4 -1
  92. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/runner_methods/multi_gpu_quantization.py +8 -3
  93. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/utils/__init__.py +1 -0
  94. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/utils/blockwise.py +70 -13
  95. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/utils/device.py +15 -0
  96. onecomp-1.3.0/onecomp/utils/lora.py +9 -0
  97. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/utils/model_inputs.py +2 -1
  98. onecomp-1.3.0/onecomp/utils/mxfp4_compat.py +73 -0
  99. onecomp-1.3.0/onecomp/utils/quant_config.py +85 -0
  100. onecomp-1.3.0/onecomp/utils/unfuse_moe.py +724 -0
  101. {onecomp-1.2.2 → onecomp-1.3.0/onecomp.egg-info}/PKG-INFO +64 -2
  102. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp.egg-info/SOURCES.txt +63 -1
  103. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp.egg-info/entry_points.txt +1 -0
  104. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp.egg-info/requires.txt +10 -0
  105. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp.egg-info/top_level.txt +3 -0
  106. {onecomp-1.2.2 → onecomp-1.3.0}/pyproject.toml +14 -1
  107. {onecomp-1.2.2 → onecomp-1.3.0}/vllm_plugins/dbf/vllm_plugin.py +105 -25
  108. onecomp-1.3.0/vllm_plugins/gptq/gptoss_wna16_moe.py +198 -0
  109. {onecomp-1.2.2 → onecomp-1.3.0}/vllm_plugins/gptq/vllm_plugin.py +151 -18
  110. onecomp-1.3.0/vllm_plugins/patches/__init__.py +5 -0
  111. onecomp-1.3.0/vllm_plugins/patches/_paths.py +19 -0
  112. onecomp-1.3.0/vllm_plugins/patches/apply_all.py +87 -0
  113. onecomp-1.3.0/vllm_plugins/patches/gpt_oss_gptq_moe.py +184 -0
  114. onecomp-1.3.0/vllm_plugins/patches/gpt_oss_wna16_bias.py +313 -0
  115. onecomp-1.3.0/vllm_plugins/utils/__init__.py +5 -0
  116. onecomp-1.3.0/vllm_plugins/utils/module.py +177 -0
  117. onecomp-1.3.0/vllm_plugins/utils/rotation.py +261 -0
  118. onecomp-1.2.2/example/post_process/example_global_ptq.py +0 -73
  119. onecomp-1.2.2/onecomp/cli.py +0 -89
  120. onecomp-1.2.2/onecomp/post_process/_base.py +0 -79
  121. onecomp-1.2.2/onecomp/quantized_model_loader.py +0 -624
  122. onecomp-1.2.2/onecomp/utils/quant_config.py +0 -28
  123. onecomp-1.2.2/onecomp/utils/unfuse_moe.py +0 -160
  124. onecomp-1.2.2/vllm_plugins/utils/module.py +0 -94
  125. {onecomp-1.2.2 → onecomp-1.3.0}/LICENSE +0 -0
  126. {onecomp-1.2.2 → onecomp-1.3.0}/benchmark/llama3-8b-gptq/quant_benchmark.py +0 -0
  127. {onecomp-1.2.2 → onecomp-1.3.0}/benchmark/llama3-8b-jointq/quant_benchmark.py +0 -0
  128. {onecomp-1.2.2 → onecomp-1.3.0}/benchmark/llama3-8b-lpcd-gptq/quant_benchmark.py +0 -0
  129. {onecomp-1.2.2 → onecomp-1.3.0}/benchmark/llama3-8b-qep-gptq/quant_benchmark.py +0 -0
  130. {onecomp-1.2.2 → onecomp-1.3.0}/benchmark/llama3-8b-various/quant_benchmark.py +0 -0
  131. {onecomp-1.2.2 → onecomp-1.3.0}/benchmark/qwen3-14b-gptq/quant_benchmark.py +0 -0
  132. {onecomp-1.2.2 → onecomp-1.3.0}/benchmark/qwen3-14b-jointq/quant_benchmark.py +0 -0
  133. {onecomp-1.2.2 → onecomp-1.3.0}/benchmark/qwen3-8b-gptq/quant_benchmark.py +0 -0
  134. {onecomp-1.2.2 → onecomp-1.3.0}/benchmark/qwen3-8b-jointq/quant_benchmark.py +0 -0
  135. {onecomp-1.2.2 → onecomp-1.3.0}/dashboard/backend/app/__init__.py +0 -0
  136. {onecomp-1.2.2 → onecomp-1.3.0}/dashboard/backend/app/api/__init__.py +0 -0
  137. {onecomp-1.2.2 → onecomp-1.3.0}/dashboard/backend/app/api/jobs.py +0 -0
  138. {onecomp-1.2.2 → onecomp-1.3.0}/dashboard/backend/app/constants.py +0 -0
  139. {onecomp-1.2.2 → onecomp-1.3.0}/dashboard/backend/app/core/__init__.py +0 -0
  140. {onecomp-1.2.2 → onecomp-1.3.0}/dashboard/backend/app/core/config.py +0 -0
  141. {onecomp-1.2.2 → onecomp-1.3.0}/dashboard/backend/app/core/database.py +0 -0
  142. {onecomp-1.2.2 → onecomp-1.3.0}/dashboard/backend/app/main.py +0 -0
  143. {onecomp-1.2.2 → onecomp-1.3.0}/dashboard/backend/app/models/__init__.py +0 -0
  144. {onecomp-1.2.2 → onecomp-1.3.0}/dashboard/backend/app/models/job.py +0 -0
  145. {onecomp-1.2.2 → onecomp-1.3.0}/dashboard/backend/app/schemas/__init__.py +0 -0
  146. {onecomp-1.2.2 → onecomp-1.3.0}/dashboard/backend/app/schemas/job.py +0 -0
  147. {onecomp-1.2.2 → onecomp-1.3.0}/dashboard/backend/app/services/__init__.py +0 -0
  148. {onecomp-1.2.2 → onecomp-1.3.0}/dashboard/backend/app/services/huggingface.py +0 -0
  149. {onecomp-1.2.2 → onecomp-1.3.0}/dashboard/backend/app/services/inference.py +0 -0
  150. {onecomp-1.2.2 → onecomp-1.3.0}/dashboard/backend/app/services/job_store.py +0 -0
  151. {onecomp-1.2.2 → onecomp-1.3.0}/dashboard/backend/app/worker/__init__.py +0 -0
  152. {onecomp-1.2.2 → onecomp-1.3.0}/dashboard/backend/app/worker/celery_app.py +0 -0
  153. {onecomp-1.2.2 → onecomp-1.3.0}/dashboard/backend/app/worker/tasks.py +0 -0
  154. {onecomp-1.2.2 → onecomp-1.3.0}/dashboard/backend/cpu_patch.py +0 -0
  155. {onecomp-1.2.2 → onecomp-1.3.0}/dashboard/backend/start_backend.py +0 -0
  156. {onecomp-1.2.2 → onecomp-1.3.0}/dashboard/backend/start_worker.py +0 -0
  157. {onecomp-1.2.2 → onecomp-1.3.0}/example/example_auto_run.py +0 -0
  158. {onecomp-1.2.2 → onecomp-1.3.0}/example/example_autobit.py +0 -0
  159. {onecomp-1.2.2 → onecomp-1.3.0}/example/example_custom_calibration.py +0 -0
  160. {onecomp-1.2.2 → onecomp-1.3.0}/example/example_gptq.py +0 -0
  161. {onecomp-1.2.2 → onecomp-1.3.0}/example/example_jointq.py +0 -0
  162. {onecomp-1.2.2 → onecomp-1.3.0}/example/example_lpcd_gptq.py +0 -0
  163. {onecomp-1.2.2 → onecomp-1.3.0}/example/example_qep_gptq.py +0 -0
  164. {onecomp-1.2.2 → onecomp-1.3.0}/example/example_save_load.py +0 -0
  165. {onecomp-1.2.2 → onecomp-1.3.0}/example/pre_process/example_llama_preprocess_rtn.py +0 -0
  166. {onecomp-1.2.2 → onecomp-1.3.0}/example/pre_process/example_preprocess_save_load.py +0 -0
  167. {onecomp-1.2.2 → onecomp-1.3.0}/example/vllm_inference/example_autobit_vllm_inference.py +0 -0
  168. {onecomp-1.2.2 → onecomp-1.3.0}/example/vllm_inference/example_jointq_vllm_inference.py +0 -0
  169. {onecomp-1.2.2/vllm_plugins → onecomp-1.3.0/llamacpp_plugins}/__init__.py +0 -0
  170. {onecomp-1.2.2 → onecomp-1.3.0}/model_validation/autobit/validate_autobit.py +0 -0
  171. {onecomp-1.2.2 → onecomp-1.3.0}/model_validation/autobit_qep/validate_autobit.py +0 -0
  172. {onecomp-1.2.2 → onecomp-1.3.0}/model_validation/gptq/validate_gptq.py +0 -0
  173. {onecomp-1.2.2 → onecomp-1.3.0}/model_validation/gptq/validate_load.py +0 -0
  174. {onecomp-1.2.2 → onecomp-1.3.0}/model_validation/gptq/validate_vllm.py +0 -0
  175. {onecomp-1.2.2 → onecomp-1.3.0}/model_validation/jointq/validate_jointq.py +0 -0
  176. {onecomp-1.2.2 → onecomp-1.3.0}/model_validation/qep_gptq/validate_gptq.py +0 -0
  177. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/__main__.py +0 -0
  178. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/analyzer/__init__.py +0 -0
  179. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/analyzer/cumulative_error.py +0 -0
  180. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/analyzer/quantization_error.py +0 -0
  181. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/analyzer/weight_outlier.py +0 -0
  182. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/calibration/__init__.py +0 -0
  183. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/calibration/_cache.py +0 -0
  184. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/calibration/c4.py +0 -0
  185. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/calibration/calibration_config.py +0 -0
  186. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/calibration/calibration_data_loader.py +0 -0
  187. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/calibration/chunking.py +0 -0
  188. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/calibration/custom.py +0 -0
  189. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/calibration/wikitext.py +0 -0
  190. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/eval/__init__.py +0 -0
  191. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/eval/__main__.py +0 -0
  192. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/eval/conf/__init__.py +0 -0
  193. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/eval/conf/eval_config.yaml +0 -0
  194. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/eval/evals/__init__.py +0 -0
  195. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/eval/evals/base.py +0 -0
  196. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/eval/evals/mt_bench/__init__.py +0 -0
  197. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/eval/evals/mt_bench/adapter.py +0 -0
  198. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/eval/evals/mt_bench/data.py +0 -0
  199. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/eval/evals/mt_bench/gen_answer.py +0 -0
  200. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/eval/evals/mt_bench/judge.py +0 -0
  201. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/eval/evals/mt_bench/radar_chart.py +0 -0
  202. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/eval/evals/mt_bench/run.py +0 -0
  203. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/eval/evals/mt_bench/show_result.py +0 -0
  204. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/eval/evals/throughput/__init__.py +0 -0
  205. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/eval/evals/throughput/adapter.py +0 -0
  206. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/eval/evals/throughput/bench.py +0 -0
  207. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/eval/evals/throughput/run.py +0 -0
  208. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/eval/orchestrator/__init__.py +0 -0
  209. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/eval/orchestrator/aggregator.py +0 -0
  210. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/eval/orchestrator/runner.py +0 -0
  211. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/eval/orchestrator/server.py +0 -0
  212. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/eval/orchestrator/subprocess_runner.py +0 -0
  213. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/eval/run_evaluate.py +0 -0
  214. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/eval/schema.py +0 -0
  215. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/eval/utils/__init__.py +0 -0
  216. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/eval/utils/model_utils.py +0 -0
  217. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/eval/utils/ports.py +0 -0
  218. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/eval/utils/resources.py +0 -0
  219. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/eval/utils/secrets.py +0 -0
  220. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/log.py +0 -0
  221. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/lpcd/__init__.py +0 -0
  222. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/lpcd/_gradient_solver.py +0 -0
  223. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/lpcd/_lpcd_config.py +0 -0
  224. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/lpcd/_metric.py +0 -0
  225. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/lpcd/arch/_llama.py +0 -0
  226. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/lpcd/arch/_llama_cf.py +0 -0
  227. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/lpcd/arch/_qwen3.py +0 -0
  228. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/post_process/__init__.py +0 -0
  229. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/post_process/_blockwise/__init__.py +0 -0
  230. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/post_process/_blockwise/generic_block_optimizer.py +0 -0
  231. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/post_process/_blockwise/gptq_block_optimizer.py +0 -0
  232. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/post_process/_blockwise/gptq_cbq_optimizer.py +0 -0
  233. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/post_process/_blockwise/helpers.py +0 -0
  234. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/post_process/_blockwise/onebit_block_optimizer.py +0 -0
  235. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/post_process/_blockwise/onebit_cbq_optimizer.py +0 -0
  236. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/post_process/_global_ptq/__init__.py +0 -0
  237. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/post_process/_global_ptq/core.py +0 -0
  238. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/post_process/_global_ptq/dbf_adapter.py +0 -0
  239. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/post_process/_global_ptq/gptq_adapter.py +0 -0
  240. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/post_process/_global_ptq/helpers.py +0 -0
  241. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/post_process/_global_ptq/losses.py +0 -0
  242. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/post_process/_global_ptq/trainer.py +0 -0
  243. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/pre_process/__init__.py +0 -0
  244. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/pre_process/hadamard_utils.py +0 -0
  245. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/pre_process/modeling_llama.py +0 -0
  246. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/pre_process/modeling_qwen3.py +0 -0
  247. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/pre_process/optimizer.py +0 -0
  248. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/pre_process/prepare_rotated_model.py +0 -0
  249. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/pre_process/preprocess_args.py +0 -0
  250. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/pre_process/quant_models.py +0 -0
  251. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/pre_process/train_rotation.py +0 -0
  252. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/qep/__init__.py +0 -0
  253. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/qep/_qep_config.py +0 -0
  254. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/qep/_quantize_with_qep.py +0 -0
  255. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/arb/__init__.py +0 -0
  256. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/arb/_arb.py +0 -0
  257. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/arb/arb_impl.py +0 -0
  258. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/autobit/__init__.py +0 -0
  259. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/autobit/activation_stats.py +0 -0
  260. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/autobit/ilp.py +0 -0
  261. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/autobit/manual.py +0 -0
  262. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/autobit/visualize.py +0 -0
  263. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/cq/__init__.py +0 -0
  264. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/cq/_cq.py +0 -0
  265. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/cq/cq_impl.py +0 -0
  266. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/dbf/__init__.py +0 -0
  267. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/dbf/admm_extended.py +0 -0
  268. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/dbf/balance.py +0 -0
  269. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/dbf/config.py +0 -0
  270. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/dbf/dbf_impl.py +0 -0
  271. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/dbf/dbf_original.py +0 -0
  272. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/dbf/fine_tune.py +0 -0
  273. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/dbf/middle.py +0 -0
  274. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/gemlite.py +0 -0
  275. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/gptq/__init__.py +0 -0
  276. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/jointq/__init__.py +0 -0
  277. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/jointq/core/__init__.py +0 -0
  278. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/jointq/core/__version__.py +0 -0
  279. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/jointq/core/clip.py +0 -0
  280. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/jointq/core/error_propagation/__init__.py +0 -0
  281. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/jointq/core/error_propagation/local_search_advanced.py +0 -0
  282. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/jointq/core/error_propagation/quantize_advanced.py +0 -0
  283. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/jointq/core/error_propagation/quantizer_advanced.py +0 -0
  284. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/jointq/core/local_search.py +0 -0
  285. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/jointq/core/quantize.py +0 -0
  286. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/jointq/core/quantize_multi_gpu.py +0 -0
  287. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/jointq/core/quantizer.py +0 -0
  288. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/jointq/core/solution.py +0 -0
  289. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/onebit/__init__.py +0 -0
  290. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/onebit/_onebit.py +0 -0
  291. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/onebit/onebit_layer.py +0 -0
  292. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/qbb/__init__.py +0 -0
  293. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/qbb/_qbb.py +0 -0
  294. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/qbb/qbb_impl.py +0 -0
  295. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/quip/__init__.py +0 -0
  296. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/quip/quant_quip.py +0 -0
  297. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/quip/quip_impl.py +0 -0
  298. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/quip/utils.py +0 -0
  299. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/quip/utils_had.py +0 -0
  300. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/quip/vector_balance.py +0 -0
  301. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/rtn/__init__.py +0 -0
  302. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/rtn/_rtn.py +0 -0
  303. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/rtn/quantizer.py +0 -0
  304. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/quantizer/rtn/rtn_impl.py +0 -0
  305. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/rotated_model_config.py +0 -0
  306. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/runner_methods/__init__.py +0 -0
  307. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/runner_methods/jointq_error_propagation.py +0 -0
  308. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/utils/accuracy.py +0 -0
  309. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/utils/activation_capture.py +0 -0
  310. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/utils/activation_check.py +0 -0
  311. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/utils/dtype.py +0 -0
  312. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/utils/perplexity.py +0 -0
  313. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/utils/quantization_progress.py +0 -0
  314. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp/utils/vram_estimator.py +0 -0
  315. {onecomp-1.2.2 → onecomp-1.3.0}/onecomp.egg-info/dependency_links.txt +0 -0
  316. {onecomp-1.2.2 → onecomp-1.3.0}/scripts/check_copyright_header.py +0 -0
  317. {onecomp-1.2.2 → onecomp-1.3.0}/scripts/check_no_japanese.py +0 -0
  318. {onecomp-1.2.2 → onecomp-1.3.0}/setup.cfg +0 -0
  319. {onecomp-1.2.2/vllm_plugins/dbf/modules → onecomp-1.3.0/vllm_plugins}/__init__.py +0 -0
  320. {onecomp-1.2.2 → onecomp-1.3.0}/vllm_plugins/dbf/__init__.py +0 -0
  321. {onecomp-1.2.2/vllm_plugins/utils → onecomp-1.3.0/vllm_plugins/dbf/modules}/__init__.py +0 -0
  322. {onecomp-1.2.2 → onecomp-1.3.0}/vllm_plugins/dbf/modules/gemlite_linear.py +0 -0
  323. {onecomp-1.2.2 → onecomp-1.3.0}/vllm_plugins/dbf/modules/naive.py +0 -0
  324. {onecomp-1.2.2 → onecomp-1.3.0}/vllm_plugins/gptq/__init__.py +0 -0
  325. {onecomp-1.2.2 → onecomp-1.3.0}/vllm_plugins/gptq/constants.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: onecomp
3
- Version: 1.2.2
3
+ Version: 1.3.0
4
4
  Summary: Python package for LLM compression
5
5
  Author: Keiji Kimura
6
6
  License: MIT License
@@ -89,6 +89,14 @@ Provides-Extra: hydra
89
89
  Requires-Dist: hydra-core; extra == "hydra"
90
90
  Provides-Extra: vllm
91
91
  Requires-Dist: vllm<0.22,>=0.10; extra == "vllm"
92
+ Requires-Dist: conch-triton-kernels; extra == "vllm"
93
+ Provides-Extra: gguf
94
+ Requires-Dist: gguf>=0.10; extra == "gguf"
95
+ Provides-Extra: llamacpp
96
+ Requires-Dist: gguf>=0.10; extra == "llamacpp"
97
+ Requires-Dist: llama-cpp-python>=0.3.0; extra == "llamacpp"
98
+ Requires-Dist: sentencepiece; extra == "llamacpp"
99
+ Requires-Dist: protobuf; extra == "llamacpp"
92
100
  Provides-Extra: eval
93
101
  Requires-Dist: hydra-core>=1.3; extra == "eval"
94
102
  Requires-Dist: omegaconf>=2.3; extra == "eval"
@@ -143,10 +151,12 @@ Full documentation is available at **[https://FujitsuResearch.github.io/OneCompr
143
151
  - **vLLM Plugin Integration**: Serve OneComp-quantized models with [vLLM](https://docs.vllm.ai/) via built-in plugins for DBF and Mixed-GPTQ quantization methods. Pair with [Open WebUI](https://github.com/open-webui/open-webui) for a ChatGPT-like chat experience on your local machine.
144
152
  - **AutoBit**: Mixed-precision quantization with ILP-based bitwidth assignment. Automatically estimates the target bitwidth from available VRAM and assigns per-layer bitwidths to minimize quantization error under the memory budget.
145
153
  - **JointQ**: Joint quantization method that optimizes weight assignments and scale parameters simultaneously for improved quantization accuracy. Supports group-wise quantization (e.g., 4-bit, groupsize=128).
154
+ - **MDBF (Multi-Envelope Double Binary Factorization)**: A binary factorization quantizer that approximates each weight matrix as a sum of multi-path sign matrices with multi-scale FP16 envelopes, generalizing DBF and LittleBit for aggressive low-bit (sub-1-bit) compression. Supports ADMM/gradient refinement, activation-aware initialization, and a GemLite-accelerated 1-bit inference path. See the [MDBF guide](https://FujitsuResearch.github.io/OneCompression/algorithms/mdbf/) for details.
146
155
  - **Block-wise PTQ**: Post-quantization block-wise distillation that minimises intermediate-representation MSE against an FP16 teacher model at Transformer-block granularity. Includes Phase 1 (greedy per-block optimisation) and Phase 2 CBQ (cross-block sliding-window optimisation). Supports GPTQ, DBF, and OneBit quantizers.
147
156
  - **LoRA SFT Post-Process**: Fine-tune quantized models with LoRA adapters for accuracy recovery or domain-specific knowledge injection. Supports SFT loss, teacher distillation, and intermediate block alignment.
148
157
  - **Rotation Preprocessing**: SpinQuant/OstQuant-based rotation preprocessing that reduces quantization error by learning optimal rotation matrices before quantization. Rotation/scaling matrices are absorbed into model weights, with online Hadamard hooks automatically registered at load time. Supports Llama and Qwen3 architectures.
149
158
  - **Web Dashboard (HPC)**: A browser-based dashboard for launching quantization jobs, deploying models, and validating chat-based inference in HPC environments. See [dashboard/README.md](dashboard/README.md) for details.
159
+ - **GGUF Export & Hugging Face Hub Integration**: Convert models to the GGUF v3 format (F16) for llama.cpp/Ollama with a dependency-free built-in writer, generate model cards with quantization recipes and evaluation results, and push save directories to the Hugging Face Hub. Supports Llama (SentencePiece or Llama-3-style BPE) and Qwen2 (BPE) architectures, including multi-EOS stop-token mapping. See the [GGUF Export guide](https://FujitsuResearch.github.io/OneCompression/user-guide/gguf-export/).
150
160
  - (TBD)
151
161
 
152
162
  ## 🤖 Supported Models
@@ -159,6 +169,8 @@ Other Hugging Face-compatible models may work but are currently untested.
159
169
  | 1 | Llama | TinyLlama, Llama-2, Llama-3 | ✅ Verified |
160
170
  | 2 | Qwen3 | Qwen3-0.6B ~ 32B | ✅ Verified |
161
171
  | 3 | Gemma | Gemma 2, Gemma 3, Gemma 4 | ✅ Verified |
172
+ | 4 | [GPT-OSS](docs/user-guide/gptoss.md) | openai/gpt-oss-20b, gpt-oss-120b | ✅ Verified |
173
+ | 5 | Qwen3.6 | Qwen3.6-27B, Qwen3.6-35B-A3B | ✅ Verified |
162
174
 
163
175
 
164
176
  > **Note:** Support for additional architectures is planned. Contributions and test reports are welcome.
@@ -379,10 +391,19 @@ uv run mkdocs serve
379
391
 
380
392
  Then open [http://127.0.0.1:8000](http://127.0.0.1:8000) in your browser.
381
393
 
394
+ ## 📓 Tutorial Notebook
395
+
396
+ Interactive walkthrough in Jupyter or [Google Colab](https://colab.research.google.com/github/FujitsuResearch/OneCompression/blob/main/notebook/01_tutorial.ipynb)
397
+ — RTN visualization, `Runner.auto_run`, and vLLM chat inference.
398
+
399
+ See [`notebook/README.md`](./notebook/README.md) for local setup, or the
400
+ [Tutorial Notebook guide](https://FujitsuResearch.github.io/OneCompression/getting-started/tutorial-notebook/) in the docs.
401
+
382
402
  ## 🚀 Examples
383
403
 
384
404
  | Category | Script | Description |
385
405
  |----------|--------|-------------|
406
+ | Tutorial | [01_tutorial.ipynb](./notebook/01_tutorial.ipynb) | Interactive notebook (Jupyter / Colab) |
386
407
  | Quantization | [example_gptq.py](./example/example_gptq.py) | GPTQ quantization |
387
408
  | | [example_qep_gptq.py](./example/example_qep_gptq.py) | GPTQ + QEP (error propagation) |
388
409
  | | [example_lpcd_gptq.py](./example/example_lpcd_gptq.py) | GPTQ + QEP + LPCD quantization |
@@ -391,14 +412,25 @@ Then open [http://127.0.0.1:8000](http://127.0.0.1:8000) in your browser.
391
412
  | | [example_auto_run.py](./example/example_auto_run.py) | AutoBit with automatic VRAM estimation |
392
413
  | Calibration | [example_custom_calibration.py](./example/example_custom_calibration.py) | Custom calibration dataset with CalibrationConfig |
393
414
  | Save / Load | [example_save_load.py](./example/example_save_load.py) | Save and load quantized models |
415
+ | | [example_reload_post_process_resave.py](./example/post_process/example_reload_post_process_resave.py) | Reload, post-process, and re-save a quantized checkpoint |
394
416
  | Rotation Preprocessing | [example_llama_preprocess_rtn.py](./example/pre_process/example_llama_preprocess_rtn.py) | Rotation preprocessing + RTN (TinyLlama) |
395
417
  | | [example_preprocess_save_load.py](./example/pre_process/example_preprocess_save_load.py) | Save and load rotation-preprocessed quantized models |
396
- | Post-Process | [example_blockwise_ptq.py](./example/post_process/example_blockwise_ptq.py) | Block-wise PTQ (GPTQ + Phase 1 & CBQ) |
418
+ | Post-Process | [example_blockwise_ptq.py](./example/post_process/example_blockwise_ptq.py) | Block-wise PTQ via `Runner.run()` with packed buffers by default |
419
+ | | [example_blockwise_global_ptq.py](./example/post_process/example_blockwise_global_ptq.py) | BlockWisePTQ → GlobalPTQ in a single Runner, followed by safetensors save/load |
420
+ | | [example_blockwise_global_ptq_staged.py](./example/post_process/example_blockwise_global_ptq_staged.py) | Staged BlockWisePTQ → GlobalPTQ across save/load boundaries with accumulated post-process metadata |
421
+ | | [example_global_ptq.py](./example/post_process/example_global_ptq.py) | Global PTQ with packed buffers by default and HF-compatible safetensors output |
422
+ | | [example_global_ptq_dbf.py](./example/post_process/example_global_ptq_dbf.py) | Global PTQ with the DBF backend and HF-compatible safetensors output |
423
+ | | [example_global_ptq_distributed.py](./example/post_process/example_global_ptq_distributed.py) | Multi-GPU Global PTQ with DeepSpeed / torchrun and safetensors output |
397
424
  | | [example_lora_sft.py](./example/post_process/example_lora_sft.py) | LoRA SFT post-quantization fine-tuning |
398
425
  | | [example_lora_sft_knowledge.py](./example/post_process/example_lora_sft_knowledge.py) | LoRA SFT knowledge injection |
426
+ | | [example_lora_sft_knowledge_jointq.py](./example/post_process/example_lora_sft_knowledge_jointq.py) | LoRA SFT knowledge injection on a JointQ-quantized model |
427
+ | | [example_lora_gptq_vllm_inference.py](./example/post_process/example_lora_gptq_vllm_inference.py) | GPTQ + LoRA SFT, HF-compatible safetensors + PEFT sidecar, and vLLM inference |
399
428
  | vLLM | [example_gptq_vllm_inference.py](./example/vllm_inference/example_gptq_vllm_inference.py) | GPTQ + QEP quantization and vLLM inference |
429
+ | | [example_gptq_vllm_qwen36_inference.py](./example/vllm_inference/example_gptq_vllm_qwen36_inference.py) | GPTQ quantization and vLLM inference for Qwen3.6 (full-wrapper save) |
430
+ | | [example_gptq_vllm_gptoss_inference.py](./example/vllm_inference/example_gptq_vllm_gptoss_inference.py) | GPTQ quantization and vLLM inference for gpt-oss |
400
431
  | | [example_jointq_vllm_inference.py](./example/vllm_inference/example_jointq_vllm_inference.py) | JointQ quantization and vLLM inference |
401
432
  | | [example_autobit_vllm_inference.py](./example/vllm_inference/example_autobit_vllm_inference.py) | AutoBit quantization and vLLM inference |
433
+ | | [example_dbf_vllm_inference.py](./example/vllm_inference/example_dbf_vllm_inference.py) | DBF quantization and vLLM inference |
402
434
 
403
435
  ## 🔌 vLLM Inference
404
436
 
@@ -415,6 +447,22 @@ pip install vllm
415
447
 
416
448
  See the [vLLM Inference guide](https://FujitsuResearch.github.io/OneCompression/user-guide/vllm-inference/) for details, including Open WebUI setup instructions.
417
449
 
450
+ ### GPT-OSS (mixed_gptq)
451
+
452
+ [GPT-OSS models](docs/user-guide/gptoss.md) (`openai/gpt-oss-20b`, `openai/gpt-oss-120b`) need extra steps beyond the
453
+ standard vLLM plugin flow:
454
+
455
+ 1. Quantize with **`GPTQ(wbits=4)`** and keep the MoE experts 4-bit via `Runner(..., moe_quant_experts=True)` with `groupsize=64`. GPT-OSS `hidden_size` (2880) is not divisible by 128, so the experts must use **`group_size=64`** for vLLM's WNA16 MoE kernel.
456
+ 2. Before `LLM(...)`, apply vLLM runtime patches: `python -m vllm_plugins.patches.apply_all`
457
+ 3. Set `export VLLM_USE_DEEP_GEMM=0`
458
+
459
+ The `--extra vllm` dependency set includes **`conch-triton-kernels`**. On NVIDIA Blackwell (B200, sm100),
460
+ vLLM 0.20 needs Conch for `mixed_gptq` linear layers when the Marlin kernel cannot handle GPT-OSS weight shapes.
461
+ If you install vLLM manually with `pip install vllm`, also run `pip install conch-triton-kernels`.
462
+
463
+ See the [GPT-OSS guide](docs/user-guide/gptoss.md) for HF save/load, patch details, and verification scripts.
464
+
465
+
418
466
 
419
467
  ## 📬 Contact Us
420
468
 
@@ -466,3 +514,17 @@ year={2025},
466
514
  url={https://arxiv.org/abs/2512.01546}
467
515
  }
468
516
  ```
517
+
518
+ MDBF (Multi-Envelope Double Binary Factorization):
519
+
520
+ ```
521
+ @misc{ichikawa2025bitsmultienvelopedoublebinary,
522
+ title={More Than Bits: Multi-Envelope Double Binary Factorization for Extreme Quantization},
523
+ author={Yuma Ichikawa and Yoshihiko Fujisawa and Yudai Fujimoto and Akira Sakai and Katsuki Fujisawa},
524
+ year={2025},
525
+ eprint={2512.24545},
526
+ archivePrefix={arXiv},
527
+ primaryClass={cs.LG},
528
+ url={https://arxiv.org/abs/2512.24545},
529
+ }
530
+ ```
@@ -38,10 +38,12 @@ Full documentation is available at **[https://FujitsuResearch.github.io/OneCompr
38
38
  - **vLLM Plugin Integration**: Serve OneComp-quantized models with [vLLM](https://docs.vllm.ai/) via built-in plugins for DBF and Mixed-GPTQ quantization methods. Pair with [Open WebUI](https://github.com/open-webui/open-webui) for a ChatGPT-like chat experience on your local machine.
39
39
  - **AutoBit**: Mixed-precision quantization with ILP-based bitwidth assignment. Automatically estimates the target bitwidth from available VRAM and assigns per-layer bitwidths to minimize quantization error under the memory budget.
40
40
  - **JointQ**: Joint quantization method that optimizes weight assignments and scale parameters simultaneously for improved quantization accuracy. Supports group-wise quantization (e.g., 4-bit, groupsize=128).
41
+ - **MDBF (Multi-Envelope Double Binary Factorization)**: A binary factorization quantizer that approximates each weight matrix as a sum of multi-path sign matrices with multi-scale FP16 envelopes, generalizing DBF and LittleBit for aggressive low-bit (sub-1-bit) compression. Supports ADMM/gradient refinement, activation-aware initialization, and a GemLite-accelerated 1-bit inference path. See the [MDBF guide](https://FujitsuResearch.github.io/OneCompression/algorithms/mdbf/) for details.
41
42
  - **Block-wise PTQ**: Post-quantization block-wise distillation that minimises intermediate-representation MSE against an FP16 teacher model at Transformer-block granularity. Includes Phase 1 (greedy per-block optimisation) and Phase 2 CBQ (cross-block sliding-window optimisation). Supports GPTQ, DBF, and OneBit quantizers.
42
43
  - **LoRA SFT Post-Process**: Fine-tune quantized models with LoRA adapters for accuracy recovery or domain-specific knowledge injection. Supports SFT loss, teacher distillation, and intermediate block alignment.
43
44
  - **Rotation Preprocessing**: SpinQuant/OstQuant-based rotation preprocessing that reduces quantization error by learning optimal rotation matrices before quantization. Rotation/scaling matrices are absorbed into model weights, with online Hadamard hooks automatically registered at load time. Supports Llama and Qwen3 architectures.
44
45
  - **Web Dashboard (HPC)**: A browser-based dashboard for launching quantization jobs, deploying models, and validating chat-based inference in HPC environments. See [dashboard/README.md](dashboard/README.md) for details.
46
+ - **GGUF Export & Hugging Face Hub Integration**: Convert models to the GGUF v3 format (F16) for llama.cpp/Ollama with a dependency-free built-in writer, generate model cards with quantization recipes and evaluation results, and push save directories to the Hugging Face Hub. Supports Llama (SentencePiece or Llama-3-style BPE) and Qwen2 (BPE) architectures, including multi-EOS stop-token mapping. See the [GGUF Export guide](https://FujitsuResearch.github.io/OneCompression/user-guide/gguf-export/).
45
47
  - (TBD)
46
48
 
47
49
  ## 🤖 Supported Models
@@ -54,6 +56,8 @@ Other Hugging Face-compatible models may work but are currently untested.
54
56
  | 1 | Llama | TinyLlama, Llama-2, Llama-3 | ✅ Verified |
55
57
  | 2 | Qwen3 | Qwen3-0.6B ~ 32B | ✅ Verified |
56
58
  | 3 | Gemma | Gemma 2, Gemma 3, Gemma 4 | ✅ Verified |
59
+ | 4 | [GPT-OSS](docs/user-guide/gptoss.md) | openai/gpt-oss-20b, gpt-oss-120b | ✅ Verified |
60
+ | 5 | Qwen3.6 | Qwen3.6-27B, Qwen3.6-35B-A3B | ✅ Verified |
57
61
 
58
62
 
59
63
  > **Note:** Support for additional architectures is planned. Contributions and test reports are welcome.
@@ -274,10 +278,19 @@ uv run mkdocs serve
274
278
 
275
279
  Then open [http://127.0.0.1:8000](http://127.0.0.1:8000) in your browser.
276
280
 
281
+ ## 📓 Tutorial Notebook
282
+
283
+ Interactive walkthrough in Jupyter or [Google Colab](https://colab.research.google.com/github/FujitsuResearch/OneCompression/blob/main/notebook/01_tutorial.ipynb)
284
+ — RTN visualization, `Runner.auto_run`, and vLLM chat inference.
285
+
286
+ See [`notebook/README.md`](./notebook/README.md) for local setup, or the
287
+ [Tutorial Notebook guide](https://FujitsuResearch.github.io/OneCompression/getting-started/tutorial-notebook/) in the docs.
288
+
277
289
  ## 🚀 Examples
278
290
 
279
291
  | Category | Script | Description |
280
292
  |----------|--------|-------------|
293
+ | Tutorial | [01_tutorial.ipynb](./notebook/01_tutorial.ipynb) | Interactive notebook (Jupyter / Colab) |
281
294
  | Quantization | [example_gptq.py](./example/example_gptq.py) | GPTQ quantization |
282
295
  | | [example_qep_gptq.py](./example/example_qep_gptq.py) | GPTQ + QEP (error propagation) |
283
296
  | | [example_lpcd_gptq.py](./example/example_lpcd_gptq.py) | GPTQ + QEP + LPCD quantization |
@@ -286,14 +299,25 @@ Then open [http://127.0.0.1:8000](http://127.0.0.1:8000) in your browser.
286
299
  | | [example_auto_run.py](./example/example_auto_run.py) | AutoBit with automatic VRAM estimation |
287
300
  | Calibration | [example_custom_calibration.py](./example/example_custom_calibration.py) | Custom calibration dataset with CalibrationConfig |
288
301
  | Save / Load | [example_save_load.py](./example/example_save_load.py) | Save and load quantized models |
302
+ | | [example_reload_post_process_resave.py](./example/post_process/example_reload_post_process_resave.py) | Reload, post-process, and re-save a quantized checkpoint |
289
303
  | Rotation Preprocessing | [example_llama_preprocess_rtn.py](./example/pre_process/example_llama_preprocess_rtn.py) | Rotation preprocessing + RTN (TinyLlama) |
290
304
  | | [example_preprocess_save_load.py](./example/pre_process/example_preprocess_save_load.py) | Save and load rotation-preprocessed quantized models |
291
- | Post-Process | [example_blockwise_ptq.py](./example/post_process/example_blockwise_ptq.py) | Block-wise PTQ (GPTQ + Phase 1 & CBQ) |
305
+ | Post-Process | [example_blockwise_ptq.py](./example/post_process/example_blockwise_ptq.py) | Block-wise PTQ via `Runner.run()` with packed buffers by default |
306
+ | | [example_blockwise_global_ptq.py](./example/post_process/example_blockwise_global_ptq.py) | BlockWisePTQ → GlobalPTQ in a single Runner, followed by safetensors save/load |
307
+ | | [example_blockwise_global_ptq_staged.py](./example/post_process/example_blockwise_global_ptq_staged.py) | Staged BlockWisePTQ → GlobalPTQ across save/load boundaries with accumulated post-process metadata |
308
+ | | [example_global_ptq.py](./example/post_process/example_global_ptq.py) | Global PTQ with packed buffers by default and HF-compatible safetensors output |
309
+ | | [example_global_ptq_dbf.py](./example/post_process/example_global_ptq_dbf.py) | Global PTQ with the DBF backend and HF-compatible safetensors output |
310
+ | | [example_global_ptq_distributed.py](./example/post_process/example_global_ptq_distributed.py) | Multi-GPU Global PTQ with DeepSpeed / torchrun and safetensors output |
292
311
  | | [example_lora_sft.py](./example/post_process/example_lora_sft.py) | LoRA SFT post-quantization fine-tuning |
293
312
  | | [example_lora_sft_knowledge.py](./example/post_process/example_lora_sft_knowledge.py) | LoRA SFT knowledge injection |
313
+ | | [example_lora_sft_knowledge_jointq.py](./example/post_process/example_lora_sft_knowledge_jointq.py) | LoRA SFT knowledge injection on a JointQ-quantized model |
314
+ | | [example_lora_gptq_vllm_inference.py](./example/post_process/example_lora_gptq_vllm_inference.py) | GPTQ + LoRA SFT, HF-compatible safetensors + PEFT sidecar, and vLLM inference |
294
315
  | vLLM | [example_gptq_vllm_inference.py](./example/vllm_inference/example_gptq_vllm_inference.py) | GPTQ + QEP quantization and vLLM inference |
316
+ | | [example_gptq_vllm_qwen36_inference.py](./example/vllm_inference/example_gptq_vllm_qwen36_inference.py) | GPTQ quantization and vLLM inference for Qwen3.6 (full-wrapper save) |
317
+ | | [example_gptq_vllm_gptoss_inference.py](./example/vllm_inference/example_gptq_vllm_gptoss_inference.py) | GPTQ quantization and vLLM inference for gpt-oss |
295
318
  | | [example_jointq_vllm_inference.py](./example/vllm_inference/example_jointq_vllm_inference.py) | JointQ quantization and vLLM inference |
296
319
  | | [example_autobit_vllm_inference.py](./example/vllm_inference/example_autobit_vllm_inference.py) | AutoBit quantization and vLLM inference |
320
+ | | [example_dbf_vllm_inference.py](./example/vllm_inference/example_dbf_vllm_inference.py) | DBF quantization and vLLM inference |
297
321
 
298
322
  ## 🔌 vLLM Inference
299
323
 
@@ -310,6 +334,22 @@ pip install vllm
310
334
 
311
335
  See the [vLLM Inference guide](https://FujitsuResearch.github.io/OneCompression/user-guide/vllm-inference/) for details, including Open WebUI setup instructions.
312
336
 
337
+ ### GPT-OSS (mixed_gptq)
338
+
339
+ [GPT-OSS models](docs/user-guide/gptoss.md) (`openai/gpt-oss-20b`, `openai/gpt-oss-120b`) need extra steps beyond the
340
+ standard vLLM plugin flow:
341
+
342
+ 1. Quantize with **`GPTQ(wbits=4)`** and keep the MoE experts 4-bit via `Runner(..., moe_quant_experts=True)` with `groupsize=64`. GPT-OSS `hidden_size` (2880) is not divisible by 128, so the experts must use **`group_size=64`** for vLLM's WNA16 MoE kernel.
343
+ 2. Before `LLM(...)`, apply vLLM runtime patches: `python -m vllm_plugins.patches.apply_all`
344
+ 3. Set `export VLLM_USE_DEEP_GEMM=0`
345
+
346
+ The `--extra vllm` dependency set includes **`conch-triton-kernels`**. On NVIDIA Blackwell (B200, sm100),
347
+ vLLM 0.20 needs Conch for `mixed_gptq` linear layers when the Marlin kernel cannot handle GPT-OSS weight shapes.
348
+ If you install vLLM manually with `pip install vllm`, also run `pip install conch-triton-kernels`.
349
+
350
+ See the [GPT-OSS guide](docs/user-guide/gptoss.md) for HF save/load, patch details, and verification scripts.
351
+
352
+
313
353
 
314
354
  ## 📬 Contact Us
315
355
 
@@ -361,3 +401,17 @@ year={2025},
361
401
  url={https://arxiv.org/abs/2512.01546}
362
402
  }
363
403
  ```
404
+
405
+ MDBF (Multi-Envelope Double Binary Factorization):
406
+
407
+ ```
408
+ @misc{ichikawa2025bitsmultienvelopedoublebinary,
409
+ title={More Than Bits: Multi-Envelope Double Binary Factorization for Extreme Quantization},
410
+ author={Yuma Ichikawa and Yoshihiko Fujisawa and Yudai Fujimoto and Akira Sakai and Katsuki Fujisawa},
411
+ year={2025},
412
+ eprint={2512.24545},
413
+ archivePrefix={arXiv},
414
+ primaryClass={cs.LG},
415
+ url={https://arxiv.org/abs/2512.24545},
416
+ }
417
+ ```
@@ -0,0 +1,12 @@
1
+ """ROCm-only workaround for vLLM 0.24.0 TritonW4A16LinearKernel.
2
+
3
+ Copyright 2025-2026 Fujitsu Ltd.
4
+
5
+ The entry point installed via ``pip install``
6
+ (declared in this package's ``pyproject.toml``) is
7
+ :func:`onecomp_vllm_v0_24_0_rocm.patch.apply`.
8
+ """
9
+
10
+ from .patch import apply
11
+
12
+ __all__ = ["apply"]
@@ -0,0 +1,238 @@
1
+ """ROCm-only workaround for vLLM 0.24.x TritonW4A16LinearKernel.
2
+
3
+ Copyright 2025-2026 Fujitsu Ltd.
4
+
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+ _APPLIED_MARKER_AUTO_GPTQ = "_onecomp_vllm_0_24_0_rocm_applied_auto_gptq"
10
+ _APPLIED_MARKER_KERNEL = "_onecomp_vllm_0_24_0_rocm_applied_kernel"
11
+
12
+ # Set after the first inference-time fixup so logs stay readable (once per process).
13
+ _LOGGED_AUTO_GPTQ_QZEROS_FIXUP = False
14
+ _LOGGED_TRITON_QZEROS_PERMUTE = False
15
+
16
+ # vLLM only attaches handlers to the ``vllm`` logger tree; ``__name__`` logs are silently dropped.
17
+ # Use a child of ``vllm`` so INFO lines appear in engine output (including EngineCore subprocesses).
18
+ _VLLM_LOGGER_NAME = "vllm.onecomp_v0_24_0_rocm"
19
+
20
+
21
+ def _get_logger():
22
+ from vllm.logger import init_logger
23
+
24
+ return init_logger(_VLLM_LOGGER_NAME)
25
+
26
+
27
+ def _is_target_env() -> tuple[bool, str]:
28
+ """Return ``(is_target, reason)`` for the current Python / vLLM env."""
29
+ try:
30
+ import vllm # noqa: F401
31
+ except ImportError as exc:
32
+ return False, f"vllm not importable ({exc})"
33
+ try:
34
+ from vllm.platforms import current_platform
35
+ except ImportError as exc:
36
+ return False, f"vllm.platforms not importable ({exc})"
37
+
38
+ if not current_platform.is_rocm():
39
+ return False, "not ROCm"
40
+
41
+ # Accept "0.24.0", "0.24.0+rocm723", "0.24.1", "0.24.0rc3.dev3", etc.
42
+ base_version = vllm.__version__.split("+", 1)[0]
43
+ if not base_version.startswith("0.24."):
44
+ return False, f"vllm base version {base_version!r} not in 0.24.x"
45
+
46
+ return True, "vllm 0.24.x on ROCm"
47
+
48
+
49
+ def _unbias_v1_zeros(zp_packed):
50
+ """Add 1 (mod 16) to every 4-bit nibble of a GPTQ-packed int32 tensor.
51
+
52
+ Parameters
53
+ ----------
54
+ zp_packed
55
+ ``[K // G, N // 8]`` int32 tensor. GPTQ sequential packing: each
56
+ int32 holds 8 consecutive N-values at bit offsets
57
+ ``[0, 4, 8, ..., 28]``.
58
+
59
+ Returns
60
+ -------
61
+ Same shape / dtype / device, with every nibble incremented by 1 mod 16.
62
+ """
63
+ import torch
64
+
65
+ shifts = torch.arange(8, device=zp_packed.device, dtype=torch.int32) * 4
66
+ nibbles = ((zp_packed.unsqueeze(-1) >> shifts) & 0xF) + 1
67
+ nibbles &= 0xF
68
+ return torch.sum(nibbles << shifts, dim=-1, dtype=torch.int32)
69
+
70
+
71
+ def _patch_auto_gptq_process_weights(logger) -> None:
72
+ """Wrap ``AutoGPTQLinearMethod.process_weights_after_loading`` with a +1
73
+ fixup on ``qzeros`` for the GPTQv1 convention.
74
+
75
+ Rationale
76
+ ---------
77
+ GPTQv1 checkpoints store ``stored_zero = real_zero - 1``. Marlin/Machete
78
+ kernels re-add the ``-1`` bias internally, but ``TritonW4A16LinearKernel``
79
+ consumes ``qzeros`` verbatim, so we need to add 1 (mod 16) to every
80
+ packed nibble before the kernel sees the parameter. This is scoped to
81
+ the AutoGPTQ path (i.e. GPTQv1 checkpoints); other quantization methods
82
+ that reuse ``TritonW4A16LinearKernel`` (e.g. compressed-tensors) do not
83
+ have the ``-1`` convention and are left untouched.
84
+ """
85
+ from vllm.model_executor.kernels.linear.mixed_precision.triton_w4a16 import (
86
+ TritonW4A16LinearKernel,
87
+ )
88
+ from vllm.model_executor.layers.quantization.auto_gptq import (
89
+ AutoGPTQConfig,
90
+ AutoGPTQLinearMethod,
91
+ )
92
+ from vllm.scalar_type import scalar_types
93
+
94
+ # Additive: unblock asymmetric GPTQ checkpoints. Stock vLLM 0.24 still
95
+ # only registers (n_bits, True) in TYPE_MAP, which makes AutoGPTQConfig
96
+ # .from_config raise ValueError for sym=False models. TritonW4A16
97
+ # already advertises uint4 (asymmetric) in SUPPORTED_QUANT_TYPES, so
98
+ # this extension does not change kernel selection for sym=True.
99
+ AutoGPTQConfig.TYPE_MAP.setdefault((4, False), scalar_types.uint4)
100
+ AutoGPTQConfig.TYPE_MAP.setdefault((8, False), scalar_types.uint8)
101
+
102
+ if getattr(
103
+ AutoGPTQLinearMethod.process_weights_after_loading,
104
+ _APPLIED_MARKER_AUTO_GPTQ,
105
+ False,
106
+ ):
107
+ logger.debug("onecomp env-patch already installed on AutoGPTQLinearMethod; skip")
108
+ return
109
+
110
+ _orig = AutoGPTQLinearMethod.process_weights_after_loading
111
+
112
+ def _patched(self, layer):
113
+ global _LOGGED_AUTO_GPTQ_QZEROS_FIXUP
114
+ # +1 fixup runs BEFORE the kernel's process_weights_after_loading so
115
+ # that we operate on the AutoGPTQ-native layout [K//G, N//8] where
116
+ # the packed axis is dim -1 (which is what ``_unbias_v1_zeros``
117
+ # assumes). The subsequent shape/permute juggling in the kernel
118
+ # PWAL preserves per-nibble values.
119
+ if isinstance(self.kernel, TritonW4A16LinearKernel):
120
+ zp = getattr(layer, "qzeros", None)
121
+ if zp is not None and getattr(zp, "data", None) is not None:
122
+ zp.data = _unbias_v1_zeros(zp.data)
123
+ if not _LOGGED_AUTO_GPTQ_QZEROS_FIXUP:
124
+ _LOGGED_AUTO_GPTQ_QZEROS_FIXUP = True
125
+ logger.info(
126
+ "onecomp env-patch active at inference: "
127
+ "AutoGPTQLinearMethod qzeros +1 fixup "
128
+ "(TritonW4A16LinearKernel, qzeros shape=%s)",
129
+ tuple(zp.data.shape),
130
+ )
131
+ return _orig(self, layer)
132
+
133
+ setattr(_patched, _APPLIED_MARKER_AUTO_GPTQ, True)
134
+ AutoGPTQLinearMethod.process_weights_after_loading = _patched
135
+ logger.info("onecomp env-patch: wrapped AutoGPTQLinearMethod.process_weights_after_loading")
136
+
137
+
138
+ def _patch_triton_w4a16_kernel_process_weights(logger) -> None:
139
+ """Wrap ``TritonW4A16LinearKernel.process_weights_after_loading`` to fix
140
+ a missing ``permute_param_layout_`` call on ``qzeros``.
141
+
142
+ Rationale
143
+ ---------
144
+ In vLLM 0.24.x the stock ``TritonW4A16LinearKernel.process_weights_after_loading``
145
+ normalizes ``qweight`` and ``scales`` via ``permute_param_layout_`` so
146
+ that both compressed-tensors (``output_dim=0, packed_dim=0``) and
147
+ AutoGPTQ (``input_dim=0, output_dim=1, packed_dim=1``) checkpoint
148
+ layouts land in the same physical form before the ``.t().contiguous()``
149
+ that the kernel expects. The block for ``qzeros`` skips this step and
150
+ calls ``.t()`` unconditionally::
151
+
152
+ # Checkpoint: [N//8, K//G] int32 (N packed at dim 0, K//G at dim 1)
153
+ # Kernel needs: [K//G, N//8] -- just transpose
154
+ replace_parameter(
155
+ layer, self.w_zp_name,
156
+ torch.nn.Parameter(zp.data.t().contiguous(), requires_grad=False),
157
+ )
158
+
159
+ That works only for the compressed-tensors layout. For an AutoGPTQ
160
+ checkpoint the parameter is already ``[K//G, N//8]``, so the ``.t()``
161
+ flips it to ``[N//8, K//G]`` and the downstream shape assertion in
162
+ ``triton_w4a16_gemm`` fires (observed as
163
+ ``AssertionError: qzeros shape mismatch: torch.Size([N//8, K//G])``).
164
+
165
+ The one-liner upstream is missing is a ``permute_param_layout_(zp,
166
+ input_dim=1, output_dim=0, packed_dim=0)`` call before the ``.t()``.
167
+ Injecting it via a wrapper here fixes AutoGPTQ without regressing the
168
+ compressed-tensors path (that path already satisfies the target
169
+ layout, so the permute is a no-op).
170
+ """
171
+ from vllm.model_executor.kernels.linear.mixed_precision.triton_w4a16 import (
172
+ TritonW4A16LinearKernel,
173
+ )
174
+ from vllm.model_executor.parameter import permute_param_layout_
175
+
176
+ if getattr(
177
+ TritonW4A16LinearKernel.process_weights_after_loading,
178
+ _APPLIED_MARKER_KERNEL,
179
+ False,
180
+ ):
181
+ logger.debug("onecomp env-patch already installed on TritonW4A16LinearKernel; skip")
182
+ return
183
+
184
+ _orig = TritonW4A16LinearKernel.process_weights_after_loading
185
+
186
+ def _patched(self, layer):
187
+ global _LOGGED_TRITON_QZEROS_PERMUTE
188
+ if self.w_zp_name is not None:
189
+ zp = getattr(layer, self.w_zp_name, None)
190
+ if zp is not None:
191
+ # Normalise both compressed-tensors and AutoGPTQ layouts to
192
+ # [N//8, K//G] with (input_dim=1, output_dim=0, packed_dim=0)
193
+ # so the stock code's subsequent zp.data.t().contiguous()
194
+ # ends up at the kernel's expected [K//G, N//8].
195
+ permute_param_layout_(zp, input_dim=1, output_dim=0, packed_dim=0)
196
+ if not _LOGGED_TRITON_QZEROS_PERMUTE:
197
+ _LOGGED_TRITON_QZEROS_PERMUTE = True
198
+ logger.info(
199
+ "onecomp env-patch active at inference: "
200
+ "TritonW4A16LinearKernel qzeros permute_param_layout_ "
201
+ "(param=%s, shape=%s)",
202
+ self.w_zp_name,
203
+ tuple(zp.data.shape),
204
+ )
205
+ return _orig(self, layer)
206
+
207
+ setattr(_patched, _APPLIED_MARKER_KERNEL, True)
208
+ TritonW4A16LinearKernel.process_weights_after_loading = _patched
209
+ logger.info("onecomp env-patch: wrapped TritonW4A16LinearKernel.process_weights_after_loading")
210
+
211
+
212
+ def apply() -> None:
213
+ """vLLM ``vllm.general_plugins`` entry point.
214
+
215
+ No-op on every platform / version other than ROCm + vLLM 0.24.x.
216
+ """
217
+ logger = _get_logger()
218
+
219
+ is_target, reason = _is_target_env()
220
+ if not is_target:
221
+ logger.debug("onecomp env-patch vllm_v0_24_0_rocm skipped: %s", reason)
222
+ return
223
+
224
+ import vllm
225
+
226
+ logger.info(
227
+ "onecomp env-patch: loading for %s (vllm %s)",
228
+ reason,
229
+ vllm.__version__,
230
+ )
231
+ _patch_auto_gptq_process_weights(logger)
232
+ _patch_triton_w4a16_kernel_process_weights(logger)
233
+ logger.info(
234
+ "onecomp env-patch installed: "
235
+ "AutoGPTQLinearMethod qzeros +1 fixup for TritonW4A16LinearKernel; "
236
+ "TritonW4A16LinearKernel qzeros layout permute injected; "
237
+ "AutoGPTQConfig.TYPE_MAP extended with asymmetric uint4/uint8.",
238
+ )
@@ -0,0 +1,57 @@
1
+ """GPTQ quantization -> direct GGUF export -> CPU inference with llama.cpp.
2
+
3
+ This example shows the OneComp CPU inference path:
4
+ 1. Quantize a model with GPTQ (4-bit symmetric, group size 128).
5
+ 2. Export it to GGUF without re-quantization (preserving the GPTQ codes).
6
+ 3. Run CPU text generation via llama-cpp-python.
7
+
8
+ Only the GGUF export and inference are CPU-bound; quantization runs on the GPU
9
+ when one is available and falls back to CPU (ModelConfig default
10
+ ``device="auto"``; both are supported).
11
+
12
+ Run:
13
+ python example/cpu_inference/example_gptq_gguf_cpu.py
14
+
15
+ Requires: pip install 'onecomp[llamacpp]' (gguf + llama-cpp-python)
16
+
17
+ Copyright 2025-2026 Fujitsu Ltd.
18
+
19
+ Author: Yuma Ichikawa
20
+
21
+ """
22
+
23
+ import os
24
+
25
+ from onecomp import GPTQ, CalibrationConfig, ModelConfig, Runner
26
+ from onecomp.cpu import LlamaCppModel, convert_gptq_to_gguf
27
+ from onecomp.log import setup_logger
28
+
29
+ MODEL_ID = "TinyLlama/TinyLlama-1.1B-intermediate-step-1431k-3T"
30
+ SAVE_DIR = "./TinyLlama-1.1B-gptq-4bit"
31
+ GGUF_PATH = "./TinyLlama-1.1B-gptq-4bit.gguf"
32
+
33
+
34
+ def main():
35
+ setup_logger()
36
+
37
+ # 1. Quantize (4-bit symmetric, group size 128, no actorder => lossless to Q4_0).
38
+ runner = Runner(
39
+ model_config=ModelConfig(model_id=MODEL_ID),
40
+ quantizer=GPTQ(wbits=4, groupsize=128, sym=True, actorder=False),
41
+ calibration_config=CalibrationConfig(num_calibration_samples=64, max_length=512),
42
+ qep=True,
43
+ )
44
+ runner.run()
45
+ runner.save_quantized_model(SAVE_DIR)
46
+
47
+ # 2. Direct GGUF export (no re-quantization; QEP-corrected codes preserved).
48
+ summary = convert_gptq_to_gguf(quantized_dir=SAVE_DIR, out_gguf=GGUF_PATH)
49
+ print("Export summary:", summary)
50
+
51
+ # 3. CPU inference.
52
+ model = LlamaCppModel(GGUF_PATH, n_ctx=1024)
53
+ print(model.generate("Fujitsu is", max_tokens=64, temperature=0.0))
54
+
55
+
56
+ if __name__ == "__main__":
57
+ main()
@@ -0,0 +1,63 @@
1
+ """Example: mixed-bit GPTQ -> mixed-precision GGUF -> llama.cpp CPU inference.
2
+
3
+ Quantizes a model with per-module bit-widths (4-bit attention, 8-bit MLP gate,
4
+ 3-bit MLP up, 2-bit MLP down) in a single GPTQ run, then exports it to ONE GGUF
5
+ where each tensor carries its own quantization type:
6
+
7
+ 4-bit sym -> Q4_0 8-bit sym -> Q8_0 3-bit -> Q3_K 2-bit -> Q2_K
8
+
9
+ The 4/8-bit layers keep the exact GPTQ codes (lossless); the 2/3-bit layers are
10
+ re-quantized to K-quants via llama-quantize (requires the llama-quantize binary).
11
+
12
+ Copyright 2025-2026 Fujitsu Ltd.
13
+
14
+ Author: Yuma Ichikawa
15
+ """
16
+
17
+ import json
18
+ import os
19
+
20
+ from transformers import AutoConfig
21
+
22
+ from llamacpp_plugins.gptq import export_mixed_gptq_gguf, plan_mixed_export
23
+ from onecomp import GPTQ, CalibrationConfig, ModelConfig, Runner, setup_logger
24
+ from onecomp.cpu import LlamaCppModel
25
+
26
+ MODEL_ID = "TinyLlama/TinyLlama-1.1B-intermediate-step-1431k-3T"
27
+ OUT_DIR = "./results/mixed-gptq"
28
+ OUT_GGUF = "./results/mixed-gptq.gguf"
29
+
30
+
31
+ def main():
32
+ setup_logger()
33
+
34
+ num_layers = AutoConfig.from_pretrained(MODEL_ID).num_hidden_layers
35
+ module_wbits = {}
36
+ for i in range(num_layers):
37
+ module_wbits[f"model.layers.{i}.mlp.gate_proj"] = 8
38
+ module_wbits[f"model.layers.{i}.mlp.up_proj"] = 3
39
+ module_wbits[f"model.layers.{i}.mlp.down_proj"] = 2
40
+
41
+ quantizer = GPTQ(wbits=4, groupsize=128, sym=True, module_wbits=module_wbits)
42
+ runner = Runner(
43
+ model_config=ModelConfig(model_id=MODEL_ID, device="cpu"),
44
+ quantizer=quantizer,
45
+ calibration_config=CalibrationConfig(num_calibration_samples=32, max_length=512),
46
+ qep=False,
47
+ )
48
+ runner.run()
49
+ runner.save_quantized_model(OUT_DIR)
50
+
51
+ # Preview the per-module GGUF routing.
52
+ for p in plan_mixed_export(OUT_DIR)[:6]:
53
+ print(f"{p.name:45s} bits={p.bits} -> {p.route}/{p.ggml_type}")
54
+
55
+ summary = export_mixed_gptq_gguf(OUT_DIR, OUT_GGUF)
56
+ print(json.dumps(summary["plan"], indent=2))
57
+
58
+ model = LlamaCppModel(OUT_GGUF, n_ctx=512)
59
+ print(model.generate("The capital of Japan is", max_tokens=32, temperature=0.0))
60
+
61
+
62
+ if __name__ == "__main__":
63
+ main()
@@ -0,0 +1,43 @@
1
+ """Serve a packed OneComp checkpoint (or GGUF) on CPU with one command.
2
+
3
+ The barrier-free path: drop in a *packed quantized* model directory and anyone
4
+ can run it on CPU behind an OpenAI-compatible API. If given a packed OneComp
5
+ checkpoint, the GGUF is auto-exported (losslessly, cached) on first launch.
6
+
7
+ Run:
8
+ # Packed OneComp GPTQ checkpoint (auto-exports to GGUF once):
9
+ python example/cpu_inference/example_serve_cpu.py ./TinyLlama-1.1B-gptq-4bit
10
+ # ...or an existing GGUF:
11
+ python example/cpu_inference/example_serve_cpu.py ./model.gguf
12
+
13
+ Then, from another shell:
14
+ curl http://localhost:8080/v1/chat/completions \
15
+ -d '{"messages":[{"role":"user","content":"Hello!"}],"max_tokens":64}'
16
+
17
+ Equivalent CLI: onecomp-gguf serve --model <path> --port 8080
18
+
19
+ Requires: pip install 'onecomp[llamacpp]' (no FastAPI/uvicorn needed)
20
+
21
+ Copyright 2025-2026 Fujitsu Ltd.
22
+
23
+ Author: Yuma Ichikawa
24
+
25
+ """
26
+
27
+ import sys
28
+
29
+ from onecomp.cpu import serve
30
+ from onecomp.log import setup_logger
31
+
32
+
33
+ def main():
34
+ setup_logger()
35
+ model_path = sys.argv[1] if len(sys.argv) > 1 else "./TinyLlama-1.1B-gptq-4bit"
36
+ port = int(sys.argv[2]) if len(sys.argv) > 2 else 8080
37
+ # host=0.0.0.0 to expose on the network; chat_format is auto-detected from
38
+ # the GGUF architecture when no chat template is embedded.
39
+ serve(model_path, host="0.0.0.0", port=port, n_ctx=2048)
40
+
41
+
42
+ if __name__ == "__main__":
43
+ main()