onecomp 1.2.0__tar.gz → 1.2.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {onecomp-1.2.0/onecomp.egg-info → onecomp-1.2.2}/PKG-INFO +7 -1
- {onecomp-1.2.0 → onecomp-1.2.2}/README.md +6 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/example/post_process/example_lora_sft.py +5 -1
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/__version__.py +1 -1
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/model_config.py +7 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/post_process/post_process_lora_sft.py +2 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/quantized_model_loader.py +41 -1
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/quantizer/_quantizer.py +44 -3
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/runner.py +8 -8
- {onecomp-1.2.0 → onecomp-1.2.2/onecomp.egg-info}/PKG-INFO +7 -1
- {onecomp-1.2.0 → onecomp-1.2.2}/LICENSE +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/benchmark/llama3-8b-gptq/quant_benchmark.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/benchmark/llama3-8b-jointq/quant_benchmark.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/benchmark/llama3-8b-lpcd-gptq/quant_benchmark.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/benchmark/llama3-8b-qep-gptq/quant_benchmark.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/benchmark/llama3-8b-various/quant_benchmark.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/benchmark/qwen3-14b-gptq/quant_benchmark.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/benchmark/qwen3-14b-jointq/quant_benchmark.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/benchmark/qwen3-8b-gptq/quant_benchmark.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/benchmark/qwen3-8b-jointq/quant_benchmark.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/dashboard/backend/app/__init__.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/dashboard/backend/app/api/__init__.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/dashboard/backend/app/api/jobs.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/dashboard/backend/app/constants.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/dashboard/backend/app/core/__init__.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/dashboard/backend/app/core/config.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/dashboard/backend/app/core/database.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/dashboard/backend/app/main.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/dashboard/backend/app/models/__init__.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/dashboard/backend/app/models/job.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/dashboard/backend/app/schemas/__init__.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/dashboard/backend/app/schemas/job.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/dashboard/backend/app/services/__init__.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/dashboard/backend/app/services/huggingface.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/dashboard/backend/app/services/inference.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/dashboard/backend/app/services/job_store.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/dashboard/backend/app/worker/__init__.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/dashboard/backend/app/worker/celery_app.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/dashboard/backend/app/worker/tasks.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/dashboard/backend/cpu_patch.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/dashboard/backend/start_backend.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/dashboard/backend/start_worker.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/example/example_auto_run.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/example/example_autobit.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/example/example_custom_calibration.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/example/example_gptq.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/example/example_jointq.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/example/example_lpcd_gptq.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/example/example_qep_gptq.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/example/example_save_load.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/example/post_process/example_blockwise_ptq.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/example/post_process/example_global_ptq.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/example/post_process/example_global_ptq_dbf.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/example/post_process/example_global_ptq_distributed.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/example/post_process/example_lora_sft_knowledge.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/example/pre_process/example_llama_preprocess_rtn.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/example/pre_process/example_preprocess_save_load.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/example/vllm_inference/example_autobit_vllm_inference.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/example/vllm_inference/example_gptq_vllm_inference.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/example/vllm_inference/example_jointq_vllm_inference.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/model_validation/autobit/validate_autobit.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/model_validation/autobit_qep/validate_autobit.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/model_validation/gptq/validate_gptq.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/model_validation/gptq/validate_load.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/model_validation/gptq/validate_vllm.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/model_validation/jointq/validate_jointq.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/model_validation/qep_gptq/validate_gptq.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/__init__.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/__main__.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/analyzer/__init__.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/analyzer/cumulative_error.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/analyzer/quantization_error.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/analyzer/weight_outlier.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/calibration/__init__.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/calibration/_cache.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/calibration/c4.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/calibration/calibration_config.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/calibration/calibration_data_loader.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/calibration/chunking.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/calibration/custom.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/calibration/wikitext.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/cli.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/eval/__init__.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/eval/__main__.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/eval/conf/__init__.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/eval/conf/eval_config.yaml +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/eval/evals/__init__.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/eval/evals/base.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/eval/evals/mt_bench/__init__.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/eval/evals/mt_bench/adapter.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/eval/evals/mt_bench/data.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/eval/evals/mt_bench/gen_answer.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/eval/evals/mt_bench/judge.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/eval/evals/mt_bench/radar_chart.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/eval/evals/mt_bench/run.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/eval/evals/mt_bench/show_result.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/eval/evals/throughput/__init__.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/eval/evals/throughput/adapter.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/eval/evals/throughput/bench.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/eval/evals/throughput/run.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/eval/orchestrator/__init__.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/eval/orchestrator/aggregator.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/eval/orchestrator/runner.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/eval/orchestrator/server.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/eval/orchestrator/subprocess_runner.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/eval/run_evaluate.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/eval/schema.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/eval/utils/__init__.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/eval/utils/model_utils.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/eval/utils/ports.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/eval/utils/resources.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/eval/utils/secrets.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/log.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/lpcd/__init__.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/lpcd/_gradient_solver.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/lpcd/_lpcd_config.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/lpcd/_lpcd_runner.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/lpcd/_metric.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/lpcd/_refiner.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/lpcd/arch/_llama.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/lpcd/arch/_llama_cf.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/lpcd/arch/_qwen3.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/post_process/__init__.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/post_process/_base.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/post_process/_blockwise/__init__.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/post_process/_blockwise/dbf_block_optimizer.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/post_process/_blockwise/dbf_cbq_optimizer.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/post_process/_blockwise/generic_block_optimizer.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/post_process/_blockwise/gptq_block_optimizer.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/post_process/_blockwise/gptq_cbq_optimizer.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/post_process/_blockwise/helpers.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/post_process/_blockwise/onebit_block_optimizer.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/post_process/_blockwise/onebit_cbq_optimizer.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/post_process/_global_ptq/__init__.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/post_process/_global_ptq/core.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/post_process/_global_ptq/dbf_adapter.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/post_process/_global_ptq/gptq_adapter.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/post_process/_global_ptq/helpers.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/post_process/_global_ptq/losses.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/post_process/_global_ptq/trainer.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/post_process/blockwise_ptq.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/post_process/global_ptq.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/post_process/global_ptq_distributed.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/pre_process/__init__.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/pre_process/hadamard_utils.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/pre_process/modeling_llama.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/pre_process/modeling_qwen3.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/pre_process/optimizer.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/pre_process/prepare_rotated_model.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/pre_process/preprocess_args.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/pre_process/quant_models.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/pre_process/rotation_utils.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/pre_process/train_rotation.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/qep/__init__.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/qep/_qep_config.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/qep/_quantize_with_qep.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/qep/_quantize_with_qep_arch.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/quantizer/__init__.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/quantizer/arb/__init__.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/quantizer/arb/_arb.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/quantizer/arb/arb_impl.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/quantizer/autobit/__init__.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/quantizer/autobit/_autobit.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/quantizer/autobit/activation_stats.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/quantizer/autobit/dbf_fallback.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/quantizer/autobit/ilp.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/quantizer/autobit/manual.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/quantizer/autobit/visualize.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/quantizer/cq/__init__.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/quantizer/cq/_cq.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/quantizer/cq/cq_impl.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/quantizer/dbf/__init__.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/quantizer/dbf/_dbf.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/quantizer/dbf/admm_extended.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/quantizer/dbf/balance.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/quantizer/dbf/config.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/quantizer/dbf/dbf_impl.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/quantizer/dbf/dbf_layer.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/quantizer/dbf/dbf_original.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/quantizer/dbf/fine_tune.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/quantizer/dbf/middle.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/quantizer/gemlite.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/quantizer/gptq/__init__.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/quantizer/gptq/_gptq.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/quantizer/gptq/config.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/quantizer/gptq/gptq_layer.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/quantizer/jointq/__init__.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/quantizer/jointq/_jointq.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/quantizer/jointq/core/__init__.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/quantizer/jointq/core/__version__.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/quantizer/jointq/core/clip.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/quantizer/jointq/core/error_propagation/__init__.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/quantizer/jointq/core/error_propagation/local_search_advanced.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/quantizer/jointq/core/error_propagation/quantize_advanced.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/quantizer/jointq/core/error_propagation/quantizer_advanced.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/quantizer/jointq/core/local_search.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/quantizer/jointq/core/quantize.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/quantizer/jointq/core/quantize_multi_gpu.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/quantizer/jointq/core/quantizer.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/quantizer/jointq/core/solution.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/quantizer/onebit/__init__.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/quantizer/onebit/_onebit.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/quantizer/onebit/onebit_impl.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/quantizer/onebit/onebit_layer.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/quantizer/qbb/__init__.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/quantizer/qbb/_qbb.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/quantizer/qbb/qbb_impl.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/quantizer/quip/__init__.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/quantizer/quip/_quip.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/quantizer/quip/quant_quip.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/quantizer/quip/quip_impl.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/quantizer/quip/utils.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/quantizer/quip/utils_had.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/quantizer/quip/vector_balance.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/quantizer/rtn/__init__.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/quantizer/rtn/_rtn.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/quantizer/rtn/quantizer.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/quantizer/rtn/rtn_impl.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/rotated_model_config.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/runner_methods/__init__.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/runner_methods/chunked_quantization.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/runner_methods/jointq_error_propagation.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/runner_methods/multi_gpu_quantization.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/utils/__init__.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/utils/accuracy.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/utils/activation_capture.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/utils/activation_check.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/utils/blockwise.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/utils/device.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/utils/dtype.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/utils/model_inputs.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/utils/perplexity.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/utils/quant_config.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/utils/quantization_progress.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/utils/unfuse_moe.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp/utils/vram_estimator.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp.egg-info/SOURCES.txt +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp.egg-info/dependency_links.txt +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp.egg-info/entry_points.txt +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp.egg-info/requires.txt +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/onecomp.egg-info/top_level.txt +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/pyproject.toml +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/scripts/check_copyright_header.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/scripts/check_no_japanese.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/setup.cfg +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/vllm_plugins/__init__.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/vllm_plugins/dbf/__init__.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/vllm_plugins/dbf/modules/__init__.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/vllm_plugins/dbf/modules/gemlite_linear.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/vllm_plugins/dbf/modules/naive.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/vllm_plugins/dbf/vllm_plugin.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/vllm_plugins/gptq/__init__.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/vllm_plugins/gptq/constants.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/vllm_plugins/gptq/vllm_plugin.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/vllm_plugins/utils/__init__.py +0 -0
- {onecomp-1.2.0 → onecomp-1.2.2}/vllm_plugins/utils/module.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: onecomp
|
|
3
|
-
Version: 1.2.
|
|
3
|
+
Version: 1.2.2
|
|
4
4
|
Summary: Python package for LLM compression
|
|
5
5
|
Author: Keiji Kimura
|
|
6
6
|
License: MIT License
|
|
@@ -416,6 +416,12 @@ pip install vllm
|
|
|
416
416
|
See the [vLLM Inference guide](https://FujitsuResearch.github.io/OneCompression/user-guide/vllm-inference/) for details, including Open WebUI setup instructions.
|
|
417
417
|
|
|
418
418
|
|
|
419
|
+
## 📬 Contact Us
|
|
420
|
+
|
|
421
|
+
- For technical questions and feature requests, please use GitHub [Issues](https://github.com/FujitsuResearch/OneCompression/issues).
|
|
422
|
+
- For security vulnerabilities, please **do not** open a public Issue. See our [Security Policy](./SECURITY.md) for how to report them privately.
|
|
423
|
+
- For collaborations, partnerships, and other inquiries, please contact us at [contact-onecompression@cs.jp.fujitsu.com](mailto:contact-onecompression@cs.jp.fujitsu.com).
|
|
424
|
+
|
|
419
425
|
## 📄 License
|
|
420
426
|
|
|
421
427
|
See [LICENSE](./LICENSE) for more details.
|
|
@@ -311,6 +311,12 @@ pip install vllm
|
|
|
311
311
|
See the [vLLM Inference guide](https://FujitsuResearch.github.io/OneCompression/user-guide/vllm-inference/) for details, including Open WebUI setup instructions.
|
|
312
312
|
|
|
313
313
|
|
|
314
|
+
## 📬 Contact Us
|
|
315
|
+
|
|
316
|
+
- For technical questions and feature requests, please use GitHub [Issues](https://github.com/FujitsuResearch/OneCompression/issues).
|
|
317
|
+
- For security vulnerabilities, please **do not** open a public Issue. See our [Security Policy](./SECURITY.md) for how to report them privately.
|
|
318
|
+
- For collaborations, partnerships, and other inquiries, please contact us at [contact-onecompression@cs.jp.fujitsu.com](mailto:contact-onecompression@cs.jp.fujitsu.com).
|
|
319
|
+
|
|
314
320
|
## 📄 License
|
|
315
321
|
|
|
316
322
|
See [LICENSE](./LICENSE) for more details.
|
|
@@ -133,7 +133,11 @@ print("\n" + "=" * 70)
|
|
|
133
133
|
print(f"Step 4: Loading model from {SAVE_DIR}")
|
|
134
134
|
print("=" * 70)
|
|
135
135
|
|
|
136
|
-
|
|
136
|
+
# SAVE_DIR was produced by this script (trusted), so we opt in to the
|
|
137
|
+
# pickle-based .pt loader. Never enable this for untrusted model.pt files.
|
|
138
|
+
loaded_model, loaded_tokenizer = load_quantized_model_pt(
|
|
139
|
+
SAVE_DIR, allow_unsafe_deserialization=True
|
|
140
|
+
)
|
|
137
141
|
print(f"Loaded model type : {type(loaded_model).__name__}")
|
|
138
142
|
print(f"Loaded model device: {next(loaded_model.parameters()).device}")
|
|
139
143
|
|
|
@@ -11,6 +11,7 @@ from logging import getLogger
|
|
|
11
11
|
import torch
|
|
12
12
|
from transformers import AutoConfig, AutoModelForCausalLM, AutoTokenizer
|
|
13
13
|
|
|
14
|
+
from .utils.device import get_default_device
|
|
14
15
|
from .utils.dtype import needs_bfloat16
|
|
15
16
|
|
|
16
17
|
try:
|
|
@@ -114,6 +115,12 @@ class ModelConfig:
|
|
|
114
115
|
self.logger.info("Model loaded with dtype=%s", next(model.parameters()).dtype)
|
|
115
116
|
return model
|
|
116
117
|
|
|
118
|
+
def get_device(self) -> torch.device:
|
|
119
|
+
"""Return a concrete torch.device for PyTorch operations."""
|
|
120
|
+
if self.device == "auto":
|
|
121
|
+
return get_default_device()
|
|
122
|
+
return torch.device(self.device)
|
|
123
|
+
|
|
117
124
|
def load_tokenizer(self):
|
|
118
125
|
"""Load the tokenizer"""
|
|
119
126
|
|
|
@@ -298,6 +298,8 @@ class PostProcessLoraSFT(PostQuantizationProcess):
|
|
|
298
298
|
>>> import torch
|
|
299
299
|
>>> from onecomp import ModelConfig, PostProcessLoraSFT
|
|
300
300
|
>>> model_config = ModelConfig(model_id="meta-llama/Llama-2-7b-hf")
|
|
301
|
+
>>> # weights_only=False uses pickle and can execute code from a
|
|
302
|
+
>>> # malicious file (CWE-502); only load trusted checkpoints.
|
|
301
303
|
>>> quantized_model = torch.load(
|
|
302
304
|
... "quantized_model.pt",
|
|
303
305
|
... map_location="cpu",
|
|
@@ -187,6 +187,7 @@ class QuantizedModelLoader:
|
|
|
187
187
|
*,
|
|
188
188
|
device_map: str = "auto",
|
|
189
189
|
local_files_only: bool = True,
|
|
190
|
+
allow_unsafe_deserialization: bool = False,
|
|
190
191
|
) -> Tuple[Any, Any]:
|
|
191
192
|
"""Load a quantized model and tokenizer saved as a PyTorch .pt file.
|
|
192
193
|
|
|
@@ -198,18 +199,39 @@ class QuantizedModelLoader:
|
|
|
198
199
|
- ``model.pt`` (serialized with ``torch.save``)
|
|
199
200
|
- Tokenizer files
|
|
200
201
|
|
|
202
|
+
.. warning::
|
|
203
|
+
This method deserializes ``model.pt`` with
|
|
204
|
+
``torch.load(..., weights_only=False)``. Because PyTorch ``.pt``
|
|
205
|
+
checkpoints use Python's ``pickle``, a maliciously crafted
|
|
206
|
+
``model.pt`` can execute arbitrary code during deserialization
|
|
207
|
+
(CWE-502). ``weights_only=False`` is required here because the
|
|
208
|
+
``.pt`` format preserves full custom module objects (e.g.
|
|
209
|
+
``LoRAGPTQLinear``) that cannot be reconstructed from tensors
|
|
210
|
+
alone. Only load ``model.pt`` files that you produced yourself
|
|
211
|
+
or obtained from a fully trusted source. For untrusted or
|
|
212
|
+
third-party models, prefer the safetensors-based
|
|
213
|
+
:meth:`load_quantized_model`, which does not execute code.
|
|
214
|
+
|
|
201
215
|
Args:
|
|
202
216
|
save_directory: Path to the saved model directory.
|
|
203
217
|
device_map: Device placement (default: ``"auto"``).
|
|
204
218
|
Set to ``""`` or ``None`` to skip device placement.
|
|
205
219
|
local_files_only: Passed to ``AutoTokenizer.from_pretrained``.
|
|
220
|
+
allow_unsafe_deserialization: Must be explicitly set to ``True``
|
|
221
|
+
to acknowledge the unsafe-deserialization risk described
|
|
222
|
+
above and permit loading. Defaults to ``False``, in which
|
|
223
|
+
case this method raises before any code can be executed.
|
|
206
224
|
|
|
207
225
|
Returns:
|
|
208
226
|
(model, tokenizer)
|
|
209
227
|
|
|
228
|
+
Raises:
|
|
229
|
+
ValueError: If ``allow_unsafe_deserialization`` is not ``True``.
|
|
230
|
+
|
|
210
231
|
Example:
|
|
211
232
|
>>> model, tokenizer = QuantizedModelLoader.load_quantized_model_pt(
|
|
212
|
-
... "./quantized_model_lora"
|
|
233
|
+
... "./quantized_model_lora",
|
|
234
|
+
... allow_unsafe_deserialization=True, # trusted source only
|
|
213
235
|
... )
|
|
214
236
|
"""
|
|
215
237
|
save_directory = os.path.abspath(save_directory)
|
|
@@ -224,6 +246,24 @@ class QuantizedModelLoader:
|
|
|
224
246
|
"(safetensors format); use load_quantized_model() instead."
|
|
225
247
|
)
|
|
226
248
|
|
|
249
|
+
if not allow_unsafe_deserialization:
|
|
250
|
+
raise ValueError(
|
|
251
|
+
f"Refusing to load '{model_path}': loading a .pt model uses "
|
|
252
|
+
"torch.load(weights_only=False), which deserializes arbitrary "
|
|
253
|
+
"Python objects via pickle and can execute code embedded in a "
|
|
254
|
+
"malicious file (CWE-502). Only load model.pt files you produced "
|
|
255
|
+
"yourself or obtained from a fully trusted source, then pass "
|
|
256
|
+
"allow_unsafe_deserialization=True to acknowledge this risk. "
|
|
257
|
+
"For untrusted or third-party models, use the safetensors-based "
|
|
258
|
+
"load_quantized_model() instead."
|
|
259
|
+
)
|
|
260
|
+
|
|
261
|
+
logger.warning(
|
|
262
|
+
"Loading '%s' with torch.load(weights_only=False); arbitrary code in "
|
|
263
|
+
"a malicious checkpoint can execute during deserialization. Ensure "
|
|
264
|
+
"this model.pt comes from a trusted source.",
|
|
265
|
+
model_path,
|
|
266
|
+
)
|
|
227
267
|
model = torch.load(model_path, map_location="cpu", weights_only=False)
|
|
228
268
|
|
|
229
269
|
if device_map:
|
|
@@ -641,22 +641,40 @@ class Quantizer(metaclass=ABCMeta):
|
|
|
641
641
|
torch.save(self.results, filepath)
|
|
642
642
|
self.logger.info("Saved quantization results to %s", filepath)
|
|
643
643
|
|
|
644
|
-
def load_results(self, filepath, weights_only=False):
|
|
644
|
+
def load_results(self, filepath, *, weights_only=False, allow_unsafe_deserialization=False):
|
|
645
645
|
"""Load the quantization results from a file into self.results.
|
|
646
646
|
|
|
647
647
|
Loads saved quantization results and stores them in self.results.
|
|
648
648
|
|
|
649
|
+
.. warning::
|
|
650
|
+
With ``weights_only=False`` this method deserializes arbitrary
|
|
651
|
+
Python objects via ``torch.load`` / ``pickle``, so a maliciously
|
|
652
|
+
crafted file can execute code during deserialization (CWE-502).
|
|
653
|
+
Only load result files you produced yourself or obtained from a
|
|
654
|
+
fully trusted source, and set ``allow_unsafe_deserialization=True``
|
|
655
|
+
to acknowledge the risk.
|
|
656
|
+
|
|
649
657
|
Args:
|
|
650
658
|
filepath (str): The path to load the results from.
|
|
651
659
|
weights_only (bool): If True, only load tensor weights (safer but limited).
|
|
652
660
|
Default is False to support loading QuantizationResult objects.
|
|
661
|
+
allow_unsafe_deserialization (bool): Required to be True when
|
|
662
|
+
``weights_only`` is False, to acknowledge the
|
|
663
|
+
unsafe-deserialization risk. Ignored when ``weights_only`` is
|
|
664
|
+
True (that path is safe). Defaults to False.
|
|
653
665
|
|
|
654
666
|
Returns:
|
|
655
667
|
dict: A dict mapping layer name -> QuantizationResult (same reference as self.results).
|
|
656
668
|
|
|
669
|
+
Raises:
|
|
670
|
+
ValueError: If ``weights_only`` is False and
|
|
671
|
+
``allow_unsafe_deserialization`` is not True.
|
|
672
|
+
|
|
657
673
|
Example:
|
|
658
674
|
>>> quantizer = JointQ()
|
|
659
|
-
>>> quantizer.load_results(
|
|
675
|
+
>>> quantizer.load_results(
|
|
676
|
+
... "quantization_results.pt", allow_unsafe_deserialization=True
|
|
677
|
+
... )
|
|
660
678
|
>>> for layer_name, result in quantizer.results.items():
|
|
661
679
|
... print(f"{layer_name}: {result.dequantized_weight.shape}")
|
|
662
680
|
|
|
@@ -667,6 +685,22 @@ class Quantizer(metaclass=ABCMeta):
|
|
|
667
685
|
- Loading files saved with older versions may fail if class
|
|
668
686
|
definitions have changed.
|
|
669
687
|
"""
|
|
688
|
+
if not weights_only and not allow_unsafe_deserialization:
|
|
689
|
+
raise ValueError(
|
|
690
|
+
f"Refusing to load '{filepath}': loading with weights_only=False "
|
|
691
|
+
"deserializes arbitrary Python objects via pickle and can execute "
|
|
692
|
+
"code embedded in a malicious file (CWE-502). Only load result "
|
|
693
|
+
"files from a fully trusted source, then pass "
|
|
694
|
+
"allow_unsafe_deserialization=True to acknowledge this risk, or "
|
|
695
|
+
"use weights_only=True for the safe (tensors-only) path."
|
|
696
|
+
)
|
|
697
|
+
if not weights_only:
|
|
698
|
+
self.logger.warning(
|
|
699
|
+
"Loading '%s' with weights_only=False; arbitrary code in a "
|
|
700
|
+
"malicious file can execute during deserialization. Ensure this "
|
|
701
|
+
"file comes from a trusted source.",
|
|
702
|
+
filepath,
|
|
703
|
+
)
|
|
670
704
|
self.results = torch.load(filepath, weights_only=weights_only)
|
|
671
705
|
self.logger.info("Loaded quantization results from %s", filepath)
|
|
672
706
|
return self.results
|
|
@@ -951,6 +985,9 @@ class ResultLoader(Quantizer):
|
|
|
951
985
|
# Optional: load precomputed results at initialization time
|
|
952
986
|
results_file: str = None
|
|
953
987
|
weights_only: bool = False
|
|
988
|
+
# Required to be True when weights_only is False, to acknowledge the
|
|
989
|
+
# unsafe-deserialization risk of torch.load(weights_only=False) (CWE-502).
|
|
990
|
+
allow_unsafe_deserialization: bool = False
|
|
954
991
|
|
|
955
992
|
# Ensure no layers are selected for quantization by default
|
|
956
993
|
target_layer_types: tuple = field(default_factory=tuple)
|
|
@@ -962,7 +999,11 @@ class ResultLoader(Quantizer):
|
|
|
962
999
|
def __post_init__(self):
|
|
963
1000
|
super().__post_init__()
|
|
964
1001
|
if self.results_file is not None:
|
|
965
|
-
self.load_results(
|
|
1002
|
+
self.load_results(
|
|
1003
|
+
self.results_file,
|
|
1004
|
+
weights_only=self.weights_only,
|
|
1005
|
+
allow_unsafe_deserialization=self.allow_unsafe_deserialization,
|
|
1006
|
+
)
|
|
966
1007
|
|
|
967
1008
|
def setup(self, model): # pylint: disable=unused-argument
|
|
968
1009
|
"""Select no layers (no-op).
|
|
@@ -327,7 +327,7 @@ class Runner:
|
|
|
327
327
|
|
|
328
328
|
# MPS device validation: only GPTQ (or AutoBitQuantizer whose
|
|
329
329
|
# candidates are all GPTQ, without DBF fallback) is supported on MPS
|
|
330
|
-
device = self.model_config.
|
|
330
|
+
device = self.model_config.get_device()
|
|
331
331
|
if is_mps_device(device):
|
|
332
332
|
if self.multi_gpu:
|
|
333
333
|
raise ValueError("multi_gpu is not supported on MPS device.")
|
|
@@ -1139,24 +1139,24 @@ class Runner:
|
|
|
1139
1139
|
tokenizer = self.model_config.load_tokenizer()
|
|
1140
1140
|
original_result = eval_function(model=model, tokenizer=tokenizer, **eval_args)
|
|
1141
1141
|
del model, tokenizer
|
|
1142
|
-
empty_cache(self.model_config.
|
|
1142
|
+
empty_cache(self.model_config.get_device())
|
|
1143
1143
|
|
|
1144
1144
|
if quantized_model:
|
|
1145
1145
|
try:
|
|
1146
1146
|
logger.info("Evaluating quantized model (%s)...", eval_name)
|
|
1147
1147
|
if self.quantized_model is not None:
|
|
1148
1148
|
model = self.quantized_model
|
|
1149
|
-
model.to(self.model_config.
|
|
1149
|
+
model.to(self.model_config.get_device())
|
|
1150
1150
|
tokenizer = self.model_config.load_tokenizer()
|
|
1151
1151
|
quantized_result = eval_function(model=model, tokenizer=tokenizer, **eval_args)
|
|
1152
1152
|
model.to("cpu")
|
|
1153
1153
|
del tokenizer
|
|
1154
1154
|
else:
|
|
1155
1155
|
model, tokenizer = self.create_quantized_model(quantizer=quantizer)
|
|
1156
|
-
model.to(self.model_config.
|
|
1156
|
+
model.to(self.model_config.get_device())
|
|
1157
1157
|
quantized_result = eval_function(model=model, tokenizer=tokenizer, **eval_args)
|
|
1158
1158
|
del model, tokenizer
|
|
1159
|
-
empty_cache(self.model_config.
|
|
1159
|
+
empty_cache(self.model_config.get_device())
|
|
1160
1160
|
except NotImplementedError:
|
|
1161
1161
|
logger.warning(
|
|
1162
1162
|
"This quantization method does not support creating a quantized model; "
|
|
@@ -1171,7 +1171,7 @@ class Runner:
|
|
|
1171
1171
|
self.update_model_weights(model, quantizer=quantizer)
|
|
1172
1172
|
dequantized_result = eval_function(model=model, tokenizer=tokenizer, **eval_args)
|
|
1173
1173
|
del model, tokenizer
|
|
1174
|
-
empty_cache(self.model_config.
|
|
1174
|
+
empty_cache(self.model_config.get_device())
|
|
1175
1175
|
|
|
1176
1176
|
return original_result, dequantized_result, quantized_result
|
|
1177
1177
|
|
|
@@ -2117,7 +2117,7 @@ class Runner:
|
|
|
2117
2117
|
)
|
|
2118
2118
|
# Release fragmented GPU memory from previous operations (e.g., run())
|
|
2119
2119
|
gc.collect()
|
|
2120
|
-
empty_cache(self.model_config.
|
|
2120
|
+
empty_cache(self.model_config.get_device())
|
|
2121
2121
|
|
|
2122
2122
|
model = self.model_config.load_model()
|
|
2123
2123
|
input_device = next(model.parameters()).device
|
|
@@ -2141,7 +2141,7 @@ class Runner:
|
|
|
2141
2141
|
)
|
|
2142
2142
|
# Release fragmented GPU memory from previous operations (e.g., run())
|
|
2143
2143
|
gc.collect()
|
|
2144
|
-
empty_cache(self.model_config.
|
|
2144
|
+
empty_cache(self.model_config.get_device())
|
|
2145
2145
|
|
|
2146
2146
|
model = self.model_config.load_model()
|
|
2147
2147
|
input_device = next(model.parameters()).device
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: onecomp
|
|
3
|
-
Version: 1.2.
|
|
3
|
+
Version: 1.2.2
|
|
4
4
|
Summary: Python package for LLM compression
|
|
5
5
|
Author: Keiji Kimura
|
|
6
6
|
License: MIT License
|
|
@@ -416,6 +416,12 @@ pip install vllm
|
|
|
416
416
|
See the [vLLM Inference guide](https://FujitsuResearch.github.io/OneCompression/user-guide/vllm-inference/) for details, including Open WebUI setup instructions.
|
|
417
417
|
|
|
418
418
|
|
|
419
|
+
## 📬 Contact Us
|
|
420
|
+
|
|
421
|
+
- For technical questions and feature requests, please use GitHub [Issues](https://github.com/FujitsuResearch/OneCompression/issues).
|
|
422
|
+
- For security vulnerabilities, please **do not** open a public Issue. See our [Security Policy](./SECURITY.md) for how to report them privately.
|
|
423
|
+
- For collaborations, partnerships, and other inquiries, please contact us at [contact-onecompression@cs.jp.fujitsu.com](mailto:contact-onecompression@cs.jp.fujitsu.com).
|
|
424
|
+
|
|
419
425
|
## 📄 License
|
|
420
426
|
|
|
421
427
|
See [LICENSE](./LICENSE) for more details.
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|