onecomp 1.3.3__tar.gz → 1.3.4__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {onecomp-1.3.3/onecomp.egg-info → onecomp-1.3.4}/PKG-INFO +7 -2
- {onecomp-1.3.3 → onecomp-1.3.4}/README.md +6 -1
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/__version__.py +1 -1
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/model_config.py +9 -2
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/qep/_quantize_with_qep_arch.py +11 -4
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/quantized_model_loader.py +12 -17
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/runner.py +5 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/utils/blockwise.py +78 -2
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/utils/unfuse_moe.py +7 -2
- {onecomp-1.3.3 → onecomp-1.3.4/onecomp.egg-info}/PKG-INFO +7 -2
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp.egg-info/SOURCES.txt +1 -0
- onecomp-1.3.4/scripts/prepare_calibration_cache.py +284 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/LICENSE +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/benchmark/llama3-8b-gptq/quant_benchmark.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/benchmark/llama3-8b-jointq/quant_benchmark.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/benchmark/llama3-8b-lpcd-gptq/quant_benchmark.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/benchmark/llama3-8b-qep-gptq/quant_benchmark.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/benchmark/llama3-8b-various/quant_benchmark.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/benchmark/qwen3-14b-gptq/quant_benchmark.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/benchmark/qwen3-14b-jointq/quant_benchmark.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/benchmark/qwen3-8b-gptq/quant_benchmark.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/benchmark/qwen3-8b-jointq/quant_benchmark.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/dashboard/backend/app/__init__.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/dashboard/backend/app/api/__init__.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/dashboard/backend/app/api/jobs.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/dashboard/backend/app/constants.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/dashboard/backend/app/core/__init__.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/dashboard/backend/app/core/config.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/dashboard/backend/app/core/database.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/dashboard/backend/app/main.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/dashboard/backend/app/models/__init__.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/dashboard/backend/app/models/job.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/dashboard/backend/app/schemas/__init__.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/dashboard/backend/app/schemas/job.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/dashboard/backend/app/services/__init__.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/dashboard/backend/app/services/huggingface.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/dashboard/backend/app/services/inference.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/dashboard/backend/app/services/job_store.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/dashboard/backend/app/worker/__init__.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/dashboard/backend/app/worker/celery_app.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/dashboard/backend/app/worker/tasks.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/dashboard/backend/cpu_patch.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/dashboard/backend/start_backend.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/dashboard/backend/start_worker.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/envs/vllm/v0_24_0_rocm/src/onecomp_vllm_v0_24_0_rocm/__init__.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/envs/vllm/v0_24_0_rocm/src/onecomp_vllm_v0_24_0_rocm/patch.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/example/cpu_inference/example_gptq_gguf_cpu.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/example/cpu_inference/example_mixed_gptq_gguf_cpu.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/example/cpu_inference/example_serve_cpu.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/example/example_auto_run.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/example/example_autobit.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/example/example_custom_calibration.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/example/example_gptq.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/example/example_jointq.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/example/example_lpcd_gptq.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/example/example_mdbf.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/example/example_qep_gptq.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/example/example_save_load.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/example/post_process/example_blockwise_global_ptq.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/example/post_process/example_blockwise_global_ptq_staged.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/example/post_process/example_blockwise_ptq.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/example/post_process/example_global_ptq.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/example/post_process/example_global_ptq_dbf.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/example/post_process/example_global_ptq_distributed.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/example/post_process/example_lora_gptq_vllm_inference.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/example/post_process/example_lora_sft.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/example/post_process/example_lora_sft_knowledge.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/example/post_process/example_lora_sft_knowledge_jointq.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/example/post_process/example_reload_post_process_resave.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/example/pre_process/example_llama_preprocess_rtn.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/example/pre_process/example_preprocess_save_load.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/example/vllm_inference/example_autobit_vllm_inference.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/example/vllm_inference/example_dbf_vllm_inference.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/example/vllm_inference/example_gptq_vllm_gptoss_inference.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/example/vllm_inference/example_gptq_vllm_inference.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/example/vllm_inference/example_gptq_vllm_qwen36_inference.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/example/vllm_inference/example_jointq_vllm_inference.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/llamacpp_plugins/__init__.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/llamacpp_plugins/gptq/__init__.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/llamacpp_plugins/gptq/constants.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/llamacpp_plugins/gptq/llamacpp_plugin.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/model_validation/autobit/validate_autobit.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/model_validation/autobit_qep/validate_autobit.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/model_validation/gptq/validate_gptq.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/model_validation/gptq/validate_load.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/model_validation/gptq/validate_vllm.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/model_validation/jointq/validate_jointq.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/model_validation/qep_gptq/validate_gptq.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/__init__.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/__main__.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/analyzer/__init__.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/analyzer/cumulative_error.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/analyzer/quantization_error.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/analyzer/weight_outlier.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/calibration/__init__.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/calibration/_cache.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/calibration/c4.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/calibration/calibration_config.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/calibration/calibration_data_loader.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/calibration/chunking.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/calibration/custom.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/calibration/wikitext.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/cli.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/cpu/__init__.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/cpu/cli.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/cpu/eval/__init__.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/cpu/eval/benchmark.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/cpu/eval/inspect_gguf.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/cpu/eval/parity.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/cpu/eval/perplexity.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/cpu/export/__init__.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/cpu/export/auto.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/cpu/export/blocks.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/cpu/export/checkpoint.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/cpu/export/dequantize.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/cpu/export/direct.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/cpu/export/fallback.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/cpu/export/rotation.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/cpu/export/skeleton.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/cpu/inference.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/cpu/llama_tooling.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/cpu/serve.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/eval/__init__.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/eval/__main__.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/eval/conf/__init__.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/eval/conf/eval_config.yaml +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/eval/evals/__init__.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/eval/evals/base.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/eval/evals/mt_bench/__init__.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/eval/evals/mt_bench/adapter.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/eval/evals/mt_bench/data.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/eval/evals/mt_bench/gen_answer.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/eval/evals/mt_bench/judge.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/eval/evals/mt_bench/radar_chart.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/eval/evals/mt_bench/run.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/eval/evals/mt_bench/show_result.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/eval/evals/throughput/__init__.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/eval/evals/throughput/adapter.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/eval/evals/throughput/bench.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/eval/evals/throughput/run.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/eval/orchestrator/__init__.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/eval/orchestrator/aggregator.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/eval/orchestrator/runner.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/eval/orchestrator/server.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/eval/orchestrator/subprocess_runner.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/eval/run_evaluate.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/eval/schema.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/eval/utils/__init__.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/eval/utils/model_utils.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/eval/utils/ports.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/eval/utils/resources.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/eval/utils/secrets.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/export/__init__.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/export/gguf_export.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/export/gguf_reader.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/export/gguf_writer.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/export/hub.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/export/model_card.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/log.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/lpcd/__init__.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/lpcd/_gradient_solver.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/lpcd/_lpcd_config.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/lpcd/_lpcd_runner.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/lpcd/_metric.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/lpcd/_refiner.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/lpcd/arch/_llama.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/lpcd/arch/_llama_cf.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/lpcd/arch/_qwen3.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/post_process/__init__.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/post_process/_base.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/post_process/_blockwise/__init__.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/post_process/_blockwise/dbf_block_optimizer.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/post_process/_blockwise/dbf_cbq_optimizer.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/post_process/_blockwise/generic_block_optimizer.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/post_process/_blockwise/gptq_block_optimizer.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/post_process/_blockwise/gptq_cbq_optimizer.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/post_process/_blockwise/helpers.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/post_process/_blockwise/onebit_block_optimizer.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/post_process/_blockwise/onebit_cbq_optimizer.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/post_process/_global_ptq/__init__.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/post_process/_global_ptq/core.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/post_process/_global_ptq/dbf_adapter.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/post_process/_global_ptq/gptq_adapter.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/post_process/_global_ptq/helpers.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/post_process/_global_ptq/losses.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/post_process/_global_ptq/trainer.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/post_process/_runtime.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/post_process/blockwise_ptq.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/post_process/global_ptq.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/post_process/global_ptq_distributed.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/post_process/post_process_lora_sft.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/pre_process/__init__.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/pre_process/hadamard_utils.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/pre_process/modeling_llama.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/pre_process/modeling_qwen3.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/pre_process/optimizer.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/pre_process/prepare_rotated_model.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/pre_process/preprocess_args.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/pre_process/quant_models.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/pre_process/rotation_utils.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/pre_process/train_rotation.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/qep/__init__.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/qep/_qep_config.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/qep/_quantize_with_qep.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/quantizer/__init__.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/quantizer/_quantizer.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/quantizer/arb/__init__.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/quantizer/arb/_arb.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/quantizer/arb/arb_impl.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/quantizer/autobit/__init__.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/quantizer/autobit/_autobit.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/quantizer/autobit/activation_stats.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/quantizer/autobit/dbf_fallback.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/quantizer/autobit/ilp.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/quantizer/autobit/manual.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/quantizer/autobit/visualize.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/quantizer/cq/__init__.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/quantizer/cq/_cq.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/quantizer/cq/cq_impl.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/quantizer/dbf/__init__.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/quantizer/dbf/_dbf.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/quantizer/dbf/admm_extended.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/quantizer/dbf/balance.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/quantizer/dbf/config.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/quantizer/dbf/dbf_impl.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/quantizer/dbf/dbf_layer.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/quantizer/dbf/dbf_original.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/quantizer/dbf/fine_tune.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/quantizer/dbf/middle.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/quantizer/gemlite.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/quantizer/gptq/__init__.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/quantizer/gptq/_gptq.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/quantizer/gptq/config.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/quantizer/gptq/gptq_layer.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/quantizer/jointq/__init__.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/quantizer/jointq/_jointq.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/quantizer/jointq/core/__init__.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/quantizer/jointq/core/__version__.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/quantizer/jointq/core/clip.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/quantizer/jointq/core/error_propagation/__init__.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/quantizer/jointq/core/error_propagation/local_search_advanced.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/quantizer/jointq/core/error_propagation/quantize_advanced.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/quantizer/jointq/core/error_propagation/quantizer_advanced.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/quantizer/jointq/core/local_search.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/quantizer/jointq/core/quantize.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/quantizer/jointq/core/quantize_multi_gpu.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/quantizer/jointq/core/quantizer.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/quantizer/jointq/core/solution.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/quantizer/mdbf/__init__.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/quantizer/mdbf/_mdbf.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/quantizer/mdbf/admm.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/quantizer/mdbf/config.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/quantizer/mdbf/gradient_refine.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/quantizer/mdbf/initialize.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/quantizer/mdbf/mdbf_impl.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/quantizer/mdbf/mdbf_layer.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/quantizer/mdbf/utils.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/quantizer/onebit/__init__.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/quantizer/onebit/_onebit.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/quantizer/onebit/onebit_impl.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/quantizer/onebit/onebit_layer.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/quantizer/qbb/__init__.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/quantizer/qbb/_qbb.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/quantizer/qbb/qbb_impl.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/quantizer/quip/__init__.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/quantizer/quip/_quip.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/quantizer/quip/quant_quip.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/quantizer/quip/quip_impl.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/quantizer/quip/utils.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/quantizer/quip/utils_had.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/quantizer/quip/vector_balance.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/quantizer/rtn/__init__.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/quantizer/rtn/_rtn.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/quantizer/rtn/quantizer.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/quantizer/rtn/rtn_impl.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/rotated_model_config.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/runner_methods/__init__.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/runner_methods/chunked_quantization.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/runner_methods/jointq_error_propagation.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/runner_methods/multi_gpu_quantization.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/utils/__init__.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/utils/accuracy.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/utils/activation_capture.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/utils/activation_check.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/utils/device.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/utils/dtype.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/utils/lora.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/utils/model_inputs.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/utils/mxfp4_compat.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/utils/perplexity.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/utils/quant_config.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/utils/quantization_progress.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp/utils/vram_estimator.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp.egg-info/dependency_links.txt +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp.egg-info/entry_points.txt +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp.egg-info/requires.txt +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/onecomp.egg-info/top_level.txt +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/pyproject.toml +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/scripts/check_copyright_header.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/scripts/check_no_japanese.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/setup.cfg +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/vllm_plugins/__init__.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/vllm_plugins/dbf/__init__.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/vllm_plugins/dbf/modules/__init__.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/vllm_plugins/dbf/modules/gemlite_linear.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/vllm_plugins/dbf/modules/naive.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/vllm_plugins/dbf/vllm_plugin.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/vllm_plugins/gptq/__init__.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/vllm_plugins/gptq/constants.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/vllm_plugins/gptq/gptoss_wna16_moe.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/vllm_plugins/gptq/vllm_plugin.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/vllm_plugins/patches/__init__.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/vllm_plugins/patches/_paths.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/vllm_plugins/patches/apply_all.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/vllm_plugins/patches/gpt_oss_gptq_moe.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/vllm_plugins/patches/gpt_oss_wna16_bias.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/vllm_plugins/utils/__init__.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/vllm_plugins/utils/module.py +0 -0
- {onecomp-1.3.3 → onecomp-1.3.4}/vllm_plugins/utils/rotation.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: onecomp
|
|
3
|
-
Version: 1.3.
|
|
3
|
+
Version: 1.3.4
|
|
4
4
|
Summary: Python package for LLM compression
|
|
5
5
|
Author: Keiji Kimura
|
|
6
6
|
License: MIT License
|
|
@@ -473,7 +473,12 @@ See the [GPT-OSS guide](docs/user-guide/gptoss.md) for HF save/load, patch detai
|
|
|
473
473
|
|
|
474
474
|
## 📄 License
|
|
475
475
|
|
|
476
|
-
|
|
476
|
+
OneComp is licensed under the [MIT License](./LICENSE).
|
|
477
|
+
|
|
478
|
+
The dependencies installed with OneComp are separate open-source software (OSS)
|
|
479
|
+
projects and are distributed under their respective licenses. Their licenses
|
|
480
|
+
may change when the dependencies are updated, so please check the license
|
|
481
|
+
terms of the installed versions as well.
|
|
477
482
|
|
|
478
483
|
## Citation
|
|
479
484
|
|
|
@@ -359,7 +359,12 @@ See the [GPT-OSS guide](docs/user-guide/gptoss.md) for HF save/load, patch detai
|
|
|
359
359
|
|
|
360
360
|
## 📄 License
|
|
361
361
|
|
|
362
|
-
|
|
362
|
+
OneComp is licensed under the [MIT License](./LICENSE).
|
|
363
|
+
|
|
364
|
+
The dependencies installed with OneComp are separate open-source software (OSS)
|
|
365
|
+
projects and are distributed under their respective licenses. Their licenses
|
|
366
|
+
may change when the dependencies are updated, so please check the license
|
|
367
|
+
terms of the installed versions as well.
|
|
363
368
|
|
|
364
369
|
## Citation
|
|
365
370
|
|
|
@@ -11,7 +11,7 @@ from logging import getLogger
|
|
|
11
11
|
import torch
|
|
12
12
|
from transformers import AutoConfig, AutoModelForCausalLM, AutoTokenizer
|
|
13
13
|
|
|
14
|
-
from .utils.device import get_default_device
|
|
14
|
+
from .utils.device import get_default_device, is_mps_device
|
|
15
15
|
from .utils.dtype import needs_bfloat16
|
|
16
16
|
|
|
17
17
|
try:
|
|
@@ -95,9 +95,14 @@ class ModelConfig:
|
|
|
95
95
|
If ``None`` (default), ``self.device`` is used.
|
|
96
96
|
"""
|
|
97
97
|
effective_device = device_map if device_map is not None else self.device
|
|
98
|
+
if effective_device == "auto":
|
|
99
|
+
target_device = get_default_device()
|
|
100
|
+
else:
|
|
101
|
+
target_device = effective_device
|
|
102
|
+
load_device = "cpu" if is_mps_device(target_device) else effective_device
|
|
98
103
|
kwargs = dict(
|
|
99
104
|
dtype=self.dtype if self.dtype == "auto" else getattr(torch, self.dtype),
|
|
100
|
-
device_map=
|
|
105
|
+
device_map=load_device,
|
|
101
106
|
)
|
|
102
107
|
|
|
103
108
|
config = self.load_config()
|
|
@@ -135,6 +140,8 @@ class ModelConfig:
|
|
|
135
140
|
raise
|
|
136
141
|
self.logger.info("AutoModelForCausalLM failed; trying AutoModelForImageTextToText.")
|
|
137
142
|
model = _AutoVLM.from_pretrained(self.get_model_id_or_path(), **kwargs)
|
|
143
|
+
if is_mps_device(target_device):
|
|
144
|
+
model = model.to(target_device)
|
|
138
145
|
model.eval()
|
|
139
146
|
self.logger.info("Model loaded with dtype=%s", next(model.parameters()).dtype)
|
|
140
147
|
return model
|
|
@@ -298,8 +298,15 @@ def _rtn_fallback_result(module: nn.Module, quantizer: Quantizer, name: str) ->
|
|
|
298
298
|
|
|
299
299
|
result_dict = run_rtn(module, wbits=wbits, groupsize=groupsize, sym=quantizer.sym)
|
|
300
300
|
|
|
301
|
-
|
|
302
|
-
|
|
301
|
+
scales = result_dict["scale"]
|
|
302
|
+
qzeros = result_dict["zero"]
|
|
303
|
+
|
|
304
|
+
if groupsize != -1:
|
|
305
|
+
# RTN's raw scale/zero are (out_features, num_groups); GPTQResult
|
|
306
|
+
# expects (num_groups, out_features).
|
|
307
|
+
scales = scales.T
|
|
308
|
+
qzeros = qzeros.T
|
|
309
|
+
|
|
303
310
|
return GPTQResult(
|
|
304
311
|
dequantized_weight=result_dict["dequantized_weight"],
|
|
305
312
|
wbits=wbits,
|
|
@@ -307,8 +314,8 @@ def _rtn_fallback_result(module: nn.Module, quantizer: Quantizer, name: str) ->
|
|
|
307
314
|
actorder=False,
|
|
308
315
|
sym=quantizer.sym,
|
|
309
316
|
qweight=result_dict["quantized_weight"],
|
|
310
|
-
scales=
|
|
311
|
-
qzeros=
|
|
317
|
+
scales=scales,
|
|
318
|
+
qzeros=qzeros,
|
|
312
319
|
perm=None,
|
|
313
320
|
)
|
|
314
321
|
|
|
@@ -123,20 +123,16 @@ class QuantizedModelLoader:
|
|
|
123
123
|
elif unfuse_moe_experts(model, logger):
|
|
124
124
|
logger.info("Unfused MoE expert tensors for quantized model load")
|
|
125
125
|
|
|
126
|
-
#
|
|
127
|
-
#
|
|
128
|
-
#
|
|
129
|
-
# ForCausalLM wrapper) while from_config exposes
|
|
130
|
-
# model.language_model.layers.* directly.
|
|
131
|
-
state_dict = cls._remap_state_dict_keys(state_dict, model)
|
|
132
|
-
|
|
133
|
-
# Replace quantized layers with empty modules and align quantized
|
|
134
|
-
# tensor keys with the actual module names in the model built from
|
|
135
|
-
# config. This is required when the saved checkpoint and the
|
|
136
|
-
# from_config model use different wrapper prefixes, e.g.
|
|
137
|
-
# model.language_model.layers.* vs model.layers.*.
|
|
126
|
+
# Replace quantized layers first. This resolves every quantizer's
|
|
127
|
+
# tensor state by layer prefix while the model still exposes the
|
|
128
|
+
# original empty Linear modules.
|
|
138
129
|
state_dict = cls._replace_quantized_layers(model, state_dict, quant_config)
|
|
139
130
|
|
|
131
|
+
# Align remaining checkpoint keys with the model built from config.
|
|
132
|
+
# Quantized keys now already use the actual module prefix, so this
|
|
133
|
+
# generic remap only handles non-quantized parameters and buffers.
|
|
134
|
+
state_dict = cls._remap_state_dict_keys(state_dict, model)
|
|
135
|
+
|
|
140
136
|
# Load all weights (quantized + non-quantized) in one go. strict=False
|
|
141
137
|
# is intentional because some wrapper-only components may be absent, but
|
|
142
138
|
# critical language-model and quantized-buffer mismatches must fail fast.
|
|
@@ -593,11 +589,10 @@ class QuantizedModelLoader:
|
|
|
593
589
|
silently skips mismatched keys and leaves layers at their empty-model
|
|
594
590
|
initial values (often all zeros for quantized buffers).
|
|
595
591
|
|
|
596
|
-
|
|
597
|
-
|
|
598
|
-
|
|
599
|
-
|
|
600
|
-
destination key to already exist in model.named_parameters().
|
|
592
|
+
Quantized layers are resolved by _replace_quantized_layers before
|
|
593
|
+
this method runs. Therefore this method only remaps ordinary model
|
|
594
|
+
parameters and buffers; quantizer-specific tensor fields are already
|
|
595
|
+
materialized under their actual module prefixes.
|
|
601
596
|
|
|
602
597
|
Args:
|
|
603
598
|
state_dict: Tensors loaded from *.safetensors.
|
|
@@ -379,6 +379,11 @@ class Runner:
|
|
|
379
379
|
if is_mps_device(device):
|
|
380
380
|
if self.multi_gpu:
|
|
381
381
|
raise ValueError("multi_gpu is not supported on MPS device.")
|
|
382
|
+
if batch_size is not None:
|
|
383
|
+
raise ValueError(
|
|
384
|
+
"MPS quantization does not support calibration_config.batch_size. "
|
|
385
|
+
"Remove batch_size from CalibrationConfig and run without chunked calibration."
|
|
386
|
+
)
|
|
382
387
|
all_quantizers = self.quantizers if self.quantizers is not None else [self.quantizer]
|
|
383
388
|
for i, q in enumerate(all_quantizers):
|
|
384
389
|
label = f"quantizers[{i}]" if self.quantizers else "quantizer"
|
|
@@ -6,6 +6,7 @@ Author: Yudai Fujimoto, Akihiro Yoshida, Yuma Ichikawa
|
|
|
6
6
|
|
|
7
7
|
"""
|
|
8
8
|
|
|
9
|
+
from collections import UserDict
|
|
9
10
|
from logging import getLogger
|
|
10
11
|
|
|
11
12
|
import torch
|
|
@@ -95,6 +96,7 @@ class Catcher(nn.Module):
|
|
|
95
96
|
_PER_LAYER_INPUTS_KEY = "_per_layer_inputs"
|
|
96
97
|
_POS_EMB_MAP_KEY = "_position_embeddings_map"
|
|
97
98
|
_ATTN_MASK_MAP_KEY = "_attention_mask_map"
|
|
99
|
+
_SHARED_KV_STATES_KEY = "shared_kv_states"
|
|
98
100
|
|
|
99
101
|
|
|
100
102
|
def _find_blocks_parent(model, blocks):
|
|
@@ -243,9 +245,15 @@ def get_blocks_and_inputs(
|
|
|
243
245
|
|
|
244
246
|
# Now capture block inputs for all calibration samples.
|
|
245
247
|
block_inps = []
|
|
246
|
-
for
|
|
248
|
+
for first in range(0, inp_ids.shape[0], batch_size):
|
|
249
|
+
last = min(first + batch_size, inp_ids.shape[0])
|
|
250
|
+
inp = inp_ids[first:last]
|
|
251
|
+
batch_model_kwargs = {
|
|
252
|
+
k: v[first:last] if isinstance(v, torch.Tensor) and v.dim() >= 1 else v
|
|
253
|
+
for k, v in model_kwargs.items()
|
|
254
|
+
}
|
|
247
255
|
try:
|
|
248
|
-
_ = model(inp, **
|
|
256
|
+
_ = model(inp, **batch_model_kwargs)
|
|
249
257
|
except StopForward:
|
|
250
258
|
block_inps.append(blocks[0].inp.cpu())
|
|
251
259
|
|
|
@@ -361,6 +369,10 @@ def move_kwargs_to_device(x, device):
|
|
|
361
369
|
return [move_kwargs_to_device(v, device) for v in x]
|
|
362
370
|
elif isinstance(x, tuple):
|
|
363
371
|
return tuple(move_kwargs_to_device(v, device) for v in x)
|
|
372
|
+
elif isinstance(x, UserDict):
|
|
373
|
+
for k, v in list(x.items()):
|
|
374
|
+
x[k] = move_kwargs_to_device(v, device)
|
|
375
|
+
return x
|
|
364
376
|
else:
|
|
365
377
|
return x
|
|
366
378
|
|
|
@@ -396,6 +408,10 @@ def expand_kwargs_batch(kwargs, batch_size):
|
|
|
396
408
|
return [_expand(t) for t in v]
|
|
397
409
|
elif isinstance(v, dict):
|
|
398
410
|
return {k: _expand(val) for k, val in v.items()}
|
|
411
|
+
elif isinstance(v, UserDict):
|
|
412
|
+
for k, val in list(v.items()):
|
|
413
|
+
v[k] = _expand(val)
|
|
414
|
+
return v
|
|
399
415
|
return v
|
|
400
416
|
|
|
401
417
|
return {k: _expand(v) for k, v in kwargs.items()}
|
|
@@ -433,9 +449,46 @@ def prepare_block_kwargs(batch_kwargs, block, pli, offset, batch_size, device):
|
|
|
433
449
|
if layer_type and layer_type in mask_map:
|
|
434
450
|
batch_kwargs["attention_mask"] = mask_map[layer_type]
|
|
435
451
|
|
|
452
|
+
# 4) Gemma4 shared KV state. Provider blocks write states into a fresh
|
|
453
|
+
# batch-local mapping; consumer blocks read the matching calibration slice.
|
|
454
|
+
shared_kv_states = batch_kwargs.get(_SHARED_KV_STATES_KEY)
|
|
455
|
+
self_attn = getattr(block, "self_attn", None)
|
|
456
|
+
if isinstance(shared_kv_states, UserDict) and getattr(
|
|
457
|
+
self_attn, "store_full_length_kv", False
|
|
458
|
+
):
|
|
459
|
+
batch_kwargs[_SHARED_KV_STATES_KEY] = UserDict()
|
|
460
|
+
elif isinstance(shared_kv_states, UserDict) and getattr(
|
|
461
|
+
self_attn, "is_kv_shared_layer", False
|
|
462
|
+
):
|
|
463
|
+
batch_kwargs[_SHARED_KV_STATES_KEY] = _slice_shared_kv_states(
|
|
464
|
+
shared_kv_states, offset, batch_size, device
|
|
465
|
+
)
|
|
466
|
+
|
|
436
467
|
return batch_kwargs
|
|
437
468
|
|
|
438
469
|
|
|
470
|
+
def _slice_shared_kv_states(shared_kv_states, offset, batch_size, device):
|
|
471
|
+
batch_shared_kv_states = UserDict()
|
|
472
|
+
for key, value in shared_kv_states.items():
|
|
473
|
+
if isinstance(value, tuple):
|
|
474
|
+
batch_shared_kv_states[key] = tuple(
|
|
475
|
+
_slice_shared_kv_tensor(v, offset, batch_size, device) for v in value
|
|
476
|
+
)
|
|
477
|
+
else:
|
|
478
|
+
batch_shared_kv_states[key] = _slice_shared_kv_tensor(
|
|
479
|
+
value, offset, batch_size, device
|
|
480
|
+
)
|
|
481
|
+
return batch_shared_kv_states
|
|
482
|
+
|
|
483
|
+
|
|
484
|
+
def _slice_shared_kv_tensor(value, offset, batch_size, device):
|
|
485
|
+
if isinstance(value, torch.Tensor):
|
|
486
|
+
if value.dim() >= 1 and value.shape[0] >= offset + batch_size:
|
|
487
|
+
value = value[offset : offset + batch_size]
|
|
488
|
+
return value.to(device)
|
|
489
|
+
return value
|
|
490
|
+
|
|
491
|
+
|
|
439
492
|
def _get_block_layer_type(block: nn.Module) -> str | None:
|
|
440
493
|
return (
|
|
441
494
|
getattr(block, "layer_type", None)
|
|
@@ -468,6 +521,9 @@ def forward_input(
|
|
|
468
521
|
pli = kwargs.get(_PER_LAYER_INPUTS_KEY)
|
|
469
522
|
next_inps = []
|
|
470
523
|
offset = 0
|
|
524
|
+
shared_kv_chunks = {}
|
|
525
|
+
self_attn = getattr(block, "self_attn", None)
|
|
526
|
+
stores_shared_kv = getattr(self_attn, "store_full_length_kv", False)
|
|
471
527
|
for inp in inps.split(batch_size):
|
|
472
528
|
bs = inp.shape[0]
|
|
473
529
|
batch_kwargs = expand_kwargs_batch(kwargs, bs)
|
|
@@ -475,7 +531,27 @@ def forward_input(
|
|
|
475
531
|
out = block(inp.to(device), **batch_kwargs)
|
|
476
532
|
out = out[0] if isinstance(out, tuple) else out
|
|
477
533
|
next_inps.append(out.cpu())
|
|
534
|
+
if stores_shared_kv:
|
|
535
|
+
for key, value in batch_kwargs[_SHARED_KV_STATES_KEY].items():
|
|
536
|
+
shared_kv_chunks.setdefault(key, []).append(
|
|
537
|
+
tuple(v.detach().cpu() for v in value)
|
|
538
|
+
if isinstance(value, tuple)
|
|
539
|
+
else value.detach().cpu()
|
|
540
|
+
)
|
|
478
541
|
offset += bs
|
|
542
|
+
|
|
543
|
+
if shared_kv_chunks:
|
|
544
|
+
shared_kv_states = kwargs.get(_SHARED_KV_STATES_KEY)
|
|
545
|
+
if not isinstance(shared_kv_states, UserDict):
|
|
546
|
+
shared_kv_states = UserDict()
|
|
547
|
+
kwargs[_SHARED_KV_STATES_KEY] = shared_kv_states
|
|
548
|
+
for key, chunks in shared_kv_chunks.items():
|
|
549
|
+
if isinstance(chunks[0], tuple):
|
|
550
|
+
shared_kv_states[key] = tuple(
|
|
551
|
+
torch.cat([chunk[i] for chunk in chunks], dim=0) for i in range(len(chunks[0]))
|
|
552
|
+
)
|
|
553
|
+
else:
|
|
554
|
+
shared_kv_states[key] = torch.cat(chunks, dim=0)
|
|
479
555
|
return torch.cat(next_inps)
|
|
480
556
|
|
|
481
557
|
|
|
@@ -55,6 +55,9 @@ class _UnfusedExperts(nn.Module):
|
|
|
55
55
|
def __getitem__(self, idx):
|
|
56
56
|
return getattr(self, str(int(idx)))
|
|
57
57
|
|
|
58
|
+
def __iter__(self):
|
|
59
|
+
return iter(self._modules.values())
|
|
60
|
+
|
|
58
61
|
def forward(
|
|
59
62
|
self,
|
|
60
63
|
hidden_states: torch.Tensor,
|
|
@@ -584,17 +587,19 @@ def _fuse_one(
|
|
|
584
587
|
expert0 = unfused[0]
|
|
585
588
|
inter = expert0.gate_proj.out_features
|
|
586
589
|
hidden = expert0.gate_proj.in_features
|
|
590
|
+
up_w0, _ = _dequantized_weight_bias(expert0.up_proj)
|
|
591
|
+
down_w0, _ = _dequantized_weight_bias(expert0.down_proj)
|
|
587
592
|
gate_up_3d = torch.empty(
|
|
588
593
|
num_experts,
|
|
589
594
|
2 * inter,
|
|
590
595
|
hidden,
|
|
591
|
-
dtype=
|
|
596
|
+
dtype=up_w0.dtype,
|
|
592
597
|
)
|
|
593
598
|
down_3d = torch.empty(
|
|
594
599
|
num_experts,
|
|
595
600
|
hidden,
|
|
596
601
|
inter,
|
|
597
|
-
dtype=
|
|
602
|
+
dtype=down_w0.dtype,
|
|
598
603
|
)
|
|
599
604
|
for i in range(num_experts):
|
|
600
605
|
expert = unfused[i]
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: onecomp
|
|
3
|
-
Version: 1.3.
|
|
3
|
+
Version: 1.3.4
|
|
4
4
|
Summary: Python package for LLM compression
|
|
5
5
|
Author: Keiji Kimura
|
|
6
6
|
License: MIT License
|
|
@@ -473,7 +473,12 @@ See the [GPT-OSS guide](docs/user-guide/gptoss.md) for HF save/load, patch detai
|
|
|
473
473
|
|
|
474
474
|
## 📄 License
|
|
475
475
|
|
|
476
|
-
|
|
476
|
+
OneComp is licensed under the [MIT License](./LICENSE).
|
|
477
|
+
|
|
478
|
+
The dependencies installed with OneComp are separate open-source software (OSS)
|
|
479
|
+
projects and are distributed under their respective licenses. Their licenses
|
|
480
|
+
may change when the dependencies are updated, so please check the license
|
|
481
|
+
terms of the installed versions as well.
|
|
477
482
|
|
|
478
483
|
## Citation
|
|
479
484
|
|
|
@@ -296,6 +296,7 @@ onecomp/utils/unfuse_moe.py
|
|
|
296
296
|
onecomp/utils/vram_estimator.py
|
|
297
297
|
scripts/check_copyright_header.py
|
|
298
298
|
scripts/check_no_japanese.py
|
|
299
|
+
scripts/prepare_calibration_cache.py
|
|
299
300
|
vllm_plugins/__init__.py
|
|
300
301
|
vllm_plugins/dbf/__init__.py
|
|
301
302
|
vllm_plugins/dbf/vllm_plugin.py
|
|
@@ -0,0 +1,284 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""Prepare and verify the fixed C4 calibration cache used by cluster CI.
|
|
3
|
+
|
|
4
|
+
Copyright 2025-2026 Fujitsu Ltd.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
import argparse
|
|
8
|
+
import fcntl
|
|
9
|
+
import hashlib
|
|
10
|
+
import json
|
|
11
|
+
import os
|
|
12
|
+
import shutil
|
|
13
|
+
import sys
|
|
14
|
+
from contextlib import contextmanager
|
|
15
|
+
from pathlib import Path
|
|
16
|
+
|
|
17
|
+
import datasets
|
|
18
|
+
from huggingface_hub import hf_hub_download
|
|
19
|
+
|
|
20
|
+
DATASET_ID = "allenai/c4"
|
|
21
|
+
DATASET_REVISION = "1588ec454efa1a09f29cd18ddd04fe05fc8653a2"
|
|
22
|
+
DATA_FILE = "en/c4-train.00001-of-01024.json.gz"
|
|
23
|
+
SOURCE_SHA256 = "b945059cd1a343cabe311881b7840a6f0363f570e745a0eff0e687e266f6b55d"
|
|
24
|
+
EXPECTED_TRAIN_ROWS = 356318
|
|
25
|
+
MANIFEST_NAME = "onecomp-calibration-manifest.json"
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def _sha256(path):
|
|
29
|
+
"""Return the SHA-256 digest of a file without loading it all into memory."""
|
|
30
|
+
|
|
31
|
+
digest = hashlib.sha256()
|
|
32
|
+
with path.open("rb") as file:
|
|
33
|
+
for chunk in iter(lambda: file.read(1024 * 1024), b""):
|
|
34
|
+
digest.update(chunk)
|
|
35
|
+
return digest.hexdigest()
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def _cache_path(cache_root):
|
|
39
|
+
"""Return the save-to-disk directory consumed by the C4 loader."""
|
|
40
|
+
|
|
41
|
+
return Path(cache_root).expanduser().resolve() / "c4"
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def _file_records(root):
|
|
45
|
+
"""Describe every persisted cache file so content changes are detectable."""
|
|
46
|
+
|
|
47
|
+
records = []
|
|
48
|
+
for path in sorted(item for item in root.rglob("*") if item.is_file()):
|
|
49
|
+
if path.name == MANIFEST_NAME:
|
|
50
|
+
# The manifest cannot include its own digest without becoming self-referential.
|
|
51
|
+
continue
|
|
52
|
+
records.append(
|
|
53
|
+
{
|
|
54
|
+
"path": path.relative_to(root).as_posix(),
|
|
55
|
+
"size": path.stat().st_size,
|
|
56
|
+
"sha256": _sha256(path),
|
|
57
|
+
}
|
|
58
|
+
)
|
|
59
|
+
return records
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
@contextmanager
|
|
63
|
+
def _cache_lock(cache_root, *, exclusive):
|
|
64
|
+
"""Coordinate cache readers with the short final installation step.
|
|
65
|
+
|
|
66
|
+
Verification and CI use a shared lock for the entire period in which cache
|
|
67
|
+
files may be read. Regeneration uses an exclusive lock only while replacing
|
|
68
|
+
the verified temporary cache, so building and hashing it does not block CI.
|
|
69
|
+
"""
|
|
70
|
+
|
|
71
|
+
cache_root = Path(cache_root).expanduser().resolve()
|
|
72
|
+
lock_path = cache_root / ".c4.lock"
|
|
73
|
+
if exclusive:
|
|
74
|
+
cache_root.mkdir(parents=True, exist_ok=True)
|
|
75
|
+
lock_path.touch(exist_ok=True)
|
|
76
|
+
mode = "r+"
|
|
77
|
+
operation = fcntl.LOCK_EX
|
|
78
|
+
else:
|
|
79
|
+
mode = "r"
|
|
80
|
+
operation = fcntl.LOCK_SH
|
|
81
|
+
|
|
82
|
+
try:
|
|
83
|
+
lock_file = lock_path.open(mode, encoding="utf-8")
|
|
84
|
+
except FileNotFoundError as exc:
|
|
85
|
+
raise FileNotFoundError(
|
|
86
|
+
f"Fixed calibration cache lock is missing: {lock_path}. "
|
|
87
|
+
"Regenerate the cache with this script."
|
|
88
|
+
) from exc
|
|
89
|
+
with lock_file:
|
|
90
|
+
fcntl.flock(lock_file, operation)
|
|
91
|
+
yield
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
def prepare(cache_root, *, local_files_only=False, force=False):
|
|
95
|
+
"""Build, verify, and transactionally install the pinned C4 cache."""
|
|
96
|
+
|
|
97
|
+
destination = _cache_path(cache_root)
|
|
98
|
+
if destination.exists() and not force:
|
|
99
|
+
raise FileExistsError(
|
|
100
|
+
f"Calibration cache already exists: {destination}. "
|
|
101
|
+
"Use --force only for an intentional regeneration."
|
|
102
|
+
)
|
|
103
|
+
|
|
104
|
+
source = Path(
|
|
105
|
+
hf_hub_download(
|
|
106
|
+
repo_id=DATASET_ID,
|
|
107
|
+
filename=DATA_FILE,
|
|
108
|
+
repo_type="dataset",
|
|
109
|
+
revision=DATASET_REVISION,
|
|
110
|
+
local_files_only=local_files_only,
|
|
111
|
+
)
|
|
112
|
+
)
|
|
113
|
+
source_sha256 = _sha256(source)
|
|
114
|
+
if source_sha256 != SOURCE_SHA256:
|
|
115
|
+
raise ValueError(
|
|
116
|
+
f"Unexpected SHA-256 for {DATA_FILE}: {source_sha256}; " f"expected {SOURCE_SHA256}"
|
|
117
|
+
)
|
|
118
|
+
|
|
119
|
+
dataset = datasets.load_dataset("json", data_files={"train": str(source)})
|
|
120
|
+
if len(dataset["train"]) != EXPECTED_TRAIN_ROWS:
|
|
121
|
+
raise ValueError(
|
|
122
|
+
f"Unexpected C4 train row count: {len(dataset['train'])}; "
|
|
123
|
+
f"expected {EXPECTED_TRAIN_ROWS}"
|
|
124
|
+
)
|
|
125
|
+
|
|
126
|
+
destination.parent.mkdir(parents=True, exist_ok=True)
|
|
127
|
+
temporary = destination.with_name(f".{destination.name}.tmp-{os.getpid()}")
|
|
128
|
+
if temporary.exists():
|
|
129
|
+
shutil.rmtree(temporary)
|
|
130
|
+
|
|
131
|
+
try:
|
|
132
|
+
dataset.save_to_disk(temporary)
|
|
133
|
+
# Fingerprints can change during save_to_disk(), so record the value that
|
|
134
|
+
# consumers will observe after loading the persisted cache.
|
|
135
|
+
persisted_dataset = datasets.load_from_disk(temporary)
|
|
136
|
+
manifest = {
|
|
137
|
+
"dataset_id": DATASET_ID,
|
|
138
|
+
"dataset_revision": DATASET_REVISION,
|
|
139
|
+
"data_files": {"train": DATA_FILE},
|
|
140
|
+
"source_sha256": source_sha256,
|
|
141
|
+
"datasets_version": datasets.__version__,
|
|
142
|
+
"files": _file_records(temporary),
|
|
143
|
+
"splits": {
|
|
144
|
+
"train": {
|
|
145
|
+
"num_rows": len(persisted_dataset["train"]),
|
|
146
|
+
"fingerprint": persisted_dataset["train"]._fingerprint,
|
|
147
|
+
"features": persisted_dataset["train"].features.to_dict(),
|
|
148
|
+
}
|
|
149
|
+
},
|
|
150
|
+
}
|
|
151
|
+
(temporary / MANIFEST_NAME).write_text(
|
|
152
|
+
json.dumps(manifest, indent=2, sort_keys=True) + "\n",
|
|
153
|
+
encoding="utf-8",
|
|
154
|
+
)
|
|
155
|
+
# Reject an incomplete or internally inconsistent cache before it can
|
|
156
|
+
# become visible at the shared destination.
|
|
157
|
+
_verify_cache(temporary)
|
|
158
|
+
|
|
159
|
+
backup = destination.with_name(f".{destination.name}.backup-{os.getpid()}")
|
|
160
|
+
with _cache_lock(cache_root, exclusive=True):
|
|
161
|
+
if destination.exists() and not force:
|
|
162
|
+
raise FileExistsError(
|
|
163
|
+
f"Calibration cache already exists: {destination}. "
|
|
164
|
+
"Use --force only for an intentional regeneration."
|
|
165
|
+
)
|
|
166
|
+
if backup.exists():
|
|
167
|
+
raise FileExistsError(f"Calibration cache backup already exists: {backup}")
|
|
168
|
+
|
|
169
|
+
moved_existing = False
|
|
170
|
+
try:
|
|
171
|
+
# Keep the previous cache available for rollback until the
|
|
172
|
+
# already-verified replacement has been installed.
|
|
173
|
+
if destination.exists():
|
|
174
|
+
destination.rename(backup)
|
|
175
|
+
moved_existing = True
|
|
176
|
+
temporary.rename(destination)
|
|
177
|
+
except Exception:
|
|
178
|
+
if moved_existing and not destination.exists():
|
|
179
|
+
backup.rename(destination)
|
|
180
|
+
raise
|
|
181
|
+
|
|
182
|
+
if moved_existing:
|
|
183
|
+
try:
|
|
184
|
+
shutil.rmtree(backup)
|
|
185
|
+
except OSError as exc:
|
|
186
|
+
print(
|
|
187
|
+
f"WARNING: failed to remove calibration cache backup: {exc}",
|
|
188
|
+
file=sys.stderr,
|
|
189
|
+
)
|
|
190
|
+
except Exception:
|
|
191
|
+
shutil.rmtree(temporary, ignore_errors=True)
|
|
192
|
+
raise
|
|
193
|
+
|
|
194
|
+
print(f"Prepared fixed C4 calibration cache: {destination}")
|
|
195
|
+
|
|
196
|
+
|
|
197
|
+
def _verify_cache(destination):
|
|
198
|
+
"""Validate the pinned source identity and every persisted cache artifact."""
|
|
199
|
+
|
|
200
|
+
manifest_path = destination / MANIFEST_NAME
|
|
201
|
+
if not manifest_path.is_file():
|
|
202
|
+
raise FileNotFoundError(f"Fixed calibration cache manifest is missing: {manifest_path}")
|
|
203
|
+
|
|
204
|
+
manifest = json.loads(manifest_path.read_text(encoding="utf-8"))
|
|
205
|
+
expected = {
|
|
206
|
+
"dataset_id": DATASET_ID,
|
|
207
|
+
"dataset_revision": DATASET_REVISION,
|
|
208
|
+
"data_files": {"train": DATA_FILE},
|
|
209
|
+
"source_sha256": SOURCE_SHA256,
|
|
210
|
+
}
|
|
211
|
+
for key, expected_value in expected.items():
|
|
212
|
+
if manifest.get(key) != expected_value:
|
|
213
|
+
raise ValueError(
|
|
214
|
+
f"Invalid fixed calibration cache manifest field {key!r}: "
|
|
215
|
+
f"{manifest.get(key)!r}; expected {expected_value!r}"
|
|
216
|
+
)
|
|
217
|
+
|
|
218
|
+
files = _file_records(destination)
|
|
219
|
+
if manifest.get("files") != files:
|
|
220
|
+
raise ValueError("Fixed calibration cache files do not match its manifest")
|
|
221
|
+
|
|
222
|
+
dataset = datasets.load_from_disk(destination)
|
|
223
|
+
if "train" not in dataset:
|
|
224
|
+
raise ValueError("Fixed calibration cache has no train split")
|
|
225
|
+
train_manifest = manifest.get("splits", {}).get("train", {})
|
|
226
|
+
if len(dataset["train"]) != EXPECTED_TRAIN_ROWS:
|
|
227
|
+
raise ValueError(
|
|
228
|
+
f"Invalid fixed calibration cache row count: {len(dataset['train'])}; "
|
|
229
|
+
f"expected {EXPECTED_TRAIN_ROWS}"
|
|
230
|
+
)
|
|
231
|
+
if train_manifest.get("num_rows") != EXPECTED_TRAIN_ROWS:
|
|
232
|
+
raise ValueError("Fixed calibration cache manifest has an invalid row count")
|
|
233
|
+
if dataset["train"]._fingerprint != train_manifest.get("fingerprint"):
|
|
234
|
+
raise ValueError("Fixed calibration cache fingerprint does not match its manifest")
|
|
235
|
+
if dataset["train"].features.to_dict() != train_manifest.get("features"):
|
|
236
|
+
raise ValueError("Fixed calibration cache schema does not match its manifest")
|
|
237
|
+
|
|
238
|
+
print(
|
|
239
|
+
"Verified fixed C4 calibration cache: "
|
|
240
|
+
f"{destination} ({len(dataset['train'])} train rows)"
|
|
241
|
+
)
|
|
242
|
+
|
|
243
|
+
|
|
244
|
+
def verify(cache_root):
|
|
245
|
+
"""Verify a cache while preventing concurrent regeneration from replacing it."""
|
|
246
|
+
|
|
247
|
+
with _cache_lock(cache_root, exclusive=False):
|
|
248
|
+
_verify_cache(_cache_path(cache_root))
|
|
249
|
+
|
|
250
|
+
|
|
251
|
+
def main():
|
|
252
|
+
parser = argparse.ArgumentParser(description=__doc__)
|
|
253
|
+
parser.add_argument(
|
|
254
|
+
"--cache-root",
|
|
255
|
+
default=os.environ.get("ONECOMP_CALIB_CACHE"),
|
|
256
|
+
help="Cache root containing the c4 directory (default: ONECOMP_CALIB_CACHE)",
|
|
257
|
+
)
|
|
258
|
+
parser.add_argument("--verify", action="store_true", help="Verify without writing")
|
|
259
|
+
parser.add_argument(
|
|
260
|
+
"--local-files-only",
|
|
261
|
+
action="store_true",
|
|
262
|
+
help="Use an already downloaded Hugging Face source shard",
|
|
263
|
+
)
|
|
264
|
+
parser.add_argument(
|
|
265
|
+
"--force",
|
|
266
|
+
action="store_true",
|
|
267
|
+
help="Replace an existing cache intentionally",
|
|
268
|
+
)
|
|
269
|
+
args = parser.parse_args()
|
|
270
|
+
if not args.cache_root:
|
|
271
|
+
parser.error("--cache-root or ONECOMP_CALIB_CACHE is required")
|
|
272
|
+
|
|
273
|
+
if args.verify:
|
|
274
|
+
verify(args.cache_root)
|
|
275
|
+
else:
|
|
276
|
+
prepare(
|
|
277
|
+
args.cache_root,
|
|
278
|
+
local_files_only=args.local_files_only,
|
|
279
|
+
force=args.force,
|
|
280
|
+
)
|
|
281
|
+
|
|
282
|
+
|
|
283
|
+
if __name__ == "__main__":
|
|
284
|
+
main()
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|