lmcache-cli 0.4.5.dev0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- lmcache/__init__.py +84 -0
- lmcache/_version.py +24 -0
- lmcache/cli/__init__.py +1 -0
- lmcache/cli/commands/__init__.py +34 -0
- lmcache/cli/commands/base.py +157 -0
- lmcache/cli/commands/bench/__init__.py +557 -0
- lmcache/cli/commands/bench/engine_bench/__init__.py +1 -0
- lmcache/cli/commands/bench/engine_bench/config.py +245 -0
- lmcache/cli/commands/bench/engine_bench/interactive/__init__.py +274 -0
- lmcache/cli/commands/bench/engine_bench/interactive/config.json +10 -0
- lmcache/cli/commands/bench/engine_bench/interactive/schema.py +352 -0
- lmcache/cli/commands/bench/engine_bench/interactive/state.py +327 -0
- lmcache/cli/commands/bench/engine_bench/interactive/terminal.py +291 -0
- lmcache/cli/commands/bench/engine_bench/progress.py +145 -0
- lmcache/cli/commands/bench/engine_bench/request_sender.py +232 -0
- lmcache/cli/commands/bench/engine_bench/stats.py +275 -0
- lmcache/cli/commands/bench/engine_bench/workloads/__init__.py +153 -0
- lmcache/cli/commands/bench/engine_bench/workloads/base.py +122 -0
- lmcache/cli/commands/bench/engine_bench/workloads/long_doc_permutator.py +435 -0
- lmcache/cli/commands/bench/engine_bench/workloads/long_doc_qa.py +281 -0
- lmcache/cli/commands/bench/engine_bench/workloads/multi_round_chat.py +337 -0
- lmcache/cli/commands/bench/engine_bench/workloads/random_prefill.py +178 -0
- lmcache/cli/commands/describe.py +310 -0
- lmcache/cli/commands/kvcache.py +133 -0
- lmcache/cli/commands/mock.py +75 -0
- lmcache/cli/commands/ping.py +113 -0
- lmcache/cli/commands/query/__init__.py +155 -0
- lmcache/cli/commands/query/prompt.py +134 -0
- lmcache/cli/commands/query/request.py +357 -0
- lmcache/cli/commands/server.py +99 -0
- lmcache/cli/commands/tool/__init__.py +63 -0
- lmcache/cli/commands/tool/cache_simulator.py +113 -0
- lmcache/cli/commands/trace/__init__.py +505 -0
- lmcache/cli/commands/trace/dispatch.py +249 -0
- lmcache/cli/commands/trace/driver.py +372 -0
- lmcache/cli/commands/trace/stats.py +289 -0
- lmcache/cli/documents/lmcache.txt +11 -0
- lmcache/cli/main.py +42 -0
- lmcache/cli/metrics/__init__.py +29 -0
- lmcache/cli/metrics/formatter.py +171 -0
- lmcache/cli/metrics/handler.py +94 -0
- lmcache/cli/metrics/metrics.py +161 -0
- lmcache/cli/metrics/section.py +77 -0
- lmcache/connections.py +173 -0
- lmcache/integration/__init__.py +2 -0
- lmcache/integration/base_service_factory.py +165 -0
- lmcache/integration/request_telemetry/__init__.py +1 -0
- lmcache/integration/request_telemetry/base.py +51 -0
- lmcache/integration/request_telemetry/factory.py +113 -0
- lmcache/integration/request_telemetry/fastapi.py +109 -0
- lmcache/integration/request_telemetry/noop.py +35 -0
- lmcache/integration/sglang/__init__.py +2 -0
- lmcache/integration/sglang/sglang_adapter.py +326 -0
- lmcache/integration/sglang/utils.py +39 -0
- lmcache/integration/vllm/__init__.py +1 -0
- lmcache/integration/vllm/lmcache_connector_v1.py +213 -0
- lmcache/integration/vllm/lmcache_connector_v1_085.py +150 -0
- lmcache/integration/vllm/lmcache_mp_connector_0180.py +1072 -0
- lmcache/integration/vllm/tests/test_mm_hash_utils.py +112 -0
- lmcache/integration/vllm/utils.py +433 -0
- lmcache/integration/vllm/vllm_multi_process_adapter.py +1090 -0
- lmcache/integration/vllm/vllm_service_factory.py +339 -0
- lmcache/integration/vllm/vllm_v1_adapter.py +1713 -0
- lmcache/logging.py +107 -0
- lmcache/native_storage_ops.pyi +230 -0
- lmcache/non_cuda_equivalents.py +1424 -0
- lmcache/observability.py +1958 -0
- lmcache/storage_backend/serde/__init__.py +1 -0
- lmcache/storage_backend/serde/cachegen_basics.py +210 -0
- lmcache/storage_backend/serde/cachegen_decoder.py +207 -0
- lmcache/storage_backend/serde/cachegen_encoder.py +394 -0
- lmcache/storage_backend/serde/serde.py +75 -0
- lmcache/tools/__init__.py +1 -0
- lmcache/tools/cache_simulator/README.md +392 -0
- lmcache/tools/cache_simulator/__init__.py +1 -0
- lmcache/tools/cache_simulator/docs/simulate_example.png +0 -0
- lmcache/tools/cache_simulator/docs/sweep_example.png +0 -0
- lmcache/tools/cache_simulator/gen_bench_dataset.py +360 -0
- lmcache/tools/cache_simulator/lru_cache.py +124 -0
- lmcache/tools/cache_simulator/plot_hit_rate.py +231 -0
- lmcache/tools/cache_simulator/simulator.py +795 -0
- lmcache/tools/controller_benchmark/README.md +161 -0
- lmcache/tools/controller_benchmark/__init__.py +1 -0
- lmcache/tools/controller_benchmark/__main__.py +331 -0
- lmcache/tools/controller_benchmark/benchmark.py +660 -0
- lmcache/tools/controller_benchmark/config.py +44 -0
- lmcache/tools/controller_benchmark/constants.py +10 -0
- lmcache/tools/controller_benchmark/handlers/__init__.py +46 -0
- lmcache/tools/controller_benchmark/handlers/admit.py +52 -0
- lmcache/tools/controller_benchmark/handlers/base.py +47 -0
- lmcache/tools/controller_benchmark/handlers/deregister.py +49 -0
- lmcache/tools/controller_benchmark/handlers/evict.py +52 -0
- lmcache/tools/controller_benchmark/handlers/heartbeat.py +56 -0
- lmcache/tools/controller_benchmark/handlers/p2p_lookup.py +47 -0
- lmcache/tools/controller_benchmark/handlers/register.py +56 -0
- lmcache/tools/mp_status_viewer/__init__.py +1 -0
- lmcache/tools/mp_status_viewer/__main__.py +95 -0
- lmcache/usage_context.py +417 -0
- lmcache/utils.py +665 -0
- lmcache/v1/__init__.py +2 -0
- lmcache/v1/api_server/__init__.py +2 -0
- lmcache/v1/api_server/__main__.py +537 -0
- lmcache/v1/basic_check.py +112 -0
- lmcache/v1/cache_controller/__init__.py +9 -0
- lmcache/v1/cache_controller/commands/__init__.py +15 -0
- lmcache/v1/cache_controller/commands/base.py +35 -0
- lmcache/v1/cache_controller/commands/full_sync.py +49 -0
- lmcache/v1/cache_controller/config.py +176 -0
- lmcache/v1/cache_controller/controller_manager.py +535 -0
- lmcache/v1/cache_controller/controllers/__init__.py +11 -0
- lmcache/v1/cache_controller/controllers/full_sync_tracker.py +473 -0
- lmcache/v1/cache_controller/controllers/kv_controller.py +439 -0
- lmcache/v1/cache_controller/controllers/registration_controller.py +282 -0
- lmcache/v1/cache_controller/executor.py +463 -0
- lmcache/v1/cache_controller/frontend/static/css/style.css +201 -0
- lmcache/v1/cache_controller/frontend/static/img/logo.png +0 -0
- lmcache/v1/cache_controller/frontend/static/index.html +234 -0
- lmcache/v1/cache_controller/frontend/static/js/controller_app.js +660 -0
- lmcache/v1/cache_controller/full_sync_sender.py +475 -0
- lmcache/v1/cache_controller/locks.py +149 -0
- lmcache/v1/cache_controller/message.py +828 -0
- lmcache/v1/cache_controller/observability.py +208 -0
- lmcache/v1/cache_controller/utils.py +679 -0
- lmcache/v1/cache_controller/worker.py +665 -0
- lmcache/v1/cache_engine.py +2058 -0
- lmcache/v1/cache_interface.py +19 -0
- lmcache/v1/check/__init__.py +74 -0
- lmcache/v1/check/check_mode_gen.py +86 -0
- lmcache/v1/check/check_mode_test_l2_adapter.py +284 -0
- lmcache/v1/check/check_mode_test_remote.py +155 -0
- lmcache/v1/check/check_mode_test_storage_manager.py +142 -0
- lmcache/v1/check/utils.py +571 -0
- lmcache/v1/compute/__init__.py +2 -0
- lmcache/v1/compute/attention/__init__.py +0 -0
- lmcache/v1/compute/attention/abstract.py +39 -0
- lmcache/v1/compute/attention/flash_attn.py +129 -0
- lmcache/v1/compute/attention/flash_infer_sparse.py +284 -0
- lmcache/v1/compute/attention/metadata.py +85 -0
- lmcache/v1/compute/attention/utils.py +14 -0
- lmcache/v1/compute/blend/__init__.py +7 -0
- lmcache/v1/compute/blend/blender.py +168 -0
- lmcache/v1/compute/blend/metadata.py +34 -0
- lmcache/v1/compute/blend/utils.py +63 -0
- lmcache/v1/compute/models/__init__.py +0 -0
- lmcache/v1/compute/models/base.py +141 -0
- lmcache/v1/compute/models/llama.py +9 -0
- lmcache/v1/compute/models/qwen3.py +24 -0
- lmcache/v1/compute/models/utils.py +68 -0
- lmcache/v1/compute/positional_encoding.py +199 -0
- lmcache/v1/config.py +848 -0
- lmcache/v1/config_base.py +848 -0
- lmcache/v1/distributed/api.py +248 -0
- lmcache/v1/distributed/config.py +321 -0
- lmcache/v1/distributed/error.py +64 -0
- lmcache/v1/distributed/eviction.py +192 -0
- lmcache/v1/distributed/eviction_policy/__init__.py +21 -0
- lmcache/v1/distributed/eviction_policy/factory.py +27 -0
- lmcache/v1/distributed/eviction_policy/lru.py +244 -0
- lmcache/v1/distributed/eviction_policy/noop.py +50 -0
- lmcache/v1/distributed/internal_api.py +170 -0
- lmcache/v1/distributed/l1_manager.py +835 -0
- lmcache/v1/distributed/l2_adapters/__init__.py +67 -0
- lmcache/v1/distributed/l2_adapters/base.py +360 -0
- lmcache/v1/distributed/l2_adapters/config.py +385 -0
- lmcache/v1/distributed/l2_adapters/factory.py +205 -0
- lmcache/v1/distributed/l2_adapters/fs_l2_adapter.py +747 -0
- lmcache/v1/distributed/l2_adapters/fs_native_l2_adapter.py +167 -0
- lmcache/v1/distributed/l2_adapters/mock_l2_adapter.py +516 -0
- lmcache/v1/distributed/l2_adapters/mooncake_store_l2_adapter.py +135 -0
- lmcache/v1/distributed/l2_adapters/native_connector_l2_adapter.py +468 -0
- lmcache/v1/distributed/l2_adapters/native_plugin_l2_adapter.py +199 -0
- lmcache/v1/distributed/l2_adapters/nixl_store_dynamic_l2_adapter.py +831 -0
- lmcache/v1/distributed/l2_adapters/nixl_store_l2_adapter.py +983 -0
- lmcache/v1/distributed/l2_adapters/plugin_l2_adapter.py +210 -0
- lmcache/v1/distributed/l2_adapters/resp_l2_adapter.py +176 -0
- lmcache/v1/distributed/memory_manager.py +179 -0
- lmcache/v1/distributed/storage_controller.py +39 -0
- lmcache/v1/distributed/storage_controllers/__init__.py +43 -0
- lmcache/v1/distributed/storage_controllers/eviction_controller.py +242 -0
- lmcache/v1/distributed/storage_controllers/prefetch_controller.py +830 -0
- lmcache/v1/distributed/storage_controllers/prefetch_policy.py +193 -0
- lmcache/v1/distributed/storage_controllers/store_controller.py +452 -0
- lmcache/v1/distributed/storage_controllers/store_policy.py +213 -0
- lmcache/v1/distributed/storage_manager.py +532 -0
- lmcache/v1/event_manager.py +145 -0
- lmcache/v1/exceptions/__init__.py +16 -0
- lmcache/v1/gpu_connector/__init__.py +126 -0
- lmcache/v1/gpu_connector/gpu_connectors.py +1906 -0
- lmcache/v1/gpu_connector/gpu_ops.py +85 -0
- lmcache/v1/gpu_connector/hpu_connector.py +326 -0
- lmcache/v1/gpu_connector/mock_gpu_connector.py +67 -0
- lmcache/v1/gpu_connector/utils.py +890 -0
- lmcache/v1/gpu_connector/xpu_connectors.py +916 -0
- lmcache/v1/health_monitor/__init__.py +1 -0
- lmcache/v1/health_monitor/base.py +587 -0
- lmcache/v1/health_monitor/checks/__init__.py +1 -0
- lmcache/v1/health_monitor/checks/remote_backend_check.py +304 -0
- lmcache/v1/health_monitor/constants.py +36 -0
- lmcache/v1/internal_api_server/__init__.py +0 -0
- lmcache/v1/internal_api_server/api_registry.py +59 -0
- lmcache/v1/internal_api_server/api_server.py +120 -0
- lmcache/v1/internal_api_server/common/__init__.py +1 -0
- lmcache/v1/internal_api_server/common/env_api.py +22 -0
- lmcache/v1/internal_api_server/common/loglevel_api.py +57 -0
- lmcache/v1/internal_api_server/common/metrics_api.py +29 -0
- lmcache/v1/internal_api_server/common/periodic_thread_api.py +138 -0
- lmcache/v1/internal_api_server/common/run_script_api.py +73 -0
- lmcache/v1/internal_api_server/common/thread_api.py +63 -0
- lmcache/v1/internal_api_server/controller/__init__.py +1 -0
- lmcache/v1/internal_api_server/controller/key_stats_api.py +81 -0
- lmcache/v1/internal_api_server/controller/worker_info_api.py +136 -0
- lmcache/v1/internal_api_server/utils.py +43 -0
- lmcache/v1/internal_api_server/vllm/__init__.py +1 -0
- lmcache/v1/internal_api_server/vllm/backend_api.py +221 -0
- lmcache/v1/internal_api_server/vllm/bypass_api.py +204 -0
- lmcache/v1/internal_api_server/vllm/cache_api.py +895 -0
- lmcache/v1/internal_api_server/vllm/chunk_statistics_api.py +141 -0
- lmcache/v1/internal_api_server/vllm/conf_api.py +147 -0
- lmcache/v1/internal_api_server/vllm/freeze_api.py +172 -0
- lmcache/v1/internal_api_server/vllm/hot_cache_api.py +184 -0
- lmcache/v1/internal_api_server/vllm/inference_api.py +65 -0
- lmcache/v1/internal_api_server/vllm/load_fs_chunks_api.py +320 -0
- lmcache/v1/internal_api_server/vllm/lookup_api.py +145 -0
- lmcache/v1/internal_api_server/vllm/version_api.py +25 -0
- lmcache/v1/kv_layer_groups.py +267 -0
- lmcache/v1/lazy_memory_allocator.py +284 -0
- lmcache/v1/lookup_client/__init__.py +25 -0
- lmcache/v1/lookup_client/abstract_client.py +77 -0
- lmcache/v1/lookup_client/async_lookup_message.py +50 -0
- lmcache/v1/lookup_client/chunk_statistics_lookup_client.py +200 -0
- lmcache/v1/lookup_client/factory.py +251 -0
- lmcache/v1/lookup_client/hit_limit_lookup_client.py +86 -0
- lmcache/v1/lookup_client/lmcache_async_lookup_client.py +407 -0
- lmcache/v1/lookup_client/lmcache_lookup_client.py +285 -0
- lmcache/v1/lookup_client/lmcache_lookup_client_bypass.py +99 -0
- lmcache/v1/lookup_client/mooncake_lookup_client.py +87 -0
- lmcache/v1/lookup_client/record_strategies/__init__.py +77 -0
- lmcache/v1/lookup_client/record_strategies/base.py +327 -0
- lmcache/v1/lookup_client/record_strategies/file_hash.py +130 -0
- lmcache/v1/lookup_client/record_strategies/memory_bloom_filter.py +81 -0
- lmcache/v1/manager.py +539 -0
- lmcache/v1/memory_management.py +2619 -0
- lmcache/v1/metadata.py +114 -0
- lmcache/v1/mp_observability/AGENTS.override.md +21 -0
- lmcache/v1/mp_observability/README.md +204 -0
- lmcache/v1/mp_observability/config.py +340 -0
- lmcache/v1/mp_observability/event.py +100 -0
- lmcache/v1/mp_observability/event_bus.py +313 -0
- lmcache/v1/mp_observability/otel_init.py +145 -0
- lmcache/v1/mp_observability/subscribers/__init__.py +28 -0
- lmcache/v1/mp_observability/subscribers/logging/__init__.py +19 -0
- lmcache/v1/mp_observability/subscribers/logging/l1.py +56 -0
- lmcache/v1/mp_observability/subscribers/logging/l2.py +73 -0
- lmcache/v1/mp_observability/subscribers/logging/lookup_hash.py +209 -0
- lmcache/v1/mp_observability/subscribers/logging/mp_server.py +90 -0
- lmcache/v1/mp_observability/subscribers/logging/sm.py +59 -0
- lmcache/v1/mp_observability/subscribers/metrics/__init__.py +20 -0
- lmcache/v1/mp_observability/subscribers/metrics/l0_lifecycle.py +290 -0
- lmcache/v1/mp_observability/subscribers/metrics/l1.py +55 -0
- lmcache/v1/mp_observability/subscribers/metrics/l1_lifecycle.py +166 -0
- lmcache/v1/mp_observability/subscribers/metrics/l2.py +121 -0
- lmcache/v1/mp_observability/subscribers/metrics/sm.py +69 -0
- lmcache/v1/mp_observability/subscribers/tracing/__init__.py +12 -0
- lmcache/v1/mp_observability/subscribers/tracing/mp_server.py +333 -0
- lmcache/v1/mp_observability/subscribers/tracing/span_registry.py +148 -0
- lmcache/v1/mp_observability/trace/__init__.py +50 -0
- lmcache/v1/mp_observability/trace/codecs.py +255 -0
- lmcache/v1/mp_observability/trace/decorator.py +147 -0
- lmcache/v1/mp_observability/trace/format.py +132 -0
- lmcache/v1/mp_observability/trace/lifecycle.py +83 -0
- lmcache/v1/mp_observability/trace/reader.py +167 -0
- lmcache/v1/mp_observability/trace/recorder.py +300 -0
- lmcache/v1/multiprocess/__init__.py +0 -0
- lmcache/v1/multiprocess/affinity_pool.py +102 -0
- lmcache/v1/multiprocess/blend_server_v2.py +891 -0
- lmcache/v1/multiprocess/config.py +253 -0
- lmcache/v1/multiprocess/custom_types.py +281 -0
- lmcache/v1/multiprocess/futures.py +194 -0
- lmcache/v1/multiprocess/gpu_context.py +511 -0
- lmcache/v1/multiprocess/http_server.py +235 -0
- lmcache/v1/multiprocess/mp_runtime_plugin_launcher.py +130 -0
- lmcache/v1/multiprocess/mq.py +732 -0
- lmcache/v1/multiprocess/protocol.py +86 -0
- lmcache/v1/multiprocess/protocols/README.md +213 -0
- lmcache/v1/multiprocess/protocols/__init__.py +127 -0
- lmcache/v1/multiprocess/protocols/base.py +89 -0
- lmcache/v1/multiprocess/protocols/blend.py +109 -0
- lmcache/v1/multiprocess/protocols/blend_v2.py +57 -0
- lmcache/v1/multiprocess/protocols/controller.py +53 -0
- lmcache/v1/multiprocess/protocols/debug.py +34 -0
- lmcache/v1/multiprocess/protocols/engine.py +146 -0
- lmcache/v1/multiprocess/protocols/observability.py +39 -0
- lmcache/v1/multiprocess/server.py +1134 -0
- lmcache/v1/multiprocess/session.py +190 -0
- lmcache/v1/multiprocess/token_hasher.py +441 -0
- lmcache/v1/offload_server/__init__.py +17 -0
- lmcache/v1/offload_server/abstract_server.py +37 -0
- lmcache/v1/offload_server/message.py +30 -0
- lmcache/v1/offload_server/zmq_server.py +122 -0
- lmcache/v1/periodic_thread.py +579 -0
- lmcache/v1/pin_monitor.py +246 -0
- lmcache/v1/plugin/__init__.py +0 -0
- lmcache/v1/plugin/runtime_plugin_launcher.py +211 -0
- lmcache/v1/protocol.py +317 -0
- lmcache/v1/rpc/__init__.py +17 -0
- lmcache/v1/rpc/transport.py +105 -0
- lmcache/v1/rpc/zmq_transport.py +213 -0
- lmcache/v1/rpc_utils.py +165 -0
- lmcache/v1/server/__init__.py +2 -0
- lmcache/v1/server/__main__.py +170 -0
- lmcache/v1/server/storage_backend/__init__.py +21 -0
- lmcache/v1/server/storage_backend/abstract_backend.py +80 -0
- lmcache/v1/server/storage_backend/local_backend.py +75 -0
- lmcache/v1/server/utils.py +21 -0
- lmcache/v1/standalone/__init__.py +1 -0
- lmcache/v1/standalone/__main__.py +583 -0
- lmcache/v1/standalone/manager.py +80 -0
- lmcache/v1/standalone/standalone_service_factory.py +86 -0
- lmcache/v1/storage_backend/__init__.py +313 -0
- lmcache/v1/storage_backend/abstract_backend.py +445 -0
- lmcache/v1/storage_backend/audit_backend.py +233 -0
- lmcache/v1/storage_backend/batched_message_sender.py +222 -0
- lmcache/v1/storage_backend/cache_policy/__init__.py +45 -0
- lmcache/v1/storage_backend/cache_policy/base_policy.py +87 -0
- lmcache/v1/storage_backend/cache_policy/fifo.py +58 -0
- lmcache/v1/storage_backend/cache_policy/lfu.py +105 -0
- lmcache/v1/storage_backend/cache_policy/lru.py +81 -0
- lmcache/v1/storage_backend/cache_policy/mru.py +61 -0
- lmcache/v1/storage_backend/connector/__init__.py +443 -0
- lmcache/v1/storage_backend/connector/audit_adapter.py +77 -0
- lmcache/v1/storage_backend/connector/audit_connector.py +320 -0
- lmcache/v1/storage_backend/connector/base_connector.py +379 -0
- lmcache/v1/storage_backend/connector/blackhole_adapter.py +21 -0
- lmcache/v1/storage_backend/connector/blackhole_connector.py +37 -0
- lmcache/v1/storage_backend/connector/eic_adapter.py +31 -0
- lmcache/v1/storage_backend/connector/eic_connector.py +757 -0
- lmcache/v1/storage_backend/connector/external_adapter.py +79 -0
- lmcache/v1/storage_backend/connector/fs_adapter.py +51 -0
- lmcache/v1/storage_backend/connector/fs_connector.py +403 -0
- lmcache/v1/storage_backend/connector/infinistore_adapter.py +56 -0
- lmcache/v1/storage_backend/connector/infinistore_connector.py +177 -0
- lmcache/v1/storage_backend/connector/instrumented_connector.py +219 -0
- lmcache/v1/storage_backend/connector/lm_adapter.py +31 -0
- lmcache/v1/storage_backend/connector/lm_connector.py +176 -0
- lmcache/v1/storage_backend/connector/mock_adapter.py +57 -0
- lmcache/v1/storage_backend/connector/mock_connector.py +349 -0
- lmcache/v1/storage_backend/connector/mooncakestore_adapter.py +43 -0
- lmcache/v1/storage_backend/connector/mooncakestore_connector.py +614 -0
- lmcache/v1/storage_backend/connector/redis_adapter.py +181 -0
- lmcache/v1/storage_backend/connector/redis_connector.py +828 -0
- lmcache/v1/storage_backend/connector/s3_adapter.py +59 -0
- lmcache/v1/storage_backend/connector/s3_connector.py +699 -0
- lmcache/v1/storage_backend/connector/sagemaker_hyperpod_adapter.py +233 -0
- lmcache/v1/storage_backend/connector/sagemaker_hyperpod_connector.py +987 -0
- lmcache/v1/storage_backend/connector/valkey_adapter.py +114 -0
- lmcache/v1/storage_backend/connector/valkey_connector.py +627 -0
- lmcache/v1/storage_backend/gds_backend.py +1199 -0
- lmcache/v1/storage_backend/job_executor/__init__.py +0 -0
- lmcache/v1/storage_backend/job_executor/base_executor.py +34 -0
- lmcache/v1/storage_backend/job_executor/pq_executor.py +235 -0
- lmcache/v1/storage_backend/local_cpu_backend.py +810 -0
- lmcache/v1/storage_backend/local_disk_backend.py +656 -0
- lmcache/v1/storage_backend/maru_backend.py +734 -0
- lmcache/v1/storage_backend/naive_serde/__init__.py +50 -0
- lmcache/v1/storage_backend/naive_serde/cachegen_basics.py +133 -0
- lmcache/v1/storage_backend/naive_serde/cachegen_decoder.py +135 -0
- lmcache/v1/storage_backend/naive_serde/cachegen_encoder.py +83 -0
- lmcache/v1/storage_backend/naive_serde/kivi_serde.py +22 -0
- lmcache/v1/storage_backend/naive_serde/naive_serde.py +18 -0
- lmcache/v1/storage_backend/naive_serde/serde.py +37 -0
- lmcache/v1/storage_backend/native_clients/connector_client_base.py +165 -0
- lmcache/v1/storage_backend/native_clients/resp_client.py +35 -0
- lmcache/v1/storage_backend/nixl_storage_backend.py +1400 -0
- lmcache/v1/storage_backend/p2p_backend.py +788 -0
- lmcache/v1/storage_backend/path_sharder.py +117 -0
- lmcache/v1/storage_backend/pd_backend.py +646 -0
- lmcache/v1/storage_backend/plugins/dax_backend.py +1443 -0
- lmcache/v1/storage_backend/plugins/rust_raw_block_backend.py +1361 -0
- lmcache/v1/storage_backend/remote_backend.py +624 -0
- lmcache/v1/storage_backend/resp_client.py +227 -0
- lmcache/v1/storage_backend/storage_backend_listener.py +19 -0
- lmcache/v1/storage_backend/storage_manager.py +1352 -0
- lmcache/v1/system_detection.py +110 -0
- lmcache/v1/token_database.py +551 -0
- lmcache/v1/transfer_channel/__init__.py +83 -0
- lmcache/v1/transfer_channel/abstract.py +285 -0
- lmcache/v1/transfer_channel/mock_memory_channel.py +156 -0
- lmcache/v1/transfer_channel/nixl_channel.py +639 -0
- lmcache/v1/transfer_channel/py_socket_channel.py +260 -0
- lmcache/v1/transfer_channel/transfer_utils.py +63 -0
- lmcache/v1/utils/__init__.py +1 -0
- lmcache/v1/utils/bloom_filter.py +109 -0
- lmcache/v1/utils/cache_utils.py +125 -0
- lmcache_cli-0.4.5.dev0.dist-info/METADATA +185 -0
- lmcache_cli-0.4.5.dev0.dist-info/RECORD +399 -0
- lmcache_cli-0.4.5.dev0.dist-info/WHEEL +5 -0
- lmcache_cli-0.4.5.dev0.dist-info/entry_points.txt +2 -0
- lmcache_cli-0.4.5.dev0.dist-info/licenses/LICENSE +201 -0
- lmcache_cli-0.4.5.dev0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,112 @@
|
|
|
1
|
+
# SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
# Standard
|
|
3
|
+
import dataclasses
|
|
4
|
+
|
|
5
|
+
# Third Party
|
|
6
|
+
import torch
|
|
7
|
+
|
|
8
|
+
# First Party
|
|
9
|
+
from lmcache.integration.vllm.utils import (
|
|
10
|
+
apply_mm_hashes_to_token_ids,
|
|
11
|
+
hex_hash_to_int16,
|
|
12
|
+
)
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
@dataclasses.dataclass(frozen=True)
|
|
16
|
+
class DummyPlaceholderRange:
|
|
17
|
+
offset: int
|
|
18
|
+
length: int
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def test_hex_hash_to_int16_accepts_hex_and_non_hex() -> None:
|
|
22
|
+
# Hex behavior preserved (with and without 0x prefix).
|
|
23
|
+
assert hex_hash_to_int16("0000") == 0
|
|
24
|
+
assert hex_hash_to_int16("ffff") == 0xFFFF
|
|
25
|
+
assert hex_hash_to_int16("0xFFFF") == 0xFFFF
|
|
26
|
+
assert hex_hash_to_int16("0x0001") == 1
|
|
27
|
+
|
|
28
|
+
# Non-hex identifiers must not raise and must be deterministic.
|
|
29
|
+
s = "chatcmpl-a2a48871c4aad192-image-0"
|
|
30
|
+
v1 = hex_hash_to_int16(s)
|
|
31
|
+
v2 = hex_hash_to_int16(s)
|
|
32
|
+
assert isinstance(v1, int)
|
|
33
|
+
assert 0 <= v1 <= 0xFFFF
|
|
34
|
+
assert v1 == v2
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def test_hex_hash_to_int16_hex_variants_whitespace_and_truncation() -> None:
|
|
38
|
+
# Whitespace should be ignored and case should not matter.
|
|
39
|
+
assert hex_hash_to_int16(" FfFf ") == 0xFFFF
|
|
40
|
+
assert hex_hash_to_int16("\n0x00aB\t") == 0x00AB
|
|
41
|
+
|
|
42
|
+
# Long hex should be truncated to 16 bits via masking.
|
|
43
|
+
assert hex_hash_to_int16("123456") == 0x3456
|
|
44
|
+
assert hex_hash_to_int16("0x123456") == 0x3456
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def test_hex_hash_to_int16_empty_and_invalid_hex_are_safe_and_deterministic() -> None:
|
|
48
|
+
# Empty (or effectively empty) values should not raise.
|
|
49
|
+
for s in ("", " ", "0x"):
|
|
50
|
+
v1 = hex_hash_to_int16(s)
|
|
51
|
+
v2 = hex_hash_to_int16(s)
|
|
52
|
+
assert isinstance(v1, int)
|
|
53
|
+
assert 0 <= v1 <= 0xFFFF
|
|
54
|
+
assert v1 == v2
|
|
55
|
+
|
|
56
|
+
# Invalid "hex-looking" strings must fall back to hashing.
|
|
57
|
+
for s in ("0xGG", "deadbeeg", "0x12xz"):
|
|
58
|
+
v1 = hex_hash_to_int16(s)
|
|
59
|
+
v2 = hex_hash_to_int16(s)
|
|
60
|
+
assert isinstance(v1, int)
|
|
61
|
+
assert 0 <= v1 <= 0xFFFF
|
|
62
|
+
assert v1 == v2
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
def test_hex_hash_to_int16_non_string_inputs_are_safe() -> None:
|
|
66
|
+
# Be defensive: callers may pass None or other non-string types.
|
|
67
|
+
for val in (None, 0, 12345, 3.14, b"deadbeef"):
|
|
68
|
+
v1 = hex_hash_to_int16(val) # type: ignore[arg-type]
|
|
69
|
+
v2 = hex_hash_to_int16(val) # type: ignore[arg-type]
|
|
70
|
+
assert isinstance(v1, int)
|
|
71
|
+
assert 0 <= v1 <= 0xFFFF
|
|
72
|
+
assert v1 == v2
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def test_apply_mm_hashes_to_token_ids_handles_non_hex_mm_hash() -> None:
|
|
76
|
+
token_ids = torch.arange(0, 10, dtype=torch.long)
|
|
77
|
+
mm_hashes = ["chatcmpl-a2a48871c4aad192-image-0"]
|
|
78
|
+
mm_positions = [DummyPlaceholderRange(offset=2, length=4)]
|
|
79
|
+
|
|
80
|
+
out = apply_mm_hashes_to_token_ids(token_ids.clone(), mm_hashes, mm_positions)
|
|
81
|
+
expected_val = hex_hash_to_int16(mm_hashes[0])
|
|
82
|
+
assert out[2:6].tolist() == [expected_val] * 4
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
def test_apply_mm_hashes_to_token_ids_out_of_bounds_is_safe() -> None:
|
|
86
|
+
token_ids = torch.zeros(5, dtype=torch.long)
|
|
87
|
+
mm_hashes = ["deadbeef"]
|
|
88
|
+
mm_positions = [DummyPlaceholderRange(offset=999, length=10)]
|
|
89
|
+
|
|
90
|
+
out = apply_mm_hashes_to_token_ids(token_ids.clone(), mm_hashes, mm_positions)
|
|
91
|
+
assert out.tolist() == token_ids.tolist()
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
def test_apply_mm_hashes_to_token_ids_multiple_placeholders_and_length_mismatch() -> (
|
|
95
|
+
None
|
|
96
|
+
):
|
|
97
|
+
token_ids = torch.zeros(12, dtype=torch.long)
|
|
98
|
+
mm_hashes = ["deadbeef", "chatcmpl-a2a48871c4aad192-image-0", "EXTRA_HASH_IGNORED"]
|
|
99
|
+
mm_positions = [
|
|
100
|
+
DummyPlaceholderRange(offset=0, length=3),
|
|
101
|
+
DummyPlaceholderRange(offset=5, length=4),
|
|
102
|
+
]
|
|
103
|
+
|
|
104
|
+
out = apply_mm_hashes_to_token_ids(token_ids.clone(), mm_hashes, mm_positions)
|
|
105
|
+
v0 = hex_hash_to_int16(mm_hashes[0])
|
|
106
|
+
v1 = hex_hash_to_int16(mm_hashes[1])
|
|
107
|
+
|
|
108
|
+
assert out[0:3].tolist() == [v0] * 3
|
|
109
|
+
assert out[5:9].tolist() == [v1] * 4
|
|
110
|
+
# Other regions remain unchanged.
|
|
111
|
+
assert out[3:5].tolist() == [0, 0]
|
|
112
|
+
assert out[9:12].tolist() == [0, 0, 0]
|
|
@@ -0,0 +1,433 @@
|
|
|
1
|
+
# SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
# Standard
|
|
3
|
+
from typing import TYPE_CHECKING, Literal, Optional, Tuple
|
|
4
|
+
import hashlib
|
|
5
|
+
import os
|
|
6
|
+
import string
|
|
7
|
+
import threading
|
|
8
|
+
|
|
9
|
+
if TYPE_CHECKING:
|
|
10
|
+
from vllm.config import ModelConfig, VllmConfig
|
|
11
|
+
from vllm.multimodal.inputs import PlaceholderRange
|
|
12
|
+
from vllm.v1.request import Request
|
|
13
|
+
|
|
14
|
+
# Third Party
|
|
15
|
+
import torch
|
|
16
|
+
|
|
17
|
+
# First Party
|
|
18
|
+
from lmcache.logging import init_logger
|
|
19
|
+
from lmcache.v1.config import LMCacheEngineConfig
|
|
20
|
+
from lmcache.v1.config_base import apply_remote_configs, fetch_remote_config
|
|
21
|
+
|
|
22
|
+
if TYPE_CHECKING:
|
|
23
|
+
# First Party
|
|
24
|
+
from lmcache.v1.gpu_connector.utils import LayoutHints
|
|
25
|
+
|
|
26
|
+
logger = init_logger(__name__)
|
|
27
|
+
ENGINE_NAME = "vllm-instance"
|
|
28
|
+
|
|
29
|
+
# Thread-safe singleton storage
|
|
30
|
+
_config_instance: Optional[LMCacheEngineConfig] = None
|
|
31
|
+
_config_lock = threading.Lock()
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def is_false(value: str) -> bool:
|
|
35
|
+
"""Check if the given string value is equivalent to 'false'."""
|
|
36
|
+
return value.lower() in ("false", "0", "no", "n", "off")
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def vllm_layout_hints() -> "LayoutHints":
|
|
40
|
+
"""Build layout_hints dict by querying vLLM at runtime."""
|
|
41
|
+
hints: dict[str, str] = {}
|
|
42
|
+
kv_layout = try_get_vllm_kv_cache_layout()
|
|
43
|
+
if kv_layout is not None:
|
|
44
|
+
hints["kv_layout"] = kv_layout
|
|
45
|
+
return hints # type: ignore[return-value]
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def try_get_vllm_kv_cache_layout() -> Literal["NHD", "HND"] | None:
|
|
49
|
+
"""Try to query the KV cache layout from vLLM at runtime.
|
|
50
|
+
|
|
51
|
+
Returns ``"NHD"`` or ``"HND"`` if vLLM is available and the layout
|
|
52
|
+
has been configured, otherwise ``None``.
|
|
53
|
+
|
|
54
|
+
Please only call this where vllm is available (i.e. not in the MP server)
|
|
55
|
+
We will print an error if we try to get vllm kv layout where vllm
|
|
56
|
+
is not available.
|
|
57
|
+
"""
|
|
58
|
+
|
|
59
|
+
# Third Party
|
|
60
|
+
try:
|
|
61
|
+
# Third Party
|
|
62
|
+
from vllm.v1.attention.backends.utils import ( # type: ignore[import-untyped]
|
|
63
|
+
get_kv_cache_layout,
|
|
64
|
+
)
|
|
65
|
+
|
|
66
|
+
return get_kv_cache_layout()
|
|
67
|
+
except Exception:
|
|
68
|
+
logger.error(
|
|
69
|
+
"vLLM is not available but tried to query kv cache "
|
|
70
|
+
"layout information, cannot get KV cache layout"
|
|
71
|
+
)
|
|
72
|
+
return None
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def lmcache_get_or_create_config() -> LMCacheEngineConfig:
|
|
76
|
+
"""Get the LMCache configuration from the environment variable
|
|
77
|
+
`LMCACHE_CONFIG_FILE`. If the environment variable is not set, this
|
|
78
|
+
function will return the default configuration.
|
|
79
|
+
|
|
80
|
+
This function is thread-safe and implements singleton pattern,
|
|
81
|
+
ensuring the configuration is loaded only once.
|
|
82
|
+
|
|
83
|
+
After loading the configuration, if 'remote_config_url' is configured,
|
|
84
|
+
this function will attempt to fetch additional configuration from the
|
|
85
|
+
remote config service. The current config and LMCACHE environment
|
|
86
|
+
variables will be sent to the service, along with 'appid' if set.
|
|
87
|
+
"""
|
|
88
|
+
global _config_instance
|
|
89
|
+
|
|
90
|
+
# Double-checked locking for thread-safe singleton
|
|
91
|
+
if _config_instance is None:
|
|
92
|
+
with _config_lock:
|
|
93
|
+
if _config_instance is None: # Check again within lock
|
|
94
|
+
if "LMCACHE_CONFIG_FILE" not in os.environ:
|
|
95
|
+
logger.warning(
|
|
96
|
+
"No LMCache configuration file is set. Trying to read"
|
|
97
|
+
" configurations from the environment variables."
|
|
98
|
+
)
|
|
99
|
+
logger.warning(
|
|
100
|
+
"You can set the configuration file through "
|
|
101
|
+
"the environment variable: LMCACHE_CONFIG_FILE"
|
|
102
|
+
)
|
|
103
|
+
_config_instance = LMCacheEngineConfig.from_env()
|
|
104
|
+
else:
|
|
105
|
+
config_file = os.environ["LMCACHE_CONFIG_FILE"]
|
|
106
|
+
logger.info(f"Loading LMCache config file {config_file}")
|
|
107
|
+
_config_instance = LMCacheEngineConfig.from_file(config_file)
|
|
108
|
+
# Update config from environment variables
|
|
109
|
+
_config_instance.update_config_from_env()
|
|
110
|
+
|
|
111
|
+
# Fetch and apply remote configuration if configured
|
|
112
|
+
remote_config_url = _config_instance.remote_config_url
|
|
113
|
+
if remote_config_url:
|
|
114
|
+
logger.info(
|
|
115
|
+
"Fetching remote configuration from %s", remote_config_url
|
|
116
|
+
)
|
|
117
|
+
app_id = _config_instance.app_id
|
|
118
|
+
remote_response = fetch_remote_config(
|
|
119
|
+
remote_config_url, app_id, _config_instance
|
|
120
|
+
)
|
|
121
|
+
if remote_response:
|
|
122
|
+
_config_instance = apply_remote_configs(
|
|
123
|
+
_config_instance, remote_response
|
|
124
|
+
)
|
|
125
|
+
else:
|
|
126
|
+
logger.warning(
|
|
127
|
+
"Failed to fetch remote configuration from %s. "
|
|
128
|
+
"Using local configuration only.",
|
|
129
|
+
remote_config_url,
|
|
130
|
+
)
|
|
131
|
+
return _config_instance
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
def hex_hash_to_int16(s: str) -> int:
|
|
135
|
+
"""
|
|
136
|
+
Convert a hash identifier into a 16-bit integer.
|
|
137
|
+
|
|
138
|
+
Historically, LMCache expected multimodal identifiers to be hex strings.
|
|
139
|
+
In practice (e.g., OpenAI-style multimodal requests), identifiers may be
|
|
140
|
+
arbitrary strings like `chatcmpl-...-image-0`. This function therefore:
|
|
141
|
+
- Parses hex strings (optionally prefixed with `0x`) as before, or
|
|
142
|
+
- Falls back to a stable string hash (SHA-256) when the input is not hex.
|
|
143
|
+
"""
|
|
144
|
+
# Be defensive: vLLM may pass non-string identifiers.
|
|
145
|
+
s = "" if s is None else str(s)
|
|
146
|
+
s_stripped = s.strip()
|
|
147
|
+
|
|
148
|
+
# Fast-path: pure hex (optionally 0x-prefixed).
|
|
149
|
+
hex_part = s_stripped[2:] if s_stripped.lower().startswith("0x") else s_stripped
|
|
150
|
+
if hex_part and all(c in string.hexdigits for c in hex_part):
|
|
151
|
+
try:
|
|
152
|
+
return int(hex_part, 16) & 0xFFFF
|
|
153
|
+
except ValueError:
|
|
154
|
+
# Extremely unlikely (e.g., oversized/odd formatting); fall back to hashing.
|
|
155
|
+
pass
|
|
156
|
+
|
|
157
|
+
# Fallback: stable 16-bit value derived from the full identifier string.
|
|
158
|
+
digest = hashlib.sha256(s_stripped.encode("utf-8")).digest()
|
|
159
|
+
return int.from_bytes(digest[:2], byteorder="big", signed=False)
|
|
160
|
+
|
|
161
|
+
|
|
162
|
+
def apply_mm_hashes_to_token_ids(
|
|
163
|
+
token_ids: torch.Tensor,
|
|
164
|
+
mm_hashes: list[str],
|
|
165
|
+
mm_positions: list["PlaceholderRange"],
|
|
166
|
+
) -> torch.Tensor:
|
|
167
|
+
"""
|
|
168
|
+
Overwrite token_ids in-place for multimodal placeholders using
|
|
169
|
+
efficient slice assignments.
|
|
170
|
+
"""
|
|
171
|
+
n = token_ids.size(0)
|
|
172
|
+
for hash_str, placeholder in zip(mm_hashes, mm_positions, strict=False):
|
|
173
|
+
start, length = placeholder.offset, placeholder.length
|
|
174
|
+
if start >= n:
|
|
175
|
+
continue
|
|
176
|
+
end = min(start + length, n)
|
|
177
|
+
token_ids[start:end] = hex_hash_to_int16(hash_str)
|
|
178
|
+
return token_ids
|
|
179
|
+
|
|
180
|
+
|
|
181
|
+
def mla_enabled(model_config: "ModelConfig") -> bool:
|
|
182
|
+
return (
|
|
183
|
+
hasattr(model_config, "use_mla")
|
|
184
|
+
and isinstance(model_config.use_mla, bool)
|
|
185
|
+
and model_config.use_mla
|
|
186
|
+
)
|
|
187
|
+
|
|
188
|
+
|
|
189
|
+
def create_lmcache_metadata(
|
|
190
|
+
vllm_config=None,
|
|
191
|
+
model_config=None,
|
|
192
|
+
parallel_config=None,
|
|
193
|
+
cache_config=None,
|
|
194
|
+
role=None,
|
|
195
|
+
):
|
|
196
|
+
"""
|
|
197
|
+
Create LMCacheMetadata from vLLM configuration.
|
|
198
|
+
|
|
199
|
+
This function extracts common metadata creation logic that was duplicated
|
|
200
|
+
across multiple files.
|
|
201
|
+
|
|
202
|
+
Args:
|
|
203
|
+
vllm_config: vLLM configuration object containing model, parallel, and
|
|
204
|
+
cache configs (alternative to individual config parameters)
|
|
205
|
+
model_config: Model configuration (alternative to vllm_config)
|
|
206
|
+
parallel_config: Parallel configuration (alternative to vllm_config)
|
|
207
|
+
cache_config: Cache configuration (alternative to vllm_config)
|
|
208
|
+
|
|
209
|
+
Returns:
|
|
210
|
+
tuple: (LMCacheMetadata, LMCacheEngineConfig)
|
|
211
|
+
"""
|
|
212
|
+
# Third Party
|
|
213
|
+
# Try to import from old location before merged https://github.com/vllm-project/vllm/pull/26908
|
|
214
|
+
try:
|
|
215
|
+
# Third Party
|
|
216
|
+
from vllm.utils.torch_utils import get_kv_cache_torch_dtype
|
|
217
|
+
except ImportError:
|
|
218
|
+
# Third Party
|
|
219
|
+
from vllm.utils import get_kv_cache_torch_dtype
|
|
220
|
+
# First Party
|
|
221
|
+
from lmcache.v1.metadata import LMCacheMetadata
|
|
222
|
+
|
|
223
|
+
config = lmcache_get_or_create_config()
|
|
224
|
+
# Support both vllm_config object and individual config parameters
|
|
225
|
+
if vllm_config is not None:
|
|
226
|
+
model_cfg = vllm_config.model_config
|
|
227
|
+
parallel_cfg = vllm_config.parallel_config
|
|
228
|
+
cache_cfg = vllm_config.cache_config
|
|
229
|
+
else:
|
|
230
|
+
model_cfg = model_config
|
|
231
|
+
parallel_cfg = parallel_config
|
|
232
|
+
cache_cfg = cache_config
|
|
233
|
+
|
|
234
|
+
# Get KV cache dtype
|
|
235
|
+
kv_dtype = get_kv_cache_torch_dtype(cache_cfg.cache_dtype, model_cfg.dtype)
|
|
236
|
+
|
|
237
|
+
# Check if MLA is enabled
|
|
238
|
+
use_mla = mla_enabled(model_cfg)
|
|
239
|
+
|
|
240
|
+
# Construct KV shape (for memory pool)
|
|
241
|
+
num_layer = model_cfg.get_num_layers(parallel_cfg)
|
|
242
|
+
chunk_size = config.chunk_size
|
|
243
|
+
num_kv_head = model_cfg.get_num_kv_heads(parallel_cfg)
|
|
244
|
+
head_size = model_cfg.get_head_size()
|
|
245
|
+
kv_shape = (num_layer, 1 if use_mla else 2, chunk_size, num_kv_head, head_size)
|
|
246
|
+
|
|
247
|
+
# Extract engine_id and kv_connector_extra_config from vllm_config if available
|
|
248
|
+
engine_id = None
|
|
249
|
+
kv_connector_extra_config = None
|
|
250
|
+
if vllm_config is not None and hasattr(vllm_config, "kv_transfer_config"):
|
|
251
|
+
kv_transfer_config = vllm_config.kv_transfer_config
|
|
252
|
+
if kv_transfer_config is not None:
|
|
253
|
+
engine_id = getattr(kv_transfer_config, "engine_id", None)
|
|
254
|
+
kv_connector_extra_config = getattr(
|
|
255
|
+
kv_transfer_config, "kv_connector_extra_config", None
|
|
256
|
+
)
|
|
257
|
+
|
|
258
|
+
# Create metadata
|
|
259
|
+
metadata = LMCacheMetadata(
|
|
260
|
+
model_name=model_cfg.model,
|
|
261
|
+
world_size=parallel_cfg.world_size,
|
|
262
|
+
local_world_size=parallel_cfg.world_size,
|
|
263
|
+
worker_id=parallel_cfg.rank,
|
|
264
|
+
local_worker_id=parallel_cfg.rank,
|
|
265
|
+
kv_dtype=kv_dtype,
|
|
266
|
+
kv_shape=kv_shape,
|
|
267
|
+
use_mla=use_mla,
|
|
268
|
+
role=role,
|
|
269
|
+
served_model_name=model_cfg.served_model_name,
|
|
270
|
+
engine_id=engine_id,
|
|
271
|
+
kv_connector_extra_config=kv_connector_extra_config,
|
|
272
|
+
)
|
|
273
|
+
|
|
274
|
+
return metadata, config
|
|
275
|
+
|
|
276
|
+
|
|
277
|
+
def extract_mm_features(
|
|
278
|
+
request: "Request", modify: bool = False
|
|
279
|
+
) -> Tuple[list[str], list["PlaceholderRange"]]:
|
|
280
|
+
"""
|
|
281
|
+
Normalize multimodal information from a Request into parallel lists.
|
|
282
|
+
|
|
283
|
+
This helper reads either:
|
|
284
|
+
1) `request.mm_features` (objects each exposing `.identifier` and
|
|
285
|
+
`.mm_position`), or
|
|
286
|
+
2) legacy fields `request.mm_hashes` and `request.mm_positions`.
|
|
287
|
+
|
|
288
|
+
It returns two equally sized lists: the multimodal hash identifiers and their
|
|
289
|
+
corresponding positions. If the request contains no multimodal info, it returns
|
|
290
|
+
`([], [])`.
|
|
291
|
+
|
|
292
|
+
Args:
|
|
293
|
+
request (Request): The source object.
|
|
294
|
+
modify (bool):
|
|
295
|
+
Controls copy semantics for the legacy-path return values.
|
|
296
|
+
- If True and legacy fields are used, shallow-copies are returned so
|
|
297
|
+
the caller can mutate the lists without affecting `request`.
|
|
298
|
+
- If False, the original legacy sequences are returned as-is
|
|
299
|
+
(zero-copy); treat them as read-only.
|
|
300
|
+
|
|
301
|
+
Returns:
|
|
302
|
+
Tuple[list[str], list[PlaceholderRange]]: (`mm_hashes`, `mm_positions`).
|
|
303
|
+
May be `([], [])` when no multimodal data is present.
|
|
304
|
+
"""
|
|
305
|
+
if getattr(request, "mm_features", None):
|
|
306
|
+
mm_hashes, mm_positions = zip(
|
|
307
|
+
*((f.identifier, f.mm_position) for f in request.mm_features), strict=False
|
|
308
|
+
)
|
|
309
|
+
return (list(mm_hashes), list(mm_positions))
|
|
310
|
+
elif getattr(request, "mm_hashes", None):
|
|
311
|
+
if modify:
|
|
312
|
+
return (request.mm_hashes.copy(), request.mm_positions.copy())
|
|
313
|
+
else:
|
|
314
|
+
return (request.mm_hashes, request.mm_positions)
|
|
315
|
+
else:
|
|
316
|
+
return ([], [])
|
|
317
|
+
|
|
318
|
+
|
|
319
|
+
def get_size_bytes(shapes: list[torch.Size], kv_dtypes: list[torch.dtype]):
|
|
320
|
+
"""
|
|
321
|
+
Calculate the size in bytes with the given shapes and dtypes.
|
|
322
|
+
"""
|
|
323
|
+
assert len(shapes) == len(kv_dtypes), (
|
|
324
|
+
f"shapes and dtypes must have the same length, "
|
|
325
|
+
f"but got {len(shapes)} and {len(kv_dtypes)}"
|
|
326
|
+
)
|
|
327
|
+
return sum(
|
|
328
|
+
shape.numel() * kv_dtype.itemsize
|
|
329
|
+
for shape, kv_dtype in zip(shapes, kv_dtypes, strict=True)
|
|
330
|
+
)
|
|
331
|
+
|
|
332
|
+
|
|
333
|
+
def get_vllm_torch_dev():
|
|
334
|
+
"""
|
|
335
|
+
Returns the torch device and device name for the vLLM engine.
|
|
336
|
+
e.g. (torch.cuda, "cuda") or (torch.xpu, "xpu")
|
|
337
|
+
"""
|
|
338
|
+
# Third Party
|
|
339
|
+
from vllm.platforms import current_platform
|
|
340
|
+
|
|
341
|
+
if current_platform.is_cuda_alike():
|
|
342
|
+
logger.info("CUDA device is available. Using CUDA for LMCache engine.")
|
|
343
|
+
torch_dev = torch.cuda
|
|
344
|
+
dev_name = "cuda"
|
|
345
|
+
elif current_platform.is_xpu():
|
|
346
|
+
logger.info("XPU device is available. Using XPU for LMCache engine.")
|
|
347
|
+
torch_dev = torch.xpu
|
|
348
|
+
dev_name = "xpu"
|
|
349
|
+
elif hasattr(torch, "hpu") and torch.hpu.is_available():
|
|
350
|
+
logger.info("HPU device is available. Using HPU for LMCache engine.")
|
|
351
|
+
torch_dev = torch.hpu
|
|
352
|
+
dev_name = "hpu"
|
|
353
|
+
else:
|
|
354
|
+
raise RuntimeError("Unsupported device platform for LMCache engine.")
|
|
355
|
+
return torch_dev, dev_name
|
|
356
|
+
|
|
357
|
+
|
|
358
|
+
def calculate_local_rank_and_world_size(vllm_config: "VllmConfig") -> Tuple[int, int]:
|
|
359
|
+
"""
|
|
360
|
+
Calculate the local worker id and local world size.
|
|
361
|
+
|
|
362
|
+
Current assumption (TODO: add custom logic in the future):
|
|
363
|
+
- Tensor Parallel is intra-node
|
|
364
|
+
- Pipeline Parallel is inter-node
|
|
365
|
+
|
|
366
|
+
Returns:
|
|
367
|
+
Tuple[int, int]: (local_worker_id, local_world_size)
|
|
368
|
+
"""
|
|
369
|
+
parallel_config = vllm_config.parallel_config
|
|
370
|
+
global_rank = parallel_config.rank
|
|
371
|
+
global_world_size = parallel_config.world_size
|
|
372
|
+
torch_dev, dev_name = get_vllm_torch_dev()
|
|
373
|
+
num_gpus = torch_dev.device_count()
|
|
374
|
+
if global_world_size <= num_gpus:
|
|
375
|
+
# single node case
|
|
376
|
+
return parallel_config.rank, parallel_config.world_size
|
|
377
|
+
else:
|
|
378
|
+
tp_size = parallel_config.tensor_parallel_size
|
|
379
|
+
pp_size = parallel_config.pipeline_parallel_size
|
|
380
|
+
local_world_size = global_world_size // pp_size
|
|
381
|
+
assert local_world_size == tp_size, (
|
|
382
|
+
"LMCache is operating under the assumption that the "
|
|
383
|
+
"local world size is equal to the tensor parallel size "
|
|
384
|
+
"in multi-node deployment."
|
|
385
|
+
)
|
|
386
|
+
local_worker_id = global_rank % local_world_size
|
|
387
|
+
return local_worker_id, local_world_size
|
|
388
|
+
|
|
389
|
+
|
|
390
|
+
def validate_mla_config(config: LMCacheEngineConfig, use_mla: bool) -> None:
|
|
391
|
+
"""Validate MLA-related configuration."""
|
|
392
|
+
if use_mla and (config.remote_serde != "naive" and config.remote_serde is not None):
|
|
393
|
+
raise ValueError("MLA only works with naive serde mode..")
|
|
394
|
+
|
|
395
|
+
if use_mla and config.use_layerwise and config.enable_blending:
|
|
396
|
+
raise ValueError(
|
|
397
|
+
"We haven't supported MLA with Cacheblend yet. Please disable blending."
|
|
398
|
+
)
|
|
399
|
+
|
|
400
|
+
|
|
401
|
+
def calculate_draft_layers(vllm_config: "VllmConfig") -> int:
|
|
402
|
+
"""Calculate the number of draft layers for speculative decoding."""
|
|
403
|
+
assert vllm_config is not None, "vllm_config required for vLLM mode"
|
|
404
|
+
|
|
405
|
+
num_draft_layers = 0
|
|
406
|
+
model_config = vllm_config.model_config
|
|
407
|
+
|
|
408
|
+
if vllm_config.speculative_config is not None:
|
|
409
|
+
logger.info(
|
|
410
|
+
"vllm_config.speculative_config: %s", vllm_config.speculative_config
|
|
411
|
+
)
|
|
412
|
+
if vllm_config.speculative_config.method == "deepseek_mtp":
|
|
413
|
+
num_draft_layers = getattr(
|
|
414
|
+
model_config.hf_config, "num_nextn_predict_layers", 0
|
|
415
|
+
)
|
|
416
|
+
elif vllm_config.speculative_config.use_eagle():
|
|
417
|
+
try:
|
|
418
|
+
draft_model_config = vllm_config.speculative_config.draft_model_config
|
|
419
|
+
num_draft_layers = draft_model_config.get_num_layers(
|
|
420
|
+
vllm_config.parallel_config
|
|
421
|
+
)
|
|
422
|
+
logger.info("EAGLE detected %d extra layer(s)", num_draft_layers)
|
|
423
|
+
except Exception:
|
|
424
|
+
logger.info(
|
|
425
|
+
"EAGLE detected, but failed to get the number of extra layers"
|
|
426
|
+
"falling back to 1"
|
|
427
|
+
)
|
|
428
|
+
num_draft_layers = 1
|
|
429
|
+
return num_draft_layers
|
|
430
|
+
|
|
431
|
+
|
|
432
|
+
def is_dp_rank0(vllm_config: "VllmConfig") -> bool:
|
|
433
|
+
return vllm_config.parallel_config.data_parallel_rank_local == 0
|