lmcache-cli 0.4.5.dev0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- lmcache/__init__.py +84 -0
- lmcache/_version.py +24 -0
- lmcache/cli/__init__.py +1 -0
- lmcache/cli/commands/__init__.py +34 -0
- lmcache/cli/commands/base.py +157 -0
- lmcache/cli/commands/bench/__init__.py +557 -0
- lmcache/cli/commands/bench/engine_bench/__init__.py +1 -0
- lmcache/cli/commands/bench/engine_bench/config.py +245 -0
- lmcache/cli/commands/bench/engine_bench/interactive/__init__.py +274 -0
- lmcache/cli/commands/bench/engine_bench/interactive/config.json +10 -0
- lmcache/cli/commands/bench/engine_bench/interactive/schema.py +352 -0
- lmcache/cli/commands/bench/engine_bench/interactive/state.py +327 -0
- lmcache/cli/commands/bench/engine_bench/interactive/terminal.py +291 -0
- lmcache/cli/commands/bench/engine_bench/progress.py +145 -0
- lmcache/cli/commands/bench/engine_bench/request_sender.py +232 -0
- lmcache/cli/commands/bench/engine_bench/stats.py +275 -0
- lmcache/cli/commands/bench/engine_bench/workloads/__init__.py +153 -0
- lmcache/cli/commands/bench/engine_bench/workloads/base.py +122 -0
- lmcache/cli/commands/bench/engine_bench/workloads/long_doc_permutator.py +435 -0
- lmcache/cli/commands/bench/engine_bench/workloads/long_doc_qa.py +281 -0
- lmcache/cli/commands/bench/engine_bench/workloads/multi_round_chat.py +337 -0
- lmcache/cli/commands/bench/engine_bench/workloads/random_prefill.py +178 -0
- lmcache/cli/commands/describe.py +310 -0
- lmcache/cli/commands/kvcache.py +133 -0
- lmcache/cli/commands/mock.py +75 -0
- lmcache/cli/commands/ping.py +113 -0
- lmcache/cli/commands/query/__init__.py +155 -0
- lmcache/cli/commands/query/prompt.py +134 -0
- lmcache/cli/commands/query/request.py +357 -0
- lmcache/cli/commands/server.py +99 -0
- lmcache/cli/commands/tool/__init__.py +63 -0
- lmcache/cli/commands/tool/cache_simulator.py +113 -0
- lmcache/cli/commands/trace/__init__.py +505 -0
- lmcache/cli/commands/trace/dispatch.py +249 -0
- lmcache/cli/commands/trace/driver.py +372 -0
- lmcache/cli/commands/trace/stats.py +289 -0
- lmcache/cli/documents/lmcache.txt +11 -0
- lmcache/cli/main.py +42 -0
- lmcache/cli/metrics/__init__.py +29 -0
- lmcache/cli/metrics/formatter.py +171 -0
- lmcache/cli/metrics/handler.py +94 -0
- lmcache/cli/metrics/metrics.py +161 -0
- lmcache/cli/metrics/section.py +77 -0
- lmcache/connections.py +173 -0
- lmcache/integration/__init__.py +2 -0
- lmcache/integration/base_service_factory.py +165 -0
- lmcache/integration/request_telemetry/__init__.py +1 -0
- lmcache/integration/request_telemetry/base.py +51 -0
- lmcache/integration/request_telemetry/factory.py +113 -0
- lmcache/integration/request_telemetry/fastapi.py +109 -0
- lmcache/integration/request_telemetry/noop.py +35 -0
- lmcache/integration/sglang/__init__.py +2 -0
- lmcache/integration/sglang/sglang_adapter.py +326 -0
- lmcache/integration/sglang/utils.py +39 -0
- lmcache/integration/vllm/__init__.py +1 -0
- lmcache/integration/vllm/lmcache_connector_v1.py +213 -0
- lmcache/integration/vllm/lmcache_connector_v1_085.py +150 -0
- lmcache/integration/vllm/lmcache_mp_connector_0180.py +1072 -0
- lmcache/integration/vllm/tests/test_mm_hash_utils.py +112 -0
- lmcache/integration/vllm/utils.py +433 -0
- lmcache/integration/vllm/vllm_multi_process_adapter.py +1090 -0
- lmcache/integration/vllm/vllm_service_factory.py +339 -0
- lmcache/integration/vllm/vllm_v1_adapter.py +1713 -0
- lmcache/logging.py +107 -0
- lmcache/native_storage_ops.pyi +230 -0
- lmcache/non_cuda_equivalents.py +1424 -0
- lmcache/observability.py +1958 -0
- lmcache/storage_backend/serde/__init__.py +1 -0
- lmcache/storage_backend/serde/cachegen_basics.py +210 -0
- lmcache/storage_backend/serde/cachegen_decoder.py +207 -0
- lmcache/storage_backend/serde/cachegen_encoder.py +394 -0
- lmcache/storage_backend/serde/serde.py +75 -0
- lmcache/tools/__init__.py +1 -0
- lmcache/tools/cache_simulator/README.md +392 -0
- lmcache/tools/cache_simulator/__init__.py +1 -0
- lmcache/tools/cache_simulator/docs/simulate_example.png +0 -0
- lmcache/tools/cache_simulator/docs/sweep_example.png +0 -0
- lmcache/tools/cache_simulator/gen_bench_dataset.py +360 -0
- lmcache/tools/cache_simulator/lru_cache.py +124 -0
- lmcache/tools/cache_simulator/plot_hit_rate.py +231 -0
- lmcache/tools/cache_simulator/simulator.py +795 -0
- lmcache/tools/controller_benchmark/README.md +161 -0
- lmcache/tools/controller_benchmark/__init__.py +1 -0
- lmcache/tools/controller_benchmark/__main__.py +331 -0
- lmcache/tools/controller_benchmark/benchmark.py +660 -0
- lmcache/tools/controller_benchmark/config.py +44 -0
- lmcache/tools/controller_benchmark/constants.py +10 -0
- lmcache/tools/controller_benchmark/handlers/__init__.py +46 -0
- lmcache/tools/controller_benchmark/handlers/admit.py +52 -0
- lmcache/tools/controller_benchmark/handlers/base.py +47 -0
- lmcache/tools/controller_benchmark/handlers/deregister.py +49 -0
- lmcache/tools/controller_benchmark/handlers/evict.py +52 -0
- lmcache/tools/controller_benchmark/handlers/heartbeat.py +56 -0
- lmcache/tools/controller_benchmark/handlers/p2p_lookup.py +47 -0
- lmcache/tools/controller_benchmark/handlers/register.py +56 -0
- lmcache/tools/mp_status_viewer/__init__.py +1 -0
- lmcache/tools/mp_status_viewer/__main__.py +95 -0
- lmcache/usage_context.py +417 -0
- lmcache/utils.py +665 -0
- lmcache/v1/__init__.py +2 -0
- lmcache/v1/api_server/__init__.py +2 -0
- lmcache/v1/api_server/__main__.py +537 -0
- lmcache/v1/basic_check.py +112 -0
- lmcache/v1/cache_controller/__init__.py +9 -0
- lmcache/v1/cache_controller/commands/__init__.py +15 -0
- lmcache/v1/cache_controller/commands/base.py +35 -0
- lmcache/v1/cache_controller/commands/full_sync.py +49 -0
- lmcache/v1/cache_controller/config.py +176 -0
- lmcache/v1/cache_controller/controller_manager.py +535 -0
- lmcache/v1/cache_controller/controllers/__init__.py +11 -0
- lmcache/v1/cache_controller/controllers/full_sync_tracker.py +473 -0
- lmcache/v1/cache_controller/controllers/kv_controller.py +439 -0
- lmcache/v1/cache_controller/controllers/registration_controller.py +282 -0
- lmcache/v1/cache_controller/executor.py +463 -0
- lmcache/v1/cache_controller/frontend/static/css/style.css +201 -0
- lmcache/v1/cache_controller/frontend/static/img/logo.png +0 -0
- lmcache/v1/cache_controller/frontend/static/index.html +234 -0
- lmcache/v1/cache_controller/frontend/static/js/controller_app.js +660 -0
- lmcache/v1/cache_controller/full_sync_sender.py +475 -0
- lmcache/v1/cache_controller/locks.py +149 -0
- lmcache/v1/cache_controller/message.py +828 -0
- lmcache/v1/cache_controller/observability.py +208 -0
- lmcache/v1/cache_controller/utils.py +679 -0
- lmcache/v1/cache_controller/worker.py +665 -0
- lmcache/v1/cache_engine.py +2058 -0
- lmcache/v1/cache_interface.py +19 -0
- lmcache/v1/check/__init__.py +74 -0
- lmcache/v1/check/check_mode_gen.py +86 -0
- lmcache/v1/check/check_mode_test_l2_adapter.py +284 -0
- lmcache/v1/check/check_mode_test_remote.py +155 -0
- lmcache/v1/check/check_mode_test_storage_manager.py +142 -0
- lmcache/v1/check/utils.py +571 -0
- lmcache/v1/compute/__init__.py +2 -0
- lmcache/v1/compute/attention/__init__.py +0 -0
- lmcache/v1/compute/attention/abstract.py +39 -0
- lmcache/v1/compute/attention/flash_attn.py +129 -0
- lmcache/v1/compute/attention/flash_infer_sparse.py +284 -0
- lmcache/v1/compute/attention/metadata.py +85 -0
- lmcache/v1/compute/attention/utils.py +14 -0
- lmcache/v1/compute/blend/__init__.py +7 -0
- lmcache/v1/compute/blend/blender.py +168 -0
- lmcache/v1/compute/blend/metadata.py +34 -0
- lmcache/v1/compute/blend/utils.py +63 -0
- lmcache/v1/compute/models/__init__.py +0 -0
- lmcache/v1/compute/models/base.py +141 -0
- lmcache/v1/compute/models/llama.py +9 -0
- lmcache/v1/compute/models/qwen3.py +24 -0
- lmcache/v1/compute/models/utils.py +68 -0
- lmcache/v1/compute/positional_encoding.py +199 -0
- lmcache/v1/config.py +848 -0
- lmcache/v1/config_base.py +848 -0
- lmcache/v1/distributed/api.py +248 -0
- lmcache/v1/distributed/config.py +321 -0
- lmcache/v1/distributed/error.py +64 -0
- lmcache/v1/distributed/eviction.py +192 -0
- lmcache/v1/distributed/eviction_policy/__init__.py +21 -0
- lmcache/v1/distributed/eviction_policy/factory.py +27 -0
- lmcache/v1/distributed/eviction_policy/lru.py +244 -0
- lmcache/v1/distributed/eviction_policy/noop.py +50 -0
- lmcache/v1/distributed/internal_api.py +170 -0
- lmcache/v1/distributed/l1_manager.py +835 -0
- lmcache/v1/distributed/l2_adapters/__init__.py +67 -0
- lmcache/v1/distributed/l2_adapters/base.py +360 -0
- lmcache/v1/distributed/l2_adapters/config.py +385 -0
- lmcache/v1/distributed/l2_adapters/factory.py +205 -0
- lmcache/v1/distributed/l2_adapters/fs_l2_adapter.py +747 -0
- lmcache/v1/distributed/l2_adapters/fs_native_l2_adapter.py +167 -0
- lmcache/v1/distributed/l2_adapters/mock_l2_adapter.py +516 -0
- lmcache/v1/distributed/l2_adapters/mooncake_store_l2_adapter.py +135 -0
- lmcache/v1/distributed/l2_adapters/native_connector_l2_adapter.py +468 -0
- lmcache/v1/distributed/l2_adapters/native_plugin_l2_adapter.py +199 -0
- lmcache/v1/distributed/l2_adapters/nixl_store_dynamic_l2_adapter.py +831 -0
- lmcache/v1/distributed/l2_adapters/nixl_store_l2_adapter.py +983 -0
- lmcache/v1/distributed/l2_adapters/plugin_l2_adapter.py +210 -0
- lmcache/v1/distributed/l2_adapters/resp_l2_adapter.py +176 -0
- lmcache/v1/distributed/memory_manager.py +179 -0
- lmcache/v1/distributed/storage_controller.py +39 -0
- lmcache/v1/distributed/storage_controllers/__init__.py +43 -0
- lmcache/v1/distributed/storage_controllers/eviction_controller.py +242 -0
- lmcache/v1/distributed/storage_controllers/prefetch_controller.py +830 -0
- lmcache/v1/distributed/storage_controllers/prefetch_policy.py +193 -0
- lmcache/v1/distributed/storage_controllers/store_controller.py +452 -0
- lmcache/v1/distributed/storage_controllers/store_policy.py +213 -0
- lmcache/v1/distributed/storage_manager.py +532 -0
- lmcache/v1/event_manager.py +145 -0
- lmcache/v1/exceptions/__init__.py +16 -0
- lmcache/v1/gpu_connector/__init__.py +126 -0
- lmcache/v1/gpu_connector/gpu_connectors.py +1906 -0
- lmcache/v1/gpu_connector/gpu_ops.py +85 -0
- lmcache/v1/gpu_connector/hpu_connector.py +326 -0
- lmcache/v1/gpu_connector/mock_gpu_connector.py +67 -0
- lmcache/v1/gpu_connector/utils.py +890 -0
- lmcache/v1/gpu_connector/xpu_connectors.py +916 -0
- lmcache/v1/health_monitor/__init__.py +1 -0
- lmcache/v1/health_monitor/base.py +587 -0
- lmcache/v1/health_monitor/checks/__init__.py +1 -0
- lmcache/v1/health_monitor/checks/remote_backend_check.py +304 -0
- lmcache/v1/health_monitor/constants.py +36 -0
- lmcache/v1/internal_api_server/__init__.py +0 -0
- lmcache/v1/internal_api_server/api_registry.py +59 -0
- lmcache/v1/internal_api_server/api_server.py +120 -0
- lmcache/v1/internal_api_server/common/__init__.py +1 -0
- lmcache/v1/internal_api_server/common/env_api.py +22 -0
- lmcache/v1/internal_api_server/common/loglevel_api.py +57 -0
- lmcache/v1/internal_api_server/common/metrics_api.py +29 -0
- lmcache/v1/internal_api_server/common/periodic_thread_api.py +138 -0
- lmcache/v1/internal_api_server/common/run_script_api.py +73 -0
- lmcache/v1/internal_api_server/common/thread_api.py +63 -0
- lmcache/v1/internal_api_server/controller/__init__.py +1 -0
- lmcache/v1/internal_api_server/controller/key_stats_api.py +81 -0
- lmcache/v1/internal_api_server/controller/worker_info_api.py +136 -0
- lmcache/v1/internal_api_server/utils.py +43 -0
- lmcache/v1/internal_api_server/vllm/__init__.py +1 -0
- lmcache/v1/internal_api_server/vllm/backend_api.py +221 -0
- lmcache/v1/internal_api_server/vllm/bypass_api.py +204 -0
- lmcache/v1/internal_api_server/vllm/cache_api.py +895 -0
- lmcache/v1/internal_api_server/vllm/chunk_statistics_api.py +141 -0
- lmcache/v1/internal_api_server/vllm/conf_api.py +147 -0
- lmcache/v1/internal_api_server/vllm/freeze_api.py +172 -0
- lmcache/v1/internal_api_server/vllm/hot_cache_api.py +184 -0
- lmcache/v1/internal_api_server/vllm/inference_api.py +65 -0
- lmcache/v1/internal_api_server/vllm/load_fs_chunks_api.py +320 -0
- lmcache/v1/internal_api_server/vllm/lookup_api.py +145 -0
- lmcache/v1/internal_api_server/vllm/version_api.py +25 -0
- lmcache/v1/kv_layer_groups.py +267 -0
- lmcache/v1/lazy_memory_allocator.py +284 -0
- lmcache/v1/lookup_client/__init__.py +25 -0
- lmcache/v1/lookup_client/abstract_client.py +77 -0
- lmcache/v1/lookup_client/async_lookup_message.py +50 -0
- lmcache/v1/lookup_client/chunk_statistics_lookup_client.py +200 -0
- lmcache/v1/lookup_client/factory.py +251 -0
- lmcache/v1/lookup_client/hit_limit_lookup_client.py +86 -0
- lmcache/v1/lookup_client/lmcache_async_lookup_client.py +407 -0
- lmcache/v1/lookup_client/lmcache_lookup_client.py +285 -0
- lmcache/v1/lookup_client/lmcache_lookup_client_bypass.py +99 -0
- lmcache/v1/lookup_client/mooncake_lookup_client.py +87 -0
- lmcache/v1/lookup_client/record_strategies/__init__.py +77 -0
- lmcache/v1/lookup_client/record_strategies/base.py +327 -0
- lmcache/v1/lookup_client/record_strategies/file_hash.py +130 -0
- lmcache/v1/lookup_client/record_strategies/memory_bloom_filter.py +81 -0
- lmcache/v1/manager.py +539 -0
- lmcache/v1/memory_management.py +2619 -0
- lmcache/v1/metadata.py +114 -0
- lmcache/v1/mp_observability/AGENTS.override.md +21 -0
- lmcache/v1/mp_observability/README.md +204 -0
- lmcache/v1/mp_observability/config.py +340 -0
- lmcache/v1/mp_observability/event.py +100 -0
- lmcache/v1/mp_observability/event_bus.py +313 -0
- lmcache/v1/mp_observability/otel_init.py +145 -0
- lmcache/v1/mp_observability/subscribers/__init__.py +28 -0
- lmcache/v1/mp_observability/subscribers/logging/__init__.py +19 -0
- lmcache/v1/mp_observability/subscribers/logging/l1.py +56 -0
- lmcache/v1/mp_observability/subscribers/logging/l2.py +73 -0
- lmcache/v1/mp_observability/subscribers/logging/lookup_hash.py +209 -0
- lmcache/v1/mp_observability/subscribers/logging/mp_server.py +90 -0
- lmcache/v1/mp_observability/subscribers/logging/sm.py +59 -0
- lmcache/v1/mp_observability/subscribers/metrics/__init__.py +20 -0
- lmcache/v1/mp_observability/subscribers/metrics/l0_lifecycle.py +290 -0
- lmcache/v1/mp_observability/subscribers/metrics/l1.py +55 -0
- lmcache/v1/mp_observability/subscribers/metrics/l1_lifecycle.py +166 -0
- lmcache/v1/mp_observability/subscribers/metrics/l2.py +121 -0
- lmcache/v1/mp_observability/subscribers/metrics/sm.py +69 -0
- lmcache/v1/mp_observability/subscribers/tracing/__init__.py +12 -0
- lmcache/v1/mp_observability/subscribers/tracing/mp_server.py +333 -0
- lmcache/v1/mp_observability/subscribers/tracing/span_registry.py +148 -0
- lmcache/v1/mp_observability/trace/__init__.py +50 -0
- lmcache/v1/mp_observability/trace/codecs.py +255 -0
- lmcache/v1/mp_observability/trace/decorator.py +147 -0
- lmcache/v1/mp_observability/trace/format.py +132 -0
- lmcache/v1/mp_observability/trace/lifecycle.py +83 -0
- lmcache/v1/mp_observability/trace/reader.py +167 -0
- lmcache/v1/mp_observability/trace/recorder.py +300 -0
- lmcache/v1/multiprocess/__init__.py +0 -0
- lmcache/v1/multiprocess/affinity_pool.py +102 -0
- lmcache/v1/multiprocess/blend_server_v2.py +891 -0
- lmcache/v1/multiprocess/config.py +253 -0
- lmcache/v1/multiprocess/custom_types.py +281 -0
- lmcache/v1/multiprocess/futures.py +194 -0
- lmcache/v1/multiprocess/gpu_context.py +511 -0
- lmcache/v1/multiprocess/http_server.py +235 -0
- lmcache/v1/multiprocess/mp_runtime_plugin_launcher.py +130 -0
- lmcache/v1/multiprocess/mq.py +732 -0
- lmcache/v1/multiprocess/protocol.py +86 -0
- lmcache/v1/multiprocess/protocols/README.md +213 -0
- lmcache/v1/multiprocess/protocols/__init__.py +127 -0
- lmcache/v1/multiprocess/protocols/base.py +89 -0
- lmcache/v1/multiprocess/protocols/blend.py +109 -0
- lmcache/v1/multiprocess/protocols/blend_v2.py +57 -0
- lmcache/v1/multiprocess/protocols/controller.py +53 -0
- lmcache/v1/multiprocess/protocols/debug.py +34 -0
- lmcache/v1/multiprocess/protocols/engine.py +146 -0
- lmcache/v1/multiprocess/protocols/observability.py +39 -0
- lmcache/v1/multiprocess/server.py +1134 -0
- lmcache/v1/multiprocess/session.py +190 -0
- lmcache/v1/multiprocess/token_hasher.py +441 -0
- lmcache/v1/offload_server/__init__.py +17 -0
- lmcache/v1/offload_server/abstract_server.py +37 -0
- lmcache/v1/offload_server/message.py +30 -0
- lmcache/v1/offload_server/zmq_server.py +122 -0
- lmcache/v1/periodic_thread.py +579 -0
- lmcache/v1/pin_monitor.py +246 -0
- lmcache/v1/plugin/__init__.py +0 -0
- lmcache/v1/plugin/runtime_plugin_launcher.py +211 -0
- lmcache/v1/protocol.py +317 -0
- lmcache/v1/rpc/__init__.py +17 -0
- lmcache/v1/rpc/transport.py +105 -0
- lmcache/v1/rpc/zmq_transport.py +213 -0
- lmcache/v1/rpc_utils.py +165 -0
- lmcache/v1/server/__init__.py +2 -0
- lmcache/v1/server/__main__.py +170 -0
- lmcache/v1/server/storage_backend/__init__.py +21 -0
- lmcache/v1/server/storage_backend/abstract_backend.py +80 -0
- lmcache/v1/server/storage_backend/local_backend.py +75 -0
- lmcache/v1/server/utils.py +21 -0
- lmcache/v1/standalone/__init__.py +1 -0
- lmcache/v1/standalone/__main__.py +583 -0
- lmcache/v1/standalone/manager.py +80 -0
- lmcache/v1/standalone/standalone_service_factory.py +86 -0
- lmcache/v1/storage_backend/__init__.py +313 -0
- lmcache/v1/storage_backend/abstract_backend.py +445 -0
- lmcache/v1/storage_backend/audit_backend.py +233 -0
- lmcache/v1/storage_backend/batched_message_sender.py +222 -0
- lmcache/v1/storage_backend/cache_policy/__init__.py +45 -0
- lmcache/v1/storage_backend/cache_policy/base_policy.py +87 -0
- lmcache/v1/storage_backend/cache_policy/fifo.py +58 -0
- lmcache/v1/storage_backend/cache_policy/lfu.py +105 -0
- lmcache/v1/storage_backend/cache_policy/lru.py +81 -0
- lmcache/v1/storage_backend/cache_policy/mru.py +61 -0
- lmcache/v1/storage_backend/connector/__init__.py +443 -0
- lmcache/v1/storage_backend/connector/audit_adapter.py +77 -0
- lmcache/v1/storage_backend/connector/audit_connector.py +320 -0
- lmcache/v1/storage_backend/connector/base_connector.py +379 -0
- lmcache/v1/storage_backend/connector/blackhole_adapter.py +21 -0
- lmcache/v1/storage_backend/connector/blackhole_connector.py +37 -0
- lmcache/v1/storage_backend/connector/eic_adapter.py +31 -0
- lmcache/v1/storage_backend/connector/eic_connector.py +757 -0
- lmcache/v1/storage_backend/connector/external_adapter.py +79 -0
- lmcache/v1/storage_backend/connector/fs_adapter.py +51 -0
- lmcache/v1/storage_backend/connector/fs_connector.py +403 -0
- lmcache/v1/storage_backend/connector/infinistore_adapter.py +56 -0
- lmcache/v1/storage_backend/connector/infinistore_connector.py +177 -0
- lmcache/v1/storage_backend/connector/instrumented_connector.py +219 -0
- lmcache/v1/storage_backend/connector/lm_adapter.py +31 -0
- lmcache/v1/storage_backend/connector/lm_connector.py +176 -0
- lmcache/v1/storage_backend/connector/mock_adapter.py +57 -0
- lmcache/v1/storage_backend/connector/mock_connector.py +349 -0
- lmcache/v1/storage_backend/connector/mooncakestore_adapter.py +43 -0
- lmcache/v1/storage_backend/connector/mooncakestore_connector.py +614 -0
- lmcache/v1/storage_backend/connector/redis_adapter.py +181 -0
- lmcache/v1/storage_backend/connector/redis_connector.py +828 -0
- lmcache/v1/storage_backend/connector/s3_adapter.py +59 -0
- lmcache/v1/storage_backend/connector/s3_connector.py +699 -0
- lmcache/v1/storage_backend/connector/sagemaker_hyperpod_adapter.py +233 -0
- lmcache/v1/storage_backend/connector/sagemaker_hyperpod_connector.py +987 -0
- lmcache/v1/storage_backend/connector/valkey_adapter.py +114 -0
- lmcache/v1/storage_backend/connector/valkey_connector.py +627 -0
- lmcache/v1/storage_backend/gds_backend.py +1199 -0
- lmcache/v1/storage_backend/job_executor/__init__.py +0 -0
- lmcache/v1/storage_backend/job_executor/base_executor.py +34 -0
- lmcache/v1/storage_backend/job_executor/pq_executor.py +235 -0
- lmcache/v1/storage_backend/local_cpu_backend.py +810 -0
- lmcache/v1/storage_backend/local_disk_backend.py +656 -0
- lmcache/v1/storage_backend/maru_backend.py +734 -0
- lmcache/v1/storage_backend/naive_serde/__init__.py +50 -0
- lmcache/v1/storage_backend/naive_serde/cachegen_basics.py +133 -0
- lmcache/v1/storage_backend/naive_serde/cachegen_decoder.py +135 -0
- lmcache/v1/storage_backend/naive_serde/cachegen_encoder.py +83 -0
- lmcache/v1/storage_backend/naive_serde/kivi_serde.py +22 -0
- lmcache/v1/storage_backend/naive_serde/naive_serde.py +18 -0
- lmcache/v1/storage_backend/naive_serde/serde.py +37 -0
- lmcache/v1/storage_backend/native_clients/connector_client_base.py +165 -0
- lmcache/v1/storage_backend/native_clients/resp_client.py +35 -0
- lmcache/v1/storage_backend/nixl_storage_backend.py +1400 -0
- lmcache/v1/storage_backend/p2p_backend.py +788 -0
- lmcache/v1/storage_backend/path_sharder.py +117 -0
- lmcache/v1/storage_backend/pd_backend.py +646 -0
- lmcache/v1/storage_backend/plugins/dax_backend.py +1443 -0
- lmcache/v1/storage_backend/plugins/rust_raw_block_backend.py +1361 -0
- lmcache/v1/storage_backend/remote_backend.py +624 -0
- lmcache/v1/storage_backend/resp_client.py +227 -0
- lmcache/v1/storage_backend/storage_backend_listener.py +19 -0
- lmcache/v1/storage_backend/storage_manager.py +1352 -0
- lmcache/v1/system_detection.py +110 -0
- lmcache/v1/token_database.py +551 -0
- lmcache/v1/transfer_channel/__init__.py +83 -0
- lmcache/v1/transfer_channel/abstract.py +285 -0
- lmcache/v1/transfer_channel/mock_memory_channel.py +156 -0
- lmcache/v1/transfer_channel/nixl_channel.py +639 -0
- lmcache/v1/transfer_channel/py_socket_channel.py +260 -0
- lmcache/v1/transfer_channel/transfer_utils.py +63 -0
- lmcache/v1/utils/__init__.py +1 -0
- lmcache/v1/utils/bloom_filter.py +109 -0
- lmcache/v1/utils/cache_utils.py +125 -0
- lmcache_cli-0.4.5.dev0.dist-info/METADATA +185 -0
- lmcache_cli-0.4.5.dev0.dist-info/RECORD +399 -0
- lmcache_cli-0.4.5.dev0.dist-info/WHEEL +5 -0
- lmcache_cli-0.4.5.dev0.dist-info/entry_points.txt +2 -0
- lmcache_cli-0.4.5.dev0.dist-info/licenses/LICENSE +201 -0
- lmcache_cli-0.4.5.dev0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,253 @@
|
|
|
1
|
+
# SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
|
|
3
|
+
"""
|
|
4
|
+
Configuration for the multiprocess (ZMQ) server and HTTP frontend.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
# Standard
|
|
8
|
+
from dataclasses import dataclass, field
|
|
9
|
+
import argparse
|
|
10
|
+
import json
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
@dataclass
|
|
14
|
+
class MPServerConfig:
|
|
15
|
+
"""Configuration for the ZMQ-based multiprocess cache server."""
|
|
16
|
+
|
|
17
|
+
host: str = "localhost"
|
|
18
|
+
"""ZMQ server host."""
|
|
19
|
+
|
|
20
|
+
port: int = 5555
|
|
21
|
+
"""ZMQ server port."""
|
|
22
|
+
|
|
23
|
+
chunk_size: int = 256
|
|
24
|
+
"""Chunk size for KV cache operations."""
|
|
25
|
+
|
|
26
|
+
max_workers: int = 1
|
|
27
|
+
"""Base number of worker threads. Sets default for both GPU and CPU pools."""
|
|
28
|
+
|
|
29
|
+
max_gpu_workers: int = 1
|
|
30
|
+
"""Worker threads for the GPU affinity pool (STORE/RETRIEVE).
|
|
31
|
+
Resolved from --max-gpu-workers or --max-workers."""
|
|
32
|
+
|
|
33
|
+
max_cpu_workers: int = 1
|
|
34
|
+
"""Worker threads for the normal (CPU) pool (LOOKUP, END_SESSION, etc.).
|
|
35
|
+
Resolved from --max-cpu-workers or --max-workers."""
|
|
36
|
+
|
|
37
|
+
hash_algorithm: str = "blake3"
|
|
38
|
+
"""Hash algorithm for token-based operations (builtin, sha256_cbor, blake3)."""
|
|
39
|
+
|
|
40
|
+
engine_type: str = "default"
|
|
41
|
+
"""Cache engine backend type
|
|
42
|
+
('default' for MPCacheEngine, 'blend' for BlendEngineV2).
|
|
43
|
+
"""
|
|
44
|
+
|
|
45
|
+
runtime_plugin_config: "RuntimePluginConfig" = field(
|
|
46
|
+
default_factory=lambda: RuntimePluginConfig()
|
|
47
|
+
)
|
|
48
|
+
"""Runtime plugin configuration (locations + extra config)."""
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
@dataclass
|
|
52
|
+
class RuntimePluginConfig:
|
|
53
|
+
"""Configuration for runtime plugins."""
|
|
54
|
+
|
|
55
|
+
locations: list[str] = field(default_factory=list)
|
|
56
|
+
"""Paths to runtime plugin scripts or directories."""
|
|
57
|
+
|
|
58
|
+
extra_config: dict = field(default_factory=dict)
|
|
59
|
+
"""Extra key-value config forwarded to runtime plugins
|
|
60
|
+
via the JSON config blob.
|
|
61
|
+
Accepts a JSON string on the command line.
|
|
62
|
+
"""
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
DEFAULT_MP_SERVER_CONFIG = MPServerConfig()
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
@dataclass
|
|
69
|
+
class HTTPFrontendConfig:
|
|
70
|
+
"""Configuration for the HTTP frontend (uvicorn/FastAPI)."""
|
|
71
|
+
|
|
72
|
+
http_host: str = "0.0.0.0"
|
|
73
|
+
"""HTTP server host."""
|
|
74
|
+
|
|
75
|
+
http_port: int = 8080
|
|
76
|
+
"""HTTP server port."""
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
DEFAULT_HTTP_FRONTEND_CONFIG = HTTPFrontendConfig()
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def add_mp_server_args(
|
|
83
|
+
parser: argparse.ArgumentParser,
|
|
84
|
+
) -> argparse.ArgumentParser:
|
|
85
|
+
"""
|
|
86
|
+
Add MP server configuration arguments to an existing parser.
|
|
87
|
+
|
|
88
|
+
Args:
|
|
89
|
+
parser: The argument parser to add arguments to.
|
|
90
|
+
|
|
91
|
+
Returns:
|
|
92
|
+
The same parser with MP server arguments added.
|
|
93
|
+
"""
|
|
94
|
+
mp_group = parser.add_argument_group(
|
|
95
|
+
"MP Server", "Configuration for the ZMQ multiprocess cache server"
|
|
96
|
+
)
|
|
97
|
+
mp_group.add_argument(
|
|
98
|
+
"--host",
|
|
99
|
+
type=str,
|
|
100
|
+
default="localhost",
|
|
101
|
+
help="Host to bind the ZMQ server. Default is localhost.",
|
|
102
|
+
)
|
|
103
|
+
mp_group.add_argument(
|
|
104
|
+
"--port",
|
|
105
|
+
type=int,
|
|
106
|
+
default=5555,
|
|
107
|
+
help="Port to bind the ZMQ server. Default is 5555.",
|
|
108
|
+
)
|
|
109
|
+
mp_group.add_argument(
|
|
110
|
+
"--chunk-size",
|
|
111
|
+
type=int,
|
|
112
|
+
default=256,
|
|
113
|
+
help="Chunk size for KV cache operations. Default is 256.",
|
|
114
|
+
)
|
|
115
|
+
mp_group.add_argument(
|
|
116
|
+
"--max-workers",
|
|
117
|
+
type=int,
|
|
118
|
+
default=1,
|
|
119
|
+
help="Base number of worker threads for both GPU and CPU pools. "
|
|
120
|
+
"Default is 1. Can be overridden per-pool with "
|
|
121
|
+
"--max-gpu-workers and --max-cpu-workers.",
|
|
122
|
+
)
|
|
123
|
+
mp_group.add_argument(
|
|
124
|
+
"--max-gpu-workers",
|
|
125
|
+
type=int,
|
|
126
|
+
default=None,
|
|
127
|
+
help="Worker threads for the GPU affinity pool (STORE/RETRIEVE). "
|
|
128
|
+
"Defaults to --max-workers if not specified.",
|
|
129
|
+
)
|
|
130
|
+
mp_group.add_argument(
|
|
131
|
+
"--max-cpu-workers",
|
|
132
|
+
type=int,
|
|
133
|
+
default=None,
|
|
134
|
+
help="Worker threads for the normal CPU pool (LOOKUP, etc.). "
|
|
135
|
+
"Defaults to --max-workers if not specified.",
|
|
136
|
+
)
|
|
137
|
+
mp_group.add_argument(
|
|
138
|
+
"--hash-algorithm",
|
|
139
|
+
type=str,
|
|
140
|
+
default="blake3",
|
|
141
|
+
help="Hash algorithm for token-based operations "
|
|
142
|
+
"(builtin, sha256_cbor, blake3). Default is blake3.",
|
|
143
|
+
)
|
|
144
|
+
mp_group.add_argument(
|
|
145
|
+
"--engine-type",
|
|
146
|
+
type=str,
|
|
147
|
+
default="default",
|
|
148
|
+
choices=["default", "blend"],
|
|
149
|
+
help="Cache engine backend type. 'default' uses MPCacheEngine, "
|
|
150
|
+
"'blend' uses BlendEngineV2 for cross-request KV reuse. "
|
|
151
|
+
"Default is 'default'.",
|
|
152
|
+
)
|
|
153
|
+
mp_group.add_argument(
|
|
154
|
+
"--runtime-plugin-locations",
|
|
155
|
+
type=str,
|
|
156
|
+
nargs="*",
|
|
157
|
+
default=[],
|
|
158
|
+
help="Paths to runtime plugin scripts or "
|
|
159
|
+
"directories to launch alongside the server.",
|
|
160
|
+
)
|
|
161
|
+
mp_group.add_argument(
|
|
162
|
+
"--runtime-plugin-config",
|
|
163
|
+
type=str,
|
|
164
|
+
default="{}",
|
|
165
|
+
help="JSON string of extra key-value config forwarded to runtime "
|
|
166
|
+
"plugins via LMCACHE_RUNTIME_PLUGIN_EXTRA_CONFIG. "
|
|
167
|
+
'Example: \'{"plugin.frontend.heartbeat_url": '
|
|
168
|
+
'"http://localhost:5000/heartbeat"}\'',
|
|
169
|
+
)
|
|
170
|
+
return parser
|
|
171
|
+
|
|
172
|
+
|
|
173
|
+
def parse_args_to_mp_server_config(
|
|
174
|
+
args: argparse.Namespace,
|
|
175
|
+
) -> MPServerConfig:
|
|
176
|
+
"""
|
|
177
|
+
Convert parsed command line arguments to an MPServerConfig.
|
|
178
|
+
|
|
179
|
+
Args:
|
|
180
|
+
args: Parsed arguments from the argument parser.
|
|
181
|
+
|
|
182
|
+
Returns:
|
|
183
|
+
MPServerConfig: The configuration object.
|
|
184
|
+
"""
|
|
185
|
+
base = args.max_workers
|
|
186
|
+
max_gpu = args.max_gpu_workers if args.max_gpu_workers is not None else base
|
|
187
|
+
max_cpu = args.max_cpu_workers if args.max_cpu_workers is not None else base
|
|
188
|
+
try:
|
|
189
|
+
plugin_extra = json.loads(getattr(args, "runtime_plugin_config", None) or "{}")
|
|
190
|
+
except json.JSONDecodeError as exc:
|
|
191
|
+
raise ValueError("--runtime-plugin-config is not valid JSON: %s" % exc) from exc
|
|
192
|
+
return MPServerConfig(
|
|
193
|
+
host=args.host,
|
|
194
|
+
port=args.port,
|
|
195
|
+
chunk_size=args.chunk_size,
|
|
196
|
+
max_workers=base,
|
|
197
|
+
max_gpu_workers=max_gpu,
|
|
198
|
+
max_cpu_workers=max_cpu,
|
|
199
|
+
hash_algorithm=args.hash_algorithm,
|
|
200
|
+
engine_type=args.engine_type,
|
|
201
|
+
runtime_plugin_config=RuntimePluginConfig(
|
|
202
|
+
locations=(args.runtime_plugin_locations or []),
|
|
203
|
+
extra_config=plugin_extra,
|
|
204
|
+
),
|
|
205
|
+
)
|
|
206
|
+
|
|
207
|
+
|
|
208
|
+
def add_http_frontend_args(
|
|
209
|
+
parser: argparse.ArgumentParser,
|
|
210
|
+
) -> argparse.ArgumentParser:
|
|
211
|
+
"""
|
|
212
|
+
Add HTTP frontend configuration arguments to an existing parser.
|
|
213
|
+
|
|
214
|
+
Args:
|
|
215
|
+
parser: The argument parser to add arguments to.
|
|
216
|
+
|
|
217
|
+
Returns:
|
|
218
|
+
The same parser with HTTP frontend arguments added.
|
|
219
|
+
"""
|
|
220
|
+
http_group = parser.add_argument_group(
|
|
221
|
+
"HTTP Frontend", "Configuration for the HTTP frontend server"
|
|
222
|
+
)
|
|
223
|
+
http_group.add_argument(
|
|
224
|
+
"--http-host",
|
|
225
|
+
type=str,
|
|
226
|
+
default="0.0.0.0",
|
|
227
|
+
help="Host to bind the HTTP server. Default is 0.0.0.0.",
|
|
228
|
+
)
|
|
229
|
+
http_group.add_argument(
|
|
230
|
+
"--http-port",
|
|
231
|
+
type=int,
|
|
232
|
+
default=8080,
|
|
233
|
+
help="Port to bind the HTTP server. Default is 8080.",
|
|
234
|
+
)
|
|
235
|
+
return parser
|
|
236
|
+
|
|
237
|
+
|
|
238
|
+
def parse_args_to_http_frontend_config(
|
|
239
|
+
args: argparse.Namespace,
|
|
240
|
+
) -> HTTPFrontendConfig:
|
|
241
|
+
"""
|
|
242
|
+
Convert parsed command line arguments to an HTTPFrontendConfig.
|
|
243
|
+
|
|
244
|
+
Args:
|
|
245
|
+
args: Parsed arguments from the argument parser.
|
|
246
|
+
|
|
247
|
+
Returns:
|
|
248
|
+
HTTPFrontendConfig: The configuration object.
|
|
249
|
+
"""
|
|
250
|
+
return HTTPFrontendConfig(
|
|
251
|
+
http_host=args.http_host,
|
|
252
|
+
http_port=args.http_port,
|
|
253
|
+
)
|
|
@@ -0,0 +1,281 @@
|
|
|
1
|
+
# SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
# Standard
|
|
3
|
+
from dataclasses import dataclass, field
|
|
4
|
+
from typing import Any, Callable
|
|
5
|
+
import pickle
|
|
6
|
+
import threading
|
|
7
|
+
|
|
8
|
+
# Third Party
|
|
9
|
+
import msgspec
|
|
10
|
+
import torch
|
|
11
|
+
|
|
12
|
+
"""
|
|
13
|
+
Defines the types and the customized encoder/decoders for inter-process
|
|
14
|
+
communications.
|
|
15
|
+
|
|
16
|
+
Key Types:
|
|
17
|
+
- IPCCacheEngineKey: Token-based cache key
|
|
18
|
+
- Contains token_ids, start, end, request_id (all required)
|
|
19
|
+
- Converted to ObjectKey for storage operations via ipc_key_to_object_keys()
|
|
20
|
+
"""
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
class CudaIPCWrapper:
|
|
24
|
+
_discovered_device_mapping: dict[str, int] = {}
|
|
25
|
+
_device_mapping_lock = threading.Lock()
|
|
26
|
+
|
|
27
|
+
@staticmethod
|
|
28
|
+
def _get_device_uuid(device_index: int) -> str:
|
|
29
|
+
"""Get the UUID of a GPU device given its index."""
|
|
30
|
+
return str(torch.cuda.get_device_properties(device_index).uuid)
|
|
31
|
+
|
|
32
|
+
@staticmethod
|
|
33
|
+
def _discover_gpu_devices():
|
|
34
|
+
"""Discover all available GPU devices and map their UUIDs to
|
|
35
|
+
the physical device ordinals.
|
|
36
|
+
"""
|
|
37
|
+
if not torch.cuda.is_available():
|
|
38
|
+
return
|
|
39
|
+
|
|
40
|
+
num_devices = torch.cuda.device_count()
|
|
41
|
+
with CudaIPCWrapper._device_mapping_lock:
|
|
42
|
+
if CudaIPCWrapper._discovered_device_mapping:
|
|
43
|
+
return # Already discovered
|
|
44
|
+
|
|
45
|
+
for i in range(num_devices):
|
|
46
|
+
device_uuid = CudaIPCWrapper._get_device_uuid(i)
|
|
47
|
+
CudaIPCWrapper._discovered_device_mapping[device_uuid] = i
|
|
48
|
+
|
|
49
|
+
@staticmethod
|
|
50
|
+
def _get_device_index_from_uuid(device_uuid: str) -> int:
|
|
51
|
+
"""Get the physical device ordinal from its UUID."""
|
|
52
|
+
CudaIPCWrapper._discover_gpu_devices()
|
|
53
|
+
|
|
54
|
+
with CudaIPCWrapper._device_mapping_lock:
|
|
55
|
+
device_index = CudaIPCWrapper._discovered_device_mapping.get(
|
|
56
|
+
device_uuid, None
|
|
57
|
+
)
|
|
58
|
+
|
|
59
|
+
if device_index is None:
|
|
60
|
+
raise RuntimeError(
|
|
61
|
+
f"Device UUID {device_uuid} not found in the discovered devices."
|
|
62
|
+
"Please make sure the process can see all the GPU devices"
|
|
63
|
+
)
|
|
64
|
+
return device_index
|
|
65
|
+
|
|
66
|
+
def __init__(self, tensor: torch.Tensor):
|
|
67
|
+
# First Party
|
|
68
|
+
from lmcache.v1.gpu_connector.utils import assert_contiguous
|
|
69
|
+
|
|
70
|
+
assert_contiguous(tensor)
|
|
71
|
+
|
|
72
|
+
storage = tensor.untyped_storage()
|
|
73
|
+
handle = storage._share_cuda_()
|
|
74
|
+
|
|
75
|
+
self.handle = handle
|
|
76
|
+
self.dtype = tensor.dtype
|
|
77
|
+
self.shape = tuple(tensor.shape)
|
|
78
|
+
self.stride = tuple(tensor.stride())
|
|
79
|
+
self.storage_offset = int(tensor.storage_offset())
|
|
80
|
+
|
|
81
|
+
device_index = tensor.device.index
|
|
82
|
+
self.device_uuid = CudaIPCWrapper._get_device_uuid(device_index)
|
|
83
|
+
|
|
84
|
+
def to_tensor(self) -> torch.Tensor:
|
|
85
|
+
"""
|
|
86
|
+
Note:
|
|
87
|
+
This function may break if torch cuda is not initialized.
|
|
88
|
+
We should call `torch.cuda.init()` before using this function.
|
|
89
|
+
"""
|
|
90
|
+
device_index = CudaIPCWrapper._get_device_index_from_uuid(self.device_uuid)
|
|
91
|
+
|
|
92
|
+
storage = torch.UntypedStorage._new_shared_cuda( # noqa: SLF001
|
|
93
|
+
device_index, *self.handle[1:]
|
|
94
|
+
)
|
|
95
|
+
|
|
96
|
+
t = torch.empty((), device=f"cuda:{device_index}", dtype=self.dtype)
|
|
97
|
+
t.set_(storage, self.storage_offset, self.shape, self.stride)
|
|
98
|
+
return t
|
|
99
|
+
|
|
100
|
+
def __eq__(self, other):
|
|
101
|
+
if not isinstance(other, CudaIPCWrapper):
|
|
102
|
+
return False
|
|
103
|
+
return (
|
|
104
|
+
self.handle == other.handle
|
|
105
|
+
and self.dtype == other.dtype
|
|
106
|
+
and self.shape == other.shape
|
|
107
|
+
and self.stride == other.stride
|
|
108
|
+
and self.storage_offset == other.storage_offset
|
|
109
|
+
and self.device_uuid == other.device_uuid
|
|
110
|
+
)
|
|
111
|
+
|
|
112
|
+
@staticmethod
|
|
113
|
+
def Serialize(obj: "CudaIPCWrapper") -> bytes:
|
|
114
|
+
return pickle.dumps(obj)
|
|
115
|
+
|
|
116
|
+
@staticmethod
|
|
117
|
+
def Deserialize(data: bytes) -> "CudaIPCWrapper":
|
|
118
|
+
return pickle.loads(data)
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
@dataclass(order=True, frozen=True)
|
|
122
|
+
class IPCCacheEngineKey:
|
|
123
|
+
"""Cache key for the IPC (multiprocess) protocol.
|
|
124
|
+
|
|
125
|
+
This key type is sent by the client over ZMQ (serialized via msgspec).
|
|
126
|
+
|
|
127
|
+
The client sends token_ids, start, end, and request_id (all required).
|
|
128
|
+
The server computes chunk hashes via TokenHasher and converts to
|
|
129
|
+
ObjectKey for storage operations using ipc_key_to_object_keys().
|
|
130
|
+
|
|
131
|
+
The request_id field is for session tracking and is NOT included
|
|
132
|
+
in equality/hash comparisons (two keys with same content but different
|
|
133
|
+
request_ids are considered equal for cache purposes).
|
|
134
|
+
"""
|
|
135
|
+
|
|
136
|
+
model_name: str
|
|
137
|
+
world_size: int
|
|
138
|
+
worker_id: int | None
|
|
139
|
+
|
|
140
|
+
token_ids: tuple[int, ...] # frozen tuple for hashability
|
|
141
|
+
start: int
|
|
142
|
+
end: int
|
|
143
|
+
|
|
144
|
+
# === Session tracking (not part of cache identity) ===
|
|
145
|
+
request_id: str = field(compare=False)
|
|
146
|
+
|
|
147
|
+
# === Per-user isolation salt (part of cache identity) ===
|
|
148
|
+
# msgspec encodes dataclasses as maps, so forward wire compatibility
|
|
149
|
+
# works by field name: an old payload without ``cache_salt`` decodes
|
|
150
|
+
# on new code using the default "". Placing the field last is a style
|
|
151
|
+
# choice — all defaulted fields must come after non-defaulted ones.
|
|
152
|
+
#
|
|
153
|
+
# Invariant: must not contain ``@``, ``/``, ``\``, or NUL, and
|
|
154
|
+
# must be <= 128 chars — same rationale as ObjectKey (see
|
|
155
|
+
# ObjectKey.cache_salt). Validated in __post_init__.
|
|
156
|
+
cache_salt: str = ""
|
|
157
|
+
|
|
158
|
+
# Duplicated from ObjectKey — cannot import ObjectKey here due to
|
|
159
|
+
# circular dependency (api.py imports IPCCacheEngineKey).
|
|
160
|
+
_SALT_FORBIDDEN_CHARS = frozenset("@/\\\x00")
|
|
161
|
+
_SALT_MAX_LEN = 128
|
|
162
|
+
|
|
163
|
+
def __post_init__(self) -> None:
|
|
164
|
+
bad = self._SALT_FORBIDDEN_CHARS & set(self.cache_salt)
|
|
165
|
+
if bad:
|
|
166
|
+
raise ValueError(
|
|
167
|
+
f"cache_salt must not contain {bad!r} (got {self.cache_salt!r})"
|
|
168
|
+
)
|
|
169
|
+
if len(self.cache_salt) > self._SALT_MAX_LEN:
|
|
170
|
+
raise ValueError(
|
|
171
|
+
f"cache_salt exceeds max length {self._SALT_MAX_LEN} "
|
|
172
|
+
f"(got {len(self.cache_salt)})"
|
|
173
|
+
)
|
|
174
|
+
|
|
175
|
+
# Helper function for unit tests only
|
|
176
|
+
@classmethod
|
|
177
|
+
def from_token_ids(
|
|
178
|
+
cls,
|
|
179
|
+
model_name: str,
|
|
180
|
+
world_size: int,
|
|
181
|
+
worker_id: int | None,
|
|
182
|
+
token_ids: list[int],
|
|
183
|
+
start: int = 0,
|
|
184
|
+
end: int = 0,
|
|
185
|
+
request_id: str = "",
|
|
186
|
+
cache_salt: str = "",
|
|
187
|
+
) -> "IPCCacheEngineKey":
|
|
188
|
+
"""Create a key from token ids. Only used by the tests."""
|
|
189
|
+
return cls(
|
|
190
|
+
model_name=model_name,
|
|
191
|
+
world_size=world_size,
|
|
192
|
+
worker_id=worker_id,
|
|
193
|
+
token_ids=tuple(token_ids),
|
|
194
|
+
start=start,
|
|
195
|
+
end=end,
|
|
196
|
+
request_id=request_id,
|
|
197
|
+
cache_salt=cache_salt,
|
|
198
|
+
)
|
|
199
|
+
|
|
200
|
+
def no_worker_id_version(self) -> "IPCCacheEngineKey":
|
|
201
|
+
"""Create a copy with worker_id=None for lookup requests."""
|
|
202
|
+
return IPCCacheEngineKey(
|
|
203
|
+
model_name=self.model_name,
|
|
204
|
+
world_size=self.world_size,
|
|
205
|
+
worker_id=None,
|
|
206
|
+
token_ids=self.token_ids,
|
|
207
|
+
start=self.start,
|
|
208
|
+
end=self.end,
|
|
209
|
+
request_id=self.request_id,
|
|
210
|
+
cache_salt=self.cache_salt,
|
|
211
|
+
)
|
|
212
|
+
|
|
213
|
+
|
|
214
|
+
# Type exports
|
|
215
|
+
KVCache = list[CudaIPCWrapper]
|
|
216
|
+
|
|
217
|
+
|
|
218
|
+
@dataclass
|
|
219
|
+
class CustomizedSerdeConfig:
|
|
220
|
+
serializer: Callable[[Any], bytes]
|
|
221
|
+
deserializer: Callable[[bytes], Any]
|
|
222
|
+
code: int
|
|
223
|
+
|
|
224
|
+
|
|
225
|
+
_CUSTOMERIZED_SERIALIZERS = {
|
|
226
|
+
CudaIPCWrapper: CustomizedSerdeConfig(
|
|
227
|
+
serializer=CudaIPCWrapper.Serialize,
|
|
228
|
+
deserializer=CudaIPCWrapper.Deserialize,
|
|
229
|
+
code=1,
|
|
230
|
+
),
|
|
231
|
+
}
|
|
232
|
+
|
|
233
|
+
|
|
234
|
+
def get_customized_encoder(type: Any) -> msgspec.msgpack.Encoder:
|
|
235
|
+
# TODO: `type` is not used here
|
|
236
|
+
def enc_hook(obj: Any) -> Any:
|
|
237
|
+
for supported_type, cfg in _CUSTOMERIZED_SERIALIZERS.items():
|
|
238
|
+
if isinstance(obj, supported_type):
|
|
239
|
+
data = cfg.serializer(obj)
|
|
240
|
+
return msgspec.msgpack.Ext(cfg.code, data)
|
|
241
|
+
raise TypeError(f"Unsupported type for serialization: {type(obj)}")
|
|
242
|
+
|
|
243
|
+
return msgspec.msgpack.Encoder(enc_hook=enc_hook)
|
|
244
|
+
|
|
245
|
+
|
|
246
|
+
def get_customized_decoder(type: Any) -> msgspec.msgpack.Decoder:
|
|
247
|
+
def ext_hook(code: int, data: bytes) -> Any:
|
|
248
|
+
for cfg in _CUSTOMERIZED_SERIALIZERS.values():
|
|
249
|
+
if cfg.code == code:
|
|
250
|
+
return cfg.deserializer(data)
|
|
251
|
+
raise TypeError(f"Unsupported ext code for deserialization: {code}")
|
|
252
|
+
|
|
253
|
+
return msgspec.msgpack.Decoder(ext_hook=ext_hook, type=type)
|
|
254
|
+
|
|
255
|
+
|
|
256
|
+
@dataclass
|
|
257
|
+
class BlockAllocationRecord:
|
|
258
|
+
"""A single per-request GPU block allocation delta from vLLM."""
|
|
259
|
+
|
|
260
|
+
req_id: str
|
|
261
|
+
new_block_ids: list[int]
|
|
262
|
+
new_token_ids: list[int]
|
|
263
|
+
|
|
264
|
+
|
|
265
|
+
@dataclass
|
|
266
|
+
class CBMatchResult:
|
|
267
|
+
"""Result of a sub-sequence match from BlendTokenRangeMatcher.
|
|
268
|
+
|
|
269
|
+
Attributes:
|
|
270
|
+
old_st: Start position in the originally registered (stored) sequence.
|
|
271
|
+
old_ed: End position in the originally registered (stored) sequence.
|
|
272
|
+
cur_st: Start position in the query sequence where the match was found.
|
|
273
|
+
cur_ed: End position in the query sequence where the match was found.
|
|
274
|
+
hash: Token hash bytes (from registration) used as the storage key.
|
|
275
|
+
"""
|
|
276
|
+
|
|
277
|
+
old_st: int
|
|
278
|
+
old_ed: int
|
|
279
|
+
cur_st: int
|
|
280
|
+
cur_ed: int
|
|
281
|
+
hash: bytes
|