lmcache-cli 0.4.5.dev0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- lmcache/__init__.py +84 -0
- lmcache/_version.py +24 -0
- lmcache/cli/__init__.py +1 -0
- lmcache/cli/commands/__init__.py +34 -0
- lmcache/cli/commands/base.py +157 -0
- lmcache/cli/commands/bench/__init__.py +557 -0
- lmcache/cli/commands/bench/engine_bench/__init__.py +1 -0
- lmcache/cli/commands/bench/engine_bench/config.py +245 -0
- lmcache/cli/commands/bench/engine_bench/interactive/__init__.py +274 -0
- lmcache/cli/commands/bench/engine_bench/interactive/config.json +10 -0
- lmcache/cli/commands/bench/engine_bench/interactive/schema.py +352 -0
- lmcache/cli/commands/bench/engine_bench/interactive/state.py +327 -0
- lmcache/cli/commands/bench/engine_bench/interactive/terminal.py +291 -0
- lmcache/cli/commands/bench/engine_bench/progress.py +145 -0
- lmcache/cli/commands/bench/engine_bench/request_sender.py +232 -0
- lmcache/cli/commands/bench/engine_bench/stats.py +275 -0
- lmcache/cli/commands/bench/engine_bench/workloads/__init__.py +153 -0
- lmcache/cli/commands/bench/engine_bench/workloads/base.py +122 -0
- lmcache/cli/commands/bench/engine_bench/workloads/long_doc_permutator.py +435 -0
- lmcache/cli/commands/bench/engine_bench/workloads/long_doc_qa.py +281 -0
- lmcache/cli/commands/bench/engine_bench/workloads/multi_round_chat.py +337 -0
- lmcache/cli/commands/bench/engine_bench/workloads/random_prefill.py +178 -0
- lmcache/cli/commands/describe.py +310 -0
- lmcache/cli/commands/kvcache.py +133 -0
- lmcache/cli/commands/mock.py +75 -0
- lmcache/cli/commands/ping.py +113 -0
- lmcache/cli/commands/query/__init__.py +155 -0
- lmcache/cli/commands/query/prompt.py +134 -0
- lmcache/cli/commands/query/request.py +357 -0
- lmcache/cli/commands/server.py +99 -0
- lmcache/cli/commands/tool/__init__.py +63 -0
- lmcache/cli/commands/tool/cache_simulator.py +113 -0
- lmcache/cli/commands/trace/__init__.py +505 -0
- lmcache/cli/commands/trace/dispatch.py +249 -0
- lmcache/cli/commands/trace/driver.py +372 -0
- lmcache/cli/commands/trace/stats.py +289 -0
- lmcache/cli/documents/lmcache.txt +11 -0
- lmcache/cli/main.py +42 -0
- lmcache/cli/metrics/__init__.py +29 -0
- lmcache/cli/metrics/formatter.py +171 -0
- lmcache/cli/metrics/handler.py +94 -0
- lmcache/cli/metrics/metrics.py +161 -0
- lmcache/cli/metrics/section.py +77 -0
- lmcache/connections.py +173 -0
- lmcache/integration/__init__.py +2 -0
- lmcache/integration/base_service_factory.py +165 -0
- lmcache/integration/request_telemetry/__init__.py +1 -0
- lmcache/integration/request_telemetry/base.py +51 -0
- lmcache/integration/request_telemetry/factory.py +113 -0
- lmcache/integration/request_telemetry/fastapi.py +109 -0
- lmcache/integration/request_telemetry/noop.py +35 -0
- lmcache/integration/sglang/__init__.py +2 -0
- lmcache/integration/sglang/sglang_adapter.py +326 -0
- lmcache/integration/sglang/utils.py +39 -0
- lmcache/integration/vllm/__init__.py +1 -0
- lmcache/integration/vllm/lmcache_connector_v1.py +213 -0
- lmcache/integration/vllm/lmcache_connector_v1_085.py +150 -0
- lmcache/integration/vllm/lmcache_mp_connector_0180.py +1072 -0
- lmcache/integration/vllm/tests/test_mm_hash_utils.py +112 -0
- lmcache/integration/vllm/utils.py +433 -0
- lmcache/integration/vllm/vllm_multi_process_adapter.py +1090 -0
- lmcache/integration/vllm/vllm_service_factory.py +339 -0
- lmcache/integration/vllm/vllm_v1_adapter.py +1713 -0
- lmcache/logging.py +107 -0
- lmcache/native_storage_ops.pyi +230 -0
- lmcache/non_cuda_equivalents.py +1424 -0
- lmcache/observability.py +1958 -0
- lmcache/storage_backend/serde/__init__.py +1 -0
- lmcache/storage_backend/serde/cachegen_basics.py +210 -0
- lmcache/storage_backend/serde/cachegen_decoder.py +207 -0
- lmcache/storage_backend/serde/cachegen_encoder.py +394 -0
- lmcache/storage_backend/serde/serde.py +75 -0
- lmcache/tools/__init__.py +1 -0
- lmcache/tools/cache_simulator/README.md +392 -0
- lmcache/tools/cache_simulator/__init__.py +1 -0
- lmcache/tools/cache_simulator/docs/simulate_example.png +0 -0
- lmcache/tools/cache_simulator/docs/sweep_example.png +0 -0
- lmcache/tools/cache_simulator/gen_bench_dataset.py +360 -0
- lmcache/tools/cache_simulator/lru_cache.py +124 -0
- lmcache/tools/cache_simulator/plot_hit_rate.py +231 -0
- lmcache/tools/cache_simulator/simulator.py +795 -0
- lmcache/tools/controller_benchmark/README.md +161 -0
- lmcache/tools/controller_benchmark/__init__.py +1 -0
- lmcache/tools/controller_benchmark/__main__.py +331 -0
- lmcache/tools/controller_benchmark/benchmark.py +660 -0
- lmcache/tools/controller_benchmark/config.py +44 -0
- lmcache/tools/controller_benchmark/constants.py +10 -0
- lmcache/tools/controller_benchmark/handlers/__init__.py +46 -0
- lmcache/tools/controller_benchmark/handlers/admit.py +52 -0
- lmcache/tools/controller_benchmark/handlers/base.py +47 -0
- lmcache/tools/controller_benchmark/handlers/deregister.py +49 -0
- lmcache/tools/controller_benchmark/handlers/evict.py +52 -0
- lmcache/tools/controller_benchmark/handlers/heartbeat.py +56 -0
- lmcache/tools/controller_benchmark/handlers/p2p_lookup.py +47 -0
- lmcache/tools/controller_benchmark/handlers/register.py +56 -0
- lmcache/tools/mp_status_viewer/__init__.py +1 -0
- lmcache/tools/mp_status_viewer/__main__.py +95 -0
- lmcache/usage_context.py +417 -0
- lmcache/utils.py +665 -0
- lmcache/v1/__init__.py +2 -0
- lmcache/v1/api_server/__init__.py +2 -0
- lmcache/v1/api_server/__main__.py +537 -0
- lmcache/v1/basic_check.py +112 -0
- lmcache/v1/cache_controller/__init__.py +9 -0
- lmcache/v1/cache_controller/commands/__init__.py +15 -0
- lmcache/v1/cache_controller/commands/base.py +35 -0
- lmcache/v1/cache_controller/commands/full_sync.py +49 -0
- lmcache/v1/cache_controller/config.py +176 -0
- lmcache/v1/cache_controller/controller_manager.py +535 -0
- lmcache/v1/cache_controller/controllers/__init__.py +11 -0
- lmcache/v1/cache_controller/controllers/full_sync_tracker.py +473 -0
- lmcache/v1/cache_controller/controllers/kv_controller.py +439 -0
- lmcache/v1/cache_controller/controllers/registration_controller.py +282 -0
- lmcache/v1/cache_controller/executor.py +463 -0
- lmcache/v1/cache_controller/frontend/static/css/style.css +201 -0
- lmcache/v1/cache_controller/frontend/static/img/logo.png +0 -0
- lmcache/v1/cache_controller/frontend/static/index.html +234 -0
- lmcache/v1/cache_controller/frontend/static/js/controller_app.js +660 -0
- lmcache/v1/cache_controller/full_sync_sender.py +475 -0
- lmcache/v1/cache_controller/locks.py +149 -0
- lmcache/v1/cache_controller/message.py +828 -0
- lmcache/v1/cache_controller/observability.py +208 -0
- lmcache/v1/cache_controller/utils.py +679 -0
- lmcache/v1/cache_controller/worker.py +665 -0
- lmcache/v1/cache_engine.py +2058 -0
- lmcache/v1/cache_interface.py +19 -0
- lmcache/v1/check/__init__.py +74 -0
- lmcache/v1/check/check_mode_gen.py +86 -0
- lmcache/v1/check/check_mode_test_l2_adapter.py +284 -0
- lmcache/v1/check/check_mode_test_remote.py +155 -0
- lmcache/v1/check/check_mode_test_storage_manager.py +142 -0
- lmcache/v1/check/utils.py +571 -0
- lmcache/v1/compute/__init__.py +2 -0
- lmcache/v1/compute/attention/__init__.py +0 -0
- lmcache/v1/compute/attention/abstract.py +39 -0
- lmcache/v1/compute/attention/flash_attn.py +129 -0
- lmcache/v1/compute/attention/flash_infer_sparse.py +284 -0
- lmcache/v1/compute/attention/metadata.py +85 -0
- lmcache/v1/compute/attention/utils.py +14 -0
- lmcache/v1/compute/blend/__init__.py +7 -0
- lmcache/v1/compute/blend/blender.py +168 -0
- lmcache/v1/compute/blend/metadata.py +34 -0
- lmcache/v1/compute/blend/utils.py +63 -0
- lmcache/v1/compute/models/__init__.py +0 -0
- lmcache/v1/compute/models/base.py +141 -0
- lmcache/v1/compute/models/llama.py +9 -0
- lmcache/v1/compute/models/qwen3.py +24 -0
- lmcache/v1/compute/models/utils.py +68 -0
- lmcache/v1/compute/positional_encoding.py +199 -0
- lmcache/v1/config.py +848 -0
- lmcache/v1/config_base.py +848 -0
- lmcache/v1/distributed/api.py +248 -0
- lmcache/v1/distributed/config.py +321 -0
- lmcache/v1/distributed/error.py +64 -0
- lmcache/v1/distributed/eviction.py +192 -0
- lmcache/v1/distributed/eviction_policy/__init__.py +21 -0
- lmcache/v1/distributed/eviction_policy/factory.py +27 -0
- lmcache/v1/distributed/eviction_policy/lru.py +244 -0
- lmcache/v1/distributed/eviction_policy/noop.py +50 -0
- lmcache/v1/distributed/internal_api.py +170 -0
- lmcache/v1/distributed/l1_manager.py +835 -0
- lmcache/v1/distributed/l2_adapters/__init__.py +67 -0
- lmcache/v1/distributed/l2_adapters/base.py +360 -0
- lmcache/v1/distributed/l2_adapters/config.py +385 -0
- lmcache/v1/distributed/l2_adapters/factory.py +205 -0
- lmcache/v1/distributed/l2_adapters/fs_l2_adapter.py +747 -0
- lmcache/v1/distributed/l2_adapters/fs_native_l2_adapter.py +167 -0
- lmcache/v1/distributed/l2_adapters/mock_l2_adapter.py +516 -0
- lmcache/v1/distributed/l2_adapters/mooncake_store_l2_adapter.py +135 -0
- lmcache/v1/distributed/l2_adapters/native_connector_l2_adapter.py +468 -0
- lmcache/v1/distributed/l2_adapters/native_plugin_l2_adapter.py +199 -0
- lmcache/v1/distributed/l2_adapters/nixl_store_dynamic_l2_adapter.py +831 -0
- lmcache/v1/distributed/l2_adapters/nixl_store_l2_adapter.py +983 -0
- lmcache/v1/distributed/l2_adapters/plugin_l2_adapter.py +210 -0
- lmcache/v1/distributed/l2_adapters/resp_l2_adapter.py +176 -0
- lmcache/v1/distributed/memory_manager.py +179 -0
- lmcache/v1/distributed/storage_controller.py +39 -0
- lmcache/v1/distributed/storage_controllers/__init__.py +43 -0
- lmcache/v1/distributed/storage_controllers/eviction_controller.py +242 -0
- lmcache/v1/distributed/storage_controllers/prefetch_controller.py +830 -0
- lmcache/v1/distributed/storage_controllers/prefetch_policy.py +193 -0
- lmcache/v1/distributed/storage_controllers/store_controller.py +452 -0
- lmcache/v1/distributed/storage_controllers/store_policy.py +213 -0
- lmcache/v1/distributed/storage_manager.py +532 -0
- lmcache/v1/event_manager.py +145 -0
- lmcache/v1/exceptions/__init__.py +16 -0
- lmcache/v1/gpu_connector/__init__.py +126 -0
- lmcache/v1/gpu_connector/gpu_connectors.py +1906 -0
- lmcache/v1/gpu_connector/gpu_ops.py +85 -0
- lmcache/v1/gpu_connector/hpu_connector.py +326 -0
- lmcache/v1/gpu_connector/mock_gpu_connector.py +67 -0
- lmcache/v1/gpu_connector/utils.py +890 -0
- lmcache/v1/gpu_connector/xpu_connectors.py +916 -0
- lmcache/v1/health_monitor/__init__.py +1 -0
- lmcache/v1/health_monitor/base.py +587 -0
- lmcache/v1/health_monitor/checks/__init__.py +1 -0
- lmcache/v1/health_monitor/checks/remote_backend_check.py +304 -0
- lmcache/v1/health_monitor/constants.py +36 -0
- lmcache/v1/internal_api_server/__init__.py +0 -0
- lmcache/v1/internal_api_server/api_registry.py +59 -0
- lmcache/v1/internal_api_server/api_server.py +120 -0
- lmcache/v1/internal_api_server/common/__init__.py +1 -0
- lmcache/v1/internal_api_server/common/env_api.py +22 -0
- lmcache/v1/internal_api_server/common/loglevel_api.py +57 -0
- lmcache/v1/internal_api_server/common/metrics_api.py +29 -0
- lmcache/v1/internal_api_server/common/periodic_thread_api.py +138 -0
- lmcache/v1/internal_api_server/common/run_script_api.py +73 -0
- lmcache/v1/internal_api_server/common/thread_api.py +63 -0
- lmcache/v1/internal_api_server/controller/__init__.py +1 -0
- lmcache/v1/internal_api_server/controller/key_stats_api.py +81 -0
- lmcache/v1/internal_api_server/controller/worker_info_api.py +136 -0
- lmcache/v1/internal_api_server/utils.py +43 -0
- lmcache/v1/internal_api_server/vllm/__init__.py +1 -0
- lmcache/v1/internal_api_server/vllm/backend_api.py +221 -0
- lmcache/v1/internal_api_server/vllm/bypass_api.py +204 -0
- lmcache/v1/internal_api_server/vllm/cache_api.py +895 -0
- lmcache/v1/internal_api_server/vllm/chunk_statistics_api.py +141 -0
- lmcache/v1/internal_api_server/vllm/conf_api.py +147 -0
- lmcache/v1/internal_api_server/vllm/freeze_api.py +172 -0
- lmcache/v1/internal_api_server/vllm/hot_cache_api.py +184 -0
- lmcache/v1/internal_api_server/vllm/inference_api.py +65 -0
- lmcache/v1/internal_api_server/vllm/load_fs_chunks_api.py +320 -0
- lmcache/v1/internal_api_server/vllm/lookup_api.py +145 -0
- lmcache/v1/internal_api_server/vllm/version_api.py +25 -0
- lmcache/v1/kv_layer_groups.py +267 -0
- lmcache/v1/lazy_memory_allocator.py +284 -0
- lmcache/v1/lookup_client/__init__.py +25 -0
- lmcache/v1/lookup_client/abstract_client.py +77 -0
- lmcache/v1/lookup_client/async_lookup_message.py +50 -0
- lmcache/v1/lookup_client/chunk_statistics_lookup_client.py +200 -0
- lmcache/v1/lookup_client/factory.py +251 -0
- lmcache/v1/lookup_client/hit_limit_lookup_client.py +86 -0
- lmcache/v1/lookup_client/lmcache_async_lookup_client.py +407 -0
- lmcache/v1/lookup_client/lmcache_lookup_client.py +285 -0
- lmcache/v1/lookup_client/lmcache_lookup_client_bypass.py +99 -0
- lmcache/v1/lookup_client/mooncake_lookup_client.py +87 -0
- lmcache/v1/lookup_client/record_strategies/__init__.py +77 -0
- lmcache/v1/lookup_client/record_strategies/base.py +327 -0
- lmcache/v1/lookup_client/record_strategies/file_hash.py +130 -0
- lmcache/v1/lookup_client/record_strategies/memory_bloom_filter.py +81 -0
- lmcache/v1/manager.py +539 -0
- lmcache/v1/memory_management.py +2619 -0
- lmcache/v1/metadata.py +114 -0
- lmcache/v1/mp_observability/AGENTS.override.md +21 -0
- lmcache/v1/mp_observability/README.md +204 -0
- lmcache/v1/mp_observability/config.py +340 -0
- lmcache/v1/mp_observability/event.py +100 -0
- lmcache/v1/mp_observability/event_bus.py +313 -0
- lmcache/v1/mp_observability/otel_init.py +145 -0
- lmcache/v1/mp_observability/subscribers/__init__.py +28 -0
- lmcache/v1/mp_observability/subscribers/logging/__init__.py +19 -0
- lmcache/v1/mp_observability/subscribers/logging/l1.py +56 -0
- lmcache/v1/mp_observability/subscribers/logging/l2.py +73 -0
- lmcache/v1/mp_observability/subscribers/logging/lookup_hash.py +209 -0
- lmcache/v1/mp_observability/subscribers/logging/mp_server.py +90 -0
- lmcache/v1/mp_observability/subscribers/logging/sm.py +59 -0
- lmcache/v1/mp_observability/subscribers/metrics/__init__.py +20 -0
- lmcache/v1/mp_observability/subscribers/metrics/l0_lifecycle.py +290 -0
- lmcache/v1/mp_observability/subscribers/metrics/l1.py +55 -0
- lmcache/v1/mp_observability/subscribers/metrics/l1_lifecycle.py +166 -0
- lmcache/v1/mp_observability/subscribers/metrics/l2.py +121 -0
- lmcache/v1/mp_observability/subscribers/metrics/sm.py +69 -0
- lmcache/v1/mp_observability/subscribers/tracing/__init__.py +12 -0
- lmcache/v1/mp_observability/subscribers/tracing/mp_server.py +333 -0
- lmcache/v1/mp_observability/subscribers/tracing/span_registry.py +148 -0
- lmcache/v1/mp_observability/trace/__init__.py +50 -0
- lmcache/v1/mp_observability/trace/codecs.py +255 -0
- lmcache/v1/mp_observability/trace/decorator.py +147 -0
- lmcache/v1/mp_observability/trace/format.py +132 -0
- lmcache/v1/mp_observability/trace/lifecycle.py +83 -0
- lmcache/v1/mp_observability/trace/reader.py +167 -0
- lmcache/v1/mp_observability/trace/recorder.py +300 -0
- lmcache/v1/multiprocess/__init__.py +0 -0
- lmcache/v1/multiprocess/affinity_pool.py +102 -0
- lmcache/v1/multiprocess/blend_server_v2.py +891 -0
- lmcache/v1/multiprocess/config.py +253 -0
- lmcache/v1/multiprocess/custom_types.py +281 -0
- lmcache/v1/multiprocess/futures.py +194 -0
- lmcache/v1/multiprocess/gpu_context.py +511 -0
- lmcache/v1/multiprocess/http_server.py +235 -0
- lmcache/v1/multiprocess/mp_runtime_plugin_launcher.py +130 -0
- lmcache/v1/multiprocess/mq.py +732 -0
- lmcache/v1/multiprocess/protocol.py +86 -0
- lmcache/v1/multiprocess/protocols/README.md +213 -0
- lmcache/v1/multiprocess/protocols/__init__.py +127 -0
- lmcache/v1/multiprocess/protocols/base.py +89 -0
- lmcache/v1/multiprocess/protocols/blend.py +109 -0
- lmcache/v1/multiprocess/protocols/blend_v2.py +57 -0
- lmcache/v1/multiprocess/protocols/controller.py +53 -0
- lmcache/v1/multiprocess/protocols/debug.py +34 -0
- lmcache/v1/multiprocess/protocols/engine.py +146 -0
- lmcache/v1/multiprocess/protocols/observability.py +39 -0
- lmcache/v1/multiprocess/server.py +1134 -0
- lmcache/v1/multiprocess/session.py +190 -0
- lmcache/v1/multiprocess/token_hasher.py +441 -0
- lmcache/v1/offload_server/__init__.py +17 -0
- lmcache/v1/offload_server/abstract_server.py +37 -0
- lmcache/v1/offload_server/message.py +30 -0
- lmcache/v1/offload_server/zmq_server.py +122 -0
- lmcache/v1/periodic_thread.py +579 -0
- lmcache/v1/pin_monitor.py +246 -0
- lmcache/v1/plugin/__init__.py +0 -0
- lmcache/v1/plugin/runtime_plugin_launcher.py +211 -0
- lmcache/v1/protocol.py +317 -0
- lmcache/v1/rpc/__init__.py +17 -0
- lmcache/v1/rpc/transport.py +105 -0
- lmcache/v1/rpc/zmq_transport.py +213 -0
- lmcache/v1/rpc_utils.py +165 -0
- lmcache/v1/server/__init__.py +2 -0
- lmcache/v1/server/__main__.py +170 -0
- lmcache/v1/server/storage_backend/__init__.py +21 -0
- lmcache/v1/server/storage_backend/abstract_backend.py +80 -0
- lmcache/v1/server/storage_backend/local_backend.py +75 -0
- lmcache/v1/server/utils.py +21 -0
- lmcache/v1/standalone/__init__.py +1 -0
- lmcache/v1/standalone/__main__.py +583 -0
- lmcache/v1/standalone/manager.py +80 -0
- lmcache/v1/standalone/standalone_service_factory.py +86 -0
- lmcache/v1/storage_backend/__init__.py +313 -0
- lmcache/v1/storage_backend/abstract_backend.py +445 -0
- lmcache/v1/storage_backend/audit_backend.py +233 -0
- lmcache/v1/storage_backend/batched_message_sender.py +222 -0
- lmcache/v1/storage_backend/cache_policy/__init__.py +45 -0
- lmcache/v1/storage_backend/cache_policy/base_policy.py +87 -0
- lmcache/v1/storage_backend/cache_policy/fifo.py +58 -0
- lmcache/v1/storage_backend/cache_policy/lfu.py +105 -0
- lmcache/v1/storage_backend/cache_policy/lru.py +81 -0
- lmcache/v1/storage_backend/cache_policy/mru.py +61 -0
- lmcache/v1/storage_backend/connector/__init__.py +443 -0
- lmcache/v1/storage_backend/connector/audit_adapter.py +77 -0
- lmcache/v1/storage_backend/connector/audit_connector.py +320 -0
- lmcache/v1/storage_backend/connector/base_connector.py +379 -0
- lmcache/v1/storage_backend/connector/blackhole_adapter.py +21 -0
- lmcache/v1/storage_backend/connector/blackhole_connector.py +37 -0
- lmcache/v1/storage_backend/connector/eic_adapter.py +31 -0
- lmcache/v1/storage_backend/connector/eic_connector.py +757 -0
- lmcache/v1/storage_backend/connector/external_adapter.py +79 -0
- lmcache/v1/storage_backend/connector/fs_adapter.py +51 -0
- lmcache/v1/storage_backend/connector/fs_connector.py +403 -0
- lmcache/v1/storage_backend/connector/infinistore_adapter.py +56 -0
- lmcache/v1/storage_backend/connector/infinistore_connector.py +177 -0
- lmcache/v1/storage_backend/connector/instrumented_connector.py +219 -0
- lmcache/v1/storage_backend/connector/lm_adapter.py +31 -0
- lmcache/v1/storage_backend/connector/lm_connector.py +176 -0
- lmcache/v1/storage_backend/connector/mock_adapter.py +57 -0
- lmcache/v1/storage_backend/connector/mock_connector.py +349 -0
- lmcache/v1/storage_backend/connector/mooncakestore_adapter.py +43 -0
- lmcache/v1/storage_backend/connector/mooncakestore_connector.py +614 -0
- lmcache/v1/storage_backend/connector/redis_adapter.py +181 -0
- lmcache/v1/storage_backend/connector/redis_connector.py +828 -0
- lmcache/v1/storage_backend/connector/s3_adapter.py +59 -0
- lmcache/v1/storage_backend/connector/s3_connector.py +699 -0
- lmcache/v1/storage_backend/connector/sagemaker_hyperpod_adapter.py +233 -0
- lmcache/v1/storage_backend/connector/sagemaker_hyperpod_connector.py +987 -0
- lmcache/v1/storage_backend/connector/valkey_adapter.py +114 -0
- lmcache/v1/storage_backend/connector/valkey_connector.py +627 -0
- lmcache/v1/storage_backend/gds_backend.py +1199 -0
- lmcache/v1/storage_backend/job_executor/__init__.py +0 -0
- lmcache/v1/storage_backend/job_executor/base_executor.py +34 -0
- lmcache/v1/storage_backend/job_executor/pq_executor.py +235 -0
- lmcache/v1/storage_backend/local_cpu_backend.py +810 -0
- lmcache/v1/storage_backend/local_disk_backend.py +656 -0
- lmcache/v1/storage_backend/maru_backend.py +734 -0
- lmcache/v1/storage_backend/naive_serde/__init__.py +50 -0
- lmcache/v1/storage_backend/naive_serde/cachegen_basics.py +133 -0
- lmcache/v1/storage_backend/naive_serde/cachegen_decoder.py +135 -0
- lmcache/v1/storage_backend/naive_serde/cachegen_encoder.py +83 -0
- lmcache/v1/storage_backend/naive_serde/kivi_serde.py +22 -0
- lmcache/v1/storage_backend/naive_serde/naive_serde.py +18 -0
- lmcache/v1/storage_backend/naive_serde/serde.py +37 -0
- lmcache/v1/storage_backend/native_clients/connector_client_base.py +165 -0
- lmcache/v1/storage_backend/native_clients/resp_client.py +35 -0
- lmcache/v1/storage_backend/nixl_storage_backend.py +1400 -0
- lmcache/v1/storage_backend/p2p_backend.py +788 -0
- lmcache/v1/storage_backend/path_sharder.py +117 -0
- lmcache/v1/storage_backend/pd_backend.py +646 -0
- lmcache/v1/storage_backend/plugins/dax_backend.py +1443 -0
- lmcache/v1/storage_backend/plugins/rust_raw_block_backend.py +1361 -0
- lmcache/v1/storage_backend/remote_backend.py +624 -0
- lmcache/v1/storage_backend/resp_client.py +227 -0
- lmcache/v1/storage_backend/storage_backend_listener.py +19 -0
- lmcache/v1/storage_backend/storage_manager.py +1352 -0
- lmcache/v1/system_detection.py +110 -0
- lmcache/v1/token_database.py +551 -0
- lmcache/v1/transfer_channel/__init__.py +83 -0
- lmcache/v1/transfer_channel/abstract.py +285 -0
- lmcache/v1/transfer_channel/mock_memory_channel.py +156 -0
- lmcache/v1/transfer_channel/nixl_channel.py +639 -0
- lmcache/v1/transfer_channel/py_socket_channel.py +260 -0
- lmcache/v1/transfer_channel/transfer_utils.py +63 -0
- lmcache/v1/utils/__init__.py +1 -0
- lmcache/v1/utils/bloom_filter.py +109 -0
- lmcache/v1/utils/cache_utils.py +125 -0
- lmcache_cli-0.4.5.dev0.dist-info/METADATA +185 -0
- lmcache_cli-0.4.5.dev0.dist-info/RECORD +399 -0
- lmcache_cli-0.4.5.dev0.dist-info/WHEEL +5 -0
- lmcache_cli-0.4.5.dev0.dist-info/entry_points.txt +2 -0
- lmcache_cli-0.4.5.dev0.dist-info/licenses/LICENSE +201 -0
- lmcache_cli-0.4.5.dev0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,53 @@
|
|
|
1
|
+
# SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
"""
|
|
3
|
+
Controller protocol definitions for cache management and configuration.
|
|
4
|
+
|
|
5
|
+
This module defines the protocol for:
|
|
6
|
+
- CLEAR: Clear all caches in the server
|
|
7
|
+
- GET_CHUNK_SIZE: Get the chunk size configuration from the server
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
# First Party
|
|
11
|
+
from lmcache.v1.multiprocess.protocols.base import HandlerType, ProtocolDefinition
|
|
12
|
+
|
|
13
|
+
# Define request names for this protocol group
|
|
14
|
+
REQUEST_NAMES = [
|
|
15
|
+
"CLEAR",
|
|
16
|
+
"GET_CHUNK_SIZE",
|
|
17
|
+
"PING",
|
|
18
|
+
]
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def get_protocol_definitions() -> dict[str, ProtocolDefinition]:
|
|
22
|
+
"""
|
|
23
|
+
Returns protocol definitions for controller operations.
|
|
24
|
+
|
|
25
|
+
Returns:
|
|
26
|
+
Dictionary mapping request names to their protocol definitions
|
|
27
|
+
"""
|
|
28
|
+
return {
|
|
29
|
+
# Clear all caches
|
|
30
|
+
# Payload: None
|
|
31
|
+
# Returns: None
|
|
32
|
+
"CLEAR": ProtocolDefinition(
|
|
33
|
+
payload_classes=[],
|
|
34
|
+
response_class=None,
|
|
35
|
+
handler_type=HandlerType.BLOCKING,
|
|
36
|
+
),
|
|
37
|
+
# Get chunk size configuration
|
|
38
|
+
# Payload: None
|
|
39
|
+
# Returns: int - The chunk size value
|
|
40
|
+
"GET_CHUNK_SIZE": ProtocolDefinition(
|
|
41
|
+
payload_classes=[],
|
|
42
|
+
response_class=int,
|
|
43
|
+
handler_type=HandlerType.SYNC,
|
|
44
|
+
),
|
|
45
|
+
# Ping
|
|
46
|
+
# Payload: None
|
|
47
|
+
# Returns: bool - Always True
|
|
48
|
+
"PING": ProtocolDefinition(
|
|
49
|
+
payload_classes=[],
|
|
50
|
+
response_class=bool,
|
|
51
|
+
handler_type=HandlerType.BLOCKING,
|
|
52
|
+
),
|
|
53
|
+
}
|
|
@@ -0,0 +1,34 @@
|
|
|
1
|
+
# SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
"""
|
|
3
|
+
Debug protocol definitions for testing and monitoring.
|
|
4
|
+
|
|
5
|
+
This module defines the protocol for:
|
|
6
|
+
- NOOP: No-operation command for testing connectivity and as a heartbeat
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
# First Party
|
|
10
|
+
from lmcache.v1.multiprocess.protocols.base import HandlerType, ProtocolDefinition
|
|
11
|
+
|
|
12
|
+
# Define request names for this protocol group
|
|
13
|
+
REQUEST_NAMES = [
|
|
14
|
+
"NOOP",
|
|
15
|
+
]
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def get_protocol_definitions() -> dict[str, ProtocolDefinition]:
|
|
19
|
+
"""
|
|
20
|
+
Returns protocol definitions for debug operations.
|
|
21
|
+
|
|
22
|
+
Returns:
|
|
23
|
+
Dictionary mapping request names to their protocol definitions
|
|
24
|
+
"""
|
|
25
|
+
return {
|
|
26
|
+
# No-operation (for testing/heartbeat)
|
|
27
|
+
# Payload: None
|
|
28
|
+
# Returns: str - A confirmation message
|
|
29
|
+
"NOOP": ProtocolDefinition(
|
|
30
|
+
payload_classes=[],
|
|
31
|
+
response_class=str,
|
|
32
|
+
handler_type=HandlerType.SYNC,
|
|
33
|
+
),
|
|
34
|
+
}
|
|
@@ -0,0 +1,146 @@
|
|
|
1
|
+
# SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
"""
|
|
3
|
+
Engine protocol definitions for core KV cache operations.
|
|
4
|
+
|
|
5
|
+
This module defines the protocol for:
|
|
6
|
+
- REGISTER_KV_CACHE: Register a KV cache instance with the server
|
|
7
|
+
- UNREGISTER_KV_CACHE: Unregister a KV cache instance
|
|
8
|
+
- STORE: Store KV cache blocks to the server
|
|
9
|
+
- RETRIEVE: Retrieve KV cache blocks from the server
|
|
10
|
+
- LOOKUP: Submit a prefix lookup and return a prefetch job ID
|
|
11
|
+
- QUERY_PREFETCH_STATUS: Poll a prefetch job for its result
|
|
12
|
+
- END_SESSION: End a session and clean up associated resources
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
# First Party
|
|
16
|
+
from lmcache.v1.gpu_connector.utils import LayoutHints
|
|
17
|
+
from lmcache.v1.multiprocess.custom_types import (
|
|
18
|
+
IPCCacheEngineKey,
|
|
19
|
+
KVCache,
|
|
20
|
+
)
|
|
21
|
+
from lmcache.v1.multiprocess.protocols.base import HandlerType, ProtocolDefinition
|
|
22
|
+
|
|
23
|
+
# Define request names for this protocol group
|
|
24
|
+
REQUEST_NAMES = [
|
|
25
|
+
"REGISTER_KV_CACHE",
|
|
26
|
+
"UNREGISTER_KV_CACHE",
|
|
27
|
+
"STORE",
|
|
28
|
+
"RETRIEVE",
|
|
29
|
+
"LOOKUP",
|
|
30
|
+
"QUERY_PREFETCH_STATUS",
|
|
31
|
+
"QUERY_PREFETCH_LOOKUP_HITS",
|
|
32
|
+
"FREE_LOOKUP_LOCKS",
|
|
33
|
+
"END_SESSION",
|
|
34
|
+
]
|
|
35
|
+
|
|
36
|
+
# Type alias for cache keys
|
|
37
|
+
KeyType = IPCCacheEngineKey
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def get_protocol_definitions() -> dict[str, ProtocolDefinition]:
|
|
41
|
+
"""
|
|
42
|
+
Returns protocol definitions for engine operations.
|
|
43
|
+
|
|
44
|
+
Returns:
|
|
45
|
+
Dictionary mapping request names to their protocol definitions
|
|
46
|
+
"""
|
|
47
|
+
return {
|
|
48
|
+
# Register KV Cache
|
|
49
|
+
# Payload:
|
|
50
|
+
# - instance_id: int - Unique identifier for the vLLM instance
|
|
51
|
+
# - kv_cache: KVCache - The KV cache configuration
|
|
52
|
+
# - model_name: str - Name of the model associated with the engine
|
|
53
|
+
# - world_size: int - World size of the engine
|
|
54
|
+
# - layout_hints: LayoutHints - See custom_types.LayoutHints.
|
|
55
|
+
# Returns: None
|
|
56
|
+
"REGISTER_KV_CACHE": ProtocolDefinition(
|
|
57
|
+
payload_classes=[int, KVCache, str, int, LayoutHints],
|
|
58
|
+
response_class=None,
|
|
59
|
+
handler_type=HandlerType.SYNC,
|
|
60
|
+
),
|
|
61
|
+
# Unregister KV Cache
|
|
62
|
+
# Payload:
|
|
63
|
+
# - instance_id: int - Unique identifier for the vLLM instance
|
|
64
|
+
# Returns: None
|
|
65
|
+
"UNREGISTER_KV_CACHE": ProtocolDefinition(
|
|
66
|
+
payload_classes=[int],
|
|
67
|
+
response_class=None,
|
|
68
|
+
handler_type=HandlerType.SYNC,
|
|
69
|
+
),
|
|
70
|
+
# Store KV cache blocks
|
|
71
|
+
# Payload:
|
|
72
|
+
# - key: KeyType - Cache key to store
|
|
73
|
+
# - instance_id: int - Unique identifier for the vLLM instance
|
|
74
|
+
# - gpu_block_ids: list[int] - GPU block IDs containing the data
|
|
75
|
+
# - event_ipc_handle: bytes - CUDA event IPC handle for synchronization
|
|
76
|
+
# Returns: tuple[bytes, bool] - (CUDA event handle, success flag)
|
|
77
|
+
"STORE": ProtocolDefinition(
|
|
78
|
+
payload_classes=[KeyType, int, list[int], bytes],
|
|
79
|
+
response_class=tuple[bytes, bool],
|
|
80
|
+
handler_type=HandlerType.BLOCKING,
|
|
81
|
+
),
|
|
82
|
+
# Retrieve KV cache blocks
|
|
83
|
+
# Payload:
|
|
84
|
+
# - key: KeyType - Cache key to retrieve
|
|
85
|
+
# - instance_id: int - Unique identifier for the vLLM instance
|
|
86
|
+
# - gpu_block_ids: list[int] - GPU block IDs to store retrieved data
|
|
87
|
+
# - event_ipc_handle: bytes - CUDA event IPC handle for synchronization
|
|
88
|
+
# - skip_first_n_tokens: int - Number of tokens to skip writing at the
|
|
89
|
+
# start of the retrieve range (to avoid overwriting APC-shared blocks)
|
|
90
|
+
# Returns: tuple[bytes, bool] - (CUDA event handle, success flag)
|
|
91
|
+
"RETRIEVE": ProtocolDefinition(
|
|
92
|
+
payload_classes=[KeyType, int, list[int], bytes, int],
|
|
93
|
+
response_class=tuple[bytes, bool],
|
|
94
|
+
handler_type=HandlerType.BLOCKING,
|
|
95
|
+
),
|
|
96
|
+
# Submit a prefix lookup; job is tracked server-side by request_id
|
|
97
|
+
# Payload:
|
|
98
|
+
# - key: KeyType - Cache key to look up
|
|
99
|
+
# - tp_size: int - Tensor-parallel size for
|
|
100
|
+
# MLA multi-reader locking
|
|
101
|
+
# Returns: None
|
|
102
|
+
"LOOKUP": ProtocolDefinition(
|
|
103
|
+
payload_classes=[KeyType, int],
|
|
104
|
+
response_class=None,
|
|
105
|
+
handler_type=HandlerType.BLOCKING,
|
|
106
|
+
),
|
|
107
|
+
# Query the status of a prefetch job by request_id
|
|
108
|
+
# Payload:
|
|
109
|
+
# - request_id: str - The external request ID passed in the lookup key
|
|
110
|
+
# Returns: int | None - Chunk count when done, None if still in progress
|
|
111
|
+
"QUERY_PREFETCH_STATUS": ProtocolDefinition(
|
|
112
|
+
payload_classes=[str],
|
|
113
|
+
response_class=int | None,
|
|
114
|
+
handler_type=HandlerType.BLOCKING,
|
|
115
|
+
),
|
|
116
|
+
# Query the lookup hit chunks before the prefetch is done
|
|
117
|
+
# Payload:
|
|
118
|
+
# - request_id: str - The external request ID passed in the lookup key
|
|
119
|
+
# Returns: int | None - Chunk count if lookup is done, None if still in progress
|
|
120
|
+
"QUERY_PREFETCH_LOOKUP_HITS": ProtocolDefinition(
|
|
121
|
+
payload_classes=[str],
|
|
122
|
+
response_class=int | None,
|
|
123
|
+
handler_type=HandlerType.BLOCKING,
|
|
124
|
+
),
|
|
125
|
+
# Free locks (release read locks without a full RETRIEVE)
|
|
126
|
+
# Payload:
|
|
127
|
+
# - key: KeyType - Cache key whose read locks
|
|
128
|
+
# to release
|
|
129
|
+
# - tp_size: int - Tensor-parallel size for
|
|
130
|
+
# MLA multi-reader locking
|
|
131
|
+
# Returns: None
|
|
132
|
+
"FREE_LOOKUP_LOCKS": ProtocolDefinition(
|
|
133
|
+
payload_classes=[KeyType, int],
|
|
134
|
+
response_class=None,
|
|
135
|
+
handler_type=HandlerType.BLOCKING,
|
|
136
|
+
),
|
|
137
|
+
# End session
|
|
138
|
+
# Payload:
|
|
139
|
+
# - request_id: str - Request ID of the session to end
|
|
140
|
+
# Returns: None
|
|
141
|
+
"END_SESSION": ProtocolDefinition(
|
|
142
|
+
payload_classes=[str],
|
|
143
|
+
response_class=None,
|
|
144
|
+
handler_type=HandlerType.BLOCKING,
|
|
145
|
+
),
|
|
146
|
+
}
|
|
@@ -0,0 +1,39 @@
|
|
|
1
|
+
# SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
"""
|
|
3
|
+
Observability protocol definitions.
|
|
4
|
+
|
|
5
|
+
This module defines protocols for:
|
|
6
|
+
- REPORT_BLOCK_ALLOCATION: Report vLLM GPU block allocation events
|
|
7
|
+
(fire-and-forget, no response)
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
# First Party
|
|
11
|
+
from lmcache.v1.multiprocess.custom_types import BlockAllocationRecord
|
|
12
|
+
from lmcache.v1.multiprocess.protocols.base import HandlerType, ProtocolDefinition
|
|
13
|
+
|
|
14
|
+
# Define request names for this protocol group
|
|
15
|
+
REQUEST_NAMES = [
|
|
16
|
+
"REPORT_BLOCK_ALLOCATION",
|
|
17
|
+
]
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def get_protocol_definitions() -> dict[str, ProtocolDefinition]:
|
|
21
|
+
"""
|
|
22
|
+
Returns protocol definitions for observability operations.
|
|
23
|
+
|
|
24
|
+
Returns:
|
|
25
|
+
Dictionary mapping request names to their protocol definitions
|
|
26
|
+
"""
|
|
27
|
+
return {
|
|
28
|
+
# Report vLLM block allocation
|
|
29
|
+
# Payload:
|
|
30
|
+
# - instance_id: int - scheduler instance ID
|
|
31
|
+
# - model_name: str - model name from the adapter
|
|
32
|
+
# - records: list[BlockAllocationRecord] - allocation records
|
|
33
|
+
# Returns: None (fire-and-forget)
|
|
34
|
+
"REPORT_BLOCK_ALLOCATION": ProtocolDefinition(
|
|
35
|
+
payload_classes=[int, str, list[BlockAllocationRecord]],
|
|
36
|
+
response_class=None,
|
|
37
|
+
handler_type=HandlerType.BLOCKING,
|
|
38
|
+
),
|
|
39
|
+
}
|