lmcache-cli 0.4.5.dev0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- lmcache/__init__.py +84 -0
- lmcache/_version.py +24 -0
- lmcache/cli/__init__.py +1 -0
- lmcache/cli/commands/__init__.py +34 -0
- lmcache/cli/commands/base.py +157 -0
- lmcache/cli/commands/bench/__init__.py +557 -0
- lmcache/cli/commands/bench/engine_bench/__init__.py +1 -0
- lmcache/cli/commands/bench/engine_bench/config.py +245 -0
- lmcache/cli/commands/bench/engine_bench/interactive/__init__.py +274 -0
- lmcache/cli/commands/bench/engine_bench/interactive/config.json +10 -0
- lmcache/cli/commands/bench/engine_bench/interactive/schema.py +352 -0
- lmcache/cli/commands/bench/engine_bench/interactive/state.py +327 -0
- lmcache/cli/commands/bench/engine_bench/interactive/terminal.py +291 -0
- lmcache/cli/commands/bench/engine_bench/progress.py +145 -0
- lmcache/cli/commands/bench/engine_bench/request_sender.py +232 -0
- lmcache/cli/commands/bench/engine_bench/stats.py +275 -0
- lmcache/cli/commands/bench/engine_bench/workloads/__init__.py +153 -0
- lmcache/cli/commands/bench/engine_bench/workloads/base.py +122 -0
- lmcache/cli/commands/bench/engine_bench/workloads/long_doc_permutator.py +435 -0
- lmcache/cli/commands/bench/engine_bench/workloads/long_doc_qa.py +281 -0
- lmcache/cli/commands/bench/engine_bench/workloads/multi_round_chat.py +337 -0
- lmcache/cli/commands/bench/engine_bench/workloads/random_prefill.py +178 -0
- lmcache/cli/commands/describe.py +310 -0
- lmcache/cli/commands/kvcache.py +133 -0
- lmcache/cli/commands/mock.py +75 -0
- lmcache/cli/commands/ping.py +113 -0
- lmcache/cli/commands/query/__init__.py +155 -0
- lmcache/cli/commands/query/prompt.py +134 -0
- lmcache/cli/commands/query/request.py +357 -0
- lmcache/cli/commands/server.py +99 -0
- lmcache/cli/commands/tool/__init__.py +63 -0
- lmcache/cli/commands/tool/cache_simulator.py +113 -0
- lmcache/cli/commands/trace/__init__.py +505 -0
- lmcache/cli/commands/trace/dispatch.py +249 -0
- lmcache/cli/commands/trace/driver.py +372 -0
- lmcache/cli/commands/trace/stats.py +289 -0
- lmcache/cli/documents/lmcache.txt +11 -0
- lmcache/cli/main.py +42 -0
- lmcache/cli/metrics/__init__.py +29 -0
- lmcache/cli/metrics/formatter.py +171 -0
- lmcache/cli/metrics/handler.py +94 -0
- lmcache/cli/metrics/metrics.py +161 -0
- lmcache/cli/metrics/section.py +77 -0
- lmcache/connections.py +173 -0
- lmcache/integration/__init__.py +2 -0
- lmcache/integration/base_service_factory.py +165 -0
- lmcache/integration/request_telemetry/__init__.py +1 -0
- lmcache/integration/request_telemetry/base.py +51 -0
- lmcache/integration/request_telemetry/factory.py +113 -0
- lmcache/integration/request_telemetry/fastapi.py +109 -0
- lmcache/integration/request_telemetry/noop.py +35 -0
- lmcache/integration/sglang/__init__.py +2 -0
- lmcache/integration/sglang/sglang_adapter.py +326 -0
- lmcache/integration/sglang/utils.py +39 -0
- lmcache/integration/vllm/__init__.py +1 -0
- lmcache/integration/vllm/lmcache_connector_v1.py +213 -0
- lmcache/integration/vllm/lmcache_connector_v1_085.py +150 -0
- lmcache/integration/vllm/lmcache_mp_connector_0180.py +1072 -0
- lmcache/integration/vllm/tests/test_mm_hash_utils.py +112 -0
- lmcache/integration/vllm/utils.py +433 -0
- lmcache/integration/vllm/vllm_multi_process_adapter.py +1090 -0
- lmcache/integration/vllm/vllm_service_factory.py +339 -0
- lmcache/integration/vllm/vllm_v1_adapter.py +1713 -0
- lmcache/logging.py +107 -0
- lmcache/native_storage_ops.pyi +230 -0
- lmcache/non_cuda_equivalents.py +1424 -0
- lmcache/observability.py +1958 -0
- lmcache/storage_backend/serde/__init__.py +1 -0
- lmcache/storage_backend/serde/cachegen_basics.py +210 -0
- lmcache/storage_backend/serde/cachegen_decoder.py +207 -0
- lmcache/storage_backend/serde/cachegen_encoder.py +394 -0
- lmcache/storage_backend/serde/serde.py +75 -0
- lmcache/tools/__init__.py +1 -0
- lmcache/tools/cache_simulator/README.md +392 -0
- lmcache/tools/cache_simulator/__init__.py +1 -0
- lmcache/tools/cache_simulator/docs/simulate_example.png +0 -0
- lmcache/tools/cache_simulator/docs/sweep_example.png +0 -0
- lmcache/tools/cache_simulator/gen_bench_dataset.py +360 -0
- lmcache/tools/cache_simulator/lru_cache.py +124 -0
- lmcache/tools/cache_simulator/plot_hit_rate.py +231 -0
- lmcache/tools/cache_simulator/simulator.py +795 -0
- lmcache/tools/controller_benchmark/README.md +161 -0
- lmcache/tools/controller_benchmark/__init__.py +1 -0
- lmcache/tools/controller_benchmark/__main__.py +331 -0
- lmcache/tools/controller_benchmark/benchmark.py +660 -0
- lmcache/tools/controller_benchmark/config.py +44 -0
- lmcache/tools/controller_benchmark/constants.py +10 -0
- lmcache/tools/controller_benchmark/handlers/__init__.py +46 -0
- lmcache/tools/controller_benchmark/handlers/admit.py +52 -0
- lmcache/tools/controller_benchmark/handlers/base.py +47 -0
- lmcache/tools/controller_benchmark/handlers/deregister.py +49 -0
- lmcache/tools/controller_benchmark/handlers/evict.py +52 -0
- lmcache/tools/controller_benchmark/handlers/heartbeat.py +56 -0
- lmcache/tools/controller_benchmark/handlers/p2p_lookup.py +47 -0
- lmcache/tools/controller_benchmark/handlers/register.py +56 -0
- lmcache/tools/mp_status_viewer/__init__.py +1 -0
- lmcache/tools/mp_status_viewer/__main__.py +95 -0
- lmcache/usage_context.py +417 -0
- lmcache/utils.py +665 -0
- lmcache/v1/__init__.py +2 -0
- lmcache/v1/api_server/__init__.py +2 -0
- lmcache/v1/api_server/__main__.py +537 -0
- lmcache/v1/basic_check.py +112 -0
- lmcache/v1/cache_controller/__init__.py +9 -0
- lmcache/v1/cache_controller/commands/__init__.py +15 -0
- lmcache/v1/cache_controller/commands/base.py +35 -0
- lmcache/v1/cache_controller/commands/full_sync.py +49 -0
- lmcache/v1/cache_controller/config.py +176 -0
- lmcache/v1/cache_controller/controller_manager.py +535 -0
- lmcache/v1/cache_controller/controllers/__init__.py +11 -0
- lmcache/v1/cache_controller/controllers/full_sync_tracker.py +473 -0
- lmcache/v1/cache_controller/controllers/kv_controller.py +439 -0
- lmcache/v1/cache_controller/controllers/registration_controller.py +282 -0
- lmcache/v1/cache_controller/executor.py +463 -0
- lmcache/v1/cache_controller/frontend/static/css/style.css +201 -0
- lmcache/v1/cache_controller/frontend/static/img/logo.png +0 -0
- lmcache/v1/cache_controller/frontend/static/index.html +234 -0
- lmcache/v1/cache_controller/frontend/static/js/controller_app.js +660 -0
- lmcache/v1/cache_controller/full_sync_sender.py +475 -0
- lmcache/v1/cache_controller/locks.py +149 -0
- lmcache/v1/cache_controller/message.py +828 -0
- lmcache/v1/cache_controller/observability.py +208 -0
- lmcache/v1/cache_controller/utils.py +679 -0
- lmcache/v1/cache_controller/worker.py +665 -0
- lmcache/v1/cache_engine.py +2058 -0
- lmcache/v1/cache_interface.py +19 -0
- lmcache/v1/check/__init__.py +74 -0
- lmcache/v1/check/check_mode_gen.py +86 -0
- lmcache/v1/check/check_mode_test_l2_adapter.py +284 -0
- lmcache/v1/check/check_mode_test_remote.py +155 -0
- lmcache/v1/check/check_mode_test_storage_manager.py +142 -0
- lmcache/v1/check/utils.py +571 -0
- lmcache/v1/compute/__init__.py +2 -0
- lmcache/v1/compute/attention/__init__.py +0 -0
- lmcache/v1/compute/attention/abstract.py +39 -0
- lmcache/v1/compute/attention/flash_attn.py +129 -0
- lmcache/v1/compute/attention/flash_infer_sparse.py +284 -0
- lmcache/v1/compute/attention/metadata.py +85 -0
- lmcache/v1/compute/attention/utils.py +14 -0
- lmcache/v1/compute/blend/__init__.py +7 -0
- lmcache/v1/compute/blend/blender.py +168 -0
- lmcache/v1/compute/blend/metadata.py +34 -0
- lmcache/v1/compute/blend/utils.py +63 -0
- lmcache/v1/compute/models/__init__.py +0 -0
- lmcache/v1/compute/models/base.py +141 -0
- lmcache/v1/compute/models/llama.py +9 -0
- lmcache/v1/compute/models/qwen3.py +24 -0
- lmcache/v1/compute/models/utils.py +68 -0
- lmcache/v1/compute/positional_encoding.py +199 -0
- lmcache/v1/config.py +848 -0
- lmcache/v1/config_base.py +848 -0
- lmcache/v1/distributed/api.py +248 -0
- lmcache/v1/distributed/config.py +321 -0
- lmcache/v1/distributed/error.py +64 -0
- lmcache/v1/distributed/eviction.py +192 -0
- lmcache/v1/distributed/eviction_policy/__init__.py +21 -0
- lmcache/v1/distributed/eviction_policy/factory.py +27 -0
- lmcache/v1/distributed/eviction_policy/lru.py +244 -0
- lmcache/v1/distributed/eviction_policy/noop.py +50 -0
- lmcache/v1/distributed/internal_api.py +170 -0
- lmcache/v1/distributed/l1_manager.py +835 -0
- lmcache/v1/distributed/l2_adapters/__init__.py +67 -0
- lmcache/v1/distributed/l2_adapters/base.py +360 -0
- lmcache/v1/distributed/l2_adapters/config.py +385 -0
- lmcache/v1/distributed/l2_adapters/factory.py +205 -0
- lmcache/v1/distributed/l2_adapters/fs_l2_adapter.py +747 -0
- lmcache/v1/distributed/l2_adapters/fs_native_l2_adapter.py +167 -0
- lmcache/v1/distributed/l2_adapters/mock_l2_adapter.py +516 -0
- lmcache/v1/distributed/l2_adapters/mooncake_store_l2_adapter.py +135 -0
- lmcache/v1/distributed/l2_adapters/native_connector_l2_adapter.py +468 -0
- lmcache/v1/distributed/l2_adapters/native_plugin_l2_adapter.py +199 -0
- lmcache/v1/distributed/l2_adapters/nixl_store_dynamic_l2_adapter.py +831 -0
- lmcache/v1/distributed/l2_adapters/nixl_store_l2_adapter.py +983 -0
- lmcache/v1/distributed/l2_adapters/plugin_l2_adapter.py +210 -0
- lmcache/v1/distributed/l2_adapters/resp_l2_adapter.py +176 -0
- lmcache/v1/distributed/memory_manager.py +179 -0
- lmcache/v1/distributed/storage_controller.py +39 -0
- lmcache/v1/distributed/storage_controllers/__init__.py +43 -0
- lmcache/v1/distributed/storage_controllers/eviction_controller.py +242 -0
- lmcache/v1/distributed/storage_controllers/prefetch_controller.py +830 -0
- lmcache/v1/distributed/storage_controllers/prefetch_policy.py +193 -0
- lmcache/v1/distributed/storage_controllers/store_controller.py +452 -0
- lmcache/v1/distributed/storage_controllers/store_policy.py +213 -0
- lmcache/v1/distributed/storage_manager.py +532 -0
- lmcache/v1/event_manager.py +145 -0
- lmcache/v1/exceptions/__init__.py +16 -0
- lmcache/v1/gpu_connector/__init__.py +126 -0
- lmcache/v1/gpu_connector/gpu_connectors.py +1906 -0
- lmcache/v1/gpu_connector/gpu_ops.py +85 -0
- lmcache/v1/gpu_connector/hpu_connector.py +326 -0
- lmcache/v1/gpu_connector/mock_gpu_connector.py +67 -0
- lmcache/v1/gpu_connector/utils.py +890 -0
- lmcache/v1/gpu_connector/xpu_connectors.py +916 -0
- lmcache/v1/health_monitor/__init__.py +1 -0
- lmcache/v1/health_monitor/base.py +587 -0
- lmcache/v1/health_monitor/checks/__init__.py +1 -0
- lmcache/v1/health_monitor/checks/remote_backend_check.py +304 -0
- lmcache/v1/health_monitor/constants.py +36 -0
- lmcache/v1/internal_api_server/__init__.py +0 -0
- lmcache/v1/internal_api_server/api_registry.py +59 -0
- lmcache/v1/internal_api_server/api_server.py +120 -0
- lmcache/v1/internal_api_server/common/__init__.py +1 -0
- lmcache/v1/internal_api_server/common/env_api.py +22 -0
- lmcache/v1/internal_api_server/common/loglevel_api.py +57 -0
- lmcache/v1/internal_api_server/common/metrics_api.py +29 -0
- lmcache/v1/internal_api_server/common/periodic_thread_api.py +138 -0
- lmcache/v1/internal_api_server/common/run_script_api.py +73 -0
- lmcache/v1/internal_api_server/common/thread_api.py +63 -0
- lmcache/v1/internal_api_server/controller/__init__.py +1 -0
- lmcache/v1/internal_api_server/controller/key_stats_api.py +81 -0
- lmcache/v1/internal_api_server/controller/worker_info_api.py +136 -0
- lmcache/v1/internal_api_server/utils.py +43 -0
- lmcache/v1/internal_api_server/vllm/__init__.py +1 -0
- lmcache/v1/internal_api_server/vllm/backend_api.py +221 -0
- lmcache/v1/internal_api_server/vllm/bypass_api.py +204 -0
- lmcache/v1/internal_api_server/vllm/cache_api.py +895 -0
- lmcache/v1/internal_api_server/vllm/chunk_statistics_api.py +141 -0
- lmcache/v1/internal_api_server/vllm/conf_api.py +147 -0
- lmcache/v1/internal_api_server/vllm/freeze_api.py +172 -0
- lmcache/v1/internal_api_server/vllm/hot_cache_api.py +184 -0
- lmcache/v1/internal_api_server/vllm/inference_api.py +65 -0
- lmcache/v1/internal_api_server/vllm/load_fs_chunks_api.py +320 -0
- lmcache/v1/internal_api_server/vllm/lookup_api.py +145 -0
- lmcache/v1/internal_api_server/vllm/version_api.py +25 -0
- lmcache/v1/kv_layer_groups.py +267 -0
- lmcache/v1/lazy_memory_allocator.py +284 -0
- lmcache/v1/lookup_client/__init__.py +25 -0
- lmcache/v1/lookup_client/abstract_client.py +77 -0
- lmcache/v1/lookup_client/async_lookup_message.py +50 -0
- lmcache/v1/lookup_client/chunk_statistics_lookup_client.py +200 -0
- lmcache/v1/lookup_client/factory.py +251 -0
- lmcache/v1/lookup_client/hit_limit_lookup_client.py +86 -0
- lmcache/v1/lookup_client/lmcache_async_lookup_client.py +407 -0
- lmcache/v1/lookup_client/lmcache_lookup_client.py +285 -0
- lmcache/v1/lookup_client/lmcache_lookup_client_bypass.py +99 -0
- lmcache/v1/lookup_client/mooncake_lookup_client.py +87 -0
- lmcache/v1/lookup_client/record_strategies/__init__.py +77 -0
- lmcache/v1/lookup_client/record_strategies/base.py +327 -0
- lmcache/v1/lookup_client/record_strategies/file_hash.py +130 -0
- lmcache/v1/lookup_client/record_strategies/memory_bloom_filter.py +81 -0
- lmcache/v1/manager.py +539 -0
- lmcache/v1/memory_management.py +2619 -0
- lmcache/v1/metadata.py +114 -0
- lmcache/v1/mp_observability/AGENTS.override.md +21 -0
- lmcache/v1/mp_observability/README.md +204 -0
- lmcache/v1/mp_observability/config.py +340 -0
- lmcache/v1/mp_observability/event.py +100 -0
- lmcache/v1/mp_observability/event_bus.py +313 -0
- lmcache/v1/mp_observability/otel_init.py +145 -0
- lmcache/v1/mp_observability/subscribers/__init__.py +28 -0
- lmcache/v1/mp_observability/subscribers/logging/__init__.py +19 -0
- lmcache/v1/mp_observability/subscribers/logging/l1.py +56 -0
- lmcache/v1/mp_observability/subscribers/logging/l2.py +73 -0
- lmcache/v1/mp_observability/subscribers/logging/lookup_hash.py +209 -0
- lmcache/v1/mp_observability/subscribers/logging/mp_server.py +90 -0
- lmcache/v1/mp_observability/subscribers/logging/sm.py +59 -0
- lmcache/v1/mp_observability/subscribers/metrics/__init__.py +20 -0
- lmcache/v1/mp_observability/subscribers/metrics/l0_lifecycle.py +290 -0
- lmcache/v1/mp_observability/subscribers/metrics/l1.py +55 -0
- lmcache/v1/mp_observability/subscribers/metrics/l1_lifecycle.py +166 -0
- lmcache/v1/mp_observability/subscribers/metrics/l2.py +121 -0
- lmcache/v1/mp_observability/subscribers/metrics/sm.py +69 -0
- lmcache/v1/mp_observability/subscribers/tracing/__init__.py +12 -0
- lmcache/v1/mp_observability/subscribers/tracing/mp_server.py +333 -0
- lmcache/v1/mp_observability/subscribers/tracing/span_registry.py +148 -0
- lmcache/v1/mp_observability/trace/__init__.py +50 -0
- lmcache/v1/mp_observability/trace/codecs.py +255 -0
- lmcache/v1/mp_observability/trace/decorator.py +147 -0
- lmcache/v1/mp_observability/trace/format.py +132 -0
- lmcache/v1/mp_observability/trace/lifecycle.py +83 -0
- lmcache/v1/mp_observability/trace/reader.py +167 -0
- lmcache/v1/mp_observability/trace/recorder.py +300 -0
- lmcache/v1/multiprocess/__init__.py +0 -0
- lmcache/v1/multiprocess/affinity_pool.py +102 -0
- lmcache/v1/multiprocess/blend_server_v2.py +891 -0
- lmcache/v1/multiprocess/config.py +253 -0
- lmcache/v1/multiprocess/custom_types.py +281 -0
- lmcache/v1/multiprocess/futures.py +194 -0
- lmcache/v1/multiprocess/gpu_context.py +511 -0
- lmcache/v1/multiprocess/http_server.py +235 -0
- lmcache/v1/multiprocess/mp_runtime_plugin_launcher.py +130 -0
- lmcache/v1/multiprocess/mq.py +732 -0
- lmcache/v1/multiprocess/protocol.py +86 -0
- lmcache/v1/multiprocess/protocols/README.md +213 -0
- lmcache/v1/multiprocess/protocols/__init__.py +127 -0
- lmcache/v1/multiprocess/protocols/base.py +89 -0
- lmcache/v1/multiprocess/protocols/blend.py +109 -0
- lmcache/v1/multiprocess/protocols/blend_v2.py +57 -0
- lmcache/v1/multiprocess/protocols/controller.py +53 -0
- lmcache/v1/multiprocess/protocols/debug.py +34 -0
- lmcache/v1/multiprocess/protocols/engine.py +146 -0
- lmcache/v1/multiprocess/protocols/observability.py +39 -0
- lmcache/v1/multiprocess/server.py +1134 -0
- lmcache/v1/multiprocess/session.py +190 -0
- lmcache/v1/multiprocess/token_hasher.py +441 -0
- lmcache/v1/offload_server/__init__.py +17 -0
- lmcache/v1/offload_server/abstract_server.py +37 -0
- lmcache/v1/offload_server/message.py +30 -0
- lmcache/v1/offload_server/zmq_server.py +122 -0
- lmcache/v1/periodic_thread.py +579 -0
- lmcache/v1/pin_monitor.py +246 -0
- lmcache/v1/plugin/__init__.py +0 -0
- lmcache/v1/plugin/runtime_plugin_launcher.py +211 -0
- lmcache/v1/protocol.py +317 -0
- lmcache/v1/rpc/__init__.py +17 -0
- lmcache/v1/rpc/transport.py +105 -0
- lmcache/v1/rpc/zmq_transport.py +213 -0
- lmcache/v1/rpc_utils.py +165 -0
- lmcache/v1/server/__init__.py +2 -0
- lmcache/v1/server/__main__.py +170 -0
- lmcache/v1/server/storage_backend/__init__.py +21 -0
- lmcache/v1/server/storage_backend/abstract_backend.py +80 -0
- lmcache/v1/server/storage_backend/local_backend.py +75 -0
- lmcache/v1/server/utils.py +21 -0
- lmcache/v1/standalone/__init__.py +1 -0
- lmcache/v1/standalone/__main__.py +583 -0
- lmcache/v1/standalone/manager.py +80 -0
- lmcache/v1/standalone/standalone_service_factory.py +86 -0
- lmcache/v1/storage_backend/__init__.py +313 -0
- lmcache/v1/storage_backend/abstract_backend.py +445 -0
- lmcache/v1/storage_backend/audit_backend.py +233 -0
- lmcache/v1/storage_backend/batched_message_sender.py +222 -0
- lmcache/v1/storage_backend/cache_policy/__init__.py +45 -0
- lmcache/v1/storage_backend/cache_policy/base_policy.py +87 -0
- lmcache/v1/storage_backend/cache_policy/fifo.py +58 -0
- lmcache/v1/storage_backend/cache_policy/lfu.py +105 -0
- lmcache/v1/storage_backend/cache_policy/lru.py +81 -0
- lmcache/v1/storage_backend/cache_policy/mru.py +61 -0
- lmcache/v1/storage_backend/connector/__init__.py +443 -0
- lmcache/v1/storage_backend/connector/audit_adapter.py +77 -0
- lmcache/v1/storage_backend/connector/audit_connector.py +320 -0
- lmcache/v1/storage_backend/connector/base_connector.py +379 -0
- lmcache/v1/storage_backend/connector/blackhole_adapter.py +21 -0
- lmcache/v1/storage_backend/connector/blackhole_connector.py +37 -0
- lmcache/v1/storage_backend/connector/eic_adapter.py +31 -0
- lmcache/v1/storage_backend/connector/eic_connector.py +757 -0
- lmcache/v1/storage_backend/connector/external_adapter.py +79 -0
- lmcache/v1/storage_backend/connector/fs_adapter.py +51 -0
- lmcache/v1/storage_backend/connector/fs_connector.py +403 -0
- lmcache/v1/storage_backend/connector/infinistore_adapter.py +56 -0
- lmcache/v1/storage_backend/connector/infinistore_connector.py +177 -0
- lmcache/v1/storage_backend/connector/instrumented_connector.py +219 -0
- lmcache/v1/storage_backend/connector/lm_adapter.py +31 -0
- lmcache/v1/storage_backend/connector/lm_connector.py +176 -0
- lmcache/v1/storage_backend/connector/mock_adapter.py +57 -0
- lmcache/v1/storage_backend/connector/mock_connector.py +349 -0
- lmcache/v1/storage_backend/connector/mooncakestore_adapter.py +43 -0
- lmcache/v1/storage_backend/connector/mooncakestore_connector.py +614 -0
- lmcache/v1/storage_backend/connector/redis_adapter.py +181 -0
- lmcache/v1/storage_backend/connector/redis_connector.py +828 -0
- lmcache/v1/storage_backend/connector/s3_adapter.py +59 -0
- lmcache/v1/storage_backend/connector/s3_connector.py +699 -0
- lmcache/v1/storage_backend/connector/sagemaker_hyperpod_adapter.py +233 -0
- lmcache/v1/storage_backend/connector/sagemaker_hyperpod_connector.py +987 -0
- lmcache/v1/storage_backend/connector/valkey_adapter.py +114 -0
- lmcache/v1/storage_backend/connector/valkey_connector.py +627 -0
- lmcache/v1/storage_backend/gds_backend.py +1199 -0
- lmcache/v1/storage_backend/job_executor/__init__.py +0 -0
- lmcache/v1/storage_backend/job_executor/base_executor.py +34 -0
- lmcache/v1/storage_backend/job_executor/pq_executor.py +235 -0
- lmcache/v1/storage_backend/local_cpu_backend.py +810 -0
- lmcache/v1/storage_backend/local_disk_backend.py +656 -0
- lmcache/v1/storage_backend/maru_backend.py +734 -0
- lmcache/v1/storage_backend/naive_serde/__init__.py +50 -0
- lmcache/v1/storage_backend/naive_serde/cachegen_basics.py +133 -0
- lmcache/v1/storage_backend/naive_serde/cachegen_decoder.py +135 -0
- lmcache/v1/storage_backend/naive_serde/cachegen_encoder.py +83 -0
- lmcache/v1/storage_backend/naive_serde/kivi_serde.py +22 -0
- lmcache/v1/storage_backend/naive_serde/naive_serde.py +18 -0
- lmcache/v1/storage_backend/naive_serde/serde.py +37 -0
- lmcache/v1/storage_backend/native_clients/connector_client_base.py +165 -0
- lmcache/v1/storage_backend/native_clients/resp_client.py +35 -0
- lmcache/v1/storage_backend/nixl_storage_backend.py +1400 -0
- lmcache/v1/storage_backend/p2p_backend.py +788 -0
- lmcache/v1/storage_backend/path_sharder.py +117 -0
- lmcache/v1/storage_backend/pd_backend.py +646 -0
- lmcache/v1/storage_backend/plugins/dax_backend.py +1443 -0
- lmcache/v1/storage_backend/plugins/rust_raw_block_backend.py +1361 -0
- lmcache/v1/storage_backend/remote_backend.py +624 -0
- lmcache/v1/storage_backend/resp_client.py +227 -0
- lmcache/v1/storage_backend/storage_backend_listener.py +19 -0
- lmcache/v1/storage_backend/storage_manager.py +1352 -0
- lmcache/v1/system_detection.py +110 -0
- lmcache/v1/token_database.py +551 -0
- lmcache/v1/transfer_channel/__init__.py +83 -0
- lmcache/v1/transfer_channel/abstract.py +285 -0
- lmcache/v1/transfer_channel/mock_memory_channel.py +156 -0
- lmcache/v1/transfer_channel/nixl_channel.py +639 -0
- lmcache/v1/transfer_channel/py_socket_channel.py +260 -0
- lmcache/v1/transfer_channel/transfer_utils.py +63 -0
- lmcache/v1/utils/__init__.py +1 -0
- lmcache/v1/utils/bloom_filter.py +109 -0
- lmcache/v1/utils/cache_utils.py +125 -0
- lmcache_cli-0.4.5.dev0.dist-info/METADATA +185 -0
- lmcache_cli-0.4.5.dev0.dist-info/RECORD +399 -0
- lmcache_cli-0.4.5.dev0.dist-info/WHEEL +5 -0
- lmcache_cli-0.4.5.dev0.dist-info/entry_points.txt +2 -0
- lmcache_cli-0.4.5.dev0.dist-info/licenses/LICENSE +201 -0
- lmcache_cli-0.4.5.dev0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,349 @@
|
|
|
1
|
+
# SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
# Standard
|
|
3
|
+
|
|
4
|
+
# Standard
|
|
5
|
+
from collections import OrderedDict
|
|
6
|
+
from dataclasses import dataclass
|
|
7
|
+
from enum import IntEnum, auto
|
|
8
|
+
from typing import List, Optional
|
|
9
|
+
import asyncio
|
|
10
|
+
|
|
11
|
+
# First Party
|
|
12
|
+
from lmcache.logging import init_logger
|
|
13
|
+
from lmcache.utils import CacheEngineKey
|
|
14
|
+
from lmcache.v1.memory_management import MemoryObj, MemoryObjMetadata, TensorMemoryObj
|
|
15
|
+
from lmcache.v1.storage_backend.connector.base_connector import RemoteConnector
|
|
16
|
+
from lmcache.v1.storage_backend.job_executor.pq_executor import AsyncPQExecutor
|
|
17
|
+
from lmcache.v1.storage_backend.local_cpu_backend import LocalCPUBackend
|
|
18
|
+
|
|
19
|
+
logger = init_logger(__name__)
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
class Priorities(IntEnum):
|
|
23
|
+
PEEK = auto()
|
|
24
|
+
PREFETCH = auto()
|
|
25
|
+
GET = auto()
|
|
26
|
+
PUT = auto()
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
@dataclass
|
|
30
|
+
class MockMemoryObj:
|
|
31
|
+
metadata: MemoryObjMetadata
|
|
32
|
+
num_bytes: int
|
|
33
|
+
|
|
34
|
+
@staticmethod
|
|
35
|
+
def from_tensor_memory_obj(tensor_memory_obj: MemoryObj) -> "MockMemoryObj":
|
|
36
|
+
assert isinstance(tensor_memory_obj, TensorMemoryObj)
|
|
37
|
+
return MockMemoryObj(
|
|
38
|
+
metadata=tensor_memory_obj.metadata,
|
|
39
|
+
num_bytes=len(tensor_memory_obj.byte_array),
|
|
40
|
+
)
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
class AsyncLRU:
|
|
44
|
+
"""
|
|
45
|
+
the async lock protects against race conditions while mimicking synchronization
|
|
46
|
+
being done on a remote server (async client)
|
|
47
|
+
"""
|
|
48
|
+
|
|
49
|
+
def __init__(self, capacity: int):
|
|
50
|
+
self.lock = asyncio.Lock()
|
|
51
|
+
# current size in bytes
|
|
52
|
+
self.size = 0
|
|
53
|
+
self.capacity = capacity * 1024**3
|
|
54
|
+
self.dict: OrderedDict[CacheEngineKey, MockMemoryObj] = OrderedDict()
|
|
55
|
+
|
|
56
|
+
async def exists(self, key: CacheEngineKey):
|
|
57
|
+
async with self.lock:
|
|
58
|
+
if key in self.dict:
|
|
59
|
+
self.dict.move_to_end(key)
|
|
60
|
+
return True
|
|
61
|
+
return False
|
|
62
|
+
|
|
63
|
+
async def get(self, key: CacheEngineKey) -> Optional[MockMemoryObj]:
|
|
64
|
+
async with self.lock:
|
|
65
|
+
if key not in self.dict:
|
|
66
|
+
return None
|
|
67
|
+
self.dict.move_to_end(key)
|
|
68
|
+
return self.dict[key]
|
|
69
|
+
|
|
70
|
+
async def batched_get(
|
|
71
|
+
self, keys: List[CacheEngineKey]
|
|
72
|
+
) -> List[Optional[MockMemoryObj]]:
|
|
73
|
+
async with self.lock:
|
|
74
|
+
return [self.dict.get(key, None) for key in keys]
|
|
75
|
+
|
|
76
|
+
async def put(self, key: CacheEngineKey, mock_obj: MockMemoryObj):
|
|
77
|
+
async with self.lock:
|
|
78
|
+
alloc_size = mock_obj.num_bytes
|
|
79
|
+
if alloc_size > self.capacity:
|
|
80
|
+
raise ValueError(
|
|
81
|
+
f"Allocation size {alloc_size} is",
|
|
82
|
+
" greater than capacity {self.capacity}",
|
|
83
|
+
)
|
|
84
|
+
if key in self.dict:
|
|
85
|
+
self.dict.move_to_end(key)
|
|
86
|
+
return None
|
|
87
|
+
self.dict[key] = mock_obj
|
|
88
|
+
while self.size + alloc_size > self.capacity:
|
|
89
|
+
_, mock_obj = self.dict.popitem(last=False)
|
|
90
|
+
self.size -= mock_obj.num_bytes
|
|
91
|
+
self.size += alloc_size
|
|
92
|
+
|
|
93
|
+
async def list(self) -> List[CacheEngineKey]:
|
|
94
|
+
async with self.lock:
|
|
95
|
+
return list(self.dict.keys())
|
|
96
|
+
|
|
97
|
+
async def close(self):
|
|
98
|
+
async with self.lock:
|
|
99
|
+
self.dict.clear()
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
class PressureManager:
|
|
103
|
+
"""
|
|
104
|
+
Manage I/O pressure of the mock connector
|
|
105
|
+
Assumption: Read and Write throughput are independent
|
|
106
|
+
Locks control overall backend throughput, not per-operation throughput
|
|
107
|
+
"""
|
|
108
|
+
|
|
109
|
+
def __init__(
|
|
110
|
+
self,
|
|
111
|
+
peeking_latency: float,
|
|
112
|
+
read_throughput: float,
|
|
113
|
+
write_throughput: float,
|
|
114
|
+
):
|
|
115
|
+
# seconds
|
|
116
|
+
self.peeking_latency = peeking_latency / 1000
|
|
117
|
+
# seconds / byte
|
|
118
|
+
self.read_latency_per_byte = (1 / read_throughput) / 1024**3
|
|
119
|
+
self.write_latency_per_byte = (1 / write_throughput) / 1024**3
|
|
120
|
+
|
|
121
|
+
self.read_lock = asyncio.Lock()
|
|
122
|
+
self.write_lock = asyncio.Lock()
|
|
123
|
+
|
|
124
|
+
async def on_exists(self):
|
|
125
|
+
# exists latency will delay everyone
|
|
126
|
+
logger.debug(f"waiting {self.peeking_latency} seconds to peek")
|
|
127
|
+
if self.peeking_latency > 0:
|
|
128
|
+
await asyncio.sleep(self.peeking_latency)
|
|
129
|
+
|
|
130
|
+
async def on_put(self, mock_obj: MockMemoryObj):
|
|
131
|
+
total_wait_time = self.write_latency_per_byte * mock_obj.num_bytes
|
|
132
|
+
logger.debug(
|
|
133
|
+
f"waiting {total_wait_time} seconds to put {mock_obj.num_bytes} bytes"
|
|
134
|
+
)
|
|
135
|
+
async with self.write_lock:
|
|
136
|
+
await asyncio.sleep(total_wait_time)
|
|
137
|
+
|
|
138
|
+
async def on_get(self, mock_obj: MockMemoryObj):
|
|
139
|
+
total_wait_time = self.read_latency_per_byte * mock_obj.num_bytes
|
|
140
|
+
logger.debug(
|
|
141
|
+
f"waiting {total_wait_time} seconds to get {mock_obj.num_bytes} bytes"
|
|
142
|
+
)
|
|
143
|
+
async with self.read_lock:
|
|
144
|
+
await asyncio.sleep(total_wait_time)
|
|
145
|
+
|
|
146
|
+
async def on_batched_get(self, mock_objs: List[Optional[MockMemoryObj]]):
|
|
147
|
+
total_bytes = 0
|
|
148
|
+
for mock_obj in mock_objs:
|
|
149
|
+
if mock_obj is None:
|
|
150
|
+
continue
|
|
151
|
+
total_bytes += mock_obj.num_bytes
|
|
152
|
+
total_wait_time = self.read_latency_per_byte * total_bytes
|
|
153
|
+
logger.debug(f"waiting {total_wait_time} seconds to get {total_bytes} bytes")
|
|
154
|
+
async with self.read_lock:
|
|
155
|
+
await asyncio.sleep(total_wait_time)
|
|
156
|
+
|
|
157
|
+
|
|
158
|
+
class MockConnector(RemoteConnector):
|
|
159
|
+
"""
|
|
160
|
+
A CPU "remote" backend that doesn't actually go through any network/DB layers and
|
|
161
|
+
let's you manually set R/W throughput and peek latency
|
|
162
|
+
"""
|
|
163
|
+
|
|
164
|
+
def __init__(
|
|
165
|
+
self,
|
|
166
|
+
url: str,
|
|
167
|
+
loop: asyncio.AbstractEventLoop,
|
|
168
|
+
local_cpu_backend: LocalCPUBackend,
|
|
169
|
+
capacity: int,
|
|
170
|
+
peeking_latency: float = 1.0,
|
|
171
|
+
read_throughput: float = 2.0,
|
|
172
|
+
write_throughput: float = 2.0,
|
|
173
|
+
):
|
|
174
|
+
"""
|
|
175
|
+
peeking_latency: latency for peeking a key (ms)
|
|
176
|
+
capacity: capacity in GB
|
|
177
|
+
read_throughput: GB/s for reading
|
|
178
|
+
write_throughput: GB/s for writing
|
|
179
|
+
"""
|
|
180
|
+
# initialize base class, which includes some common attributes
|
|
181
|
+
super().__init__(local_cpu_backend.config, local_cpu_backend.metadata)
|
|
182
|
+
|
|
183
|
+
self.loop = loop
|
|
184
|
+
self.local_cpu_backend = local_cpu_backend
|
|
185
|
+
|
|
186
|
+
self.lru_store = AsyncLRU(capacity)
|
|
187
|
+
|
|
188
|
+
self.pressure_manager = PressureManager(
|
|
189
|
+
peeking_latency=peeking_latency,
|
|
190
|
+
read_throughput=read_throughput,
|
|
191
|
+
write_throughput=write_throughput,
|
|
192
|
+
)
|
|
193
|
+
|
|
194
|
+
# only for the __repr__ string
|
|
195
|
+
self.capacity = capacity
|
|
196
|
+
self.peeking_latency = peeking_latency
|
|
197
|
+
self.read_throughput = read_throughput
|
|
198
|
+
self.write_throughput = write_throughput
|
|
199
|
+
|
|
200
|
+
# only for the async loading codepath
|
|
201
|
+
# MockConnector is naturally async so we don't need to use a thread pool
|
|
202
|
+
# for avoiding R-W interference
|
|
203
|
+
self.pq_executor = AsyncPQExecutor(loop)
|
|
204
|
+
|
|
205
|
+
async def _exists(self, key: CacheEngineKey) -> bool:
|
|
206
|
+
await self.pressure_manager.on_exists()
|
|
207
|
+
return await self.lru_store.exists(key)
|
|
208
|
+
|
|
209
|
+
async def exists(self, key: CacheEngineKey) -> bool:
|
|
210
|
+
return await self.pq_executor.submit_job(
|
|
211
|
+
self._exists, key=key, priority=Priorities.PEEK
|
|
212
|
+
)
|
|
213
|
+
|
|
214
|
+
def exists_sync(self, key: CacheEngineKey) -> bool:
|
|
215
|
+
"""Synchronous exists check without async lock (for testing purposes)"""
|
|
216
|
+
return key in self.lru_store.dict
|
|
217
|
+
|
|
218
|
+
async def _get(self, key: CacheEngineKey) -> Optional[MemoryObj]:
|
|
219
|
+
mock_obj = await self.lru_store.get(key)
|
|
220
|
+
if mock_obj is None:
|
|
221
|
+
return None
|
|
222
|
+
await self.pressure_manager.on_get(mock_obj)
|
|
223
|
+
metadata = mock_obj.metadata
|
|
224
|
+
memory_obj = self.local_cpu_backend.allocate(
|
|
225
|
+
metadata.shape,
|
|
226
|
+
metadata.dtype,
|
|
227
|
+
metadata.fmt,
|
|
228
|
+
)
|
|
229
|
+
|
|
230
|
+
if memory_obj is None:
|
|
231
|
+
logger.warning("Failed to allocate memory during remote receive")
|
|
232
|
+
return None
|
|
233
|
+
return memory_obj
|
|
234
|
+
|
|
235
|
+
async def get(self, key: CacheEngineKey) -> Optional[MemoryObj]:
|
|
236
|
+
return await self.pq_executor.submit_job(
|
|
237
|
+
self._get, key=key, priority=Priorities.GET
|
|
238
|
+
)
|
|
239
|
+
|
|
240
|
+
async def _put(self, key: CacheEngineKey, memory_obj: MemoryObj):
|
|
241
|
+
mock_obj = MockMemoryObj.from_tensor_memory_obj(memory_obj)
|
|
242
|
+
await self.lru_store.put(key, mock_obj)
|
|
243
|
+
await self.pressure_manager.on_put(mock_obj)
|
|
244
|
+
|
|
245
|
+
async def put(self, key: CacheEngineKey, memory_obj: MemoryObj):
|
|
246
|
+
await self.pq_executor.submit_job(
|
|
247
|
+
self._put, key=key, memory_obj=memory_obj, priority=Priorities.PUT
|
|
248
|
+
)
|
|
249
|
+
|
|
250
|
+
async def list(self) -> List[str]:
|
|
251
|
+
keys = await self.lru_store.list()
|
|
252
|
+
return [k.to_string() for k in keys]
|
|
253
|
+
|
|
254
|
+
def support_batched_get(self) -> bool:
|
|
255
|
+
return True
|
|
256
|
+
|
|
257
|
+
async def _batched_get(
|
|
258
|
+
self, keys: List[CacheEngineKey]
|
|
259
|
+
) -> List[Optional[MemoryObj]]:
|
|
260
|
+
mock_objs = await self.lru_store.batched_get(keys)
|
|
261
|
+
await self.pressure_manager.on_batched_get(mock_objs)
|
|
262
|
+
memory_objs = []
|
|
263
|
+
|
|
264
|
+
for i, mock_obj in enumerate(mock_objs):
|
|
265
|
+
if mock_obj is None:
|
|
266
|
+
logger.warning(
|
|
267
|
+
f"Mock object is None on {i}",
|
|
268
|
+
f" out of {len(mock_objs)} objects",
|
|
269
|
+
)
|
|
270
|
+
break
|
|
271
|
+
metadata = mock_obj.metadata
|
|
272
|
+
memory_obj = self.local_cpu_backend.allocate(
|
|
273
|
+
metadata.shape,
|
|
274
|
+
metadata.dtype,
|
|
275
|
+
metadata.fmt,
|
|
276
|
+
)
|
|
277
|
+
if memory_obj is None:
|
|
278
|
+
logger.warning(
|
|
279
|
+
"Failed to allocate memory even with",
|
|
280
|
+
f" busy loop on {i} out of {len(mock_objs)} objects",
|
|
281
|
+
)
|
|
282
|
+
break
|
|
283
|
+
memory_objs.append(memory_obj)
|
|
284
|
+
|
|
285
|
+
return memory_objs
|
|
286
|
+
|
|
287
|
+
async def batched_get(
|
|
288
|
+
self, keys: List[CacheEngineKey]
|
|
289
|
+
) -> List[Optional[MemoryObj]]:
|
|
290
|
+
return await self.pq_executor.submit_job(
|
|
291
|
+
self._batched_get, keys=keys, priority=Priorities.GET
|
|
292
|
+
)
|
|
293
|
+
|
|
294
|
+
def support_batched_async_contains(self) -> bool:
|
|
295
|
+
return True
|
|
296
|
+
|
|
297
|
+
async def _batched_async_contains(
|
|
298
|
+
self,
|
|
299
|
+
lookup_id: str,
|
|
300
|
+
keys: List[CacheEngineKey],
|
|
301
|
+
pin: bool = False,
|
|
302
|
+
) -> int:
|
|
303
|
+
num_hit_counts = 0
|
|
304
|
+
for key in keys:
|
|
305
|
+
await self.pressure_manager.on_exists()
|
|
306
|
+
if not await self.lru_store.exists(key):
|
|
307
|
+
return num_hit_counts
|
|
308
|
+
num_hit_counts += 1
|
|
309
|
+
return num_hit_counts
|
|
310
|
+
|
|
311
|
+
async def batched_async_contains(
|
|
312
|
+
self,
|
|
313
|
+
lookup_id: str,
|
|
314
|
+
keys: List[CacheEngineKey],
|
|
315
|
+
pin: bool = False,
|
|
316
|
+
) -> int:
|
|
317
|
+
return await self.pq_executor.submit_job(
|
|
318
|
+
self._batched_async_contains,
|
|
319
|
+
lookup_id=lookup_id,
|
|
320
|
+
keys=keys,
|
|
321
|
+
pin=pin,
|
|
322
|
+
priority=Priorities.PEEK,
|
|
323
|
+
)
|
|
324
|
+
|
|
325
|
+
def support_batched_get_non_blocking(self) -> bool:
|
|
326
|
+
return True
|
|
327
|
+
|
|
328
|
+
async def batched_get_non_blocking(
|
|
329
|
+
self,
|
|
330
|
+
lookup_id: str,
|
|
331
|
+
keys: List[CacheEngineKey],
|
|
332
|
+
) -> List[MemoryObj]:
|
|
333
|
+
# batched get is already async and the non-blocking element is handled
|
|
334
|
+
# in the StorageManager
|
|
335
|
+
return await self.pq_executor.submit_job(
|
|
336
|
+
self._batched_get, keys=keys, priority=Priorities.PREFETCH
|
|
337
|
+
)
|
|
338
|
+
|
|
339
|
+
async def close(self):
|
|
340
|
+
await self.lru_store.close()
|
|
341
|
+
await self.pq_executor.shutdown(wait=True)
|
|
342
|
+
|
|
343
|
+
def __repr__(self) -> str:
|
|
344
|
+
return (
|
|
345
|
+
f"MockConnector(capacity={self.capacity}GB, "
|
|
346
|
+
f"peeking_latency={self.peeking_latency}ms, "
|
|
347
|
+
f"read_throughput={self.read_throughput}GB/s, "
|
|
348
|
+
f"write_throughput={self.write_throughput}GB/s)"
|
|
349
|
+
)
|
|
@@ -0,0 +1,43 @@
|
|
|
1
|
+
# SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
# First Party
|
|
3
|
+
from lmcache.logging import init_logger
|
|
4
|
+
from lmcache.v1.storage_backend.connector import (
|
|
5
|
+
ConnectorAdapter,
|
|
6
|
+
ConnectorContext,
|
|
7
|
+
extract_plugin_type,
|
|
8
|
+
)
|
|
9
|
+
from lmcache.v1.storage_backend.connector.base_connector import (
|
|
10
|
+
RemoteConnector,
|
|
11
|
+
)
|
|
12
|
+
|
|
13
|
+
logger = init_logger(__name__)
|
|
14
|
+
|
|
15
|
+
PLUGIN_TYPE = "mooncakestore"
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
class MooncakestoreConnectorAdapter(ConnectorAdapter):
|
|
19
|
+
"""Adapter for Mooncakestore connectors."""
|
|
20
|
+
|
|
21
|
+
def __init__(self) -> None:
|
|
22
|
+
super().__init__("mooncakestore://")
|
|
23
|
+
|
|
24
|
+
def can_parse(self, url: str) -> bool:
|
|
25
|
+
if url.startswith(self.schema):
|
|
26
|
+
return True
|
|
27
|
+
if url.startswith("plugin://"):
|
|
28
|
+
pname = url[len("plugin://") :]
|
|
29
|
+
return extract_plugin_type(pname) == PLUGIN_TYPE
|
|
30
|
+
return False
|
|
31
|
+
|
|
32
|
+
def create_connector(self, context: ConnectorContext) -> RemoteConnector:
|
|
33
|
+
# Local
|
|
34
|
+
from .mooncakestore_connector import MooncakestoreConnector
|
|
35
|
+
|
|
36
|
+
logger.info("Creating Mooncakestore connector")
|
|
37
|
+
|
|
38
|
+
return MooncakestoreConnector(
|
|
39
|
+
loop=context.loop,
|
|
40
|
+
local_cpu_backend=context.local_cpu_backend,
|
|
41
|
+
lmcache_config=context.config,
|
|
42
|
+
plugin_name=context.plugin_name,
|
|
43
|
+
)
|