lmcache-cli 0.4.5.dev0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- lmcache/__init__.py +84 -0
- lmcache/_version.py +24 -0
- lmcache/cli/__init__.py +1 -0
- lmcache/cli/commands/__init__.py +34 -0
- lmcache/cli/commands/base.py +157 -0
- lmcache/cli/commands/bench/__init__.py +557 -0
- lmcache/cli/commands/bench/engine_bench/__init__.py +1 -0
- lmcache/cli/commands/bench/engine_bench/config.py +245 -0
- lmcache/cli/commands/bench/engine_bench/interactive/__init__.py +274 -0
- lmcache/cli/commands/bench/engine_bench/interactive/config.json +10 -0
- lmcache/cli/commands/bench/engine_bench/interactive/schema.py +352 -0
- lmcache/cli/commands/bench/engine_bench/interactive/state.py +327 -0
- lmcache/cli/commands/bench/engine_bench/interactive/terminal.py +291 -0
- lmcache/cli/commands/bench/engine_bench/progress.py +145 -0
- lmcache/cli/commands/bench/engine_bench/request_sender.py +232 -0
- lmcache/cli/commands/bench/engine_bench/stats.py +275 -0
- lmcache/cli/commands/bench/engine_bench/workloads/__init__.py +153 -0
- lmcache/cli/commands/bench/engine_bench/workloads/base.py +122 -0
- lmcache/cli/commands/bench/engine_bench/workloads/long_doc_permutator.py +435 -0
- lmcache/cli/commands/bench/engine_bench/workloads/long_doc_qa.py +281 -0
- lmcache/cli/commands/bench/engine_bench/workloads/multi_round_chat.py +337 -0
- lmcache/cli/commands/bench/engine_bench/workloads/random_prefill.py +178 -0
- lmcache/cli/commands/describe.py +310 -0
- lmcache/cli/commands/kvcache.py +133 -0
- lmcache/cli/commands/mock.py +75 -0
- lmcache/cli/commands/ping.py +113 -0
- lmcache/cli/commands/query/__init__.py +155 -0
- lmcache/cli/commands/query/prompt.py +134 -0
- lmcache/cli/commands/query/request.py +357 -0
- lmcache/cli/commands/server.py +99 -0
- lmcache/cli/commands/tool/__init__.py +63 -0
- lmcache/cli/commands/tool/cache_simulator.py +113 -0
- lmcache/cli/commands/trace/__init__.py +505 -0
- lmcache/cli/commands/trace/dispatch.py +249 -0
- lmcache/cli/commands/trace/driver.py +372 -0
- lmcache/cli/commands/trace/stats.py +289 -0
- lmcache/cli/documents/lmcache.txt +11 -0
- lmcache/cli/main.py +42 -0
- lmcache/cli/metrics/__init__.py +29 -0
- lmcache/cli/metrics/formatter.py +171 -0
- lmcache/cli/metrics/handler.py +94 -0
- lmcache/cli/metrics/metrics.py +161 -0
- lmcache/cli/metrics/section.py +77 -0
- lmcache/connections.py +173 -0
- lmcache/integration/__init__.py +2 -0
- lmcache/integration/base_service_factory.py +165 -0
- lmcache/integration/request_telemetry/__init__.py +1 -0
- lmcache/integration/request_telemetry/base.py +51 -0
- lmcache/integration/request_telemetry/factory.py +113 -0
- lmcache/integration/request_telemetry/fastapi.py +109 -0
- lmcache/integration/request_telemetry/noop.py +35 -0
- lmcache/integration/sglang/__init__.py +2 -0
- lmcache/integration/sglang/sglang_adapter.py +326 -0
- lmcache/integration/sglang/utils.py +39 -0
- lmcache/integration/vllm/__init__.py +1 -0
- lmcache/integration/vllm/lmcache_connector_v1.py +213 -0
- lmcache/integration/vllm/lmcache_connector_v1_085.py +150 -0
- lmcache/integration/vllm/lmcache_mp_connector_0180.py +1072 -0
- lmcache/integration/vllm/tests/test_mm_hash_utils.py +112 -0
- lmcache/integration/vllm/utils.py +433 -0
- lmcache/integration/vllm/vllm_multi_process_adapter.py +1090 -0
- lmcache/integration/vllm/vllm_service_factory.py +339 -0
- lmcache/integration/vllm/vllm_v1_adapter.py +1713 -0
- lmcache/logging.py +107 -0
- lmcache/native_storage_ops.pyi +230 -0
- lmcache/non_cuda_equivalents.py +1424 -0
- lmcache/observability.py +1958 -0
- lmcache/storage_backend/serde/__init__.py +1 -0
- lmcache/storage_backend/serde/cachegen_basics.py +210 -0
- lmcache/storage_backend/serde/cachegen_decoder.py +207 -0
- lmcache/storage_backend/serde/cachegen_encoder.py +394 -0
- lmcache/storage_backend/serde/serde.py +75 -0
- lmcache/tools/__init__.py +1 -0
- lmcache/tools/cache_simulator/README.md +392 -0
- lmcache/tools/cache_simulator/__init__.py +1 -0
- lmcache/tools/cache_simulator/docs/simulate_example.png +0 -0
- lmcache/tools/cache_simulator/docs/sweep_example.png +0 -0
- lmcache/tools/cache_simulator/gen_bench_dataset.py +360 -0
- lmcache/tools/cache_simulator/lru_cache.py +124 -0
- lmcache/tools/cache_simulator/plot_hit_rate.py +231 -0
- lmcache/tools/cache_simulator/simulator.py +795 -0
- lmcache/tools/controller_benchmark/README.md +161 -0
- lmcache/tools/controller_benchmark/__init__.py +1 -0
- lmcache/tools/controller_benchmark/__main__.py +331 -0
- lmcache/tools/controller_benchmark/benchmark.py +660 -0
- lmcache/tools/controller_benchmark/config.py +44 -0
- lmcache/tools/controller_benchmark/constants.py +10 -0
- lmcache/tools/controller_benchmark/handlers/__init__.py +46 -0
- lmcache/tools/controller_benchmark/handlers/admit.py +52 -0
- lmcache/tools/controller_benchmark/handlers/base.py +47 -0
- lmcache/tools/controller_benchmark/handlers/deregister.py +49 -0
- lmcache/tools/controller_benchmark/handlers/evict.py +52 -0
- lmcache/tools/controller_benchmark/handlers/heartbeat.py +56 -0
- lmcache/tools/controller_benchmark/handlers/p2p_lookup.py +47 -0
- lmcache/tools/controller_benchmark/handlers/register.py +56 -0
- lmcache/tools/mp_status_viewer/__init__.py +1 -0
- lmcache/tools/mp_status_viewer/__main__.py +95 -0
- lmcache/usage_context.py +417 -0
- lmcache/utils.py +665 -0
- lmcache/v1/__init__.py +2 -0
- lmcache/v1/api_server/__init__.py +2 -0
- lmcache/v1/api_server/__main__.py +537 -0
- lmcache/v1/basic_check.py +112 -0
- lmcache/v1/cache_controller/__init__.py +9 -0
- lmcache/v1/cache_controller/commands/__init__.py +15 -0
- lmcache/v1/cache_controller/commands/base.py +35 -0
- lmcache/v1/cache_controller/commands/full_sync.py +49 -0
- lmcache/v1/cache_controller/config.py +176 -0
- lmcache/v1/cache_controller/controller_manager.py +535 -0
- lmcache/v1/cache_controller/controllers/__init__.py +11 -0
- lmcache/v1/cache_controller/controllers/full_sync_tracker.py +473 -0
- lmcache/v1/cache_controller/controllers/kv_controller.py +439 -0
- lmcache/v1/cache_controller/controllers/registration_controller.py +282 -0
- lmcache/v1/cache_controller/executor.py +463 -0
- lmcache/v1/cache_controller/frontend/static/css/style.css +201 -0
- lmcache/v1/cache_controller/frontend/static/img/logo.png +0 -0
- lmcache/v1/cache_controller/frontend/static/index.html +234 -0
- lmcache/v1/cache_controller/frontend/static/js/controller_app.js +660 -0
- lmcache/v1/cache_controller/full_sync_sender.py +475 -0
- lmcache/v1/cache_controller/locks.py +149 -0
- lmcache/v1/cache_controller/message.py +828 -0
- lmcache/v1/cache_controller/observability.py +208 -0
- lmcache/v1/cache_controller/utils.py +679 -0
- lmcache/v1/cache_controller/worker.py +665 -0
- lmcache/v1/cache_engine.py +2058 -0
- lmcache/v1/cache_interface.py +19 -0
- lmcache/v1/check/__init__.py +74 -0
- lmcache/v1/check/check_mode_gen.py +86 -0
- lmcache/v1/check/check_mode_test_l2_adapter.py +284 -0
- lmcache/v1/check/check_mode_test_remote.py +155 -0
- lmcache/v1/check/check_mode_test_storage_manager.py +142 -0
- lmcache/v1/check/utils.py +571 -0
- lmcache/v1/compute/__init__.py +2 -0
- lmcache/v1/compute/attention/__init__.py +0 -0
- lmcache/v1/compute/attention/abstract.py +39 -0
- lmcache/v1/compute/attention/flash_attn.py +129 -0
- lmcache/v1/compute/attention/flash_infer_sparse.py +284 -0
- lmcache/v1/compute/attention/metadata.py +85 -0
- lmcache/v1/compute/attention/utils.py +14 -0
- lmcache/v1/compute/blend/__init__.py +7 -0
- lmcache/v1/compute/blend/blender.py +168 -0
- lmcache/v1/compute/blend/metadata.py +34 -0
- lmcache/v1/compute/blend/utils.py +63 -0
- lmcache/v1/compute/models/__init__.py +0 -0
- lmcache/v1/compute/models/base.py +141 -0
- lmcache/v1/compute/models/llama.py +9 -0
- lmcache/v1/compute/models/qwen3.py +24 -0
- lmcache/v1/compute/models/utils.py +68 -0
- lmcache/v1/compute/positional_encoding.py +199 -0
- lmcache/v1/config.py +848 -0
- lmcache/v1/config_base.py +848 -0
- lmcache/v1/distributed/api.py +248 -0
- lmcache/v1/distributed/config.py +321 -0
- lmcache/v1/distributed/error.py +64 -0
- lmcache/v1/distributed/eviction.py +192 -0
- lmcache/v1/distributed/eviction_policy/__init__.py +21 -0
- lmcache/v1/distributed/eviction_policy/factory.py +27 -0
- lmcache/v1/distributed/eviction_policy/lru.py +244 -0
- lmcache/v1/distributed/eviction_policy/noop.py +50 -0
- lmcache/v1/distributed/internal_api.py +170 -0
- lmcache/v1/distributed/l1_manager.py +835 -0
- lmcache/v1/distributed/l2_adapters/__init__.py +67 -0
- lmcache/v1/distributed/l2_adapters/base.py +360 -0
- lmcache/v1/distributed/l2_adapters/config.py +385 -0
- lmcache/v1/distributed/l2_adapters/factory.py +205 -0
- lmcache/v1/distributed/l2_adapters/fs_l2_adapter.py +747 -0
- lmcache/v1/distributed/l2_adapters/fs_native_l2_adapter.py +167 -0
- lmcache/v1/distributed/l2_adapters/mock_l2_adapter.py +516 -0
- lmcache/v1/distributed/l2_adapters/mooncake_store_l2_adapter.py +135 -0
- lmcache/v1/distributed/l2_adapters/native_connector_l2_adapter.py +468 -0
- lmcache/v1/distributed/l2_adapters/native_plugin_l2_adapter.py +199 -0
- lmcache/v1/distributed/l2_adapters/nixl_store_dynamic_l2_adapter.py +831 -0
- lmcache/v1/distributed/l2_adapters/nixl_store_l2_adapter.py +983 -0
- lmcache/v1/distributed/l2_adapters/plugin_l2_adapter.py +210 -0
- lmcache/v1/distributed/l2_adapters/resp_l2_adapter.py +176 -0
- lmcache/v1/distributed/memory_manager.py +179 -0
- lmcache/v1/distributed/storage_controller.py +39 -0
- lmcache/v1/distributed/storage_controllers/__init__.py +43 -0
- lmcache/v1/distributed/storage_controllers/eviction_controller.py +242 -0
- lmcache/v1/distributed/storage_controllers/prefetch_controller.py +830 -0
- lmcache/v1/distributed/storage_controllers/prefetch_policy.py +193 -0
- lmcache/v1/distributed/storage_controllers/store_controller.py +452 -0
- lmcache/v1/distributed/storage_controllers/store_policy.py +213 -0
- lmcache/v1/distributed/storage_manager.py +532 -0
- lmcache/v1/event_manager.py +145 -0
- lmcache/v1/exceptions/__init__.py +16 -0
- lmcache/v1/gpu_connector/__init__.py +126 -0
- lmcache/v1/gpu_connector/gpu_connectors.py +1906 -0
- lmcache/v1/gpu_connector/gpu_ops.py +85 -0
- lmcache/v1/gpu_connector/hpu_connector.py +326 -0
- lmcache/v1/gpu_connector/mock_gpu_connector.py +67 -0
- lmcache/v1/gpu_connector/utils.py +890 -0
- lmcache/v1/gpu_connector/xpu_connectors.py +916 -0
- lmcache/v1/health_monitor/__init__.py +1 -0
- lmcache/v1/health_monitor/base.py +587 -0
- lmcache/v1/health_monitor/checks/__init__.py +1 -0
- lmcache/v1/health_monitor/checks/remote_backend_check.py +304 -0
- lmcache/v1/health_monitor/constants.py +36 -0
- lmcache/v1/internal_api_server/__init__.py +0 -0
- lmcache/v1/internal_api_server/api_registry.py +59 -0
- lmcache/v1/internal_api_server/api_server.py +120 -0
- lmcache/v1/internal_api_server/common/__init__.py +1 -0
- lmcache/v1/internal_api_server/common/env_api.py +22 -0
- lmcache/v1/internal_api_server/common/loglevel_api.py +57 -0
- lmcache/v1/internal_api_server/common/metrics_api.py +29 -0
- lmcache/v1/internal_api_server/common/periodic_thread_api.py +138 -0
- lmcache/v1/internal_api_server/common/run_script_api.py +73 -0
- lmcache/v1/internal_api_server/common/thread_api.py +63 -0
- lmcache/v1/internal_api_server/controller/__init__.py +1 -0
- lmcache/v1/internal_api_server/controller/key_stats_api.py +81 -0
- lmcache/v1/internal_api_server/controller/worker_info_api.py +136 -0
- lmcache/v1/internal_api_server/utils.py +43 -0
- lmcache/v1/internal_api_server/vllm/__init__.py +1 -0
- lmcache/v1/internal_api_server/vllm/backend_api.py +221 -0
- lmcache/v1/internal_api_server/vllm/bypass_api.py +204 -0
- lmcache/v1/internal_api_server/vllm/cache_api.py +895 -0
- lmcache/v1/internal_api_server/vllm/chunk_statistics_api.py +141 -0
- lmcache/v1/internal_api_server/vllm/conf_api.py +147 -0
- lmcache/v1/internal_api_server/vllm/freeze_api.py +172 -0
- lmcache/v1/internal_api_server/vllm/hot_cache_api.py +184 -0
- lmcache/v1/internal_api_server/vllm/inference_api.py +65 -0
- lmcache/v1/internal_api_server/vllm/load_fs_chunks_api.py +320 -0
- lmcache/v1/internal_api_server/vllm/lookup_api.py +145 -0
- lmcache/v1/internal_api_server/vllm/version_api.py +25 -0
- lmcache/v1/kv_layer_groups.py +267 -0
- lmcache/v1/lazy_memory_allocator.py +284 -0
- lmcache/v1/lookup_client/__init__.py +25 -0
- lmcache/v1/lookup_client/abstract_client.py +77 -0
- lmcache/v1/lookup_client/async_lookup_message.py +50 -0
- lmcache/v1/lookup_client/chunk_statistics_lookup_client.py +200 -0
- lmcache/v1/lookup_client/factory.py +251 -0
- lmcache/v1/lookup_client/hit_limit_lookup_client.py +86 -0
- lmcache/v1/lookup_client/lmcache_async_lookup_client.py +407 -0
- lmcache/v1/lookup_client/lmcache_lookup_client.py +285 -0
- lmcache/v1/lookup_client/lmcache_lookup_client_bypass.py +99 -0
- lmcache/v1/lookup_client/mooncake_lookup_client.py +87 -0
- lmcache/v1/lookup_client/record_strategies/__init__.py +77 -0
- lmcache/v1/lookup_client/record_strategies/base.py +327 -0
- lmcache/v1/lookup_client/record_strategies/file_hash.py +130 -0
- lmcache/v1/lookup_client/record_strategies/memory_bloom_filter.py +81 -0
- lmcache/v1/manager.py +539 -0
- lmcache/v1/memory_management.py +2619 -0
- lmcache/v1/metadata.py +114 -0
- lmcache/v1/mp_observability/AGENTS.override.md +21 -0
- lmcache/v1/mp_observability/README.md +204 -0
- lmcache/v1/mp_observability/config.py +340 -0
- lmcache/v1/mp_observability/event.py +100 -0
- lmcache/v1/mp_observability/event_bus.py +313 -0
- lmcache/v1/mp_observability/otel_init.py +145 -0
- lmcache/v1/mp_observability/subscribers/__init__.py +28 -0
- lmcache/v1/mp_observability/subscribers/logging/__init__.py +19 -0
- lmcache/v1/mp_observability/subscribers/logging/l1.py +56 -0
- lmcache/v1/mp_observability/subscribers/logging/l2.py +73 -0
- lmcache/v1/mp_observability/subscribers/logging/lookup_hash.py +209 -0
- lmcache/v1/mp_observability/subscribers/logging/mp_server.py +90 -0
- lmcache/v1/mp_observability/subscribers/logging/sm.py +59 -0
- lmcache/v1/mp_observability/subscribers/metrics/__init__.py +20 -0
- lmcache/v1/mp_observability/subscribers/metrics/l0_lifecycle.py +290 -0
- lmcache/v1/mp_observability/subscribers/metrics/l1.py +55 -0
- lmcache/v1/mp_observability/subscribers/metrics/l1_lifecycle.py +166 -0
- lmcache/v1/mp_observability/subscribers/metrics/l2.py +121 -0
- lmcache/v1/mp_observability/subscribers/metrics/sm.py +69 -0
- lmcache/v1/mp_observability/subscribers/tracing/__init__.py +12 -0
- lmcache/v1/mp_observability/subscribers/tracing/mp_server.py +333 -0
- lmcache/v1/mp_observability/subscribers/tracing/span_registry.py +148 -0
- lmcache/v1/mp_observability/trace/__init__.py +50 -0
- lmcache/v1/mp_observability/trace/codecs.py +255 -0
- lmcache/v1/mp_observability/trace/decorator.py +147 -0
- lmcache/v1/mp_observability/trace/format.py +132 -0
- lmcache/v1/mp_observability/trace/lifecycle.py +83 -0
- lmcache/v1/mp_observability/trace/reader.py +167 -0
- lmcache/v1/mp_observability/trace/recorder.py +300 -0
- lmcache/v1/multiprocess/__init__.py +0 -0
- lmcache/v1/multiprocess/affinity_pool.py +102 -0
- lmcache/v1/multiprocess/blend_server_v2.py +891 -0
- lmcache/v1/multiprocess/config.py +253 -0
- lmcache/v1/multiprocess/custom_types.py +281 -0
- lmcache/v1/multiprocess/futures.py +194 -0
- lmcache/v1/multiprocess/gpu_context.py +511 -0
- lmcache/v1/multiprocess/http_server.py +235 -0
- lmcache/v1/multiprocess/mp_runtime_plugin_launcher.py +130 -0
- lmcache/v1/multiprocess/mq.py +732 -0
- lmcache/v1/multiprocess/protocol.py +86 -0
- lmcache/v1/multiprocess/protocols/README.md +213 -0
- lmcache/v1/multiprocess/protocols/__init__.py +127 -0
- lmcache/v1/multiprocess/protocols/base.py +89 -0
- lmcache/v1/multiprocess/protocols/blend.py +109 -0
- lmcache/v1/multiprocess/protocols/blend_v2.py +57 -0
- lmcache/v1/multiprocess/protocols/controller.py +53 -0
- lmcache/v1/multiprocess/protocols/debug.py +34 -0
- lmcache/v1/multiprocess/protocols/engine.py +146 -0
- lmcache/v1/multiprocess/protocols/observability.py +39 -0
- lmcache/v1/multiprocess/server.py +1134 -0
- lmcache/v1/multiprocess/session.py +190 -0
- lmcache/v1/multiprocess/token_hasher.py +441 -0
- lmcache/v1/offload_server/__init__.py +17 -0
- lmcache/v1/offload_server/abstract_server.py +37 -0
- lmcache/v1/offload_server/message.py +30 -0
- lmcache/v1/offload_server/zmq_server.py +122 -0
- lmcache/v1/periodic_thread.py +579 -0
- lmcache/v1/pin_monitor.py +246 -0
- lmcache/v1/plugin/__init__.py +0 -0
- lmcache/v1/plugin/runtime_plugin_launcher.py +211 -0
- lmcache/v1/protocol.py +317 -0
- lmcache/v1/rpc/__init__.py +17 -0
- lmcache/v1/rpc/transport.py +105 -0
- lmcache/v1/rpc/zmq_transport.py +213 -0
- lmcache/v1/rpc_utils.py +165 -0
- lmcache/v1/server/__init__.py +2 -0
- lmcache/v1/server/__main__.py +170 -0
- lmcache/v1/server/storage_backend/__init__.py +21 -0
- lmcache/v1/server/storage_backend/abstract_backend.py +80 -0
- lmcache/v1/server/storage_backend/local_backend.py +75 -0
- lmcache/v1/server/utils.py +21 -0
- lmcache/v1/standalone/__init__.py +1 -0
- lmcache/v1/standalone/__main__.py +583 -0
- lmcache/v1/standalone/manager.py +80 -0
- lmcache/v1/standalone/standalone_service_factory.py +86 -0
- lmcache/v1/storage_backend/__init__.py +313 -0
- lmcache/v1/storage_backend/abstract_backend.py +445 -0
- lmcache/v1/storage_backend/audit_backend.py +233 -0
- lmcache/v1/storage_backend/batched_message_sender.py +222 -0
- lmcache/v1/storage_backend/cache_policy/__init__.py +45 -0
- lmcache/v1/storage_backend/cache_policy/base_policy.py +87 -0
- lmcache/v1/storage_backend/cache_policy/fifo.py +58 -0
- lmcache/v1/storage_backend/cache_policy/lfu.py +105 -0
- lmcache/v1/storage_backend/cache_policy/lru.py +81 -0
- lmcache/v1/storage_backend/cache_policy/mru.py +61 -0
- lmcache/v1/storage_backend/connector/__init__.py +443 -0
- lmcache/v1/storage_backend/connector/audit_adapter.py +77 -0
- lmcache/v1/storage_backend/connector/audit_connector.py +320 -0
- lmcache/v1/storage_backend/connector/base_connector.py +379 -0
- lmcache/v1/storage_backend/connector/blackhole_adapter.py +21 -0
- lmcache/v1/storage_backend/connector/blackhole_connector.py +37 -0
- lmcache/v1/storage_backend/connector/eic_adapter.py +31 -0
- lmcache/v1/storage_backend/connector/eic_connector.py +757 -0
- lmcache/v1/storage_backend/connector/external_adapter.py +79 -0
- lmcache/v1/storage_backend/connector/fs_adapter.py +51 -0
- lmcache/v1/storage_backend/connector/fs_connector.py +403 -0
- lmcache/v1/storage_backend/connector/infinistore_adapter.py +56 -0
- lmcache/v1/storage_backend/connector/infinistore_connector.py +177 -0
- lmcache/v1/storage_backend/connector/instrumented_connector.py +219 -0
- lmcache/v1/storage_backend/connector/lm_adapter.py +31 -0
- lmcache/v1/storage_backend/connector/lm_connector.py +176 -0
- lmcache/v1/storage_backend/connector/mock_adapter.py +57 -0
- lmcache/v1/storage_backend/connector/mock_connector.py +349 -0
- lmcache/v1/storage_backend/connector/mooncakestore_adapter.py +43 -0
- lmcache/v1/storage_backend/connector/mooncakestore_connector.py +614 -0
- lmcache/v1/storage_backend/connector/redis_adapter.py +181 -0
- lmcache/v1/storage_backend/connector/redis_connector.py +828 -0
- lmcache/v1/storage_backend/connector/s3_adapter.py +59 -0
- lmcache/v1/storage_backend/connector/s3_connector.py +699 -0
- lmcache/v1/storage_backend/connector/sagemaker_hyperpod_adapter.py +233 -0
- lmcache/v1/storage_backend/connector/sagemaker_hyperpod_connector.py +987 -0
- lmcache/v1/storage_backend/connector/valkey_adapter.py +114 -0
- lmcache/v1/storage_backend/connector/valkey_connector.py +627 -0
- lmcache/v1/storage_backend/gds_backend.py +1199 -0
- lmcache/v1/storage_backend/job_executor/__init__.py +0 -0
- lmcache/v1/storage_backend/job_executor/base_executor.py +34 -0
- lmcache/v1/storage_backend/job_executor/pq_executor.py +235 -0
- lmcache/v1/storage_backend/local_cpu_backend.py +810 -0
- lmcache/v1/storage_backend/local_disk_backend.py +656 -0
- lmcache/v1/storage_backend/maru_backend.py +734 -0
- lmcache/v1/storage_backend/naive_serde/__init__.py +50 -0
- lmcache/v1/storage_backend/naive_serde/cachegen_basics.py +133 -0
- lmcache/v1/storage_backend/naive_serde/cachegen_decoder.py +135 -0
- lmcache/v1/storage_backend/naive_serde/cachegen_encoder.py +83 -0
- lmcache/v1/storage_backend/naive_serde/kivi_serde.py +22 -0
- lmcache/v1/storage_backend/naive_serde/naive_serde.py +18 -0
- lmcache/v1/storage_backend/naive_serde/serde.py +37 -0
- lmcache/v1/storage_backend/native_clients/connector_client_base.py +165 -0
- lmcache/v1/storage_backend/native_clients/resp_client.py +35 -0
- lmcache/v1/storage_backend/nixl_storage_backend.py +1400 -0
- lmcache/v1/storage_backend/p2p_backend.py +788 -0
- lmcache/v1/storage_backend/path_sharder.py +117 -0
- lmcache/v1/storage_backend/pd_backend.py +646 -0
- lmcache/v1/storage_backend/plugins/dax_backend.py +1443 -0
- lmcache/v1/storage_backend/plugins/rust_raw_block_backend.py +1361 -0
- lmcache/v1/storage_backend/remote_backend.py +624 -0
- lmcache/v1/storage_backend/resp_client.py +227 -0
- lmcache/v1/storage_backend/storage_backend_listener.py +19 -0
- lmcache/v1/storage_backend/storage_manager.py +1352 -0
- lmcache/v1/system_detection.py +110 -0
- lmcache/v1/token_database.py +551 -0
- lmcache/v1/transfer_channel/__init__.py +83 -0
- lmcache/v1/transfer_channel/abstract.py +285 -0
- lmcache/v1/transfer_channel/mock_memory_channel.py +156 -0
- lmcache/v1/transfer_channel/nixl_channel.py +639 -0
- lmcache/v1/transfer_channel/py_socket_channel.py +260 -0
- lmcache/v1/transfer_channel/transfer_utils.py +63 -0
- lmcache/v1/utils/__init__.py +1 -0
- lmcache/v1/utils/bloom_filter.py +109 -0
- lmcache/v1/utils/cache_utils.py +125 -0
- lmcache_cli-0.4.5.dev0.dist-info/METADATA +185 -0
- lmcache_cli-0.4.5.dev0.dist-info/RECORD +399 -0
- lmcache_cli-0.4.5.dev0.dist-info/WHEEL +5 -0
- lmcache_cli-0.4.5.dev0.dist-info/entry_points.txt +2 -0
- lmcache_cli-0.4.5.dev0.dist-info/licenses/LICENSE +201 -0
- lmcache_cli-0.4.5.dev0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,445 @@
|
|
|
1
|
+
# SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
# Standard
|
|
3
|
+
from concurrent.futures import Future
|
|
4
|
+
from typing import TYPE_CHECKING, Any, Callable, List, Optional, Sequence, Union
|
|
5
|
+
import abc
|
|
6
|
+
import asyncio
|
|
7
|
+
|
|
8
|
+
# Third Party
|
|
9
|
+
import torch
|
|
10
|
+
|
|
11
|
+
# First Party
|
|
12
|
+
from lmcache.utils import CacheEngineKey
|
|
13
|
+
from lmcache.v1.config import LMCacheEngineConfig
|
|
14
|
+
from lmcache.v1.memory_management import (
|
|
15
|
+
MemoryAllocatorInterface,
|
|
16
|
+
MemoryFormat,
|
|
17
|
+
MemoryObj,
|
|
18
|
+
)
|
|
19
|
+
from lmcache.v1.metadata import LMCacheMetadata
|
|
20
|
+
|
|
21
|
+
if TYPE_CHECKING:
|
|
22
|
+
# First Party
|
|
23
|
+
from lmcache.v1.storage_backend import LocalCPUBackend
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
class StorageBackendInterface(metaclass=abc.ABCMeta):
|
|
27
|
+
def __init__(
|
|
28
|
+
self,
|
|
29
|
+
dst_device: str = "cuda",
|
|
30
|
+
):
|
|
31
|
+
"""
|
|
32
|
+
Initialize the storage backend.
|
|
33
|
+
|
|
34
|
+
:param dst_device: the device where the blocking retrieved KV is stored,
|
|
35
|
+
could be either "cpu", "cuda", or "cuda:0", "cuda:1", etc.
|
|
36
|
+
|
|
37
|
+
:raise: RuntimeError if the device is not valid
|
|
38
|
+
"""
|
|
39
|
+
try:
|
|
40
|
+
torch.device(dst_device)
|
|
41
|
+
except RuntimeError:
|
|
42
|
+
raise
|
|
43
|
+
|
|
44
|
+
self.dst_device = dst_device
|
|
45
|
+
|
|
46
|
+
@abc.abstractmethod
|
|
47
|
+
def contains(self, key: CacheEngineKey, pin: bool = False) -> bool:
|
|
48
|
+
"""
|
|
49
|
+
Check whether key is in the storage backend.
|
|
50
|
+
|
|
51
|
+
:param CacheEngineKey key: The key of the MemoryObj.
|
|
52
|
+
|
|
53
|
+
:param bool pin: Whether to pin the key.
|
|
54
|
+
If True, the corresponding KV cache will be
|
|
55
|
+
pinned in the storage backend.
|
|
56
|
+
|
|
57
|
+
:return: True if the key exists, False otherwise.
|
|
58
|
+
"""
|
|
59
|
+
raise NotImplementedError
|
|
60
|
+
|
|
61
|
+
@abc.abstractmethod
|
|
62
|
+
def exists_in_put_tasks(self, key: CacheEngineKey) -> bool:
|
|
63
|
+
"""
|
|
64
|
+
Check whether key is in the ongoing put tasks.
|
|
65
|
+
"""
|
|
66
|
+
raise NotImplementedError
|
|
67
|
+
|
|
68
|
+
# NOTE (Jiayi): Using batched interface allows the underlying implementation
|
|
69
|
+
# have more flexibility to do optimizations.
|
|
70
|
+
@abc.abstractmethod
|
|
71
|
+
def batched_submit_put_task(
|
|
72
|
+
self,
|
|
73
|
+
keys: Sequence[CacheEngineKey],
|
|
74
|
+
objs: List[MemoryObj],
|
|
75
|
+
transfer_spec: Any = None,
|
|
76
|
+
on_complete_callback: Optional[Callable[[CacheEngineKey], None]] = None,
|
|
77
|
+
) -> Union[List[Future], None]:
|
|
78
|
+
"""
|
|
79
|
+
An async function to put the MemoryObj into the storage backend.
|
|
80
|
+
|
|
81
|
+
:param List[CacheEngineKey] keys: The keys of the MemoryObjs.
|
|
82
|
+
:param List[MemoryObj] objs: The MemoryObjs to be stored.
|
|
83
|
+
:param Any transfer_spec: Optional transfer specification.
|
|
84
|
+
:param on_complete_callback: Optional callback invoked once per key
|
|
85
|
+
after the backend finishes persisting the KV chunk for that key.
|
|
86
|
+
For batched puts, the callback is invoked separately for each key
|
|
87
|
+
when that key completes (not once per batch). Callback exceptions
|
|
88
|
+
are caught and logged. Backends that cannot use this callback may
|
|
89
|
+
ignore it.
|
|
90
|
+
|
|
91
|
+
:return: Union[List[Future], None]: A list of `Future` objects if the
|
|
92
|
+
storage persistence operation is asynchronous and is successful.
|
|
93
|
+
`None` if the operation is synchronous, or the asynchronous fails
|
|
94
|
+
or is skipped.
|
|
95
|
+
|
|
96
|
+
:note: This function will have the side effect that modifies the
|
|
97
|
+
underlying key-value mappings in the storage backend. The side
|
|
98
|
+
effect may change the result of lookup and get.
|
|
99
|
+
"""
|
|
100
|
+
raise NotImplementedError
|
|
101
|
+
|
|
102
|
+
async def async_batched_submit_put_task(
|
|
103
|
+
self,
|
|
104
|
+
keys: Sequence[CacheEngineKey],
|
|
105
|
+
objs: List[MemoryObj],
|
|
106
|
+
transfer_spec: Any = None,
|
|
107
|
+
on_complete_callback: Optional[Callable[[CacheEngineKey], None]] = None,
|
|
108
|
+
) -> None:
|
|
109
|
+
"""
|
|
110
|
+
An async version of batched_submit_put_task.
|
|
111
|
+
|
|
112
|
+
:param on_complete_callback: Optional callback invoked once per key
|
|
113
|
+
after the backend finishes persisting the KV chunk for that key.
|
|
114
|
+
"""
|
|
115
|
+
raise NotImplementedError
|
|
116
|
+
|
|
117
|
+
@abc.abstractmethod
|
|
118
|
+
def get_blocking(
|
|
119
|
+
self,
|
|
120
|
+
key: CacheEngineKey,
|
|
121
|
+
) -> Optional[MemoryObj]:
|
|
122
|
+
"""
|
|
123
|
+
A blocking function to get the kv cache from the storage backend.
|
|
124
|
+
|
|
125
|
+
:param CacheEngineKey key: The key of the MemoryObj.
|
|
126
|
+
|
|
127
|
+
:return: MemoryObj. None if the key does not exist.
|
|
128
|
+
"""
|
|
129
|
+
raise NotImplementedError
|
|
130
|
+
|
|
131
|
+
def get_non_blocking(
|
|
132
|
+
self,
|
|
133
|
+
key: CacheEngineKey,
|
|
134
|
+
location: Optional[str] = None,
|
|
135
|
+
) -> Optional[Future]:
|
|
136
|
+
"""
|
|
137
|
+
A non-blocking function to get the kv cache from the storage backend.
|
|
138
|
+
"""
|
|
139
|
+
raise NotImplementedError
|
|
140
|
+
|
|
141
|
+
async def batched_async_contains(
|
|
142
|
+
self,
|
|
143
|
+
lookup_id: str,
|
|
144
|
+
keys: List[CacheEngineKey],
|
|
145
|
+
pin: bool = False,
|
|
146
|
+
) -> int:
|
|
147
|
+
"""
|
|
148
|
+
Check whether keys are in the storage backend.
|
|
149
|
+
|
|
150
|
+
:param List[CacheEngineKey] keys: The keys of the MemoryObjs.
|
|
151
|
+
|
|
152
|
+
:param bool pin: Whether to pin the keys.
|
|
153
|
+
If True, the corresponding KV caches will be
|
|
154
|
+
pinned in the storage backend.
|
|
155
|
+
|
|
156
|
+
:return: The number of keys that exist in the storage backend.
|
|
157
|
+
"""
|
|
158
|
+
raise NotImplementedError
|
|
159
|
+
|
|
160
|
+
async def batched_get_non_blocking(
|
|
161
|
+
self,
|
|
162
|
+
lookup_id: str,
|
|
163
|
+
keys: list[CacheEngineKey],
|
|
164
|
+
transfer_spec: Any = None,
|
|
165
|
+
) -> list[MemoryObj]:
|
|
166
|
+
"""
|
|
167
|
+
A non-blcocking function to get the kv cache from the storage backend.
|
|
168
|
+
|
|
169
|
+
:param list[CacheEngineKey] keys: The keys of the list of MemoryObjs.
|
|
170
|
+
|
|
171
|
+
:return: a list of Memoryobjs.
|
|
172
|
+
"""
|
|
173
|
+
raise NotImplementedError
|
|
174
|
+
|
|
175
|
+
# NOTE(Jiayi): Please re-implement this method if the storage backend
|
|
176
|
+
# can benefit from batched get.
|
|
177
|
+
def batched_get_blocking(
|
|
178
|
+
self,
|
|
179
|
+
keys: List[CacheEngineKey],
|
|
180
|
+
) -> List[Optional[MemoryObj]]:
|
|
181
|
+
"""
|
|
182
|
+
A blocking function to get the kv cache from the storage backend.
|
|
183
|
+
|
|
184
|
+
:param List[CacheEngineKey] keys: The keys of the MemoryObjs.
|
|
185
|
+
|
|
186
|
+
:return: a list of memory objects.
|
|
187
|
+
"""
|
|
188
|
+
mem_objs = []
|
|
189
|
+
for key in keys:
|
|
190
|
+
mem_objs.append(self.get_blocking(key))
|
|
191
|
+
return mem_objs
|
|
192
|
+
|
|
193
|
+
@abc.abstractmethod
|
|
194
|
+
def pin(
|
|
195
|
+
self,
|
|
196
|
+
key: CacheEngineKey,
|
|
197
|
+
) -> bool:
|
|
198
|
+
"""
|
|
199
|
+
Pin a memory object so it will not be evicted.
|
|
200
|
+
|
|
201
|
+
:param CacheEngineKey key: The key of the MemoryObj.
|
|
202
|
+
|
|
203
|
+
:return: a bool indicates whether pin is successful.
|
|
204
|
+
"""
|
|
205
|
+
raise NotImplementedError
|
|
206
|
+
|
|
207
|
+
@abc.abstractmethod
|
|
208
|
+
def unpin(
|
|
209
|
+
self,
|
|
210
|
+
key: CacheEngineKey,
|
|
211
|
+
) -> bool:
|
|
212
|
+
"""
|
|
213
|
+
Unpin a memory object so it can be evicted.
|
|
214
|
+
|
|
215
|
+
:param CacheEngineKey key: The key of the MemoryObj.
|
|
216
|
+
|
|
217
|
+
:return: a bool indicates whether unpin is successful.
|
|
218
|
+
"""
|
|
219
|
+
raise NotImplementedError
|
|
220
|
+
|
|
221
|
+
@abc.abstractmethod
|
|
222
|
+
def remove(self, key: CacheEngineKey, force: bool = True) -> bool:
|
|
223
|
+
"""
|
|
224
|
+
remove a memory object.
|
|
225
|
+
|
|
226
|
+
:param CacheEngineKey key: The key of the MemoryObj.
|
|
227
|
+
:param bool force: Whether to it is a forced remove from the external.
|
|
228
|
+
|
|
229
|
+
:return: a bool indicates whether remove is successful.
|
|
230
|
+
"""
|
|
231
|
+
raise NotImplementedError
|
|
232
|
+
|
|
233
|
+
# TODO(Jiayi): Optimize batched remove
|
|
234
|
+
def batched_remove(
|
|
235
|
+
self,
|
|
236
|
+
keys: list[CacheEngineKey],
|
|
237
|
+
force: bool = True,
|
|
238
|
+
) -> int:
|
|
239
|
+
"""
|
|
240
|
+
Remove a list of memory objects.
|
|
241
|
+
|
|
242
|
+
:param list[CacheEngineKey] keys: The keys of the MemoryObjs.
|
|
243
|
+
:param bool force: Whether to force remove the memory objects.
|
|
244
|
+
|
|
245
|
+
:return: a int indicates the number of removed memory objects.
|
|
246
|
+
"""
|
|
247
|
+
num_removed = 0
|
|
248
|
+
for key in keys:
|
|
249
|
+
num_removed += self.remove(key, force=force)
|
|
250
|
+
return num_removed
|
|
251
|
+
|
|
252
|
+
@abc.abstractmethod
|
|
253
|
+
def get_allocator_backend(self) -> "AllocatorBackendInterface":
|
|
254
|
+
"""
|
|
255
|
+
Get the allocator backend that is used by the current storage backend
|
|
256
|
+
to allocate memory objects during `get` operations.
|
|
257
|
+
|
|
258
|
+
:return: an instance of AllocateBackendInterface
|
|
259
|
+
"""
|
|
260
|
+
raise NotImplementedError
|
|
261
|
+
|
|
262
|
+
@abc.abstractmethod
|
|
263
|
+
def close(
|
|
264
|
+
self,
|
|
265
|
+
) -> None:
|
|
266
|
+
"""
|
|
267
|
+
Close the storage backend.
|
|
268
|
+
"""
|
|
269
|
+
raise NotImplementedError
|
|
270
|
+
|
|
271
|
+
def batched_contains(
|
|
272
|
+
self,
|
|
273
|
+
keys: List[CacheEngineKey],
|
|
274
|
+
pin: bool = False,
|
|
275
|
+
) -> int:
|
|
276
|
+
"""
|
|
277
|
+
Check whether the keys are in the storage backend.
|
|
278
|
+
|
|
279
|
+
:param List[CacheEngineKey] keys: The keys of the MemoryObj.
|
|
280
|
+
|
|
281
|
+
:param bool pin: Whether to pin the key.
|
|
282
|
+
If True, the corresponding KV cache will be
|
|
283
|
+
pinned in the storage backend.
|
|
284
|
+
|
|
285
|
+
:return: Return hit chunks by prefix match.
|
|
286
|
+
"""
|
|
287
|
+
hit_chunks = 0
|
|
288
|
+
for key in keys:
|
|
289
|
+
if not self.contains(key, pin):
|
|
290
|
+
break
|
|
291
|
+
hit_chunks += 1
|
|
292
|
+
return hit_chunks
|
|
293
|
+
|
|
294
|
+
def touch_cache(self) -> None:
|
|
295
|
+
"""
|
|
296
|
+
Update cache policy with keys that were accessed during a request.
|
|
297
|
+
|
|
298
|
+
This method is called to update the cache eviction policy with the
|
|
299
|
+
keys that were accessed in the most recent request, typically to
|
|
300
|
+
implement LRU or similar eviction strategies.
|
|
301
|
+
|
|
302
|
+
Default implementation does nothing. Backends that support
|
|
303
|
+
cache eviction policies should override this method.
|
|
304
|
+
|
|
305
|
+
:return: None
|
|
306
|
+
"""
|
|
307
|
+
raise NotImplementedError
|
|
308
|
+
|
|
309
|
+
|
|
310
|
+
class AllocatorBackendInterface(StorageBackendInterface):
|
|
311
|
+
"""
|
|
312
|
+
AllocatorBackendInterface extends the StorageBackendInterface with
|
|
313
|
+
the ability to actively allocate the memory objects.
|
|
314
|
+
"""
|
|
315
|
+
|
|
316
|
+
@abc.abstractmethod
|
|
317
|
+
def initialize_allocator(
|
|
318
|
+
self, config: LMCacheEngineConfig, metadata: LMCacheMetadata
|
|
319
|
+
) -> MemoryAllocatorInterface:
|
|
320
|
+
"""
|
|
321
|
+
Create the correct memory allocator for the current storage backend
|
|
322
|
+
|
|
323
|
+
Args:
|
|
324
|
+
config: The cache engine config
|
|
325
|
+
metadata: the cache engine metadata
|
|
326
|
+
|
|
327
|
+
Returns:
|
|
328
|
+
The memory allocator for this storage backend
|
|
329
|
+
"""
|
|
330
|
+
raise NotImplementedError
|
|
331
|
+
|
|
332
|
+
@abc.abstractmethod
|
|
333
|
+
def get_memory_allocator(self) -> MemoryAllocatorInterface:
|
|
334
|
+
"""
|
|
335
|
+
Returns:
|
|
336
|
+
The underlying memory allocator
|
|
337
|
+
"""
|
|
338
|
+
raise NotImplementedError
|
|
339
|
+
|
|
340
|
+
@abc.abstractmethod
|
|
341
|
+
def allocate(
|
|
342
|
+
self,
|
|
343
|
+
shapes: Union[torch.Size, list[torch.Size]],
|
|
344
|
+
dtypes: Union[torch.dtype, list[torch.dtype]],
|
|
345
|
+
fmt: MemoryFormat = MemoryFormat.KV_2LTD,
|
|
346
|
+
eviction: bool = True,
|
|
347
|
+
busy_loop: bool = True,
|
|
348
|
+
) -> Optional[MemoryObj]:
|
|
349
|
+
"""
|
|
350
|
+
Allocates memory in the backend to hold a tensor of the given shape.
|
|
351
|
+
|
|
352
|
+
:param Union[torch.Size, list[torch.Size]] shapes:
|
|
353
|
+
The shape of the tensor to allocate.
|
|
354
|
+
:param Union[torch.dtype, list[torch.dtype]] dtypes:
|
|
355
|
+
The dtype of the tensor to allocate.
|
|
356
|
+
:param MemoryFormat fmt: The format of the memory to allocate.
|
|
357
|
+
:param bool eviction: whether to enable eviction when allocating.
|
|
358
|
+
:param bool busy_loop: whether to enable a busy loop to wait
|
|
359
|
+
for in-progress store operations to finish and release the
|
|
360
|
+
memory space for retrieve.
|
|
361
|
+
|
|
362
|
+
:return: A MemoryObj wrapping the allocated memory. Returns
|
|
363
|
+
None if the allocation failed.
|
|
364
|
+
|
|
365
|
+
:rtype: Optional[MemoryObj]
|
|
366
|
+
"""
|
|
367
|
+
raise NotImplementedError
|
|
368
|
+
|
|
369
|
+
@abc.abstractmethod
|
|
370
|
+
def batched_allocate(
|
|
371
|
+
self,
|
|
372
|
+
shapes: Union[torch.Size, list[torch.Size]],
|
|
373
|
+
dtypes: Union[torch.dtype, list[torch.dtype]],
|
|
374
|
+
batch_size: int,
|
|
375
|
+
fmt: MemoryFormat = MemoryFormat.KV_2LTD,
|
|
376
|
+
eviction: bool = True,
|
|
377
|
+
busy_loop: bool = True,
|
|
378
|
+
) -> Optional[list[MemoryObj]]:
|
|
379
|
+
"""
|
|
380
|
+
Allocates memory in the backend to hold a tensor of the given shape
|
|
381
|
+
in a batched manner. The allocated memory objects will have the same
|
|
382
|
+
shape, dtype, and format.
|
|
383
|
+
|
|
384
|
+
:param Union[torch.Size, list[torch.Size]] shapes:
|
|
385
|
+
The shape of the tensor to allocate.
|
|
386
|
+
:param Union[torch.dtype, list[torch.dtype]] dtypes:
|
|
387
|
+
The dtype of the tensor to allocate.
|
|
388
|
+
:param int batch_size: The number of memory objects to allocate.
|
|
389
|
+
:param MemoryFormat fmt: The format of the memory to allocate.
|
|
390
|
+
:param bool eviction: whether to enable eviction when allocating.
|
|
391
|
+
:param bool busy_loop: whether to enable a busy loop to wait
|
|
392
|
+
for in-progress store operations to finish and release the
|
|
393
|
+
memory space for retrieve.
|
|
394
|
+
|
|
395
|
+
:return: A MemoryObj wrapping the allocated memory. Returns
|
|
396
|
+
None if the allocation failed.
|
|
397
|
+
|
|
398
|
+
:rtype: Optional[MemoryObj]
|
|
399
|
+
"""
|
|
400
|
+
raise NotImplementedError
|
|
401
|
+
|
|
402
|
+
def calculate_chunk_budget(self) -> int:
|
|
403
|
+
"""
|
|
404
|
+
Calculate the chunk budget for the allocator backend.
|
|
405
|
+
"""
|
|
406
|
+
raise NotImplementedError
|
|
407
|
+
|
|
408
|
+
|
|
409
|
+
class StoragePluginInterface(StorageBackendInterface):
|
|
410
|
+
"""The Configurable Storage Backend Interface needs to be implemented
|
|
411
|
+
when you want to add a storage backend in a configurable or plug and play
|
|
412
|
+
fashion."""
|
|
413
|
+
|
|
414
|
+
def __init__(
|
|
415
|
+
self,
|
|
416
|
+
dst_device: str = "cuda",
|
|
417
|
+
config: Optional[LMCacheEngineConfig] = None,
|
|
418
|
+
metadata: Optional[LMCacheMetadata] = None,
|
|
419
|
+
local_cpu_backend: Optional["LocalCPUBackend"] = None,
|
|
420
|
+
loop: Optional[asyncio.AbstractEventLoop] = None,
|
|
421
|
+
):
|
|
422
|
+
"""
|
|
423
|
+
Initialize a configurable storage backend. This constructor will be called
|
|
424
|
+
when loading the configurable storage backends from the configuration file.
|
|
425
|
+
|
|
426
|
+
:param str dst_device: The target device for tensor operations
|
|
427
|
+
(e.g., "cuda" or "cpu").
|
|
428
|
+
:param LMCacheEngineConfig config: Optional configuration object for the
|
|
429
|
+
cache engine.
|
|
430
|
+
:param LMCacheMetadata metadata: Optional metadata describing the cache
|
|
431
|
+
engine state or version.
|
|
432
|
+
:param LocalCPUBackend local_cpu_backend: Optional backend for local CPU-based
|
|
433
|
+
inference or caching.
|
|
434
|
+
:param asyncio.AbstractEventLoop loop: Optional asyncio event loop for
|
|
435
|
+
asynchronous operations.
|
|
436
|
+
"""
|
|
437
|
+
super().__init__(dst_device=dst_device)
|
|
438
|
+
self.config = config
|
|
439
|
+
self.metadata = metadata
|
|
440
|
+
self.local_cpu_backend = local_cpu_backend
|
|
441
|
+
self.loop = loop
|
|
442
|
+
|
|
443
|
+
|
|
444
|
+
# TODO: Alias for backwards compatibility - remove when applicable
|
|
445
|
+
ConfigurableStorageBackendInterface = StoragePluginInterface
|
|
@@ -0,0 +1,233 @@
|
|
|
1
|
+
# SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
|
|
3
|
+
# Standard
|
|
4
|
+
from typing import Any, Callable, List, Optional, Sequence
|
|
5
|
+
import time
|
|
6
|
+
|
|
7
|
+
# First Party
|
|
8
|
+
from lmcache.logging import init_logger
|
|
9
|
+
from lmcache.utils import CacheEngineKey
|
|
10
|
+
from lmcache.v1.memory_management import MemoryObj
|
|
11
|
+
from lmcache.v1.storage_backend.abstract_backend import StorageBackendInterface
|
|
12
|
+
|
|
13
|
+
logger = init_logger(__name__)
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
class AuditBackend(StorageBackendInterface):
|
|
17
|
+
"""
|
|
18
|
+
Audit wrapper for StorageBackend that logs operations and measures performance.
|
|
19
|
+
"""
|
|
20
|
+
|
|
21
|
+
def __init__(self, real_backend: StorageBackendInterface):
|
|
22
|
+
super().__init__(dst_device=real_backend.dst_device)
|
|
23
|
+
self.real_backend = real_backend
|
|
24
|
+
self.logger = logger.getChild("audit")
|
|
25
|
+
self.logger.info(
|
|
26
|
+
f"[AUDIT_BACKEND] Initialized for backend: {str(real_backend)}"
|
|
27
|
+
)
|
|
28
|
+
|
|
29
|
+
def _log_operation(
|
|
30
|
+
self,
|
|
31
|
+
op_name: str,
|
|
32
|
+
start_time: float,
|
|
33
|
+
key: Optional[CacheEngineKey] = None,
|
|
34
|
+
success: bool = True,
|
|
35
|
+
result=None,
|
|
36
|
+
error=None,
|
|
37
|
+
size=None,
|
|
38
|
+
):
|
|
39
|
+
"""Helper method to log operation results."""
|
|
40
|
+
cost = (time.perf_counter() - start_time) * 1000
|
|
41
|
+
backend_name = str(self.real_backend)
|
|
42
|
+
|
|
43
|
+
if error:
|
|
44
|
+
self.logger.error(
|
|
45
|
+
f"[AUDIT_BACKEND][{backend_name}]:{op_name}|FAILED|"
|
|
46
|
+
f"Key:{key}|Error:{str(error)}"
|
|
47
|
+
)
|
|
48
|
+
elif success:
|
|
49
|
+
log_msg = (
|
|
50
|
+
f"[AUDIT_BACKEND][{backend_name}]:{op_name}|SUCCESS|Cost:{cost:.2f}ms"
|
|
51
|
+
)
|
|
52
|
+
if key:
|
|
53
|
+
log_msg += f"|Key:{key}"
|
|
54
|
+
if size is not None:
|
|
55
|
+
log_msg += f"|Size:{size}"
|
|
56
|
+
if result is not None:
|
|
57
|
+
log_msg += f"|Result:{result}"
|
|
58
|
+
self.logger.info(log_msg)
|
|
59
|
+
|
|
60
|
+
def contains(self, key: CacheEngineKey, pin: bool = False) -> bool:
|
|
61
|
+
"""Check key existence with audit logging."""
|
|
62
|
+
self.logger.debug(f"[AUDIT_BACKEND] Checking contains for key: {key}")
|
|
63
|
+
start_time = time.perf_counter()
|
|
64
|
+
try:
|
|
65
|
+
result = self.real_backend.contains(key, pin)
|
|
66
|
+
self._log_operation("CONTAINS", start_time, key, True, result)
|
|
67
|
+
return result
|
|
68
|
+
except Exception as e:
|
|
69
|
+
self._log_operation("CONTAINS", start_time, key, False, error=e)
|
|
70
|
+
raise
|
|
71
|
+
|
|
72
|
+
def get_blocking(self, key: CacheEngineKey) -> Optional[MemoryObj]:
|
|
73
|
+
"""Retrieve data with audit logging."""
|
|
74
|
+
self.logger.debug(f"[AUDIT_BACKEND] Getting data for key: {key}")
|
|
75
|
+
start_time = time.perf_counter()
|
|
76
|
+
try:
|
|
77
|
+
result = self.real_backend.get_blocking(key)
|
|
78
|
+
size = len(result.byte_array) if result else 0
|
|
79
|
+
self._log_operation(
|
|
80
|
+
"GET", start_time, key, True, result=result is not None, size=size
|
|
81
|
+
)
|
|
82
|
+
return result
|
|
83
|
+
except Exception as e:
|
|
84
|
+
self._log_operation("GET", start_time, key, False, error=e)
|
|
85
|
+
raise
|
|
86
|
+
|
|
87
|
+
def close(self) -> None:
|
|
88
|
+
"""Close backend with audit logging."""
|
|
89
|
+
self.logger.debug("[AUDIT_BACKEND] Closing backend")
|
|
90
|
+
start_time = time.perf_counter()
|
|
91
|
+
try:
|
|
92
|
+
self.real_backend.close()
|
|
93
|
+
self._log_operation("CLOSE", start_time, None, True)
|
|
94
|
+
except Exception as e:
|
|
95
|
+
self._log_operation("CLOSE", start_time, None, False, error=e)
|
|
96
|
+
raise
|
|
97
|
+
|
|
98
|
+
# Implement other required methods following the same pattern
|
|
99
|
+
def exists_in_put_tasks(self, key: CacheEngineKey) -> bool:
|
|
100
|
+
start_time = time.perf_counter()
|
|
101
|
+
try:
|
|
102
|
+
result = self.real_backend.exists_in_put_tasks(key)
|
|
103
|
+
self._log_operation("EXISTS_IN_PUT_TASKS", start_time, key, True, result)
|
|
104
|
+
return result
|
|
105
|
+
except Exception as e:
|
|
106
|
+
self._log_operation("EXISTS_IN_PUT_TASKS", start_time, key, False, error=e)
|
|
107
|
+
raise
|
|
108
|
+
|
|
109
|
+
def batched_submit_put_task(
|
|
110
|
+
self,
|
|
111
|
+
keys: Sequence[CacheEngineKey],
|
|
112
|
+
memory_objs: List[MemoryObj],
|
|
113
|
+
transfer_spec: Any = None,
|
|
114
|
+
on_complete_callback: Optional[Callable[[CacheEngineKey], None]] = None,
|
|
115
|
+
) -> None:
|
|
116
|
+
sizes = [len(obj.byte_array) for obj in memory_objs]
|
|
117
|
+
start_time = time.perf_counter()
|
|
118
|
+
try:
|
|
119
|
+
self.real_backend.batched_submit_put_task(
|
|
120
|
+
keys, memory_objs, transfer_spec, on_complete_callback
|
|
121
|
+
)
|
|
122
|
+
self._log_operation(
|
|
123
|
+
"BATCHED_SUBMIT_PUT_TASK", start_time, None, True, size=sum(sizes)
|
|
124
|
+
)
|
|
125
|
+
except Exception as e:
|
|
126
|
+
self._log_operation(
|
|
127
|
+
"BATCHED_SUBMIT_PUT_TASK", start_time, None, False, error=e
|
|
128
|
+
)
|
|
129
|
+
raise
|
|
130
|
+
|
|
131
|
+
async def batched_get_non_blocking(
|
|
132
|
+
self,
|
|
133
|
+
lookup_id: str,
|
|
134
|
+
keys: list[CacheEngineKey],
|
|
135
|
+
transfer_spec: Any = None,
|
|
136
|
+
) -> list[MemoryObj]:
|
|
137
|
+
start_time = time.perf_counter()
|
|
138
|
+
try:
|
|
139
|
+
result = await self.real_backend.batched_get_non_blocking(lookup_id, keys)
|
|
140
|
+
self._log_operation("BATCHED_GET_NON_BLOCKING", start_time, None, True)
|
|
141
|
+
return result
|
|
142
|
+
except Exception as e:
|
|
143
|
+
self._log_operation(
|
|
144
|
+
"BATCHED_GET_NON_BLOCKING", start_time, None, False, error=e
|
|
145
|
+
)
|
|
146
|
+
raise
|
|
147
|
+
|
|
148
|
+
async def batched_async_contains(
|
|
149
|
+
self,
|
|
150
|
+
lookup_id: str,
|
|
151
|
+
keys: list[CacheEngineKey],
|
|
152
|
+
pin: bool = False,
|
|
153
|
+
) -> int:
|
|
154
|
+
start_time = time.perf_counter()
|
|
155
|
+
try:
|
|
156
|
+
result = await self.real_backend.batched_async_contains(
|
|
157
|
+
lookup_id, keys, pin
|
|
158
|
+
)
|
|
159
|
+
self._log_operation("BATCHED_ASYNC_CONTAINS", start_time, None, True)
|
|
160
|
+
return result
|
|
161
|
+
except Exception as e:
|
|
162
|
+
self._log_operation(
|
|
163
|
+
"BATCHED_ASYNC_CONTAINS", start_time, None, False, error=e
|
|
164
|
+
)
|
|
165
|
+
raise
|
|
166
|
+
|
|
167
|
+
def pin(self, key: CacheEngineKey) -> bool:
|
|
168
|
+
start_time = time.perf_counter()
|
|
169
|
+
try:
|
|
170
|
+
result = self.real_backend.pin(key)
|
|
171
|
+
self._log_operation("PIN", start_time, key, True, result)
|
|
172
|
+
return result
|
|
173
|
+
except Exception as e:
|
|
174
|
+
self._log_operation("PIN", start_time, key, False, error=e)
|
|
175
|
+
raise
|
|
176
|
+
|
|
177
|
+
def unpin(self, key: CacheEngineKey) -> bool:
|
|
178
|
+
start_time = time.perf_counter()
|
|
179
|
+
try:
|
|
180
|
+
result = self.real_backend.unpin(key)
|
|
181
|
+
self._log_operation("UNPIN", start_time, key, True, result)
|
|
182
|
+
return result
|
|
183
|
+
except Exception as e:
|
|
184
|
+
self._log_operation("UNPIN", start_time, key, False, error=e)
|
|
185
|
+
raise
|
|
186
|
+
|
|
187
|
+
def batched_get_blocking(
|
|
188
|
+
self,
|
|
189
|
+
keys: List[CacheEngineKey],
|
|
190
|
+
) -> List[Optional[MemoryObj]]:
|
|
191
|
+
start_time = time.perf_counter()
|
|
192
|
+
try:
|
|
193
|
+
result = self.real_backend.batched_get_blocking(keys)
|
|
194
|
+
self._log_operation(
|
|
195
|
+
"BATCHED_GET_BLOCKING",
|
|
196
|
+
start_time,
|
|
197
|
+
None,
|
|
198
|
+
True,
|
|
199
|
+
result=len(result) if result is not None else 0,
|
|
200
|
+
)
|
|
201
|
+
return result
|
|
202
|
+
except Exception as e:
|
|
203
|
+
self._log_operation(
|
|
204
|
+
"BATCHED_GET_BLOCKING", start_time, None, False, error=e
|
|
205
|
+
)
|
|
206
|
+
raise
|
|
207
|
+
|
|
208
|
+
def remove(self, key: CacheEngineKey, free_obj: bool = True) -> bool:
|
|
209
|
+
start_time = time.perf_counter()
|
|
210
|
+
try:
|
|
211
|
+
result = self.real_backend.remove(key, free_obj)
|
|
212
|
+
self._log_operation("REMOVE", start_time, key, True, result)
|
|
213
|
+
return result
|
|
214
|
+
except Exception as e:
|
|
215
|
+
self._log_operation("REMOVE", start_time, key, False, error=e)
|
|
216
|
+
raise
|
|
217
|
+
|
|
218
|
+
def batched_remove(
|
|
219
|
+
self,
|
|
220
|
+
keys: list[CacheEngineKey],
|
|
221
|
+
free_obj: bool = True,
|
|
222
|
+
) -> int:
|
|
223
|
+
start_time = time.perf_counter()
|
|
224
|
+
try:
|
|
225
|
+
result = self.real_backend.batched_remove(keys, free_obj)
|
|
226
|
+
self._log_operation("BATCHED_REMOVE", start_time, None, True, result)
|
|
227
|
+
return result
|
|
228
|
+
except Exception as e:
|
|
229
|
+
self._log_operation("BATCHED_REMOVE", start_time, None, False, error=e)
|
|
230
|
+
raise
|
|
231
|
+
|
|
232
|
+
def get_allocator_backend(self):
|
|
233
|
+
return self.real_backend.get_allocator_backend()
|