lmcache-cli 0.4.5.dev0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- lmcache/__init__.py +84 -0
- lmcache/_version.py +24 -0
- lmcache/cli/__init__.py +1 -0
- lmcache/cli/commands/__init__.py +34 -0
- lmcache/cli/commands/base.py +157 -0
- lmcache/cli/commands/bench/__init__.py +557 -0
- lmcache/cli/commands/bench/engine_bench/__init__.py +1 -0
- lmcache/cli/commands/bench/engine_bench/config.py +245 -0
- lmcache/cli/commands/bench/engine_bench/interactive/__init__.py +274 -0
- lmcache/cli/commands/bench/engine_bench/interactive/config.json +10 -0
- lmcache/cli/commands/bench/engine_bench/interactive/schema.py +352 -0
- lmcache/cli/commands/bench/engine_bench/interactive/state.py +327 -0
- lmcache/cli/commands/bench/engine_bench/interactive/terminal.py +291 -0
- lmcache/cli/commands/bench/engine_bench/progress.py +145 -0
- lmcache/cli/commands/bench/engine_bench/request_sender.py +232 -0
- lmcache/cli/commands/bench/engine_bench/stats.py +275 -0
- lmcache/cli/commands/bench/engine_bench/workloads/__init__.py +153 -0
- lmcache/cli/commands/bench/engine_bench/workloads/base.py +122 -0
- lmcache/cli/commands/bench/engine_bench/workloads/long_doc_permutator.py +435 -0
- lmcache/cli/commands/bench/engine_bench/workloads/long_doc_qa.py +281 -0
- lmcache/cli/commands/bench/engine_bench/workloads/multi_round_chat.py +337 -0
- lmcache/cli/commands/bench/engine_bench/workloads/random_prefill.py +178 -0
- lmcache/cli/commands/describe.py +310 -0
- lmcache/cli/commands/kvcache.py +133 -0
- lmcache/cli/commands/mock.py +75 -0
- lmcache/cli/commands/ping.py +113 -0
- lmcache/cli/commands/query/__init__.py +155 -0
- lmcache/cli/commands/query/prompt.py +134 -0
- lmcache/cli/commands/query/request.py +357 -0
- lmcache/cli/commands/server.py +99 -0
- lmcache/cli/commands/tool/__init__.py +63 -0
- lmcache/cli/commands/tool/cache_simulator.py +113 -0
- lmcache/cli/commands/trace/__init__.py +505 -0
- lmcache/cli/commands/trace/dispatch.py +249 -0
- lmcache/cli/commands/trace/driver.py +372 -0
- lmcache/cli/commands/trace/stats.py +289 -0
- lmcache/cli/documents/lmcache.txt +11 -0
- lmcache/cli/main.py +42 -0
- lmcache/cli/metrics/__init__.py +29 -0
- lmcache/cli/metrics/formatter.py +171 -0
- lmcache/cli/metrics/handler.py +94 -0
- lmcache/cli/metrics/metrics.py +161 -0
- lmcache/cli/metrics/section.py +77 -0
- lmcache/connections.py +173 -0
- lmcache/integration/__init__.py +2 -0
- lmcache/integration/base_service_factory.py +165 -0
- lmcache/integration/request_telemetry/__init__.py +1 -0
- lmcache/integration/request_telemetry/base.py +51 -0
- lmcache/integration/request_telemetry/factory.py +113 -0
- lmcache/integration/request_telemetry/fastapi.py +109 -0
- lmcache/integration/request_telemetry/noop.py +35 -0
- lmcache/integration/sglang/__init__.py +2 -0
- lmcache/integration/sglang/sglang_adapter.py +326 -0
- lmcache/integration/sglang/utils.py +39 -0
- lmcache/integration/vllm/__init__.py +1 -0
- lmcache/integration/vllm/lmcache_connector_v1.py +213 -0
- lmcache/integration/vllm/lmcache_connector_v1_085.py +150 -0
- lmcache/integration/vllm/lmcache_mp_connector_0180.py +1072 -0
- lmcache/integration/vllm/tests/test_mm_hash_utils.py +112 -0
- lmcache/integration/vllm/utils.py +433 -0
- lmcache/integration/vllm/vllm_multi_process_adapter.py +1090 -0
- lmcache/integration/vllm/vllm_service_factory.py +339 -0
- lmcache/integration/vllm/vllm_v1_adapter.py +1713 -0
- lmcache/logging.py +107 -0
- lmcache/native_storage_ops.pyi +230 -0
- lmcache/non_cuda_equivalents.py +1424 -0
- lmcache/observability.py +1958 -0
- lmcache/storage_backend/serde/__init__.py +1 -0
- lmcache/storage_backend/serde/cachegen_basics.py +210 -0
- lmcache/storage_backend/serde/cachegen_decoder.py +207 -0
- lmcache/storage_backend/serde/cachegen_encoder.py +394 -0
- lmcache/storage_backend/serde/serde.py +75 -0
- lmcache/tools/__init__.py +1 -0
- lmcache/tools/cache_simulator/README.md +392 -0
- lmcache/tools/cache_simulator/__init__.py +1 -0
- lmcache/tools/cache_simulator/docs/simulate_example.png +0 -0
- lmcache/tools/cache_simulator/docs/sweep_example.png +0 -0
- lmcache/tools/cache_simulator/gen_bench_dataset.py +360 -0
- lmcache/tools/cache_simulator/lru_cache.py +124 -0
- lmcache/tools/cache_simulator/plot_hit_rate.py +231 -0
- lmcache/tools/cache_simulator/simulator.py +795 -0
- lmcache/tools/controller_benchmark/README.md +161 -0
- lmcache/tools/controller_benchmark/__init__.py +1 -0
- lmcache/tools/controller_benchmark/__main__.py +331 -0
- lmcache/tools/controller_benchmark/benchmark.py +660 -0
- lmcache/tools/controller_benchmark/config.py +44 -0
- lmcache/tools/controller_benchmark/constants.py +10 -0
- lmcache/tools/controller_benchmark/handlers/__init__.py +46 -0
- lmcache/tools/controller_benchmark/handlers/admit.py +52 -0
- lmcache/tools/controller_benchmark/handlers/base.py +47 -0
- lmcache/tools/controller_benchmark/handlers/deregister.py +49 -0
- lmcache/tools/controller_benchmark/handlers/evict.py +52 -0
- lmcache/tools/controller_benchmark/handlers/heartbeat.py +56 -0
- lmcache/tools/controller_benchmark/handlers/p2p_lookup.py +47 -0
- lmcache/tools/controller_benchmark/handlers/register.py +56 -0
- lmcache/tools/mp_status_viewer/__init__.py +1 -0
- lmcache/tools/mp_status_viewer/__main__.py +95 -0
- lmcache/usage_context.py +417 -0
- lmcache/utils.py +665 -0
- lmcache/v1/__init__.py +2 -0
- lmcache/v1/api_server/__init__.py +2 -0
- lmcache/v1/api_server/__main__.py +537 -0
- lmcache/v1/basic_check.py +112 -0
- lmcache/v1/cache_controller/__init__.py +9 -0
- lmcache/v1/cache_controller/commands/__init__.py +15 -0
- lmcache/v1/cache_controller/commands/base.py +35 -0
- lmcache/v1/cache_controller/commands/full_sync.py +49 -0
- lmcache/v1/cache_controller/config.py +176 -0
- lmcache/v1/cache_controller/controller_manager.py +535 -0
- lmcache/v1/cache_controller/controllers/__init__.py +11 -0
- lmcache/v1/cache_controller/controllers/full_sync_tracker.py +473 -0
- lmcache/v1/cache_controller/controllers/kv_controller.py +439 -0
- lmcache/v1/cache_controller/controllers/registration_controller.py +282 -0
- lmcache/v1/cache_controller/executor.py +463 -0
- lmcache/v1/cache_controller/frontend/static/css/style.css +201 -0
- lmcache/v1/cache_controller/frontend/static/img/logo.png +0 -0
- lmcache/v1/cache_controller/frontend/static/index.html +234 -0
- lmcache/v1/cache_controller/frontend/static/js/controller_app.js +660 -0
- lmcache/v1/cache_controller/full_sync_sender.py +475 -0
- lmcache/v1/cache_controller/locks.py +149 -0
- lmcache/v1/cache_controller/message.py +828 -0
- lmcache/v1/cache_controller/observability.py +208 -0
- lmcache/v1/cache_controller/utils.py +679 -0
- lmcache/v1/cache_controller/worker.py +665 -0
- lmcache/v1/cache_engine.py +2058 -0
- lmcache/v1/cache_interface.py +19 -0
- lmcache/v1/check/__init__.py +74 -0
- lmcache/v1/check/check_mode_gen.py +86 -0
- lmcache/v1/check/check_mode_test_l2_adapter.py +284 -0
- lmcache/v1/check/check_mode_test_remote.py +155 -0
- lmcache/v1/check/check_mode_test_storage_manager.py +142 -0
- lmcache/v1/check/utils.py +571 -0
- lmcache/v1/compute/__init__.py +2 -0
- lmcache/v1/compute/attention/__init__.py +0 -0
- lmcache/v1/compute/attention/abstract.py +39 -0
- lmcache/v1/compute/attention/flash_attn.py +129 -0
- lmcache/v1/compute/attention/flash_infer_sparse.py +284 -0
- lmcache/v1/compute/attention/metadata.py +85 -0
- lmcache/v1/compute/attention/utils.py +14 -0
- lmcache/v1/compute/blend/__init__.py +7 -0
- lmcache/v1/compute/blend/blender.py +168 -0
- lmcache/v1/compute/blend/metadata.py +34 -0
- lmcache/v1/compute/blend/utils.py +63 -0
- lmcache/v1/compute/models/__init__.py +0 -0
- lmcache/v1/compute/models/base.py +141 -0
- lmcache/v1/compute/models/llama.py +9 -0
- lmcache/v1/compute/models/qwen3.py +24 -0
- lmcache/v1/compute/models/utils.py +68 -0
- lmcache/v1/compute/positional_encoding.py +199 -0
- lmcache/v1/config.py +848 -0
- lmcache/v1/config_base.py +848 -0
- lmcache/v1/distributed/api.py +248 -0
- lmcache/v1/distributed/config.py +321 -0
- lmcache/v1/distributed/error.py +64 -0
- lmcache/v1/distributed/eviction.py +192 -0
- lmcache/v1/distributed/eviction_policy/__init__.py +21 -0
- lmcache/v1/distributed/eviction_policy/factory.py +27 -0
- lmcache/v1/distributed/eviction_policy/lru.py +244 -0
- lmcache/v1/distributed/eviction_policy/noop.py +50 -0
- lmcache/v1/distributed/internal_api.py +170 -0
- lmcache/v1/distributed/l1_manager.py +835 -0
- lmcache/v1/distributed/l2_adapters/__init__.py +67 -0
- lmcache/v1/distributed/l2_adapters/base.py +360 -0
- lmcache/v1/distributed/l2_adapters/config.py +385 -0
- lmcache/v1/distributed/l2_adapters/factory.py +205 -0
- lmcache/v1/distributed/l2_adapters/fs_l2_adapter.py +747 -0
- lmcache/v1/distributed/l2_adapters/fs_native_l2_adapter.py +167 -0
- lmcache/v1/distributed/l2_adapters/mock_l2_adapter.py +516 -0
- lmcache/v1/distributed/l2_adapters/mooncake_store_l2_adapter.py +135 -0
- lmcache/v1/distributed/l2_adapters/native_connector_l2_adapter.py +468 -0
- lmcache/v1/distributed/l2_adapters/native_plugin_l2_adapter.py +199 -0
- lmcache/v1/distributed/l2_adapters/nixl_store_dynamic_l2_adapter.py +831 -0
- lmcache/v1/distributed/l2_adapters/nixl_store_l2_adapter.py +983 -0
- lmcache/v1/distributed/l2_adapters/plugin_l2_adapter.py +210 -0
- lmcache/v1/distributed/l2_adapters/resp_l2_adapter.py +176 -0
- lmcache/v1/distributed/memory_manager.py +179 -0
- lmcache/v1/distributed/storage_controller.py +39 -0
- lmcache/v1/distributed/storage_controllers/__init__.py +43 -0
- lmcache/v1/distributed/storage_controllers/eviction_controller.py +242 -0
- lmcache/v1/distributed/storage_controllers/prefetch_controller.py +830 -0
- lmcache/v1/distributed/storage_controllers/prefetch_policy.py +193 -0
- lmcache/v1/distributed/storage_controllers/store_controller.py +452 -0
- lmcache/v1/distributed/storage_controllers/store_policy.py +213 -0
- lmcache/v1/distributed/storage_manager.py +532 -0
- lmcache/v1/event_manager.py +145 -0
- lmcache/v1/exceptions/__init__.py +16 -0
- lmcache/v1/gpu_connector/__init__.py +126 -0
- lmcache/v1/gpu_connector/gpu_connectors.py +1906 -0
- lmcache/v1/gpu_connector/gpu_ops.py +85 -0
- lmcache/v1/gpu_connector/hpu_connector.py +326 -0
- lmcache/v1/gpu_connector/mock_gpu_connector.py +67 -0
- lmcache/v1/gpu_connector/utils.py +890 -0
- lmcache/v1/gpu_connector/xpu_connectors.py +916 -0
- lmcache/v1/health_monitor/__init__.py +1 -0
- lmcache/v1/health_monitor/base.py +587 -0
- lmcache/v1/health_monitor/checks/__init__.py +1 -0
- lmcache/v1/health_monitor/checks/remote_backend_check.py +304 -0
- lmcache/v1/health_monitor/constants.py +36 -0
- lmcache/v1/internal_api_server/__init__.py +0 -0
- lmcache/v1/internal_api_server/api_registry.py +59 -0
- lmcache/v1/internal_api_server/api_server.py +120 -0
- lmcache/v1/internal_api_server/common/__init__.py +1 -0
- lmcache/v1/internal_api_server/common/env_api.py +22 -0
- lmcache/v1/internal_api_server/common/loglevel_api.py +57 -0
- lmcache/v1/internal_api_server/common/metrics_api.py +29 -0
- lmcache/v1/internal_api_server/common/periodic_thread_api.py +138 -0
- lmcache/v1/internal_api_server/common/run_script_api.py +73 -0
- lmcache/v1/internal_api_server/common/thread_api.py +63 -0
- lmcache/v1/internal_api_server/controller/__init__.py +1 -0
- lmcache/v1/internal_api_server/controller/key_stats_api.py +81 -0
- lmcache/v1/internal_api_server/controller/worker_info_api.py +136 -0
- lmcache/v1/internal_api_server/utils.py +43 -0
- lmcache/v1/internal_api_server/vllm/__init__.py +1 -0
- lmcache/v1/internal_api_server/vllm/backend_api.py +221 -0
- lmcache/v1/internal_api_server/vllm/bypass_api.py +204 -0
- lmcache/v1/internal_api_server/vllm/cache_api.py +895 -0
- lmcache/v1/internal_api_server/vllm/chunk_statistics_api.py +141 -0
- lmcache/v1/internal_api_server/vllm/conf_api.py +147 -0
- lmcache/v1/internal_api_server/vllm/freeze_api.py +172 -0
- lmcache/v1/internal_api_server/vllm/hot_cache_api.py +184 -0
- lmcache/v1/internal_api_server/vllm/inference_api.py +65 -0
- lmcache/v1/internal_api_server/vllm/load_fs_chunks_api.py +320 -0
- lmcache/v1/internal_api_server/vllm/lookup_api.py +145 -0
- lmcache/v1/internal_api_server/vllm/version_api.py +25 -0
- lmcache/v1/kv_layer_groups.py +267 -0
- lmcache/v1/lazy_memory_allocator.py +284 -0
- lmcache/v1/lookup_client/__init__.py +25 -0
- lmcache/v1/lookup_client/abstract_client.py +77 -0
- lmcache/v1/lookup_client/async_lookup_message.py +50 -0
- lmcache/v1/lookup_client/chunk_statistics_lookup_client.py +200 -0
- lmcache/v1/lookup_client/factory.py +251 -0
- lmcache/v1/lookup_client/hit_limit_lookup_client.py +86 -0
- lmcache/v1/lookup_client/lmcache_async_lookup_client.py +407 -0
- lmcache/v1/lookup_client/lmcache_lookup_client.py +285 -0
- lmcache/v1/lookup_client/lmcache_lookup_client_bypass.py +99 -0
- lmcache/v1/lookup_client/mooncake_lookup_client.py +87 -0
- lmcache/v1/lookup_client/record_strategies/__init__.py +77 -0
- lmcache/v1/lookup_client/record_strategies/base.py +327 -0
- lmcache/v1/lookup_client/record_strategies/file_hash.py +130 -0
- lmcache/v1/lookup_client/record_strategies/memory_bloom_filter.py +81 -0
- lmcache/v1/manager.py +539 -0
- lmcache/v1/memory_management.py +2619 -0
- lmcache/v1/metadata.py +114 -0
- lmcache/v1/mp_observability/AGENTS.override.md +21 -0
- lmcache/v1/mp_observability/README.md +204 -0
- lmcache/v1/mp_observability/config.py +340 -0
- lmcache/v1/mp_observability/event.py +100 -0
- lmcache/v1/mp_observability/event_bus.py +313 -0
- lmcache/v1/mp_observability/otel_init.py +145 -0
- lmcache/v1/mp_observability/subscribers/__init__.py +28 -0
- lmcache/v1/mp_observability/subscribers/logging/__init__.py +19 -0
- lmcache/v1/mp_observability/subscribers/logging/l1.py +56 -0
- lmcache/v1/mp_observability/subscribers/logging/l2.py +73 -0
- lmcache/v1/mp_observability/subscribers/logging/lookup_hash.py +209 -0
- lmcache/v1/mp_observability/subscribers/logging/mp_server.py +90 -0
- lmcache/v1/mp_observability/subscribers/logging/sm.py +59 -0
- lmcache/v1/mp_observability/subscribers/metrics/__init__.py +20 -0
- lmcache/v1/mp_observability/subscribers/metrics/l0_lifecycle.py +290 -0
- lmcache/v1/mp_observability/subscribers/metrics/l1.py +55 -0
- lmcache/v1/mp_observability/subscribers/metrics/l1_lifecycle.py +166 -0
- lmcache/v1/mp_observability/subscribers/metrics/l2.py +121 -0
- lmcache/v1/mp_observability/subscribers/metrics/sm.py +69 -0
- lmcache/v1/mp_observability/subscribers/tracing/__init__.py +12 -0
- lmcache/v1/mp_observability/subscribers/tracing/mp_server.py +333 -0
- lmcache/v1/mp_observability/subscribers/tracing/span_registry.py +148 -0
- lmcache/v1/mp_observability/trace/__init__.py +50 -0
- lmcache/v1/mp_observability/trace/codecs.py +255 -0
- lmcache/v1/mp_observability/trace/decorator.py +147 -0
- lmcache/v1/mp_observability/trace/format.py +132 -0
- lmcache/v1/mp_observability/trace/lifecycle.py +83 -0
- lmcache/v1/mp_observability/trace/reader.py +167 -0
- lmcache/v1/mp_observability/trace/recorder.py +300 -0
- lmcache/v1/multiprocess/__init__.py +0 -0
- lmcache/v1/multiprocess/affinity_pool.py +102 -0
- lmcache/v1/multiprocess/blend_server_v2.py +891 -0
- lmcache/v1/multiprocess/config.py +253 -0
- lmcache/v1/multiprocess/custom_types.py +281 -0
- lmcache/v1/multiprocess/futures.py +194 -0
- lmcache/v1/multiprocess/gpu_context.py +511 -0
- lmcache/v1/multiprocess/http_server.py +235 -0
- lmcache/v1/multiprocess/mp_runtime_plugin_launcher.py +130 -0
- lmcache/v1/multiprocess/mq.py +732 -0
- lmcache/v1/multiprocess/protocol.py +86 -0
- lmcache/v1/multiprocess/protocols/README.md +213 -0
- lmcache/v1/multiprocess/protocols/__init__.py +127 -0
- lmcache/v1/multiprocess/protocols/base.py +89 -0
- lmcache/v1/multiprocess/protocols/blend.py +109 -0
- lmcache/v1/multiprocess/protocols/blend_v2.py +57 -0
- lmcache/v1/multiprocess/protocols/controller.py +53 -0
- lmcache/v1/multiprocess/protocols/debug.py +34 -0
- lmcache/v1/multiprocess/protocols/engine.py +146 -0
- lmcache/v1/multiprocess/protocols/observability.py +39 -0
- lmcache/v1/multiprocess/server.py +1134 -0
- lmcache/v1/multiprocess/session.py +190 -0
- lmcache/v1/multiprocess/token_hasher.py +441 -0
- lmcache/v1/offload_server/__init__.py +17 -0
- lmcache/v1/offload_server/abstract_server.py +37 -0
- lmcache/v1/offload_server/message.py +30 -0
- lmcache/v1/offload_server/zmq_server.py +122 -0
- lmcache/v1/periodic_thread.py +579 -0
- lmcache/v1/pin_monitor.py +246 -0
- lmcache/v1/plugin/__init__.py +0 -0
- lmcache/v1/plugin/runtime_plugin_launcher.py +211 -0
- lmcache/v1/protocol.py +317 -0
- lmcache/v1/rpc/__init__.py +17 -0
- lmcache/v1/rpc/transport.py +105 -0
- lmcache/v1/rpc/zmq_transport.py +213 -0
- lmcache/v1/rpc_utils.py +165 -0
- lmcache/v1/server/__init__.py +2 -0
- lmcache/v1/server/__main__.py +170 -0
- lmcache/v1/server/storage_backend/__init__.py +21 -0
- lmcache/v1/server/storage_backend/abstract_backend.py +80 -0
- lmcache/v1/server/storage_backend/local_backend.py +75 -0
- lmcache/v1/server/utils.py +21 -0
- lmcache/v1/standalone/__init__.py +1 -0
- lmcache/v1/standalone/__main__.py +583 -0
- lmcache/v1/standalone/manager.py +80 -0
- lmcache/v1/standalone/standalone_service_factory.py +86 -0
- lmcache/v1/storage_backend/__init__.py +313 -0
- lmcache/v1/storage_backend/abstract_backend.py +445 -0
- lmcache/v1/storage_backend/audit_backend.py +233 -0
- lmcache/v1/storage_backend/batched_message_sender.py +222 -0
- lmcache/v1/storage_backend/cache_policy/__init__.py +45 -0
- lmcache/v1/storage_backend/cache_policy/base_policy.py +87 -0
- lmcache/v1/storage_backend/cache_policy/fifo.py +58 -0
- lmcache/v1/storage_backend/cache_policy/lfu.py +105 -0
- lmcache/v1/storage_backend/cache_policy/lru.py +81 -0
- lmcache/v1/storage_backend/cache_policy/mru.py +61 -0
- lmcache/v1/storage_backend/connector/__init__.py +443 -0
- lmcache/v1/storage_backend/connector/audit_adapter.py +77 -0
- lmcache/v1/storage_backend/connector/audit_connector.py +320 -0
- lmcache/v1/storage_backend/connector/base_connector.py +379 -0
- lmcache/v1/storage_backend/connector/blackhole_adapter.py +21 -0
- lmcache/v1/storage_backend/connector/blackhole_connector.py +37 -0
- lmcache/v1/storage_backend/connector/eic_adapter.py +31 -0
- lmcache/v1/storage_backend/connector/eic_connector.py +757 -0
- lmcache/v1/storage_backend/connector/external_adapter.py +79 -0
- lmcache/v1/storage_backend/connector/fs_adapter.py +51 -0
- lmcache/v1/storage_backend/connector/fs_connector.py +403 -0
- lmcache/v1/storage_backend/connector/infinistore_adapter.py +56 -0
- lmcache/v1/storage_backend/connector/infinistore_connector.py +177 -0
- lmcache/v1/storage_backend/connector/instrumented_connector.py +219 -0
- lmcache/v1/storage_backend/connector/lm_adapter.py +31 -0
- lmcache/v1/storage_backend/connector/lm_connector.py +176 -0
- lmcache/v1/storage_backend/connector/mock_adapter.py +57 -0
- lmcache/v1/storage_backend/connector/mock_connector.py +349 -0
- lmcache/v1/storage_backend/connector/mooncakestore_adapter.py +43 -0
- lmcache/v1/storage_backend/connector/mooncakestore_connector.py +614 -0
- lmcache/v1/storage_backend/connector/redis_adapter.py +181 -0
- lmcache/v1/storage_backend/connector/redis_connector.py +828 -0
- lmcache/v1/storage_backend/connector/s3_adapter.py +59 -0
- lmcache/v1/storage_backend/connector/s3_connector.py +699 -0
- lmcache/v1/storage_backend/connector/sagemaker_hyperpod_adapter.py +233 -0
- lmcache/v1/storage_backend/connector/sagemaker_hyperpod_connector.py +987 -0
- lmcache/v1/storage_backend/connector/valkey_adapter.py +114 -0
- lmcache/v1/storage_backend/connector/valkey_connector.py +627 -0
- lmcache/v1/storage_backend/gds_backend.py +1199 -0
- lmcache/v1/storage_backend/job_executor/__init__.py +0 -0
- lmcache/v1/storage_backend/job_executor/base_executor.py +34 -0
- lmcache/v1/storage_backend/job_executor/pq_executor.py +235 -0
- lmcache/v1/storage_backend/local_cpu_backend.py +810 -0
- lmcache/v1/storage_backend/local_disk_backend.py +656 -0
- lmcache/v1/storage_backend/maru_backend.py +734 -0
- lmcache/v1/storage_backend/naive_serde/__init__.py +50 -0
- lmcache/v1/storage_backend/naive_serde/cachegen_basics.py +133 -0
- lmcache/v1/storage_backend/naive_serde/cachegen_decoder.py +135 -0
- lmcache/v1/storage_backend/naive_serde/cachegen_encoder.py +83 -0
- lmcache/v1/storage_backend/naive_serde/kivi_serde.py +22 -0
- lmcache/v1/storage_backend/naive_serde/naive_serde.py +18 -0
- lmcache/v1/storage_backend/naive_serde/serde.py +37 -0
- lmcache/v1/storage_backend/native_clients/connector_client_base.py +165 -0
- lmcache/v1/storage_backend/native_clients/resp_client.py +35 -0
- lmcache/v1/storage_backend/nixl_storage_backend.py +1400 -0
- lmcache/v1/storage_backend/p2p_backend.py +788 -0
- lmcache/v1/storage_backend/path_sharder.py +117 -0
- lmcache/v1/storage_backend/pd_backend.py +646 -0
- lmcache/v1/storage_backend/plugins/dax_backend.py +1443 -0
- lmcache/v1/storage_backend/plugins/rust_raw_block_backend.py +1361 -0
- lmcache/v1/storage_backend/remote_backend.py +624 -0
- lmcache/v1/storage_backend/resp_client.py +227 -0
- lmcache/v1/storage_backend/storage_backend_listener.py +19 -0
- lmcache/v1/storage_backend/storage_manager.py +1352 -0
- lmcache/v1/system_detection.py +110 -0
- lmcache/v1/token_database.py +551 -0
- lmcache/v1/transfer_channel/__init__.py +83 -0
- lmcache/v1/transfer_channel/abstract.py +285 -0
- lmcache/v1/transfer_channel/mock_memory_channel.py +156 -0
- lmcache/v1/transfer_channel/nixl_channel.py +639 -0
- lmcache/v1/transfer_channel/py_socket_channel.py +260 -0
- lmcache/v1/transfer_channel/transfer_utils.py +63 -0
- lmcache/v1/utils/__init__.py +1 -0
- lmcache/v1/utils/bloom_filter.py +109 -0
- lmcache/v1/utils/cache_utils.py +125 -0
- lmcache_cli-0.4.5.dev0.dist-info/METADATA +185 -0
- lmcache_cli-0.4.5.dev0.dist-info/RECORD +399 -0
- lmcache_cli-0.4.5.dev0.dist-info/WHEEL +5 -0
- lmcache_cli-0.4.5.dev0.dist-info/entry_points.txt +2 -0
- lmcache_cli-0.4.5.dev0.dist-info/licenses/LICENSE +201 -0
- lmcache_cli-0.4.5.dev0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,222 @@
|
|
|
1
|
+
# SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
# Standard
|
|
3
|
+
from typing import TYPE_CHECKING, List, Optional
|
|
4
|
+
import queue
|
|
5
|
+
import threading
|
|
6
|
+
|
|
7
|
+
# First Party
|
|
8
|
+
from lmcache.logging import init_logger
|
|
9
|
+
from lmcache.observability import PrometheusLogger
|
|
10
|
+
from lmcache.v1.cache_controller.message import (
|
|
11
|
+
BatchedKVOperationMsg,
|
|
12
|
+
KVOpEvent,
|
|
13
|
+
OpType,
|
|
14
|
+
)
|
|
15
|
+
from lmcache.v1.config import LMCacheEngineConfig
|
|
16
|
+
from lmcache.v1.metadata import LMCacheMetadata
|
|
17
|
+
|
|
18
|
+
if TYPE_CHECKING:
|
|
19
|
+
# First Party
|
|
20
|
+
from lmcache.v1.cache_controller.worker import LMCacheWorker
|
|
21
|
+
|
|
22
|
+
logger = init_logger(__name__)
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
class BatchedMessageSender:
|
|
26
|
+
"""
|
|
27
|
+
Batched message sender for KVOperation.
|
|
28
|
+
|
|
29
|
+
This class accumulates KV admit/evict messages and sends them in batches
|
|
30
|
+
to reduce communication overhead. Messages are flushed when either:
|
|
31
|
+
1. The batch size threshold is reached (default: 50 messages)
|
|
32
|
+
2. The timeout period expires (default: 0.01 seconds)
|
|
33
|
+
|
|
34
|
+
Each message is assigned a unique, monotonically increasing sequence number
|
|
35
|
+
to enable the receiver to detect missing or out-of-order messages.
|
|
36
|
+
|
|
37
|
+
Design rationale:
|
|
38
|
+
- Uses a SINGLE queue for both admit and evict messages to maintain strict
|
|
39
|
+
order consistency. This is critical because operations like
|
|
40
|
+
admit(key) -> evict(key) -> admit(key) must be processed in exact order
|
|
41
|
+
to avoid race conditions and state inconsistencies on the receiver side.
|
|
42
|
+
|
|
43
|
+
Thread-safe: Uses locks to protect internal queue and sequence counter.
|
|
44
|
+
|
|
45
|
+
Args:
|
|
46
|
+
metadata: Metadata for the worker
|
|
47
|
+
config: Configuration for the worker
|
|
48
|
+
location: Location of the worker
|
|
49
|
+
lmcache_worker: The worker to send messages to. If None, batching is disabled.
|
|
50
|
+
"""
|
|
51
|
+
|
|
52
|
+
def __init__(
|
|
53
|
+
self,
|
|
54
|
+
metadata: LMCacheMetadata,
|
|
55
|
+
config: LMCacheEngineConfig,
|
|
56
|
+
location: str,
|
|
57
|
+
lmcache_worker: "LMCacheWorker",
|
|
58
|
+
):
|
|
59
|
+
self.batch_size = config.get_extra_config_value("kv_msg_batch_size", 50)
|
|
60
|
+
self.batch_timeout = config.get_extra_config_value("kv_msg_batch_timeout", 0.01)
|
|
61
|
+
self.lmcache_worker = lmcache_worker
|
|
62
|
+
|
|
63
|
+
# Common fields shared by all operations in the batch
|
|
64
|
+
self.instance_id = config.lmcache_instance_id
|
|
65
|
+
self.worker_id = metadata.worker_id
|
|
66
|
+
self.location = location
|
|
67
|
+
|
|
68
|
+
# Use thread-safe queue for producer-consumer pattern
|
|
69
|
+
self.message_queue: queue.Queue[KVOpEvent] = queue.Queue()
|
|
70
|
+
self.sequence_number = 0
|
|
71
|
+
self.sequence_lock = threading.Lock()
|
|
72
|
+
|
|
73
|
+
# Condition variable for coordinating producer and consumer
|
|
74
|
+
self.cv = threading.Condition()
|
|
75
|
+
self.running = False
|
|
76
|
+
self.thread: Optional[threading.Thread] = None
|
|
77
|
+
|
|
78
|
+
self._start_background_thread()
|
|
79
|
+
|
|
80
|
+
self._setup_metrics()
|
|
81
|
+
|
|
82
|
+
def _setup_metrics(self):
|
|
83
|
+
"""Setup metrics for monitoring queue size."""
|
|
84
|
+
prometheus_logger = PrometheusLogger.GetInstanceOrNone()
|
|
85
|
+
if prometheus_logger is not None:
|
|
86
|
+
prometheus_logger.kv_msg_queue_size.set_function(
|
|
87
|
+
lambda: self.message_queue.qsize()
|
|
88
|
+
)
|
|
89
|
+
|
|
90
|
+
def _start_background_thread(self):
|
|
91
|
+
"""Start background thread for periodic flushing."""
|
|
92
|
+
self.running = True
|
|
93
|
+
self.thread = threading.Thread(
|
|
94
|
+
target=self._consumer_loop, daemon=True, name="batched-msg-sender-thread"
|
|
95
|
+
)
|
|
96
|
+
self.thread.start()
|
|
97
|
+
|
|
98
|
+
def _consumer_loop(self):
|
|
99
|
+
"""Consumer loop that drains queue and sends batched messages."""
|
|
100
|
+
while self.running:
|
|
101
|
+
with self.cv:
|
|
102
|
+
# Wait for timeout or notification from producer
|
|
103
|
+
self.cv.wait(timeout=self.batch_timeout)
|
|
104
|
+
|
|
105
|
+
# Check if we have messages to process while holding the lock
|
|
106
|
+
# This prevents race conditions but we'll release lock
|
|
107
|
+
# before blocking operations
|
|
108
|
+
if self.message_queue.empty():
|
|
109
|
+
continue
|
|
110
|
+
|
|
111
|
+
# Drain the queue without holding the lock to avoid blocking producers
|
|
112
|
+
# This improves performance during the actual message processing
|
|
113
|
+
self._drain_and_send()
|
|
114
|
+
|
|
115
|
+
def _get_next_sequence_number(self) -> int:
|
|
116
|
+
"""Get next sequence number for message tracking.
|
|
117
|
+
|
|
118
|
+
Thread-safe: Uses dedicated lock for sequence number generation.
|
|
119
|
+
"""
|
|
120
|
+
with self.sequence_lock:
|
|
121
|
+
seq = self.sequence_number
|
|
122
|
+
self.sequence_number += 1
|
|
123
|
+
return seq
|
|
124
|
+
|
|
125
|
+
def add_kv_op(
|
|
126
|
+
self,
|
|
127
|
+
op_type: OpType,
|
|
128
|
+
key: int,
|
|
129
|
+
):
|
|
130
|
+
"""Add a KV operation to the batch queue.
|
|
131
|
+
|
|
132
|
+
Producer method: Adds operation to queue and notifies consumer
|
|
133
|
+
when batch size threshold is reached.
|
|
134
|
+
|
|
135
|
+
Args:
|
|
136
|
+
op_type: Operation type (ADMIT or EVICT)
|
|
137
|
+
key: Chunk hash key
|
|
138
|
+
"""
|
|
139
|
+
# Create operation without sequence number (will be assigned during drain)
|
|
140
|
+
op = KVOpEvent(op_type=op_type, key=key, seq_num=-1)
|
|
141
|
+
|
|
142
|
+
# Thread-safe queue put
|
|
143
|
+
self.message_queue.put(op)
|
|
144
|
+
|
|
145
|
+
# Notify consumer if batch size threshold is reached
|
|
146
|
+
if self.message_queue.qsize() >= self.batch_size:
|
|
147
|
+
with self.cv:
|
|
148
|
+
self.cv.notify()
|
|
149
|
+
|
|
150
|
+
def _drain_and_send(self):
|
|
151
|
+
"""Drain the queue and send all messages in a batch.
|
|
152
|
+
|
|
153
|
+
This method is called by the consumer thread to collect all pending
|
|
154
|
+
operations from the queue and send them as a single batched message.
|
|
155
|
+
"""
|
|
156
|
+
ops_to_send: List[KVOpEvent] = []
|
|
157
|
+
|
|
158
|
+
# Drain all messages from the queue using blocking get with timeout
|
|
159
|
+
# This ensures we don't miss any messages due to race conditions
|
|
160
|
+
while True:
|
|
161
|
+
try:
|
|
162
|
+
# Use a small timeout to avoid blocking indefinitely
|
|
163
|
+
op = self.message_queue.get(timeout=0.001)
|
|
164
|
+
# Assign sequence number at drain time to ensure strict ordering
|
|
165
|
+
op.seq_num = self._get_next_sequence_number()
|
|
166
|
+
ops_to_send.append(op)
|
|
167
|
+
except queue.Empty:
|
|
168
|
+
# Queue is empty, break the loop
|
|
169
|
+
break
|
|
170
|
+
|
|
171
|
+
if not ops_to_send:
|
|
172
|
+
return
|
|
173
|
+
|
|
174
|
+
try:
|
|
175
|
+
# Ensure common fields are set
|
|
176
|
+
assert self.instance_id is not None, "instance_id must be set"
|
|
177
|
+
assert self.worker_id is not None, "worker_id must be set"
|
|
178
|
+
assert self.location is not None, "location must be set"
|
|
179
|
+
|
|
180
|
+
# Create batched message with common fields and lightweight operations
|
|
181
|
+
# This reduces redundancy: common fields are sent once instead of N times
|
|
182
|
+
batched_msg = BatchedKVOperationMsg(
|
|
183
|
+
instance_id=self.instance_id,
|
|
184
|
+
worker_id=self.worker_id,
|
|
185
|
+
location=self.location,
|
|
186
|
+
operations=ops_to_send,
|
|
187
|
+
)
|
|
188
|
+
self.lmcache_worker.put_msg(batched_msg)
|
|
189
|
+
finally:
|
|
190
|
+
# Mark all tasks as done regardless of success/failure
|
|
191
|
+
# This ensures flush() doesn't hang if put_msg fails
|
|
192
|
+
for _ in ops_to_send:
|
|
193
|
+
self.message_queue.task_done()
|
|
194
|
+
|
|
195
|
+
def flush(self):
|
|
196
|
+
"""Manually flush all pending messages.
|
|
197
|
+
|
|
198
|
+
This method ensures all pending messages in the queue are processed
|
|
199
|
+
before returning. It triggers the consumer thread and waits for the
|
|
200
|
+
queue to be empty.
|
|
201
|
+
"""
|
|
202
|
+
with self.cv:
|
|
203
|
+
self.cv.notify()
|
|
204
|
+
|
|
205
|
+
self.message_queue.join()
|
|
206
|
+
|
|
207
|
+
def close(self):
|
|
208
|
+
"""Close the batched message sender and flush remaining messages."""
|
|
209
|
+
self.flush()
|
|
210
|
+
self.running = False
|
|
211
|
+
|
|
212
|
+
# Wake up consumer thread to exit
|
|
213
|
+
with self.cv:
|
|
214
|
+
self.cv.notify()
|
|
215
|
+
|
|
216
|
+
# Wait for thread to finish
|
|
217
|
+
if self.thread is not None and self.thread.is_alive():
|
|
218
|
+
self.thread.join(timeout=1.0)
|
|
219
|
+
if self.thread.is_alive():
|
|
220
|
+
logger.warning(
|
|
221
|
+
"Batched message sender thread did not terminate within timeout"
|
|
222
|
+
)
|
|
@@ -0,0 +1,45 @@
|
|
|
1
|
+
# SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
# Standard
|
|
3
|
+
from typing import Dict, Type
|
|
4
|
+
|
|
5
|
+
# First Party
|
|
6
|
+
from lmcache.v1.storage_backend.cache_policy.base_policy import BaseCachePolicy
|
|
7
|
+
from lmcache.v1.storage_backend.cache_policy.fifo import FIFOCachePolicy
|
|
8
|
+
from lmcache.v1.storage_backend.cache_policy.lfu import LFUCachePolicy
|
|
9
|
+
from lmcache.v1.storage_backend.cache_policy.lru import LRUCachePolicy
|
|
10
|
+
from lmcache.v1.storage_backend.cache_policy.mru import MRUCachePolicy
|
|
11
|
+
|
|
12
|
+
# Cache policy mapping
|
|
13
|
+
POLICY_MAPPING: Dict[str, Type[BaseCachePolicy]] = {
|
|
14
|
+
"LRU": LRUCachePolicy,
|
|
15
|
+
"LFU": LFUCachePolicy,
|
|
16
|
+
"FIFO": FIFOCachePolicy,
|
|
17
|
+
"MRU": MRUCachePolicy,
|
|
18
|
+
}
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def get_cache_policy(policy_name: str) -> BaseCachePolicy:
|
|
22
|
+
"""
|
|
23
|
+
Factory function to get the cache policy instance based on the policy name.
|
|
24
|
+
|
|
25
|
+
Args:
|
|
26
|
+
policy_name: Name of the cache policy (case-insensitive, e.g., "LRU", "lru").
|
|
27
|
+
|
|
28
|
+
Returns:
|
|
29
|
+
Instance of the corresponding cache policy.
|
|
30
|
+
|
|
31
|
+
Raises:
|
|
32
|
+
ValueError: If the policy name is not supported.
|
|
33
|
+
"""
|
|
34
|
+
if not policy_name:
|
|
35
|
+
raise ValueError("Cache policy name cannot be empty")
|
|
36
|
+
|
|
37
|
+
upper_policy_name = policy_name.upper()
|
|
38
|
+
|
|
39
|
+
try:
|
|
40
|
+
return POLICY_MAPPING[upper_policy_name]()
|
|
41
|
+
except KeyError:
|
|
42
|
+
raise ValueError(
|
|
43
|
+
f"Unknown cache policy: {upper_policy_name}."
|
|
44
|
+
f" Supported policies are: {list(POLICY_MAPPING.keys())}"
|
|
45
|
+
) from None
|
|
@@ -0,0 +1,87 @@
|
|
|
1
|
+
# SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
# Standard
|
|
3
|
+
from collections.abc import MutableMapping
|
|
4
|
+
from typing import Generic, TypeVar
|
|
5
|
+
import abc
|
|
6
|
+
|
|
7
|
+
KeyType = TypeVar("KeyType")
|
|
8
|
+
MapType = TypeVar("MapType", bound=MutableMapping)
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
class BaseCachePolicy(Generic[KeyType, MapType], metaclass=abc.ABCMeta):
|
|
12
|
+
"""
|
|
13
|
+
Interface for cache policy.
|
|
14
|
+
"""
|
|
15
|
+
|
|
16
|
+
@abc.abstractmethod
|
|
17
|
+
def init_mutable_mapping(self) -> MapType:
|
|
18
|
+
"""
|
|
19
|
+
Initialize a mutable mapping for cache storage.
|
|
20
|
+
|
|
21
|
+
Return:
|
|
22
|
+
A mutable mapping that can be used to store cache entries.
|
|
23
|
+
"""
|
|
24
|
+
raise NotImplementedError
|
|
25
|
+
|
|
26
|
+
# TODO(Jiayi): we need to unify the `Any` type in the `MutableMapping`
|
|
27
|
+
@abc.abstractmethod
|
|
28
|
+
def update_on_hit(
|
|
29
|
+
self,
|
|
30
|
+
key: KeyType,
|
|
31
|
+
cache_dict: MapType,
|
|
32
|
+
) -> None:
|
|
33
|
+
"""
|
|
34
|
+
Update cache_dict and internal states when a cache is used
|
|
35
|
+
|
|
36
|
+
Input:
|
|
37
|
+
key: an object of KeyType
|
|
38
|
+
cache_dict: a dict consists of current cache
|
|
39
|
+
"""
|
|
40
|
+
raise NotImplementedError
|
|
41
|
+
|
|
42
|
+
# TODO(Jiayi): we need to unify the `Any` type in the `MutableMapping`
|
|
43
|
+
@abc.abstractmethod
|
|
44
|
+
def update_on_put(
|
|
45
|
+
self,
|
|
46
|
+
key: KeyType,
|
|
47
|
+
) -> None:
|
|
48
|
+
"""
|
|
49
|
+
Update cache_dict and internal states when a cache is stored
|
|
50
|
+
|
|
51
|
+
Input:
|
|
52
|
+
key: an object of KeyType
|
|
53
|
+
"""
|
|
54
|
+
raise NotImplementedError
|
|
55
|
+
|
|
56
|
+
# TODO(Jiayi): we need to unify the `Any` type in the `MutableMapping`
|
|
57
|
+
@abc.abstractmethod
|
|
58
|
+
def update_on_force_evict(
|
|
59
|
+
self,
|
|
60
|
+
key: KeyType,
|
|
61
|
+
) -> None:
|
|
62
|
+
"""
|
|
63
|
+
Update internal states when a cache is force evicted
|
|
64
|
+
|
|
65
|
+
Input:
|
|
66
|
+
key: an object of KeyType
|
|
67
|
+
"""
|
|
68
|
+
raise NotImplementedError
|
|
69
|
+
|
|
70
|
+
# TODO(Jiayi): we need to unify the `Any` type in the `MutableMapping`
|
|
71
|
+
@abc.abstractmethod
|
|
72
|
+
def get_evict_candidates(
|
|
73
|
+
self,
|
|
74
|
+
cache_dict: MapType,
|
|
75
|
+
num_candidates: int = 1,
|
|
76
|
+
) -> list[KeyType]:
|
|
77
|
+
"""
|
|
78
|
+
Evict cache when a new cache comes and the storage is full
|
|
79
|
+
|
|
80
|
+
Input:
|
|
81
|
+
cache_dict: a dict consists of current cache
|
|
82
|
+
num_candidates: number of candidates to be evicted
|
|
83
|
+
|
|
84
|
+
Return:
|
|
85
|
+
return a list of keys to be evicted
|
|
86
|
+
"""
|
|
87
|
+
raise NotImplementedError
|
|
@@ -0,0 +1,58 @@
|
|
|
1
|
+
# SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
# Standard
|
|
3
|
+
from typing import Any
|
|
4
|
+
|
|
5
|
+
# First Party
|
|
6
|
+
from lmcache.logging import init_logger
|
|
7
|
+
from lmcache.v1.storage_backend.cache_policy.base_policy import BaseCachePolicy, KeyType
|
|
8
|
+
|
|
9
|
+
logger = init_logger(__name__)
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
class FIFOCachePolicy(BaseCachePolicy[KeyType, dict[KeyType, Any]]):
|
|
13
|
+
"""
|
|
14
|
+
FIFO cache policy.
|
|
15
|
+
"""
|
|
16
|
+
|
|
17
|
+
def __init__(self):
|
|
18
|
+
logger.info("Initializing FIFOCachePolicy")
|
|
19
|
+
|
|
20
|
+
def init_mutable_mapping(self) -> dict[KeyType, Any]:
|
|
21
|
+
# NOTE(Jiayi): python dict maintains insertion order.
|
|
22
|
+
return {}
|
|
23
|
+
|
|
24
|
+
def update_on_hit(
|
|
25
|
+
self,
|
|
26
|
+
key: KeyType,
|
|
27
|
+
cache_dict: dict[KeyType, Any],
|
|
28
|
+
) -> None:
|
|
29
|
+
pass
|
|
30
|
+
|
|
31
|
+
def update_on_put(
|
|
32
|
+
self,
|
|
33
|
+
key: KeyType,
|
|
34
|
+
) -> None:
|
|
35
|
+
pass
|
|
36
|
+
|
|
37
|
+
def update_on_force_evict(
|
|
38
|
+
self,
|
|
39
|
+
key: KeyType,
|
|
40
|
+
) -> None:
|
|
41
|
+
pass
|
|
42
|
+
|
|
43
|
+
# NOTE(Jiayi): We do best effort to get eviction candidates so the number
|
|
44
|
+
# of returned keys mignt be smaller than num_candidates.
|
|
45
|
+
def get_evict_candidates(
|
|
46
|
+
self,
|
|
47
|
+
cache_dict: dict[KeyType, Any],
|
|
48
|
+
num_candidates: int = 1,
|
|
49
|
+
) -> list[KeyType]:
|
|
50
|
+
evict_keys = []
|
|
51
|
+
for key, cache in cache_dict.items():
|
|
52
|
+
if not cache.can_evict:
|
|
53
|
+
continue
|
|
54
|
+
evict_keys.append(key)
|
|
55
|
+
if len(evict_keys) == num_candidates:
|
|
56
|
+
break
|
|
57
|
+
|
|
58
|
+
return evict_keys
|
|
@@ -0,0 +1,105 @@
|
|
|
1
|
+
# SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
|
|
3
|
+
# Standard
|
|
4
|
+
from typing import Any
|
|
5
|
+
|
|
6
|
+
# Third Party
|
|
7
|
+
from sortedcontainers import SortedDict
|
|
8
|
+
|
|
9
|
+
# First Party
|
|
10
|
+
from lmcache.logging import init_logger
|
|
11
|
+
from lmcache.v1.storage_backend.cache_policy.base_policy import BaseCachePolicy, KeyType
|
|
12
|
+
|
|
13
|
+
logger = init_logger(__name__)
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
class LFUCachePolicy(BaseCachePolicy[KeyType, dict[KeyType, Any]]):
|
|
17
|
+
"""
|
|
18
|
+
LFU cache policy.
|
|
19
|
+
"""
|
|
20
|
+
|
|
21
|
+
# NOTE(Jiayi): We use `sorted dict` + `bucket` to implement LFU.
|
|
22
|
+
# NOTE(Jiayi): We use FIFO for entries with the same frequency.
|
|
23
|
+
def __init__(self):
|
|
24
|
+
# TODO(Jiayi): `SortedDict` is log(N).
|
|
25
|
+
# A way to make it O(1) is to use a dict and keep track min freuency.
|
|
26
|
+
# However, this requires us keep another data structures to keep track
|
|
27
|
+
# of the pinned keys.
|
|
28
|
+
self.freq_to_keys = SortedDict()
|
|
29
|
+
|
|
30
|
+
# TODO(Jiayi): We can optimize this a bit by using `key_to_val_freq`
|
|
31
|
+
self.key_to_freq = {}
|
|
32
|
+
|
|
33
|
+
logger.info("Initializing LFUCachePolicy")
|
|
34
|
+
|
|
35
|
+
def init_mutable_mapping(self) -> dict[KeyType, Any]:
|
|
36
|
+
return {}
|
|
37
|
+
|
|
38
|
+
def update_on_hit(
|
|
39
|
+
self,
|
|
40
|
+
key: KeyType,
|
|
41
|
+
cache_dict: dict[KeyType, Any],
|
|
42
|
+
) -> None:
|
|
43
|
+
curr_freq = self.key_to_freq[key]
|
|
44
|
+
self.freq_to_keys[curr_freq].pop(key)
|
|
45
|
+
if not self.freq_to_keys[curr_freq]:
|
|
46
|
+
self.freq_to_keys.pop(curr_freq)
|
|
47
|
+
|
|
48
|
+
curr_freq += 1
|
|
49
|
+
self.key_to_freq[key] = curr_freq
|
|
50
|
+
|
|
51
|
+
if curr_freq not in self.freq_to_keys:
|
|
52
|
+
self.freq_to_keys[curr_freq] = {key: None}
|
|
53
|
+
else:
|
|
54
|
+
self.freq_to_keys[curr_freq][key] = None
|
|
55
|
+
|
|
56
|
+
def update_on_put(
|
|
57
|
+
self,
|
|
58
|
+
key: KeyType,
|
|
59
|
+
) -> None:
|
|
60
|
+
# Initialize the frequency for the new key.
|
|
61
|
+
self.key_to_freq[key] = 1
|
|
62
|
+
if 1 not in self.freq_to_keys:
|
|
63
|
+
self.freq_to_keys[1] = {key: None}
|
|
64
|
+
else:
|
|
65
|
+
self.freq_to_keys[1][key] = None
|
|
66
|
+
|
|
67
|
+
def update_on_force_evict(
|
|
68
|
+
self,
|
|
69
|
+
key: KeyType,
|
|
70
|
+
) -> None:
|
|
71
|
+
freq = self.key_to_freq.pop(key, None)
|
|
72
|
+
if not freq:
|
|
73
|
+
return
|
|
74
|
+
self.freq_to_keys[freq].pop(key)
|
|
75
|
+
if not self.freq_to_keys[freq]:
|
|
76
|
+
self.freq_to_keys.pop(freq)
|
|
77
|
+
|
|
78
|
+
# NOTE(Jiayi): We do best effort to get eviction candidates so the number
|
|
79
|
+
# of returned keys mignt be smaller than num_candidates.
|
|
80
|
+
def get_evict_candidates(
|
|
81
|
+
self,
|
|
82
|
+
cache_dict: dict[KeyType, Any],
|
|
83
|
+
num_candidates: int = 1,
|
|
84
|
+
) -> list[KeyType]:
|
|
85
|
+
evict_keys = []
|
|
86
|
+
evict_freqs = []
|
|
87
|
+
for curr_min_freq, fifo_keys in self.freq_to_keys.items():
|
|
88
|
+
for key in fifo_keys:
|
|
89
|
+
if not cache_dict[key].can_evict:
|
|
90
|
+
continue
|
|
91
|
+
evict_keys.append(key)
|
|
92
|
+
evict_freqs.append(curr_min_freq)
|
|
93
|
+
self.key_to_freq.pop(key)
|
|
94
|
+
if len(evict_keys) == num_candidates:
|
|
95
|
+
break
|
|
96
|
+
|
|
97
|
+
if len(evict_keys) == num_candidates:
|
|
98
|
+
break
|
|
99
|
+
|
|
100
|
+
for freq, key in zip(evict_freqs, evict_keys, strict=False):
|
|
101
|
+
self.freq_to_keys[freq].pop(key)
|
|
102
|
+
if not self.freq_to_keys[freq]:
|
|
103
|
+
self.freq_to_keys.pop(freq)
|
|
104
|
+
|
|
105
|
+
return evict_keys
|
|
@@ -0,0 +1,81 @@
|
|
|
1
|
+
# SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
# Standard
|
|
3
|
+
from collections import OrderedDict
|
|
4
|
+
from typing import Any, Dict
|
|
5
|
+
import time
|
|
6
|
+
|
|
7
|
+
# First Party
|
|
8
|
+
from lmcache.logging import init_logger
|
|
9
|
+
from lmcache.observability import LMCStatsMonitor
|
|
10
|
+
from lmcache.utils import CacheEngineKey
|
|
11
|
+
from lmcache.v1.storage_backend.cache_policy.base_policy import BaseCachePolicy, KeyType
|
|
12
|
+
|
|
13
|
+
logger = init_logger(__name__)
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
class LRUCachePolicy(BaseCachePolicy[KeyType, OrderedDict[KeyType, Any]]):
|
|
17
|
+
"""
|
|
18
|
+
LRU cache policy.
|
|
19
|
+
"""
|
|
20
|
+
|
|
21
|
+
def __init__(self):
|
|
22
|
+
logger.info("Initializing LRUCachePolicy")
|
|
23
|
+
self.chunk_hash_to_init_timestamp: Dict[Any, float] = {}
|
|
24
|
+
self.stats_monitor = LMCStatsMonitor.GetOrCreate()
|
|
25
|
+
self.max_num_chunk_hash = 12500000
|
|
26
|
+
|
|
27
|
+
def init_mutable_mapping(self) -> OrderedDict[KeyType, Any]:
|
|
28
|
+
return OrderedDict()
|
|
29
|
+
|
|
30
|
+
def update_chunk_hash_dict(self, key: KeyType) -> None:
|
|
31
|
+
curr_time = time.time()
|
|
32
|
+
# HACK: doing type conversion here
|
|
33
|
+
key_hash: Any = key
|
|
34
|
+
if isinstance(key, CacheEngineKey):
|
|
35
|
+
key_hash = key.chunk_hash
|
|
36
|
+
|
|
37
|
+
if init_timestamp := self.chunk_hash_to_init_timestamp.get(key_hash, None):
|
|
38
|
+
time_interval = curr_time - init_timestamp
|
|
39
|
+
self.stats_monitor.on_chunk_reuse(time_interval)
|
|
40
|
+
else:
|
|
41
|
+
if len(self.chunk_hash_to_init_timestamp) >= self.max_num_chunk_hash:
|
|
42
|
+
self.chunk_hash_to_init_timestamp.clear()
|
|
43
|
+
self.chunk_hash_to_init_timestamp[key_hash] = curr_time
|
|
44
|
+
|
|
45
|
+
def update_on_hit(
|
|
46
|
+
self,
|
|
47
|
+
key: KeyType,
|
|
48
|
+
cache_dict: OrderedDict[KeyType, Any],
|
|
49
|
+
) -> None:
|
|
50
|
+
self.update_chunk_hash_dict(key)
|
|
51
|
+
cache_dict.move_to_end(key)
|
|
52
|
+
|
|
53
|
+
def update_on_put(
|
|
54
|
+
self,
|
|
55
|
+
key: KeyType,
|
|
56
|
+
) -> None:
|
|
57
|
+
self.update_chunk_hash_dict(key)
|
|
58
|
+
pass
|
|
59
|
+
|
|
60
|
+
def update_on_force_evict(
|
|
61
|
+
self,
|
|
62
|
+
key: KeyType,
|
|
63
|
+
) -> None:
|
|
64
|
+
pass
|
|
65
|
+
|
|
66
|
+
# NOTE(Jiayi): We do best effort to get eviction candidates so the number
|
|
67
|
+
# of returned keys mignt be smaller than num_candidates.
|
|
68
|
+
def get_evict_candidates(
|
|
69
|
+
self,
|
|
70
|
+
cache_dict: OrderedDict[KeyType, Any],
|
|
71
|
+
num_candidates: int = 1,
|
|
72
|
+
) -> list[KeyType]:
|
|
73
|
+
evict_keys = []
|
|
74
|
+
for key, cache in cache_dict.items():
|
|
75
|
+
if not cache.can_evict:
|
|
76
|
+
continue
|
|
77
|
+
evict_keys.append(key)
|
|
78
|
+
if len(evict_keys) == num_candidates:
|
|
79
|
+
break
|
|
80
|
+
|
|
81
|
+
return evict_keys
|
|
@@ -0,0 +1,61 @@
|
|
|
1
|
+
# SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
# Standard
|
|
3
|
+
from collections import OrderedDict
|
|
4
|
+
from typing import Any
|
|
5
|
+
|
|
6
|
+
# First Party
|
|
7
|
+
from lmcache.logging import init_logger
|
|
8
|
+
from lmcache.v1.storage_backend.cache_policy.base_policy import BaseCachePolicy, KeyType
|
|
9
|
+
|
|
10
|
+
logger = init_logger(__name__)
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
class MRUCachePolicy(BaseCachePolicy[KeyType, OrderedDict[KeyType, Any]]):
|
|
14
|
+
"""
|
|
15
|
+
MRU cache policy.
|
|
16
|
+
"""
|
|
17
|
+
|
|
18
|
+
def __init__(self):
|
|
19
|
+
logger.info("Initializing MRUCachePolicy")
|
|
20
|
+
|
|
21
|
+
def init_mutable_mapping(self) -> OrderedDict[KeyType, Any]:
|
|
22
|
+
return OrderedDict()
|
|
23
|
+
|
|
24
|
+
def update_on_hit(
|
|
25
|
+
self,
|
|
26
|
+
key: KeyType,
|
|
27
|
+
cache_dict: OrderedDict[KeyType, Any],
|
|
28
|
+
) -> None:
|
|
29
|
+
# since MRU evicts from the back, the logic is same as LRU.
|
|
30
|
+
cache_dict.move_to_end(key, last=True)
|
|
31
|
+
|
|
32
|
+
def update_on_put(
|
|
33
|
+
self,
|
|
34
|
+
key: KeyType,
|
|
35
|
+
) -> None:
|
|
36
|
+
# No action needed for MRU on put, as the key is already at the back.
|
|
37
|
+
pass
|
|
38
|
+
|
|
39
|
+
def update_on_force_evict(
|
|
40
|
+
self,
|
|
41
|
+
key: KeyType,
|
|
42
|
+
) -> None:
|
|
43
|
+
pass
|
|
44
|
+
|
|
45
|
+
# NOTE(Jiayi): We do best effort to get eviction candidates so the number
|
|
46
|
+
# of returned keys mignt be smaller than num_candidates.
|
|
47
|
+
def get_evict_candidates(
|
|
48
|
+
self,
|
|
49
|
+
cache_dict: OrderedDict[KeyType, Any],
|
|
50
|
+
num_candidates: int = 1,
|
|
51
|
+
) -> list[KeyType]:
|
|
52
|
+
evict_keys = []
|
|
53
|
+
# Since the most recent object is at the end, we reverse the order here
|
|
54
|
+
for key, cache in reversed(cache_dict.items()):
|
|
55
|
+
if not cache.can_evict:
|
|
56
|
+
continue
|
|
57
|
+
evict_keys.append(key)
|
|
58
|
+
if len(evict_keys) == num_candidates:
|
|
59
|
+
break
|
|
60
|
+
|
|
61
|
+
return evict_keys
|