lmcache-cli 0.4.5.dev0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- lmcache/__init__.py +84 -0
- lmcache/_version.py +24 -0
- lmcache/cli/__init__.py +1 -0
- lmcache/cli/commands/__init__.py +34 -0
- lmcache/cli/commands/base.py +157 -0
- lmcache/cli/commands/bench/__init__.py +557 -0
- lmcache/cli/commands/bench/engine_bench/__init__.py +1 -0
- lmcache/cli/commands/bench/engine_bench/config.py +245 -0
- lmcache/cli/commands/bench/engine_bench/interactive/__init__.py +274 -0
- lmcache/cli/commands/bench/engine_bench/interactive/config.json +10 -0
- lmcache/cli/commands/bench/engine_bench/interactive/schema.py +352 -0
- lmcache/cli/commands/bench/engine_bench/interactive/state.py +327 -0
- lmcache/cli/commands/bench/engine_bench/interactive/terminal.py +291 -0
- lmcache/cli/commands/bench/engine_bench/progress.py +145 -0
- lmcache/cli/commands/bench/engine_bench/request_sender.py +232 -0
- lmcache/cli/commands/bench/engine_bench/stats.py +275 -0
- lmcache/cli/commands/bench/engine_bench/workloads/__init__.py +153 -0
- lmcache/cli/commands/bench/engine_bench/workloads/base.py +122 -0
- lmcache/cli/commands/bench/engine_bench/workloads/long_doc_permutator.py +435 -0
- lmcache/cli/commands/bench/engine_bench/workloads/long_doc_qa.py +281 -0
- lmcache/cli/commands/bench/engine_bench/workloads/multi_round_chat.py +337 -0
- lmcache/cli/commands/bench/engine_bench/workloads/random_prefill.py +178 -0
- lmcache/cli/commands/describe.py +310 -0
- lmcache/cli/commands/kvcache.py +133 -0
- lmcache/cli/commands/mock.py +75 -0
- lmcache/cli/commands/ping.py +113 -0
- lmcache/cli/commands/query/__init__.py +155 -0
- lmcache/cli/commands/query/prompt.py +134 -0
- lmcache/cli/commands/query/request.py +357 -0
- lmcache/cli/commands/server.py +99 -0
- lmcache/cli/commands/tool/__init__.py +63 -0
- lmcache/cli/commands/tool/cache_simulator.py +113 -0
- lmcache/cli/commands/trace/__init__.py +505 -0
- lmcache/cli/commands/trace/dispatch.py +249 -0
- lmcache/cli/commands/trace/driver.py +372 -0
- lmcache/cli/commands/trace/stats.py +289 -0
- lmcache/cli/documents/lmcache.txt +11 -0
- lmcache/cli/main.py +42 -0
- lmcache/cli/metrics/__init__.py +29 -0
- lmcache/cli/metrics/formatter.py +171 -0
- lmcache/cli/metrics/handler.py +94 -0
- lmcache/cli/metrics/metrics.py +161 -0
- lmcache/cli/metrics/section.py +77 -0
- lmcache/connections.py +173 -0
- lmcache/integration/__init__.py +2 -0
- lmcache/integration/base_service_factory.py +165 -0
- lmcache/integration/request_telemetry/__init__.py +1 -0
- lmcache/integration/request_telemetry/base.py +51 -0
- lmcache/integration/request_telemetry/factory.py +113 -0
- lmcache/integration/request_telemetry/fastapi.py +109 -0
- lmcache/integration/request_telemetry/noop.py +35 -0
- lmcache/integration/sglang/__init__.py +2 -0
- lmcache/integration/sglang/sglang_adapter.py +326 -0
- lmcache/integration/sglang/utils.py +39 -0
- lmcache/integration/vllm/__init__.py +1 -0
- lmcache/integration/vllm/lmcache_connector_v1.py +213 -0
- lmcache/integration/vllm/lmcache_connector_v1_085.py +150 -0
- lmcache/integration/vllm/lmcache_mp_connector_0180.py +1072 -0
- lmcache/integration/vllm/tests/test_mm_hash_utils.py +112 -0
- lmcache/integration/vllm/utils.py +433 -0
- lmcache/integration/vllm/vllm_multi_process_adapter.py +1090 -0
- lmcache/integration/vllm/vllm_service_factory.py +339 -0
- lmcache/integration/vllm/vllm_v1_adapter.py +1713 -0
- lmcache/logging.py +107 -0
- lmcache/native_storage_ops.pyi +230 -0
- lmcache/non_cuda_equivalents.py +1424 -0
- lmcache/observability.py +1958 -0
- lmcache/storage_backend/serde/__init__.py +1 -0
- lmcache/storage_backend/serde/cachegen_basics.py +210 -0
- lmcache/storage_backend/serde/cachegen_decoder.py +207 -0
- lmcache/storage_backend/serde/cachegen_encoder.py +394 -0
- lmcache/storage_backend/serde/serde.py +75 -0
- lmcache/tools/__init__.py +1 -0
- lmcache/tools/cache_simulator/README.md +392 -0
- lmcache/tools/cache_simulator/__init__.py +1 -0
- lmcache/tools/cache_simulator/docs/simulate_example.png +0 -0
- lmcache/tools/cache_simulator/docs/sweep_example.png +0 -0
- lmcache/tools/cache_simulator/gen_bench_dataset.py +360 -0
- lmcache/tools/cache_simulator/lru_cache.py +124 -0
- lmcache/tools/cache_simulator/plot_hit_rate.py +231 -0
- lmcache/tools/cache_simulator/simulator.py +795 -0
- lmcache/tools/controller_benchmark/README.md +161 -0
- lmcache/tools/controller_benchmark/__init__.py +1 -0
- lmcache/tools/controller_benchmark/__main__.py +331 -0
- lmcache/tools/controller_benchmark/benchmark.py +660 -0
- lmcache/tools/controller_benchmark/config.py +44 -0
- lmcache/tools/controller_benchmark/constants.py +10 -0
- lmcache/tools/controller_benchmark/handlers/__init__.py +46 -0
- lmcache/tools/controller_benchmark/handlers/admit.py +52 -0
- lmcache/tools/controller_benchmark/handlers/base.py +47 -0
- lmcache/tools/controller_benchmark/handlers/deregister.py +49 -0
- lmcache/tools/controller_benchmark/handlers/evict.py +52 -0
- lmcache/tools/controller_benchmark/handlers/heartbeat.py +56 -0
- lmcache/tools/controller_benchmark/handlers/p2p_lookup.py +47 -0
- lmcache/tools/controller_benchmark/handlers/register.py +56 -0
- lmcache/tools/mp_status_viewer/__init__.py +1 -0
- lmcache/tools/mp_status_viewer/__main__.py +95 -0
- lmcache/usage_context.py +417 -0
- lmcache/utils.py +665 -0
- lmcache/v1/__init__.py +2 -0
- lmcache/v1/api_server/__init__.py +2 -0
- lmcache/v1/api_server/__main__.py +537 -0
- lmcache/v1/basic_check.py +112 -0
- lmcache/v1/cache_controller/__init__.py +9 -0
- lmcache/v1/cache_controller/commands/__init__.py +15 -0
- lmcache/v1/cache_controller/commands/base.py +35 -0
- lmcache/v1/cache_controller/commands/full_sync.py +49 -0
- lmcache/v1/cache_controller/config.py +176 -0
- lmcache/v1/cache_controller/controller_manager.py +535 -0
- lmcache/v1/cache_controller/controllers/__init__.py +11 -0
- lmcache/v1/cache_controller/controllers/full_sync_tracker.py +473 -0
- lmcache/v1/cache_controller/controllers/kv_controller.py +439 -0
- lmcache/v1/cache_controller/controllers/registration_controller.py +282 -0
- lmcache/v1/cache_controller/executor.py +463 -0
- lmcache/v1/cache_controller/frontend/static/css/style.css +201 -0
- lmcache/v1/cache_controller/frontend/static/img/logo.png +0 -0
- lmcache/v1/cache_controller/frontend/static/index.html +234 -0
- lmcache/v1/cache_controller/frontend/static/js/controller_app.js +660 -0
- lmcache/v1/cache_controller/full_sync_sender.py +475 -0
- lmcache/v1/cache_controller/locks.py +149 -0
- lmcache/v1/cache_controller/message.py +828 -0
- lmcache/v1/cache_controller/observability.py +208 -0
- lmcache/v1/cache_controller/utils.py +679 -0
- lmcache/v1/cache_controller/worker.py +665 -0
- lmcache/v1/cache_engine.py +2058 -0
- lmcache/v1/cache_interface.py +19 -0
- lmcache/v1/check/__init__.py +74 -0
- lmcache/v1/check/check_mode_gen.py +86 -0
- lmcache/v1/check/check_mode_test_l2_adapter.py +284 -0
- lmcache/v1/check/check_mode_test_remote.py +155 -0
- lmcache/v1/check/check_mode_test_storage_manager.py +142 -0
- lmcache/v1/check/utils.py +571 -0
- lmcache/v1/compute/__init__.py +2 -0
- lmcache/v1/compute/attention/__init__.py +0 -0
- lmcache/v1/compute/attention/abstract.py +39 -0
- lmcache/v1/compute/attention/flash_attn.py +129 -0
- lmcache/v1/compute/attention/flash_infer_sparse.py +284 -0
- lmcache/v1/compute/attention/metadata.py +85 -0
- lmcache/v1/compute/attention/utils.py +14 -0
- lmcache/v1/compute/blend/__init__.py +7 -0
- lmcache/v1/compute/blend/blender.py +168 -0
- lmcache/v1/compute/blend/metadata.py +34 -0
- lmcache/v1/compute/blend/utils.py +63 -0
- lmcache/v1/compute/models/__init__.py +0 -0
- lmcache/v1/compute/models/base.py +141 -0
- lmcache/v1/compute/models/llama.py +9 -0
- lmcache/v1/compute/models/qwen3.py +24 -0
- lmcache/v1/compute/models/utils.py +68 -0
- lmcache/v1/compute/positional_encoding.py +199 -0
- lmcache/v1/config.py +848 -0
- lmcache/v1/config_base.py +848 -0
- lmcache/v1/distributed/api.py +248 -0
- lmcache/v1/distributed/config.py +321 -0
- lmcache/v1/distributed/error.py +64 -0
- lmcache/v1/distributed/eviction.py +192 -0
- lmcache/v1/distributed/eviction_policy/__init__.py +21 -0
- lmcache/v1/distributed/eviction_policy/factory.py +27 -0
- lmcache/v1/distributed/eviction_policy/lru.py +244 -0
- lmcache/v1/distributed/eviction_policy/noop.py +50 -0
- lmcache/v1/distributed/internal_api.py +170 -0
- lmcache/v1/distributed/l1_manager.py +835 -0
- lmcache/v1/distributed/l2_adapters/__init__.py +67 -0
- lmcache/v1/distributed/l2_adapters/base.py +360 -0
- lmcache/v1/distributed/l2_adapters/config.py +385 -0
- lmcache/v1/distributed/l2_adapters/factory.py +205 -0
- lmcache/v1/distributed/l2_adapters/fs_l2_adapter.py +747 -0
- lmcache/v1/distributed/l2_adapters/fs_native_l2_adapter.py +167 -0
- lmcache/v1/distributed/l2_adapters/mock_l2_adapter.py +516 -0
- lmcache/v1/distributed/l2_adapters/mooncake_store_l2_adapter.py +135 -0
- lmcache/v1/distributed/l2_adapters/native_connector_l2_adapter.py +468 -0
- lmcache/v1/distributed/l2_adapters/native_plugin_l2_adapter.py +199 -0
- lmcache/v1/distributed/l2_adapters/nixl_store_dynamic_l2_adapter.py +831 -0
- lmcache/v1/distributed/l2_adapters/nixl_store_l2_adapter.py +983 -0
- lmcache/v1/distributed/l2_adapters/plugin_l2_adapter.py +210 -0
- lmcache/v1/distributed/l2_adapters/resp_l2_adapter.py +176 -0
- lmcache/v1/distributed/memory_manager.py +179 -0
- lmcache/v1/distributed/storage_controller.py +39 -0
- lmcache/v1/distributed/storage_controllers/__init__.py +43 -0
- lmcache/v1/distributed/storage_controllers/eviction_controller.py +242 -0
- lmcache/v1/distributed/storage_controllers/prefetch_controller.py +830 -0
- lmcache/v1/distributed/storage_controllers/prefetch_policy.py +193 -0
- lmcache/v1/distributed/storage_controllers/store_controller.py +452 -0
- lmcache/v1/distributed/storage_controllers/store_policy.py +213 -0
- lmcache/v1/distributed/storage_manager.py +532 -0
- lmcache/v1/event_manager.py +145 -0
- lmcache/v1/exceptions/__init__.py +16 -0
- lmcache/v1/gpu_connector/__init__.py +126 -0
- lmcache/v1/gpu_connector/gpu_connectors.py +1906 -0
- lmcache/v1/gpu_connector/gpu_ops.py +85 -0
- lmcache/v1/gpu_connector/hpu_connector.py +326 -0
- lmcache/v1/gpu_connector/mock_gpu_connector.py +67 -0
- lmcache/v1/gpu_connector/utils.py +890 -0
- lmcache/v1/gpu_connector/xpu_connectors.py +916 -0
- lmcache/v1/health_monitor/__init__.py +1 -0
- lmcache/v1/health_monitor/base.py +587 -0
- lmcache/v1/health_monitor/checks/__init__.py +1 -0
- lmcache/v1/health_monitor/checks/remote_backend_check.py +304 -0
- lmcache/v1/health_monitor/constants.py +36 -0
- lmcache/v1/internal_api_server/__init__.py +0 -0
- lmcache/v1/internal_api_server/api_registry.py +59 -0
- lmcache/v1/internal_api_server/api_server.py +120 -0
- lmcache/v1/internal_api_server/common/__init__.py +1 -0
- lmcache/v1/internal_api_server/common/env_api.py +22 -0
- lmcache/v1/internal_api_server/common/loglevel_api.py +57 -0
- lmcache/v1/internal_api_server/common/metrics_api.py +29 -0
- lmcache/v1/internal_api_server/common/periodic_thread_api.py +138 -0
- lmcache/v1/internal_api_server/common/run_script_api.py +73 -0
- lmcache/v1/internal_api_server/common/thread_api.py +63 -0
- lmcache/v1/internal_api_server/controller/__init__.py +1 -0
- lmcache/v1/internal_api_server/controller/key_stats_api.py +81 -0
- lmcache/v1/internal_api_server/controller/worker_info_api.py +136 -0
- lmcache/v1/internal_api_server/utils.py +43 -0
- lmcache/v1/internal_api_server/vllm/__init__.py +1 -0
- lmcache/v1/internal_api_server/vllm/backend_api.py +221 -0
- lmcache/v1/internal_api_server/vllm/bypass_api.py +204 -0
- lmcache/v1/internal_api_server/vllm/cache_api.py +895 -0
- lmcache/v1/internal_api_server/vllm/chunk_statistics_api.py +141 -0
- lmcache/v1/internal_api_server/vllm/conf_api.py +147 -0
- lmcache/v1/internal_api_server/vllm/freeze_api.py +172 -0
- lmcache/v1/internal_api_server/vllm/hot_cache_api.py +184 -0
- lmcache/v1/internal_api_server/vllm/inference_api.py +65 -0
- lmcache/v1/internal_api_server/vllm/load_fs_chunks_api.py +320 -0
- lmcache/v1/internal_api_server/vllm/lookup_api.py +145 -0
- lmcache/v1/internal_api_server/vllm/version_api.py +25 -0
- lmcache/v1/kv_layer_groups.py +267 -0
- lmcache/v1/lazy_memory_allocator.py +284 -0
- lmcache/v1/lookup_client/__init__.py +25 -0
- lmcache/v1/lookup_client/abstract_client.py +77 -0
- lmcache/v1/lookup_client/async_lookup_message.py +50 -0
- lmcache/v1/lookup_client/chunk_statistics_lookup_client.py +200 -0
- lmcache/v1/lookup_client/factory.py +251 -0
- lmcache/v1/lookup_client/hit_limit_lookup_client.py +86 -0
- lmcache/v1/lookup_client/lmcache_async_lookup_client.py +407 -0
- lmcache/v1/lookup_client/lmcache_lookup_client.py +285 -0
- lmcache/v1/lookup_client/lmcache_lookup_client_bypass.py +99 -0
- lmcache/v1/lookup_client/mooncake_lookup_client.py +87 -0
- lmcache/v1/lookup_client/record_strategies/__init__.py +77 -0
- lmcache/v1/lookup_client/record_strategies/base.py +327 -0
- lmcache/v1/lookup_client/record_strategies/file_hash.py +130 -0
- lmcache/v1/lookup_client/record_strategies/memory_bloom_filter.py +81 -0
- lmcache/v1/manager.py +539 -0
- lmcache/v1/memory_management.py +2619 -0
- lmcache/v1/metadata.py +114 -0
- lmcache/v1/mp_observability/AGENTS.override.md +21 -0
- lmcache/v1/mp_observability/README.md +204 -0
- lmcache/v1/mp_observability/config.py +340 -0
- lmcache/v1/mp_observability/event.py +100 -0
- lmcache/v1/mp_observability/event_bus.py +313 -0
- lmcache/v1/mp_observability/otel_init.py +145 -0
- lmcache/v1/mp_observability/subscribers/__init__.py +28 -0
- lmcache/v1/mp_observability/subscribers/logging/__init__.py +19 -0
- lmcache/v1/mp_observability/subscribers/logging/l1.py +56 -0
- lmcache/v1/mp_observability/subscribers/logging/l2.py +73 -0
- lmcache/v1/mp_observability/subscribers/logging/lookup_hash.py +209 -0
- lmcache/v1/mp_observability/subscribers/logging/mp_server.py +90 -0
- lmcache/v1/mp_observability/subscribers/logging/sm.py +59 -0
- lmcache/v1/mp_observability/subscribers/metrics/__init__.py +20 -0
- lmcache/v1/mp_observability/subscribers/metrics/l0_lifecycle.py +290 -0
- lmcache/v1/mp_observability/subscribers/metrics/l1.py +55 -0
- lmcache/v1/mp_observability/subscribers/metrics/l1_lifecycle.py +166 -0
- lmcache/v1/mp_observability/subscribers/metrics/l2.py +121 -0
- lmcache/v1/mp_observability/subscribers/metrics/sm.py +69 -0
- lmcache/v1/mp_observability/subscribers/tracing/__init__.py +12 -0
- lmcache/v1/mp_observability/subscribers/tracing/mp_server.py +333 -0
- lmcache/v1/mp_observability/subscribers/tracing/span_registry.py +148 -0
- lmcache/v1/mp_observability/trace/__init__.py +50 -0
- lmcache/v1/mp_observability/trace/codecs.py +255 -0
- lmcache/v1/mp_observability/trace/decorator.py +147 -0
- lmcache/v1/mp_observability/trace/format.py +132 -0
- lmcache/v1/mp_observability/trace/lifecycle.py +83 -0
- lmcache/v1/mp_observability/trace/reader.py +167 -0
- lmcache/v1/mp_observability/trace/recorder.py +300 -0
- lmcache/v1/multiprocess/__init__.py +0 -0
- lmcache/v1/multiprocess/affinity_pool.py +102 -0
- lmcache/v1/multiprocess/blend_server_v2.py +891 -0
- lmcache/v1/multiprocess/config.py +253 -0
- lmcache/v1/multiprocess/custom_types.py +281 -0
- lmcache/v1/multiprocess/futures.py +194 -0
- lmcache/v1/multiprocess/gpu_context.py +511 -0
- lmcache/v1/multiprocess/http_server.py +235 -0
- lmcache/v1/multiprocess/mp_runtime_plugin_launcher.py +130 -0
- lmcache/v1/multiprocess/mq.py +732 -0
- lmcache/v1/multiprocess/protocol.py +86 -0
- lmcache/v1/multiprocess/protocols/README.md +213 -0
- lmcache/v1/multiprocess/protocols/__init__.py +127 -0
- lmcache/v1/multiprocess/protocols/base.py +89 -0
- lmcache/v1/multiprocess/protocols/blend.py +109 -0
- lmcache/v1/multiprocess/protocols/blend_v2.py +57 -0
- lmcache/v1/multiprocess/protocols/controller.py +53 -0
- lmcache/v1/multiprocess/protocols/debug.py +34 -0
- lmcache/v1/multiprocess/protocols/engine.py +146 -0
- lmcache/v1/multiprocess/protocols/observability.py +39 -0
- lmcache/v1/multiprocess/server.py +1134 -0
- lmcache/v1/multiprocess/session.py +190 -0
- lmcache/v1/multiprocess/token_hasher.py +441 -0
- lmcache/v1/offload_server/__init__.py +17 -0
- lmcache/v1/offload_server/abstract_server.py +37 -0
- lmcache/v1/offload_server/message.py +30 -0
- lmcache/v1/offload_server/zmq_server.py +122 -0
- lmcache/v1/periodic_thread.py +579 -0
- lmcache/v1/pin_monitor.py +246 -0
- lmcache/v1/plugin/__init__.py +0 -0
- lmcache/v1/plugin/runtime_plugin_launcher.py +211 -0
- lmcache/v1/protocol.py +317 -0
- lmcache/v1/rpc/__init__.py +17 -0
- lmcache/v1/rpc/transport.py +105 -0
- lmcache/v1/rpc/zmq_transport.py +213 -0
- lmcache/v1/rpc_utils.py +165 -0
- lmcache/v1/server/__init__.py +2 -0
- lmcache/v1/server/__main__.py +170 -0
- lmcache/v1/server/storage_backend/__init__.py +21 -0
- lmcache/v1/server/storage_backend/abstract_backend.py +80 -0
- lmcache/v1/server/storage_backend/local_backend.py +75 -0
- lmcache/v1/server/utils.py +21 -0
- lmcache/v1/standalone/__init__.py +1 -0
- lmcache/v1/standalone/__main__.py +583 -0
- lmcache/v1/standalone/manager.py +80 -0
- lmcache/v1/standalone/standalone_service_factory.py +86 -0
- lmcache/v1/storage_backend/__init__.py +313 -0
- lmcache/v1/storage_backend/abstract_backend.py +445 -0
- lmcache/v1/storage_backend/audit_backend.py +233 -0
- lmcache/v1/storage_backend/batched_message_sender.py +222 -0
- lmcache/v1/storage_backend/cache_policy/__init__.py +45 -0
- lmcache/v1/storage_backend/cache_policy/base_policy.py +87 -0
- lmcache/v1/storage_backend/cache_policy/fifo.py +58 -0
- lmcache/v1/storage_backend/cache_policy/lfu.py +105 -0
- lmcache/v1/storage_backend/cache_policy/lru.py +81 -0
- lmcache/v1/storage_backend/cache_policy/mru.py +61 -0
- lmcache/v1/storage_backend/connector/__init__.py +443 -0
- lmcache/v1/storage_backend/connector/audit_adapter.py +77 -0
- lmcache/v1/storage_backend/connector/audit_connector.py +320 -0
- lmcache/v1/storage_backend/connector/base_connector.py +379 -0
- lmcache/v1/storage_backend/connector/blackhole_adapter.py +21 -0
- lmcache/v1/storage_backend/connector/blackhole_connector.py +37 -0
- lmcache/v1/storage_backend/connector/eic_adapter.py +31 -0
- lmcache/v1/storage_backend/connector/eic_connector.py +757 -0
- lmcache/v1/storage_backend/connector/external_adapter.py +79 -0
- lmcache/v1/storage_backend/connector/fs_adapter.py +51 -0
- lmcache/v1/storage_backend/connector/fs_connector.py +403 -0
- lmcache/v1/storage_backend/connector/infinistore_adapter.py +56 -0
- lmcache/v1/storage_backend/connector/infinistore_connector.py +177 -0
- lmcache/v1/storage_backend/connector/instrumented_connector.py +219 -0
- lmcache/v1/storage_backend/connector/lm_adapter.py +31 -0
- lmcache/v1/storage_backend/connector/lm_connector.py +176 -0
- lmcache/v1/storage_backend/connector/mock_adapter.py +57 -0
- lmcache/v1/storage_backend/connector/mock_connector.py +349 -0
- lmcache/v1/storage_backend/connector/mooncakestore_adapter.py +43 -0
- lmcache/v1/storage_backend/connector/mooncakestore_connector.py +614 -0
- lmcache/v1/storage_backend/connector/redis_adapter.py +181 -0
- lmcache/v1/storage_backend/connector/redis_connector.py +828 -0
- lmcache/v1/storage_backend/connector/s3_adapter.py +59 -0
- lmcache/v1/storage_backend/connector/s3_connector.py +699 -0
- lmcache/v1/storage_backend/connector/sagemaker_hyperpod_adapter.py +233 -0
- lmcache/v1/storage_backend/connector/sagemaker_hyperpod_connector.py +987 -0
- lmcache/v1/storage_backend/connector/valkey_adapter.py +114 -0
- lmcache/v1/storage_backend/connector/valkey_connector.py +627 -0
- lmcache/v1/storage_backend/gds_backend.py +1199 -0
- lmcache/v1/storage_backend/job_executor/__init__.py +0 -0
- lmcache/v1/storage_backend/job_executor/base_executor.py +34 -0
- lmcache/v1/storage_backend/job_executor/pq_executor.py +235 -0
- lmcache/v1/storage_backend/local_cpu_backend.py +810 -0
- lmcache/v1/storage_backend/local_disk_backend.py +656 -0
- lmcache/v1/storage_backend/maru_backend.py +734 -0
- lmcache/v1/storage_backend/naive_serde/__init__.py +50 -0
- lmcache/v1/storage_backend/naive_serde/cachegen_basics.py +133 -0
- lmcache/v1/storage_backend/naive_serde/cachegen_decoder.py +135 -0
- lmcache/v1/storage_backend/naive_serde/cachegen_encoder.py +83 -0
- lmcache/v1/storage_backend/naive_serde/kivi_serde.py +22 -0
- lmcache/v1/storage_backend/naive_serde/naive_serde.py +18 -0
- lmcache/v1/storage_backend/naive_serde/serde.py +37 -0
- lmcache/v1/storage_backend/native_clients/connector_client_base.py +165 -0
- lmcache/v1/storage_backend/native_clients/resp_client.py +35 -0
- lmcache/v1/storage_backend/nixl_storage_backend.py +1400 -0
- lmcache/v1/storage_backend/p2p_backend.py +788 -0
- lmcache/v1/storage_backend/path_sharder.py +117 -0
- lmcache/v1/storage_backend/pd_backend.py +646 -0
- lmcache/v1/storage_backend/plugins/dax_backend.py +1443 -0
- lmcache/v1/storage_backend/plugins/rust_raw_block_backend.py +1361 -0
- lmcache/v1/storage_backend/remote_backend.py +624 -0
- lmcache/v1/storage_backend/resp_client.py +227 -0
- lmcache/v1/storage_backend/storage_backend_listener.py +19 -0
- lmcache/v1/storage_backend/storage_manager.py +1352 -0
- lmcache/v1/system_detection.py +110 -0
- lmcache/v1/token_database.py +551 -0
- lmcache/v1/transfer_channel/__init__.py +83 -0
- lmcache/v1/transfer_channel/abstract.py +285 -0
- lmcache/v1/transfer_channel/mock_memory_channel.py +156 -0
- lmcache/v1/transfer_channel/nixl_channel.py +639 -0
- lmcache/v1/transfer_channel/py_socket_channel.py +260 -0
- lmcache/v1/transfer_channel/transfer_utils.py +63 -0
- lmcache/v1/utils/__init__.py +1 -0
- lmcache/v1/utils/bloom_filter.py +109 -0
- lmcache/v1/utils/cache_utils.py +125 -0
- lmcache_cli-0.4.5.dev0.dist-info/METADATA +185 -0
- lmcache_cli-0.4.5.dev0.dist-info/RECORD +399 -0
- lmcache_cli-0.4.5.dev0.dist-info/WHEEL +5 -0
- lmcache_cli-0.4.5.dev0.dist-info/entry_points.txt +2 -0
- lmcache_cli-0.4.5.dev0.dist-info/licenses/LICENSE +201 -0
- lmcache_cli-0.4.5.dev0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,835 @@
|
|
|
1
|
+
# SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
"""
|
|
3
|
+
Managing objects and memory for L1 cache
|
|
4
|
+
"""
|
|
5
|
+
|
|
6
|
+
# Standard
|
|
7
|
+
from dataclasses import dataclass
|
|
8
|
+
from typing import Literal
|
|
9
|
+
import threading
|
|
10
|
+
|
|
11
|
+
# First Party
|
|
12
|
+
from lmcache.logging import init_logger
|
|
13
|
+
from lmcache.native_storage_ops import TTLLock
|
|
14
|
+
from lmcache.v1.distributed.api import MemoryLayoutDesc, ObjectKey
|
|
15
|
+
from lmcache.v1.distributed.config import L1ManagerConfig
|
|
16
|
+
from lmcache.v1.distributed.error import L1Error
|
|
17
|
+
from lmcache.v1.distributed.internal_api import L1ManagerListener
|
|
18
|
+
from lmcache.v1.distributed.memory_manager import L1MemoryManager
|
|
19
|
+
from lmcache.v1.memory_management import MemoryObj
|
|
20
|
+
from lmcache.v1.mp_observability.event import Event, EventType
|
|
21
|
+
from lmcache.v1.mp_observability.event_bus import get_event_bus
|
|
22
|
+
|
|
23
|
+
logger = init_logger(__name__)
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
# Internal classes and helper functions
|
|
27
|
+
@dataclass
|
|
28
|
+
class L1ObjectState:
|
|
29
|
+
"""
|
|
30
|
+
The internal state of an object in L1 cache
|
|
31
|
+
"""
|
|
32
|
+
|
|
33
|
+
memory_obj: MemoryObj
|
|
34
|
+
""" The memory object stored in L1 cache. """
|
|
35
|
+
|
|
36
|
+
write_lock: TTLLock
|
|
37
|
+
""" Whether the object is write-locked. """
|
|
38
|
+
|
|
39
|
+
read_lock: TTLLock
|
|
40
|
+
""" The read lock with TTL for the object. """
|
|
41
|
+
|
|
42
|
+
is_temporary: bool
|
|
43
|
+
""" Whether the object is temporary (need to be deleted after read). """
|
|
44
|
+
|
|
45
|
+
def available_for_read(self) -> bool:
|
|
46
|
+
"""Check if the object is available for read.
|
|
47
|
+
|
|
48
|
+
Returns:
|
|
49
|
+
True if the object is not write-locked, False otherwise.
|
|
50
|
+
"""
|
|
51
|
+
return not self.write_lock.is_locked()
|
|
52
|
+
|
|
53
|
+
def available_for_write(self) -> bool:
|
|
54
|
+
"""Check if the object is available for write.
|
|
55
|
+
|
|
56
|
+
Returns:
|
|
57
|
+
True if the object is not write-locked and has no read locks
|
|
58
|
+
and is not a temporary object, False otherwise.
|
|
59
|
+
"""
|
|
60
|
+
|
|
61
|
+
return (
|
|
62
|
+
not self.write_lock.is_locked()
|
|
63
|
+
and not self.read_lock.is_locked()
|
|
64
|
+
and not self.is_temporary
|
|
65
|
+
)
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def l1_mgr_synchronized(func):
|
|
69
|
+
"""
|
|
70
|
+
Decorator to mark L1Manager methods as thread-safe
|
|
71
|
+
"""
|
|
72
|
+
|
|
73
|
+
def wrapper(self: "L1Manager", *args, **kwargs):
|
|
74
|
+
with self._lock:
|
|
75
|
+
return func(self, *args, **kwargs)
|
|
76
|
+
|
|
77
|
+
return wrapper
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
L1OperationResult = tuple[L1Error, MemoryObj | None]
|
|
81
|
+
|
|
82
|
+
# Upper bound for the count parameter in reserve_read / finish_read
|
|
83
|
+
# to prevent a single call from holding the global lock for too long.
|
|
84
|
+
MAX_READ_LOCK_COUNT = 128
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
def _validate_extra_count(extra_count: int) -> int:
|
|
88
|
+
"""Validate and clamp extra_count.
|
|
89
|
+
|
|
90
|
+
Args:
|
|
91
|
+
extra_count: Extra lock count on top of the
|
|
92
|
+
default 1 lock.
|
|
93
|
+
|
|
94
|
+
Returns:
|
|
95
|
+
Clamped value in [0, MAX_READ_LOCK_COUNT - 1].
|
|
96
|
+
"""
|
|
97
|
+
if extra_count < 0:
|
|
98
|
+
logger.warning(
|
|
99
|
+
"L1Manager: extra_count=%d is invalid, clamping to 0",
|
|
100
|
+
extra_count,
|
|
101
|
+
)
|
|
102
|
+
return 0
|
|
103
|
+
upper = MAX_READ_LOCK_COUNT - 1
|
|
104
|
+
if extra_count > upper:
|
|
105
|
+
logger.warning(
|
|
106
|
+
"L1Manager: extra_count=%d exceeds limit=%d, clamping",
|
|
107
|
+
extra_count,
|
|
108
|
+
upper,
|
|
109
|
+
)
|
|
110
|
+
return upper
|
|
111
|
+
return extra_count
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
# Main classes
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
class L1Manager:
|
|
118
|
+
"""
|
|
119
|
+
Object lifecycle state machine for L1 cache
|
|
120
|
+
|
|
121
|
+
+--------+
|
|
122
|
+
| None | <---------------------------------------+
|
|
123
|
+
+--------+ |
|
|
124
|
+
| ^ |
|
|
125
|
+
| | (write lock expired) | delete()
|
|
126
|
+
| | |
|
|
127
|
+
reserve | +----------------------+ |
|
|
128
|
+
write() | | |
|
|
129
|
+
v | |
|
|
130
|
+
+--------------+ +-----------+ |
|
|
131
|
+
| write_locked | | |---------------+
|
|
132
|
+
| |---------->| ready |
|
|
133
|
+
| | finish_ | |---------------+
|
|
134
|
+
+--------------+ write() +-----------+ |
|
|
135
|
+
^ | |
|
|
136
|
+
| | reserve_read() | finish_read()
|
|
137
|
+
+--------------------------+ | (if count becomes 0)
|
|
138
|
+
reserve_write() | |
|
|
139
|
+
v |
|
|
140
|
+
+-----------------+ |
|
|
141
|
+
| read_locked |-----------+
|
|
142
|
+
| (count = 1) |
|
|
143
|
+
+-----------------+
|
|
144
|
+
| ^
|
|
145
|
+
reserve_read() | | finish_read()
|
|
146
|
+
v |
|
|
147
|
+
+-----------------+
|
|
148
|
+
| read_locked |
|
|
149
|
+
| (count = 2) |
|
|
150
|
+
+-----------------+
|
|
151
|
+
| ^
|
|
152
|
+
reserve_read() | | finish_read()
|
|
153
|
+
v |
|
|
154
|
+
(...) (...)
|
|
155
|
+
(Higher Counts)
|
|
156
|
+
|
|
157
|
+
For every operation on list of keys, the operation is atomic
|
|
158
|
+
"""
|
|
159
|
+
|
|
160
|
+
def __init__(self, config: L1ManagerConfig):
|
|
161
|
+
self._lock = threading.Lock()
|
|
162
|
+
|
|
163
|
+
self._objects: dict[ObjectKey, L1ObjectState] = {}
|
|
164
|
+
|
|
165
|
+
self._memory_manager = L1MemoryManager(config.memory_config)
|
|
166
|
+
|
|
167
|
+
self._write_ttl_seconds = config.write_ttl_seconds
|
|
168
|
+
self._read_ttl_seconds = config.read_ttl_seconds
|
|
169
|
+
|
|
170
|
+
self._registered_listeners: list[L1ManagerListener] = []
|
|
171
|
+
|
|
172
|
+
self._event_bus = get_event_bus()
|
|
173
|
+
|
|
174
|
+
def register_listener(self, listener: L1ManagerListener) -> None:
|
|
175
|
+
"""Register a listener for L1Manager events.
|
|
176
|
+
|
|
177
|
+
Args:
|
|
178
|
+
listener: The listener to register.
|
|
179
|
+
"""
|
|
180
|
+
with self._lock:
|
|
181
|
+
self._registered_listeners.append(listener)
|
|
182
|
+
|
|
183
|
+
@l1_mgr_synchronized
|
|
184
|
+
def reserve_read(
|
|
185
|
+
self,
|
|
186
|
+
keys: list[ObjectKey],
|
|
187
|
+
extra_count: int = 0,
|
|
188
|
+
) -> dict[ObjectKey, L1OperationResult]:
|
|
189
|
+
"""Reserve read access for the given keys.
|
|
190
|
+
|
|
191
|
+
Args:
|
|
192
|
+
keys: The list of object keys to reserve
|
|
193
|
+
read access for.
|
|
194
|
+
extra_count: Extra read locks on top of the
|
|
195
|
+
default 1 lock. Total locks acquired per
|
|
196
|
+
key = 1 + extra_count. Useful when multiple
|
|
197
|
+
workers each consume one read lock for the
|
|
198
|
+
same key (e.g. MLA models with TP > 1).
|
|
199
|
+
|
|
200
|
+
Returns:
|
|
201
|
+
A dictionary mapping each object key to a tuple
|
|
202
|
+
of (L1Error, Optional[MemoryObj]).
|
|
203
|
+
|
|
204
|
+
Errors:
|
|
205
|
+
KEY_NOT_EXIST: The key does not exist.
|
|
206
|
+
KEY_NOT_READABLE: The key exists but is not
|
|
207
|
+
readable.
|
|
208
|
+
"""
|
|
209
|
+
extra_count = _validate_extra_count(extra_count)
|
|
210
|
+
total = 1 + extra_count
|
|
211
|
+
ret: dict[ObjectKey, L1OperationResult] = {}
|
|
212
|
+
successful_keys: list[ObjectKey] = []
|
|
213
|
+
for key in keys:
|
|
214
|
+
entry = self._objects.get(key, None)
|
|
215
|
+
if entry is None:
|
|
216
|
+
ret[key] = (L1Error.KEY_NOT_EXIST, None)
|
|
217
|
+
continue
|
|
218
|
+
|
|
219
|
+
if not entry.available_for_read():
|
|
220
|
+
ret[key] = (L1Error.KEY_NOT_READABLE, None)
|
|
221
|
+
continue
|
|
222
|
+
|
|
223
|
+
# TODO(perf): support a count argument in
|
|
224
|
+
# TTLLock.lock() to avoid Python for-loop
|
|
225
|
+
# overhead (TTLLock is C++ std::atomic).
|
|
226
|
+
for _ in range(total):
|
|
227
|
+
entry.read_lock.lock()
|
|
228
|
+
ret[key] = (L1Error.SUCCESS, entry.memory_obj)
|
|
229
|
+
successful_keys.append(key)
|
|
230
|
+
|
|
231
|
+
for listener in self._registered_listeners:
|
|
232
|
+
listener.on_l1_keys_reserved_read(successful_keys)
|
|
233
|
+
self._event_bus.publish(
|
|
234
|
+
Event(
|
|
235
|
+
event_type=EventType.L1_READ_RESERVED,
|
|
236
|
+
metadata={"keys": successful_keys},
|
|
237
|
+
)
|
|
238
|
+
)
|
|
239
|
+
return ret
|
|
240
|
+
|
|
241
|
+
@l1_mgr_synchronized
|
|
242
|
+
def unsafe_read(
|
|
243
|
+
self,
|
|
244
|
+
keys: list[ObjectKey],
|
|
245
|
+
) -> dict[ObjectKey, L1OperationResult]:
|
|
246
|
+
"""Unsafe read the read-locked objects without adding new read locks.
|
|
247
|
+
|
|
248
|
+
This method does not acquire read locks. Therefore, the caller need
|
|
249
|
+
to make sure the `unsafe_read` is called between `reserve_read` and
|
|
250
|
+
`finish_read` calls.
|
|
251
|
+
|
|
252
|
+
Args:
|
|
253
|
+
keys: The list of object keys to read.
|
|
254
|
+
|
|
255
|
+
Returns:
|
|
256
|
+
A dictionary mapping each object key to a tuple of
|
|
257
|
+
(L1Error, Optional[MemoryObj]).
|
|
258
|
+
|
|
259
|
+
Errors:
|
|
260
|
+
KEY_NOT_EXIST: The key does not exist.
|
|
261
|
+
KEY_NOT_READABLE: The key is not readable (in this case, not read-locked).
|
|
262
|
+
"""
|
|
263
|
+
ret: dict[ObjectKey, L1OperationResult] = {}
|
|
264
|
+
|
|
265
|
+
for key in keys:
|
|
266
|
+
entry = self._objects.get(key, None)
|
|
267
|
+
if entry is None:
|
|
268
|
+
ret[key] = (L1Error.KEY_NOT_EXIST, None)
|
|
269
|
+
continue
|
|
270
|
+
|
|
271
|
+
if not entry.read_lock.is_locked():
|
|
272
|
+
ret[key] = (L1Error.KEY_NOT_READABLE, None)
|
|
273
|
+
continue
|
|
274
|
+
|
|
275
|
+
ret[key] = (L1Error.SUCCESS, entry.memory_obj)
|
|
276
|
+
|
|
277
|
+
return ret
|
|
278
|
+
|
|
279
|
+
@l1_mgr_synchronized
|
|
280
|
+
def finish_read(
|
|
281
|
+
self,
|
|
282
|
+
keys: list[ObjectKey],
|
|
283
|
+
extra_count: int = 0,
|
|
284
|
+
) -> dict[ObjectKey, L1Error]:
|
|
285
|
+
"""Finish read access for the given keys.
|
|
286
|
+
|
|
287
|
+
Will delete the object if it is temporary and read
|
|
288
|
+
count reaches zero.
|
|
289
|
+
|
|
290
|
+
Args:
|
|
291
|
+
keys: The list of object keys to finish read
|
|
292
|
+
access for.
|
|
293
|
+
extra_count: Extra read locks to release on top
|
|
294
|
+
of the default 1. Must match the
|
|
295
|
+
``extra_count`` used in the corresponding
|
|
296
|
+
``reserve_read`` call.
|
|
297
|
+
|
|
298
|
+
Returns:
|
|
299
|
+
A dictionary mapping each object key to an
|
|
300
|
+
L1Error.
|
|
301
|
+
|
|
302
|
+
Errors:
|
|
303
|
+
KEY_NOT_EXIST: The key does not exist.
|
|
304
|
+
KEY_IN_WRONG_STATE: The key is write-locked or
|
|
305
|
+
non-read-locked, which means the reader may
|
|
306
|
+
read inconsistent data.
|
|
307
|
+
"""
|
|
308
|
+
extra_count = _validate_extra_count(extra_count)
|
|
309
|
+
total = 1 + extra_count
|
|
310
|
+
need_to_free: list[MemoryObj] = []
|
|
311
|
+
need_to_free_keys: list[ObjectKey] = []
|
|
312
|
+
ret: dict[ObjectKey, L1Error] = {}
|
|
313
|
+
successful_keys: list[ObjectKey] = []
|
|
314
|
+
|
|
315
|
+
for key in keys:
|
|
316
|
+
entry = self._objects.get(key, None)
|
|
317
|
+
if entry is None:
|
|
318
|
+
logger.warning(
|
|
319
|
+
"L1Manager: finish read on non-existing key %s, "
|
|
320
|
+
"potential inconsistent data might be read",
|
|
321
|
+
key,
|
|
322
|
+
)
|
|
323
|
+
ret[key] = L1Error.KEY_NOT_EXIST
|
|
324
|
+
continue
|
|
325
|
+
|
|
326
|
+
if entry.write_lock.is_locked():
|
|
327
|
+
logger.warning(
|
|
328
|
+
"L1Manager: finish read on write-locked key %s, "
|
|
329
|
+
"potential inconsistent data might be read",
|
|
330
|
+
key,
|
|
331
|
+
)
|
|
332
|
+
ret[key] = L1Error.KEY_IN_WRONG_STATE
|
|
333
|
+
continue
|
|
334
|
+
|
|
335
|
+
if not entry.read_lock.is_locked():
|
|
336
|
+
logger.warning(
|
|
337
|
+
"L1Manager: finish read on non-read-locked key %s, "
|
|
338
|
+
"potential inconsistent data might be read",
|
|
339
|
+
key,
|
|
340
|
+
)
|
|
341
|
+
ret[key] = L1Error.KEY_IN_WRONG_STATE
|
|
342
|
+
continue
|
|
343
|
+
|
|
344
|
+
# TODO(perf): support a count argument in
|
|
345
|
+
# TTLLock.unlock() to avoid Python for-loop
|
|
346
|
+
# overhead (TTLLock is C++ std::atomic).
|
|
347
|
+
for _ in range(total):
|
|
348
|
+
entry.read_lock.unlock()
|
|
349
|
+
if entry.is_temporary and not entry.read_lock.is_locked():
|
|
350
|
+
# NOTE: temporary objects shouldn't have write-locks
|
|
351
|
+
need_to_free.append(entry.memory_obj)
|
|
352
|
+
need_to_free_keys.append(key)
|
|
353
|
+
del self._objects[key]
|
|
354
|
+
|
|
355
|
+
ret[key] = L1Error.SUCCESS
|
|
356
|
+
successful_keys.append(key)
|
|
357
|
+
|
|
358
|
+
self._memory_manager.free(need_to_free)
|
|
359
|
+
|
|
360
|
+
for listener in self._registered_listeners:
|
|
361
|
+
listener.on_l1_keys_read_finished(successful_keys)
|
|
362
|
+
listener.on_l1_keys_deleted_by_manager(need_to_free_keys)
|
|
363
|
+
self._event_bus.publish(
|
|
364
|
+
Event(
|
|
365
|
+
event_type=EventType.L1_READ_FINISHED,
|
|
366
|
+
metadata={"keys": successful_keys},
|
|
367
|
+
)
|
|
368
|
+
)
|
|
369
|
+
self._event_bus.publish(
|
|
370
|
+
Event(
|
|
371
|
+
event_type=EventType.L1_KEYS_EVICTED,
|
|
372
|
+
metadata={"keys": need_to_free_keys},
|
|
373
|
+
)
|
|
374
|
+
)
|
|
375
|
+
|
|
376
|
+
return ret
|
|
377
|
+
|
|
378
|
+
@l1_mgr_synchronized
|
|
379
|
+
def reserve_write(
|
|
380
|
+
self,
|
|
381
|
+
keys: list[ObjectKey],
|
|
382
|
+
is_temporary: list[bool],
|
|
383
|
+
layout_desc: MemoryLayoutDesc,
|
|
384
|
+
mode: Literal["new", "update", "all"] = "all",
|
|
385
|
+
) -> dict[ObjectKey, L1OperationResult]:
|
|
386
|
+
"""Reserve write access for the given keys.
|
|
387
|
+
|
|
388
|
+
Args:
|
|
389
|
+
keys: The list of object keys to reserve write access for.
|
|
390
|
+
is_temporary: The list of booleans indicating whether each key is
|
|
391
|
+
temporary.
|
|
392
|
+
shape_spec: The memory layout description for the objects to be
|
|
393
|
+
allocated.
|
|
394
|
+
mode (Literal["new", "update", "all"]): Reservation mode.
|
|
395
|
+
- "new": Reserve only new objects that do not exist.
|
|
396
|
+
- "update": Reserve only existing objects for update.
|
|
397
|
+
- "all": Reserve all writable objects regardless of existence.
|
|
398
|
+
|
|
399
|
+
Returns:
|
|
400
|
+
A dictionary mapping each object key to a tuple of
|
|
401
|
+
(L1Error, Optional[MemoryObj]).
|
|
402
|
+
|
|
403
|
+
Errors:
|
|
404
|
+
KEY_NOT_WRITABLE: The key exists but is not writable.
|
|
405
|
+
OUT_OF_MEMORY: Not enough memory to allocate for the object.
|
|
406
|
+
"""
|
|
407
|
+
need_to_allocate: list[tuple[ObjectKey, bool]] = []
|
|
408
|
+
ret: dict[ObjectKey, L1OperationResult] = {}
|
|
409
|
+
successful_keys: list[ObjectKey] = []
|
|
410
|
+
|
|
411
|
+
for key, is_temp in zip(keys, is_temporary, strict=False):
|
|
412
|
+
entry = self._objects.get(key, None)
|
|
413
|
+
if entry is None:
|
|
414
|
+
need_to_allocate.append((key, is_temp))
|
|
415
|
+
continue
|
|
416
|
+
|
|
417
|
+
if mode == "new":
|
|
418
|
+
ret[key] = (L1Error.KEY_NOT_WRITABLE, None)
|
|
419
|
+
continue
|
|
420
|
+
|
|
421
|
+
if not entry.available_for_write():
|
|
422
|
+
ret[key] = (L1Error.KEY_NOT_WRITABLE, None)
|
|
423
|
+
continue
|
|
424
|
+
|
|
425
|
+
entry.write_lock.lock()
|
|
426
|
+
ret[key] = (L1Error.SUCCESS, entry.memory_obj)
|
|
427
|
+
successful_keys.append(key)
|
|
428
|
+
|
|
429
|
+
# Early return if no allocation is needed
|
|
430
|
+
if len(need_to_allocate) == 0:
|
|
431
|
+
return ret
|
|
432
|
+
|
|
433
|
+
# Don't allow allocation in "update" mode
|
|
434
|
+
if mode == "update":
|
|
435
|
+
for key, _ in need_to_allocate:
|
|
436
|
+
ret[key] = (L1Error.KEY_NOT_WRITABLE, None)
|
|
437
|
+
return ret
|
|
438
|
+
|
|
439
|
+
err, allocated_objs = self._memory_manager.allocate(
|
|
440
|
+
layout_desc, len(need_to_allocate)
|
|
441
|
+
)
|
|
442
|
+
|
|
443
|
+
if err != L1Error.SUCCESS:
|
|
444
|
+
for key, _ in need_to_allocate:
|
|
445
|
+
ret[key] = (L1Error.OUT_OF_MEMORY, None)
|
|
446
|
+
|
|
447
|
+
# Free the memory if partial allocation succeeded
|
|
448
|
+
if allocated_objs:
|
|
449
|
+
self._memory_manager.free(allocated_objs)
|
|
450
|
+
|
|
451
|
+
else:
|
|
452
|
+
for (key, is_temp), mem_obj in zip(
|
|
453
|
+
need_to_allocate, allocated_objs, strict=False
|
|
454
|
+
):
|
|
455
|
+
self._objects[key] = L1ObjectState(
|
|
456
|
+
memory_obj=mem_obj,
|
|
457
|
+
write_lock=TTLLock(self._write_ttl_seconds),
|
|
458
|
+
read_lock=TTLLock(self._read_ttl_seconds),
|
|
459
|
+
is_temporary=is_temp,
|
|
460
|
+
)
|
|
461
|
+
self._objects[key].write_lock.lock()
|
|
462
|
+
ret[key] = (L1Error.SUCCESS, mem_obj)
|
|
463
|
+
successful_keys.append(key)
|
|
464
|
+
|
|
465
|
+
for listener in self._registered_listeners:
|
|
466
|
+
listener.on_l1_keys_reserved_write(successful_keys)
|
|
467
|
+
self._event_bus.publish(
|
|
468
|
+
Event(
|
|
469
|
+
event_type=EventType.L1_WRITE_RESERVED,
|
|
470
|
+
metadata={"keys": successful_keys},
|
|
471
|
+
)
|
|
472
|
+
)
|
|
473
|
+
return ret
|
|
474
|
+
|
|
475
|
+
@l1_mgr_synchronized
|
|
476
|
+
def finish_write(
|
|
477
|
+
self,
|
|
478
|
+
keys: list[ObjectKey],
|
|
479
|
+
) -> dict[ObjectKey, L1Error]:
|
|
480
|
+
"""Finish write access for the given keys.
|
|
481
|
+
|
|
482
|
+
Args:
|
|
483
|
+
keys: The list of object keys to finish write access for.
|
|
484
|
+
|
|
485
|
+
Returns:
|
|
486
|
+
A dictionary mapping each object key to an L1Error.
|
|
487
|
+
|
|
488
|
+
Errors:
|
|
489
|
+
KEY_NOT_EXIST: The key does not exist.
|
|
490
|
+
KEY_IN_WRONG_STATE: The key is not write-locked, or it's read-locked,
|
|
491
|
+
which means the writer may have caused inconsistent data.
|
|
492
|
+
"""
|
|
493
|
+
ret: dict[ObjectKey, L1Error] = {}
|
|
494
|
+
successful_keys: list[ObjectKey] = []
|
|
495
|
+
|
|
496
|
+
for key in keys:
|
|
497
|
+
entry = self._objects.get(key, None)
|
|
498
|
+
if entry is None:
|
|
499
|
+
ret[key] = L1Error.KEY_NOT_EXIST
|
|
500
|
+
continue
|
|
501
|
+
|
|
502
|
+
if not entry.write_lock.is_locked():
|
|
503
|
+
logger.warning(
|
|
504
|
+
"L1Manager: finish write on non-write-locked key %s, "
|
|
505
|
+
"potential inconsistent data might be written",
|
|
506
|
+
key,
|
|
507
|
+
)
|
|
508
|
+
ret[key] = L1Error.KEY_IN_WRONG_STATE
|
|
509
|
+
continue
|
|
510
|
+
|
|
511
|
+
if entry.read_lock.is_locked():
|
|
512
|
+
logger.warning(
|
|
513
|
+
"L1Manager: finish write on read-locked key %s, "
|
|
514
|
+
"potential inconsistent data might be written",
|
|
515
|
+
key,
|
|
516
|
+
)
|
|
517
|
+
ret[key] = L1Error.KEY_IN_WRONG_STATE
|
|
518
|
+
continue
|
|
519
|
+
|
|
520
|
+
entry.write_lock.unlock()
|
|
521
|
+
ret[key] = L1Error.SUCCESS
|
|
522
|
+
successful_keys.append(key)
|
|
523
|
+
|
|
524
|
+
for listener in self._registered_listeners:
|
|
525
|
+
listener.on_l1_keys_write_finished(successful_keys)
|
|
526
|
+
self._event_bus.publish(
|
|
527
|
+
Event(
|
|
528
|
+
event_type=EventType.L1_WRITE_FINISHED,
|
|
529
|
+
metadata={"keys": successful_keys},
|
|
530
|
+
)
|
|
531
|
+
)
|
|
532
|
+
return ret
|
|
533
|
+
|
|
534
|
+
@l1_mgr_synchronized
|
|
535
|
+
def finish_write_and_reserve_read(
|
|
536
|
+
self,
|
|
537
|
+
keys: list[ObjectKey],
|
|
538
|
+
extra_count: int = 0,
|
|
539
|
+
) -> dict[ObjectKey, L1OperationResult]:
|
|
540
|
+
"""Atomically finish write and acquire read lock for the given keys.
|
|
541
|
+
|
|
542
|
+
This is used by the prefetch controller after successfully loading
|
|
543
|
+
data from L2 into write-reserved L1 buffers. It transitions the
|
|
544
|
+
object from write-locked to read-locked in a single atomic step,
|
|
545
|
+
preventing a race window where eviction could interfere.
|
|
546
|
+
|
|
547
|
+
Args:
|
|
548
|
+
keys: Keys to transition from write-locked to read-locked.
|
|
549
|
+
extra_count: Extra read locks on top of the default 1 lock.
|
|
550
|
+
Total locks acquired per key = 1 + extra_count. Useful
|
|
551
|
+
when multiple TP workers each consume one read lock for
|
|
552
|
+
the same key (e.g. MLA models with TP > 1).
|
|
553
|
+
|
|
554
|
+
Returns:
|
|
555
|
+
A dictionary mapping each object key to a tuple of
|
|
556
|
+
(L1Error, Optional[MemoryObj]).
|
|
557
|
+
|
|
558
|
+
Errors:
|
|
559
|
+
KEY_NOT_EXIST: The key does not exist.
|
|
560
|
+
KEY_IN_WRONG_STATE: The key is not write-locked, or it already
|
|
561
|
+
has read locks.
|
|
562
|
+
"""
|
|
563
|
+
extra_count = _validate_extra_count(extra_count)
|
|
564
|
+
total = 1 + extra_count
|
|
565
|
+
ret: dict[ObjectKey, L1OperationResult] = {}
|
|
566
|
+
successful_keys: list[ObjectKey] = []
|
|
567
|
+
|
|
568
|
+
for key in keys:
|
|
569
|
+
entry = self._objects.get(key, None)
|
|
570
|
+
if entry is None:
|
|
571
|
+
ret[key] = (L1Error.KEY_NOT_EXIST, None)
|
|
572
|
+
continue
|
|
573
|
+
|
|
574
|
+
if not entry.write_lock.is_locked():
|
|
575
|
+
logger.warning(
|
|
576
|
+
"L1Manager: finish_write_and_reserve_read on "
|
|
577
|
+
"non-write-locked key %s",
|
|
578
|
+
key,
|
|
579
|
+
)
|
|
580
|
+
ret[key] = (L1Error.KEY_IN_WRONG_STATE, None)
|
|
581
|
+
continue
|
|
582
|
+
|
|
583
|
+
if entry.read_lock.is_locked():
|
|
584
|
+
logger.warning(
|
|
585
|
+
"L1Manager: finish_write_and_reserve_read on read-locked key %s",
|
|
586
|
+
key,
|
|
587
|
+
)
|
|
588
|
+
ret[key] = (L1Error.KEY_IN_WRONG_STATE, None)
|
|
589
|
+
continue
|
|
590
|
+
|
|
591
|
+
entry.write_lock.unlock()
|
|
592
|
+
for _ in range(total):
|
|
593
|
+
entry.read_lock.lock()
|
|
594
|
+
ret[key] = (L1Error.SUCCESS, entry.memory_obj)
|
|
595
|
+
successful_keys.append(key)
|
|
596
|
+
|
|
597
|
+
for listener in self._registered_listeners:
|
|
598
|
+
listener.on_l1_keys_finish_write_and_reserve_read(successful_keys)
|
|
599
|
+
self._event_bus.publish(
|
|
600
|
+
Event(
|
|
601
|
+
event_type=EventType.L1_WRITE_FINISHED_AND_READ_RESERVED,
|
|
602
|
+
metadata={"keys": successful_keys},
|
|
603
|
+
)
|
|
604
|
+
)
|
|
605
|
+
return ret
|
|
606
|
+
|
|
607
|
+
@l1_mgr_synchronized
|
|
608
|
+
def delete(self, keys: list[ObjectKey]) -> dict[ObjectKey, L1Error]:
|
|
609
|
+
"""Delete the given keys from L1 cache.
|
|
610
|
+
|
|
611
|
+
Args:
|
|
612
|
+
keys: The list of object keys to delete.
|
|
613
|
+
|
|
614
|
+
Returns:
|
|
615
|
+
A dictionary mapping each object key to an L1Error.
|
|
616
|
+
|
|
617
|
+
Errors:
|
|
618
|
+
KEY_NOT_EXIST: The key does not exist.
|
|
619
|
+
KEY_IS_LOCKED: The key is locked (either write-locked or read-locked
|
|
620
|
+
and cannot be deleted).
|
|
621
|
+
"""
|
|
622
|
+
need_to_free: list[MemoryObj] = []
|
|
623
|
+
ret: dict[ObjectKey, L1Error] = {}
|
|
624
|
+
successful_keys: list[ObjectKey] = []
|
|
625
|
+
|
|
626
|
+
for key in keys:
|
|
627
|
+
entry = self._objects.get(key, None)
|
|
628
|
+
if entry is None:
|
|
629
|
+
ret[key] = L1Error.KEY_NOT_EXIST
|
|
630
|
+
continue
|
|
631
|
+
|
|
632
|
+
if entry.read_lock.is_locked() or entry.write_lock.is_locked():
|
|
633
|
+
ret[key] = L1Error.KEY_IS_LOCKED
|
|
634
|
+
continue
|
|
635
|
+
|
|
636
|
+
need_to_free.append(entry.memory_obj)
|
|
637
|
+
del self._objects[key]
|
|
638
|
+
ret[key] = L1Error.SUCCESS
|
|
639
|
+
successful_keys.append(key)
|
|
640
|
+
|
|
641
|
+
self._memory_manager.free(need_to_free)
|
|
642
|
+
|
|
643
|
+
for listener in self._registered_listeners:
|
|
644
|
+
listener.on_l1_keys_deleted_by_manager(successful_keys)
|
|
645
|
+
self._event_bus.publish(
|
|
646
|
+
Event(
|
|
647
|
+
event_type=EventType.L1_KEYS_EVICTED,
|
|
648
|
+
metadata={"keys": successful_keys},
|
|
649
|
+
)
|
|
650
|
+
)
|
|
651
|
+
return ret
|
|
652
|
+
|
|
653
|
+
def touch_keys(self, keys: list[ObjectKey]):
|
|
654
|
+
"""Touch the given keys, marking the keys as accessed(retrieved or stored).
|
|
655
|
+
|
|
656
|
+
Args:
|
|
657
|
+
keys: The list of object keys to touch.
|
|
658
|
+
"""
|
|
659
|
+
for listener in self._registered_listeners:
|
|
660
|
+
listener.on_l1_keys_accessed(keys)
|
|
661
|
+
|
|
662
|
+
@l1_mgr_synchronized
|
|
663
|
+
def clear(self, force: bool = False) -> None:
|
|
664
|
+
"""Clear objects from L1 cache.
|
|
665
|
+
|
|
666
|
+
Args:
|
|
667
|
+
force: If True, clear ALL objects including locked ones.
|
|
668
|
+
This may corrupt in-flight store/prefetch operations.
|
|
669
|
+
If False (default), only clear unlocked objects, keeping
|
|
670
|
+
write-locked and read-locked objects intact.
|
|
671
|
+
"""
|
|
672
|
+
if force:
|
|
673
|
+
logger.warning(
|
|
674
|
+
"L1Manager: force-clearing all %d objects "
|
|
675
|
+
"(including locked ones). This may corrupt in-flight "
|
|
676
|
+
"store/prefetch operations — use with caution.",
|
|
677
|
+
len(self._objects),
|
|
678
|
+
)
|
|
679
|
+
all_keys = list(self._objects.keys())
|
|
680
|
+
all_memory_objs = [entry.memory_obj for entry in self._objects.values()]
|
|
681
|
+
self._memory_manager.free(all_memory_objs)
|
|
682
|
+
self._objects.clear()
|
|
683
|
+
for listener in self._registered_listeners:
|
|
684
|
+
listener.on_l1_keys_deleted_by_manager(all_keys)
|
|
685
|
+
self._event_bus.publish(
|
|
686
|
+
Event(
|
|
687
|
+
event_type=EventType.L1_KEYS_EVICTED,
|
|
688
|
+
metadata={"keys": all_keys},
|
|
689
|
+
)
|
|
690
|
+
)
|
|
691
|
+
logger.info(
|
|
692
|
+
"L1Manager: cleared %d objects, 0 remaining.",
|
|
693
|
+
len(all_keys),
|
|
694
|
+
)
|
|
695
|
+
return
|
|
696
|
+
|
|
697
|
+
keys_to_clear: list[ObjectKey] = []
|
|
698
|
+
objs_to_free: list[MemoryObj] = []
|
|
699
|
+
locked_count = 0
|
|
700
|
+
|
|
701
|
+
for key, entry in list(self._objects.items()):
|
|
702
|
+
if entry.write_lock.is_locked() or entry.read_lock.is_locked():
|
|
703
|
+
locked_count += 1
|
|
704
|
+
continue
|
|
705
|
+
keys_to_clear.append(key)
|
|
706
|
+
objs_to_free.append(entry.memory_obj)
|
|
707
|
+
|
|
708
|
+
for key in keys_to_clear:
|
|
709
|
+
del self._objects[key]
|
|
710
|
+
|
|
711
|
+
self._memory_manager.free(objs_to_free)
|
|
712
|
+
|
|
713
|
+
if keys_to_clear:
|
|
714
|
+
for listener in self._registered_listeners:
|
|
715
|
+
listener.on_l1_keys_deleted_by_manager(keys_to_clear)
|
|
716
|
+
self._event_bus.publish(
|
|
717
|
+
Event(
|
|
718
|
+
event_type=EventType.L1_KEYS_EVICTED,
|
|
719
|
+
metadata={"keys": keys_to_clear},
|
|
720
|
+
)
|
|
721
|
+
)
|
|
722
|
+
|
|
723
|
+
logger.info(
|
|
724
|
+
"L1Manager: cleared %d objects, %d locked objects remaining.",
|
|
725
|
+
len(keys_to_clear),
|
|
726
|
+
locked_count,
|
|
727
|
+
)
|
|
728
|
+
|
|
729
|
+
def is_key_evictable(self, key: ObjectKey) -> bool:
|
|
730
|
+
"""Check if a key is eligible for eviction (not locked).
|
|
731
|
+
|
|
732
|
+
This method does NOT acquire the global L1Manager lock.
|
|
733
|
+
L1Manager.delete() will check again and safely reject a key
|
|
734
|
+
that became locked between the check and the actual deletion.
|
|
735
|
+
|
|
736
|
+
Args:
|
|
737
|
+
key: The object key to check.
|
|
738
|
+
|
|
739
|
+
Returns:
|
|
740
|
+
True if the key exists and is not locked (neither read-locked
|
|
741
|
+
nor write-locked), False otherwise.
|
|
742
|
+
"""
|
|
743
|
+
entry = self._objects.get(key, None)
|
|
744
|
+
if entry is None:
|
|
745
|
+
return False
|
|
746
|
+
return not entry.read_lock.is_locked() and not entry.write_lock.is_locked()
|
|
747
|
+
|
|
748
|
+
def get_memory_usage(self) -> tuple[int, int]:
|
|
749
|
+
"""Get the current memory usage of L1 cache.
|
|
750
|
+
|
|
751
|
+
Returns:
|
|
752
|
+
A tuple of (used_memory_bytes, total_memory_bytes).
|
|
753
|
+
|
|
754
|
+
Note:
|
|
755
|
+
In the future, we many want to make a "callback" based mechanism
|
|
756
|
+
via "L1ManagerListener" to notify the memory usage changes.
|
|
757
|
+
"""
|
|
758
|
+
return self._memory_manager.get_memory_usage()
|
|
759
|
+
|
|
760
|
+
def get_l1_memory_desc(self):
|
|
761
|
+
"""Return an L1MemoryDesc describing the underlying L1 memory buffer."""
|
|
762
|
+
return self._memory_manager.get_l1_memory_desc()
|
|
763
|
+
|
|
764
|
+
def close(self) -> None:
|
|
765
|
+
"""Close the L1Manager and free all resources."""
|
|
766
|
+
with self._lock:
|
|
767
|
+
all_memory_objs = [entry.memory_obj for entry in self._objects.values()]
|
|
768
|
+
self._memory_manager.free(all_memory_objs)
|
|
769
|
+
self._objects.clear()
|
|
770
|
+
|
|
771
|
+
self._memory_manager.close()
|
|
772
|
+
|
|
773
|
+
# Status reporting
|
|
774
|
+
@l1_mgr_synchronized
|
|
775
|
+
def report_status(self) -> dict:
|
|
776
|
+
"""Return a status dict describing L1 cache state."""
|
|
777
|
+
write_locked = 0
|
|
778
|
+
read_locked = 0
|
|
779
|
+
temporary = 0
|
|
780
|
+
for entry in self._objects.values():
|
|
781
|
+
if entry.write_lock.is_locked():
|
|
782
|
+
write_locked += 1
|
|
783
|
+
if entry.read_lock.is_locked():
|
|
784
|
+
read_locked += 1
|
|
785
|
+
if entry.is_temporary:
|
|
786
|
+
temporary += 1
|
|
787
|
+
used, total = self._memory_manager.get_memory_usage()
|
|
788
|
+
return {
|
|
789
|
+
"is_healthy": self._memory_manager.memcheck(),
|
|
790
|
+
"total_object_count": len(self._objects),
|
|
791
|
+
"write_locked_count": write_locked,
|
|
792
|
+
"read_locked_count": read_locked,
|
|
793
|
+
"temporary_count": temporary,
|
|
794
|
+
"memory_used_bytes": used,
|
|
795
|
+
"memory_total_bytes": total,
|
|
796
|
+
"memory_usage_ratio": used / total if total > 0 else 0.0,
|
|
797
|
+
"write_ttl_seconds": self._write_ttl_seconds,
|
|
798
|
+
"read_ttl_seconds": self._read_ttl_seconds,
|
|
799
|
+
}
|
|
800
|
+
|
|
801
|
+
# Debugging APIs
|
|
802
|
+
@l1_mgr_synchronized
|
|
803
|
+
def get_object_state(self, key: ObjectKey) -> L1ObjectState | None:
|
|
804
|
+
"""Get the internal state of the object with the given key.
|
|
805
|
+
|
|
806
|
+
Args:
|
|
807
|
+
key: The object key.
|
|
808
|
+
|
|
809
|
+
Returns:
|
|
810
|
+
The L1ObjectState if the object exists, None otherwise.
|
|
811
|
+
"""
|
|
812
|
+
return self._objects.get(key, None)
|
|
813
|
+
|
|
814
|
+
@l1_mgr_synchronized
|
|
815
|
+
def memcheck(self) -> bool:
|
|
816
|
+
"""Perform memory check for L1 cache."""
|
|
817
|
+
mem_check_result = self._memory_manager.memcheck()
|
|
818
|
+
|
|
819
|
+
# Log the locked objects for debugging
|
|
820
|
+
num_write_locked = 0
|
|
821
|
+
num_read_locked = 0
|
|
822
|
+
for key, entry in self._objects.items():
|
|
823
|
+
if entry.write_lock.is_locked():
|
|
824
|
+
num_write_locked += 1
|
|
825
|
+
if entry.read_lock.is_locked():
|
|
826
|
+
num_read_locked += 1
|
|
827
|
+
|
|
828
|
+
logger.info(
|
|
829
|
+
"L1Manager memcheck: total objects = %d, write-locked = %d, "
|
|
830
|
+
"read-locked = %d",
|
|
831
|
+
len(self._objects),
|
|
832
|
+
num_write_locked,
|
|
833
|
+
num_read_locked,
|
|
834
|
+
)
|
|
835
|
+
return mem_check_result
|