lmcache-cli 0.4.5.dev0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- lmcache/__init__.py +84 -0
- lmcache/_version.py +24 -0
- lmcache/cli/__init__.py +1 -0
- lmcache/cli/commands/__init__.py +34 -0
- lmcache/cli/commands/base.py +157 -0
- lmcache/cli/commands/bench/__init__.py +557 -0
- lmcache/cli/commands/bench/engine_bench/__init__.py +1 -0
- lmcache/cli/commands/bench/engine_bench/config.py +245 -0
- lmcache/cli/commands/bench/engine_bench/interactive/__init__.py +274 -0
- lmcache/cli/commands/bench/engine_bench/interactive/config.json +10 -0
- lmcache/cli/commands/bench/engine_bench/interactive/schema.py +352 -0
- lmcache/cli/commands/bench/engine_bench/interactive/state.py +327 -0
- lmcache/cli/commands/bench/engine_bench/interactive/terminal.py +291 -0
- lmcache/cli/commands/bench/engine_bench/progress.py +145 -0
- lmcache/cli/commands/bench/engine_bench/request_sender.py +232 -0
- lmcache/cli/commands/bench/engine_bench/stats.py +275 -0
- lmcache/cli/commands/bench/engine_bench/workloads/__init__.py +153 -0
- lmcache/cli/commands/bench/engine_bench/workloads/base.py +122 -0
- lmcache/cli/commands/bench/engine_bench/workloads/long_doc_permutator.py +435 -0
- lmcache/cli/commands/bench/engine_bench/workloads/long_doc_qa.py +281 -0
- lmcache/cli/commands/bench/engine_bench/workloads/multi_round_chat.py +337 -0
- lmcache/cli/commands/bench/engine_bench/workloads/random_prefill.py +178 -0
- lmcache/cli/commands/describe.py +310 -0
- lmcache/cli/commands/kvcache.py +133 -0
- lmcache/cli/commands/mock.py +75 -0
- lmcache/cli/commands/ping.py +113 -0
- lmcache/cli/commands/query/__init__.py +155 -0
- lmcache/cli/commands/query/prompt.py +134 -0
- lmcache/cli/commands/query/request.py +357 -0
- lmcache/cli/commands/server.py +99 -0
- lmcache/cli/commands/tool/__init__.py +63 -0
- lmcache/cli/commands/tool/cache_simulator.py +113 -0
- lmcache/cli/commands/trace/__init__.py +505 -0
- lmcache/cli/commands/trace/dispatch.py +249 -0
- lmcache/cli/commands/trace/driver.py +372 -0
- lmcache/cli/commands/trace/stats.py +289 -0
- lmcache/cli/documents/lmcache.txt +11 -0
- lmcache/cli/main.py +42 -0
- lmcache/cli/metrics/__init__.py +29 -0
- lmcache/cli/metrics/formatter.py +171 -0
- lmcache/cli/metrics/handler.py +94 -0
- lmcache/cli/metrics/metrics.py +161 -0
- lmcache/cli/metrics/section.py +77 -0
- lmcache/connections.py +173 -0
- lmcache/integration/__init__.py +2 -0
- lmcache/integration/base_service_factory.py +165 -0
- lmcache/integration/request_telemetry/__init__.py +1 -0
- lmcache/integration/request_telemetry/base.py +51 -0
- lmcache/integration/request_telemetry/factory.py +113 -0
- lmcache/integration/request_telemetry/fastapi.py +109 -0
- lmcache/integration/request_telemetry/noop.py +35 -0
- lmcache/integration/sglang/__init__.py +2 -0
- lmcache/integration/sglang/sglang_adapter.py +326 -0
- lmcache/integration/sglang/utils.py +39 -0
- lmcache/integration/vllm/__init__.py +1 -0
- lmcache/integration/vllm/lmcache_connector_v1.py +213 -0
- lmcache/integration/vllm/lmcache_connector_v1_085.py +150 -0
- lmcache/integration/vllm/lmcache_mp_connector_0180.py +1072 -0
- lmcache/integration/vllm/tests/test_mm_hash_utils.py +112 -0
- lmcache/integration/vllm/utils.py +433 -0
- lmcache/integration/vllm/vllm_multi_process_adapter.py +1090 -0
- lmcache/integration/vllm/vllm_service_factory.py +339 -0
- lmcache/integration/vllm/vllm_v1_adapter.py +1713 -0
- lmcache/logging.py +107 -0
- lmcache/native_storage_ops.pyi +230 -0
- lmcache/non_cuda_equivalents.py +1424 -0
- lmcache/observability.py +1958 -0
- lmcache/storage_backend/serde/__init__.py +1 -0
- lmcache/storage_backend/serde/cachegen_basics.py +210 -0
- lmcache/storage_backend/serde/cachegen_decoder.py +207 -0
- lmcache/storage_backend/serde/cachegen_encoder.py +394 -0
- lmcache/storage_backend/serde/serde.py +75 -0
- lmcache/tools/__init__.py +1 -0
- lmcache/tools/cache_simulator/README.md +392 -0
- lmcache/tools/cache_simulator/__init__.py +1 -0
- lmcache/tools/cache_simulator/docs/simulate_example.png +0 -0
- lmcache/tools/cache_simulator/docs/sweep_example.png +0 -0
- lmcache/tools/cache_simulator/gen_bench_dataset.py +360 -0
- lmcache/tools/cache_simulator/lru_cache.py +124 -0
- lmcache/tools/cache_simulator/plot_hit_rate.py +231 -0
- lmcache/tools/cache_simulator/simulator.py +795 -0
- lmcache/tools/controller_benchmark/README.md +161 -0
- lmcache/tools/controller_benchmark/__init__.py +1 -0
- lmcache/tools/controller_benchmark/__main__.py +331 -0
- lmcache/tools/controller_benchmark/benchmark.py +660 -0
- lmcache/tools/controller_benchmark/config.py +44 -0
- lmcache/tools/controller_benchmark/constants.py +10 -0
- lmcache/tools/controller_benchmark/handlers/__init__.py +46 -0
- lmcache/tools/controller_benchmark/handlers/admit.py +52 -0
- lmcache/tools/controller_benchmark/handlers/base.py +47 -0
- lmcache/tools/controller_benchmark/handlers/deregister.py +49 -0
- lmcache/tools/controller_benchmark/handlers/evict.py +52 -0
- lmcache/tools/controller_benchmark/handlers/heartbeat.py +56 -0
- lmcache/tools/controller_benchmark/handlers/p2p_lookup.py +47 -0
- lmcache/tools/controller_benchmark/handlers/register.py +56 -0
- lmcache/tools/mp_status_viewer/__init__.py +1 -0
- lmcache/tools/mp_status_viewer/__main__.py +95 -0
- lmcache/usage_context.py +417 -0
- lmcache/utils.py +665 -0
- lmcache/v1/__init__.py +2 -0
- lmcache/v1/api_server/__init__.py +2 -0
- lmcache/v1/api_server/__main__.py +537 -0
- lmcache/v1/basic_check.py +112 -0
- lmcache/v1/cache_controller/__init__.py +9 -0
- lmcache/v1/cache_controller/commands/__init__.py +15 -0
- lmcache/v1/cache_controller/commands/base.py +35 -0
- lmcache/v1/cache_controller/commands/full_sync.py +49 -0
- lmcache/v1/cache_controller/config.py +176 -0
- lmcache/v1/cache_controller/controller_manager.py +535 -0
- lmcache/v1/cache_controller/controllers/__init__.py +11 -0
- lmcache/v1/cache_controller/controllers/full_sync_tracker.py +473 -0
- lmcache/v1/cache_controller/controllers/kv_controller.py +439 -0
- lmcache/v1/cache_controller/controllers/registration_controller.py +282 -0
- lmcache/v1/cache_controller/executor.py +463 -0
- lmcache/v1/cache_controller/frontend/static/css/style.css +201 -0
- lmcache/v1/cache_controller/frontend/static/img/logo.png +0 -0
- lmcache/v1/cache_controller/frontend/static/index.html +234 -0
- lmcache/v1/cache_controller/frontend/static/js/controller_app.js +660 -0
- lmcache/v1/cache_controller/full_sync_sender.py +475 -0
- lmcache/v1/cache_controller/locks.py +149 -0
- lmcache/v1/cache_controller/message.py +828 -0
- lmcache/v1/cache_controller/observability.py +208 -0
- lmcache/v1/cache_controller/utils.py +679 -0
- lmcache/v1/cache_controller/worker.py +665 -0
- lmcache/v1/cache_engine.py +2058 -0
- lmcache/v1/cache_interface.py +19 -0
- lmcache/v1/check/__init__.py +74 -0
- lmcache/v1/check/check_mode_gen.py +86 -0
- lmcache/v1/check/check_mode_test_l2_adapter.py +284 -0
- lmcache/v1/check/check_mode_test_remote.py +155 -0
- lmcache/v1/check/check_mode_test_storage_manager.py +142 -0
- lmcache/v1/check/utils.py +571 -0
- lmcache/v1/compute/__init__.py +2 -0
- lmcache/v1/compute/attention/__init__.py +0 -0
- lmcache/v1/compute/attention/abstract.py +39 -0
- lmcache/v1/compute/attention/flash_attn.py +129 -0
- lmcache/v1/compute/attention/flash_infer_sparse.py +284 -0
- lmcache/v1/compute/attention/metadata.py +85 -0
- lmcache/v1/compute/attention/utils.py +14 -0
- lmcache/v1/compute/blend/__init__.py +7 -0
- lmcache/v1/compute/blend/blender.py +168 -0
- lmcache/v1/compute/blend/metadata.py +34 -0
- lmcache/v1/compute/blend/utils.py +63 -0
- lmcache/v1/compute/models/__init__.py +0 -0
- lmcache/v1/compute/models/base.py +141 -0
- lmcache/v1/compute/models/llama.py +9 -0
- lmcache/v1/compute/models/qwen3.py +24 -0
- lmcache/v1/compute/models/utils.py +68 -0
- lmcache/v1/compute/positional_encoding.py +199 -0
- lmcache/v1/config.py +848 -0
- lmcache/v1/config_base.py +848 -0
- lmcache/v1/distributed/api.py +248 -0
- lmcache/v1/distributed/config.py +321 -0
- lmcache/v1/distributed/error.py +64 -0
- lmcache/v1/distributed/eviction.py +192 -0
- lmcache/v1/distributed/eviction_policy/__init__.py +21 -0
- lmcache/v1/distributed/eviction_policy/factory.py +27 -0
- lmcache/v1/distributed/eviction_policy/lru.py +244 -0
- lmcache/v1/distributed/eviction_policy/noop.py +50 -0
- lmcache/v1/distributed/internal_api.py +170 -0
- lmcache/v1/distributed/l1_manager.py +835 -0
- lmcache/v1/distributed/l2_adapters/__init__.py +67 -0
- lmcache/v1/distributed/l2_adapters/base.py +360 -0
- lmcache/v1/distributed/l2_adapters/config.py +385 -0
- lmcache/v1/distributed/l2_adapters/factory.py +205 -0
- lmcache/v1/distributed/l2_adapters/fs_l2_adapter.py +747 -0
- lmcache/v1/distributed/l2_adapters/fs_native_l2_adapter.py +167 -0
- lmcache/v1/distributed/l2_adapters/mock_l2_adapter.py +516 -0
- lmcache/v1/distributed/l2_adapters/mooncake_store_l2_adapter.py +135 -0
- lmcache/v1/distributed/l2_adapters/native_connector_l2_adapter.py +468 -0
- lmcache/v1/distributed/l2_adapters/native_plugin_l2_adapter.py +199 -0
- lmcache/v1/distributed/l2_adapters/nixl_store_dynamic_l2_adapter.py +831 -0
- lmcache/v1/distributed/l2_adapters/nixl_store_l2_adapter.py +983 -0
- lmcache/v1/distributed/l2_adapters/plugin_l2_adapter.py +210 -0
- lmcache/v1/distributed/l2_adapters/resp_l2_adapter.py +176 -0
- lmcache/v1/distributed/memory_manager.py +179 -0
- lmcache/v1/distributed/storage_controller.py +39 -0
- lmcache/v1/distributed/storage_controllers/__init__.py +43 -0
- lmcache/v1/distributed/storage_controllers/eviction_controller.py +242 -0
- lmcache/v1/distributed/storage_controllers/prefetch_controller.py +830 -0
- lmcache/v1/distributed/storage_controllers/prefetch_policy.py +193 -0
- lmcache/v1/distributed/storage_controllers/store_controller.py +452 -0
- lmcache/v1/distributed/storage_controllers/store_policy.py +213 -0
- lmcache/v1/distributed/storage_manager.py +532 -0
- lmcache/v1/event_manager.py +145 -0
- lmcache/v1/exceptions/__init__.py +16 -0
- lmcache/v1/gpu_connector/__init__.py +126 -0
- lmcache/v1/gpu_connector/gpu_connectors.py +1906 -0
- lmcache/v1/gpu_connector/gpu_ops.py +85 -0
- lmcache/v1/gpu_connector/hpu_connector.py +326 -0
- lmcache/v1/gpu_connector/mock_gpu_connector.py +67 -0
- lmcache/v1/gpu_connector/utils.py +890 -0
- lmcache/v1/gpu_connector/xpu_connectors.py +916 -0
- lmcache/v1/health_monitor/__init__.py +1 -0
- lmcache/v1/health_monitor/base.py +587 -0
- lmcache/v1/health_monitor/checks/__init__.py +1 -0
- lmcache/v1/health_monitor/checks/remote_backend_check.py +304 -0
- lmcache/v1/health_monitor/constants.py +36 -0
- lmcache/v1/internal_api_server/__init__.py +0 -0
- lmcache/v1/internal_api_server/api_registry.py +59 -0
- lmcache/v1/internal_api_server/api_server.py +120 -0
- lmcache/v1/internal_api_server/common/__init__.py +1 -0
- lmcache/v1/internal_api_server/common/env_api.py +22 -0
- lmcache/v1/internal_api_server/common/loglevel_api.py +57 -0
- lmcache/v1/internal_api_server/common/metrics_api.py +29 -0
- lmcache/v1/internal_api_server/common/periodic_thread_api.py +138 -0
- lmcache/v1/internal_api_server/common/run_script_api.py +73 -0
- lmcache/v1/internal_api_server/common/thread_api.py +63 -0
- lmcache/v1/internal_api_server/controller/__init__.py +1 -0
- lmcache/v1/internal_api_server/controller/key_stats_api.py +81 -0
- lmcache/v1/internal_api_server/controller/worker_info_api.py +136 -0
- lmcache/v1/internal_api_server/utils.py +43 -0
- lmcache/v1/internal_api_server/vllm/__init__.py +1 -0
- lmcache/v1/internal_api_server/vllm/backend_api.py +221 -0
- lmcache/v1/internal_api_server/vllm/bypass_api.py +204 -0
- lmcache/v1/internal_api_server/vllm/cache_api.py +895 -0
- lmcache/v1/internal_api_server/vllm/chunk_statistics_api.py +141 -0
- lmcache/v1/internal_api_server/vllm/conf_api.py +147 -0
- lmcache/v1/internal_api_server/vllm/freeze_api.py +172 -0
- lmcache/v1/internal_api_server/vllm/hot_cache_api.py +184 -0
- lmcache/v1/internal_api_server/vllm/inference_api.py +65 -0
- lmcache/v1/internal_api_server/vllm/load_fs_chunks_api.py +320 -0
- lmcache/v1/internal_api_server/vllm/lookup_api.py +145 -0
- lmcache/v1/internal_api_server/vllm/version_api.py +25 -0
- lmcache/v1/kv_layer_groups.py +267 -0
- lmcache/v1/lazy_memory_allocator.py +284 -0
- lmcache/v1/lookup_client/__init__.py +25 -0
- lmcache/v1/lookup_client/abstract_client.py +77 -0
- lmcache/v1/lookup_client/async_lookup_message.py +50 -0
- lmcache/v1/lookup_client/chunk_statistics_lookup_client.py +200 -0
- lmcache/v1/lookup_client/factory.py +251 -0
- lmcache/v1/lookup_client/hit_limit_lookup_client.py +86 -0
- lmcache/v1/lookup_client/lmcache_async_lookup_client.py +407 -0
- lmcache/v1/lookup_client/lmcache_lookup_client.py +285 -0
- lmcache/v1/lookup_client/lmcache_lookup_client_bypass.py +99 -0
- lmcache/v1/lookup_client/mooncake_lookup_client.py +87 -0
- lmcache/v1/lookup_client/record_strategies/__init__.py +77 -0
- lmcache/v1/lookup_client/record_strategies/base.py +327 -0
- lmcache/v1/lookup_client/record_strategies/file_hash.py +130 -0
- lmcache/v1/lookup_client/record_strategies/memory_bloom_filter.py +81 -0
- lmcache/v1/manager.py +539 -0
- lmcache/v1/memory_management.py +2619 -0
- lmcache/v1/metadata.py +114 -0
- lmcache/v1/mp_observability/AGENTS.override.md +21 -0
- lmcache/v1/mp_observability/README.md +204 -0
- lmcache/v1/mp_observability/config.py +340 -0
- lmcache/v1/mp_observability/event.py +100 -0
- lmcache/v1/mp_observability/event_bus.py +313 -0
- lmcache/v1/mp_observability/otel_init.py +145 -0
- lmcache/v1/mp_observability/subscribers/__init__.py +28 -0
- lmcache/v1/mp_observability/subscribers/logging/__init__.py +19 -0
- lmcache/v1/mp_observability/subscribers/logging/l1.py +56 -0
- lmcache/v1/mp_observability/subscribers/logging/l2.py +73 -0
- lmcache/v1/mp_observability/subscribers/logging/lookup_hash.py +209 -0
- lmcache/v1/mp_observability/subscribers/logging/mp_server.py +90 -0
- lmcache/v1/mp_observability/subscribers/logging/sm.py +59 -0
- lmcache/v1/mp_observability/subscribers/metrics/__init__.py +20 -0
- lmcache/v1/mp_observability/subscribers/metrics/l0_lifecycle.py +290 -0
- lmcache/v1/mp_observability/subscribers/metrics/l1.py +55 -0
- lmcache/v1/mp_observability/subscribers/metrics/l1_lifecycle.py +166 -0
- lmcache/v1/mp_observability/subscribers/metrics/l2.py +121 -0
- lmcache/v1/mp_observability/subscribers/metrics/sm.py +69 -0
- lmcache/v1/mp_observability/subscribers/tracing/__init__.py +12 -0
- lmcache/v1/mp_observability/subscribers/tracing/mp_server.py +333 -0
- lmcache/v1/mp_observability/subscribers/tracing/span_registry.py +148 -0
- lmcache/v1/mp_observability/trace/__init__.py +50 -0
- lmcache/v1/mp_observability/trace/codecs.py +255 -0
- lmcache/v1/mp_observability/trace/decorator.py +147 -0
- lmcache/v1/mp_observability/trace/format.py +132 -0
- lmcache/v1/mp_observability/trace/lifecycle.py +83 -0
- lmcache/v1/mp_observability/trace/reader.py +167 -0
- lmcache/v1/mp_observability/trace/recorder.py +300 -0
- lmcache/v1/multiprocess/__init__.py +0 -0
- lmcache/v1/multiprocess/affinity_pool.py +102 -0
- lmcache/v1/multiprocess/blend_server_v2.py +891 -0
- lmcache/v1/multiprocess/config.py +253 -0
- lmcache/v1/multiprocess/custom_types.py +281 -0
- lmcache/v1/multiprocess/futures.py +194 -0
- lmcache/v1/multiprocess/gpu_context.py +511 -0
- lmcache/v1/multiprocess/http_server.py +235 -0
- lmcache/v1/multiprocess/mp_runtime_plugin_launcher.py +130 -0
- lmcache/v1/multiprocess/mq.py +732 -0
- lmcache/v1/multiprocess/protocol.py +86 -0
- lmcache/v1/multiprocess/protocols/README.md +213 -0
- lmcache/v1/multiprocess/protocols/__init__.py +127 -0
- lmcache/v1/multiprocess/protocols/base.py +89 -0
- lmcache/v1/multiprocess/protocols/blend.py +109 -0
- lmcache/v1/multiprocess/protocols/blend_v2.py +57 -0
- lmcache/v1/multiprocess/protocols/controller.py +53 -0
- lmcache/v1/multiprocess/protocols/debug.py +34 -0
- lmcache/v1/multiprocess/protocols/engine.py +146 -0
- lmcache/v1/multiprocess/protocols/observability.py +39 -0
- lmcache/v1/multiprocess/server.py +1134 -0
- lmcache/v1/multiprocess/session.py +190 -0
- lmcache/v1/multiprocess/token_hasher.py +441 -0
- lmcache/v1/offload_server/__init__.py +17 -0
- lmcache/v1/offload_server/abstract_server.py +37 -0
- lmcache/v1/offload_server/message.py +30 -0
- lmcache/v1/offload_server/zmq_server.py +122 -0
- lmcache/v1/periodic_thread.py +579 -0
- lmcache/v1/pin_monitor.py +246 -0
- lmcache/v1/plugin/__init__.py +0 -0
- lmcache/v1/plugin/runtime_plugin_launcher.py +211 -0
- lmcache/v1/protocol.py +317 -0
- lmcache/v1/rpc/__init__.py +17 -0
- lmcache/v1/rpc/transport.py +105 -0
- lmcache/v1/rpc/zmq_transport.py +213 -0
- lmcache/v1/rpc_utils.py +165 -0
- lmcache/v1/server/__init__.py +2 -0
- lmcache/v1/server/__main__.py +170 -0
- lmcache/v1/server/storage_backend/__init__.py +21 -0
- lmcache/v1/server/storage_backend/abstract_backend.py +80 -0
- lmcache/v1/server/storage_backend/local_backend.py +75 -0
- lmcache/v1/server/utils.py +21 -0
- lmcache/v1/standalone/__init__.py +1 -0
- lmcache/v1/standalone/__main__.py +583 -0
- lmcache/v1/standalone/manager.py +80 -0
- lmcache/v1/standalone/standalone_service_factory.py +86 -0
- lmcache/v1/storage_backend/__init__.py +313 -0
- lmcache/v1/storage_backend/abstract_backend.py +445 -0
- lmcache/v1/storage_backend/audit_backend.py +233 -0
- lmcache/v1/storage_backend/batched_message_sender.py +222 -0
- lmcache/v1/storage_backend/cache_policy/__init__.py +45 -0
- lmcache/v1/storage_backend/cache_policy/base_policy.py +87 -0
- lmcache/v1/storage_backend/cache_policy/fifo.py +58 -0
- lmcache/v1/storage_backend/cache_policy/lfu.py +105 -0
- lmcache/v1/storage_backend/cache_policy/lru.py +81 -0
- lmcache/v1/storage_backend/cache_policy/mru.py +61 -0
- lmcache/v1/storage_backend/connector/__init__.py +443 -0
- lmcache/v1/storage_backend/connector/audit_adapter.py +77 -0
- lmcache/v1/storage_backend/connector/audit_connector.py +320 -0
- lmcache/v1/storage_backend/connector/base_connector.py +379 -0
- lmcache/v1/storage_backend/connector/blackhole_adapter.py +21 -0
- lmcache/v1/storage_backend/connector/blackhole_connector.py +37 -0
- lmcache/v1/storage_backend/connector/eic_adapter.py +31 -0
- lmcache/v1/storage_backend/connector/eic_connector.py +757 -0
- lmcache/v1/storage_backend/connector/external_adapter.py +79 -0
- lmcache/v1/storage_backend/connector/fs_adapter.py +51 -0
- lmcache/v1/storage_backend/connector/fs_connector.py +403 -0
- lmcache/v1/storage_backend/connector/infinistore_adapter.py +56 -0
- lmcache/v1/storage_backend/connector/infinistore_connector.py +177 -0
- lmcache/v1/storage_backend/connector/instrumented_connector.py +219 -0
- lmcache/v1/storage_backend/connector/lm_adapter.py +31 -0
- lmcache/v1/storage_backend/connector/lm_connector.py +176 -0
- lmcache/v1/storage_backend/connector/mock_adapter.py +57 -0
- lmcache/v1/storage_backend/connector/mock_connector.py +349 -0
- lmcache/v1/storage_backend/connector/mooncakestore_adapter.py +43 -0
- lmcache/v1/storage_backend/connector/mooncakestore_connector.py +614 -0
- lmcache/v1/storage_backend/connector/redis_adapter.py +181 -0
- lmcache/v1/storage_backend/connector/redis_connector.py +828 -0
- lmcache/v1/storage_backend/connector/s3_adapter.py +59 -0
- lmcache/v1/storage_backend/connector/s3_connector.py +699 -0
- lmcache/v1/storage_backend/connector/sagemaker_hyperpod_adapter.py +233 -0
- lmcache/v1/storage_backend/connector/sagemaker_hyperpod_connector.py +987 -0
- lmcache/v1/storage_backend/connector/valkey_adapter.py +114 -0
- lmcache/v1/storage_backend/connector/valkey_connector.py +627 -0
- lmcache/v1/storage_backend/gds_backend.py +1199 -0
- lmcache/v1/storage_backend/job_executor/__init__.py +0 -0
- lmcache/v1/storage_backend/job_executor/base_executor.py +34 -0
- lmcache/v1/storage_backend/job_executor/pq_executor.py +235 -0
- lmcache/v1/storage_backend/local_cpu_backend.py +810 -0
- lmcache/v1/storage_backend/local_disk_backend.py +656 -0
- lmcache/v1/storage_backend/maru_backend.py +734 -0
- lmcache/v1/storage_backend/naive_serde/__init__.py +50 -0
- lmcache/v1/storage_backend/naive_serde/cachegen_basics.py +133 -0
- lmcache/v1/storage_backend/naive_serde/cachegen_decoder.py +135 -0
- lmcache/v1/storage_backend/naive_serde/cachegen_encoder.py +83 -0
- lmcache/v1/storage_backend/naive_serde/kivi_serde.py +22 -0
- lmcache/v1/storage_backend/naive_serde/naive_serde.py +18 -0
- lmcache/v1/storage_backend/naive_serde/serde.py +37 -0
- lmcache/v1/storage_backend/native_clients/connector_client_base.py +165 -0
- lmcache/v1/storage_backend/native_clients/resp_client.py +35 -0
- lmcache/v1/storage_backend/nixl_storage_backend.py +1400 -0
- lmcache/v1/storage_backend/p2p_backend.py +788 -0
- lmcache/v1/storage_backend/path_sharder.py +117 -0
- lmcache/v1/storage_backend/pd_backend.py +646 -0
- lmcache/v1/storage_backend/plugins/dax_backend.py +1443 -0
- lmcache/v1/storage_backend/plugins/rust_raw_block_backend.py +1361 -0
- lmcache/v1/storage_backend/remote_backend.py +624 -0
- lmcache/v1/storage_backend/resp_client.py +227 -0
- lmcache/v1/storage_backend/storage_backend_listener.py +19 -0
- lmcache/v1/storage_backend/storage_manager.py +1352 -0
- lmcache/v1/system_detection.py +110 -0
- lmcache/v1/token_database.py +551 -0
- lmcache/v1/transfer_channel/__init__.py +83 -0
- lmcache/v1/transfer_channel/abstract.py +285 -0
- lmcache/v1/transfer_channel/mock_memory_channel.py +156 -0
- lmcache/v1/transfer_channel/nixl_channel.py +639 -0
- lmcache/v1/transfer_channel/py_socket_channel.py +260 -0
- lmcache/v1/transfer_channel/transfer_utils.py +63 -0
- lmcache/v1/utils/__init__.py +1 -0
- lmcache/v1/utils/bloom_filter.py +109 -0
- lmcache/v1/utils/cache_utils.py +125 -0
- lmcache_cli-0.4.5.dev0.dist-info/METADATA +185 -0
- lmcache_cli-0.4.5.dev0.dist-info/RECORD +399 -0
- lmcache_cli-0.4.5.dev0.dist-info/WHEEL +5 -0
- lmcache_cli-0.4.5.dev0.dist-info/entry_points.txt +2 -0
- lmcache_cli-0.4.5.dev0.dist-info/licenses/LICENSE +201 -0
- lmcache_cli-0.4.5.dev0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,475 @@
|
|
|
1
|
+
# SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
# Standard
|
|
3
|
+
from dataclasses import dataclass
|
|
4
|
+
from typing import TYPE_CHECKING, List, Optional
|
|
5
|
+
import asyncio
|
|
6
|
+
import random
|
|
7
|
+
import uuid
|
|
8
|
+
|
|
9
|
+
# First Party
|
|
10
|
+
from lmcache.logging import init_logger
|
|
11
|
+
from lmcache.v1.cache_controller.message import (
|
|
12
|
+
FullSyncBatchMsg,
|
|
13
|
+
FullSyncEndMsg,
|
|
14
|
+
FullSyncStartMsg,
|
|
15
|
+
FullSyncStartRetMsg,
|
|
16
|
+
FullSyncStatusMsg,
|
|
17
|
+
FullSyncStatusRetMsg,
|
|
18
|
+
)
|
|
19
|
+
from lmcache.v1.config import LMCacheEngineConfig
|
|
20
|
+
|
|
21
|
+
if TYPE_CHECKING:
|
|
22
|
+
# First Party
|
|
23
|
+
from lmcache.v1.cache_controller.worker import LMCacheWorker
|
|
24
|
+
from lmcache.v1.cache_engine import LMCacheEngine
|
|
25
|
+
from lmcache.v1.storage_backend.local_cpu_backend import LocalCPUBackend
|
|
26
|
+
|
|
27
|
+
logger = init_logger(__name__)
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
@dataclass
|
|
31
|
+
class SyncInitResult:
|
|
32
|
+
"""Result of sync initialization"""
|
|
33
|
+
|
|
34
|
+
sync_id: str
|
|
35
|
+
keys: List[int]
|
|
36
|
+
total_keys: int
|
|
37
|
+
batch_count: int
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
@dataclass
|
|
41
|
+
class BatchInfo:
|
|
42
|
+
"""Information about a batch for resending"""
|
|
43
|
+
|
|
44
|
+
batch_id: int
|
|
45
|
+
start_idx: int
|
|
46
|
+
end_idx: int
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
class FullSyncSender:
|
|
50
|
+
"""
|
|
51
|
+
Handles full sync of hot_cache keys to the Controller.
|
|
52
|
+
|
|
53
|
+
This class manages the process of sending all keys from the local hot_cache
|
|
54
|
+
to the Controller when a full sync is requested (e.g., after Controller restart).
|
|
55
|
+
|
|
56
|
+
The sync process:
|
|
57
|
+
1. Enter freeze mode (prevent allocations)
|
|
58
|
+
2. Add random startup delay (avoid thundering herd)
|
|
59
|
+
3. Send FullSyncStartMsg and wait for confirmation
|
|
60
|
+
4. Send keys in batches via FullSyncBatchMsg
|
|
61
|
+
5. Send FullSyncEndMsg
|
|
62
|
+
6. Poll for completion status
|
|
63
|
+
7. Exit freeze mode when threshold is reached
|
|
64
|
+
"""
|
|
65
|
+
|
|
66
|
+
def __init__(
|
|
67
|
+
self,
|
|
68
|
+
config: LMCacheEngineConfig,
|
|
69
|
+
worker: "LMCacheWorker",
|
|
70
|
+
lmcache_engine: "LMCacheEngine",
|
|
71
|
+
local_cpu_backend: "LocalCPUBackend",
|
|
72
|
+
):
|
|
73
|
+
# Configuration
|
|
74
|
+
self.batch_size = config.get_extra_config_value("full_sync_batch_size", 2000)
|
|
75
|
+
self.batch_interval_ms = config.get_extra_config_value(
|
|
76
|
+
"full_sync_batch_interval_ms", 5
|
|
77
|
+
)
|
|
78
|
+
self.startup_delay_range_s = config.get_extra_config_value(
|
|
79
|
+
"full_sync_startup_delay_s", 5.0
|
|
80
|
+
)
|
|
81
|
+
self.status_poll_interval_s = config.get_extra_config_value(
|
|
82
|
+
"full_sync_status_poll_interval_s", 5.0
|
|
83
|
+
)
|
|
84
|
+
self.max_retry_count = config.get_extra_config_value(
|
|
85
|
+
"full_sync_max_retry_count", 3
|
|
86
|
+
)
|
|
87
|
+
self.retry_delay_s = config.get_extra_config_value(
|
|
88
|
+
"full_sync_retry_delay_s", 1.0
|
|
89
|
+
)
|
|
90
|
+
self.max_poll_attempts = config.get_extra_config_value(
|
|
91
|
+
"full_sync_max_poll_attempts", 60
|
|
92
|
+
)
|
|
93
|
+
|
|
94
|
+
# Dependencies
|
|
95
|
+
self.worker = worker
|
|
96
|
+
self.lmcache_engine = lmcache_engine
|
|
97
|
+
self.local_cpu_backend = local_cpu_backend
|
|
98
|
+
self.config = config
|
|
99
|
+
|
|
100
|
+
# State
|
|
101
|
+
self._is_syncing = False
|
|
102
|
+
self._current_sync_id: Optional[str] = None
|
|
103
|
+
|
|
104
|
+
@property
|
|
105
|
+
def instance_id(self) -> str:
|
|
106
|
+
return self.config.lmcache_instance_id
|
|
107
|
+
|
|
108
|
+
@property
|
|
109
|
+
def worker_id(self) -> int:
|
|
110
|
+
return self.worker.worker_id
|
|
111
|
+
|
|
112
|
+
@property
|
|
113
|
+
def location(self) -> str:
|
|
114
|
+
return str(self.local_cpu_backend)
|
|
115
|
+
|
|
116
|
+
def _generate_sync_id(self) -> str:
|
|
117
|
+
"""Generate a unique sync session ID"""
|
|
118
|
+
return f"{self.instance_id}_{self.worker_id}_{uuid.uuid4().hex[:8]}"
|
|
119
|
+
|
|
120
|
+
def _get_all_hot_cache_keys(self) -> List[int]:
|
|
121
|
+
"""Get all chunk hashes from the hot cache"""
|
|
122
|
+
keys = self.local_cpu_backend.get_keys()
|
|
123
|
+
return [key.chunk_hash for key in keys]
|
|
124
|
+
|
|
125
|
+
async def _send_sync_start(
|
|
126
|
+
self, sync_id: str, total_keys: int, batch_count: int
|
|
127
|
+
) -> Optional[FullSyncStartRetMsg]:
|
|
128
|
+
"""Send FullSyncStartMsg and wait for confirmation"""
|
|
129
|
+
msg = FullSyncStartMsg(
|
|
130
|
+
instance_id=self.instance_id,
|
|
131
|
+
worker_id=self.worker_id,
|
|
132
|
+
location=self.location,
|
|
133
|
+
sync_id=sync_id,
|
|
134
|
+
total_keys=total_keys,
|
|
135
|
+
batch_count=batch_count,
|
|
136
|
+
)
|
|
137
|
+
|
|
138
|
+
try:
|
|
139
|
+
ret_msg = await self.worker.async_put_and_wait_msg(msg)
|
|
140
|
+
if isinstance(ret_msg, FullSyncStartRetMsg):
|
|
141
|
+
return ret_msg
|
|
142
|
+
else:
|
|
143
|
+
logger.error(
|
|
144
|
+
"Unexpected response type for FullSyncStartMsg: %s", type(ret_msg)
|
|
145
|
+
)
|
|
146
|
+
return None
|
|
147
|
+
except Exception as e:
|
|
148
|
+
logger.error("Error sending FullSyncStartMsg: %s", e)
|
|
149
|
+
return None
|
|
150
|
+
|
|
151
|
+
def _send_sync_batch(self, sync_id: str, batch_id: int, keys: List[int]) -> None:
|
|
152
|
+
"""Send a batch of keys via PUSH mode"""
|
|
153
|
+
msg = FullSyncBatchMsg(
|
|
154
|
+
instance_id=self.instance_id,
|
|
155
|
+
worker_id=self.worker_id,
|
|
156
|
+
location=self.location,
|
|
157
|
+
sync_id=sync_id,
|
|
158
|
+
batch_id=batch_id,
|
|
159
|
+
keys=keys,
|
|
160
|
+
)
|
|
161
|
+
self.worker.put_msg(msg)
|
|
162
|
+
|
|
163
|
+
def _send_sync_end(self, sync_id: str, actual_total_keys: int) -> None:
|
|
164
|
+
"""Send FullSyncEndMsg via PUSH mode"""
|
|
165
|
+
msg = FullSyncEndMsg(
|
|
166
|
+
instance_id=self.instance_id,
|
|
167
|
+
worker_id=self.worker_id,
|
|
168
|
+
location=self.location,
|
|
169
|
+
sync_id=sync_id,
|
|
170
|
+
actual_total_keys=actual_total_keys,
|
|
171
|
+
)
|
|
172
|
+
self.worker.put_msg(msg)
|
|
173
|
+
|
|
174
|
+
async def _query_sync_status(self, sync_id: str) -> Optional[FullSyncStatusRetMsg]:
|
|
175
|
+
"""Query sync status from controller"""
|
|
176
|
+
msg = FullSyncStatusMsg(
|
|
177
|
+
instance_id=self.instance_id,
|
|
178
|
+
worker_id=self.worker_id,
|
|
179
|
+
sync_id=sync_id,
|
|
180
|
+
)
|
|
181
|
+
|
|
182
|
+
try:
|
|
183
|
+
ret_msg = await self.worker.async_put_and_wait_msg(msg)
|
|
184
|
+
if isinstance(ret_msg, FullSyncStatusRetMsg):
|
|
185
|
+
return ret_msg
|
|
186
|
+
else:
|
|
187
|
+
logger.error(
|
|
188
|
+
"Unexpected response type for FullSyncStatusMsg: %s", type(ret_msg)
|
|
189
|
+
)
|
|
190
|
+
return None
|
|
191
|
+
except Exception as e:
|
|
192
|
+
logger.error("Error querying sync status: %s", e)
|
|
193
|
+
return None
|
|
194
|
+
|
|
195
|
+
async def _initialize_sync(
|
|
196
|
+
self, reason: Optional[str] = None
|
|
197
|
+
) -> Optional[SyncInitResult]:
|
|
198
|
+
"""
|
|
199
|
+
Initialize the sync process.
|
|
200
|
+
|
|
201
|
+
Handles startup delay, entering freeze mode, getting keys,
|
|
202
|
+
and sending the start message with retry.
|
|
203
|
+
|
|
204
|
+
Args:
|
|
205
|
+
reason: The reason for full sync
|
|
206
|
+
|
|
207
|
+
Returns:
|
|
208
|
+
SyncInitResult if initialization succeeded, None otherwise
|
|
209
|
+
"""
|
|
210
|
+
# Step 1: Random startup delay to avoid thundering herd
|
|
211
|
+
delay = random.uniform(0, self.startup_delay_range_s)
|
|
212
|
+
logger.info("Full sync startup delay: %.2fs", delay)
|
|
213
|
+
await asyncio.sleep(delay)
|
|
214
|
+
|
|
215
|
+
# Step 2: Enter freeze mode
|
|
216
|
+
logger.info("Entering freeze mode for full sync.")
|
|
217
|
+
self.lmcache_engine.freeze(True)
|
|
218
|
+
|
|
219
|
+
# Step 3: Get all keys from hot cache
|
|
220
|
+
keys = self._get_all_hot_cache_keys()
|
|
221
|
+
total_keys = len(keys)
|
|
222
|
+
batch_count = (total_keys + self.batch_size - 1) // self.batch_size
|
|
223
|
+
batch_count = max(batch_count, 1) # At least 1 batch even if empty
|
|
224
|
+
|
|
225
|
+
logger.info(
|
|
226
|
+
"Full sync: total_keys=%d, batch_size=%d, batch_count=%d",
|
|
227
|
+
total_keys,
|
|
228
|
+
self.batch_size,
|
|
229
|
+
batch_count,
|
|
230
|
+
)
|
|
231
|
+
|
|
232
|
+
# Step 4: Generate sync ID and send start message with retry
|
|
233
|
+
sync_id = self._generate_sync_id()
|
|
234
|
+
self._current_sync_id = sync_id
|
|
235
|
+
|
|
236
|
+
start_accepted = False
|
|
237
|
+
for attempt in range(self.max_retry_count):
|
|
238
|
+
ret_msg = await self._send_sync_start(sync_id, total_keys, batch_count)
|
|
239
|
+
if ret_msg is not None and ret_msg.accepted:
|
|
240
|
+
start_accepted = True
|
|
241
|
+
break
|
|
242
|
+
logger.warning(
|
|
243
|
+
"FullSyncStart not accepted, attempt %d/%d, error: %s",
|
|
244
|
+
attempt + 1,
|
|
245
|
+
self.max_retry_count,
|
|
246
|
+
ret_msg.error_msg if ret_msg else "No response",
|
|
247
|
+
)
|
|
248
|
+
await asyncio.sleep(self.retry_delay_s)
|
|
249
|
+
|
|
250
|
+
if not start_accepted:
|
|
251
|
+
logger.error(
|
|
252
|
+
"Failed to start full sync after %d attempts", self.max_retry_count
|
|
253
|
+
)
|
|
254
|
+
return None
|
|
255
|
+
|
|
256
|
+
return SyncInitResult(
|
|
257
|
+
sync_id=sync_id,
|
|
258
|
+
keys=keys,
|
|
259
|
+
total_keys=total_keys,
|
|
260
|
+
batch_count=batch_count,
|
|
261
|
+
)
|
|
262
|
+
|
|
263
|
+
async def _send_key_batches(
|
|
264
|
+
self, sync_id: str, keys: List[int], batch_count: int
|
|
265
|
+
) -> int:
|
|
266
|
+
"""
|
|
267
|
+
Send keys in batches to the controller.
|
|
268
|
+
|
|
269
|
+
Args:
|
|
270
|
+
sync_id: The sync session ID
|
|
271
|
+
keys: List of all keys to send
|
|
272
|
+
batch_count: Number of batches to send
|
|
273
|
+
|
|
274
|
+
Returns:
|
|
275
|
+
Total number of keys sent
|
|
276
|
+
"""
|
|
277
|
+
total_keys = len(keys)
|
|
278
|
+
|
|
279
|
+
for batch_id in range(batch_count):
|
|
280
|
+
start_idx = batch_id * self.batch_size
|
|
281
|
+
end_idx = min(start_idx + self.batch_size, total_keys)
|
|
282
|
+
batch_keys = keys[start_idx:end_idx]
|
|
283
|
+
|
|
284
|
+
self._send_sync_batch(sync_id, batch_id, batch_keys)
|
|
285
|
+
|
|
286
|
+
logger.debug(
|
|
287
|
+
"Sent batch %d/%d with %d keys",
|
|
288
|
+
batch_id + 1,
|
|
289
|
+
batch_count,
|
|
290
|
+
len(batch_keys),
|
|
291
|
+
)
|
|
292
|
+
|
|
293
|
+
# Small delay between batches to avoid overwhelming controller
|
|
294
|
+
if self.batch_interval_ms > 0 and batch_id < batch_count - 1:
|
|
295
|
+
await asyncio.sleep(self.batch_interval_ms / 1000.0)
|
|
296
|
+
|
|
297
|
+
# Send end message
|
|
298
|
+
self._send_sync_end(sync_id, total_keys)
|
|
299
|
+
logger.info("Full sync batches sent, total_keys=%d", total_keys)
|
|
300
|
+
|
|
301
|
+
return total_keys
|
|
302
|
+
|
|
303
|
+
async def _resend_missing_batches(
|
|
304
|
+
self, sync_id: str, keys: List[int], missing_batches: List[int]
|
|
305
|
+
) -> None:
|
|
306
|
+
"""
|
|
307
|
+
Resend missing batches to the controller.
|
|
308
|
+
|
|
309
|
+
Args:
|
|
310
|
+
sync_id: The sync session ID
|
|
311
|
+
keys: List of all keys
|
|
312
|
+
missing_batches: List of missing batch IDs to resend
|
|
313
|
+
"""
|
|
314
|
+
total_keys = len(keys)
|
|
315
|
+
|
|
316
|
+
logger.info(
|
|
317
|
+
"Resending %d missing batches: %s",
|
|
318
|
+
len(missing_batches),
|
|
319
|
+
missing_batches,
|
|
320
|
+
)
|
|
321
|
+
|
|
322
|
+
for batch_id in missing_batches:
|
|
323
|
+
start_idx = batch_id * self.batch_size
|
|
324
|
+
end_idx = min(start_idx + self.batch_size, total_keys)
|
|
325
|
+
batch_keys = keys[start_idx:end_idx]
|
|
326
|
+
|
|
327
|
+
self._send_sync_batch(sync_id, batch_id, batch_keys)
|
|
328
|
+
|
|
329
|
+
logger.debug(
|
|
330
|
+
"Resent batch %d with %d keys",
|
|
331
|
+
batch_id,
|
|
332
|
+
len(batch_keys),
|
|
333
|
+
)
|
|
334
|
+
|
|
335
|
+
# Small delay between batches
|
|
336
|
+
if self.batch_interval_ms > 0:
|
|
337
|
+
await asyncio.sleep(self.batch_interval_ms / 1000.0)
|
|
338
|
+
|
|
339
|
+
logger.info("Finished resending %d missing batches", len(missing_batches))
|
|
340
|
+
|
|
341
|
+
async def _poll_for_completion(
|
|
342
|
+
self, sync_id: str, keys: List[int], total_keys: int
|
|
343
|
+
) -> bool:
|
|
344
|
+
"""
|
|
345
|
+
Poll for sync completion status and resend missing batches if needed.
|
|
346
|
+
|
|
347
|
+
Args:
|
|
348
|
+
sync_id: The sync session ID
|
|
349
|
+
keys: List of all keys (needed for resending missing batches)
|
|
350
|
+
total_keys: Total number of keys
|
|
351
|
+
|
|
352
|
+
Returns:
|
|
353
|
+
True if sync completed and can exit freeze mode, False on timeout
|
|
354
|
+
"""
|
|
355
|
+
resend_count = 0
|
|
356
|
+
# TODO(baoloongmao): This can be an individual config
|
|
357
|
+
max_resend_attempts = self.max_retry_count
|
|
358
|
+
|
|
359
|
+
for poll_attempt in range(self.max_poll_attempts):
|
|
360
|
+
await asyncio.sleep(self.status_poll_interval_s)
|
|
361
|
+
|
|
362
|
+
status = await self._query_sync_status(sync_id)
|
|
363
|
+
if status is None:
|
|
364
|
+
logger.warning(
|
|
365
|
+
"Failed to query sync status, attempt %d", poll_attempt + 1
|
|
366
|
+
)
|
|
367
|
+
continue
|
|
368
|
+
|
|
369
|
+
logger.info(
|
|
370
|
+
"Sync status: is_complete=%s, global_progress=%.1f%%, "
|
|
371
|
+
"can_exit_freeze=%s, missing_batches=%s",
|
|
372
|
+
status.is_complete,
|
|
373
|
+
status.global_progress * 100,
|
|
374
|
+
status.can_exit_freeze,
|
|
375
|
+
status.missing_batches if status.missing_batches else "none",
|
|
376
|
+
)
|
|
377
|
+
|
|
378
|
+
if status.can_exit_freeze:
|
|
379
|
+
return True
|
|
380
|
+
|
|
381
|
+
# Handle missing batches - resend them
|
|
382
|
+
if status.missing_batches and resend_count < max_resend_attempts:
|
|
383
|
+
resend_count += 1
|
|
384
|
+
logger.warning(
|
|
385
|
+
"Controller reported missing batches, resending "
|
|
386
|
+
"(attempt %d/%d): %s",
|
|
387
|
+
resend_count,
|
|
388
|
+
max_resend_attempts,
|
|
389
|
+
status.missing_batches,
|
|
390
|
+
)
|
|
391
|
+
await self._resend_missing_batches(
|
|
392
|
+
sync_id, keys, status.missing_batches
|
|
393
|
+
)
|
|
394
|
+
# Resend end message after resending missing batches
|
|
395
|
+
self._send_sync_end(sync_id, total_keys)
|
|
396
|
+
elif status.missing_batches and resend_count >= max_resend_attempts:
|
|
397
|
+
logger.error(
|
|
398
|
+
"Max resend attempts reached (%d), "
|
|
399
|
+
"giving up on missing batches: %s",
|
|
400
|
+
max_resend_attempts,
|
|
401
|
+
status.missing_batches,
|
|
402
|
+
)
|
|
403
|
+
|
|
404
|
+
# TODO(baoloongmao): Use heartbeat to detect controller failure
|
|
405
|
+
# and exit freeze mode if necessary
|
|
406
|
+
logger.warning("Full sync status poll timeout, exiting freeze mode anyway")
|
|
407
|
+
return False
|
|
408
|
+
|
|
409
|
+
async def start_full_sync(self, reason: Optional[str] = None) -> bool:
|
|
410
|
+
"""
|
|
411
|
+
Start the full sync process.
|
|
412
|
+
|
|
413
|
+
This method orchestrates the full sync by delegating to helper methods:
|
|
414
|
+
1. _initialize_sync: Startup delay, freeze mode, get keys, send start msg
|
|
415
|
+
2. _send_key_batches: Send all keys in batches
|
|
416
|
+
3. _poll_for_completion: Poll for sync completion status
|
|
417
|
+
|
|
418
|
+
Args:
|
|
419
|
+
reason: The reason for full sync (e.g., "controller_restart")
|
|
420
|
+
|
|
421
|
+
Returns:
|
|
422
|
+
True if sync completed successfully, False otherwise
|
|
423
|
+
"""
|
|
424
|
+
if self._is_syncing:
|
|
425
|
+
logger.warning("Full sync already in progress, skipping")
|
|
426
|
+
return False
|
|
427
|
+
|
|
428
|
+
self._is_syncing = True
|
|
429
|
+
self._current_sync_id = None
|
|
430
|
+
|
|
431
|
+
logger.info(
|
|
432
|
+
"Starting full sync for worker %s:%s, reason: %s",
|
|
433
|
+
self.instance_id,
|
|
434
|
+
self.worker_id,
|
|
435
|
+
reason,
|
|
436
|
+
)
|
|
437
|
+
|
|
438
|
+
try:
|
|
439
|
+
# Step 1: Initialize sync (delay, freeze, get keys, send start)
|
|
440
|
+
init_result = await self._initialize_sync(reason)
|
|
441
|
+
if init_result is None:
|
|
442
|
+
return False
|
|
443
|
+
|
|
444
|
+
# Step 2: Send keys in batches
|
|
445
|
+
await self._send_key_batches(
|
|
446
|
+
init_result.sync_id,
|
|
447
|
+
init_result.keys,
|
|
448
|
+
init_result.batch_count,
|
|
449
|
+
)
|
|
450
|
+
|
|
451
|
+
# Step 3: Poll for completion status (with resend support)
|
|
452
|
+
await self._poll_for_completion(
|
|
453
|
+
init_result.sync_id,
|
|
454
|
+
init_result.keys,
|
|
455
|
+
init_result.total_keys,
|
|
456
|
+
)
|
|
457
|
+
|
|
458
|
+
logger.info("Full sync completed successfully")
|
|
459
|
+
return True
|
|
460
|
+
|
|
461
|
+
except Exception as e:
|
|
462
|
+
logger.error("Error during full sync: %s", e)
|
|
463
|
+
return False
|
|
464
|
+
|
|
465
|
+
finally:
|
|
466
|
+
# Always clean up state, regardless of success or failure
|
|
467
|
+
logger.info("Exiting freeze mode after full sync.")
|
|
468
|
+
self.lmcache_engine.freeze(False)
|
|
469
|
+
self._is_syncing = False
|
|
470
|
+
self._current_sync_id = None
|
|
471
|
+
|
|
472
|
+
@property
|
|
473
|
+
def is_syncing(self) -> bool:
|
|
474
|
+
"""Check if full sync is currently in progress"""
|
|
475
|
+
return self._is_syncing
|
|
@@ -0,0 +1,149 @@
|
|
|
1
|
+
# SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
"""
|
|
3
|
+
Lock utilities for thread-safe operations.
|
|
4
|
+
|
|
5
|
+
This module provides thread synchronization primitives with timeout support.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
# Standard
|
|
9
|
+
from contextlib import contextmanager
|
|
10
|
+
from typing import Optional
|
|
11
|
+
import threading
|
|
12
|
+
import time
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
class RWLockTimeoutError(Exception):
|
|
16
|
+
"""Exception raised when a lock acquisition times out."""
|
|
17
|
+
|
|
18
|
+
pass
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
class RWLockWithTimeout:
|
|
22
|
+
"""
|
|
23
|
+
A simple read-write lock with timeout support.
|
|
24
|
+
Multiple readers can hold the lock simultaneously, but only one writer.
|
|
25
|
+
|
|
26
|
+
Note: This lock is NOT reentrant.
|
|
27
|
+
"""
|
|
28
|
+
|
|
29
|
+
def __init__(self):
|
|
30
|
+
self._readers = 0
|
|
31
|
+
self._writers_waiting = 0
|
|
32
|
+
self._writer_active = False
|
|
33
|
+
self._condition = threading.Condition(threading.Lock())
|
|
34
|
+
|
|
35
|
+
def acquire_read(self, timeout: Optional[float] = None) -> bool:
|
|
36
|
+
"""Acquire a read lock with optional timeout."""
|
|
37
|
+
deadline = time.monotonic() + timeout if timeout is not None else None
|
|
38
|
+
|
|
39
|
+
with self._condition:
|
|
40
|
+
# Note: Sustained write operations may starve reads if writers
|
|
41
|
+
# continuously arrive while readers are waiting
|
|
42
|
+
while self._writer_active or self._writers_waiting > 0:
|
|
43
|
+
if deadline is not None and time.monotonic() >= deadline:
|
|
44
|
+
return False
|
|
45
|
+
remaining = deadline - time.monotonic() if deadline else None
|
|
46
|
+
if remaining is not None and remaining <= 0:
|
|
47
|
+
return False
|
|
48
|
+
self._condition.wait(timeout=remaining)
|
|
49
|
+
self._readers += 1
|
|
50
|
+
return True
|
|
51
|
+
|
|
52
|
+
def release_read(self):
|
|
53
|
+
"""Release a read lock."""
|
|
54
|
+
with self._condition:
|
|
55
|
+
self._readers -= 1
|
|
56
|
+
if self._readers == 0:
|
|
57
|
+
self._condition.notify_all()
|
|
58
|
+
|
|
59
|
+
def acquire_write(self, timeout: Optional[float] = None) -> bool:
|
|
60
|
+
"""Acquire a write lock with optional timeout."""
|
|
61
|
+
deadline = time.monotonic() + timeout if timeout is not None else None
|
|
62
|
+
|
|
63
|
+
with self._condition:
|
|
64
|
+
self._writers_waiting += 1
|
|
65
|
+
try:
|
|
66
|
+
while self._readers > 0 or self._writer_active:
|
|
67
|
+
if deadline is not None and time.monotonic() >= deadline:
|
|
68
|
+
return False
|
|
69
|
+
remaining = deadline - time.monotonic() if deadline else None
|
|
70
|
+
if remaining is not None and remaining <= 0:
|
|
71
|
+
return False
|
|
72
|
+
self._condition.wait(timeout=remaining)
|
|
73
|
+
self._writer_active = True
|
|
74
|
+
return True
|
|
75
|
+
finally:
|
|
76
|
+
self._writers_waiting -= 1
|
|
77
|
+
|
|
78
|
+
def release_write(self):
|
|
79
|
+
"""Release a write lock."""
|
|
80
|
+
with self._condition:
|
|
81
|
+
self._writer_active = False
|
|
82
|
+
self._condition.notify_all()
|
|
83
|
+
|
|
84
|
+
@contextmanager
|
|
85
|
+
def read_lock(self, timeout: Optional[float] = None):
|
|
86
|
+
"""Context manager for read lock with timeout.
|
|
87
|
+
|
|
88
|
+
Args:
|
|
89
|
+
timeout: Timeout in seconds. None means wait forever.
|
|
90
|
+
"""
|
|
91
|
+
if not self.acquire_read(timeout):
|
|
92
|
+
raise RWLockTimeoutError("Failed to acquire read lock within timeout")
|
|
93
|
+
try:
|
|
94
|
+
yield
|
|
95
|
+
finally:
|
|
96
|
+
self.release_read()
|
|
97
|
+
|
|
98
|
+
@contextmanager
|
|
99
|
+
def write_lock(self, timeout: Optional[float] = None):
|
|
100
|
+
"""Context manager for write lock with timeout.
|
|
101
|
+
|
|
102
|
+
Args:
|
|
103
|
+
timeout: Timeout in seconds. None means wait forever.
|
|
104
|
+
"""
|
|
105
|
+
if not self.acquire_write(timeout):
|
|
106
|
+
raise RWLockTimeoutError("Failed to acquire write lock within timeout")
|
|
107
|
+
try:
|
|
108
|
+
yield
|
|
109
|
+
finally:
|
|
110
|
+
self.release_write()
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
class FastLockWithTimeout:
|
|
114
|
+
"""
|
|
115
|
+
A fast lock with timeout support for WorkerNode.
|
|
116
|
+
Optimized for high frequency operations on small critical sections.
|
|
117
|
+
Uses non-blocking fast path for better performance.
|
|
118
|
+
|
|
119
|
+
Note: This lock is NOT reentrant.
|
|
120
|
+
"""
|
|
121
|
+
|
|
122
|
+
__slots__ = ("_lock",)
|
|
123
|
+
|
|
124
|
+
def __init__(self):
|
|
125
|
+
self._lock = threading.Lock()
|
|
126
|
+
|
|
127
|
+
def acquire(self, timeout: Optional[float] = None) -> bool:
|
|
128
|
+
"""Acquire the lock with optional timeout."""
|
|
129
|
+
if timeout is None:
|
|
130
|
+
return self._lock.acquire()
|
|
131
|
+
return self._lock.acquire(timeout=timeout)
|
|
132
|
+
|
|
133
|
+
def release(self):
|
|
134
|
+
"""Release the lock."""
|
|
135
|
+
self._lock.release()
|
|
136
|
+
|
|
137
|
+
def __enter__(self):
|
|
138
|
+
# Fast path: try non-blocking acquire first (no context switch overhead)
|
|
139
|
+
if self._lock.acquire(blocking=False):
|
|
140
|
+
return self
|
|
141
|
+
# Slow path: reduced timeout for faster failure detection
|
|
142
|
+
if not self._lock.acquire(timeout=10): # 10s timeout
|
|
143
|
+
# TODO(baoloongmao): Mark as operation failed for metrics
|
|
144
|
+
# and schedule full sync
|
|
145
|
+
raise RWLockTimeoutError("Failed to acquire WorkerNode lock within 10s")
|
|
146
|
+
return self
|
|
147
|
+
|
|
148
|
+
def __exit__(self, exc_type, exc_val, exc_tb):
|
|
149
|
+
self._lock.release()
|