lmcache-cli 0.4.5.dev0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- lmcache/__init__.py +84 -0
- lmcache/_version.py +24 -0
- lmcache/cli/__init__.py +1 -0
- lmcache/cli/commands/__init__.py +34 -0
- lmcache/cli/commands/base.py +157 -0
- lmcache/cli/commands/bench/__init__.py +557 -0
- lmcache/cli/commands/bench/engine_bench/__init__.py +1 -0
- lmcache/cli/commands/bench/engine_bench/config.py +245 -0
- lmcache/cli/commands/bench/engine_bench/interactive/__init__.py +274 -0
- lmcache/cli/commands/bench/engine_bench/interactive/config.json +10 -0
- lmcache/cli/commands/bench/engine_bench/interactive/schema.py +352 -0
- lmcache/cli/commands/bench/engine_bench/interactive/state.py +327 -0
- lmcache/cli/commands/bench/engine_bench/interactive/terminal.py +291 -0
- lmcache/cli/commands/bench/engine_bench/progress.py +145 -0
- lmcache/cli/commands/bench/engine_bench/request_sender.py +232 -0
- lmcache/cli/commands/bench/engine_bench/stats.py +275 -0
- lmcache/cli/commands/bench/engine_bench/workloads/__init__.py +153 -0
- lmcache/cli/commands/bench/engine_bench/workloads/base.py +122 -0
- lmcache/cli/commands/bench/engine_bench/workloads/long_doc_permutator.py +435 -0
- lmcache/cli/commands/bench/engine_bench/workloads/long_doc_qa.py +281 -0
- lmcache/cli/commands/bench/engine_bench/workloads/multi_round_chat.py +337 -0
- lmcache/cli/commands/bench/engine_bench/workloads/random_prefill.py +178 -0
- lmcache/cli/commands/describe.py +310 -0
- lmcache/cli/commands/kvcache.py +133 -0
- lmcache/cli/commands/mock.py +75 -0
- lmcache/cli/commands/ping.py +113 -0
- lmcache/cli/commands/query/__init__.py +155 -0
- lmcache/cli/commands/query/prompt.py +134 -0
- lmcache/cli/commands/query/request.py +357 -0
- lmcache/cli/commands/server.py +99 -0
- lmcache/cli/commands/tool/__init__.py +63 -0
- lmcache/cli/commands/tool/cache_simulator.py +113 -0
- lmcache/cli/commands/trace/__init__.py +505 -0
- lmcache/cli/commands/trace/dispatch.py +249 -0
- lmcache/cli/commands/trace/driver.py +372 -0
- lmcache/cli/commands/trace/stats.py +289 -0
- lmcache/cli/documents/lmcache.txt +11 -0
- lmcache/cli/main.py +42 -0
- lmcache/cli/metrics/__init__.py +29 -0
- lmcache/cli/metrics/formatter.py +171 -0
- lmcache/cli/metrics/handler.py +94 -0
- lmcache/cli/metrics/metrics.py +161 -0
- lmcache/cli/metrics/section.py +77 -0
- lmcache/connections.py +173 -0
- lmcache/integration/__init__.py +2 -0
- lmcache/integration/base_service_factory.py +165 -0
- lmcache/integration/request_telemetry/__init__.py +1 -0
- lmcache/integration/request_telemetry/base.py +51 -0
- lmcache/integration/request_telemetry/factory.py +113 -0
- lmcache/integration/request_telemetry/fastapi.py +109 -0
- lmcache/integration/request_telemetry/noop.py +35 -0
- lmcache/integration/sglang/__init__.py +2 -0
- lmcache/integration/sglang/sglang_adapter.py +326 -0
- lmcache/integration/sglang/utils.py +39 -0
- lmcache/integration/vllm/__init__.py +1 -0
- lmcache/integration/vllm/lmcache_connector_v1.py +213 -0
- lmcache/integration/vllm/lmcache_connector_v1_085.py +150 -0
- lmcache/integration/vllm/lmcache_mp_connector_0180.py +1072 -0
- lmcache/integration/vllm/tests/test_mm_hash_utils.py +112 -0
- lmcache/integration/vllm/utils.py +433 -0
- lmcache/integration/vllm/vllm_multi_process_adapter.py +1090 -0
- lmcache/integration/vllm/vllm_service_factory.py +339 -0
- lmcache/integration/vllm/vllm_v1_adapter.py +1713 -0
- lmcache/logging.py +107 -0
- lmcache/native_storage_ops.pyi +230 -0
- lmcache/non_cuda_equivalents.py +1424 -0
- lmcache/observability.py +1958 -0
- lmcache/storage_backend/serde/__init__.py +1 -0
- lmcache/storage_backend/serde/cachegen_basics.py +210 -0
- lmcache/storage_backend/serde/cachegen_decoder.py +207 -0
- lmcache/storage_backend/serde/cachegen_encoder.py +394 -0
- lmcache/storage_backend/serde/serde.py +75 -0
- lmcache/tools/__init__.py +1 -0
- lmcache/tools/cache_simulator/README.md +392 -0
- lmcache/tools/cache_simulator/__init__.py +1 -0
- lmcache/tools/cache_simulator/docs/simulate_example.png +0 -0
- lmcache/tools/cache_simulator/docs/sweep_example.png +0 -0
- lmcache/tools/cache_simulator/gen_bench_dataset.py +360 -0
- lmcache/tools/cache_simulator/lru_cache.py +124 -0
- lmcache/tools/cache_simulator/plot_hit_rate.py +231 -0
- lmcache/tools/cache_simulator/simulator.py +795 -0
- lmcache/tools/controller_benchmark/README.md +161 -0
- lmcache/tools/controller_benchmark/__init__.py +1 -0
- lmcache/tools/controller_benchmark/__main__.py +331 -0
- lmcache/tools/controller_benchmark/benchmark.py +660 -0
- lmcache/tools/controller_benchmark/config.py +44 -0
- lmcache/tools/controller_benchmark/constants.py +10 -0
- lmcache/tools/controller_benchmark/handlers/__init__.py +46 -0
- lmcache/tools/controller_benchmark/handlers/admit.py +52 -0
- lmcache/tools/controller_benchmark/handlers/base.py +47 -0
- lmcache/tools/controller_benchmark/handlers/deregister.py +49 -0
- lmcache/tools/controller_benchmark/handlers/evict.py +52 -0
- lmcache/tools/controller_benchmark/handlers/heartbeat.py +56 -0
- lmcache/tools/controller_benchmark/handlers/p2p_lookup.py +47 -0
- lmcache/tools/controller_benchmark/handlers/register.py +56 -0
- lmcache/tools/mp_status_viewer/__init__.py +1 -0
- lmcache/tools/mp_status_viewer/__main__.py +95 -0
- lmcache/usage_context.py +417 -0
- lmcache/utils.py +665 -0
- lmcache/v1/__init__.py +2 -0
- lmcache/v1/api_server/__init__.py +2 -0
- lmcache/v1/api_server/__main__.py +537 -0
- lmcache/v1/basic_check.py +112 -0
- lmcache/v1/cache_controller/__init__.py +9 -0
- lmcache/v1/cache_controller/commands/__init__.py +15 -0
- lmcache/v1/cache_controller/commands/base.py +35 -0
- lmcache/v1/cache_controller/commands/full_sync.py +49 -0
- lmcache/v1/cache_controller/config.py +176 -0
- lmcache/v1/cache_controller/controller_manager.py +535 -0
- lmcache/v1/cache_controller/controllers/__init__.py +11 -0
- lmcache/v1/cache_controller/controllers/full_sync_tracker.py +473 -0
- lmcache/v1/cache_controller/controllers/kv_controller.py +439 -0
- lmcache/v1/cache_controller/controllers/registration_controller.py +282 -0
- lmcache/v1/cache_controller/executor.py +463 -0
- lmcache/v1/cache_controller/frontend/static/css/style.css +201 -0
- lmcache/v1/cache_controller/frontend/static/img/logo.png +0 -0
- lmcache/v1/cache_controller/frontend/static/index.html +234 -0
- lmcache/v1/cache_controller/frontend/static/js/controller_app.js +660 -0
- lmcache/v1/cache_controller/full_sync_sender.py +475 -0
- lmcache/v1/cache_controller/locks.py +149 -0
- lmcache/v1/cache_controller/message.py +828 -0
- lmcache/v1/cache_controller/observability.py +208 -0
- lmcache/v1/cache_controller/utils.py +679 -0
- lmcache/v1/cache_controller/worker.py +665 -0
- lmcache/v1/cache_engine.py +2058 -0
- lmcache/v1/cache_interface.py +19 -0
- lmcache/v1/check/__init__.py +74 -0
- lmcache/v1/check/check_mode_gen.py +86 -0
- lmcache/v1/check/check_mode_test_l2_adapter.py +284 -0
- lmcache/v1/check/check_mode_test_remote.py +155 -0
- lmcache/v1/check/check_mode_test_storage_manager.py +142 -0
- lmcache/v1/check/utils.py +571 -0
- lmcache/v1/compute/__init__.py +2 -0
- lmcache/v1/compute/attention/__init__.py +0 -0
- lmcache/v1/compute/attention/abstract.py +39 -0
- lmcache/v1/compute/attention/flash_attn.py +129 -0
- lmcache/v1/compute/attention/flash_infer_sparse.py +284 -0
- lmcache/v1/compute/attention/metadata.py +85 -0
- lmcache/v1/compute/attention/utils.py +14 -0
- lmcache/v1/compute/blend/__init__.py +7 -0
- lmcache/v1/compute/blend/blender.py +168 -0
- lmcache/v1/compute/blend/metadata.py +34 -0
- lmcache/v1/compute/blend/utils.py +63 -0
- lmcache/v1/compute/models/__init__.py +0 -0
- lmcache/v1/compute/models/base.py +141 -0
- lmcache/v1/compute/models/llama.py +9 -0
- lmcache/v1/compute/models/qwen3.py +24 -0
- lmcache/v1/compute/models/utils.py +68 -0
- lmcache/v1/compute/positional_encoding.py +199 -0
- lmcache/v1/config.py +848 -0
- lmcache/v1/config_base.py +848 -0
- lmcache/v1/distributed/api.py +248 -0
- lmcache/v1/distributed/config.py +321 -0
- lmcache/v1/distributed/error.py +64 -0
- lmcache/v1/distributed/eviction.py +192 -0
- lmcache/v1/distributed/eviction_policy/__init__.py +21 -0
- lmcache/v1/distributed/eviction_policy/factory.py +27 -0
- lmcache/v1/distributed/eviction_policy/lru.py +244 -0
- lmcache/v1/distributed/eviction_policy/noop.py +50 -0
- lmcache/v1/distributed/internal_api.py +170 -0
- lmcache/v1/distributed/l1_manager.py +835 -0
- lmcache/v1/distributed/l2_adapters/__init__.py +67 -0
- lmcache/v1/distributed/l2_adapters/base.py +360 -0
- lmcache/v1/distributed/l2_adapters/config.py +385 -0
- lmcache/v1/distributed/l2_adapters/factory.py +205 -0
- lmcache/v1/distributed/l2_adapters/fs_l2_adapter.py +747 -0
- lmcache/v1/distributed/l2_adapters/fs_native_l2_adapter.py +167 -0
- lmcache/v1/distributed/l2_adapters/mock_l2_adapter.py +516 -0
- lmcache/v1/distributed/l2_adapters/mooncake_store_l2_adapter.py +135 -0
- lmcache/v1/distributed/l2_adapters/native_connector_l2_adapter.py +468 -0
- lmcache/v1/distributed/l2_adapters/native_plugin_l2_adapter.py +199 -0
- lmcache/v1/distributed/l2_adapters/nixl_store_dynamic_l2_adapter.py +831 -0
- lmcache/v1/distributed/l2_adapters/nixl_store_l2_adapter.py +983 -0
- lmcache/v1/distributed/l2_adapters/plugin_l2_adapter.py +210 -0
- lmcache/v1/distributed/l2_adapters/resp_l2_adapter.py +176 -0
- lmcache/v1/distributed/memory_manager.py +179 -0
- lmcache/v1/distributed/storage_controller.py +39 -0
- lmcache/v1/distributed/storage_controllers/__init__.py +43 -0
- lmcache/v1/distributed/storage_controllers/eviction_controller.py +242 -0
- lmcache/v1/distributed/storage_controllers/prefetch_controller.py +830 -0
- lmcache/v1/distributed/storage_controllers/prefetch_policy.py +193 -0
- lmcache/v1/distributed/storage_controllers/store_controller.py +452 -0
- lmcache/v1/distributed/storage_controllers/store_policy.py +213 -0
- lmcache/v1/distributed/storage_manager.py +532 -0
- lmcache/v1/event_manager.py +145 -0
- lmcache/v1/exceptions/__init__.py +16 -0
- lmcache/v1/gpu_connector/__init__.py +126 -0
- lmcache/v1/gpu_connector/gpu_connectors.py +1906 -0
- lmcache/v1/gpu_connector/gpu_ops.py +85 -0
- lmcache/v1/gpu_connector/hpu_connector.py +326 -0
- lmcache/v1/gpu_connector/mock_gpu_connector.py +67 -0
- lmcache/v1/gpu_connector/utils.py +890 -0
- lmcache/v1/gpu_connector/xpu_connectors.py +916 -0
- lmcache/v1/health_monitor/__init__.py +1 -0
- lmcache/v1/health_monitor/base.py +587 -0
- lmcache/v1/health_monitor/checks/__init__.py +1 -0
- lmcache/v1/health_monitor/checks/remote_backend_check.py +304 -0
- lmcache/v1/health_monitor/constants.py +36 -0
- lmcache/v1/internal_api_server/__init__.py +0 -0
- lmcache/v1/internal_api_server/api_registry.py +59 -0
- lmcache/v1/internal_api_server/api_server.py +120 -0
- lmcache/v1/internal_api_server/common/__init__.py +1 -0
- lmcache/v1/internal_api_server/common/env_api.py +22 -0
- lmcache/v1/internal_api_server/common/loglevel_api.py +57 -0
- lmcache/v1/internal_api_server/common/metrics_api.py +29 -0
- lmcache/v1/internal_api_server/common/periodic_thread_api.py +138 -0
- lmcache/v1/internal_api_server/common/run_script_api.py +73 -0
- lmcache/v1/internal_api_server/common/thread_api.py +63 -0
- lmcache/v1/internal_api_server/controller/__init__.py +1 -0
- lmcache/v1/internal_api_server/controller/key_stats_api.py +81 -0
- lmcache/v1/internal_api_server/controller/worker_info_api.py +136 -0
- lmcache/v1/internal_api_server/utils.py +43 -0
- lmcache/v1/internal_api_server/vllm/__init__.py +1 -0
- lmcache/v1/internal_api_server/vllm/backend_api.py +221 -0
- lmcache/v1/internal_api_server/vllm/bypass_api.py +204 -0
- lmcache/v1/internal_api_server/vllm/cache_api.py +895 -0
- lmcache/v1/internal_api_server/vllm/chunk_statistics_api.py +141 -0
- lmcache/v1/internal_api_server/vllm/conf_api.py +147 -0
- lmcache/v1/internal_api_server/vllm/freeze_api.py +172 -0
- lmcache/v1/internal_api_server/vllm/hot_cache_api.py +184 -0
- lmcache/v1/internal_api_server/vllm/inference_api.py +65 -0
- lmcache/v1/internal_api_server/vllm/load_fs_chunks_api.py +320 -0
- lmcache/v1/internal_api_server/vllm/lookup_api.py +145 -0
- lmcache/v1/internal_api_server/vllm/version_api.py +25 -0
- lmcache/v1/kv_layer_groups.py +267 -0
- lmcache/v1/lazy_memory_allocator.py +284 -0
- lmcache/v1/lookup_client/__init__.py +25 -0
- lmcache/v1/lookup_client/abstract_client.py +77 -0
- lmcache/v1/lookup_client/async_lookup_message.py +50 -0
- lmcache/v1/lookup_client/chunk_statistics_lookup_client.py +200 -0
- lmcache/v1/lookup_client/factory.py +251 -0
- lmcache/v1/lookup_client/hit_limit_lookup_client.py +86 -0
- lmcache/v1/lookup_client/lmcache_async_lookup_client.py +407 -0
- lmcache/v1/lookup_client/lmcache_lookup_client.py +285 -0
- lmcache/v1/lookup_client/lmcache_lookup_client_bypass.py +99 -0
- lmcache/v1/lookup_client/mooncake_lookup_client.py +87 -0
- lmcache/v1/lookup_client/record_strategies/__init__.py +77 -0
- lmcache/v1/lookup_client/record_strategies/base.py +327 -0
- lmcache/v1/lookup_client/record_strategies/file_hash.py +130 -0
- lmcache/v1/lookup_client/record_strategies/memory_bloom_filter.py +81 -0
- lmcache/v1/manager.py +539 -0
- lmcache/v1/memory_management.py +2619 -0
- lmcache/v1/metadata.py +114 -0
- lmcache/v1/mp_observability/AGENTS.override.md +21 -0
- lmcache/v1/mp_observability/README.md +204 -0
- lmcache/v1/mp_observability/config.py +340 -0
- lmcache/v1/mp_observability/event.py +100 -0
- lmcache/v1/mp_observability/event_bus.py +313 -0
- lmcache/v1/mp_observability/otel_init.py +145 -0
- lmcache/v1/mp_observability/subscribers/__init__.py +28 -0
- lmcache/v1/mp_observability/subscribers/logging/__init__.py +19 -0
- lmcache/v1/mp_observability/subscribers/logging/l1.py +56 -0
- lmcache/v1/mp_observability/subscribers/logging/l2.py +73 -0
- lmcache/v1/mp_observability/subscribers/logging/lookup_hash.py +209 -0
- lmcache/v1/mp_observability/subscribers/logging/mp_server.py +90 -0
- lmcache/v1/mp_observability/subscribers/logging/sm.py +59 -0
- lmcache/v1/mp_observability/subscribers/metrics/__init__.py +20 -0
- lmcache/v1/mp_observability/subscribers/metrics/l0_lifecycle.py +290 -0
- lmcache/v1/mp_observability/subscribers/metrics/l1.py +55 -0
- lmcache/v1/mp_observability/subscribers/metrics/l1_lifecycle.py +166 -0
- lmcache/v1/mp_observability/subscribers/metrics/l2.py +121 -0
- lmcache/v1/mp_observability/subscribers/metrics/sm.py +69 -0
- lmcache/v1/mp_observability/subscribers/tracing/__init__.py +12 -0
- lmcache/v1/mp_observability/subscribers/tracing/mp_server.py +333 -0
- lmcache/v1/mp_observability/subscribers/tracing/span_registry.py +148 -0
- lmcache/v1/mp_observability/trace/__init__.py +50 -0
- lmcache/v1/mp_observability/trace/codecs.py +255 -0
- lmcache/v1/mp_observability/trace/decorator.py +147 -0
- lmcache/v1/mp_observability/trace/format.py +132 -0
- lmcache/v1/mp_observability/trace/lifecycle.py +83 -0
- lmcache/v1/mp_observability/trace/reader.py +167 -0
- lmcache/v1/mp_observability/trace/recorder.py +300 -0
- lmcache/v1/multiprocess/__init__.py +0 -0
- lmcache/v1/multiprocess/affinity_pool.py +102 -0
- lmcache/v1/multiprocess/blend_server_v2.py +891 -0
- lmcache/v1/multiprocess/config.py +253 -0
- lmcache/v1/multiprocess/custom_types.py +281 -0
- lmcache/v1/multiprocess/futures.py +194 -0
- lmcache/v1/multiprocess/gpu_context.py +511 -0
- lmcache/v1/multiprocess/http_server.py +235 -0
- lmcache/v1/multiprocess/mp_runtime_plugin_launcher.py +130 -0
- lmcache/v1/multiprocess/mq.py +732 -0
- lmcache/v1/multiprocess/protocol.py +86 -0
- lmcache/v1/multiprocess/protocols/README.md +213 -0
- lmcache/v1/multiprocess/protocols/__init__.py +127 -0
- lmcache/v1/multiprocess/protocols/base.py +89 -0
- lmcache/v1/multiprocess/protocols/blend.py +109 -0
- lmcache/v1/multiprocess/protocols/blend_v2.py +57 -0
- lmcache/v1/multiprocess/protocols/controller.py +53 -0
- lmcache/v1/multiprocess/protocols/debug.py +34 -0
- lmcache/v1/multiprocess/protocols/engine.py +146 -0
- lmcache/v1/multiprocess/protocols/observability.py +39 -0
- lmcache/v1/multiprocess/server.py +1134 -0
- lmcache/v1/multiprocess/session.py +190 -0
- lmcache/v1/multiprocess/token_hasher.py +441 -0
- lmcache/v1/offload_server/__init__.py +17 -0
- lmcache/v1/offload_server/abstract_server.py +37 -0
- lmcache/v1/offload_server/message.py +30 -0
- lmcache/v1/offload_server/zmq_server.py +122 -0
- lmcache/v1/periodic_thread.py +579 -0
- lmcache/v1/pin_monitor.py +246 -0
- lmcache/v1/plugin/__init__.py +0 -0
- lmcache/v1/plugin/runtime_plugin_launcher.py +211 -0
- lmcache/v1/protocol.py +317 -0
- lmcache/v1/rpc/__init__.py +17 -0
- lmcache/v1/rpc/transport.py +105 -0
- lmcache/v1/rpc/zmq_transport.py +213 -0
- lmcache/v1/rpc_utils.py +165 -0
- lmcache/v1/server/__init__.py +2 -0
- lmcache/v1/server/__main__.py +170 -0
- lmcache/v1/server/storage_backend/__init__.py +21 -0
- lmcache/v1/server/storage_backend/abstract_backend.py +80 -0
- lmcache/v1/server/storage_backend/local_backend.py +75 -0
- lmcache/v1/server/utils.py +21 -0
- lmcache/v1/standalone/__init__.py +1 -0
- lmcache/v1/standalone/__main__.py +583 -0
- lmcache/v1/standalone/manager.py +80 -0
- lmcache/v1/standalone/standalone_service_factory.py +86 -0
- lmcache/v1/storage_backend/__init__.py +313 -0
- lmcache/v1/storage_backend/abstract_backend.py +445 -0
- lmcache/v1/storage_backend/audit_backend.py +233 -0
- lmcache/v1/storage_backend/batched_message_sender.py +222 -0
- lmcache/v1/storage_backend/cache_policy/__init__.py +45 -0
- lmcache/v1/storage_backend/cache_policy/base_policy.py +87 -0
- lmcache/v1/storage_backend/cache_policy/fifo.py +58 -0
- lmcache/v1/storage_backend/cache_policy/lfu.py +105 -0
- lmcache/v1/storage_backend/cache_policy/lru.py +81 -0
- lmcache/v1/storage_backend/cache_policy/mru.py +61 -0
- lmcache/v1/storage_backend/connector/__init__.py +443 -0
- lmcache/v1/storage_backend/connector/audit_adapter.py +77 -0
- lmcache/v1/storage_backend/connector/audit_connector.py +320 -0
- lmcache/v1/storage_backend/connector/base_connector.py +379 -0
- lmcache/v1/storage_backend/connector/blackhole_adapter.py +21 -0
- lmcache/v1/storage_backend/connector/blackhole_connector.py +37 -0
- lmcache/v1/storage_backend/connector/eic_adapter.py +31 -0
- lmcache/v1/storage_backend/connector/eic_connector.py +757 -0
- lmcache/v1/storage_backend/connector/external_adapter.py +79 -0
- lmcache/v1/storage_backend/connector/fs_adapter.py +51 -0
- lmcache/v1/storage_backend/connector/fs_connector.py +403 -0
- lmcache/v1/storage_backend/connector/infinistore_adapter.py +56 -0
- lmcache/v1/storage_backend/connector/infinistore_connector.py +177 -0
- lmcache/v1/storage_backend/connector/instrumented_connector.py +219 -0
- lmcache/v1/storage_backend/connector/lm_adapter.py +31 -0
- lmcache/v1/storage_backend/connector/lm_connector.py +176 -0
- lmcache/v1/storage_backend/connector/mock_adapter.py +57 -0
- lmcache/v1/storage_backend/connector/mock_connector.py +349 -0
- lmcache/v1/storage_backend/connector/mooncakestore_adapter.py +43 -0
- lmcache/v1/storage_backend/connector/mooncakestore_connector.py +614 -0
- lmcache/v1/storage_backend/connector/redis_adapter.py +181 -0
- lmcache/v1/storage_backend/connector/redis_connector.py +828 -0
- lmcache/v1/storage_backend/connector/s3_adapter.py +59 -0
- lmcache/v1/storage_backend/connector/s3_connector.py +699 -0
- lmcache/v1/storage_backend/connector/sagemaker_hyperpod_adapter.py +233 -0
- lmcache/v1/storage_backend/connector/sagemaker_hyperpod_connector.py +987 -0
- lmcache/v1/storage_backend/connector/valkey_adapter.py +114 -0
- lmcache/v1/storage_backend/connector/valkey_connector.py +627 -0
- lmcache/v1/storage_backend/gds_backend.py +1199 -0
- lmcache/v1/storage_backend/job_executor/__init__.py +0 -0
- lmcache/v1/storage_backend/job_executor/base_executor.py +34 -0
- lmcache/v1/storage_backend/job_executor/pq_executor.py +235 -0
- lmcache/v1/storage_backend/local_cpu_backend.py +810 -0
- lmcache/v1/storage_backend/local_disk_backend.py +656 -0
- lmcache/v1/storage_backend/maru_backend.py +734 -0
- lmcache/v1/storage_backend/naive_serde/__init__.py +50 -0
- lmcache/v1/storage_backend/naive_serde/cachegen_basics.py +133 -0
- lmcache/v1/storage_backend/naive_serde/cachegen_decoder.py +135 -0
- lmcache/v1/storage_backend/naive_serde/cachegen_encoder.py +83 -0
- lmcache/v1/storage_backend/naive_serde/kivi_serde.py +22 -0
- lmcache/v1/storage_backend/naive_serde/naive_serde.py +18 -0
- lmcache/v1/storage_backend/naive_serde/serde.py +37 -0
- lmcache/v1/storage_backend/native_clients/connector_client_base.py +165 -0
- lmcache/v1/storage_backend/native_clients/resp_client.py +35 -0
- lmcache/v1/storage_backend/nixl_storage_backend.py +1400 -0
- lmcache/v1/storage_backend/p2p_backend.py +788 -0
- lmcache/v1/storage_backend/path_sharder.py +117 -0
- lmcache/v1/storage_backend/pd_backend.py +646 -0
- lmcache/v1/storage_backend/plugins/dax_backend.py +1443 -0
- lmcache/v1/storage_backend/plugins/rust_raw_block_backend.py +1361 -0
- lmcache/v1/storage_backend/remote_backend.py +624 -0
- lmcache/v1/storage_backend/resp_client.py +227 -0
- lmcache/v1/storage_backend/storage_backend_listener.py +19 -0
- lmcache/v1/storage_backend/storage_manager.py +1352 -0
- lmcache/v1/system_detection.py +110 -0
- lmcache/v1/token_database.py +551 -0
- lmcache/v1/transfer_channel/__init__.py +83 -0
- lmcache/v1/transfer_channel/abstract.py +285 -0
- lmcache/v1/transfer_channel/mock_memory_channel.py +156 -0
- lmcache/v1/transfer_channel/nixl_channel.py +639 -0
- lmcache/v1/transfer_channel/py_socket_channel.py +260 -0
- lmcache/v1/transfer_channel/transfer_utils.py +63 -0
- lmcache/v1/utils/__init__.py +1 -0
- lmcache/v1/utils/bloom_filter.py +109 -0
- lmcache/v1/utils/cache_utils.py +125 -0
- lmcache_cli-0.4.5.dev0.dist-info/METADATA +185 -0
- lmcache_cli-0.4.5.dev0.dist-info/RECORD +399 -0
- lmcache_cli-0.4.5.dev0.dist-info/WHEEL +5 -0
- lmcache_cli-0.4.5.dev0.dist-info/entry_points.txt +2 -0
- lmcache_cli-0.4.5.dev0.dist-info/licenses/LICENSE +201 -0
- lmcache_cli-0.4.5.dev0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,535 @@
|
|
|
1
|
+
# SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
# Standard
|
|
3
|
+
from typing import Optional, Union
|
|
4
|
+
import asyncio
|
|
5
|
+
import json
|
|
6
|
+
import threading
|
|
7
|
+
import time
|
|
8
|
+
|
|
9
|
+
# Third Party
|
|
10
|
+
import msgspec
|
|
11
|
+
import zmq
|
|
12
|
+
|
|
13
|
+
# First Party
|
|
14
|
+
from lmcache.logging import init_logger
|
|
15
|
+
from lmcache.v1.cache_controller.controllers import KVController, RegistrationController
|
|
16
|
+
from lmcache.v1.cache_controller.executor import LMCacheClusterExecutor
|
|
17
|
+
from lmcache.v1.cache_controller.observability import (
|
|
18
|
+
PrometheusLogger,
|
|
19
|
+
SocketMetricsContext,
|
|
20
|
+
SocketType,
|
|
21
|
+
)
|
|
22
|
+
from lmcache.v1.rpc_utils import (
|
|
23
|
+
get_ip,
|
|
24
|
+
get_zmq_context,
|
|
25
|
+
get_zmq_socket,
|
|
26
|
+
)
|
|
27
|
+
|
|
28
|
+
from lmcache.v1.cache_controller.message import ( # isort: skip
|
|
29
|
+
BatchedKVOperationMsg,
|
|
30
|
+
BatchedP2PLookupMsg,
|
|
31
|
+
CheckFinishMsg,
|
|
32
|
+
ClearMsg,
|
|
33
|
+
CompressMsg,
|
|
34
|
+
DecompressMsg,
|
|
35
|
+
DeRegisterMsg,
|
|
36
|
+
ErrorMsg,
|
|
37
|
+
FullSyncBatchMsg,
|
|
38
|
+
FullSyncEndMsg,
|
|
39
|
+
FullSyncStartMsg,
|
|
40
|
+
FullSyncStatusMsg,
|
|
41
|
+
HealthMsg,
|
|
42
|
+
HeartbeatMsg,
|
|
43
|
+
LookupMsg,
|
|
44
|
+
MoveMsg,
|
|
45
|
+
Msg,
|
|
46
|
+
MsgBase,
|
|
47
|
+
OrchMsg,
|
|
48
|
+
OrchRetMsg,
|
|
49
|
+
PinMsg,
|
|
50
|
+
QueryInstMsg,
|
|
51
|
+
QueryWorkerInfoMsg,
|
|
52
|
+
RegisterMsg,
|
|
53
|
+
WorkerMsg,
|
|
54
|
+
WorkerReqMsg,
|
|
55
|
+
WorkerReqRetMsg,
|
|
56
|
+
)
|
|
57
|
+
|
|
58
|
+
logger = init_logger(__name__)
|
|
59
|
+
|
|
60
|
+
# TODO(Jiayi): Need to align the message types. For example,
|
|
61
|
+
# a controller should take in an control message and return
|
|
62
|
+
# a control message.
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
class LMCacheControllerManager:
|
|
66
|
+
def __init__(
|
|
67
|
+
self,
|
|
68
|
+
controller_urls: dict[str, str],
|
|
69
|
+
health_check_interval: int,
|
|
70
|
+
lmcache_worker_timeout: int,
|
|
71
|
+
full_sync_completion_threshold: float = 0.8,
|
|
72
|
+
full_sync_timeout_s: float = 300.0,
|
|
73
|
+
):
|
|
74
|
+
# Initialize stats logger
|
|
75
|
+
prometheus_labels = {
|
|
76
|
+
"role": "controller",
|
|
77
|
+
}
|
|
78
|
+
PrometheusLogger.GetOrCreate(prometheus_labels)
|
|
79
|
+
self.zmq_context = get_zmq_context()
|
|
80
|
+
self.controller_urls = controller_urls
|
|
81
|
+
# TODO(Jiayi): We might need multiple sockets if there are more
|
|
82
|
+
# controllers. For now, we use a single socket to receive messages
|
|
83
|
+
# for all controllers.
|
|
84
|
+
# Similarly we might need more sockets to handle different control
|
|
85
|
+
# messages. For now, we use one socket to handle all control messages.
|
|
86
|
+
|
|
87
|
+
# TODO(Jiayi): Another thing is that we might need to decoupe the
|
|
88
|
+
# interactions among `handle_worker_message`, `handle_control_message`
|
|
89
|
+
# and `handle_orchestration_message`. For example, in
|
|
90
|
+
# `handle_orchestration_message`, we might need to call
|
|
91
|
+
# `issue_control_message`. This will make the system less concurrent.
|
|
92
|
+
|
|
93
|
+
# Micro controllers
|
|
94
|
+
self.controller_pull_socket = get_zmq_socket(
|
|
95
|
+
self.zmq_context,
|
|
96
|
+
self.controller_urls["pull"],
|
|
97
|
+
protocol="tcp",
|
|
98
|
+
role=zmq.PULL, # type: ignore[attr-defined]
|
|
99
|
+
bind_or_connect="bind",
|
|
100
|
+
)
|
|
101
|
+
|
|
102
|
+
if self.controller_urls["reply"] is not None:
|
|
103
|
+
self.controller_reply_socket = get_zmq_socket(
|
|
104
|
+
self.zmq_context,
|
|
105
|
+
self.controller_urls["reply"],
|
|
106
|
+
protocol="tcp",
|
|
107
|
+
role=zmq.ROUTER, # type: ignore[attr-defined]
|
|
108
|
+
bind_or_connect="bind",
|
|
109
|
+
)
|
|
110
|
+
|
|
111
|
+
# Dedicated heartbeat socket to avoid blocking from other requests
|
|
112
|
+
if self.controller_urls.get("heartbeat") is not None:
|
|
113
|
+
self.controller_heartbeat_socket = get_zmq_socket(
|
|
114
|
+
self.zmq_context,
|
|
115
|
+
self.controller_urls["heartbeat"],
|
|
116
|
+
protocol="tcp",
|
|
117
|
+
role=zmq.ROUTER, # type: ignore[attr-defined]
|
|
118
|
+
bind_or_connect="bind",
|
|
119
|
+
)
|
|
120
|
+
else:
|
|
121
|
+
self.controller_heartbeat_socket = None
|
|
122
|
+
self.reg_controller = RegistrationController()
|
|
123
|
+
self.kv_controller = KVController(
|
|
124
|
+
registry=self.reg_controller.registry,
|
|
125
|
+
full_sync_completion_threshold=full_sync_completion_threshold,
|
|
126
|
+
full_sync_timeout_s=full_sync_timeout_s,
|
|
127
|
+
)
|
|
128
|
+
|
|
129
|
+
# Cluster executor
|
|
130
|
+
self.cluster_executor = LMCacheClusterExecutor(
|
|
131
|
+
reg_controller=self.reg_controller,
|
|
132
|
+
)
|
|
133
|
+
|
|
134
|
+
# post initialization of controllers
|
|
135
|
+
self.kv_controller.post_init(
|
|
136
|
+
reg_controller=self.reg_controller,
|
|
137
|
+
cluster_executor=self.cluster_executor,
|
|
138
|
+
)
|
|
139
|
+
self.reg_controller.post_init(
|
|
140
|
+
kv_controller=self.kv_controller,
|
|
141
|
+
cluster_executor=self.cluster_executor,
|
|
142
|
+
)
|
|
143
|
+
self.health_check_interval = health_check_interval
|
|
144
|
+
self.lmcache_worker_timeout = lmcache_worker_timeout
|
|
145
|
+
|
|
146
|
+
if self.health_check_interval > 0:
|
|
147
|
+
logger.info(
|
|
148
|
+
"Start health check thread, interval: %s", self.health_check_interval
|
|
149
|
+
)
|
|
150
|
+
self.loop = asyncio.new_event_loop()
|
|
151
|
+
self.thread = threading.Thread(
|
|
152
|
+
target=self.loop.run_forever,
|
|
153
|
+
daemon=True,
|
|
154
|
+
name="controller-health-thread",
|
|
155
|
+
)
|
|
156
|
+
self.thread.start()
|
|
157
|
+
asyncio.run_coroutine_threadsafe(self.health_check(), self.loop)
|
|
158
|
+
|
|
159
|
+
# Setup socket message count metrics
|
|
160
|
+
self._setup_socket_metrics()
|
|
161
|
+
|
|
162
|
+
async def handle_worker_message(self, msg: WorkerMsg) -> None:
|
|
163
|
+
if isinstance(msg, RegisterMsg):
|
|
164
|
+
await self.reg_controller.register(msg)
|
|
165
|
+
elif isinstance(msg, DeRegisterMsg):
|
|
166
|
+
await self.reg_controller.deregister(msg)
|
|
167
|
+
elif isinstance(msg, BatchedKVOperationMsg):
|
|
168
|
+
await self.kv_controller.handle_batched_kv_operations(msg)
|
|
169
|
+
elif isinstance(msg, FullSyncBatchMsg):
|
|
170
|
+
await self.kv_controller.handle_full_sync_batch(msg)
|
|
171
|
+
elif isinstance(msg, FullSyncEndMsg):
|
|
172
|
+
await self.kv_controller.handle_full_sync_end(msg)
|
|
173
|
+
else:
|
|
174
|
+
logger.error(f"Unknown worker message type: {msg}")
|
|
175
|
+
|
|
176
|
+
async def handle_worker_req_message(
|
|
177
|
+
self, msg: WorkerReqMsg
|
|
178
|
+
) -> Union[WorkerReqRetMsg, ErrorMsg]:
|
|
179
|
+
ret_msg: Union[WorkerReqRetMsg, ErrorMsg]
|
|
180
|
+
if isinstance(msg, RegisterMsg):
|
|
181
|
+
# Build extra_config with heartbeat_url if available
|
|
182
|
+
extra_config: dict[str, str] = {}
|
|
183
|
+
if self.controller_urls.get("heartbeat") is not None:
|
|
184
|
+
# Convert bind address (e.g., "0.0.0.0:8082" or "*:8082")
|
|
185
|
+
# to a connectable address using actual controller IP
|
|
186
|
+
heartbeat_bind_url = self.controller_urls["heartbeat"]
|
|
187
|
+
heartbeat_url = self._convert_bind_to_connect_url(
|
|
188
|
+
heartbeat_bind_url, worker_ip=msg.ip
|
|
189
|
+
)
|
|
190
|
+
extra_config["heartbeat_url"] = heartbeat_url
|
|
191
|
+
logger.debug(
|
|
192
|
+
"Returning heartbeat_url to worker: %s (bind: %s)",
|
|
193
|
+
heartbeat_url,
|
|
194
|
+
heartbeat_bind_url,
|
|
195
|
+
)
|
|
196
|
+
ret_msg = await self.reg_controller.register(msg, extra_config)
|
|
197
|
+
elif isinstance(msg, BatchedP2PLookupMsg):
|
|
198
|
+
ret_msg = await self.kv_controller.batched_p2p_lookup(msg)
|
|
199
|
+
elif isinstance(msg, HeartbeatMsg):
|
|
200
|
+
ret_msg = await self.reg_controller.heartbeat(msg)
|
|
201
|
+
elif isinstance(msg, FullSyncStartMsg):
|
|
202
|
+
ret_msg = await self.kv_controller.handle_full_sync_start(msg)
|
|
203
|
+
elif isinstance(msg, FullSyncStatusMsg):
|
|
204
|
+
ret_msg = await self.kv_controller.handle_full_sync_status(msg)
|
|
205
|
+
else:
|
|
206
|
+
logger.error(f"Unknown worker request message type: {msg}")
|
|
207
|
+
ret_msg = ErrorMsg(error=f"Unknown message type: {type(msg)}")
|
|
208
|
+
return ret_msg
|
|
209
|
+
|
|
210
|
+
async def handle_orchestration_message(self, msg: OrchMsg) -> OrchRetMsg:
|
|
211
|
+
if isinstance(msg, LookupMsg):
|
|
212
|
+
return await self.kv_controller.lookup(msg)
|
|
213
|
+
elif isinstance(msg, HealthMsg):
|
|
214
|
+
return await self.reg_controller.health(msg)
|
|
215
|
+
elif isinstance(msg, QueryInstMsg):
|
|
216
|
+
return await self.reg_controller.get_instance_id(msg)
|
|
217
|
+
elif isinstance(msg, ClearMsg):
|
|
218
|
+
return await self.kv_controller.clear(msg)
|
|
219
|
+
elif isinstance(msg, PinMsg):
|
|
220
|
+
return await self.kv_controller.pin(msg)
|
|
221
|
+
elif isinstance(msg, CompressMsg):
|
|
222
|
+
return await self.kv_controller.compress(msg)
|
|
223
|
+
elif isinstance(msg, DecompressMsg):
|
|
224
|
+
return await self.kv_controller.decompress(msg)
|
|
225
|
+
elif isinstance(msg, MoveMsg):
|
|
226
|
+
return await self.kv_controller.move(msg)
|
|
227
|
+
elif isinstance(msg, CheckFinishMsg):
|
|
228
|
+
# FIXME(Jiayi): This `check_finish` thing
|
|
229
|
+
# shouldn't be implemented in kv_controller.
|
|
230
|
+
return await self.kv_controller.check_finish(msg)
|
|
231
|
+
elif isinstance(msg, QueryWorkerInfoMsg):
|
|
232
|
+
return await self.reg_controller.query_worker_info(msg)
|
|
233
|
+
else:
|
|
234
|
+
logger.error(f"Unknown orchestration message type: {msg}")
|
|
235
|
+
raise RuntimeError(f"Unknown orchestration message type: {msg}")
|
|
236
|
+
|
|
237
|
+
def _setup_socket_metrics(self):
|
|
238
|
+
"""Setup metrics for socket message counts."""
|
|
239
|
+
# Initialize message counters for observability
|
|
240
|
+
self.pull_socket_message_count = 0
|
|
241
|
+
self.reply_socket_message_count = 0
|
|
242
|
+
|
|
243
|
+
# Initialize active request counters
|
|
244
|
+
self.pull_socket_active_requests = 0
|
|
245
|
+
self.reply_socket_active_requests = 0
|
|
246
|
+
|
|
247
|
+
prometheus_logger = PrometheusLogger.GetInstanceOrNone()
|
|
248
|
+
if prometheus_logger is not None:
|
|
249
|
+
prometheus_logger.pull_socket_message_count.set_function(
|
|
250
|
+
lambda: self.pull_socket_message_count
|
|
251
|
+
)
|
|
252
|
+
prometheus_logger.reply_socket_message_count.set_function(
|
|
253
|
+
lambda: self.reply_socket_message_count
|
|
254
|
+
)
|
|
255
|
+
|
|
256
|
+
# Socket queue/backlog metrics
|
|
257
|
+
prometheus_logger.pull_socket_has_pending.set_function(
|
|
258
|
+
lambda: self._check_socket_has_pending(self.controller_pull_socket)
|
|
259
|
+
)
|
|
260
|
+
if self.controller_urls["reply"] is not None:
|
|
261
|
+
prometheus_logger.reply_socket_has_pending.set_function(
|
|
262
|
+
lambda: self._check_socket_has_pending(self.controller_reply_socket)
|
|
263
|
+
)
|
|
264
|
+
|
|
265
|
+
# Active request metrics
|
|
266
|
+
prometheus_logger.pull_socket_active_requests.set_function(
|
|
267
|
+
lambda: self.pull_socket_active_requests
|
|
268
|
+
)
|
|
269
|
+
prometheus_logger.reply_socket_active_requests.set_function(
|
|
270
|
+
lambda: self.reply_socket_active_requests
|
|
271
|
+
)
|
|
272
|
+
|
|
273
|
+
def _convert_bind_to_connect_url(
|
|
274
|
+
self, bind_url: str, worker_ip: Optional[str] = None
|
|
275
|
+
) -> str:
|
|
276
|
+
"""Convert a bind address to a connectable address.
|
|
277
|
+
|
|
278
|
+
Bind addresses like "0.0.0.0:port" or "*:port" cannot be used
|
|
279
|
+
by workers to connect. We need to replace them with the actual
|
|
280
|
+
controller IP address.
|
|
281
|
+
|
|
282
|
+
If worker_ip is provided and matches the controller's IP, use
|
|
283
|
+
127.0.0.1 for loopback connection (more reliable than external IP).
|
|
284
|
+
|
|
285
|
+
Args:
|
|
286
|
+
bind_url: The bind URL (e.g., "0.0.0.0:8082" or "*:8082")
|
|
287
|
+
worker_ip: The IP address of the requesting worker (optional)
|
|
288
|
+
|
|
289
|
+
Returns:
|
|
290
|
+
A connectable URL (e.g., "192.168.1.100:8082" or "127.0.0.1:8082")
|
|
291
|
+
"""
|
|
292
|
+
if ":" not in bind_url:
|
|
293
|
+
return bind_url
|
|
294
|
+
|
|
295
|
+
host, port = bind_url.rsplit(":", 1)
|
|
296
|
+
# Replace bind-all addresses with actual IP
|
|
297
|
+
if host in ("0.0.0.0", "*", ""):
|
|
298
|
+
actual_ip = get_ip()
|
|
299
|
+
# If worker is on the same machine, use loopback for reliability
|
|
300
|
+
# This handles cases where external IP (e.g., VPN) doesn't support
|
|
301
|
+
# local loopback connections
|
|
302
|
+
if worker_ip is not None and worker_ip == actual_ip:
|
|
303
|
+
logger.debug(
|
|
304
|
+
"Worker IP %s matches controller IP, using 127.0.0.1",
|
|
305
|
+
worker_ip,
|
|
306
|
+
)
|
|
307
|
+
return f"127.0.0.1:{port}"
|
|
308
|
+
return f"{actual_ip}:{port}"
|
|
309
|
+
return bind_url
|
|
310
|
+
|
|
311
|
+
def _check_socket_has_pending(self, socket) -> int:
|
|
312
|
+
"""Check if socket has pending messages.
|
|
313
|
+
|
|
314
|
+
Returns:
|
|
315
|
+
1 if socket has pending messages, 0 otherwise
|
|
316
|
+
"""
|
|
317
|
+
try:
|
|
318
|
+
events = socket.get(zmq.EVENTS) # type: ignore[attr-defined]
|
|
319
|
+
# Check if POLLIN flag is set (indicates readable/pending messages)
|
|
320
|
+
has_pending = 1 if (events & zmq.POLLIN) else 0 # type: ignore[attr-defined]
|
|
321
|
+
return has_pending
|
|
322
|
+
except Exception as e:
|
|
323
|
+
logger.error(f"Error checking socket pending status: {e}")
|
|
324
|
+
return 0
|
|
325
|
+
|
|
326
|
+
async def handle_batched_push_request(self, socket) -> Optional[MsgBase]:
|
|
327
|
+
while True:
|
|
328
|
+
parts = await socket.recv_multipart()
|
|
329
|
+
part_count = len(parts)
|
|
330
|
+
with SocketMetricsContext(self, SocketType.PULL, part_count):
|
|
331
|
+
for part in parts:
|
|
332
|
+
# Parse message based on format
|
|
333
|
+
if part.startswith(b"{"):
|
|
334
|
+
# JSON format - typically from external systems
|
|
335
|
+
# like Mooncake
|
|
336
|
+
msg_dict = json.loads(part)
|
|
337
|
+
msg = msgspec.convert(msg_dict, type=Msg)
|
|
338
|
+
else:
|
|
339
|
+
# MessagePack format - internal LMCache communication
|
|
340
|
+
msg = msgspec.msgpack.decode(part, type=Msg)
|
|
341
|
+
if isinstance(msg, WorkerMsg):
|
|
342
|
+
await self.handle_worker_message(msg)
|
|
343
|
+
|
|
344
|
+
# FIXME(Jiayi): The abstraction of control messages
|
|
345
|
+
# might not be necessary.
|
|
346
|
+
# elif isinstance(msg, ControlMsg):
|
|
347
|
+
# await self.issue_control_message(msg)
|
|
348
|
+
elif isinstance(msg, OrchMsg):
|
|
349
|
+
await self.handle_orchestration_message(msg)
|
|
350
|
+
else:
|
|
351
|
+
logger.error(f"Unknown message type: {type(msg)}")
|
|
352
|
+
|
|
353
|
+
async def handle_batched_req_request(self, socket) -> Optional[MsgBase]:
|
|
354
|
+
"""Handle requests on ROUTER socket.
|
|
355
|
+
|
|
356
|
+
ROUTER socket receives multi-part messages:
|
|
357
|
+
[identity, empty_frame, payload]
|
|
358
|
+
and must reply with the same identity frame.
|
|
359
|
+
"""
|
|
360
|
+
while True:
|
|
361
|
+
frames = await socket.recv_multipart()
|
|
362
|
+
with SocketMetricsContext(self, SocketType.REPLY):
|
|
363
|
+
identity = None
|
|
364
|
+
try:
|
|
365
|
+
# ROUTER socket: [identity, empty_frame, payload]
|
|
366
|
+
if len(frames) < 3:
|
|
367
|
+
logger.error(
|
|
368
|
+
"Invalid ROUTER message format, expected >= 3 frames, "
|
|
369
|
+
"got %d",
|
|
370
|
+
len(frames),
|
|
371
|
+
)
|
|
372
|
+
continue
|
|
373
|
+
identity = frames[0]
|
|
374
|
+
# frames[1] is empty delimiter
|
|
375
|
+
part = frames[2]
|
|
376
|
+
|
|
377
|
+
# Parse message based on format
|
|
378
|
+
if part.startswith(b"{"):
|
|
379
|
+
# JSON format - typically from external systems like Mooncake
|
|
380
|
+
msg_dict = json.loads(part)
|
|
381
|
+
msg = msgspec.convert(msg_dict, type=Msg)
|
|
382
|
+
else:
|
|
383
|
+
# MessagePack format - internal LMCache communication
|
|
384
|
+
msg = msgspec.msgpack.decode(part, type=Msg)
|
|
385
|
+
|
|
386
|
+
if isinstance(msg, WorkerReqMsg):
|
|
387
|
+
ret_msg = await self.handle_worker_req_message(msg)
|
|
388
|
+
# Reply with identity frame for ROUTER socket
|
|
389
|
+
await socket.send_multipart(
|
|
390
|
+
[identity, b"", msgspec.msgpack.encode(ret_msg)]
|
|
391
|
+
)
|
|
392
|
+
else:
|
|
393
|
+
logger.error("Unknown message type: %s", type(msg))
|
|
394
|
+
err_msg = ErrorMsg(error=f"Unknown message type: {type(msg)}")
|
|
395
|
+
await socket.send_multipart(
|
|
396
|
+
[identity, b"", msgspec.msgpack.encode(err_msg)]
|
|
397
|
+
)
|
|
398
|
+
except (
|
|
399
|
+
json.JSONDecodeError,
|
|
400
|
+
msgspec.DecodeError,
|
|
401
|
+
msgspec.ValidationError,
|
|
402
|
+
zmq.ZMQError,
|
|
403
|
+
) as e:
|
|
404
|
+
logger.error("Error handling request message: %s", e, exc_info=True)
|
|
405
|
+
err_msg = ErrorMsg(error=str(e))
|
|
406
|
+
# Try to reply with error if we have identity
|
|
407
|
+
if identity is not None:
|
|
408
|
+
await socket.send_multipart(
|
|
409
|
+
[identity, b"", msgspec.msgpack.encode(err_msg)]
|
|
410
|
+
)
|
|
411
|
+
|
|
412
|
+
async def handle_heartbeat_request(self, socket) -> None:
|
|
413
|
+
"""Handle heartbeat requests on dedicated ROUTER socket.
|
|
414
|
+
|
|
415
|
+
This runs on a separate socket to ensure heartbeats are processed
|
|
416
|
+
without being blocked by other requests.
|
|
417
|
+
|
|
418
|
+
ROUTER socket receives multi-part messages:
|
|
419
|
+
[identity, empty_frame, payload]
|
|
420
|
+
"""
|
|
421
|
+
logger.info("Heartbeat handler task started, waiting for heartbeat requests...")
|
|
422
|
+
while True:
|
|
423
|
+
frames = await socket.recv_multipart()
|
|
424
|
+
logger.debug("Received heartbeat request with %d frames", len(frames))
|
|
425
|
+
with SocketMetricsContext(self, SocketType.REPLY):
|
|
426
|
+
identity = None
|
|
427
|
+
try:
|
|
428
|
+
# ROUTER socket: [identity, empty_frame, payload]
|
|
429
|
+
if len(frames) < 3:
|
|
430
|
+
logger.error(
|
|
431
|
+
"Invalid heartbeat ROUTER message format, "
|
|
432
|
+
"expected >= 3 frames, got %d",
|
|
433
|
+
len(frames),
|
|
434
|
+
)
|
|
435
|
+
continue
|
|
436
|
+
identity = frames[0]
|
|
437
|
+
part = frames[2]
|
|
438
|
+
|
|
439
|
+
if part.startswith(b"{"):
|
|
440
|
+
msg_dict = json.loads(part)
|
|
441
|
+
msg = msgspec.convert(msg_dict, type=Msg)
|
|
442
|
+
else:
|
|
443
|
+
msg = msgspec.msgpack.decode(part, type=Msg)
|
|
444
|
+
|
|
445
|
+
if isinstance(msg, HeartbeatMsg):
|
|
446
|
+
ret_msg = await self.reg_controller.heartbeat(msg)
|
|
447
|
+
await socket.send_multipart(
|
|
448
|
+
[identity, b"", msgspec.msgpack.encode(ret_msg)]
|
|
449
|
+
)
|
|
450
|
+
else:
|
|
451
|
+
logger.error(
|
|
452
|
+
"Unexpected message type on heartbeat socket: %s",
|
|
453
|
+
type(msg),
|
|
454
|
+
)
|
|
455
|
+
err_msg = ErrorMsg(
|
|
456
|
+
error=f"Expected HeartbeatMsg, got {type(msg)}"
|
|
457
|
+
)
|
|
458
|
+
await socket.send_multipart(
|
|
459
|
+
[identity, b"", msgspec.msgpack.encode(err_msg)]
|
|
460
|
+
)
|
|
461
|
+
except (
|
|
462
|
+
json.JSONDecodeError,
|
|
463
|
+
msgspec.DecodeError,
|
|
464
|
+
msgspec.ValidationError,
|
|
465
|
+
zmq.ZMQError,
|
|
466
|
+
) as e:
|
|
467
|
+
logger.error(
|
|
468
|
+
"Error handling heartbeat request: %s", e, exc_info=True
|
|
469
|
+
)
|
|
470
|
+
err_msg = ErrorMsg(error=str(e))
|
|
471
|
+
# Try to reply with error if we have identity
|
|
472
|
+
if identity is not None:
|
|
473
|
+
await socket.send_multipart(
|
|
474
|
+
[identity, b"", msgspec.msgpack.encode(err_msg)]
|
|
475
|
+
)
|
|
476
|
+
|
|
477
|
+
async def health_check(self):
|
|
478
|
+
while True:
|
|
479
|
+
await asyncio.sleep(self.health_check_interval)
|
|
480
|
+
worker_infos = self.reg_controller.registry.get_all_worker_infos_cached(1)
|
|
481
|
+
for worker_info in worker_infos:
|
|
482
|
+
if (
|
|
483
|
+
time.time() - worker_info.last_heartbeat_time
|
|
484
|
+
> self.lmcache_worker_timeout
|
|
485
|
+
):
|
|
486
|
+
logger.warning(
|
|
487
|
+
"Worker %s_%s last heartbeat time: %s, "
|
|
488
|
+
"current time: %s, more than %s seconds",
|
|
489
|
+
worker_info.instance_id,
|
|
490
|
+
worker_info.worker_id,
|
|
491
|
+
worker_info.last_heartbeat_time,
|
|
492
|
+
time.time(),
|
|
493
|
+
self.lmcache_worker_timeout,
|
|
494
|
+
)
|
|
495
|
+
# Perform a full deregister to clean up all associated resources.
|
|
496
|
+
deregister_msg = DeRegisterMsg(
|
|
497
|
+
instance_id=worker_info.instance_id,
|
|
498
|
+
worker_id=worker_info.worker_id,
|
|
499
|
+
ip=worker_info.ip,
|
|
500
|
+
port=worker_info.port,
|
|
501
|
+
)
|
|
502
|
+
await self.reg_controller.deregister(deregister_msg)
|
|
503
|
+
|
|
504
|
+
async def start_all(self):
|
|
505
|
+
tasks = []
|
|
506
|
+
if self.controller_urls["reply"] is not None:
|
|
507
|
+
logger.info(
|
|
508
|
+
"Starting reply request handler on %s", self.controller_urls["reply"]
|
|
509
|
+
)
|
|
510
|
+
tasks.append(self.handle_batched_req_request(self.controller_reply_socket))
|
|
511
|
+
if self.controller_heartbeat_socket is not None:
|
|
512
|
+
logger.info(
|
|
513
|
+
"Starting heartbeat request handler on %s",
|
|
514
|
+
self.controller_urls.get("heartbeat"),
|
|
515
|
+
)
|
|
516
|
+
tasks.append(
|
|
517
|
+
self.handle_heartbeat_request(self.controller_heartbeat_socket)
|
|
518
|
+
)
|
|
519
|
+
else:
|
|
520
|
+
logger.warning("Heartbeat socket is None, heartbeat handler not started!")
|
|
521
|
+
logger.info("Starting pull request handler on %s", self.controller_urls["pull"])
|
|
522
|
+
tasks.append(self.handle_batched_push_request(self.controller_pull_socket))
|
|
523
|
+
await asyncio.gather(
|
|
524
|
+
*tasks,
|
|
525
|
+
return_exceptions=True,
|
|
526
|
+
)
|
|
527
|
+
|
|
528
|
+
def close(self):
|
|
529
|
+
"""Clean up all resources owned by the controller manager."""
|
|
530
|
+
if hasattr(self, "controller_pull_socket"):
|
|
531
|
+
self.controller_pull_socket.close()
|
|
532
|
+
if hasattr(self, "controller_reply_socket"):
|
|
533
|
+
self.controller_reply_socket.close()
|
|
534
|
+
if hasattr(self, "zmq_context"):
|
|
535
|
+
self.zmq_context.destroy()
|
|
@@ -0,0 +1,11 @@
|
|
|
1
|
+
# SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
# First Party
|
|
3
|
+
from lmcache.v1.cache_controller.controllers.kv_controller import KVController
|
|
4
|
+
from lmcache.v1.cache_controller.controllers.registration_controller import ( # noqa: E501
|
|
5
|
+
RegistrationController,
|
|
6
|
+
)
|
|
7
|
+
|
|
8
|
+
__all__ = [
|
|
9
|
+
"KVController",
|
|
10
|
+
"RegistrationController",
|
|
11
|
+
]
|