lmcache-cli 0.4.5.dev0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- lmcache/__init__.py +84 -0
- lmcache/_version.py +24 -0
- lmcache/cli/__init__.py +1 -0
- lmcache/cli/commands/__init__.py +34 -0
- lmcache/cli/commands/base.py +157 -0
- lmcache/cli/commands/bench/__init__.py +557 -0
- lmcache/cli/commands/bench/engine_bench/__init__.py +1 -0
- lmcache/cli/commands/bench/engine_bench/config.py +245 -0
- lmcache/cli/commands/bench/engine_bench/interactive/__init__.py +274 -0
- lmcache/cli/commands/bench/engine_bench/interactive/config.json +10 -0
- lmcache/cli/commands/bench/engine_bench/interactive/schema.py +352 -0
- lmcache/cli/commands/bench/engine_bench/interactive/state.py +327 -0
- lmcache/cli/commands/bench/engine_bench/interactive/terminal.py +291 -0
- lmcache/cli/commands/bench/engine_bench/progress.py +145 -0
- lmcache/cli/commands/bench/engine_bench/request_sender.py +232 -0
- lmcache/cli/commands/bench/engine_bench/stats.py +275 -0
- lmcache/cli/commands/bench/engine_bench/workloads/__init__.py +153 -0
- lmcache/cli/commands/bench/engine_bench/workloads/base.py +122 -0
- lmcache/cli/commands/bench/engine_bench/workloads/long_doc_permutator.py +435 -0
- lmcache/cli/commands/bench/engine_bench/workloads/long_doc_qa.py +281 -0
- lmcache/cli/commands/bench/engine_bench/workloads/multi_round_chat.py +337 -0
- lmcache/cli/commands/bench/engine_bench/workloads/random_prefill.py +178 -0
- lmcache/cli/commands/describe.py +310 -0
- lmcache/cli/commands/kvcache.py +133 -0
- lmcache/cli/commands/mock.py +75 -0
- lmcache/cli/commands/ping.py +113 -0
- lmcache/cli/commands/query/__init__.py +155 -0
- lmcache/cli/commands/query/prompt.py +134 -0
- lmcache/cli/commands/query/request.py +357 -0
- lmcache/cli/commands/server.py +99 -0
- lmcache/cli/commands/tool/__init__.py +63 -0
- lmcache/cli/commands/tool/cache_simulator.py +113 -0
- lmcache/cli/commands/trace/__init__.py +505 -0
- lmcache/cli/commands/trace/dispatch.py +249 -0
- lmcache/cli/commands/trace/driver.py +372 -0
- lmcache/cli/commands/trace/stats.py +289 -0
- lmcache/cli/documents/lmcache.txt +11 -0
- lmcache/cli/main.py +42 -0
- lmcache/cli/metrics/__init__.py +29 -0
- lmcache/cli/metrics/formatter.py +171 -0
- lmcache/cli/metrics/handler.py +94 -0
- lmcache/cli/metrics/metrics.py +161 -0
- lmcache/cli/metrics/section.py +77 -0
- lmcache/connections.py +173 -0
- lmcache/integration/__init__.py +2 -0
- lmcache/integration/base_service_factory.py +165 -0
- lmcache/integration/request_telemetry/__init__.py +1 -0
- lmcache/integration/request_telemetry/base.py +51 -0
- lmcache/integration/request_telemetry/factory.py +113 -0
- lmcache/integration/request_telemetry/fastapi.py +109 -0
- lmcache/integration/request_telemetry/noop.py +35 -0
- lmcache/integration/sglang/__init__.py +2 -0
- lmcache/integration/sglang/sglang_adapter.py +326 -0
- lmcache/integration/sglang/utils.py +39 -0
- lmcache/integration/vllm/__init__.py +1 -0
- lmcache/integration/vllm/lmcache_connector_v1.py +213 -0
- lmcache/integration/vllm/lmcache_connector_v1_085.py +150 -0
- lmcache/integration/vllm/lmcache_mp_connector_0180.py +1072 -0
- lmcache/integration/vllm/tests/test_mm_hash_utils.py +112 -0
- lmcache/integration/vllm/utils.py +433 -0
- lmcache/integration/vllm/vllm_multi_process_adapter.py +1090 -0
- lmcache/integration/vllm/vllm_service_factory.py +339 -0
- lmcache/integration/vllm/vllm_v1_adapter.py +1713 -0
- lmcache/logging.py +107 -0
- lmcache/native_storage_ops.pyi +230 -0
- lmcache/non_cuda_equivalents.py +1424 -0
- lmcache/observability.py +1958 -0
- lmcache/storage_backend/serde/__init__.py +1 -0
- lmcache/storage_backend/serde/cachegen_basics.py +210 -0
- lmcache/storage_backend/serde/cachegen_decoder.py +207 -0
- lmcache/storage_backend/serde/cachegen_encoder.py +394 -0
- lmcache/storage_backend/serde/serde.py +75 -0
- lmcache/tools/__init__.py +1 -0
- lmcache/tools/cache_simulator/README.md +392 -0
- lmcache/tools/cache_simulator/__init__.py +1 -0
- lmcache/tools/cache_simulator/docs/simulate_example.png +0 -0
- lmcache/tools/cache_simulator/docs/sweep_example.png +0 -0
- lmcache/tools/cache_simulator/gen_bench_dataset.py +360 -0
- lmcache/tools/cache_simulator/lru_cache.py +124 -0
- lmcache/tools/cache_simulator/plot_hit_rate.py +231 -0
- lmcache/tools/cache_simulator/simulator.py +795 -0
- lmcache/tools/controller_benchmark/README.md +161 -0
- lmcache/tools/controller_benchmark/__init__.py +1 -0
- lmcache/tools/controller_benchmark/__main__.py +331 -0
- lmcache/tools/controller_benchmark/benchmark.py +660 -0
- lmcache/tools/controller_benchmark/config.py +44 -0
- lmcache/tools/controller_benchmark/constants.py +10 -0
- lmcache/tools/controller_benchmark/handlers/__init__.py +46 -0
- lmcache/tools/controller_benchmark/handlers/admit.py +52 -0
- lmcache/tools/controller_benchmark/handlers/base.py +47 -0
- lmcache/tools/controller_benchmark/handlers/deregister.py +49 -0
- lmcache/tools/controller_benchmark/handlers/evict.py +52 -0
- lmcache/tools/controller_benchmark/handlers/heartbeat.py +56 -0
- lmcache/tools/controller_benchmark/handlers/p2p_lookup.py +47 -0
- lmcache/tools/controller_benchmark/handlers/register.py +56 -0
- lmcache/tools/mp_status_viewer/__init__.py +1 -0
- lmcache/tools/mp_status_viewer/__main__.py +95 -0
- lmcache/usage_context.py +417 -0
- lmcache/utils.py +665 -0
- lmcache/v1/__init__.py +2 -0
- lmcache/v1/api_server/__init__.py +2 -0
- lmcache/v1/api_server/__main__.py +537 -0
- lmcache/v1/basic_check.py +112 -0
- lmcache/v1/cache_controller/__init__.py +9 -0
- lmcache/v1/cache_controller/commands/__init__.py +15 -0
- lmcache/v1/cache_controller/commands/base.py +35 -0
- lmcache/v1/cache_controller/commands/full_sync.py +49 -0
- lmcache/v1/cache_controller/config.py +176 -0
- lmcache/v1/cache_controller/controller_manager.py +535 -0
- lmcache/v1/cache_controller/controllers/__init__.py +11 -0
- lmcache/v1/cache_controller/controllers/full_sync_tracker.py +473 -0
- lmcache/v1/cache_controller/controllers/kv_controller.py +439 -0
- lmcache/v1/cache_controller/controllers/registration_controller.py +282 -0
- lmcache/v1/cache_controller/executor.py +463 -0
- lmcache/v1/cache_controller/frontend/static/css/style.css +201 -0
- lmcache/v1/cache_controller/frontend/static/img/logo.png +0 -0
- lmcache/v1/cache_controller/frontend/static/index.html +234 -0
- lmcache/v1/cache_controller/frontend/static/js/controller_app.js +660 -0
- lmcache/v1/cache_controller/full_sync_sender.py +475 -0
- lmcache/v1/cache_controller/locks.py +149 -0
- lmcache/v1/cache_controller/message.py +828 -0
- lmcache/v1/cache_controller/observability.py +208 -0
- lmcache/v1/cache_controller/utils.py +679 -0
- lmcache/v1/cache_controller/worker.py +665 -0
- lmcache/v1/cache_engine.py +2058 -0
- lmcache/v1/cache_interface.py +19 -0
- lmcache/v1/check/__init__.py +74 -0
- lmcache/v1/check/check_mode_gen.py +86 -0
- lmcache/v1/check/check_mode_test_l2_adapter.py +284 -0
- lmcache/v1/check/check_mode_test_remote.py +155 -0
- lmcache/v1/check/check_mode_test_storage_manager.py +142 -0
- lmcache/v1/check/utils.py +571 -0
- lmcache/v1/compute/__init__.py +2 -0
- lmcache/v1/compute/attention/__init__.py +0 -0
- lmcache/v1/compute/attention/abstract.py +39 -0
- lmcache/v1/compute/attention/flash_attn.py +129 -0
- lmcache/v1/compute/attention/flash_infer_sparse.py +284 -0
- lmcache/v1/compute/attention/metadata.py +85 -0
- lmcache/v1/compute/attention/utils.py +14 -0
- lmcache/v1/compute/blend/__init__.py +7 -0
- lmcache/v1/compute/blend/blender.py +168 -0
- lmcache/v1/compute/blend/metadata.py +34 -0
- lmcache/v1/compute/blend/utils.py +63 -0
- lmcache/v1/compute/models/__init__.py +0 -0
- lmcache/v1/compute/models/base.py +141 -0
- lmcache/v1/compute/models/llama.py +9 -0
- lmcache/v1/compute/models/qwen3.py +24 -0
- lmcache/v1/compute/models/utils.py +68 -0
- lmcache/v1/compute/positional_encoding.py +199 -0
- lmcache/v1/config.py +848 -0
- lmcache/v1/config_base.py +848 -0
- lmcache/v1/distributed/api.py +248 -0
- lmcache/v1/distributed/config.py +321 -0
- lmcache/v1/distributed/error.py +64 -0
- lmcache/v1/distributed/eviction.py +192 -0
- lmcache/v1/distributed/eviction_policy/__init__.py +21 -0
- lmcache/v1/distributed/eviction_policy/factory.py +27 -0
- lmcache/v1/distributed/eviction_policy/lru.py +244 -0
- lmcache/v1/distributed/eviction_policy/noop.py +50 -0
- lmcache/v1/distributed/internal_api.py +170 -0
- lmcache/v1/distributed/l1_manager.py +835 -0
- lmcache/v1/distributed/l2_adapters/__init__.py +67 -0
- lmcache/v1/distributed/l2_adapters/base.py +360 -0
- lmcache/v1/distributed/l2_adapters/config.py +385 -0
- lmcache/v1/distributed/l2_adapters/factory.py +205 -0
- lmcache/v1/distributed/l2_adapters/fs_l2_adapter.py +747 -0
- lmcache/v1/distributed/l2_adapters/fs_native_l2_adapter.py +167 -0
- lmcache/v1/distributed/l2_adapters/mock_l2_adapter.py +516 -0
- lmcache/v1/distributed/l2_adapters/mooncake_store_l2_adapter.py +135 -0
- lmcache/v1/distributed/l2_adapters/native_connector_l2_adapter.py +468 -0
- lmcache/v1/distributed/l2_adapters/native_plugin_l2_adapter.py +199 -0
- lmcache/v1/distributed/l2_adapters/nixl_store_dynamic_l2_adapter.py +831 -0
- lmcache/v1/distributed/l2_adapters/nixl_store_l2_adapter.py +983 -0
- lmcache/v1/distributed/l2_adapters/plugin_l2_adapter.py +210 -0
- lmcache/v1/distributed/l2_adapters/resp_l2_adapter.py +176 -0
- lmcache/v1/distributed/memory_manager.py +179 -0
- lmcache/v1/distributed/storage_controller.py +39 -0
- lmcache/v1/distributed/storage_controllers/__init__.py +43 -0
- lmcache/v1/distributed/storage_controllers/eviction_controller.py +242 -0
- lmcache/v1/distributed/storage_controllers/prefetch_controller.py +830 -0
- lmcache/v1/distributed/storage_controllers/prefetch_policy.py +193 -0
- lmcache/v1/distributed/storage_controllers/store_controller.py +452 -0
- lmcache/v1/distributed/storage_controllers/store_policy.py +213 -0
- lmcache/v1/distributed/storage_manager.py +532 -0
- lmcache/v1/event_manager.py +145 -0
- lmcache/v1/exceptions/__init__.py +16 -0
- lmcache/v1/gpu_connector/__init__.py +126 -0
- lmcache/v1/gpu_connector/gpu_connectors.py +1906 -0
- lmcache/v1/gpu_connector/gpu_ops.py +85 -0
- lmcache/v1/gpu_connector/hpu_connector.py +326 -0
- lmcache/v1/gpu_connector/mock_gpu_connector.py +67 -0
- lmcache/v1/gpu_connector/utils.py +890 -0
- lmcache/v1/gpu_connector/xpu_connectors.py +916 -0
- lmcache/v1/health_monitor/__init__.py +1 -0
- lmcache/v1/health_monitor/base.py +587 -0
- lmcache/v1/health_monitor/checks/__init__.py +1 -0
- lmcache/v1/health_monitor/checks/remote_backend_check.py +304 -0
- lmcache/v1/health_monitor/constants.py +36 -0
- lmcache/v1/internal_api_server/__init__.py +0 -0
- lmcache/v1/internal_api_server/api_registry.py +59 -0
- lmcache/v1/internal_api_server/api_server.py +120 -0
- lmcache/v1/internal_api_server/common/__init__.py +1 -0
- lmcache/v1/internal_api_server/common/env_api.py +22 -0
- lmcache/v1/internal_api_server/common/loglevel_api.py +57 -0
- lmcache/v1/internal_api_server/common/metrics_api.py +29 -0
- lmcache/v1/internal_api_server/common/periodic_thread_api.py +138 -0
- lmcache/v1/internal_api_server/common/run_script_api.py +73 -0
- lmcache/v1/internal_api_server/common/thread_api.py +63 -0
- lmcache/v1/internal_api_server/controller/__init__.py +1 -0
- lmcache/v1/internal_api_server/controller/key_stats_api.py +81 -0
- lmcache/v1/internal_api_server/controller/worker_info_api.py +136 -0
- lmcache/v1/internal_api_server/utils.py +43 -0
- lmcache/v1/internal_api_server/vllm/__init__.py +1 -0
- lmcache/v1/internal_api_server/vllm/backend_api.py +221 -0
- lmcache/v1/internal_api_server/vllm/bypass_api.py +204 -0
- lmcache/v1/internal_api_server/vllm/cache_api.py +895 -0
- lmcache/v1/internal_api_server/vllm/chunk_statistics_api.py +141 -0
- lmcache/v1/internal_api_server/vllm/conf_api.py +147 -0
- lmcache/v1/internal_api_server/vllm/freeze_api.py +172 -0
- lmcache/v1/internal_api_server/vllm/hot_cache_api.py +184 -0
- lmcache/v1/internal_api_server/vllm/inference_api.py +65 -0
- lmcache/v1/internal_api_server/vllm/load_fs_chunks_api.py +320 -0
- lmcache/v1/internal_api_server/vllm/lookup_api.py +145 -0
- lmcache/v1/internal_api_server/vllm/version_api.py +25 -0
- lmcache/v1/kv_layer_groups.py +267 -0
- lmcache/v1/lazy_memory_allocator.py +284 -0
- lmcache/v1/lookup_client/__init__.py +25 -0
- lmcache/v1/lookup_client/abstract_client.py +77 -0
- lmcache/v1/lookup_client/async_lookup_message.py +50 -0
- lmcache/v1/lookup_client/chunk_statistics_lookup_client.py +200 -0
- lmcache/v1/lookup_client/factory.py +251 -0
- lmcache/v1/lookup_client/hit_limit_lookup_client.py +86 -0
- lmcache/v1/lookup_client/lmcache_async_lookup_client.py +407 -0
- lmcache/v1/lookup_client/lmcache_lookup_client.py +285 -0
- lmcache/v1/lookup_client/lmcache_lookup_client_bypass.py +99 -0
- lmcache/v1/lookup_client/mooncake_lookup_client.py +87 -0
- lmcache/v1/lookup_client/record_strategies/__init__.py +77 -0
- lmcache/v1/lookup_client/record_strategies/base.py +327 -0
- lmcache/v1/lookup_client/record_strategies/file_hash.py +130 -0
- lmcache/v1/lookup_client/record_strategies/memory_bloom_filter.py +81 -0
- lmcache/v1/manager.py +539 -0
- lmcache/v1/memory_management.py +2619 -0
- lmcache/v1/metadata.py +114 -0
- lmcache/v1/mp_observability/AGENTS.override.md +21 -0
- lmcache/v1/mp_observability/README.md +204 -0
- lmcache/v1/mp_observability/config.py +340 -0
- lmcache/v1/mp_observability/event.py +100 -0
- lmcache/v1/mp_observability/event_bus.py +313 -0
- lmcache/v1/mp_observability/otel_init.py +145 -0
- lmcache/v1/mp_observability/subscribers/__init__.py +28 -0
- lmcache/v1/mp_observability/subscribers/logging/__init__.py +19 -0
- lmcache/v1/mp_observability/subscribers/logging/l1.py +56 -0
- lmcache/v1/mp_observability/subscribers/logging/l2.py +73 -0
- lmcache/v1/mp_observability/subscribers/logging/lookup_hash.py +209 -0
- lmcache/v1/mp_observability/subscribers/logging/mp_server.py +90 -0
- lmcache/v1/mp_observability/subscribers/logging/sm.py +59 -0
- lmcache/v1/mp_observability/subscribers/metrics/__init__.py +20 -0
- lmcache/v1/mp_observability/subscribers/metrics/l0_lifecycle.py +290 -0
- lmcache/v1/mp_observability/subscribers/metrics/l1.py +55 -0
- lmcache/v1/mp_observability/subscribers/metrics/l1_lifecycle.py +166 -0
- lmcache/v1/mp_observability/subscribers/metrics/l2.py +121 -0
- lmcache/v1/mp_observability/subscribers/metrics/sm.py +69 -0
- lmcache/v1/mp_observability/subscribers/tracing/__init__.py +12 -0
- lmcache/v1/mp_observability/subscribers/tracing/mp_server.py +333 -0
- lmcache/v1/mp_observability/subscribers/tracing/span_registry.py +148 -0
- lmcache/v1/mp_observability/trace/__init__.py +50 -0
- lmcache/v1/mp_observability/trace/codecs.py +255 -0
- lmcache/v1/mp_observability/trace/decorator.py +147 -0
- lmcache/v1/mp_observability/trace/format.py +132 -0
- lmcache/v1/mp_observability/trace/lifecycle.py +83 -0
- lmcache/v1/mp_observability/trace/reader.py +167 -0
- lmcache/v1/mp_observability/trace/recorder.py +300 -0
- lmcache/v1/multiprocess/__init__.py +0 -0
- lmcache/v1/multiprocess/affinity_pool.py +102 -0
- lmcache/v1/multiprocess/blend_server_v2.py +891 -0
- lmcache/v1/multiprocess/config.py +253 -0
- lmcache/v1/multiprocess/custom_types.py +281 -0
- lmcache/v1/multiprocess/futures.py +194 -0
- lmcache/v1/multiprocess/gpu_context.py +511 -0
- lmcache/v1/multiprocess/http_server.py +235 -0
- lmcache/v1/multiprocess/mp_runtime_plugin_launcher.py +130 -0
- lmcache/v1/multiprocess/mq.py +732 -0
- lmcache/v1/multiprocess/protocol.py +86 -0
- lmcache/v1/multiprocess/protocols/README.md +213 -0
- lmcache/v1/multiprocess/protocols/__init__.py +127 -0
- lmcache/v1/multiprocess/protocols/base.py +89 -0
- lmcache/v1/multiprocess/protocols/blend.py +109 -0
- lmcache/v1/multiprocess/protocols/blend_v2.py +57 -0
- lmcache/v1/multiprocess/protocols/controller.py +53 -0
- lmcache/v1/multiprocess/protocols/debug.py +34 -0
- lmcache/v1/multiprocess/protocols/engine.py +146 -0
- lmcache/v1/multiprocess/protocols/observability.py +39 -0
- lmcache/v1/multiprocess/server.py +1134 -0
- lmcache/v1/multiprocess/session.py +190 -0
- lmcache/v1/multiprocess/token_hasher.py +441 -0
- lmcache/v1/offload_server/__init__.py +17 -0
- lmcache/v1/offload_server/abstract_server.py +37 -0
- lmcache/v1/offload_server/message.py +30 -0
- lmcache/v1/offload_server/zmq_server.py +122 -0
- lmcache/v1/periodic_thread.py +579 -0
- lmcache/v1/pin_monitor.py +246 -0
- lmcache/v1/plugin/__init__.py +0 -0
- lmcache/v1/plugin/runtime_plugin_launcher.py +211 -0
- lmcache/v1/protocol.py +317 -0
- lmcache/v1/rpc/__init__.py +17 -0
- lmcache/v1/rpc/transport.py +105 -0
- lmcache/v1/rpc/zmq_transport.py +213 -0
- lmcache/v1/rpc_utils.py +165 -0
- lmcache/v1/server/__init__.py +2 -0
- lmcache/v1/server/__main__.py +170 -0
- lmcache/v1/server/storage_backend/__init__.py +21 -0
- lmcache/v1/server/storage_backend/abstract_backend.py +80 -0
- lmcache/v1/server/storage_backend/local_backend.py +75 -0
- lmcache/v1/server/utils.py +21 -0
- lmcache/v1/standalone/__init__.py +1 -0
- lmcache/v1/standalone/__main__.py +583 -0
- lmcache/v1/standalone/manager.py +80 -0
- lmcache/v1/standalone/standalone_service_factory.py +86 -0
- lmcache/v1/storage_backend/__init__.py +313 -0
- lmcache/v1/storage_backend/abstract_backend.py +445 -0
- lmcache/v1/storage_backend/audit_backend.py +233 -0
- lmcache/v1/storage_backend/batched_message_sender.py +222 -0
- lmcache/v1/storage_backend/cache_policy/__init__.py +45 -0
- lmcache/v1/storage_backend/cache_policy/base_policy.py +87 -0
- lmcache/v1/storage_backend/cache_policy/fifo.py +58 -0
- lmcache/v1/storage_backend/cache_policy/lfu.py +105 -0
- lmcache/v1/storage_backend/cache_policy/lru.py +81 -0
- lmcache/v1/storage_backend/cache_policy/mru.py +61 -0
- lmcache/v1/storage_backend/connector/__init__.py +443 -0
- lmcache/v1/storage_backend/connector/audit_adapter.py +77 -0
- lmcache/v1/storage_backend/connector/audit_connector.py +320 -0
- lmcache/v1/storage_backend/connector/base_connector.py +379 -0
- lmcache/v1/storage_backend/connector/blackhole_adapter.py +21 -0
- lmcache/v1/storage_backend/connector/blackhole_connector.py +37 -0
- lmcache/v1/storage_backend/connector/eic_adapter.py +31 -0
- lmcache/v1/storage_backend/connector/eic_connector.py +757 -0
- lmcache/v1/storage_backend/connector/external_adapter.py +79 -0
- lmcache/v1/storage_backend/connector/fs_adapter.py +51 -0
- lmcache/v1/storage_backend/connector/fs_connector.py +403 -0
- lmcache/v1/storage_backend/connector/infinistore_adapter.py +56 -0
- lmcache/v1/storage_backend/connector/infinistore_connector.py +177 -0
- lmcache/v1/storage_backend/connector/instrumented_connector.py +219 -0
- lmcache/v1/storage_backend/connector/lm_adapter.py +31 -0
- lmcache/v1/storage_backend/connector/lm_connector.py +176 -0
- lmcache/v1/storage_backend/connector/mock_adapter.py +57 -0
- lmcache/v1/storage_backend/connector/mock_connector.py +349 -0
- lmcache/v1/storage_backend/connector/mooncakestore_adapter.py +43 -0
- lmcache/v1/storage_backend/connector/mooncakestore_connector.py +614 -0
- lmcache/v1/storage_backend/connector/redis_adapter.py +181 -0
- lmcache/v1/storage_backend/connector/redis_connector.py +828 -0
- lmcache/v1/storage_backend/connector/s3_adapter.py +59 -0
- lmcache/v1/storage_backend/connector/s3_connector.py +699 -0
- lmcache/v1/storage_backend/connector/sagemaker_hyperpod_adapter.py +233 -0
- lmcache/v1/storage_backend/connector/sagemaker_hyperpod_connector.py +987 -0
- lmcache/v1/storage_backend/connector/valkey_adapter.py +114 -0
- lmcache/v1/storage_backend/connector/valkey_connector.py +627 -0
- lmcache/v1/storage_backend/gds_backend.py +1199 -0
- lmcache/v1/storage_backend/job_executor/__init__.py +0 -0
- lmcache/v1/storage_backend/job_executor/base_executor.py +34 -0
- lmcache/v1/storage_backend/job_executor/pq_executor.py +235 -0
- lmcache/v1/storage_backend/local_cpu_backend.py +810 -0
- lmcache/v1/storage_backend/local_disk_backend.py +656 -0
- lmcache/v1/storage_backend/maru_backend.py +734 -0
- lmcache/v1/storage_backend/naive_serde/__init__.py +50 -0
- lmcache/v1/storage_backend/naive_serde/cachegen_basics.py +133 -0
- lmcache/v1/storage_backend/naive_serde/cachegen_decoder.py +135 -0
- lmcache/v1/storage_backend/naive_serde/cachegen_encoder.py +83 -0
- lmcache/v1/storage_backend/naive_serde/kivi_serde.py +22 -0
- lmcache/v1/storage_backend/naive_serde/naive_serde.py +18 -0
- lmcache/v1/storage_backend/naive_serde/serde.py +37 -0
- lmcache/v1/storage_backend/native_clients/connector_client_base.py +165 -0
- lmcache/v1/storage_backend/native_clients/resp_client.py +35 -0
- lmcache/v1/storage_backend/nixl_storage_backend.py +1400 -0
- lmcache/v1/storage_backend/p2p_backend.py +788 -0
- lmcache/v1/storage_backend/path_sharder.py +117 -0
- lmcache/v1/storage_backend/pd_backend.py +646 -0
- lmcache/v1/storage_backend/plugins/dax_backend.py +1443 -0
- lmcache/v1/storage_backend/plugins/rust_raw_block_backend.py +1361 -0
- lmcache/v1/storage_backend/remote_backend.py +624 -0
- lmcache/v1/storage_backend/resp_client.py +227 -0
- lmcache/v1/storage_backend/storage_backend_listener.py +19 -0
- lmcache/v1/storage_backend/storage_manager.py +1352 -0
- lmcache/v1/system_detection.py +110 -0
- lmcache/v1/token_database.py +551 -0
- lmcache/v1/transfer_channel/__init__.py +83 -0
- lmcache/v1/transfer_channel/abstract.py +285 -0
- lmcache/v1/transfer_channel/mock_memory_channel.py +156 -0
- lmcache/v1/transfer_channel/nixl_channel.py +639 -0
- lmcache/v1/transfer_channel/py_socket_channel.py +260 -0
- lmcache/v1/transfer_channel/transfer_utils.py +63 -0
- lmcache/v1/utils/__init__.py +1 -0
- lmcache/v1/utils/bloom_filter.py +109 -0
- lmcache/v1/utils/cache_utils.py +125 -0
- lmcache_cli-0.4.5.dev0.dist-info/METADATA +185 -0
- lmcache_cli-0.4.5.dev0.dist-info/RECORD +399 -0
- lmcache_cli-0.4.5.dev0.dist-info/WHEEL +5 -0
- lmcache_cli-0.4.5.dev0.dist-info/entry_points.txt +2 -0
- lmcache_cli-0.4.5.dev0.dist-info/licenses/LICENSE +201 -0
- lmcache_cli-0.4.5.dev0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,473 @@
|
|
|
1
|
+
# SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
# Standard
|
|
3
|
+
from typing import Optional, Tuple
|
|
4
|
+
import time
|
|
5
|
+
|
|
6
|
+
# First Party
|
|
7
|
+
from lmcache.logging import init_logger
|
|
8
|
+
from lmcache.v1.cache_controller.utils import (
|
|
9
|
+
FullSyncState,
|
|
10
|
+
RegistryTree,
|
|
11
|
+
WorkerNode,
|
|
12
|
+
WorkerSyncInfo,
|
|
13
|
+
)
|
|
14
|
+
|
|
15
|
+
logger = init_logger(__name__)
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
class FullSyncTracker:
|
|
19
|
+
"""
|
|
20
|
+
Tracks full sync state for all workers.
|
|
21
|
+
|
|
22
|
+
This class manages the state of full sync operations, including:
|
|
23
|
+
- Tracking which workers need full sync
|
|
24
|
+
- Monitoring sync progress
|
|
25
|
+
- Handling sync timeout
|
|
26
|
+
- Determining when freeze mode can be exited
|
|
27
|
+
"""
|
|
28
|
+
|
|
29
|
+
def __init__(
|
|
30
|
+
self,
|
|
31
|
+
registry_tree: RegistryTree,
|
|
32
|
+
completion_threshold: float = 0.8,
|
|
33
|
+
sync_timeout_s: float = 300.0,
|
|
34
|
+
):
|
|
35
|
+
"""
|
|
36
|
+
Initialize the FullSyncTracker.
|
|
37
|
+
|
|
38
|
+
Args:
|
|
39
|
+
registry_tree: The registry tree containing worker nodes
|
|
40
|
+
completion_threshold: Percentage of workers that need to complete
|
|
41
|
+
sync before others can exit freeze mode (default: 80%)
|
|
42
|
+
sync_timeout_s: Timeout in seconds for a single worker's sync
|
|
43
|
+
(default: 300s)
|
|
44
|
+
"""
|
|
45
|
+
self.registry_tree = registry_tree
|
|
46
|
+
self.completion_threshold = completion_threshold
|
|
47
|
+
self.sync_timeout_s = sync_timeout_s
|
|
48
|
+
|
|
49
|
+
# Flag to indicate if controller just restarted and needs full sync
|
|
50
|
+
self._need_full_sync_all = True
|
|
51
|
+
|
|
52
|
+
def _get_sync_info(
|
|
53
|
+
self, instance_id: str, worker_id: int
|
|
54
|
+
) -> Optional[WorkerSyncInfo]:
|
|
55
|
+
"""Get sync info for a worker from the registry tree."""
|
|
56
|
+
worker_node = self.registry_tree.get_worker(instance_id, worker_id)
|
|
57
|
+
if worker_node is None:
|
|
58
|
+
return None
|
|
59
|
+
return worker_node.sync_info
|
|
60
|
+
|
|
61
|
+
def _set_sync_info(
|
|
62
|
+
self, instance_id: str, worker_id: int, sync_info: Optional[WorkerSyncInfo]
|
|
63
|
+
) -> bool:
|
|
64
|
+
"""Set sync info for a worker. Returns True if successful."""
|
|
65
|
+
worker_node = self.registry_tree.get_worker(instance_id, worker_id)
|
|
66
|
+
if worker_node is None:
|
|
67
|
+
return False
|
|
68
|
+
worker_node.sync_info = sync_info
|
|
69
|
+
return True
|
|
70
|
+
|
|
71
|
+
def set_need_full_sync_all(self, need: bool) -> None:
|
|
72
|
+
"""Set whether all workers need full sync (e.g., after controller restart)"""
|
|
73
|
+
self._need_full_sync_all = need
|
|
74
|
+
logger.info("Set need_full_sync_all to %s", need)
|
|
75
|
+
|
|
76
|
+
def _get_all_workers_cached(
|
|
77
|
+
self, timeout_seconds: Optional[float] = None
|
|
78
|
+
) -> list[tuple[str, WorkerNode]]:
|
|
79
|
+
"""Get all registered workers from the registry tree."""
|
|
80
|
+
return self.registry_tree.get_all_worker_nodes_cached(timeout_seconds)
|
|
81
|
+
|
|
82
|
+
def should_request_full_sync(
|
|
83
|
+
self, instance_id: str, worker_id: int
|
|
84
|
+
) -> Tuple[bool, Optional[str]]:
|
|
85
|
+
"""
|
|
86
|
+
Check if a worker should perform full sync.
|
|
87
|
+
|
|
88
|
+
Returns:
|
|
89
|
+
Tuple of (need_sync, reason)
|
|
90
|
+
"""
|
|
91
|
+
# Case 1: Controller just restarted, all workers need sync
|
|
92
|
+
if self._need_full_sync_all:
|
|
93
|
+
sync_info = self._get_sync_info(instance_id, worker_id)
|
|
94
|
+
if sync_info is None or sync_info.state not in (
|
|
95
|
+
FullSyncState.SYNCING,
|
|
96
|
+
FullSyncState.COMPLETED,
|
|
97
|
+
):
|
|
98
|
+
return True, "controller_restart"
|
|
99
|
+
|
|
100
|
+
# Case 2: Worker sync failed/timeout, needs retry
|
|
101
|
+
sync_info = self._get_sync_info(instance_id, worker_id)
|
|
102
|
+
if sync_info is not None and sync_info.state == FullSyncState.FAILED:
|
|
103
|
+
return True, "sync_failed_retry"
|
|
104
|
+
|
|
105
|
+
return False, None
|
|
106
|
+
|
|
107
|
+
def is_worker_syncing(self, instance_id: str, worker_id: int) -> bool:
|
|
108
|
+
"""
|
|
109
|
+
Check if a worker is currently in sync state.
|
|
110
|
+
|
|
111
|
+
When a worker is syncing, incremental events should be discarded.
|
|
112
|
+
"""
|
|
113
|
+
sync_info = self._get_sync_info(instance_id, worker_id)
|
|
114
|
+
if sync_info is None:
|
|
115
|
+
return False
|
|
116
|
+
return sync_info.state == FullSyncState.SYNCING
|
|
117
|
+
|
|
118
|
+
def get_sync_id(self, instance_id: str, worker_id: int) -> Optional[str]:
|
|
119
|
+
"""
|
|
120
|
+
Get sync ID for a worker.
|
|
121
|
+
|
|
122
|
+
Returns sync ID if worker is syncing, None otherwise.
|
|
123
|
+
"""
|
|
124
|
+
sync_info = self._get_sync_info(instance_id, worker_id)
|
|
125
|
+
if sync_info is None:
|
|
126
|
+
return None
|
|
127
|
+
return sync_info.sync_id
|
|
128
|
+
|
|
129
|
+
def start_sync(
|
|
130
|
+
self,
|
|
131
|
+
instance_id: str,
|
|
132
|
+
worker_id: int,
|
|
133
|
+
sync_id: str,
|
|
134
|
+
total_keys: int,
|
|
135
|
+
batch_count: int,
|
|
136
|
+
) -> bool:
|
|
137
|
+
"""
|
|
138
|
+
Start sync for a worker.
|
|
139
|
+
|
|
140
|
+
Returns:
|
|
141
|
+
True if sync started successfully, False otherwise
|
|
142
|
+
"""
|
|
143
|
+
report_id = (instance_id, worker_id)
|
|
144
|
+
current_time = time.time()
|
|
145
|
+
|
|
146
|
+
# Check if already syncing with different sync_id
|
|
147
|
+
existing_sync = self._get_sync_info(instance_id, worker_id)
|
|
148
|
+
if existing_sync is not None and existing_sync.state == FullSyncState.SYNCING:
|
|
149
|
+
if existing_sync.sync_id != sync_id:
|
|
150
|
+
logger.warning(
|
|
151
|
+
"Worker %s already syncing with different sync_id: "
|
|
152
|
+
"existing=%s, new=%s",
|
|
153
|
+
report_id,
|
|
154
|
+
existing_sync.sync_id,
|
|
155
|
+
sync_id,
|
|
156
|
+
)
|
|
157
|
+
return False
|
|
158
|
+
|
|
159
|
+
new_sync_info = WorkerSyncInfo(
|
|
160
|
+
sync_id=sync_id,
|
|
161
|
+
state=FullSyncState.SYNCING,
|
|
162
|
+
start_time=current_time,
|
|
163
|
+
expected_total_keys=total_keys,
|
|
164
|
+
expected_batch_count=batch_count,
|
|
165
|
+
last_activity_time=current_time,
|
|
166
|
+
)
|
|
167
|
+
if not self._set_sync_info(instance_id, worker_id, new_sync_info):
|
|
168
|
+
logger.warning(
|
|
169
|
+
"Failed to start sync for worker %s: worker not found", report_id
|
|
170
|
+
)
|
|
171
|
+
return False
|
|
172
|
+
|
|
173
|
+
logger.info(
|
|
174
|
+
"Started full sync for worker %s: sync_id=%s, "
|
|
175
|
+
"expected_keys=%d, expected_batches=%d",
|
|
176
|
+
report_id,
|
|
177
|
+
sync_id,
|
|
178
|
+
total_keys,
|
|
179
|
+
batch_count,
|
|
180
|
+
)
|
|
181
|
+
return True
|
|
182
|
+
|
|
183
|
+
def receive_batch(
|
|
184
|
+
self,
|
|
185
|
+
instance_id: str,
|
|
186
|
+
worker_id: int,
|
|
187
|
+
sync_id: str,
|
|
188
|
+
batch_id: int,
|
|
189
|
+
keys_count: int,
|
|
190
|
+
) -> bool:
|
|
191
|
+
"""
|
|
192
|
+
Record receipt of a sync batch.
|
|
193
|
+
|
|
194
|
+
Returns:
|
|
195
|
+
True if batch was recorded, False if invalid
|
|
196
|
+
"""
|
|
197
|
+
report_id = (instance_id, worker_id)
|
|
198
|
+
sync_info = self._get_sync_info(instance_id, worker_id)
|
|
199
|
+
|
|
200
|
+
if sync_info is None:
|
|
201
|
+
logger.warning(
|
|
202
|
+
"Received batch for unknown sync session: worker=%s, sync_id=%s",
|
|
203
|
+
report_id,
|
|
204
|
+
sync_id,
|
|
205
|
+
)
|
|
206
|
+
return False
|
|
207
|
+
|
|
208
|
+
if sync_info.sync_id != sync_id:
|
|
209
|
+
logger.warning(
|
|
210
|
+
"Sync ID mismatch: expected=%s, received=%s",
|
|
211
|
+
sync_info.sync_id,
|
|
212
|
+
sync_id,
|
|
213
|
+
)
|
|
214
|
+
return False
|
|
215
|
+
|
|
216
|
+
if sync_info.state != FullSyncState.SYNCING:
|
|
217
|
+
logger.warning(
|
|
218
|
+
"Received batch for non-syncing worker: worker=%s, state=%s",
|
|
219
|
+
report_id,
|
|
220
|
+
sync_info.state,
|
|
221
|
+
)
|
|
222
|
+
return False
|
|
223
|
+
|
|
224
|
+
sync_info.received_batches.add(batch_id)
|
|
225
|
+
sync_info.received_keys_count += keys_count
|
|
226
|
+
sync_info.last_activity_time = time.time()
|
|
227
|
+
|
|
228
|
+
logger.debug(
|
|
229
|
+
"Received batch %d for worker %s: keys=%d, total_received=%d",
|
|
230
|
+
batch_id,
|
|
231
|
+
report_id,
|
|
232
|
+
keys_count,
|
|
233
|
+
sync_info.received_keys_count,
|
|
234
|
+
)
|
|
235
|
+
return True
|
|
236
|
+
|
|
237
|
+
def complete_sync(
|
|
238
|
+
self,
|
|
239
|
+
instance_id: str,
|
|
240
|
+
worker_id: int,
|
|
241
|
+
sync_id: str,
|
|
242
|
+
actual_total_keys: int,
|
|
243
|
+
) -> bool:
|
|
244
|
+
"""
|
|
245
|
+
Mark sync as completed for a worker.
|
|
246
|
+
|
|
247
|
+
Returns:
|
|
248
|
+
True if completion was successful, False otherwise
|
|
249
|
+
"""
|
|
250
|
+
report_id = (instance_id, worker_id)
|
|
251
|
+
sync_info = self._get_sync_info(instance_id, worker_id)
|
|
252
|
+
|
|
253
|
+
if sync_info is None:
|
|
254
|
+
logger.warning(
|
|
255
|
+
"Received sync end for unknown session: worker=%s, sync_id=%s",
|
|
256
|
+
report_id,
|
|
257
|
+
sync_id,
|
|
258
|
+
)
|
|
259
|
+
return False
|
|
260
|
+
|
|
261
|
+
if sync_info.sync_id != sync_id:
|
|
262
|
+
logger.warning(
|
|
263
|
+
"Sync ID mismatch on completion: expected=%s, received=%s",
|
|
264
|
+
sync_info.sync_id,
|
|
265
|
+
sync_id,
|
|
266
|
+
)
|
|
267
|
+
return False
|
|
268
|
+
|
|
269
|
+
# Verify key count
|
|
270
|
+
if sync_info.received_keys_count != actual_total_keys:
|
|
271
|
+
logger.warning(
|
|
272
|
+
"Key count mismatch on completion: received=%d, reported=%d",
|
|
273
|
+
sync_info.received_keys_count,
|
|
274
|
+
actual_total_keys,
|
|
275
|
+
)
|
|
276
|
+
# Still mark as completed but log the discrepancy
|
|
277
|
+
|
|
278
|
+
sync_info.state = FullSyncState.COMPLETED
|
|
279
|
+
sync_info.last_activity_time = time.time()
|
|
280
|
+
|
|
281
|
+
logger.info(
|
|
282
|
+
"Completed full sync for worker %s: sync_id=%s, "
|
|
283
|
+
"received_keys=%d, batches=%d",
|
|
284
|
+
report_id,
|
|
285
|
+
sync_id,
|
|
286
|
+
sync_info.received_keys_count,
|
|
287
|
+
len(sync_info.received_batches),
|
|
288
|
+
)
|
|
289
|
+
return True
|
|
290
|
+
|
|
291
|
+
def mark_failed(self, instance_id: str, worker_id: int, reason: str) -> None:
|
|
292
|
+
"""Mark a worker's sync as failed"""
|
|
293
|
+
report_id = (instance_id, worker_id)
|
|
294
|
+
sync_info = self._get_sync_info(instance_id, worker_id)
|
|
295
|
+
|
|
296
|
+
if sync_info is not None:
|
|
297
|
+
sync_info.state = FullSyncState.FAILED
|
|
298
|
+
logger.warning(
|
|
299
|
+
"Marked sync as failed for worker %s: reason=%s", report_id, reason
|
|
300
|
+
)
|
|
301
|
+
|
|
302
|
+
# Only for testing
|
|
303
|
+
def check_sync_timeout(self) -> None:
|
|
304
|
+
"""
|
|
305
|
+
Check for sync timeouts and mark failed workers.
|
|
306
|
+
|
|
307
|
+
This should be called periodically (e.g., in health check loop).
|
|
308
|
+
"""
|
|
309
|
+
current_time = time.time()
|
|
310
|
+
for instance_id, worker_node in self._get_all_workers_cached():
|
|
311
|
+
sync_info = worker_node.sync_info
|
|
312
|
+
if sync_info is not None and sync_info.state == FullSyncState.SYNCING:
|
|
313
|
+
if current_time - sync_info.last_activity_time > self.sync_timeout_s:
|
|
314
|
+
self.mark_failed(
|
|
315
|
+
instance_id,
|
|
316
|
+
worker_node.worker_id,
|
|
317
|
+
f"timeout after {self.sync_timeout_s}s",
|
|
318
|
+
)
|
|
319
|
+
|
|
320
|
+
def get_global_progress(self) -> float:
|
|
321
|
+
"""
|
|
322
|
+
Get the global sync progress.
|
|
323
|
+
|
|
324
|
+
Returns:
|
|
325
|
+
Progress as a float between 0.0 and 1.0
|
|
326
|
+
|
|
327
|
+
Note: Uses cached worker list for Prometheus metrics efficiency.
|
|
328
|
+
|
|
329
|
+
Progress calculation:
|
|
330
|
+
- Denominator: total number of all workers
|
|
331
|
+
- Numerator: workers that are ready to serve (COMPLETED or no sync info needed)
|
|
332
|
+
- Workers in SYNCING or FAILED state are NOT considered ready
|
|
333
|
+
"""
|
|
334
|
+
all_workers = self._get_all_workers_cached()
|
|
335
|
+
if not all_workers:
|
|
336
|
+
return 0.0
|
|
337
|
+
|
|
338
|
+
total = len(all_workers)
|
|
339
|
+
ready_count = sum(
|
|
340
|
+
1
|
|
341
|
+
for _, worker_node in all_workers
|
|
342
|
+
if worker_node.sync_info is None
|
|
343
|
+
or worker_node.sync_info.state == FullSyncState.COMPLETED
|
|
344
|
+
)
|
|
345
|
+
|
|
346
|
+
# Progress = ready / total
|
|
347
|
+
return ready_count / total
|
|
348
|
+
|
|
349
|
+
def get_completed_count(self) -> int:
|
|
350
|
+
"""Get count of workers that have completed sync.
|
|
351
|
+
|
|
352
|
+
Note: Uses cached worker list for Prometheus metrics efficiency.
|
|
353
|
+
"""
|
|
354
|
+
return sum(
|
|
355
|
+
1
|
|
356
|
+
for _, worker_node in self._get_all_workers_cached()
|
|
357
|
+
if worker_node.sync_info is not None
|
|
358
|
+
and worker_node.sync_info.state == FullSyncState.COMPLETED
|
|
359
|
+
)
|
|
360
|
+
|
|
361
|
+
def get_syncing_count(self) -> int:
|
|
362
|
+
"""Get count of workers currently syncing.
|
|
363
|
+
|
|
364
|
+
Note: Uses cached worker list for Prometheus metrics efficiency.
|
|
365
|
+
"""
|
|
366
|
+
return sum(
|
|
367
|
+
1
|
|
368
|
+
for _, worker_node in self._get_all_workers_cached()
|
|
369
|
+
if worker_node.sync_info is not None
|
|
370
|
+
and worker_node.sync_info.state == FullSyncState.SYNCING
|
|
371
|
+
)
|
|
372
|
+
|
|
373
|
+
def can_exit_freeze(self, progress: Optional[float] = None) -> bool:
|
|
374
|
+
"""
|
|
375
|
+
Check if the completion threshold is reached and freeze mode can be exited.
|
|
376
|
+
|
|
377
|
+
Args:
|
|
378
|
+
progress: Pre-computed global progress. If None, will be computed.
|
|
379
|
+
|
|
380
|
+
Returns:
|
|
381
|
+
True if enough workers have completed sync
|
|
382
|
+
"""
|
|
383
|
+
if progress is None:
|
|
384
|
+
progress = self.get_global_progress()
|
|
385
|
+
can_exit = progress >= self.completion_threshold
|
|
386
|
+
|
|
387
|
+
if can_exit and self._need_full_sync_all:
|
|
388
|
+
# Once threshold is reached, disable the global full sync flag
|
|
389
|
+
logger.info(
|
|
390
|
+
"Full sync completion threshold reached (%.1f%%), "
|
|
391
|
+
"disabling need_full_sync_all",
|
|
392
|
+
progress * 100,
|
|
393
|
+
)
|
|
394
|
+
self._need_full_sync_all = False
|
|
395
|
+
|
|
396
|
+
return can_exit
|
|
397
|
+
|
|
398
|
+
def get_total_missing_batches_count(self) -> int:
|
|
399
|
+
"""
|
|
400
|
+
Get total count of missing batches across all syncing workers.
|
|
401
|
+
|
|
402
|
+
Returns:
|
|
403
|
+
Total number of missing batches
|
|
404
|
+
|
|
405
|
+
Note: Uses cached worker list for Prometheus metrics efficiency.
|
|
406
|
+
"""
|
|
407
|
+
total = 0
|
|
408
|
+
for instance_id, worker_node in self._get_all_workers_cached():
|
|
409
|
+
sync_info = worker_node.sync_info
|
|
410
|
+
if sync_info is not None and sync_info.state == FullSyncState.SYNCING:
|
|
411
|
+
expected_batches = set(range(sync_info.expected_batch_count))
|
|
412
|
+
missing = expected_batches - sync_info.received_batches
|
|
413
|
+
total += len(missing)
|
|
414
|
+
return total
|
|
415
|
+
|
|
416
|
+
def get_missing_batches(
|
|
417
|
+
self, instance_id: str, worker_id: int, sync_id: str
|
|
418
|
+
) -> list[int]:
|
|
419
|
+
"""
|
|
420
|
+
Get list of missing batch IDs that need to be resent.
|
|
421
|
+
|
|
422
|
+
Args:
|
|
423
|
+
instance_id: The instance ID
|
|
424
|
+
worker_id: The worker ID
|
|
425
|
+
sync_id: The sync session ID
|
|
426
|
+
|
|
427
|
+
Returns:
|
|
428
|
+
List of missing batch IDs, empty if sync is complete or invalid
|
|
429
|
+
"""
|
|
430
|
+
sync_info = self._get_sync_info(instance_id, worker_id)
|
|
431
|
+
|
|
432
|
+
if sync_info is None:
|
|
433
|
+
return []
|
|
434
|
+
|
|
435
|
+
# Check sync_id matches
|
|
436
|
+
if sync_info.sync_id != sync_id:
|
|
437
|
+
return []
|
|
438
|
+
|
|
439
|
+
# If already completed, no missing batches
|
|
440
|
+
if sync_info.state == FullSyncState.COMPLETED:
|
|
441
|
+
return []
|
|
442
|
+
|
|
443
|
+
# If not syncing, no missing batches
|
|
444
|
+
if sync_info.state != FullSyncState.SYNCING:
|
|
445
|
+
return []
|
|
446
|
+
|
|
447
|
+
# Calculate missing batches
|
|
448
|
+
expected_batches = set(range(sync_info.expected_batch_count))
|
|
449
|
+
missing = expected_batches - sync_info.received_batches
|
|
450
|
+
|
|
451
|
+
return sorted(missing)
|
|
452
|
+
|
|
453
|
+
def get_sync_status(
|
|
454
|
+
self, instance_id: str, worker_id: int, sync_id: str
|
|
455
|
+
) -> Tuple[bool, float, bool, list[int]]:
|
|
456
|
+
"""
|
|
457
|
+
Get sync status for a specific worker.
|
|
458
|
+
|
|
459
|
+
Returns:
|
|
460
|
+
Tuple of (is_complete, global_progress, can_exit_freeze, missing_batches)
|
|
461
|
+
"""
|
|
462
|
+
sync_info = self._get_sync_info(instance_id, worker_id)
|
|
463
|
+
|
|
464
|
+
is_complete = (
|
|
465
|
+
sync_info is not None
|
|
466
|
+
and sync_info.sync_id == sync_id
|
|
467
|
+
and sync_info.state == FullSyncState.COMPLETED
|
|
468
|
+
)
|
|
469
|
+
global_progress = self.get_global_progress()
|
|
470
|
+
can_exit = self.can_exit_freeze(global_progress)
|
|
471
|
+
missing_batches = self.get_missing_batches(instance_id, worker_id, sync_id)
|
|
472
|
+
|
|
473
|
+
return is_complete, global_progress, can_exit, missing_batches
|