lmcache-cli 0.4.5.dev0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- lmcache/__init__.py +84 -0
- lmcache/_version.py +24 -0
- lmcache/cli/__init__.py +1 -0
- lmcache/cli/commands/__init__.py +34 -0
- lmcache/cli/commands/base.py +157 -0
- lmcache/cli/commands/bench/__init__.py +557 -0
- lmcache/cli/commands/bench/engine_bench/__init__.py +1 -0
- lmcache/cli/commands/bench/engine_bench/config.py +245 -0
- lmcache/cli/commands/bench/engine_bench/interactive/__init__.py +274 -0
- lmcache/cli/commands/bench/engine_bench/interactive/config.json +10 -0
- lmcache/cli/commands/bench/engine_bench/interactive/schema.py +352 -0
- lmcache/cli/commands/bench/engine_bench/interactive/state.py +327 -0
- lmcache/cli/commands/bench/engine_bench/interactive/terminal.py +291 -0
- lmcache/cli/commands/bench/engine_bench/progress.py +145 -0
- lmcache/cli/commands/bench/engine_bench/request_sender.py +232 -0
- lmcache/cli/commands/bench/engine_bench/stats.py +275 -0
- lmcache/cli/commands/bench/engine_bench/workloads/__init__.py +153 -0
- lmcache/cli/commands/bench/engine_bench/workloads/base.py +122 -0
- lmcache/cli/commands/bench/engine_bench/workloads/long_doc_permutator.py +435 -0
- lmcache/cli/commands/bench/engine_bench/workloads/long_doc_qa.py +281 -0
- lmcache/cli/commands/bench/engine_bench/workloads/multi_round_chat.py +337 -0
- lmcache/cli/commands/bench/engine_bench/workloads/random_prefill.py +178 -0
- lmcache/cli/commands/describe.py +310 -0
- lmcache/cli/commands/kvcache.py +133 -0
- lmcache/cli/commands/mock.py +75 -0
- lmcache/cli/commands/ping.py +113 -0
- lmcache/cli/commands/query/__init__.py +155 -0
- lmcache/cli/commands/query/prompt.py +134 -0
- lmcache/cli/commands/query/request.py +357 -0
- lmcache/cli/commands/server.py +99 -0
- lmcache/cli/commands/tool/__init__.py +63 -0
- lmcache/cli/commands/tool/cache_simulator.py +113 -0
- lmcache/cli/commands/trace/__init__.py +505 -0
- lmcache/cli/commands/trace/dispatch.py +249 -0
- lmcache/cli/commands/trace/driver.py +372 -0
- lmcache/cli/commands/trace/stats.py +289 -0
- lmcache/cli/documents/lmcache.txt +11 -0
- lmcache/cli/main.py +42 -0
- lmcache/cli/metrics/__init__.py +29 -0
- lmcache/cli/metrics/formatter.py +171 -0
- lmcache/cli/metrics/handler.py +94 -0
- lmcache/cli/metrics/metrics.py +161 -0
- lmcache/cli/metrics/section.py +77 -0
- lmcache/connections.py +173 -0
- lmcache/integration/__init__.py +2 -0
- lmcache/integration/base_service_factory.py +165 -0
- lmcache/integration/request_telemetry/__init__.py +1 -0
- lmcache/integration/request_telemetry/base.py +51 -0
- lmcache/integration/request_telemetry/factory.py +113 -0
- lmcache/integration/request_telemetry/fastapi.py +109 -0
- lmcache/integration/request_telemetry/noop.py +35 -0
- lmcache/integration/sglang/__init__.py +2 -0
- lmcache/integration/sglang/sglang_adapter.py +326 -0
- lmcache/integration/sglang/utils.py +39 -0
- lmcache/integration/vllm/__init__.py +1 -0
- lmcache/integration/vllm/lmcache_connector_v1.py +213 -0
- lmcache/integration/vllm/lmcache_connector_v1_085.py +150 -0
- lmcache/integration/vllm/lmcache_mp_connector_0180.py +1072 -0
- lmcache/integration/vllm/tests/test_mm_hash_utils.py +112 -0
- lmcache/integration/vllm/utils.py +433 -0
- lmcache/integration/vllm/vllm_multi_process_adapter.py +1090 -0
- lmcache/integration/vllm/vllm_service_factory.py +339 -0
- lmcache/integration/vllm/vllm_v1_adapter.py +1713 -0
- lmcache/logging.py +107 -0
- lmcache/native_storage_ops.pyi +230 -0
- lmcache/non_cuda_equivalents.py +1424 -0
- lmcache/observability.py +1958 -0
- lmcache/storage_backend/serde/__init__.py +1 -0
- lmcache/storage_backend/serde/cachegen_basics.py +210 -0
- lmcache/storage_backend/serde/cachegen_decoder.py +207 -0
- lmcache/storage_backend/serde/cachegen_encoder.py +394 -0
- lmcache/storage_backend/serde/serde.py +75 -0
- lmcache/tools/__init__.py +1 -0
- lmcache/tools/cache_simulator/README.md +392 -0
- lmcache/tools/cache_simulator/__init__.py +1 -0
- lmcache/tools/cache_simulator/docs/simulate_example.png +0 -0
- lmcache/tools/cache_simulator/docs/sweep_example.png +0 -0
- lmcache/tools/cache_simulator/gen_bench_dataset.py +360 -0
- lmcache/tools/cache_simulator/lru_cache.py +124 -0
- lmcache/tools/cache_simulator/plot_hit_rate.py +231 -0
- lmcache/tools/cache_simulator/simulator.py +795 -0
- lmcache/tools/controller_benchmark/README.md +161 -0
- lmcache/tools/controller_benchmark/__init__.py +1 -0
- lmcache/tools/controller_benchmark/__main__.py +331 -0
- lmcache/tools/controller_benchmark/benchmark.py +660 -0
- lmcache/tools/controller_benchmark/config.py +44 -0
- lmcache/tools/controller_benchmark/constants.py +10 -0
- lmcache/tools/controller_benchmark/handlers/__init__.py +46 -0
- lmcache/tools/controller_benchmark/handlers/admit.py +52 -0
- lmcache/tools/controller_benchmark/handlers/base.py +47 -0
- lmcache/tools/controller_benchmark/handlers/deregister.py +49 -0
- lmcache/tools/controller_benchmark/handlers/evict.py +52 -0
- lmcache/tools/controller_benchmark/handlers/heartbeat.py +56 -0
- lmcache/tools/controller_benchmark/handlers/p2p_lookup.py +47 -0
- lmcache/tools/controller_benchmark/handlers/register.py +56 -0
- lmcache/tools/mp_status_viewer/__init__.py +1 -0
- lmcache/tools/mp_status_viewer/__main__.py +95 -0
- lmcache/usage_context.py +417 -0
- lmcache/utils.py +665 -0
- lmcache/v1/__init__.py +2 -0
- lmcache/v1/api_server/__init__.py +2 -0
- lmcache/v1/api_server/__main__.py +537 -0
- lmcache/v1/basic_check.py +112 -0
- lmcache/v1/cache_controller/__init__.py +9 -0
- lmcache/v1/cache_controller/commands/__init__.py +15 -0
- lmcache/v1/cache_controller/commands/base.py +35 -0
- lmcache/v1/cache_controller/commands/full_sync.py +49 -0
- lmcache/v1/cache_controller/config.py +176 -0
- lmcache/v1/cache_controller/controller_manager.py +535 -0
- lmcache/v1/cache_controller/controllers/__init__.py +11 -0
- lmcache/v1/cache_controller/controllers/full_sync_tracker.py +473 -0
- lmcache/v1/cache_controller/controllers/kv_controller.py +439 -0
- lmcache/v1/cache_controller/controllers/registration_controller.py +282 -0
- lmcache/v1/cache_controller/executor.py +463 -0
- lmcache/v1/cache_controller/frontend/static/css/style.css +201 -0
- lmcache/v1/cache_controller/frontend/static/img/logo.png +0 -0
- lmcache/v1/cache_controller/frontend/static/index.html +234 -0
- lmcache/v1/cache_controller/frontend/static/js/controller_app.js +660 -0
- lmcache/v1/cache_controller/full_sync_sender.py +475 -0
- lmcache/v1/cache_controller/locks.py +149 -0
- lmcache/v1/cache_controller/message.py +828 -0
- lmcache/v1/cache_controller/observability.py +208 -0
- lmcache/v1/cache_controller/utils.py +679 -0
- lmcache/v1/cache_controller/worker.py +665 -0
- lmcache/v1/cache_engine.py +2058 -0
- lmcache/v1/cache_interface.py +19 -0
- lmcache/v1/check/__init__.py +74 -0
- lmcache/v1/check/check_mode_gen.py +86 -0
- lmcache/v1/check/check_mode_test_l2_adapter.py +284 -0
- lmcache/v1/check/check_mode_test_remote.py +155 -0
- lmcache/v1/check/check_mode_test_storage_manager.py +142 -0
- lmcache/v1/check/utils.py +571 -0
- lmcache/v1/compute/__init__.py +2 -0
- lmcache/v1/compute/attention/__init__.py +0 -0
- lmcache/v1/compute/attention/abstract.py +39 -0
- lmcache/v1/compute/attention/flash_attn.py +129 -0
- lmcache/v1/compute/attention/flash_infer_sparse.py +284 -0
- lmcache/v1/compute/attention/metadata.py +85 -0
- lmcache/v1/compute/attention/utils.py +14 -0
- lmcache/v1/compute/blend/__init__.py +7 -0
- lmcache/v1/compute/blend/blender.py +168 -0
- lmcache/v1/compute/blend/metadata.py +34 -0
- lmcache/v1/compute/blend/utils.py +63 -0
- lmcache/v1/compute/models/__init__.py +0 -0
- lmcache/v1/compute/models/base.py +141 -0
- lmcache/v1/compute/models/llama.py +9 -0
- lmcache/v1/compute/models/qwen3.py +24 -0
- lmcache/v1/compute/models/utils.py +68 -0
- lmcache/v1/compute/positional_encoding.py +199 -0
- lmcache/v1/config.py +848 -0
- lmcache/v1/config_base.py +848 -0
- lmcache/v1/distributed/api.py +248 -0
- lmcache/v1/distributed/config.py +321 -0
- lmcache/v1/distributed/error.py +64 -0
- lmcache/v1/distributed/eviction.py +192 -0
- lmcache/v1/distributed/eviction_policy/__init__.py +21 -0
- lmcache/v1/distributed/eviction_policy/factory.py +27 -0
- lmcache/v1/distributed/eviction_policy/lru.py +244 -0
- lmcache/v1/distributed/eviction_policy/noop.py +50 -0
- lmcache/v1/distributed/internal_api.py +170 -0
- lmcache/v1/distributed/l1_manager.py +835 -0
- lmcache/v1/distributed/l2_adapters/__init__.py +67 -0
- lmcache/v1/distributed/l2_adapters/base.py +360 -0
- lmcache/v1/distributed/l2_adapters/config.py +385 -0
- lmcache/v1/distributed/l2_adapters/factory.py +205 -0
- lmcache/v1/distributed/l2_adapters/fs_l2_adapter.py +747 -0
- lmcache/v1/distributed/l2_adapters/fs_native_l2_adapter.py +167 -0
- lmcache/v1/distributed/l2_adapters/mock_l2_adapter.py +516 -0
- lmcache/v1/distributed/l2_adapters/mooncake_store_l2_adapter.py +135 -0
- lmcache/v1/distributed/l2_adapters/native_connector_l2_adapter.py +468 -0
- lmcache/v1/distributed/l2_adapters/native_plugin_l2_adapter.py +199 -0
- lmcache/v1/distributed/l2_adapters/nixl_store_dynamic_l2_adapter.py +831 -0
- lmcache/v1/distributed/l2_adapters/nixl_store_l2_adapter.py +983 -0
- lmcache/v1/distributed/l2_adapters/plugin_l2_adapter.py +210 -0
- lmcache/v1/distributed/l2_adapters/resp_l2_adapter.py +176 -0
- lmcache/v1/distributed/memory_manager.py +179 -0
- lmcache/v1/distributed/storage_controller.py +39 -0
- lmcache/v1/distributed/storage_controllers/__init__.py +43 -0
- lmcache/v1/distributed/storage_controllers/eviction_controller.py +242 -0
- lmcache/v1/distributed/storage_controllers/prefetch_controller.py +830 -0
- lmcache/v1/distributed/storage_controllers/prefetch_policy.py +193 -0
- lmcache/v1/distributed/storage_controllers/store_controller.py +452 -0
- lmcache/v1/distributed/storage_controllers/store_policy.py +213 -0
- lmcache/v1/distributed/storage_manager.py +532 -0
- lmcache/v1/event_manager.py +145 -0
- lmcache/v1/exceptions/__init__.py +16 -0
- lmcache/v1/gpu_connector/__init__.py +126 -0
- lmcache/v1/gpu_connector/gpu_connectors.py +1906 -0
- lmcache/v1/gpu_connector/gpu_ops.py +85 -0
- lmcache/v1/gpu_connector/hpu_connector.py +326 -0
- lmcache/v1/gpu_connector/mock_gpu_connector.py +67 -0
- lmcache/v1/gpu_connector/utils.py +890 -0
- lmcache/v1/gpu_connector/xpu_connectors.py +916 -0
- lmcache/v1/health_monitor/__init__.py +1 -0
- lmcache/v1/health_monitor/base.py +587 -0
- lmcache/v1/health_monitor/checks/__init__.py +1 -0
- lmcache/v1/health_monitor/checks/remote_backend_check.py +304 -0
- lmcache/v1/health_monitor/constants.py +36 -0
- lmcache/v1/internal_api_server/__init__.py +0 -0
- lmcache/v1/internal_api_server/api_registry.py +59 -0
- lmcache/v1/internal_api_server/api_server.py +120 -0
- lmcache/v1/internal_api_server/common/__init__.py +1 -0
- lmcache/v1/internal_api_server/common/env_api.py +22 -0
- lmcache/v1/internal_api_server/common/loglevel_api.py +57 -0
- lmcache/v1/internal_api_server/common/metrics_api.py +29 -0
- lmcache/v1/internal_api_server/common/periodic_thread_api.py +138 -0
- lmcache/v1/internal_api_server/common/run_script_api.py +73 -0
- lmcache/v1/internal_api_server/common/thread_api.py +63 -0
- lmcache/v1/internal_api_server/controller/__init__.py +1 -0
- lmcache/v1/internal_api_server/controller/key_stats_api.py +81 -0
- lmcache/v1/internal_api_server/controller/worker_info_api.py +136 -0
- lmcache/v1/internal_api_server/utils.py +43 -0
- lmcache/v1/internal_api_server/vllm/__init__.py +1 -0
- lmcache/v1/internal_api_server/vllm/backend_api.py +221 -0
- lmcache/v1/internal_api_server/vllm/bypass_api.py +204 -0
- lmcache/v1/internal_api_server/vllm/cache_api.py +895 -0
- lmcache/v1/internal_api_server/vllm/chunk_statistics_api.py +141 -0
- lmcache/v1/internal_api_server/vllm/conf_api.py +147 -0
- lmcache/v1/internal_api_server/vllm/freeze_api.py +172 -0
- lmcache/v1/internal_api_server/vllm/hot_cache_api.py +184 -0
- lmcache/v1/internal_api_server/vllm/inference_api.py +65 -0
- lmcache/v1/internal_api_server/vllm/load_fs_chunks_api.py +320 -0
- lmcache/v1/internal_api_server/vllm/lookup_api.py +145 -0
- lmcache/v1/internal_api_server/vllm/version_api.py +25 -0
- lmcache/v1/kv_layer_groups.py +267 -0
- lmcache/v1/lazy_memory_allocator.py +284 -0
- lmcache/v1/lookup_client/__init__.py +25 -0
- lmcache/v1/lookup_client/abstract_client.py +77 -0
- lmcache/v1/lookup_client/async_lookup_message.py +50 -0
- lmcache/v1/lookup_client/chunk_statistics_lookup_client.py +200 -0
- lmcache/v1/lookup_client/factory.py +251 -0
- lmcache/v1/lookup_client/hit_limit_lookup_client.py +86 -0
- lmcache/v1/lookup_client/lmcache_async_lookup_client.py +407 -0
- lmcache/v1/lookup_client/lmcache_lookup_client.py +285 -0
- lmcache/v1/lookup_client/lmcache_lookup_client_bypass.py +99 -0
- lmcache/v1/lookup_client/mooncake_lookup_client.py +87 -0
- lmcache/v1/lookup_client/record_strategies/__init__.py +77 -0
- lmcache/v1/lookup_client/record_strategies/base.py +327 -0
- lmcache/v1/lookup_client/record_strategies/file_hash.py +130 -0
- lmcache/v1/lookup_client/record_strategies/memory_bloom_filter.py +81 -0
- lmcache/v1/manager.py +539 -0
- lmcache/v1/memory_management.py +2619 -0
- lmcache/v1/metadata.py +114 -0
- lmcache/v1/mp_observability/AGENTS.override.md +21 -0
- lmcache/v1/mp_observability/README.md +204 -0
- lmcache/v1/mp_observability/config.py +340 -0
- lmcache/v1/mp_observability/event.py +100 -0
- lmcache/v1/mp_observability/event_bus.py +313 -0
- lmcache/v1/mp_observability/otel_init.py +145 -0
- lmcache/v1/mp_observability/subscribers/__init__.py +28 -0
- lmcache/v1/mp_observability/subscribers/logging/__init__.py +19 -0
- lmcache/v1/mp_observability/subscribers/logging/l1.py +56 -0
- lmcache/v1/mp_observability/subscribers/logging/l2.py +73 -0
- lmcache/v1/mp_observability/subscribers/logging/lookup_hash.py +209 -0
- lmcache/v1/mp_observability/subscribers/logging/mp_server.py +90 -0
- lmcache/v1/mp_observability/subscribers/logging/sm.py +59 -0
- lmcache/v1/mp_observability/subscribers/metrics/__init__.py +20 -0
- lmcache/v1/mp_observability/subscribers/metrics/l0_lifecycle.py +290 -0
- lmcache/v1/mp_observability/subscribers/metrics/l1.py +55 -0
- lmcache/v1/mp_observability/subscribers/metrics/l1_lifecycle.py +166 -0
- lmcache/v1/mp_observability/subscribers/metrics/l2.py +121 -0
- lmcache/v1/mp_observability/subscribers/metrics/sm.py +69 -0
- lmcache/v1/mp_observability/subscribers/tracing/__init__.py +12 -0
- lmcache/v1/mp_observability/subscribers/tracing/mp_server.py +333 -0
- lmcache/v1/mp_observability/subscribers/tracing/span_registry.py +148 -0
- lmcache/v1/mp_observability/trace/__init__.py +50 -0
- lmcache/v1/mp_observability/trace/codecs.py +255 -0
- lmcache/v1/mp_observability/trace/decorator.py +147 -0
- lmcache/v1/mp_observability/trace/format.py +132 -0
- lmcache/v1/mp_observability/trace/lifecycle.py +83 -0
- lmcache/v1/mp_observability/trace/reader.py +167 -0
- lmcache/v1/mp_observability/trace/recorder.py +300 -0
- lmcache/v1/multiprocess/__init__.py +0 -0
- lmcache/v1/multiprocess/affinity_pool.py +102 -0
- lmcache/v1/multiprocess/blend_server_v2.py +891 -0
- lmcache/v1/multiprocess/config.py +253 -0
- lmcache/v1/multiprocess/custom_types.py +281 -0
- lmcache/v1/multiprocess/futures.py +194 -0
- lmcache/v1/multiprocess/gpu_context.py +511 -0
- lmcache/v1/multiprocess/http_server.py +235 -0
- lmcache/v1/multiprocess/mp_runtime_plugin_launcher.py +130 -0
- lmcache/v1/multiprocess/mq.py +732 -0
- lmcache/v1/multiprocess/protocol.py +86 -0
- lmcache/v1/multiprocess/protocols/README.md +213 -0
- lmcache/v1/multiprocess/protocols/__init__.py +127 -0
- lmcache/v1/multiprocess/protocols/base.py +89 -0
- lmcache/v1/multiprocess/protocols/blend.py +109 -0
- lmcache/v1/multiprocess/protocols/blend_v2.py +57 -0
- lmcache/v1/multiprocess/protocols/controller.py +53 -0
- lmcache/v1/multiprocess/protocols/debug.py +34 -0
- lmcache/v1/multiprocess/protocols/engine.py +146 -0
- lmcache/v1/multiprocess/protocols/observability.py +39 -0
- lmcache/v1/multiprocess/server.py +1134 -0
- lmcache/v1/multiprocess/session.py +190 -0
- lmcache/v1/multiprocess/token_hasher.py +441 -0
- lmcache/v1/offload_server/__init__.py +17 -0
- lmcache/v1/offload_server/abstract_server.py +37 -0
- lmcache/v1/offload_server/message.py +30 -0
- lmcache/v1/offload_server/zmq_server.py +122 -0
- lmcache/v1/periodic_thread.py +579 -0
- lmcache/v1/pin_monitor.py +246 -0
- lmcache/v1/plugin/__init__.py +0 -0
- lmcache/v1/plugin/runtime_plugin_launcher.py +211 -0
- lmcache/v1/protocol.py +317 -0
- lmcache/v1/rpc/__init__.py +17 -0
- lmcache/v1/rpc/transport.py +105 -0
- lmcache/v1/rpc/zmq_transport.py +213 -0
- lmcache/v1/rpc_utils.py +165 -0
- lmcache/v1/server/__init__.py +2 -0
- lmcache/v1/server/__main__.py +170 -0
- lmcache/v1/server/storage_backend/__init__.py +21 -0
- lmcache/v1/server/storage_backend/abstract_backend.py +80 -0
- lmcache/v1/server/storage_backend/local_backend.py +75 -0
- lmcache/v1/server/utils.py +21 -0
- lmcache/v1/standalone/__init__.py +1 -0
- lmcache/v1/standalone/__main__.py +583 -0
- lmcache/v1/standalone/manager.py +80 -0
- lmcache/v1/standalone/standalone_service_factory.py +86 -0
- lmcache/v1/storage_backend/__init__.py +313 -0
- lmcache/v1/storage_backend/abstract_backend.py +445 -0
- lmcache/v1/storage_backend/audit_backend.py +233 -0
- lmcache/v1/storage_backend/batched_message_sender.py +222 -0
- lmcache/v1/storage_backend/cache_policy/__init__.py +45 -0
- lmcache/v1/storage_backend/cache_policy/base_policy.py +87 -0
- lmcache/v1/storage_backend/cache_policy/fifo.py +58 -0
- lmcache/v1/storage_backend/cache_policy/lfu.py +105 -0
- lmcache/v1/storage_backend/cache_policy/lru.py +81 -0
- lmcache/v1/storage_backend/cache_policy/mru.py +61 -0
- lmcache/v1/storage_backend/connector/__init__.py +443 -0
- lmcache/v1/storage_backend/connector/audit_adapter.py +77 -0
- lmcache/v1/storage_backend/connector/audit_connector.py +320 -0
- lmcache/v1/storage_backend/connector/base_connector.py +379 -0
- lmcache/v1/storage_backend/connector/blackhole_adapter.py +21 -0
- lmcache/v1/storage_backend/connector/blackhole_connector.py +37 -0
- lmcache/v1/storage_backend/connector/eic_adapter.py +31 -0
- lmcache/v1/storage_backend/connector/eic_connector.py +757 -0
- lmcache/v1/storage_backend/connector/external_adapter.py +79 -0
- lmcache/v1/storage_backend/connector/fs_adapter.py +51 -0
- lmcache/v1/storage_backend/connector/fs_connector.py +403 -0
- lmcache/v1/storage_backend/connector/infinistore_adapter.py +56 -0
- lmcache/v1/storage_backend/connector/infinistore_connector.py +177 -0
- lmcache/v1/storage_backend/connector/instrumented_connector.py +219 -0
- lmcache/v1/storage_backend/connector/lm_adapter.py +31 -0
- lmcache/v1/storage_backend/connector/lm_connector.py +176 -0
- lmcache/v1/storage_backend/connector/mock_adapter.py +57 -0
- lmcache/v1/storage_backend/connector/mock_connector.py +349 -0
- lmcache/v1/storage_backend/connector/mooncakestore_adapter.py +43 -0
- lmcache/v1/storage_backend/connector/mooncakestore_connector.py +614 -0
- lmcache/v1/storage_backend/connector/redis_adapter.py +181 -0
- lmcache/v1/storage_backend/connector/redis_connector.py +828 -0
- lmcache/v1/storage_backend/connector/s3_adapter.py +59 -0
- lmcache/v1/storage_backend/connector/s3_connector.py +699 -0
- lmcache/v1/storage_backend/connector/sagemaker_hyperpod_adapter.py +233 -0
- lmcache/v1/storage_backend/connector/sagemaker_hyperpod_connector.py +987 -0
- lmcache/v1/storage_backend/connector/valkey_adapter.py +114 -0
- lmcache/v1/storage_backend/connector/valkey_connector.py +627 -0
- lmcache/v1/storage_backend/gds_backend.py +1199 -0
- lmcache/v1/storage_backend/job_executor/__init__.py +0 -0
- lmcache/v1/storage_backend/job_executor/base_executor.py +34 -0
- lmcache/v1/storage_backend/job_executor/pq_executor.py +235 -0
- lmcache/v1/storage_backend/local_cpu_backend.py +810 -0
- lmcache/v1/storage_backend/local_disk_backend.py +656 -0
- lmcache/v1/storage_backend/maru_backend.py +734 -0
- lmcache/v1/storage_backend/naive_serde/__init__.py +50 -0
- lmcache/v1/storage_backend/naive_serde/cachegen_basics.py +133 -0
- lmcache/v1/storage_backend/naive_serde/cachegen_decoder.py +135 -0
- lmcache/v1/storage_backend/naive_serde/cachegen_encoder.py +83 -0
- lmcache/v1/storage_backend/naive_serde/kivi_serde.py +22 -0
- lmcache/v1/storage_backend/naive_serde/naive_serde.py +18 -0
- lmcache/v1/storage_backend/naive_serde/serde.py +37 -0
- lmcache/v1/storage_backend/native_clients/connector_client_base.py +165 -0
- lmcache/v1/storage_backend/native_clients/resp_client.py +35 -0
- lmcache/v1/storage_backend/nixl_storage_backend.py +1400 -0
- lmcache/v1/storage_backend/p2p_backend.py +788 -0
- lmcache/v1/storage_backend/path_sharder.py +117 -0
- lmcache/v1/storage_backend/pd_backend.py +646 -0
- lmcache/v1/storage_backend/plugins/dax_backend.py +1443 -0
- lmcache/v1/storage_backend/plugins/rust_raw_block_backend.py +1361 -0
- lmcache/v1/storage_backend/remote_backend.py +624 -0
- lmcache/v1/storage_backend/resp_client.py +227 -0
- lmcache/v1/storage_backend/storage_backend_listener.py +19 -0
- lmcache/v1/storage_backend/storage_manager.py +1352 -0
- lmcache/v1/system_detection.py +110 -0
- lmcache/v1/token_database.py +551 -0
- lmcache/v1/transfer_channel/__init__.py +83 -0
- lmcache/v1/transfer_channel/abstract.py +285 -0
- lmcache/v1/transfer_channel/mock_memory_channel.py +156 -0
- lmcache/v1/transfer_channel/nixl_channel.py +639 -0
- lmcache/v1/transfer_channel/py_socket_channel.py +260 -0
- lmcache/v1/transfer_channel/transfer_utils.py +63 -0
- lmcache/v1/utils/__init__.py +1 -0
- lmcache/v1/utils/bloom_filter.py +109 -0
- lmcache/v1/utils/cache_utils.py +125 -0
- lmcache_cli-0.4.5.dev0.dist-info/METADATA +185 -0
- lmcache_cli-0.4.5.dev0.dist-info/RECORD +399 -0
- lmcache_cli-0.4.5.dev0.dist-info/WHEEL +5 -0
- lmcache_cli-0.4.5.dev0.dist-info/entry_points.txt +2 -0
- lmcache_cli-0.4.5.dev0.dist-info/licenses/LICENSE +201 -0
- lmcache_cli-0.4.5.dev0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,138 @@
|
|
|
1
|
+
# SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
"""
|
|
3
|
+
API endpoint for monitoring periodic threads.
|
|
4
|
+
"""
|
|
5
|
+
|
|
6
|
+
# Standard
|
|
7
|
+
from typing import Optional
|
|
8
|
+
|
|
9
|
+
# Third Party
|
|
10
|
+
from fastapi import APIRouter, Query
|
|
11
|
+
from starlette.requests import Request
|
|
12
|
+
from starlette.responses import JSONResponse
|
|
13
|
+
|
|
14
|
+
# First Party
|
|
15
|
+
from lmcache.v1.periodic_thread import PeriodicThreadRegistry, ThreadLevel
|
|
16
|
+
|
|
17
|
+
router = APIRouter()
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
@router.get("/periodic-threads")
|
|
21
|
+
async def get_periodic_threads(
|
|
22
|
+
request: Request,
|
|
23
|
+
level: Optional[str] = Query(
|
|
24
|
+
None,
|
|
25
|
+
description="Filter by thread level (critical, high, medium, low)",
|
|
26
|
+
),
|
|
27
|
+
running_only: bool = Query(
|
|
28
|
+
False,
|
|
29
|
+
description="Only show running threads",
|
|
30
|
+
),
|
|
31
|
+
active_only: bool = Query(
|
|
32
|
+
False,
|
|
33
|
+
description="Only show active threads",
|
|
34
|
+
),
|
|
35
|
+
):
|
|
36
|
+
"""
|
|
37
|
+
Get information about registered periodic threads.
|
|
38
|
+
|
|
39
|
+
Returns a summary of all periodic threads including:
|
|
40
|
+
- Total, running, and active counts by level
|
|
41
|
+
- Individual thread status with last run time and summary
|
|
42
|
+
"""
|
|
43
|
+
registry = PeriodicThreadRegistry.get_instance()
|
|
44
|
+
|
|
45
|
+
# Get all threads
|
|
46
|
+
if level:
|
|
47
|
+
try:
|
|
48
|
+
thread_level = ThreadLevel(level.lower())
|
|
49
|
+
threads = registry.get_by_level(thread_level)
|
|
50
|
+
except ValueError:
|
|
51
|
+
return JSONResponse(
|
|
52
|
+
status_code=400,
|
|
53
|
+
content={
|
|
54
|
+
"error": f"Invalid level: {level}. "
|
|
55
|
+
f"Valid values: critical, high, medium, low"
|
|
56
|
+
},
|
|
57
|
+
)
|
|
58
|
+
else:
|
|
59
|
+
threads = registry.get_all()
|
|
60
|
+
|
|
61
|
+
# Apply filters
|
|
62
|
+
if running_only:
|
|
63
|
+
threads = [t for t in threads if t.is_running]
|
|
64
|
+
if active_only:
|
|
65
|
+
threads = [t for t in threads if t.is_active]
|
|
66
|
+
|
|
67
|
+
# Build response
|
|
68
|
+
thread_statuses = [t.get_status() for t in threads]
|
|
69
|
+
|
|
70
|
+
# Get summary
|
|
71
|
+
summary = registry.get_summary()
|
|
72
|
+
|
|
73
|
+
return JSONResponse(
|
|
74
|
+
content={
|
|
75
|
+
"summary": {
|
|
76
|
+
"total_count": summary["total_count"],
|
|
77
|
+
"running_count": summary["running_count"],
|
|
78
|
+
"active_count": summary["active_count"],
|
|
79
|
+
"by_level": summary["by_level"],
|
|
80
|
+
},
|
|
81
|
+
"threads": thread_statuses,
|
|
82
|
+
}
|
|
83
|
+
)
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
@router.get("/periodic-threads/{thread_name}")
|
|
87
|
+
async def get_periodic_thread(
|
|
88
|
+
request: Request,
|
|
89
|
+
thread_name: str,
|
|
90
|
+
):
|
|
91
|
+
"""
|
|
92
|
+
Get detailed information about a specific periodic thread.
|
|
93
|
+
"""
|
|
94
|
+
registry = PeriodicThreadRegistry.get_instance()
|
|
95
|
+
thread = registry.get(thread_name)
|
|
96
|
+
|
|
97
|
+
if thread is None:
|
|
98
|
+
return JSONResponse(
|
|
99
|
+
status_code=404,
|
|
100
|
+
content={"error": f"Thread not found: {thread_name}"},
|
|
101
|
+
)
|
|
102
|
+
|
|
103
|
+
return JSONResponse(content=thread.get_status())
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
@router.get("/periodic-threads-health")
|
|
107
|
+
async def get_periodic_threads_health(request: Request):
|
|
108
|
+
"""
|
|
109
|
+
Quick health check for periodic threads.
|
|
110
|
+
|
|
111
|
+
Returns:
|
|
112
|
+
- healthy: True if all critical/high level threads are active
|
|
113
|
+
- unhealthy_threads: List of inactive critical/high threads
|
|
114
|
+
"""
|
|
115
|
+
registry = PeriodicThreadRegistry.get_instance()
|
|
116
|
+
|
|
117
|
+
unhealthy_threads = []
|
|
118
|
+
|
|
119
|
+
# Check critical and high level threads
|
|
120
|
+
for level in [ThreadLevel.CRITICAL, ThreadLevel.HIGH]:
|
|
121
|
+
for thread in registry.get_by_level(level):
|
|
122
|
+
if thread.is_running and not thread.is_active:
|
|
123
|
+
unhealthy_threads.append(
|
|
124
|
+
{
|
|
125
|
+
"name": thread.name,
|
|
126
|
+
"level": thread.level.value,
|
|
127
|
+
"last_run_ago": thread.get_status().get("last_run_ago"),
|
|
128
|
+
"interval": thread.interval,
|
|
129
|
+
}
|
|
130
|
+
)
|
|
131
|
+
|
|
132
|
+
return JSONResponse(
|
|
133
|
+
content={
|
|
134
|
+
"healthy": len(unhealthy_threads) == 0,
|
|
135
|
+
"unhealthy_count": len(unhealthy_threads),
|
|
136
|
+
"unhealthy_threads": unhealthy_threads,
|
|
137
|
+
}
|
|
138
|
+
)
|
|
@@ -0,0 +1,73 @@
|
|
|
1
|
+
# SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
# Standard
|
|
3
|
+
from typing import Any
|
|
4
|
+
import importlib
|
|
5
|
+
|
|
6
|
+
# Third Party
|
|
7
|
+
from fastapi import APIRouter
|
|
8
|
+
from starlette.requests import Request
|
|
9
|
+
from starlette.responses import PlainTextResponse
|
|
10
|
+
|
|
11
|
+
# First Party
|
|
12
|
+
from lmcache.logging import init_logger
|
|
13
|
+
|
|
14
|
+
logger = init_logger(__name__)
|
|
15
|
+
|
|
16
|
+
router = APIRouter()
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
@router.post("/run_script")
|
|
20
|
+
async def run_script(request: Request):
|
|
21
|
+
form_data = await request.form()
|
|
22
|
+
script_file = form_data.get("script")
|
|
23
|
+
|
|
24
|
+
if not script_file or not hasattr(script_file, "file"):
|
|
25
|
+
return PlainTextResponse("No script file provided", status_code=400)
|
|
26
|
+
|
|
27
|
+
script_content = await script_file.read()
|
|
28
|
+
|
|
29
|
+
try:
|
|
30
|
+
# Get allowed imports from config
|
|
31
|
+
config = request.app.state.lmcache_adapter.config
|
|
32
|
+
allowed_imports = config.script_allowed_imports or []
|
|
33
|
+
|
|
34
|
+
# Pre-import allowed modules
|
|
35
|
+
allowed_modules = {}
|
|
36
|
+
for module_name in allowed_imports:
|
|
37
|
+
try:
|
|
38
|
+
module = importlib.import_module(module_name)
|
|
39
|
+
allowed_modules[module_name] = module
|
|
40
|
+
logger.info(f"Imported allowed module: {module_name}")
|
|
41
|
+
except ImportError as e:
|
|
42
|
+
logger.warning(f"Failed to import module {module_name}: {e}")
|
|
43
|
+
|
|
44
|
+
# Create custom __import__ function that only allows configured modules
|
|
45
|
+
def restricted_import(name, globals=None, locals=None, fromlist=(), level=0):
|
|
46
|
+
if name in allowed_modules:
|
|
47
|
+
return allowed_modules[name]
|
|
48
|
+
raise ImportError(f"Import of '{name}' is not allowed")
|
|
49
|
+
|
|
50
|
+
restricted_globals = {
|
|
51
|
+
"__builtins__": {
|
|
52
|
+
"print": print,
|
|
53
|
+
"str": str,
|
|
54
|
+
"int": int,
|
|
55
|
+
"float": float,
|
|
56
|
+
"list": list,
|
|
57
|
+
"dict": dict,
|
|
58
|
+
"tuple": tuple,
|
|
59
|
+
"set": set,
|
|
60
|
+
"__import__": restricted_import,
|
|
61
|
+
},
|
|
62
|
+
"app": request.app,
|
|
63
|
+
}
|
|
64
|
+
|
|
65
|
+
restricted_locals: dict[str, Any] = {}
|
|
66
|
+
|
|
67
|
+
exec(script_content, restricted_globals, restricted_locals)
|
|
68
|
+
|
|
69
|
+
result = restricted_locals.get("result", "Script executed successfully")
|
|
70
|
+
return PlainTextResponse(str(result), media_type="text/plain")
|
|
71
|
+
|
|
72
|
+
except Exception as e:
|
|
73
|
+
return PlainTextResponse(f"Error executing script: {str(e)}", status_code=500)
|
|
@@ -0,0 +1,63 @@
|
|
|
1
|
+
# SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
# Standard
|
|
3
|
+
from typing import Optional
|
|
4
|
+
import sys
|
|
5
|
+
import threading
|
|
6
|
+
import traceback
|
|
7
|
+
|
|
8
|
+
# Third Party
|
|
9
|
+
from fastapi import APIRouter, Query
|
|
10
|
+
from starlette.requests import Request
|
|
11
|
+
from starlette.responses import PlainTextResponse
|
|
12
|
+
|
|
13
|
+
router = APIRouter()
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
@router.get("/threads")
|
|
17
|
+
async def get_threads(
|
|
18
|
+
request: Request,
|
|
19
|
+
name: Optional[str] = Query(
|
|
20
|
+
None, description="Filter by thread name (fuzzy match)"
|
|
21
|
+
),
|
|
22
|
+
thread_id: Optional[int] = Query(None, description="Filter by thread ID"),
|
|
23
|
+
):
|
|
24
|
+
"""Return information about active threads with optional filtering"""
|
|
25
|
+
threads = threading.enumerate()
|
|
26
|
+
|
|
27
|
+
filtered_threads = []
|
|
28
|
+
for t in threads:
|
|
29
|
+
# Apply filters
|
|
30
|
+
if name and name.lower() not in t.name.lower():
|
|
31
|
+
continue
|
|
32
|
+
if thread_id and t.ident != thread_id:
|
|
33
|
+
continue
|
|
34
|
+
filtered_threads.append(t)
|
|
35
|
+
|
|
36
|
+
thread_info = []
|
|
37
|
+
|
|
38
|
+
for t in filtered_threads:
|
|
39
|
+
# Basic thread info with creation time
|
|
40
|
+
info = f"Thread: {t}\n"
|
|
41
|
+
|
|
42
|
+
# Get stack trace if available
|
|
43
|
+
try:
|
|
44
|
+
stack_frames = (
|
|
45
|
+
sys._current_frames().get(t.ident) if t.ident is not None else None
|
|
46
|
+
)
|
|
47
|
+
if stack_frames:
|
|
48
|
+
stack_trace = traceback.format_stack(stack_frames)
|
|
49
|
+
info += "Stack trace:\n" + "".join(stack_trace)
|
|
50
|
+
else:
|
|
51
|
+
info += "No stack trace available\n"
|
|
52
|
+
except AttributeError:
|
|
53
|
+
info += "Stack trace unavailable\n"
|
|
54
|
+
|
|
55
|
+
thread_info.append(info)
|
|
56
|
+
|
|
57
|
+
# Add summary section
|
|
58
|
+
summary = "\n\n=== Thread Summary ===\n"
|
|
59
|
+
summary += f"Total threads: {len(filtered_threads)}\n"
|
|
60
|
+
|
|
61
|
+
return PlainTextResponse(
|
|
62
|
+
content="\n\n".join(thread_info) + summary, media_type="text/plain"
|
|
63
|
+
)
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
# SPDX-License-Identifier: Apache-2.0
|
|
@@ -0,0 +1,81 @@
|
|
|
1
|
+
# SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
# Standard
|
|
3
|
+
from typing import List
|
|
4
|
+
|
|
5
|
+
# Third Party
|
|
6
|
+
from fastapi import APIRouter, HTTPException, Request
|
|
7
|
+
from pydantic import BaseModel
|
|
8
|
+
|
|
9
|
+
router = APIRouter()
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
class InstanceKeyStats(BaseModel):
|
|
13
|
+
instance_id: str
|
|
14
|
+
key_count: int
|
|
15
|
+
worker_count: int
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
class KeyStatsResponse(BaseModel):
|
|
19
|
+
total_key_count: int
|
|
20
|
+
total_instance_count: int
|
|
21
|
+
total_worker_count: int
|
|
22
|
+
instances: List[InstanceKeyStats]
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
@router.get("/controller/key-stats")
|
|
26
|
+
async def get_key_stats(request: Request):
|
|
27
|
+
"""
|
|
28
|
+
Get key statistics across all instances and workers.
|
|
29
|
+
|
|
30
|
+
Returns:
|
|
31
|
+
- Total key count across all instances
|
|
32
|
+
- Total instance count
|
|
33
|
+
- Total worker count
|
|
34
|
+
- Key count per instance
|
|
35
|
+
"""
|
|
36
|
+
try:
|
|
37
|
+
controller_manager = getattr(
|
|
38
|
+
request.app.state, "lmcache_controller_manager", None
|
|
39
|
+
)
|
|
40
|
+
|
|
41
|
+
if controller_manager is None:
|
|
42
|
+
raise HTTPException(
|
|
43
|
+
status_code=503, detail="Controller manager not available"
|
|
44
|
+
)
|
|
45
|
+
|
|
46
|
+
reg_controller = controller_manager.reg_controller
|
|
47
|
+
registry = reg_controller.registry
|
|
48
|
+
|
|
49
|
+
# Get total key count
|
|
50
|
+
total_key_count = registry.get_total_kv_count()
|
|
51
|
+
|
|
52
|
+
# Get instances and their key counts
|
|
53
|
+
instances = []
|
|
54
|
+
total_instance_count = 0
|
|
55
|
+
total_worker_count = 0
|
|
56
|
+
|
|
57
|
+
for instance_id, instance_node in registry.instances.items():
|
|
58
|
+
total_instance_count += 1
|
|
59
|
+
workers = instance_node.workers.values()
|
|
60
|
+
num_workers = len(workers)
|
|
61
|
+
instance_key_count = sum(w.get_kv_count() for w in workers)
|
|
62
|
+
total_worker_count += num_workers
|
|
63
|
+
instances.append(
|
|
64
|
+
InstanceKeyStats(
|
|
65
|
+
instance_id=instance_id,
|
|
66
|
+
key_count=instance_key_count,
|
|
67
|
+
worker_count=num_workers,
|
|
68
|
+
)
|
|
69
|
+
)
|
|
70
|
+
|
|
71
|
+
return KeyStatsResponse(
|
|
72
|
+
total_key_count=total_key_count,
|
|
73
|
+
total_instance_count=total_instance_count,
|
|
74
|
+
total_worker_count=total_worker_count,
|
|
75
|
+
instances=instances,
|
|
76
|
+
)
|
|
77
|
+
|
|
78
|
+
except HTTPException:
|
|
79
|
+
raise
|
|
80
|
+
except Exception as e:
|
|
81
|
+
raise HTTPException(status_code=500, detail=str(e)) from None
|
|
@@ -0,0 +1,136 @@
|
|
|
1
|
+
# SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
# Standard
|
|
3
|
+
from typing import Annotated, Optional
|
|
4
|
+
|
|
5
|
+
# Third Party
|
|
6
|
+
from fastapi import APIRouter, HTTPException, Query, Request
|
|
7
|
+
from pydantic import BaseModel
|
|
8
|
+
|
|
9
|
+
router = APIRouter()
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
class WorkerInfoResponse(BaseModel):
|
|
13
|
+
instance_id: str
|
|
14
|
+
worker_id: int
|
|
15
|
+
ip: str
|
|
16
|
+
port: int
|
|
17
|
+
peer_init_url: Optional[str]
|
|
18
|
+
registration_time: float
|
|
19
|
+
last_heartbeat_time: float
|
|
20
|
+
key_count: int
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
class WorkerListResponse(BaseModel):
|
|
24
|
+
workers: list[WorkerInfoResponse]
|
|
25
|
+
total_count: int
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
@router.get("/controller/workers")
|
|
29
|
+
async def get_workers(
|
|
30
|
+
request: Request,
|
|
31
|
+
instance_id: Annotated[Optional[str], Query()] = None,
|
|
32
|
+
worker_id: Annotated[Optional[int], Query()] = None,
|
|
33
|
+
):
|
|
34
|
+
"""
|
|
35
|
+
Get worker information with flexible query parameters.
|
|
36
|
+
|
|
37
|
+
- No parameters: List all registered workers across all instances
|
|
38
|
+
- instance_id only: List all workers for a specific instance
|
|
39
|
+
- instance_id and worker_id: Get detailed info about a specific worker
|
|
40
|
+
|
|
41
|
+
Args:
|
|
42
|
+
instance_id: Optional instance ID to filter workers
|
|
43
|
+
worker_id: Optional worker ID to get specific worker details
|
|
44
|
+
"""
|
|
45
|
+
try:
|
|
46
|
+
controller_manager = getattr(
|
|
47
|
+
request.app.state, "lmcache_controller_manager", None
|
|
48
|
+
)
|
|
49
|
+
|
|
50
|
+
if controller_manager is None:
|
|
51
|
+
raise HTTPException(
|
|
52
|
+
status_code=503, detail="Controller manager not available"
|
|
53
|
+
)
|
|
54
|
+
|
|
55
|
+
reg_controller = controller_manager.reg_controller
|
|
56
|
+
|
|
57
|
+
# Case 1: Get specific worker by instance_id and worker_id
|
|
58
|
+
if instance_id is not None and worker_id is not None:
|
|
59
|
+
worker_node = reg_controller.registry.get_worker(instance_id, worker_id)
|
|
60
|
+
if worker_node is None:
|
|
61
|
+
raise HTTPException(
|
|
62
|
+
status_code=404,
|
|
63
|
+
detail=f"Worker ({instance_id}, {worker_id}) not found",
|
|
64
|
+
)
|
|
65
|
+
|
|
66
|
+
worker_info = worker_node.to_worker_info(instance_id)
|
|
67
|
+
key_count = worker_node.get_kv_count()
|
|
68
|
+
return WorkerInfoResponse(
|
|
69
|
+
instance_id=worker_info.instance_id,
|
|
70
|
+
worker_id=worker_info.worker_id,
|
|
71
|
+
ip=worker_info.ip,
|
|
72
|
+
port=worker_info.port,
|
|
73
|
+
peer_init_url=worker_info.peer_init_url,
|
|
74
|
+
registration_time=worker_info.registration_time,
|
|
75
|
+
last_heartbeat_time=worker_info.last_heartbeat_time,
|
|
76
|
+
key_count=key_count,
|
|
77
|
+
)
|
|
78
|
+
|
|
79
|
+
# Case 2: Get all workers for a specific instance
|
|
80
|
+
elif instance_id is not None:
|
|
81
|
+
instance_node = reg_controller.registry.get_instance(instance_id)
|
|
82
|
+
if instance_node is None:
|
|
83
|
+
raise HTTPException(
|
|
84
|
+
status_code=404,
|
|
85
|
+
detail=f"No workers found for instance {instance_id}",
|
|
86
|
+
)
|
|
87
|
+
|
|
88
|
+
worker_infos = instance_node.get_all_worker_infos()
|
|
89
|
+
workers = []
|
|
90
|
+
for worker_info in worker_infos:
|
|
91
|
+
worker_node = reg_controller.registry.get_worker(
|
|
92
|
+
instance_id, worker_info.worker_id
|
|
93
|
+
)
|
|
94
|
+
key_count = worker_node.get_kv_count() if worker_node else 0
|
|
95
|
+
workers.append(
|
|
96
|
+
WorkerInfoResponse(
|
|
97
|
+
instance_id=worker_info.instance_id,
|
|
98
|
+
worker_id=worker_info.worker_id,
|
|
99
|
+
ip=worker_info.ip,
|
|
100
|
+
port=worker_info.port,
|
|
101
|
+
peer_init_url=worker_info.peer_init_url,
|
|
102
|
+
registration_time=worker_info.registration_time,
|
|
103
|
+
last_heartbeat_time=worker_info.last_heartbeat_time,
|
|
104
|
+
key_count=key_count,
|
|
105
|
+
)
|
|
106
|
+
)
|
|
107
|
+
|
|
108
|
+
return WorkerListResponse(workers=workers, total_count=len(workers))
|
|
109
|
+
|
|
110
|
+
# Case 3: Get all workers across all instances
|
|
111
|
+
else:
|
|
112
|
+
worker_infos = reg_controller.registry.get_all_worker_infos_cached()
|
|
113
|
+
workers = []
|
|
114
|
+
for worker_info in worker_infos:
|
|
115
|
+
worker_node = reg_controller.registry.get_worker(
|
|
116
|
+
worker_info.instance_id, worker_info.worker_id
|
|
117
|
+
)
|
|
118
|
+
key_count = worker_node.get_kv_count() if worker_node else 0
|
|
119
|
+
workers.append(
|
|
120
|
+
WorkerInfoResponse(
|
|
121
|
+
instance_id=worker_info.instance_id,
|
|
122
|
+
worker_id=worker_info.worker_id,
|
|
123
|
+
ip=worker_info.ip,
|
|
124
|
+
port=worker_info.port,
|
|
125
|
+
peer_init_url=worker_info.peer_init_url,
|
|
126
|
+
registration_time=worker_info.registration_time,
|
|
127
|
+
last_heartbeat_time=worker_info.last_heartbeat_time,
|
|
128
|
+
key_count=key_count,
|
|
129
|
+
)
|
|
130
|
+
)
|
|
131
|
+
|
|
132
|
+
return WorkerListResponse(workers=workers, total_count=len(workers))
|
|
133
|
+
except HTTPException:
|
|
134
|
+
raise
|
|
135
|
+
except Exception as e:
|
|
136
|
+
raise HTTPException(status_code=500, detail=str(e)) from None
|
|
@@ -0,0 +1,43 @@
|
|
|
1
|
+
# SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
# Standard
|
|
3
|
+
from typing import Dict, List
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
def get_all_server_infos(config, worker_count) -> List[Dict[str, str]]:
|
|
7
|
+
"""
|
|
8
|
+
Generate a list of server information (scheduler and workers) based on the config.
|
|
9
|
+
|
|
10
|
+
Args:
|
|
11
|
+
config: The configuration object containing server details.
|
|
12
|
+
worker_count: The number of worker servers.
|
|
13
|
+
|
|
14
|
+
Returns:
|
|
15
|
+
List[Dict[str, str]]: A JSON list with server information.
|
|
16
|
+
"""
|
|
17
|
+
servers = []
|
|
18
|
+
include_index_list = getattr(config, "internal_api_server_include_index_list", None)
|
|
19
|
+
socket_path_prefix = getattr(config, "internal_api_server_socket_path_prefix", None)
|
|
20
|
+
|
|
21
|
+
# Add scheduler info (index 0)
|
|
22
|
+
if include_index_list is None or 0 in include_index_list:
|
|
23
|
+
port = config.internal_api_server_port_start
|
|
24
|
+
server_info = {
|
|
25
|
+
"name": f"{config.lmcache_instance_id}_scheduler",
|
|
26
|
+
"host": config.internal_api_server_host,
|
|
27
|
+
"port": f"{socket_path_prefix}_{port}" if socket_path_prefix else port,
|
|
28
|
+
}
|
|
29
|
+
servers.append(server_info)
|
|
30
|
+
|
|
31
|
+
# Add workers info (index 1 to worker_count)
|
|
32
|
+
for worker_id in range(worker_count):
|
|
33
|
+
port_offset = 1 + worker_id
|
|
34
|
+
if include_index_list is None or port_offset in include_index_list:
|
|
35
|
+
port = config.internal_api_server_port_start + port_offset
|
|
36
|
+
server_info = {
|
|
37
|
+
"name": f"{config.lmcache_instance_id}_worker{worker_id}",
|
|
38
|
+
"host": config.internal_api_server_host,
|
|
39
|
+
"port": f"{socket_path_prefix}_{port}" if socket_path_prefix else port,
|
|
40
|
+
}
|
|
41
|
+
servers.append(server_info)
|
|
42
|
+
|
|
43
|
+
return servers
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
# SPDX-License-Identifier: Apache-2.0
|