lmcache-cli 0.4.5.dev0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- lmcache/__init__.py +84 -0
- lmcache/_version.py +24 -0
- lmcache/cli/__init__.py +1 -0
- lmcache/cli/commands/__init__.py +34 -0
- lmcache/cli/commands/base.py +157 -0
- lmcache/cli/commands/bench/__init__.py +557 -0
- lmcache/cli/commands/bench/engine_bench/__init__.py +1 -0
- lmcache/cli/commands/bench/engine_bench/config.py +245 -0
- lmcache/cli/commands/bench/engine_bench/interactive/__init__.py +274 -0
- lmcache/cli/commands/bench/engine_bench/interactive/config.json +10 -0
- lmcache/cli/commands/bench/engine_bench/interactive/schema.py +352 -0
- lmcache/cli/commands/bench/engine_bench/interactive/state.py +327 -0
- lmcache/cli/commands/bench/engine_bench/interactive/terminal.py +291 -0
- lmcache/cli/commands/bench/engine_bench/progress.py +145 -0
- lmcache/cli/commands/bench/engine_bench/request_sender.py +232 -0
- lmcache/cli/commands/bench/engine_bench/stats.py +275 -0
- lmcache/cli/commands/bench/engine_bench/workloads/__init__.py +153 -0
- lmcache/cli/commands/bench/engine_bench/workloads/base.py +122 -0
- lmcache/cli/commands/bench/engine_bench/workloads/long_doc_permutator.py +435 -0
- lmcache/cli/commands/bench/engine_bench/workloads/long_doc_qa.py +281 -0
- lmcache/cli/commands/bench/engine_bench/workloads/multi_round_chat.py +337 -0
- lmcache/cli/commands/bench/engine_bench/workloads/random_prefill.py +178 -0
- lmcache/cli/commands/describe.py +310 -0
- lmcache/cli/commands/kvcache.py +133 -0
- lmcache/cli/commands/mock.py +75 -0
- lmcache/cli/commands/ping.py +113 -0
- lmcache/cli/commands/query/__init__.py +155 -0
- lmcache/cli/commands/query/prompt.py +134 -0
- lmcache/cli/commands/query/request.py +357 -0
- lmcache/cli/commands/server.py +99 -0
- lmcache/cli/commands/tool/__init__.py +63 -0
- lmcache/cli/commands/tool/cache_simulator.py +113 -0
- lmcache/cli/commands/trace/__init__.py +505 -0
- lmcache/cli/commands/trace/dispatch.py +249 -0
- lmcache/cli/commands/trace/driver.py +372 -0
- lmcache/cli/commands/trace/stats.py +289 -0
- lmcache/cli/documents/lmcache.txt +11 -0
- lmcache/cli/main.py +42 -0
- lmcache/cli/metrics/__init__.py +29 -0
- lmcache/cli/metrics/formatter.py +171 -0
- lmcache/cli/metrics/handler.py +94 -0
- lmcache/cli/metrics/metrics.py +161 -0
- lmcache/cli/metrics/section.py +77 -0
- lmcache/connections.py +173 -0
- lmcache/integration/__init__.py +2 -0
- lmcache/integration/base_service_factory.py +165 -0
- lmcache/integration/request_telemetry/__init__.py +1 -0
- lmcache/integration/request_telemetry/base.py +51 -0
- lmcache/integration/request_telemetry/factory.py +113 -0
- lmcache/integration/request_telemetry/fastapi.py +109 -0
- lmcache/integration/request_telemetry/noop.py +35 -0
- lmcache/integration/sglang/__init__.py +2 -0
- lmcache/integration/sglang/sglang_adapter.py +326 -0
- lmcache/integration/sglang/utils.py +39 -0
- lmcache/integration/vllm/__init__.py +1 -0
- lmcache/integration/vllm/lmcache_connector_v1.py +213 -0
- lmcache/integration/vllm/lmcache_connector_v1_085.py +150 -0
- lmcache/integration/vllm/lmcache_mp_connector_0180.py +1072 -0
- lmcache/integration/vllm/tests/test_mm_hash_utils.py +112 -0
- lmcache/integration/vllm/utils.py +433 -0
- lmcache/integration/vllm/vllm_multi_process_adapter.py +1090 -0
- lmcache/integration/vllm/vllm_service_factory.py +339 -0
- lmcache/integration/vllm/vllm_v1_adapter.py +1713 -0
- lmcache/logging.py +107 -0
- lmcache/native_storage_ops.pyi +230 -0
- lmcache/non_cuda_equivalents.py +1424 -0
- lmcache/observability.py +1958 -0
- lmcache/storage_backend/serde/__init__.py +1 -0
- lmcache/storage_backend/serde/cachegen_basics.py +210 -0
- lmcache/storage_backend/serde/cachegen_decoder.py +207 -0
- lmcache/storage_backend/serde/cachegen_encoder.py +394 -0
- lmcache/storage_backend/serde/serde.py +75 -0
- lmcache/tools/__init__.py +1 -0
- lmcache/tools/cache_simulator/README.md +392 -0
- lmcache/tools/cache_simulator/__init__.py +1 -0
- lmcache/tools/cache_simulator/docs/simulate_example.png +0 -0
- lmcache/tools/cache_simulator/docs/sweep_example.png +0 -0
- lmcache/tools/cache_simulator/gen_bench_dataset.py +360 -0
- lmcache/tools/cache_simulator/lru_cache.py +124 -0
- lmcache/tools/cache_simulator/plot_hit_rate.py +231 -0
- lmcache/tools/cache_simulator/simulator.py +795 -0
- lmcache/tools/controller_benchmark/README.md +161 -0
- lmcache/tools/controller_benchmark/__init__.py +1 -0
- lmcache/tools/controller_benchmark/__main__.py +331 -0
- lmcache/tools/controller_benchmark/benchmark.py +660 -0
- lmcache/tools/controller_benchmark/config.py +44 -0
- lmcache/tools/controller_benchmark/constants.py +10 -0
- lmcache/tools/controller_benchmark/handlers/__init__.py +46 -0
- lmcache/tools/controller_benchmark/handlers/admit.py +52 -0
- lmcache/tools/controller_benchmark/handlers/base.py +47 -0
- lmcache/tools/controller_benchmark/handlers/deregister.py +49 -0
- lmcache/tools/controller_benchmark/handlers/evict.py +52 -0
- lmcache/tools/controller_benchmark/handlers/heartbeat.py +56 -0
- lmcache/tools/controller_benchmark/handlers/p2p_lookup.py +47 -0
- lmcache/tools/controller_benchmark/handlers/register.py +56 -0
- lmcache/tools/mp_status_viewer/__init__.py +1 -0
- lmcache/tools/mp_status_viewer/__main__.py +95 -0
- lmcache/usage_context.py +417 -0
- lmcache/utils.py +665 -0
- lmcache/v1/__init__.py +2 -0
- lmcache/v1/api_server/__init__.py +2 -0
- lmcache/v1/api_server/__main__.py +537 -0
- lmcache/v1/basic_check.py +112 -0
- lmcache/v1/cache_controller/__init__.py +9 -0
- lmcache/v1/cache_controller/commands/__init__.py +15 -0
- lmcache/v1/cache_controller/commands/base.py +35 -0
- lmcache/v1/cache_controller/commands/full_sync.py +49 -0
- lmcache/v1/cache_controller/config.py +176 -0
- lmcache/v1/cache_controller/controller_manager.py +535 -0
- lmcache/v1/cache_controller/controllers/__init__.py +11 -0
- lmcache/v1/cache_controller/controllers/full_sync_tracker.py +473 -0
- lmcache/v1/cache_controller/controllers/kv_controller.py +439 -0
- lmcache/v1/cache_controller/controllers/registration_controller.py +282 -0
- lmcache/v1/cache_controller/executor.py +463 -0
- lmcache/v1/cache_controller/frontend/static/css/style.css +201 -0
- lmcache/v1/cache_controller/frontend/static/img/logo.png +0 -0
- lmcache/v1/cache_controller/frontend/static/index.html +234 -0
- lmcache/v1/cache_controller/frontend/static/js/controller_app.js +660 -0
- lmcache/v1/cache_controller/full_sync_sender.py +475 -0
- lmcache/v1/cache_controller/locks.py +149 -0
- lmcache/v1/cache_controller/message.py +828 -0
- lmcache/v1/cache_controller/observability.py +208 -0
- lmcache/v1/cache_controller/utils.py +679 -0
- lmcache/v1/cache_controller/worker.py +665 -0
- lmcache/v1/cache_engine.py +2058 -0
- lmcache/v1/cache_interface.py +19 -0
- lmcache/v1/check/__init__.py +74 -0
- lmcache/v1/check/check_mode_gen.py +86 -0
- lmcache/v1/check/check_mode_test_l2_adapter.py +284 -0
- lmcache/v1/check/check_mode_test_remote.py +155 -0
- lmcache/v1/check/check_mode_test_storage_manager.py +142 -0
- lmcache/v1/check/utils.py +571 -0
- lmcache/v1/compute/__init__.py +2 -0
- lmcache/v1/compute/attention/__init__.py +0 -0
- lmcache/v1/compute/attention/abstract.py +39 -0
- lmcache/v1/compute/attention/flash_attn.py +129 -0
- lmcache/v1/compute/attention/flash_infer_sparse.py +284 -0
- lmcache/v1/compute/attention/metadata.py +85 -0
- lmcache/v1/compute/attention/utils.py +14 -0
- lmcache/v1/compute/blend/__init__.py +7 -0
- lmcache/v1/compute/blend/blender.py +168 -0
- lmcache/v1/compute/blend/metadata.py +34 -0
- lmcache/v1/compute/blend/utils.py +63 -0
- lmcache/v1/compute/models/__init__.py +0 -0
- lmcache/v1/compute/models/base.py +141 -0
- lmcache/v1/compute/models/llama.py +9 -0
- lmcache/v1/compute/models/qwen3.py +24 -0
- lmcache/v1/compute/models/utils.py +68 -0
- lmcache/v1/compute/positional_encoding.py +199 -0
- lmcache/v1/config.py +848 -0
- lmcache/v1/config_base.py +848 -0
- lmcache/v1/distributed/api.py +248 -0
- lmcache/v1/distributed/config.py +321 -0
- lmcache/v1/distributed/error.py +64 -0
- lmcache/v1/distributed/eviction.py +192 -0
- lmcache/v1/distributed/eviction_policy/__init__.py +21 -0
- lmcache/v1/distributed/eviction_policy/factory.py +27 -0
- lmcache/v1/distributed/eviction_policy/lru.py +244 -0
- lmcache/v1/distributed/eviction_policy/noop.py +50 -0
- lmcache/v1/distributed/internal_api.py +170 -0
- lmcache/v1/distributed/l1_manager.py +835 -0
- lmcache/v1/distributed/l2_adapters/__init__.py +67 -0
- lmcache/v1/distributed/l2_adapters/base.py +360 -0
- lmcache/v1/distributed/l2_adapters/config.py +385 -0
- lmcache/v1/distributed/l2_adapters/factory.py +205 -0
- lmcache/v1/distributed/l2_adapters/fs_l2_adapter.py +747 -0
- lmcache/v1/distributed/l2_adapters/fs_native_l2_adapter.py +167 -0
- lmcache/v1/distributed/l2_adapters/mock_l2_adapter.py +516 -0
- lmcache/v1/distributed/l2_adapters/mooncake_store_l2_adapter.py +135 -0
- lmcache/v1/distributed/l2_adapters/native_connector_l2_adapter.py +468 -0
- lmcache/v1/distributed/l2_adapters/native_plugin_l2_adapter.py +199 -0
- lmcache/v1/distributed/l2_adapters/nixl_store_dynamic_l2_adapter.py +831 -0
- lmcache/v1/distributed/l2_adapters/nixl_store_l2_adapter.py +983 -0
- lmcache/v1/distributed/l2_adapters/plugin_l2_adapter.py +210 -0
- lmcache/v1/distributed/l2_adapters/resp_l2_adapter.py +176 -0
- lmcache/v1/distributed/memory_manager.py +179 -0
- lmcache/v1/distributed/storage_controller.py +39 -0
- lmcache/v1/distributed/storage_controllers/__init__.py +43 -0
- lmcache/v1/distributed/storage_controllers/eviction_controller.py +242 -0
- lmcache/v1/distributed/storage_controllers/prefetch_controller.py +830 -0
- lmcache/v1/distributed/storage_controllers/prefetch_policy.py +193 -0
- lmcache/v1/distributed/storage_controllers/store_controller.py +452 -0
- lmcache/v1/distributed/storage_controllers/store_policy.py +213 -0
- lmcache/v1/distributed/storage_manager.py +532 -0
- lmcache/v1/event_manager.py +145 -0
- lmcache/v1/exceptions/__init__.py +16 -0
- lmcache/v1/gpu_connector/__init__.py +126 -0
- lmcache/v1/gpu_connector/gpu_connectors.py +1906 -0
- lmcache/v1/gpu_connector/gpu_ops.py +85 -0
- lmcache/v1/gpu_connector/hpu_connector.py +326 -0
- lmcache/v1/gpu_connector/mock_gpu_connector.py +67 -0
- lmcache/v1/gpu_connector/utils.py +890 -0
- lmcache/v1/gpu_connector/xpu_connectors.py +916 -0
- lmcache/v1/health_monitor/__init__.py +1 -0
- lmcache/v1/health_monitor/base.py +587 -0
- lmcache/v1/health_monitor/checks/__init__.py +1 -0
- lmcache/v1/health_monitor/checks/remote_backend_check.py +304 -0
- lmcache/v1/health_monitor/constants.py +36 -0
- lmcache/v1/internal_api_server/__init__.py +0 -0
- lmcache/v1/internal_api_server/api_registry.py +59 -0
- lmcache/v1/internal_api_server/api_server.py +120 -0
- lmcache/v1/internal_api_server/common/__init__.py +1 -0
- lmcache/v1/internal_api_server/common/env_api.py +22 -0
- lmcache/v1/internal_api_server/common/loglevel_api.py +57 -0
- lmcache/v1/internal_api_server/common/metrics_api.py +29 -0
- lmcache/v1/internal_api_server/common/periodic_thread_api.py +138 -0
- lmcache/v1/internal_api_server/common/run_script_api.py +73 -0
- lmcache/v1/internal_api_server/common/thread_api.py +63 -0
- lmcache/v1/internal_api_server/controller/__init__.py +1 -0
- lmcache/v1/internal_api_server/controller/key_stats_api.py +81 -0
- lmcache/v1/internal_api_server/controller/worker_info_api.py +136 -0
- lmcache/v1/internal_api_server/utils.py +43 -0
- lmcache/v1/internal_api_server/vllm/__init__.py +1 -0
- lmcache/v1/internal_api_server/vllm/backend_api.py +221 -0
- lmcache/v1/internal_api_server/vllm/bypass_api.py +204 -0
- lmcache/v1/internal_api_server/vllm/cache_api.py +895 -0
- lmcache/v1/internal_api_server/vllm/chunk_statistics_api.py +141 -0
- lmcache/v1/internal_api_server/vllm/conf_api.py +147 -0
- lmcache/v1/internal_api_server/vllm/freeze_api.py +172 -0
- lmcache/v1/internal_api_server/vllm/hot_cache_api.py +184 -0
- lmcache/v1/internal_api_server/vllm/inference_api.py +65 -0
- lmcache/v1/internal_api_server/vllm/load_fs_chunks_api.py +320 -0
- lmcache/v1/internal_api_server/vllm/lookup_api.py +145 -0
- lmcache/v1/internal_api_server/vllm/version_api.py +25 -0
- lmcache/v1/kv_layer_groups.py +267 -0
- lmcache/v1/lazy_memory_allocator.py +284 -0
- lmcache/v1/lookup_client/__init__.py +25 -0
- lmcache/v1/lookup_client/abstract_client.py +77 -0
- lmcache/v1/lookup_client/async_lookup_message.py +50 -0
- lmcache/v1/lookup_client/chunk_statistics_lookup_client.py +200 -0
- lmcache/v1/lookup_client/factory.py +251 -0
- lmcache/v1/lookup_client/hit_limit_lookup_client.py +86 -0
- lmcache/v1/lookup_client/lmcache_async_lookup_client.py +407 -0
- lmcache/v1/lookup_client/lmcache_lookup_client.py +285 -0
- lmcache/v1/lookup_client/lmcache_lookup_client_bypass.py +99 -0
- lmcache/v1/lookup_client/mooncake_lookup_client.py +87 -0
- lmcache/v1/lookup_client/record_strategies/__init__.py +77 -0
- lmcache/v1/lookup_client/record_strategies/base.py +327 -0
- lmcache/v1/lookup_client/record_strategies/file_hash.py +130 -0
- lmcache/v1/lookup_client/record_strategies/memory_bloom_filter.py +81 -0
- lmcache/v1/manager.py +539 -0
- lmcache/v1/memory_management.py +2619 -0
- lmcache/v1/metadata.py +114 -0
- lmcache/v1/mp_observability/AGENTS.override.md +21 -0
- lmcache/v1/mp_observability/README.md +204 -0
- lmcache/v1/mp_observability/config.py +340 -0
- lmcache/v1/mp_observability/event.py +100 -0
- lmcache/v1/mp_observability/event_bus.py +313 -0
- lmcache/v1/mp_observability/otel_init.py +145 -0
- lmcache/v1/mp_observability/subscribers/__init__.py +28 -0
- lmcache/v1/mp_observability/subscribers/logging/__init__.py +19 -0
- lmcache/v1/mp_observability/subscribers/logging/l1.py +56 -0
- lmcache/v1/mp_observability/subscribers/logging/l2.py +73 -0
- lmcache/v1/mp_observability/subscribers/logging/lookup_hash.py +209 -0
- lmcache/v1/mp_observability/subscribers/logging/mp_server.py +90 -0
- lmcache/v1/mp_observability/subscribers/logging/sm.py +59 -0
- lmcache/v1/mp_observability/subscribers/metrics/__init__.py +20 -0
- lmcache/v1/mp_observability/subscribers/metrics/l0_lifecycle.py +290 -0
- lmcache/v1/mp_observability/subscribers/metrics/l1.py +55 -0
- lmcache/v1/mp_observability/subscribers/metrics/l1_lifecycle.py +166 -0
- lmcache/v1/mp_observability/subscribers/metrics/l2.py +121 -0
- lmcache/v1/mp_observability/subscribers/metrics/sm.py +69 -0
- lmcache/v1/mp_observability/subscribers/tracing/__init__.py +12 -0
- lmcache/v1/mp_observability/subscribers/tracing/mp_server.py +333 -0
- lmcache/v1/mp_observability/subscribers/tracing/span_registry.py +148 -0
- lmcache/v1/mp_observability/trace/__init__.py +50 -0
- lmcache/v1/mp_observability/trace/codecs.py +255 -0
- lmcache/v1/mp_observability/trace/decorator.py +147 -0
- lmcache/v1/mp_observability/trace/format.py +132 -0
- lmcache/v1/mp_observability/trace/lifecycle.py +83 -0
- lmcache/v1/mp_observability/trace/reader.py +167 -0
- lmcache/v1/mp_observability/trace/recorder.py +300 -0
- lmcache/v1/multiprocess/__init__.py +0 -0
- lmcache/v1/multiprocess/affinity_pool.py +102 -0
- lmcache/v1/multiprocess/blend_server_v2.py +891 -0
- lmcache/v1/multiprocess/config.py +253 -0
- lmcache/v1/multiprocess/custom_types.py +281 -0
- lmcache/v1/multiprocess/futures.py +194 -0
- lmcache/v1/multiprocess/gpu_context.py +511 -0
- lmcache/v1/multiprocess/http_server.py +235 -0
- lmcache/v1/multiprocess/mp_runtime_plugin_launcher.py +130 -0
- lmcache/v1/multiprocess/mq.py +732 -0
- lmcache/v1/multiprocess/protocol.py +86 -0
- lmcache/v1/multiprocess/protocols/README.md +213 -0
- lmcache/v1/multiprocess/protocols/__init__.py +127 -0
- lmcache/v1/multiprocess/protocols/base.py +89 -0
- lmcache/v1/multiprocess/protocols/blend.py +109 -0
- lmcache/v1/multiprocess/protocols/blend_v2.py +57 -0
- lmcache/v1/multiprocess/protocols/controller.py +53 -0
- lmcache/v1/multiprocess/protocols/debug.py +34 -0
- lmcache/v1/multiprocess/protocols/engine.py +146 -0
- lmcache/v1/multiprocess/protocols/observability.py +39 -0
- lmcache/v1/multiprocess/server.py +1134 -0
- lmcache/v1/multiprocess/session.py +190 -0
- lmcache/v1/multiprocess/token_hasher.py +441 -0
- lmcache/v1/offload_server/__init__.py +17 -0
- lmcache/v1/offload_server/abstract_server.py +37 -0
- lmcache/v1/offload_server/message.py +30 -0
- lmcache/v1/offload_server/zmq_server.py +122 -0
- lmcache/v1/periodic_thread.py +579 -0
- lmcache/v1/pin_monitor.py +246 -0
- lmcache/v1/plugin/__init__.py +0 -0
- lmcache/v1/plugin/runtime_plugin_launcher.py +211 -0
- lmcache/v1/protocol.py +317 -0
- lmcache/v1/rpc/__init__.py +17 -0
- lmcache/v1/rpc/transport.py +105 -0
- lmcache/v1/rpc/zmq_transport.py +213 -0
- lmcache/v1/rpc_utils.py +165 -0
- lmcache/v1/server/__init__.py +2 -0
- lmcache/v1/server/__main__.py +170 -0
- lmcache/v1/server/storage_backend/__init__.py +21 -0
- lmcache/v1/server/storage_backend/abstract_backend.py +80 -0
- lmcache/v1/server/storage_backend/local_backend.py +75 -0
- lmcache/v1/server/utils.py +21 -0
- lmcache/v1/standalone/__init__.py +1 -0
- lmcache/v1/standalone/__main__.py +583 -0
- lmcache/v1/standalone/manager.py +80 -0
- lmcache/v1/standalone/standalone_service_factory.py +86 -0
- lmcache/v1/storage_backend/__init__.py +313 -0
- lmcache/v1/storage_backend/abstract_backend.py +445 -0
- lmcache/v1/storage_backend/audit_backend.py +233 -0
- lmcache/v1/storage_backend/batched_message_sender.py +222 -0
- lmcache/v1/storage_backend/cache_policy/__init__.py +45 -0
- lmcache/v1/storage_backend/cache_policy/base_policy.py +87 -0
- lmcache/v1/storage_backend/cache_policy/fifo.py +58 -0
- lmcache/v1/storage_backend/cache_policy/lfu.py +105 -0
- lmcache/v1/storage_backend/cache_policy/lru.py +81 -0
- lmcache/v1/storage_backend/cache_policy/mru.py +61 -0
- lmcache/v1/storage_backend/connector/__init__.py +443 -0
- lmcache/v1/storage_backend/connector/audit_adapter.py +77 -0
- lmcache/v1/storage_backend/connector/audit_connector.py +320 -0
- lmcache/v1/storage_backend/connector/base_connector.py +379 -0
- lmcache/v1/storage_backend/connector/blackhole_adapter.py +21 -0
- lmcache/v1/storage_backend/connector/blackhole_connector.py +37 -0
- lmcache/v1/storage_backend/connector/eic_adapter.py +31 -0
- lmcache/v1/storage_backend/connector/eic_connector.py +757 -0
- lmcache/v1/storage_backend/connector/external_adapter.py +79 -0
- lmcache/v1/storage_backend/connector/fs_adapter.py +51 -0
- lmcache/v1/storage_backend/connector/fs_connector.py +403 -0
- lmcache/v1/storage_backend/connector/infinistore_adapter.py +56 -0
- lmcache/v1/storage_backend/connector/infinistore_connector.py +177 -0
- lmcache/v1/storage_backend/connector/instrumented_connector.py +219 -0
- lmcache/v1/storage_backend/connector/lm_adapter.py +31 -0
- lmcache/v1/storage_backend/connector/lm_connector.py +176 -0
- lmcache/v1/storage_backend/connector/mock_adapter.py +57 -0
- lmcache/v1/storage_backend/connector/mock_connector.py +349 -0
- lmcache/v1/storage_backend/connector/mooncakestore_adapter.py +43 -0
- lmcache/v1/storage_backend/connector/mooncakestore_connector.py +614 -0
- lmcache/v1/storage_backend/connector/redis_adapter.py +181 -0
- lmcache/v1/storage_backend/connector/redis_connector.py +828 -0
- lmcache/v1/storage_backend/connector/s3_adapter.py +59 -0
- lmcache/v1/storage_backend/connector/s3_connector.py +699 -0
- lmcache/v1/storage_backend/connector/sagemaker_hyperpod_adapter.py +233 -0
- lmcache/v1/storage_backend/connector/sagemaker_hyperpod_connector.py +987 -0
- lmcache/v1/storage_backend/connector/valkey_adapter.py +114 -0
- lmcache/v1/storage_backend/connector/valkey_connector.py +627 -0
- lmcache/v1/storage_backend/gds_backend.py +1199 -0
- lmcache/v1/storage_backend/job_executor/__init__.py +0 -0
- lmcache/v1/storage_backend/job_executor/base_executor.py +34 -0
- lmcache/v1/storage_backend/job_executor/pq_executor.py +235 -0
- lmcache/v1/storage_backend/local_cpu_backend.py +810 -0
- lmcache/v1/storage_backend/local_disk_backend.py +656 -0
- lmcache/v1/storage_backend/maru_backend.py +734 -0
- lmcache/v1/storage_backend/naive_serde/__init__.py +50 -0
- lmcache/v1/storage_backend/naive_serde/cachegen_basics.py +133 -0
- lmcache/v1/storage_backend/naive_serde/cachegen_decoder.py +135 -0
- lmcache/v1/storage_backend/naive_serde/cachegen_encoder.py +83 -0
- lmcache/v1/storage_backend/naive_serde/kivi_serde.py +22 -0
- lmcache/v1/storage_backend/naive_serde/naive_serde.py +18 -0
- lmcache/v1/storage_backend/naive_serde/serde.py +37 -0
- lmcache/v1/storage_backend/native_clients/connector_client_base.py +165 -0
- lmcache/v1/storage_backend/native_clients/resp_client.py +35 -0
- lmcache/v1/storage_backend/nixl_storage_backend.py +1400 -0
- lmcache/v1/storage_backend/p2p_backend.py +788 -0
- lmcache/v1/storage_backend/path_sharder.py +117 -0
- lmcache/v1/storage_backend/pd_backend.py +646 -0
- lmcache/v1/storage_backend/plugins/dax_backend.py +1443 -0
- lmcache/v1/storage_backend/plugins/rust_raw_block_backend.py +1361 -0
- lmcache/v1/storage_backend/remote_backend.py +624 -0
- lmcache/v1/storage_backend/resp_client.py +227 -0
- lmcache/v1/storage_backend/storage_backend_listener.py +19 -0
- lmcache/v1/storage_backend/storage_manager.py +1352 -0
- lmcache/v1/system_detection.py +110 -0
- lmcache/v1/token_database.py +551 -0
- lmcache/v1/transfer_channel/__init__.py +83 -0
- lmcache/v1/transfer_channel/abstract.py +285 -0
- lmcache/v1/transfer_channel/mock_memory_channel.py +156 -0
- lmcache/v1/transfer_channel/nixl_channel.py +639 -0
- lmcache/v1/transfer_channel/py_socket_channel.py +260 -0
- lmcache/v1/transfer_channel/transfer_utils.py +63 -0
- lmcache/v1/utils/__init__.py +1 -0
- lmcache/v1/utils/bloom_filter.py +109 -0
- lmcache/v1/utils/cache_utils.py +125 -0
- lmcache_cli-0.4.5.dev0.dist-info/METADATA +185 -0
- lmcache_cli-0.4.5.dev0.dist-info/RECORD +399 -0
- lmcache_cli-0.4.5.dev0.dist-info/WHEEL +5 -0
- lmcache_cli-0.4.5.dev0.dist-info/entry_points.txt +2 -0
- lmcache_cli-0.4.5.dev0.dist-info/licenses/LICENSE +201 -0
- lmcache_cli-0.4.5.dev0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,65 @@
|
|
|
1
|
+
# SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
# Standard
|
|
3
|
+
from typing import Optional
|
|
4
|
+
import json
|
|
5
|
+
|
|
6
|
+
# Third Party
|
|
7
|
+
from fastapi import APIRouter
|
|
8
|
+
from starlette.requests import Request
|
|
9
|
+
from starlette.responses import PlainTextResponse
|
|
10
|
+
|
|
11
|
+
router = APIRouter()
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
@router.get("/inference_info")
|
|
15
|
+
async def get_inference_info(request: Request, format: Optional[str] = None):
|
|
16
|
+
"""
|
|
17
|
+
Get inference information including vLLM config and LMCache details
|
|
18
|
+
|
|
19
|
+
Args:
|
|
20
|
+
format: Optional format parameter (currently unused, for future extension)
|
|
21
|
+
|
|
22
|
+
Returns:
|
|
23
|
+
PlainTextResponse: JSON string containing inference information
|
|
24
|
+
"""
|
|
25
|
+
lmcache_adapter = request.app.state.lmcache_adapter
|
|
26
|
+
|
|
27
|
+
try:
|
|
28
|
+
inference_info = lmcache_adapter.get_inference_info()
|
|
29
|
+
return PlainTextResponse(
|
|
30
|
+
content=json.dumps(inference_info, indent=2, default=str),
|
|
31
|
+
media_type="application/json",
|
|
32
|
+
)
|
|
33
|
+
except Exception as e:
|
|
34
|
+
error_info = {"error": "Failed to get inference info", "message": str(e)}
|
|
35
|
+
return PlainTextResponse(
|
|
36
|
+
content=json.dumps(error_info, indent=2),
|
|
37
|
+
media_type="application/json",
|
|
38
|
+
status_code=500,
|
|
39
|
+
)
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
@router.get("/inference_version")
|
|
43
|
+
async def get_inference_version(request: Request):
|
|
44
|
+
"""
|
|
45
|
+
Get vLLM version information
|
|
46
|
+
|
|
47
|
+
Returns:
|
|
48
|
+
PlainTextResponse: vLLM version string
|
|
49
|
+
"""
|
|
50
|
+
lmcache_adapter = request.app.state.lmcache_adapter
|
|
51
|
+
|
|
52
|
+
try:
|
|
53
|
+
version_info = lmcache_adapter.get_inference_version()
|
|
54
|
+
version_response = {"vllm_version": version_info}
|
|
55
|
+
return PlainTextResponse(
|
|
56
|
+
content=json.dumps(version_response, indent=2),
|
|
57
|
+
media_type="application/json",
|
|
58
|
+
)
|
|
59
|
+
except Exception as e:
|
|
60
|
+
error_info = {"error": "Failed to get inference version", "message": str(e)}
|
|
61
|
+
return PlainTextResponse(
|
|
62
|
+
content=json.dumps(error_info, indent=2),
|
|
63
|
+
media_type="application/json",
|
|
64
|
+
status_code=500,
|
|
65
|
+
)
|
|
@@ -0,0 +1,320 @@
|
|
|
1
|
+
# SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
# Standard
|
|
3
|
+
from typing import Dict, List, Optional
|
|
4
|
+
import asyncio
|
|
5
|
+
import json
|
|
6
|
+
|
|
7
|
+
# Third Party
|
|
8
|
+
from fastapi import APIRouter, HTTPException
|
|
9
|
+
from pydantic import BaseModel
|
|
10
|
+
from starlette.requests import Request
|
|
11
|
+
from starlette.responses import PlainTextResponse
|
|
12
|
+
|
|
13
|
+
# First Party
|
|
14
|
+
from lmcache.logging import init_logger
|
|
15
|
+
from lmcache.utils import CacheEngineKey
|
|
16
|
+
from lmcache.v1.config import LMCacheEngineConfig
|
|
17
|
+
from lmcache.v1.storage_backend.remote_backend import RemoteBackend
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
class LoadFSChunksRequest(BaseModel):
|
|
21
|
+
"""Request model for loading FS chunks."""
|
|
22
|
+
|
|
23
|
+
config_path: str
|
|
24
|
+
max_chunks: Optional[int] = None
|
|
25
|
+
max_failed_keys: int = 10
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
class LoadFSChunksResponse(BaseModel):
|
|
29
|
+
"""Response model for load-fs-chunks endpoint."""
|
|
30
|
+
|
|
31
|
+
status: str
|
|
32
|
+
loaded_chunks: int
|
|
33
|
+
total_files: int
|
|
34
|
+
failed_keys: List[str]
|
|
35
|
+
config_path: str
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
class ErrorResponse(BaseModel):
|
|
39
|
+
"""Error response model for load-fs-chunks endpoint."""
|
|
40
|
+
|
|
41
|
+
error: str
|
|
42
|
+
message: str
|
|
43
|
+
config_path: Optional[str] = None
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
router = APIRouter()
|
|
47
|
+
logger = init_logger(__name__)
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
@router.post(
|
|
51
|
+
"/cache/load-fs-chunks",
|
|
52
|
+
summary="Load chunks from FSConnector into hot cache",
|
|
53
|
+
description="""
|
|
54
|
+
Load chunk files from FSConnector storage into LocalCPUBackend's hot cache.
|
|
55
|
+
""",
|
|
56
|
+
responses={
|
|
57
|
+
200: {
|
|
58
|
+
"model": LoadFSChunksResponse,
|
|
59
|
+
"description": "Chunks loaded successfully",
|
|
60
|
+
},
|
|
61
|
+
400: {"model": ErrorResponse, "description": "Invalid configuration file"},
|
|
62
|
+
500: {"model": ErrorResponse, "description": "Internal server error"},
|
|
63
|
+
503: {"model": ErrorResponse, "description": "LMCache engine not configured"},
|
|
64
|
+
},
|
|
65
|
+
tags=["cache-management"],
|
|
66
|
+
)
|
|
67
|
+
async def load_fs_chunks(
|
|
68
|
+
request: Request,
|
|
69
|
+
request_body: LoadFSChunksRequest,
|
|
70
|
+
):
|
|
71
|
+
"""
|
|
72
|
+
Load chunk files from FSConnector into LocalCPUBackend hot cache.
|
|
73
|
+
|
|
74
|
+
This endpoint loads all chunk files from the specified FSConnector directory
|
|
75
|
+
into the LocalCPUBackend's hot cache by:
|
|
76
|
+
1. Loading configuration from the specified config file
|
|
77
|
+
2. Initializing RemoteBackend with FSConnector
|
|
78
|
+
3. Listing all chunk files in the FSConnector directory
|
|
79
|
+
4. Constructing CacheEngineKey from filenames
|
|
80
|
+
5. Loading MemoryObj from files and putting into hot cache
|
|
81
|
+
|
|
82
|
+
Args:
|
|
83
|
+
request: The FastAPI request object containing application state
|
|
84
|
+
request_body: Request body containing:
|
|
85
|
+
- config_path: Path to LMCache engine configuration file
|
|
86
|
+
- max_chunks: Optional limit on number of chunks to load
|
|
87
|
+
- max_failed_keys: Maximum failed keys to report (default: 10)
|
|
88
|
+
|
|
89
|
+
Returns:
|
|
90
|
+
PlainTextResponse: JSON response with loading statistics
|
|
91
|
+
|
|
92
|
+
Raises:
|
|
93
|
+
HTTPException: Various error conditions with appropriate status codes
|
|
94
|
+
|
|
95
|
+
Example Request:
|
|
96
|
+
```bash
|
|
97
|
+
curl -X POST "http://localhost:8000/cache/load-fs-chunks" \
|
|
98
|
+
-H "Content-Type: application/json" \
|
|
99
|
+
-d '{
|
|
100
|
+
"config_path": "/path/to/lmcache.yaml",
|
|
101
|
+
"max_chunks": 100,
|
|
102
|
+
"max_failed_keys": 10
|
|
103
|
+
}'
|
|
104
|
+
```
|
|
105
|
+
|
|
106
|
+
Example Response (Success):
|
|
107
|
+
```json
|
|
108
|
+
{
|
|
109
|
+
"status": "success",
|
|
110
|
+
"loaded_chunks": 95,
|
|
111
|
+
"total_files": 100,
|
|
112
|
+
"failed_keys": ["key1", "key2"],
|
|
113
|
+
"config_path": "/path/to/lmcache.yaml"
|
|
114
|
+
}
|
|
115
|
+
```
|
|
116
|
+
|
|
117
|
+
Example Response (Error):
|
|
118
|
+
```json
|
|
119
|
+
{
|
|
120
|
+
"error": "Failed to load chunks from FSConnector",
|
|
121
|
+
"message": "Configuration file not found",
|
|
122
|
+
"config_path": "/path/to/lmcache.yaml"
|
|
123
|
+
}
|
|
124
|
+
```
|
|
125
|
+
"""
|
|
126
|
+
lmcache_adapter = request.app.state.lmcache_adapter
|
|
127
|
+
lmcache_engine = getattr(lmcache_adapter, "lmcache_engine", None)
|
|
128
|
+
|
|
129
|
+
if not lmcache_engine:
|
|
130
|
+
error_info = {
|
|
131
|
+
"error": "/cache/load-fs-chunks API is unavailable",
|
|
132
|
+
"message": "LMCache engine not configured.",
|
|
133
|
+
}
|
|
134
|
+
return PlainTextResponse(
|
|
135
|
+
content=json.dumps(error_info, indent=2),
|
|
136
|
+
media_type="application/json",
|
|
137
|
+
status_code=503,
|
|
138
|
+
)
|
|
139
|
+
|
|
140
|
+
remote_backend = None
|
|
141
|
+
try:
|
|
142
|
+
config = await _load_config_from_file(request_body.config_path)
|
|
143
|
+
local_cpu_backend = lmcache_engine.storage_manager.allocator_backend
|
|
144
|
+
|
|
145
|
+
remote_backend = await _initialize_remote_backend(
|
|
146
|
+
config,
|
|
147
|
+
lmcache_engine.metadata,
|
|
148
|
+
local_cpu_backend,
|
|
149
|
+
lmcache_engine.storage_manager.loop,
|
|
150
|
+
)
|
|
151
|
+
|
|
152
|
+
result = await _load_chunks_from_fs_connector(
|
|
153
|
+
remote_backend,
|
|
154
|
+
local_cpu_backend,
|
|
155
|
+
request_body.max_chunks,
|
|
156
|
+
request_body.max_failed_keys,
|
|
157
|
+
)
|
|
158
|
+
|
|
159
|
+
success_info = {
|
|
160
|
+
"status": "success",
|
|
161
|
+
"loaded_chunks": result["loaded_chunks"],
|
|
162
|
+
"total_files": result["total_files"],
|
|
163
|
+
"failed_keys": result["failed_keys"],
|
|
164
|
+
"config_path": request_body.config_path,
|
|
165
|
+
}
|
|
166
|
+
|
|
167
|
+
return PlainTextResponse(
|
|
168
|
+
content=json.dumps(success_info, indent=2),
|
|
169
|
+
media_type="application/json",
|
|
170
|
+
)
|
|
171
|
+
|
|
172
|
+
except Exception as e:
|
|
173
|
+
if isinstance(e, HTTPException):
|
|
174
|
+
raise
|
|
175
|
+
logger.error("Unexpected error in load_fs_chunks: %s", e, exc_info=True)
|
|
176
|
+
error_info = {
|
|
177
|
+
"error": "Failed to load chunks from FSConnector",
|
|
178
|
+
"message": str(e),
|
|
179
|
+
"config_path": request_body.config_path,
|
|
180
|
+
}
|
|
181
|
+
return PlainTextResponse(
|
|
182
|
+
content=json.dumps(error_info, indent=2),
|
|
183
|
+
media_type="application/json",
|
|
184
|
+
status_code=500,
|
|
185
|
+
)
|
|
186
|
+
finally:
|
|
187
|
+
if remote_backend is not None:
|
|
188
|
+
remote_backend.close()
|
|
189
|
+
|
|
190
|
+
|
|
191
|
+
async def _load_config_from_file(config_path: str) -> LMCacheEngineConfig:
|
|
192
|
+
"""Load configuration from yaml file."""
|
|
193
|
+
try:
|
|
194
|
+
return LMCacheEngineConfig.from_file(config_path)
|
|
195
|
+
except Exception as e:
|
|
196
|
+
logger.error("Failed to load config from %s: %s", config_path, e)
|
|
197
|
+
raise HTTPException(
|
|
198
|
+
status_code=400, detail="Invalid configuration file: %s" % str(e)
|
|
199
|
+
) from e
|
|
200
|
+
|
|
201
|
+
|
|
202
|
+
async def _initialize_remote_backend(
|
|
203
|
+
config: LMCacheEngineConfig, metadata, local_cpu_backend, loop
|
|
204
|
+
) -> RemoteBackend:
|
|
205
|
+
"""Initialize RemoteBackend with FSConnector."""
|
|
206
|
+
try:
|
|
207
|
+
remote_backend = RemoteBackend(
|
|
208
|
+
config=config,
|
|
209
|
+
metadata=metadata,
|
|
210
|
+
loop=loop,
|
|
211
|
+
local_cpu_backend=local_cpu_backend,
|
|
212
|
+
dst_device="cpu",
|
|
213
|
+
)
|
|
214
|
+
remote_backend.init_connection()
|
|
215
|
+
return remote_backend
|
|
216
|
+
except Exception as e:
|
|
217
|
+
logger.error("Failed to initialize RemoteBackend: %s", e)
|
|
218
|
+
raise HTTPException(
|
|
219
|
+
status_code=500, detail="Failed to initialize RemoteBackend: %s" % str(e)
|
|
220
|
+
) from e
|
|
221
|
+
|
|
222
|
+
|
|
223
|
+
async def _load_chunks_from_fs_connector(
|
|
224
|
+
remote_backend: RemoteBackend,
|
|
225
|
+
local_cpu_backend,
|
|
226
|
+
max_chunks: Optional[int] = None,
|
|
227
|
+
max_failed_keys: int = 10,
|
|
228
|
+
) -> Dict:
|
|
229
|
+
"""Load chunks from FSConnector into LocalCPUBackend."""
|
|
230
|
+
connector = remote_backend.connection
|
|
231
|
+
if not connector:
|
|
232
|
+
raise HTTPException(status_code=500, detail="FSConnector not initialized")
|
|
233
|
+
|
|
234
|
+
try:
|
|
235
|
+
chunk_files = await connector.list()
|
|
236
|
+
total_files = len(chunk_files)
|
|
237
|
+
|
|
238
|
+
if max_chunks:
|
|
239
|
+
chunk_files = chunk_files[:max_chunks]
|
|
240
|
+
|
|
241
|
+
logger.info("Found %d chunk files to load", len(chunk_files))
|
|
242
|
+
|
|
243
|
+
loaded_chunks = 0
|
|
244
|
+
failed_keys: List[str] = []
|
|
245
|
+
|
|
246
|
+
# Use semaphore to control concurrency (max 10 concurrent tasks)
|
|
247
|
+
semaphore = asyncio.Semaphore(10)
|
|
248
|
+
|
|
249
|
+
async def process_chunk(chunk_filename: str) -> bool:
|
|
250
|
+
"""
|
|
251
|
+
Process a single chunk file from FSConnector.
|
|
252
|
+
|
|
253
|
+
This function is called for each chunk file and performs:
|
|
254
|
+
- Transforms filename to CacheEngineKey format
|
|
255
|
+
- Loads MemoryObj data from remote backend
|
|
256
|
+
- Places chunk into local CPU backend hot cache
|
|
257
|
+
- Handles errors and tracks failed keys
|
|
258
|
+
|
|
259
|
+
Args:
|
|
260
|
+
chunk_filename: Name of the chunk file from FSConnector
|
|
261
|
+
|
|
262
|
+
Returns:
|
|
263
|
+
bool: True if chunk was successfully loaded, False otherwise
|
|
264
|
+
|
|
265
|
+
Note:
|
|
266
|
+
This function runs with semaphore control to limit concurrency
|
|
267
|
+
and ensure system stability during bulk loading operations.
|
|
268
|
+
"""
|
|
269
|
+
async with semaphore:
|
|
270
|
+
key_str = chunk_filename.replace("-SEP-", "/")
|
|
271
|
+
try:
|
|
272
|
+
key = CacheEngineKey.from_string(key_str)
|
|
273
|
+
|
|
274
|
+
# Get data from remote backend
|
|
275
|
+
memory_obj = await asyncio.get_event_loop().run_in_executor(
|
|
276
|
+
None, remote_backend.get_blocking, key
|
|
277
|
+
)
|
|
278
|
+
|
|
279
|
+
if memory_obj is None:
|
|
280
|
+
failed_keys.append(key_str)
|
|
281
|
+
logger.warning("Failed to load chunk: %s", key_str)
|
|
282
|
+
return False
|
|
283
|
+
|
|
284
|
+
# Put into local cpu backend and immediately release reference
|
|
285
|
+
local_cpu_backend.submit_put_task(key, memory_obj)
|
|
286
|
+
memory_obj.ref_count_down()
|
|
287
|
+
return True
|
|
288
|
+
|
|
289
|
+
except Exception as e:
|
|
290
|
+
failed_keys.append(key_str)
|
|
291
|
+
logger.warning("Error processing chunk %s: %s", key_str, e)
|
|
292
|
+
return False
|
|
293
|
+
|
|
294
|
+
# Process all chunks concurrently
|
|
295
|
+
tasks = [process_chunk(chunk_filename) for chunk_filename in chunk_files]
|
|
296
|
+
results = await asyncio.gather(*tasks, return_exceptions=True)
|
|
297
|
+
|
|
298
|
+
# Count successful loads
|
|
299
|
+
loaded_chunks = sum(1 for result in results if result is True)
|
|
300
|
+
|
|
301
|
+
if loaded_chunks > 0 and loaded_chunks % 100 == 0:
|
|
302
|
+
logger.info("Loaded %d chunks...", loaded_chunks)
|
|
303
|
+
|
|
304
|
+
logger.info(
|
|
305
|
+
"Successfully loaded %d chunks from %d files",
|
|
306
|
+
loaded_chunks,
|
|
307
|
+
total_files,
|
|
308
|
+
)
|
|
309
|
+
|
|
310
|
+
return {
|
|
311
|
+
"loaded_chunks": loaded_chunks,
|
|
312
|
+
"total_files": total_files,
|
|
313
|
+
"failed_keys": failed_keys[:max_failed_keys],
|
|
314
|
+
}
|
|
315
|
+
|
|
316
|
+
except Exception as e:
|
|
317
|
+
logger.error("Error in chunk loading process: %s", e)
|
|
318
|
+
raise HTTPException(
|
|
319
|
+
status_code=500, detail="Chunk loading failed: %s" % str(e)
|
|
320
|
+
) from e
|
|
@@ -0,0 +1,145 @@
|
|
|
1
|
+
# SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
# Standard
|
|
3
|
+
import json
|
|
4
|
+
|
|
5
|
+
# Third Party
|
|
6
|
+
from fastapi import APIRouter
|
|
7
|
+
from starlette.requests import Request
|
|
8
|
+
from starlette.responses import PlainTextResponse
|
|
9
|
+
|
|
10
|
+
router = APIRouter()
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
def _json_response(data: dict, status_code: int = 200) -> PlainTextResponse:
|
|
14
|
+
"""Helper to create JSON response."""
|
|
15
|
+
return PlainTextResponse(
|
|
16
|
+
content=json.dumps(data, indent=2),
|
|
17
|
+
media_type="application/json",
|
|
18
|
+
status_code=status_code,
|
|
19
|
+
)
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def _get_role(adapter) -> str:
|
|
23
|
+
"""Get the role from the manager's engine metadata.
|
|
24
|
+
|
|
25
|
+
Returns "scheduler", "worker", or "unknown".
|
|
26
|
+
"""
|
|
27
|
+
metadata = getattr(adapter, "lmcache_engine_metadata", None)
|
|
28
|
+
if metadata is not None:
|
|
29
|
+
role = getattr(metadata, "role", None)
|
|
30
|
+
if role is not None:
|
|
31
|
+
return role
|
|
32
|
+
return "unknown"
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
@router.get("/lookup/info")
|
|
36
|
+
async def get_lookup_info(request: Request):
|
|
37
|
+
"""
|
|
38
|
+
Get information about the current lookup client and server.
|
|
39
|
+
|
|
40
|
+
Example:
|
|
41
|
+
curl http://localhost:6999/lookup/info
|
|
42
|
+
"""
|
|
43
|
+
adapter = request.app.state.lmcache_adapter
|
|
44
|
+
if not hasattr(adapter, "get_lookup_info"):
|
|
45
|
+
return _json_response({"error": "API unavailable"}, 503)
|
|
46
|
+
return _json_response(adapter.get_lookup_info())
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
@router.post("/lookup/close")
|
|
50
|
+
async def close_lookup(request: Request):
|
|
51
|
+
"""
|
|
52
|
+
Close the current lookup client (scheduler) or server (worker).
|
|
53
|
+
|
|
54
|
+
Example:
|
|
55
|
+
curl -X POST http://localhost:6999/lookup/close
|
|
56
|
+
"""
|
|
57
|
+
adapter = request.app.state.lmcache_adapter
|
|
58
|
+
role = _get_role(adapter)
|
|
59
|
+
|
|
60
|
+
if role == "scheduler":
|
|
61
|
+
if not hasattr(adapter, "close_lookup_client"):
|
|
62
|
+
return _json_response({"error": "API unavailable"}, 503)
|
|
63
|
+
result = adapter.close_lookup_client()
|
|
64
|
+
elif role == "worker":
|
|
65
|
+
if not hasattr(adapter, "close_lookup_server"):
|
|
66
|
+
return _json_response({"error": "API unavailable"}, 503)
|
|
67
|
+
result = adapter.close_lookup_server()
|
|
68
|
+
else:
|
|
69
|
+
return _json_response({"error": "Unknown role"}, 400)
|
|
70
|
+
|
|
71
|
+
result["role"] = role
|
|
72
|
+
return _json_response(result)
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
@router.post("/lookup/create")
|
|
76
|
+
async def create_lookup(request: Request, dryrun: bool = False):
|
|
77
|
+
"""
|
|
78
|
+
Create a new lookup client (scheduler) or server (worker).
|
|
79
|
+
|
|
80
|
+
Args:
|
|
81
|
+
dryrun: If true, only show what would be created without creating it.
|
|
82
|
+
|
|
83
|
+
Example:
|
|
84
|
+
# Actually create
|
|
85
|
+
curl -X POST http://localhost:6999/lookup/create
|
|
86
|
+
|
|
87
|
+
# Dryrun - show what would be created
|
|
88
|
+
curl -X POST "http://localhost:6999/lookup/create?dryrun=true"
|
|
89
|
+
"""
|
|
90
|
+
adapter = request.app.state.lmcache_adapter
|
|
91
|
+
role = _get_role(adapter)
|
|
92
|
+
|
|
93
|
+
if role == "scheduler":
|
|
94
|
+
if not hasattr(adapter, "create_lookup_client"):
|
|
95
|
+
return _json_response({"error": "API unavailable"}, 503)
|
|
96
|
+
result = adapter.create_lookup_client(dryrun=dryrun)
|
|
97
|
+
elif role == "worker":
|
|
98
|
+
if not hasattr(adapter, "create_lookup_server"):
|
|
99
|
+
return _json_response({"error": "API unavailable"}, 503)
|
|
100
|
+
result = adapter.create_lookup_server(dryrun=dryrun)
|
|
101
|
+
else:
|
|
102
|
+
return _json_response({"error": "Unknown role"}, 400)
|
|
103
|
+
|
|
104
|
+
result["role"] = role
|
|
105
|
+
if "error" in result:
|
|
106
|
+
return _json_response(result, 400)
|
|
107
|
+
return _json_response(result)
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
@router.post("/lookup/recreate")
|
|
111
|
+
async def recreate_lookup(request: Request):
|
|
112
|
+
"""
|
|
113
|
+
Recreate the lookup client (scheduler) or server (worker).
|
|
114
|
+
|
|
115
|
+
This is equivalent to calling /lookup/close + /lookup/create.
|
|
116
|
+
|
|
117
|
+
IMPORTANT: Update configuration via /conf API before calling this.
|
|
118
|
+
|
|
119
|
+
Example:
|
|
120
|
+
# Step 1: Update config
|
|
121
|
+
curl -X PUT http://localhost:6999/conf \\
|
|
122
|
+
curl -X POST http://localhost:6999/conf \\
|
|
123
|
+
-d '{"enable_scheduler_bypass_lookup": true}'
|
|
124
|
+
|
|
125
|
+
# Step 2: Recreate
|
|
126
|
+
curl -X POST http://localhost:6999/lookup/recreate
|
|
127
|
+
"""
|
|
128
|
+
adapter = request.app.state.lmcache_adapter
|
|
129
|
+
role = _get_role(adapter)
|
|
130
|
+
|
|
131
|
+
if role == "scheduler":
|
|
132
|
+
if not hasattr(adapter, "recreate_lookup_client"):
|
|
133
|
+
return _json_response({"error": "API unavailable"}, 503)
|
|
134
|
+
result = adapter.recreate_lookup_client()
|
|
135
|
+
elif role == "worker":
|
|
136
|
+
if not hasattr(adapter, "recreate_lookup_server"):
|
|
137
|
+
return _json_response({"error": "API unavailable"}, 503)
|
|
138
|
+
result = adapter.recreate_lookup_server()
|
|
139
|
+
else:
|
|
140
|
+
return _json_response({"error": "Unknown role"}, 400)
|
|
141
|
+
|
|
142
|
+
result["role"] = role
|
|
143
|
+
if "error" in result:
|
|
144
|
+
return _json_response(result, 400)
|
|
145
|
+
return _json_response(result)
|
|
@@ -0,0 +1,25 @@
|
|
|
1
|
+
# SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
# Standard
|
|
3
|
+
|
|
4
|
+
# Third Party
|
|
5
|
+
from fastapi import APIRouter
|
|
6
|
+
|
|
7
|
+
# First Party
|
|
8
|
+
from lmcache import utils
|
|
9
|
+
|
|
10
|
+
router = APIRouter()
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
@router.get("/lmc_version")
|
|
14
|
+
async def get_lmc_version():
|
|
15
|
+
return utils.VERSION
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
@router.get("/commit_id")
|
|
19
|
+
async def get_commit_id():
|
|
20
|
+
return utils.COMMIT_ID
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
@router.get("/version")
|
|
24
|
+
async def get_version():
|
|
25
|
+
return utils.get_version()
|