lmcache-cli 0.4.5.dev0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- lmcache/__init__.py +84 -0
- lmcache/_version.py +24 -0
- lmcache/cli/__init__.py +1 -0
- lmcache/cli/commands/__init__.py +34 -0
- lmcache/cli/commands/base.py +157 -0
- lmcache/cli/commands/bench/__init__.py +557 -0
- lmcache/cli/commands/bench/engine_bench/__init__.py +1 -0
- lmcache/cli/commands/bench/engine_bench/config.py +245 -0
- lmcache/cli/commands/bench/engine_bench/interactive/__init__.py +274 -0
- lmcache/cli/commands/bench/engine_bench/interactive/config.json +10 -0
- lmcache/cli/commands/bench/engine_bench/interactive/schema.py +352 -0
- lmcache/cli/commands/bench/engine_bench/interactive/state.py +327 -0
- lmcache/cli/commands/bench/engine_bench/interactive/terminal.py +291 -0
- lmcache/cli/commands/bench/engine_bench/progress.py +145 -0
- lmcache/cli/commands/bench/engine_bench/request_sender.py +232 -0
- lmcache/cli/commands/bench/engine_bench/stats.py +275 -0
- lmcache/cli/commands/bench/engine_bench/workloads/__init__.py +153 -0
- lmcache/cli/commands/bench/engine_bench/workloads/base.py +122 -0
- lmcache/cli/commands/bench/engine_bench/workloads/long_doc_permutator.py +435 -0
- lmcache/cli/commands/bench/engine_bench/workloads/long_doc_qa.py +281 -0
- lmcache/cli/commands/bench/engine_bench/workloads/multi_round_chat.py +337 -0
- lmcache/cli/commands/bench/engine_bench/workloads/random_prefill.py +178 -0
- lmcache/cli/commands/describe.py +310 -0
- lmcache/cli/commands/kvcache.py +133 -0
- lmcache/cli/commands/mock.py +75 -0
- lmcache/cli/commands/ping.py +113 -0
- lmcache/cli/commands/query/__init__.py +155 -0
- lmcache/cli/commands/query/prompt.py +134 -0
- lmcache/cli/commands/query/request.py +357 -0
- lmcache/cli/commands/server.py +99 -0
- lmcache/cli/commands/tool/__init__.py +63 -0
- lmcache/cli/commands/tool/cache_simulator.py +113 -0
- lmcache/cli/commands/trace/__init__.py +505 -0
- lmcache/cli/commands/trace/dispatch.py +249 -0
- lmcache/cli/commands/trace/driver.py +372 -0
- lmcache/cli/commands/trace/stats.py +289 -0
- lmcache/cli/documents/lmcache.txt +11 -0
- lmcache/cli/main.py +42 -0
- lmcache/cli/metrics/__init__.py +29 -0
- lmcache/cli/metrics/formatter.py +171 -0
- lmcache/cli/metrics/handler.py +94 -0
- lmcache/cli/metrics/metrics.py +161 -0
- lmcache/cli/metrics/section.py +77 -0
- lmcache/connections.py +173 -0
- lmcache/integration/__init__.py +2 -0
- lmcache/integration/base_service_factory.py +165 -0
- lmcache/integration/request_telemetry/__init__.py +1 -0
- lmcache/integration/request_telemetry/base.py +51 -0
- lmcache/integration/request_telemetry/factory.py +113 -0
- lmcache/integration/request_telemetry/fastapi.py +109 -0
- lmcache/integration/request_telemetry/noop.py +35 -0
- lmcache/integration/sglang/__init__.py +2 -0
- lmcache/integration/sglang/sglang_adapter.py +326 -0
- lmcache/integration/sglang/utils.py +39 -0
- lmcache/integration/vllm/__init__.py +1 -0
- lmcache/integration/vllm/lmcache_connector_v1.py +213 -0
- lmcache/integration/vllm/lmcache_connector_v1_085.py +150 -0
- lmcache/integration/vllm/lmcache_mp_connector_0180.py +1072 -0
- lmcache/integration/vllm/tests/test_mm_hash_utils.py +112 -0
- lmcache/integration/vllm/utils.py +433 -0
- lmcache/integration/vllm/vllm_multi_process_adapter.py +1090 -0
- lmcache/integration/vllm/vllm_service_factory.py +339 -0
- lmcache/integration/vllm/vllm_v1_adapter.py +1713 -0
- lmcache/logging.py +107 -0
- lmcache/native_storage_ops.pyi +230 -0
- lmcache/non_cuda_equivalents.py +1424 -0
- lmcache/observability.py +1958 -0
- lmcache/storage_backend/serde/__init__.py +1 -0
- lmcache/storage_backend/serde/cachegen_basics.py +210 -0
- lmcache/storage_backend/serde/cachegen_decoder.py +207 -0
- lmcache/storage_backend/serde/cachegen_encoder.py +394 -0
- lmcache/storage_backend/serde/serde.py +75 -0
- lmcache/tools/__init__.py +1 -0
- lmcache/tools/cache_simulator/README.md +392 -0
- lmcache/tools/cache_simulator/__init__.py +1 -0
- lmcache/tools/cache_simulator/docs/simulate_example.png +0 -0
- lmcache/tools/cache_simulator/docs/sweep_example.png +0 -0
- lmcache/tools/cache_simulator/gen_bench_dataset.py +360 -0
- lmcache/tools/cache_simulator/lru_cache.py +124 -0
- lmcache/tools/cache_simulator/plot_hit_rate.py +231 -0
- lmcache/tools/cache_simulator/simulator.py +795 -0
- lmcache/tools/controller_benchmark/README.md +161 -0
- lmcache/tools/controller_benchmark/__init__.py +1 -0
- lmcache/tools/controller_benchmark/__main__.py +331 -0
- lmcache/tools/controller_benchmark/benchmark.py +660 -0
- lmcache/tools/controller_benchmark/config.py +44 -0
- lmcache/tools/controller_benchmark/constants.py +10 -0
- lmcache/tools/controller_benchmark/handlers/__init__.py +46 -0
- lmcache/tools/controller_benchmark/handlers/admit.py +52 -0
- lmcache/tools/controller_benchmark/handlers/base.py +47 -0
- lmcache/tools/controller_benchmark/handlers/deregister.py +49 -0
- lmcache/tools/controller_benchmark/handlers/evict.py +52 -0
- lmcache/tools/controller_benchmark/handlers/heartbeat.py +56 -0
- lmcache/tools/controller_benchmark/handlers/p2p_lookup.py +47 -0
- lmcache/tools/controller_benchmark/handlers/register.py +56 -0
- lmcache/tools/mp_status_viewer/__init__.py +1 -0
- lmcache/tools/mp_status_viewer/__main__.py +95 -0
- lmcache/usage_context.py +417 -0
- lmcache/utils.py +665 -0
- lmcache/v1/__init__.py +2 -0
- lmcache/v1/api_server/__init__.py +2 -0
- lmcache/v1/api_server/__main__.py +537 -0
- lmcache/v1/basic_check.py +112 -0
- lmcache/v1/cache_controller/__init__.py +9 -0
- lmcache/v1/cache_controller/commands/__init__.py +15 -0
- lmcache/v1/cache_controller/commands/base.py +35 -0
- lmcache/v1/cache_controller/commands/full_sync.py +49 -0
- lmcache/v1/cache_controller/config.py +176 -0
- lmcache/v1/cache_controller/controller_manager.py +535 -0
- lmcache/v1/cache_controller/controllers/__init__.py +11 -0
- lmcache/v1/cache_controller/controllers/full_sync_tracker.py +473 -0
- lmcache/v1/cache_controller/controllers/kv_controller.py +439 -0
- lmcache/v1/cache_controller/controllers/registration_controller.py +282 -0
- lmcache/v1/cache_controller/executor.py +463 -0
- lmcache/v1/cache_controller/frontend/static/css/style.css +201 -0
- lmcache/v1/cache_controller/frontend/static/img/logo.png +0 -0
- lmcache/v1/cache_controller/frontend/static/index.html +234 -0
- lmcache/v1/cache_controller/frontend/static/js/controller_app.js +660 -0
- lmcache/v1/cache_controller/full_sync_sender.py +475 -0
- lmcache/v1/cache_controller/locks.py +149 -0
- lmcache/v1/cache_controller/message.py +828 -0
- lmcache/v1/cache_controller/observability.py +208 -0
- lmcache/v1/cache_controller/utils.py +679 -0
- lmcache/v1/cache_controller/worker.py +665 -0
- lmcache/v1/cache_engine.py +2058 -0
- lmcache/v1/cache_interface.py +19 -0
- lmcache/v1/check/__init__.py +74 -0
- lmcache/v1/check/check_mode_gen.py +86 -0
- lmcache/v1/check/check_mode_test_l2_adapter.py +284 -0
- lmcache/v1/check/check_mode_test_remote.py +155 -0
- lmcache/v1/check/check_mode_test_storage_manager.py +142 -0
- lmcache/v1/check/utils.py +571 -0
- lmcache/v1/compute/__init__.py +2 -0
- lmcache/v1/compute/attention/__init__.py +0 -0
- lmcache/v1/compute/attention/abstract.py +39 -0
- lmcache/v1/compute/attention/flash_attn.py +129 -0
- lmcache/v1/compute/attention/flash_infer_sparse.py +284 -0
- lmcache/v1/compute/attention/metadata.py +85 -0
- lmcache/v1/compute/attention/utils.py +14 -0
- lmcache/v1/compute/blend/__init__.py +7 -0
- lmcache/v1/compute/blend/blender.py +168 -0
- lmcache/v1/compute/blend/metadata.py +34 -0
- lmcache/v1/compute/blend/utils.py +63 -0
- lmcache/v1/compute/models/__init__.py +0 -0
- lmcache/v1/compute/models/base.py +141 -0
- lmcache/v1/compute/models/llama.py +9 -0
- lmcache/v1/compute/models/qwen3.py +24 -0
- lmcache/v1/compute/models/utils.py +68 -0
- lmcache/v1/compute/positional_encoding.py +199 -0
- lmcache/v1/config.py +848 -0
- lmcache/v1/config_base.py +848 -0
- lmcache/v1/distributed/api.py +248 -0
- lmcache/v1/distributed/config.py +321 -0
- lmcache/v1/distributed/error.py +64 -0
- lmcache/v1/distributed/eviction.py +192 -0
- lmcache/v1/distributed/eviction_policy/__init__.py +21 -0
- lmcache/v1/distributed/eviction_policy/factory.py +27 -0
- lmcache/v1/distributed/eviction_policy/lru.py +244 -0
- lmcache/v1/distributed/eviction_policy/noop.py +50 -0
- lmcache/v1/distributed/internal_api.py +170 -0
- lmcache/v1/distributed/l1_manager.py +835 -0
- lmcache/v1/distributed/l2_adapters/__init__.py +67 -0
- lmcache/v1/distributed/l2_adapters/base.py +360 -0
- lmcache/v1/distributed/l2_adapters/config.py +385 -0
- lmcache/v1/distributed/l2_adapters/factory.py +205 -0
- lmcache/v1/distributed/l2_adapters/fs_l2_adapter.py +747 -0
- lmcache/v1/distributed/l2_adapters/fs_native_l2_adapter.py +167 -0
- lmcache/v1/distributed/l2_adapters/mock_l2_adapter.py +516 -0
- lmcache/v1/distributed/l2_adapters/mooncake_store_l2_adapter.py +135 -0
- lmcache/v1/distributed/l2_adapters/native_connector_l2_adapter.py +468 -0
- lmcache/v1/distributed/l2_adapters/native_plugin_l2_adapter.py +199 -0
- lmcache/v1/distributed/l2_adapters/nixl_store_dynamic_l2_adapter.py +831 -0
- lmcache/v1/distributed/l2_adapters/nixl_store_l2_adapter.py +983 -0
- lmcache/v1/distributed/l2_adapters/plugin_l2_adapter.py +210 -0
- lmcache/v1/distributed/l2_adapters/resp_l2_adapter.py +176 -0
- lmcache/v1/distributed/memory_manager.py +179 -0
- lmcache/v1/distributed/storage_controller.py +39 -0
- lmcache/v1/distributed/storage_controllers/__init__.py +43 -0
- lmcache/v1/distributed/storage_controllers/eviction_controller.py +242 -0
- lmcache/v1/distributed/storage_controllers/prefetch_controller.py +830 -0
- lmcache/v1/distributed/storage_controllers/prefetch_policy.py +193 -0
- lmcache/v1/distributed/storage_controllers/store_controller.py +452 -0
- lmcache/v1/distributed/storage_controllers/store_policy.py +213 -0
- lmcache/v1/distributed/storage_manager.py +532 -0
- lmcache/v1/event_manager.py +145 -0
- lmcache/v1/exceptions/__init__.py +16 -0
- lmcache/v1/gpu_connector/__init__.py +126 -0
- lmcache/v1/gpu_connector/gpu_connectors.py +1906 -0
- lmcache/v1/gpu_connector/gpu_ops.py +85 -0
- lmcache/v1/gpu_connector/hpu_connector.py +326 -0
- lmcache/v1/gpu_connector/mock_gpu_connector.py +67 -0
- lmcache/v1/gpu_connector/utils.py +890 -0
- lmcache/v1/gpu_connector/xpu_connectors.py +916 -0
- lmcache/v1/health_monitor/__init__.py +1 -0
- lmcache/v1/health_monitor/base.py +587 -0
- lmcache/v1/health_monitor/checks/__init__.py +1 -0
- lmcache/v1/health_monitor/checks/remote_backend_check.py +304 -0
- lmcache/v1/health_monitor/constants.py +36 -0
- lmcache/v1/internal_api_server/__init__.py +0 -0
- lmcache/v1/internal_api_server/api_registry.py +59 -0
- lmcache/v1/internal_api_server/api_server.py +120 -0
- lmcache/v1/internal_api_server/common/__init__.py +1 -0
- lmcache/v1/internal_api_server/common/env_api.py +22 -0
- lmcache/v1/internal_api_server/common/loglevel_api.py +57 -0
- lmcache/v1/internal_api_server/common/metrics_api.py +29 -0
- lmcache/v1/internal_api_server/common/periodic_thread_api.py +138 -0
- lmcache/v1/internal_api_server/common/run_script_api.py +73 -0
- lmcache/v1/internal_api_server/common/thread_api.py +63 -0
- lmcache/v1/internal_api_server/controller/__init__.py +1 -0
- lmcache/v1/internal_api_server/controller/key_stats_api.py +81 -0
- lmcache/v1/internal_api_server/controller/worker_info_api.py +136 -0
- lmcache/v1/internal_api_server/utils.py +43 -0
- lmcache/v1/internal_api_server/vllm/__init__.py +1 -0
- lmcache/v1/internal_api_server/vllm/backend_api.py +221 -0
- lmcache/v1/internal_api_server/vllm/bypass_api.py +204 -0
- lmcache/v1/internal_api_server/vllm/cache_api.py +895 -0
- lmcache/v1/internal_api_server/vllm/chunk_statistics_api.py +141 -0
- lmcache/v1/internal_api_server/vllm/conf_api.py +147 -0
- lmcache/v1/internal_api_server/vllm/freeze_api.py +172 -0
- lmcache/v1/internal_api_server/vllm/hot_cache_api.py +184 -0
- lmcache/v1/internal_api_server/vllm/inference_api.py +65 -0
- lmcache/v1/internal_api_server/vllm/load_fs_chunks_api.py +320 -0
- lmcache/v1/internal_api_server/vllm/lookup_api.py +145 -0
- lmcache/v1/internal_api_server/vllm/version_api.py +25 -0
- lmcache/v1/kv_layer_groups.py +267 -0
- lmcache/v1/lazy_memory_allocator.py +284 -0
- lmcache/v1/lookup_client/__init__.py +25 -0
- lmcache/v1/lookup_client/abstract_client.py +77 -0
- lmcache/v1/lookup_client/async_lookup_message.py +50 -0
- lmcache/v1/lookup_client/chunk_statistics_lookup_client.py +200 -0
- lmcache/v1/lookup_client/factory.py +251 -0
- lmcache/v1/lookup_client/hit_limit_lookup_client.py +86 -0
- lmcache/v1/lookup_client/lmcache_async_lookup_client.py +407 -0
- lmcache/v1/lookup_client/lmcache_lookup_client.py +285 -0
- lmcache/v1/lookup_client/lmcache_lookup_client_bypass.py +99 -0
- lmcache/v1/lookup_client/mooncake_lookup_client.py +87 -0
- lmcache/v1/lookup_client/record_strategies/__init__.py +77 -0
- lmcache/v1/lookup_client/record_strategies/base.py +327 -0
- lmcache/v1/lookup_client/record_strategies/file_hash.py +130 -0
- lmcache/v1/lookup_client/record_strategies/memory_bloom_filter.py +81 -0
- lmcache/v1/manager.py +539 -0
- lmcache/v1/memory_management.py +2619 -0
- lmcache/v1/metadata.py +114 -0
- lmcache/v1/mp_observability/AGENTS.override.md +21 -0
- lmcache/v1/mp_observability/README.md +204 -0
- lmcache/v1/mp_observability/config.py +340 -0
- lmcache/v1/mp_observability/event.py +100 -0
- lmcache/v1/mp_observability/event_bus.py +313 -0
- lmcache/v1/mp_observability/otel_init.py +145 -0
- lmcache/v1/mp_observability/subscribers/__init__.py +28 -0
- lmcache/v1/mp_observability/subscribers/logging/__init__.py +19 -0
- lmcache/v1/mp_observability/subscribers/logging/l1.py +56 -0
- lmcache/v1/mp_observability/subscribers/logging/l2.py +73 -0
- lmcache/v1/mp_observability/subscribers/logging/lookup_hash.py +209 -0
- lmcache/v1/mp_observability/subscribers/logging/mp_server.py +90 -0
- lmcache/v1/mp_observability/subscribers/logging/sm.py +59 -0
- lmcache/v1/mp_observability/subscribers/metrics/__init__.py +20 -0
- lmcache/v1/mp_observability/subscribers/metrics/l0_lifecycle.py +290 -0
- lmcache/v1/mp_observability/subscribers/metrics/l1.py +55 -0
- lmcache/v1/mp_observability/subscribers/metrics/l1_lifecycle.py +166 -0
- lmcache/v1/mp_observability/subscribers/metrics/l2.py +121 -0
- lmcache/v1/mp_observability/subscribers/metrics/sm.py +69 -0
- lmcache/v1/mp_observability/subscribers/tracing/__init__.py +12 -0
- lmcache/v1/mp_observability/subscribers/tracing/mp_server.py +333 -0
- lmcache/v1/mp_observability/subscribers/tracing/span_registry.py +148 -0
- lmcache/v1/mp_observability/trace/__init__.py +50 -0
- lmcache/v1/mp_observability/trace/codecs.py +255 -0
- lmcache/v1/mp_observability/trace/decorator.py +147 -0
- lmcache/v1/mp_observability/trace/format.py +132 -0
- lmcache/v1/mp_observability/trace/lifecycle.py +83 -0
- lmcache/v1/mp_observability/trace/reader.py +167 -0
- lmcache/v1/mp_observability/trace/recorder.py +300 -0
- lmcache/v1/multiprocess/__init__.py +0 -0
- lmcache/v1/multiprocess/affinity_pool.py +102 -0
- lmcache/v1/multiprocess/blend_server_v2.py +891 -0
- lmcache/v1/multiprocess/config.py +253 -0
- lmcache/v1/multiprocess/custom_types.py +281 -0
- lmcache/v1/multiprocess/futures.py +194 -0
- lmcache/v1/multiprocess/gpu_context.py +511 -0
- lmcache/v1/multiprocess/http_server.py +235 -0
- lmcache/v1/multiprocess/mp_runtime_plugin_launcher.py +130 -0
- lmcache/v1/multiprocess/mq.py +732 -0
- lmcache/v1/multiprocess/protocol.py +86 -0
- lmcache/v1/multiprocess/protocols/README.md +213 -0
- lmcache/v1/multiprocess/protocols/__init__.py +127 -0
- lmcache/v1/multiprocess/protocols/base.py +89 -0
- lmcache/v1/multiprocess/protocols/blend.py +109 -0
- lmcache/v1/multiprocess/protocols/blend_v2.py +57 -0
- lmcache/v1/multiprocess/protocols/controller.py +53 -0
- lmcache/v1/multiprocess/protocols/debug.py +34 -0
- lmcache/v1/multiprocess/protocols/engine.py +146 -0
- lmcache/v1/multiprocess/protocols/observability.py +39 -0
- lmcache/v1/multiprocess/server.py +1134 -0
- lmcache/v1/multiprocess/session.py +190 -0
- lmcache/v1/multiprocess/token_hasher.py +441 -0
- lmcache/v1/offload_server/__init__.py +17 -0
- lmcache/v1/offload_server/abstract_server.py +37 -0
- lmcache/v1/offload_server/message.py +30 -0
- lmcache/v1/offload_server/zmq_server.py +122 -0
- lmcache/v1/periodic_thread.py +579 -0
- lmcache/v1/pin_monitor.py +246 -0
- lmcache/v1/plugin/__init__.py +0 -0
- lmcache/v1/plugin/runtime_plugin_launcher.py +211 -0
- lmcache/v1/protocol.py +317 -0
- lmcache/v1/rpc/__init__.py +17 -0
- lmcache/v1/rpc/transport.py +105 -0
- lmcache/v1/rpc/zmq_transport.py +213 -0
- lmcache/v1/rpc_utils.py +165 -0
- lmcache/v1/server/__init__.py +2 -0
- lmcache/v1/server/__main__.py +170 -0
- lmcache/v1/server/storage_backend/__init__.py +21 -0
- lmcache/v1/server/storage_backend/abstract_backend.py +80 -0
- lmcache/v1/server/storage_backend/local_backend.py +75 -0
- lmcache/v1/server/utils.py +21 -0
- lmcache/v1/standalone/__init__.py +1 -0
- lmcache/v1/standalone/__main__.py +583 -0
- lmcache/v1/standalone/manager.py +80 -0
- lmcache/v1/standalone/standalone_service_factory.py +86 -0
- lmcache/v1/storage_backend/__init__.py +313 -0
- lmcache/v1/storage_backend/abstract_backend.py +445 -0
- lmcache/v1/storage_backend/audit_backend.py +233 -0
- lmcache/v1/storage_backend/batched_message_sender.py +222 -0
- lmcache/v1/storage_backend/cache_policy/__init__.py +45 -0
- lmcache/v1/storage_backend/cache_policy/base_policy.py +87 -0
- lmcache/v1/storage_backend/cache_policy/fifo.py +58 -0
- lmcache/v1/storage_backend/cache_policy/lfu.py +105 -0
- lmcache/v1/storage_backend/cache_policy/lru.py +81 -0
- lmcache/v1/storage_backend/cache_policy/mru.py +61 -0
- lmcache/v1/storage_backend/connector/__init__.py +443 -0
- lmcache/v1/storage_backend/connector/audit_adapter.py +77 -0
- lmcache/v1/storage_backend/connector/audit_connector.py +320 -0
- lmcache/v1/storage_backend/connector/base_connector.py +379 -0
- lmcache/v1/storage_backend/connector/blackhole_adapter.py +21 -0
- lmcache/v1/storage_backend/connector/blackhole_connector.py +37 -0
- lmcache/v1/storage_backend/connector/eic_adapter.py +31 -0
- lmcache/v1/storage_backend/connector/eic_connector.py +757 -0
- lmcache/v1/storage_backend/connector/external_adapter.py +79 -0
- lmcache/v1/storage_backend/connector/fs_adapter.py +51 -0
- lmcache/v1/storage_backend/connector/fs_connector.py +403 -0
- lmcache/v1/storage_backend/connector/infinistore_adapter.py +56 -0
- lmcache/v1/storage_backend/connector/infinistore_connector.py +177 -0
- lmcache/v1/storage_backend/connector/instrumented_connector.py +219 -0
- lmcache/v1/storage_backend/connector/lm_adapter.py +31 -0
- lmcache/v1/storage_backend/connector/lm_connector.py +176 -0
- lmcache/v1/storage_backend/connector/mock_adapter.py +57 -0
- lmcache/v1/storage_backend/connector/mock_connector.py +349 -0
- lmcache/v1/storage_backend/connector/mooncakestore_adapter.py +43 -0
- lmcache/v1/storage_backend/connector/mooncakestore_connector.py +614 -0
- lmcache/v1/storage_backend/connector/redis_adapter.py +181 -0
- lmcache/v1/storage_backend/connector/redis_connector.py +828 -0
- lmcache/v1/storage_backend/connector/s3_adapter.py +59 -0
- lmcache/v1/storage_backend/connector/s3_connector.py +699 -0
- lmcache/v1/storage_backend/connector/sagemaker_hyperpod_adapter.py +233 -0
- lmcache/v1/storage_backend/connector/sagemaker_hyperpod_connector.py +987 -0
- lmcache/v1/storage_backend/connector/valkey_adapter.py +114 -0
- lmcache/v1/storage_backend/connector/valkey_connector.py +627 -0
- lmcache/v1/storage_backend/gds_backend.py +1199 -0
- lmcache/v1/storage_backend/job_executor/__init__.py +0 -0
- lmcache/v1/storage_backend/job_executor/base_executor.py +34 -0
- lmcache/v1/storage_backend/job_executor/pq_executor.py +235 -0
- lmcache/v1/storage_backend/local_cpu_backend.py +810 -0
- lmcache/v1/storage_backend/local_disk_backend.py +656 -0
- lmcache/v1/storage_backend/maru_backend.py +734 -0
- lmcache/v1/storage_backend/naive_serde/__init__.py +50 -0
- lmcache/v1/storage_backend/naive_serde/cachegen_basics.py +133 -0
- lmcache/v1/storage_backend/naive_serde/cachegen_decoder.py +135 -0
- lmcache/v1/storage_backend/naive_serde/cachegen_encoder.py +83 -0
- lmcache/v1/storage_backend/naive_serde/kivi_serde.py +22 -0
- lmcache/v1/storage_backend/naive_serde/naive_serde.py +18 -0
- lmcache/v1/storage_backend/naive_serde/serde.py +37 -0
- lmcache/v1/storage_backend/native_clients/connector_client_base.py +165 -0
- lmcache/v1/storage_backend/native_clients/resp_client.py +35 -0
- lmcache/v1/storage_backend/nixl_storage_backend.py +1400 -0
- lmcache/v1/storage_backend/p2p_backend.py +788 -0
- lmcache/v1/storage_backend/path_sharder.py +117 -0
- lmcache/v1/storage_backend/pd_backend.py +646 -0
- lmcache/v1/storage_backend/plugins/dax_backend.py +1443 -0
- lmcache/v1/storage_backend/plugins/rust_raw_block_backend.py +1361 -0
- lmcache/v1/storage_backend/remote_backend.py +624 -0
- lmcache/v1/storage_backend/resp_client.py +227 -0
- lmcache/v1/storage_backend/storage_backend_listener.py +19 -0
- lmcache/v1/storage_backend/storage_manager.py +1352 -0
- lmcache/v1/system_detection.py +110 -0
- lmcache/v1/token_database.py +551 -0
- lmcache/v1/transfer_channel/__init__.py +83 -0
- lmcache/v1/transfer_channel/abstract.py +285 -0
- lmcache/v1/transfer_channel/mock_memory_channel.py +156 -0
- lmcache/v1/transfer_channel/nixl_channel.py +639 -0
- lmcache/v1/transfer_channel/py_socket_channel.py +260 -0
- lmcache/v1/transfer_channel/transfer_utils.py +63 -0
- lmcache/v1/utils/__init__.py +1 -0
- lmcache/v1/utils/bloom_filter.py +109 -0
- lmcache/v1/utils/cache_utils.py +125 -0
- lmcache_cli-0.4.5.dev0.dist-info/METADATA +185 -0
- lmcache_cli-0.4.5.dev0.dist-info/RECORD +399 -0
- lmcache_cli-0.4.5.dev0.dist-info/WHEEL +5 -0
- lmcache_cli-0.4.5.dev0.dist-info/entry_points.txt +2 -0
- lmcache_cli-0.4.5.dev0.dist-info/licenses/LICENSE +201 -0
- lmcache_cli-0.4.5.dev0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,177 @@
|
|
|
1
|
+
# SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
# Standard
|
|
3
|
+
from functools import reduce
|
|
4
|
+
from typing import List, Optional, Union, no_type_check
|
|
5
|
+
import asyncio
|
|
6
|
+
import ctypes
|
|
7
|
+
import operator
|
|
8
|
+
|
|
9
|
+
# Third Party
|
|
10
|
+
import infinistore
|
|
11
|
+
import torch
|
|
12
|
+
|
|
13
|
+
# First Party
|
|
14
|
+
from lmcache.logging import init_logger
|
|
15
|
+
from lmcache.utils import CacheEngineKey
|
|
16
|
+
from lmcache.v1.memory_management import MemoryObj
|
|
17
|
+
|
|
18
|
+
# reuse
|
|
19
|
+
from lmcache.v1.protocol import RemoteMetadata
|
|
20
|
+
from lmcache.v1.storage_backend.connector.base_connector import RemoteConnector
|
|
21
|
+
from lmcache.v1.storage_backend.local_cpu_backend import LocalCPUBackend
|
|
22
|
+
|
|
23
|
+
logger = init_logger(__name__)
|
|
24
|
+
|
|
25
|
+
MAX_BUFFER_SIZE = 40 << 20 # 40MB
|
|
26
|
+
METADATA_BYTES_LEN = 28
|
|
27
|
+
MAX_BUFFER_CNT = 16
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def _get_ptr(mv: Union[bytearray, memoryview]) -> int:
|
|
31
|
+
return ctypes.addressof(ctypes.c_char.from_buffer(mv))
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
class InfinistoreConnector(RemoteConnector):
|
|
35
|
+
def __init__(
|
|
36
|
+
self,
|
|
37
|
+
host: str,
|
|
38
|
+
port: int,
|
|
39
|
+
dev_name: str,
|
|
40
|
+
link_type: str,
|
|
41
|
+
loop: asyncio.AbstractEventLoop,
|
|
42
|
+
memory_allocator: LocalCPUBackend,
|
|
43
|
+
):
|
|
44
|
+
# initialize base class, which includes some common attributes
|
|
45
|
+
super().__init__(memory_allocator.config, memory_allocator.metadata)
|
|
46
|
+
|
|
47
|
+
config = infinistore.ClientConfig(
|
|
48
|
+
host_addr=host,
|
|
49
|
+
service_port=port,
|
|
50
|
+
log_level="info",
|
|
51
|
+
connection_type=infinistore.TYPE_RDMA,
|
|
52
|
+
ib_port=1,
|
|
53
|
+
link_type=link_type,
|
|
54
|
+
dev_name=dev_name,
|
|
55
|
+
)
|
|
56
|
+
|
|
57
|
+
self.rdma_conn = infinistore.InfinityConnection(config)
|
|
58
|
+
|
|
59
|
+
self.loop = loop
|
|
60
|
+
self.rdma_conn.connect()
|
|
61
|
+
|
|
62
|
+
self.send_buffers = []
|
|
63
|
+
self.recv_buffers = []
|
|
64
|
+
self.send_queue: asyncio.Queue[int] = asyncio.Queue(maxsize=MAX_BUFFER_CNT)
|
|
65
|
+
self.recv_queue: asyncio.Queue[int] = asyncio.Queue(maxsize=MAX_BUFFER_CNT)
|
|
66
|
+
|
|
67
|
+
self.buffer_size = MAX_BUFFER_SIZE
|
|
68
|
+
self.memory_allocator = memory_allocator
|
|
69
|
+
|
|
70
|
+
for i in range(MAX_BUFFER_CNT):
|
|
71
|
+
send_buffer = bytearray(self.buffer_size)
|
|
72
|
+
self.rdma_conn.register_mr(_get_ptr(send_buffer), self.buffer_size)
|
|
73
|
+
self.send_buffers.append(send_buffer)
|
|
74
|
+
self.send_queue.put_nowait(i)
|
|
75
|
+
|
|
76
|
+
recv_buffer = bytearray(self.buffer_size)
|
|
77
|
+
self.rdma_conn.register_mr(_get_ptr(recv_buffer), self.buffer_size)
|
|
78
|
+
self.recv_buffers.append(recv_buffer)
|
|
79
|
+
self.recv_queue.put_nowait(i)
|
|
80
|
+
|
|
81
|
+
async def exists(self, key: CacheEngineKey) -> bool:
|
|
82
|
+
def blocking_io():
|
|
83
|
+
return self.rdma_conn.check_exist(key.to_string())
|
|
84
|
+
|
|
85
|
+
return await self.loop.run_in_executor(None, blocking_io)
|
|
86
|
+
|
|
87
|
+
def exists_sync(self, key: CacheEngineKey) -> bool:
|
|
88
|
+
return self.rdma_conn.check_exist(key.to_string())
|
|
89
|
+
|
|
90
|
+
async def get(self, key: CacheEngineKey) -> Optional[MemoryObj]:
|
|
91
|
+
key_str = key.to_string()
|
|
92
|
+
|
|
93
|
+
buf_idx = await self.recv_queue.get()
|
|
94
|
+
buffer = self.recv_buffers[buf_idx]
|
|
95
|
+
try:
|
|
96
|
+
await self.rdma_conn.rdma_read_cache_async(
|
|
97
|
+
[(key_str, 0)], self.buffer_size, _get_ptr(buffer)
|
|
98
|
+
)
|
|
99
|
+
except Exception as e:
|
|
100
|
+
logger.warning(f"get failed: {e}")
|
|
101
|
+
self.recv_queue.put_nowait(buf_idx)
|
|
102
|
+
return None
|
|
103
|
+
|
|
104
|
+
metadata = RemoteMetadata.deserialize(buffer)
|
|
105
|
+
|
|
106
|
+
num_elements = reduce(operator.mul, metadata.shapes[0])
|
|
107
|
+
assert len(metadata.dtypes) == 1
|
|
108
|
+
temp_tensor = torch.frombuffer(
|
|
109
|
+
buffer,
|
|
110
|
+
dtype=metadata.dtypes[0],
|
|
111
|
+
offset=METADATA_BYTES_LEN,
|
|
112
|
+
count=num_elements,
|
|
113
|
+
).reshape(metadata.shapes[0])
|
|
114
|
+
|
|
115
|
+
memory_obj = self.memory_allocator.allocate(
|
|
116
|
+
metadata.shapes[0],
|
|
117
|
+
metadata.dtypes[0],
|
|
118
|
+
metadata.fmt,
|
|
119
|
+
)
|
|
120
|
+
|
|
121
|
+
assert memory_obj is not None
|
|
122
|
+
assert memory_obj.tensor is not None
|
|
123
|
+
|
|
124
|
+
# deep copy to pinned memory
|
|
125
|
+
# and hot cache will reference this memory obj
|
|
126
|
+
memory_obj.tensor.copy_(temp_tensor)
|
|
127
|
+
|
|
128
|
+
logger.debug(f"get key: {key_str} done, {memory_obj.get_shape()}")
|
|
129
|
+
self.recv_queue.put_nowait(buf_idx)
|
|
130
|
+
|
|
131
|
+
return memory_obj
|
|
132
|
+
|
|
133
|
+
async def put(self, key: CacheEngineKey, memory_obj: MemoryObj):
|
|
134
|
+
key_str = key.to_string()
|
|
135
|
+
|
|
136
|
+
kv_bytes = memory_obj.byte_array
|
|
137
|
+
kv_shapes = memory_obj.get_shapes()
|
|
138
|
+
kv_dtypes = memory_obj.get_dtypes()
|
|
139
|
+
memory_format = memory_obj.get_memory_format()
|
|
140
|
+
|
|
141
|
+
buf_idx = await self.send_queue.get()
|
|
142
|
+
buffer = self.send_buffers[buf_idx]
|
|
143
|
+
|
|
144
|
+
RemoteMetadata(
|
|
145
|
+
len(kv_bytes), kv_shapes, kv_dtypes, memory_format
|
|
146
|
+
).serialize_into(buffer)
|
|
147
|
+
|
|
148
|
+
buffer[METADATA_BYTES_LEN : METADATA_BYTES_LEN + len(kv_bytes)] = kv_bytes
|
|
149
|
+
|
|
150
|
+
size = memory_obj.get_physical_size()
|
|
151
|
+
|
|
152
|
+
if size + METADATA_BYTES_LEN > self.buffer_size:
|
|
153
|
+
raise ValueError(
|
|
154
|
+
f"Value size ({size + METADATA_BYTES_LEN} bytes)"
|
|
155
|
+
f"exceeds the maximum allowed size"
|
|
156
|
+
f"({self.buffer_size} bytes). Please decrease chunk_size."
|
|
157
|
+
)
|
|
158
|
+
try:
|
|
159
|
+
await self.rdma_conn.rdma_write_cache_async(
|
|
160
|
+
[(key_str, 0)], METADATA_BYTES_LEN + size, _get_ptr(buffer)
|
|
161
|
+
)
|
|
162
|
+
except Exception as e:
|
|
163
|
+
logger.warning(f"exception happens in rdma_write_cache_async kv_bytes {e}")
|
|
164
|
+
return
|
|
165
|
+
finally:
|
|
166
|
+
self.send_queue.put_nowait(buf_idx)
|
|
167
|
+
|
|
168
|
+
logger.debug(f"put key: {key.to_string()}, {memory_obj.get_shape()}")
|
|
169
|
+
|
|
170
|
+
# TODO
|
|
171
|
+
@no_type_check
|
|
172
|
+
async def list(self) -> List[str]:
|
|
173
|
+
pass
|
|
174
|
+
|
|
175
|
+
async def close(self):
|
|
176
|
+
self.rdma_conn.close()
|
|
177
|
+
logger.info("Closed the infinistore connection")
|
|
@@ -0,0 +1,219 @@
|
|
|
1
|
+
# SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
# Standard
|
|
3
|
+
from typing import List, Optional
|
|
4
|
+
import time
|
|
5
|
+
|
|
6
|
+
# First Party
|
|
7
|
+
from lmcache.logging import init_logger
|
|
8
|
+
from lmcache.observability import LMCStatsMonitor
|
|
9
|
+
from lmcache.utils import CacheEngineKey
|
|
10
|
+
from lmcache.v1.memory_management import MemoryObj
|
|
11
|
+
from lmcache.v1.storage_backend.connector.base_connector import RemoteConnector
|
|
12
|
+
|
|
13
|
+
logger = init_logger(__name__)
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
class InstrumentedRemoteConnector(RemoteConnector):
|
|
17
|
+
"""
|
|
18
|
+
A connector that instruments the underlying connector with
|
|
19
|
+
metrics collection and logging capabilities.
|
|
20
|
+
"""
|
|
21
|
+
|
|
22
|
+
def __init__(self, connector: RemoteConnector):
|
|
23
|
+
self._connector = connector
|
|
24
|
+
self._stats_monitor = LMCStatsMonitor.GetOrCreate()
|
|
25
|
+
self.name = self.__repr__()
|
|
26
|
+
|
|
27
|
+
async def put(self, key: CacheEngineKey, memory_obj: MemoryObj) -> None:
|
|
28
|
+
obj_size = memory_obj.get_size()
|
|
29
|
+
begin = time.perf_counter()
|
|
30
|
+
try:
|
|
31
|
+
await self._connector.put(key, memory_obj)
|
|
32
|
+
finally:
|
|
33
|
+
# Ensure reference count is decreased even if exception occurs
|
|
34
|
+
memory_obj.ref_count_down()
|
|
35
|
+
|
|
36
|
+
end = time.perf_counter()
|
|
37
|
+
self._stats_monitor.update_interval_remote_time_to_put((end - begin) * 1000)
|
|
38
|
+
self._stats_monitor.update_interval_remote_write_metrics(obj_size)
|
|
39
|
+
logger.debug(
|
|
40
|
+
"[%s]Bytes offloaded: %.3f MBytes in %.3f ms",
|
|
41
|
+
self.name,
|
|
42
|
+
obj_size / 1e6,
|
|
43
|
+
(end - begin) * 1000,
|
|
44
|
+
)
|
|
45
|
+
|
|
46
|
+
async def get(self, key: CacheEngineKey) -> Optional[MemoryObj]:
|
|
47
|
+
begin = time.perf_counter()
|
|
48
|
+
memory_obj = await self._connector.get(key)
|
|
49
|
+
end = time.perf_counter()
|
|
50
|
+
duration = end - begin
|
|
51
|
+
|
|
52
|
+
retrieve_stats = self._stats_monitor.get_current_retrieve_stats()
|
|
53
|
+
if (
|
|
54
|
+
retrieve_stats is not None
|
|
55
|
+
and "remote_backend_individual_get_stats" in retrieve_stats.detailed_metrics
|
|
56
|
+
):
|
|
57
|
+
retrieve_stats.detailed_metrics["remote_backend_individual_get_stats"][
|
|
58
|
+
key
|
|
59
|
+
] = {"instrumented_connector_get_time": duration}
|
|
60
|
+
|
|
61
|
+
if memory_obj is not None:
|
|
62
|
+
obj_size = memory_obj.get_size()
|
|
63
|
+
self._stats_monitor.update_interval_remote_read_metrics(obj_size)
|
|
64
|
+
logger.debug(
|
|
65
|
+
"[%s]Bytes loaded: %.3f MBytes in %.3f ms",
|
|
66
|
+
self.name,
|
|
67
|
+
obj_size / 1e6,
|
|
68
|
+
duration * 1000,
|
|
69
|
+
)
|
|
70
|
+
return memory_obj
|
|
71
|
+
|
|
72
|
+
# Delegate all other methods to the underlying connector
|
|
73
|
+
async def exists(self, key: CacheEngineKey) -> bool:
|
|
74
|
+
return await self._connector.exists(key)
|
|
75
|
+
|
|
76
|
+
def exists_sync(self, key: CacheEngineKey) -> bool:
|
|
77
|
+
return self._connector.exists_sync(key)
|
|
78
|
+
|
|
79
|
+
async def list(self) -> List[str]:
|
|
80
|
+
return await self._connector.list()
|
|
81
|
+
|
|
82
|
+
async def close(self) -> None:
|
|
83
|
+
await self._connector.close()
|
|
84
|
+
|
|
85
|
+
def getWrappedConnector(self) -> RemoteConnector:
|
|
86
|
+
return self._connector
|
|
87
|
+
|
|
88
|
+
def support_ping(self) -> bool:
|
|
89
|
+
return self._connector.support_ping()
|
|
90
|
+
|
|
91
|
+
async def ping(self) -> int:
|
|
92
|
+
return await self._connector.ping()
|
|
93
|
+
|
|
94
|
+
def support_batched_put(self) -> bool:
|
|
95
|
+
return self._connector.support_batched_put()
|
|
96
|
+
|
|
97
|
+
def support_batched_get(self) -> bool:
|
|
98
|
+
return self._connector.support_batched_get()
|
|
99
|
+
|
|
100
|
+
def support_batched_async_contains(self) -> bool:
|
|
101
|
+
return self._connector.support_batched_async_contains()
|
|
102
|
+
|
|
103
|
+
async def batched_async_contains(
|
|
104
|
+
self,
|
|
105
|
+
lookup_id: str,
|
|
106
|
+
keys: List[CacheEngineKey],
|
|
107
|
+
pin: bool = False,
|
|
108
|
+
) -> int:
|
|
109
|
+
return await self._connector.batched_async_contains(lookup_id, keys, pin)
|
|
110
|
+
|
|
111
|
+
def support_batched_get_non_blocking(self) -> bool:
|
|
112
|
+
return self._connector.support_batched_get_non_blocking()
|
|
113
|
+
|
|
114
|
+
async def batched_get_non_blocking(
|
|
115
|
+
self,
|
|
116
|
+
lookup_id: str,
|
|
117
|
+
keys: List[CacheEngineKey],
|
|
118
|
+
) -> List[MemoryObj]:
|
|
119
|
+
begin = time.perf_counter()
|
|
120
|
+
memory_objs = await self._connector.batched_get_non_blocking(lookup_id, keys)
|
|
121
|
+
end = time.perf_counter()
|
|
122
|
+
duration = end - begin
|
|
123
|
+
|
|
124
|
+
total_size = sum(
|
|
125
|
+
memory_obj.get_size()
|
|
126
|
+
for memory_obj in memory_objs
|
|
127
|
+
if memory_obj is not None
|
|
128
|
+
)
|
|
129
|
+
if total_size > 0:
|
|
130
|
+
self._stats_monitor.update_interval_remote_read_metrics(total_size)
|
|
131
|
+
logger.debug(
|
|
132
|
+
"[%s]Bytes loaded: %.3f MBytes in %.3f ms",
|
|
133
|
+
self.name,
|
|
134
|
+
total_size / 1e6,
|
|
135
|
+
duration * 1000,
|
|
136
|
+
)
|
|
137
|
+
return memory_objs
|
|
138
|
+
|
|
139
|
+
async def batched_get(
|
|
140
|
+
self, keys: List[CacheEngineKey]
|
|
141
|
+
) -> List[Optional[MemoryObj]]:
|
|
142
|
+
begin = time.perf_counter()
|
|
143
|
+
memory_objs = await self._connector.batched_get(keys)
|
|
144
|
+
end = time.perf_counter()
|
|
145
|
+
duration = end - begin
|
|
146
|
+
self._stats_monitor.update_interval_remote_time_to_get(duration * 1000)
|
|
147
|
+
|
|
148
|
+
retrieve_stats = self._stats_monitor.get_current_retrieve_stats()
|
|
149
|
+
if retrieve_stats is not None:
|
|
150
|
+
retrieve_stats.detailed_metrics[
|
|
151
|
+
"instrumented_connector_batched_get_time"
|
|
152
|
+
] = (
|
|
153
|
+
retrieve_stats.detailed_metrics.get(
|
|
154
|
+
"instrumented_connector_batched_get_time", 0.0
|
|
155
|
+
)
|
|
156
|
+
+ duration
|
|
157
|
+
)
|
|
158
|
+
|
|
159
|
+
total_size = sum(
|
|
160
|
+
memory_obj.get_size()
|
|
161
|
+
for memory_obj in memory_objs
|
|
162
|
+
if memory_obj is not None
|
|
163
|
+
)
|
|
164
|
+
if total_size > 0:
|
|
165
|
+
self._stats_monitor.update_interval_remote_read_metrics(total_size)
|
|
166
|
+
logger.debug(
|
|
167
|
+
"[%s]Bytes loaded: %.3f MBytes in %.3f ms",
|
|
168
|
+
self.name,
|
|
169
|
+
total_size / 1e6,
|
|
170
|
+
duration * 1000,
|
|
171
|
+
)
|
|
172
|
+
return memory_objs
|
|
173
|
+
|
|
174
|
+
async def batched_put(
|
|
175
|
+
self, keys: List[CacheEngineKey], memory_objs: List[MemoryObj]
|
|
176
|
+
):
|
|
177
|
+
total_size = sum(
|
|
178
|
+
memory_obj.get_size()
|
|
179
|
+
for memory_obj in memory_objs
|
|
180
|
+
if memory_obj is not None
|
|
181
|
+
)
|
|
182
|
+
begin = time.perf_counter()
|
|
183
|
+
try:
|
|
184
|
+
await self._connector.batched_put(keys, memory_objs)
|
|
185
|
+
except Exception as e:
|
|
186
|
+
logger.warning(f"batched put error: {e}")
|
|
187
|
+
finally:
|
|
188
|
+
for memory_obj in memory_objs:
|
|
189
|
+
memory_obj.ref_count_down()
|
|
190
|
+
|
|
191
|
+
end = time.perf_counter()
|
|
192
|
+
self._stats_monitor.update_interval_remote_time_to_put((end - begin) * 1000)
|
|
193
|
+
self._stats_monitor.update_interval_remote_write_metrics(total_size)
|
|
194
|
+
logger.debug(
|
|
195
|
+
"[%s]Bytes offloaded: %.3f MBytes in %.3f ms",
|
|
196
|
+
self.name,
|
|
197
|
+
total_size / 1e6,
|
|
198
|
+
(end - begin) * 1000,
|
|
199
|
+
)
|
|
200
|
+
|
|
201
|
+
def remove_sync(self, key: CacheEngineKey) -> bool:
|
|
202
|
+
return self._connector.remove_sync(key)
|
|
203
|
+
|
|
204
|
+
def batched_contains(self, keys: List[CacheEngineKey]) -> int:
|
|
205
|
+
return self._connector.batched_contains(keys)
|
|
206
|
+
|
|
207
|
+
def support_batched_contains(self) -> bool:
|
|
208
|
+
return self._connector.support_batched_contains()
|
|
209
|
+
|
|
210
|
+
def reshape_partial_chunk(
|
|
211
|
+
self, memory_obj: MemoryObj, bytes_read: int
|
|
212
|
+
) -> MemoryObj:
|
|
213
|
+
return self._connector.reshape_partial_chunk(memory_obj, bytes_read)
|
|
214
|
+
|
|
215
|
+
def post_init(self):
|
|
216
|
+
return self._connector.post_init()
|
|
217
|
+
|
|
218
|
+
def __repr__(self) -> str:
|
|
219
|
+
return f"InstrumentedRemoteConnector({self._connector})"
|
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
# SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
# First Party
|
|
3
|
+
from lmcache.logging import init_logger
|
|
4
|
+
from lmcache.v1.storage_backend.connector import (
|
|
5
|
+
ConnectorAdapter,
|
|
6
|
+
ConnectorContext,
|
|
7
|
+
parse_remote_url,
|
|
8
|
+
)
|
|
9
|
+
from lmcache.v1.storage_backend.connector.base_connector import RemoteConnector
|
|
10
|
+
|
|
11
|
+
logger = init_logger(__name__)
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
class LMServerConnectorAdapter(ConnectorAdapter):
|
|
15
|
+
"""Adapter for LM Server connectors."""
|
|
16
|
+
|
|
17
|
+
def __init__(self) -> None:
|
|
18
|
+
super().__init__("lm://")
|
|
19
|
+
|
|
20
|
+
def create_connector(self, context: ConnectorContext) -> RemoteConnector:
|
|
21
|
+
# Local
|
|
22
|
+
from .lm_connector import LMCServerConnector
|
|
23
|
+
|
|
24
|
+
logger.info(f"Creating LM Server connector for URL: {context.url}")
|
|
25
|
+
parse_url = parse_remote_url(context.url)
|
|
26
|
+
return LMCServerConnector(
|
|
27
|
+
host=parse_url.host,
|
|
28
|
+
port=parse_url.port,
|
|
29
|
+
loop=context.loop,
|
|
30
|
+
local_cpu_backend=context.local_cpu_backend,
|
|
31
|
+
)
|
|
@@ -0,0 +1,176 @@
|
|
|
1
|
+
# SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
# Standard
|
|
3
|
+
from typing import List, Optional, no_type_check
|
|
4
|
+
import asyncio
|
|
5
|
+
import socket
|
|
6
|
+
|
|
7
|
+
# Third Party
|
|
8
|
+
import torch
|
|
9
|
+
|
|
10
|
+
# First Party
|
|
11
|
+
from lmcache.logging import init_logger
|
|
12
|
+
from lmcache.utils import CacheEngineKey, _lmcache_nvtx_annotate
|
|
13
|
+
from lmcache.v1.memory_management import MemoryFormat, MemoryObj
|
|
14
|
+
from lmcache.v1.protocol import (
|
|
15
|
+
ClientCommand,
|
|
16
|
+
ClientMetaMessage,
|
|
17
|
+
ServerMetaMessage,
|
|
18
|
+
ServerReturnCode,
|
|
19
|
+
)
|
|
20
|
+
from lmcache.v1.storage_backend.connector.base_connector import RemoteConnector
|
|
21
|
+
from lmcache.v1.storage_backend.local_cpu_backend import LocalCPUBackend
|
|
22
|
+
|
|
23
|
+
logger = init_logger(__name__)
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
# TODO: performance optimization for this class, consider using C/C++/Rust
|
|
27
|
+
# for communication + deserialization
|
|
28
|
+
class LMCServerConnector(RemoteConnector):
|
|
29
|
+
def __init__(
|
|
30
|
+
self,
|
|
31
|
+
host: str,
|
|
32
|
+
port: int,
|
|
33
|
+
loop: asyncio.AbstractEventLoop,
|
|
34
|
+
local_cpu_backend: LocalCPUBackend,
|
|
35
|
+
):
|
|
36
|
+
# NOTE(Jiayi): According to Python documentation:
|
|
37
|
+
# https://docs.python.org/3/library/asyncio-eventloop.html
|
|
38
|
+
# In general, protocol implementations that use transport-based APIs
|
|
39
|
+
# such as loop.create_connection() and loop.create_server() are faster
|
|
40
|
+
# than implementations that work with sockets.
|
|
41
|
+
# However, we use socket here as we need to use the socket.recv_into()
|
|
42
|
+
# to reduce memory copy.
|
|
43
|
+
|
|
44
|
+
# initialize base class, which includes some common attributes
|
|
45
|
+
super().__init__(local_cpu_backend.config, local_cpu_backend.metadata)
|
|
46
|
+
|
|
47
|
+
self.client_socket = socket.socket(socket.AF_INET, socket.SOCK_STREAM)
|
|
48
|
+
self.client_socket.connect((host, port))
|
|
49
|
+
# loop.sock_recv_into(sock, buf)
|
|
50
|
+
|
|
51
|
+
self.loop = loop
|
|
52
|
+
self.local_cpu_backend = local_cpu_backend
|
|
53
|
+
|
|
54
|
+
self.async_socket_lock = asyncio.Lock()
|
|
55
|
+
|
|
56
|
+
# TODO(Jiayi): This should be an async function
|
|
57
|
+
def receive_all(self, meta: ServerMetaMessage) -> Optional[MemoryObj]:
|
|
58
|
+
received = 0
|
|
59
|
+
n = meta.length
|
|
60
|
+
|
|
61
|
+
# TODO(Jiayi): Format will be used once we support
|
|
62
|
+
# compressed memory format
|
|
63
|
+
memory_obj = self.local_cpu_backend.allocate(
|
|
64
|
+
meta.shape,
|
|
65
|
+
meta.dtype,
|
|
66
|
+
meta.fmt,
|
|
67
|
+
)
|
|
68
|
+
if memory_obj is None:
|
|
69
|
+
logger.warning("Failed to allocate memory during remote receive")
|
|
70
|
+
return None
|
|
71
|
+
|
|
72
|
+
buffer = memory_obj.byte_array
|
|
73
|
+
view = memoryview(buffer)
|
|
74
|
+
|
|
75
|
+
while received < n:
|
|
76
|
+
num_bytes = self.client_socket.recv_into(view[received:], n - received)
|
|
77
|
+
if num_bytes == 0:
|
|
78
|
+
return None
|
|
79
|
+
received += num_bytes
|
|
80
|
+
|
|
81
|
+
return memory_obj
|
|
82
|
+
|
|
83
|
+
async def exists(self, key: CacheEngineKey) -> bool:
|
|
84
|
+
# logger.debug("Call to exists()!")
|
|
85
|
+
|
|
86
|
+
async with self.async_socket_lock:
|
|
87
|
+
self.client_socket.sendall(
|
|
88
|
+
ClientMetaMessage(
|
|
89
|
+
ClientCommand.EXIST,
|
|
90
|
+
key,
|
|
91
|
+
0,
|
|
92
|
+
MemoryFormat(1),
|
|
93
|
+
torch.float16,
|
|
94
|
+
torch.Size([0, 0, 0, 0]),
|
|
95
|
+
).serialize()
|
|
96
|
+
)
|
|
97
|
+
|
|
98
|
+
response = self.client_socket.recv(ServerMetaMessage.packlength())
|
|
99
|
+
|
|
100
|
+
return ServerMetaMessage.deserialize(response).code == ServerReturnCode.SUCCESS
|
|
101
|
+
|
|
102
|
+
def exists_sync(self, key: CacheEngineKey) -> bool:
|
|
103
|
+
future = asyncio.run_coroutine_threadsafe(self.exists(key), self.loop)
|
|
104
|
+
try:
|
|
105
|
+
res = future.result()
|
|
106
|
+
return res
|
|
107
|
+
except Exception as e:
|
|
108
|
+
logger.warning(f"lm connector failed in exists: {e}")
|
|
109
|
+
return False
|
|
110
|
+
|
|
111
|
+
async def put(
|
|
112
|
+
self,
|
|
113
|
+
key: CacheEngineKey,
|
|
114
|
+
memory_obj: MemoryObj,
|
|
115
|
+
):
|
|
116
|
+
# logger.debug("Async call to put()!")
|
|
117
|
+
|
|
118
|
+
kv_bytes = memory_obj.byte_array
|
|
119
|
+
kv_shape = memory_obj.get_shape()
|
|
120
|
+
kv_dtype = memory_obj.get_dtype()
|
|
121
|
+
memory_format = memory_obj.get_memory_format()
|
|
122
|
+
|
|
123
|
+
async with self.async_socket_lock:
|
|
124
|
+
await self.loop.sock_sendall(
|
|
125
|
+
self.client_socket,
|
|
126
|
+
ClientMetaMessage(
|
|
127
|
+
ClientCommand.PUT,
|
|
128
|
+
key,
|
|
129
|
+
len(kv_bytes),
|
|
130
|
+
memory_format,
|
|
131
|
+
kv_dtype,
|
|
132
|
+
kv_shape,
|
|
133
|
+
).serialize(),
|
|
134
|
+
)
|
|
135
|
+
|
|
136
|
+
await self.loop.sock_sendall(self.client_socket, kv_bytes)
|
|
137
|
+
|
|
138
|
+
# TODO(Jiayi): This should be an async function
|
|
139
|
+
@_lmcache_nvtx_annotate
|
|
140
|
+
async def get(self, key: CacheEngineKey) -> Optional[MemoryObj]:
|
|
141
|
+
# NOTE(Jiayi): Not using any await in the following as
|
|
142
|
+
# we don't want to yield control to other tasks which could
|
|
143
|
+
# sacrifice the performance loading to trade the performance of
|
|
144
|
+
# saving
|
|
145
|
+
async with self.async_socket_lock:
|
|
146
|
+
self.client_socket.sendall(
|
|
147
|
+
ClientMetaMessage(
|
|
148
|
+
ClientCommand.GET,
|
|
149
|
+
key,
|
|
150
|
+
0,
|
|
151
|
+
MemoryFormat(1),
|
|
152
|
+
torch.float16,
|
|
153
|
+
torch.Size([0, 0, 0, 0]),
|
|
154
|
+
).serialize()
|
|
155
|
+
)
|
|
156
|
+
|
|
157
|
+
data = self.client_socket.recv(ServerMetaMessage.packlength())
|
|
158
|
+
|
|
159
|
+
meta = ServerMetaMessage.deserialize(data)
|
|
160
|
+
if meta.code != ServerReturnCode.SUCCESS:
|
|
161
|
+
return None
|
|
162
|
+
|
|
163
|
+
async with self.async_socket_lock:
|
|
164
|
+
memory_obj = self.receive_all(meta)
|
|
165
|
+
|
|
166
|
+
return memory_obj
|
|
167
|
+
|
|
168
|
+
# TODO
|
|
169
|
+
@no_type_check
|
|
170
|
+
async def list(self) -> List[str]:
|
|
171
|
+
pass
|
|
172
|
+
|
|
173
|
+
async def close(self):
|
|
174
|
+
async with self.async_socket_lock:
|
|
175
|
+
self.client_socket.close()
|
|
176
|
+
logger.info("Closed the lmserver connection")
|
|
@@ -0,0 +1,57 @@
|
|
|
1
|
+
# SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
# Standard
|
|
3
|
+
from urllib.parse import parse_qs, urlparse
|
|
4
|
+
|
|
5
|
+
# First Party
|
|
6
|
+
from lmcache.logging import init_logger
|
|
7
|
+
from lmcache.v1.storage_backend.connector import (
|
|
8
|
+
ConnectorAdapter,
|
|
9
|
+
ConnectorContext,
|
|
10
|
+
)
|
|
11
|
+
from lmcache.v1.storage_backend.connector.base_connector import RemoteConnector
|
|
12
|
+
|
|
13
|
+
logger = init_logger(__name__)
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
class MockConnectorAdapter(ConnectorAdapter):
|
|
17
|
+
"""Adapter for Mock Connector"""
|
|
18
|
+
|
|
19
|
+
def __init__(self) -> None:
|
|
20
|
+
super().__init__("mock://")
|
|
21
|
+
|
|
22
|
+
def create_connector(self, context: ConnectorContext) -> RemoteConnector:
|
|
23
|
+
# Local import to avoid circular dependencies
|
|
24
|
+
# Local
|
|
25
|
+
from .mock_connector import MockConnector
|
|
26
|
+
|
|
27
|
+
logger.info(f"Creating Mock connector for URL: {context.url}")
|
|
28
|
+
|
|
29
|
+
parsed = urlparse(context.url)
|
|
30
|
+
# capacity is provided as the netloc in URLs like: mock://100/?...
|
|
31
|
+
if not parsed.netloc:
|
|
32
|
+
raise ValueError(
|
|
33
|
+
"mock connector requires capacity in GB as netloc, e.g. mock://100/?..."
|
|
34
|
+
)
|
|
35
|
+
try:
|
|
36
|
+
capacity_gb = int(parsed.netloc)
|
|
37
|
+
except ValueError as e:
|
|
38
|
+
raise ValueError(
|
|
39
|
+
f"Invalid capacity '{parsed.netloc}' for",
|
|
40
|
+
" mock connector; must be an integer (GB).",
|
|
41
|
+
) from e
|
|
42
|
+
|
|
43
|
+
params = parse_qs(parsed.query) if parsed.query else {}
|
|
44
|
+
# Defaults
|
|
45
|
+
peeking_latency_ms = float(params.get("peeking_latency", ["1"])[0])
|
|
46
|
+
read_throughput_gbps = float(params.get("read_throughput", ["2"])[0])
|
|
47
|
+
write_throughput_gbps = float(params.get("write_throughput", ["2"])[0])
|
|
48
|
+
|
|
49
|
+
return MockConnector(
|
|
50
|
+
url=context.url,
|
|
51
|
+
loop=context.loop,
|
|
52
|
+
local_cpu_backend=context.local_cpu_backend,
|
|
53
|
+
capacity=capacity_gb,
|
|
54
|
+
read_throughput=read_throughput_gbps,
|
|
55
|
+
write_throughput=write_throughput_gbps,
|
|
56
|
+
peeking_latency=peeking_latency_ms,
|
|
57
|
+
)
|