lmcache-cli 0.4.5.dev0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- lmcache/__init__.py +84 -0
- lmcache/_version.py +24 -0
- lmcache/cli/__init__.py +1 -0
- lmcache/cli/commands/__init__.py +34 -0
- lmcache/cli/commands/base.py +157 -0
- lmcache/cli/commands/bench/__init__.py +557 -0
- lmcache/cli/commands/bench/engine_bench/__init__.py +1 -0
- lmcache/cli/commands/bench/engine_bench/config.py +245 -0
- lmcache/cli/commands/bench/engine_bench/interactive/__init__.py +274 -0
- lmcache/cli/commands/bench/engine_bench/interactive/config.json +10 -0
- lmcache/cli/commands/bench/engine_bench/interactive/schema.py +352 -0
- lmcache/cli/commands/bench/engine_bench/interactive/state.py +327 -0
- lmcache/cli/commands/bench/engine_bench/interactive/terminal.py +291 -0
- lmcache/cli/commands/bench/engine_bench/progress.py +145 -0
- lmcache/cli/commands/bench/engine_bench/request_sender.py +232 -0
- lmcache/cli/commands/bench/engine_bench/stats.py +275 -0
- lmcache/cli/commands/bench/engine_bench/workloads/__init__.py +153 -0
- lmcache/cli/commands/bench/engine_bench/workloads/base.py +122 -0
- lmcache/cli/commands/bench/engine_bench/workloads/long_doc_permutator.py +435 -0
- lmcache/cli/commands/bench/engine_bench/workloads/long_doc_qa.py +281 -0
- lmcache/cli/commands/bench/engine_bench/workloads/multi_round_chat.py +337 -0
- lmcache/cli/commands/bench/engine_bench/workloads/random_prefill.py +178 -0
- lmcache/cli/commands/describe.py +310 -0
- lmcache/cli/commands/kvcache.py +133 -0
- lmcache/cli/commands/mock.py +75 -0
- lmcache/cli/commands/ping.py +113 -0
- lmcache/cli/commands/query/__init__.py +155 -0
- lmcache/cli/commands/query/prompt.py +134 -0
- lmcache/cli/commands/query/request.py +357 -0
- lmcache/cli/commands/server.py +99 -0
- lmcache/cli/commands/tool/__init__.py +63 -0
- lmcache/cli/commands/tool/cache_simulator.py +113 -0
- lmcache/cli/commands/trace/__init__.py +505 -0
- lmcache/cli/commands/trace/dispatch.py +249 -0
- lmcache/cli/commands/trace/driver.py +372 -0
- lmcache/cli/commands/trace/stats.py +289 -0
- lmcache/cli/documents/lmcache.txt +11 -0
- lmcache/cli/main.py +42 -0
- lmcache/cli/metrics/__init__.py +29 -0
- lmcache/cli/metrics/formatter.py +171 -0
- lmcache/cli/metrics/handler.py +94 -0
- lmcache/cli/metrics/metrics.py +161 -0
- lmcache/cli/metrics/section.py +77 -0
- lmcache/connections.py +173 -0
- lmcache/integration/__init__.py +2 -0
- lmcache/integration/base_service_factory.py +165 -0
- lmcache/integration/request_telemetry/__init__.py +1 -0
- lmcache/integration/request_telemetry/base.py +51 -0
- lmcache/integration/request_telemetry/factory.py +113 -0
- lmcache/integration/request_telemetry/fastapi.py +109 -0
- lmcache/integration/request_telemetry/noop.py +35 -0
- lmcache/integration/sglang/__init__.py +2 -0
- lmcache/integration/sglang/sglang_adapter.py +326 -0
- lmcache/integration/sglang/utils.py +39 -0
- lmcache/integration/vllm/__init__.py +1 -0
- lmcache/integration/vllm/lmcache_connector_v1.py +213 -0
- lmcache/integration/vllm/lmcache_connector_v1_085.py +150 -0
- lmcache/integration/vllm/lmcache_mp_connector_0180.py +1072 -0
- lmcache/integration/vllm/tests/test_mm_hash_utils.py +112 -0
- lmcache/integration/vllm/utils.py +433 -0
- lmcache/integration/vllm/vllm_multi_process_adapter.py +1090 -0
- lmcache/integration/vllm/vllm_service_factory.py +339 -0
- lmcache/integration/vllm/vllm_v1_adapter.py +1713 -0
- lmcache/logging.py +107 -0
- lmcache/native_storage_ops.pyi +230 -0
- lmcache/non_cuda_equivalents.py +1424 -0
- lmcache/observability.py +1958 -0
- lmcache/storage_backend/serde/__init__.py +1 -0
- lmcache/storage_backend/serde/cachegen_basics.py +210 -0
- lmcache/storage_backend/serde/cachegen_decoder.py +207 -0
- lmcache/storage_backend/serde/cachegen_encoder.py +394 -0
- lmcache/storage_backend/serde/serde.py +75 -0
- lmcache/tools/__init__.py +1 -0
- lmcache/tools/cache_simulator/README.md +392 -0
- lmcache/tools/cache_simulator/__init__.py +1 -0
- lmcache/tools/cache_simulator/docs/simulate_example.png +0 -0
- lmcache/tools/cache_simulator/docs/sweep_example.png +0 -0
- lmcache/tools/cache_simulator/gen_bench_dataset.py +360 -0
- lmcache/tools/cache_simulator/lru_cache.py +124 -0
- lmcache/tools/cache_simulator/plot_hit_rate.py +231 -0
- lmcache/tools/cache_simulator/simulator.py +795 -0
- lmcache/tools/controller_benchmark/README.md +161 -0
- lmcache/tools/controller_benchmark/__init__.py +1 -0
- lmcache/tools/controller_benchmark/__main__.py +331 -0
- lmcache/tools/controller_benchmark/benchmark.py +660 -0
- lmcache/tools/controller_benchmark/config.py +44 -0
- lmcache/tools/controller_benchmark/constants.py +10 -0
- lmcache/tools/controller_benchmark/handlers/__init__.py +46 -0
- lmcache/tools/controller_benchmark/handlers/admit.py +52 -0
- lmcache/tools/controller_benchmark/handlers/base.py +47 -0
- lmcache/tools/controller_benchmark/handlers/deregister.py +49 -0
- lmcache/tools/controller_benchmark/handlers/evict.py +52 -0
- lmcache/tools/controller_benchmark/handlers/heartbeat.py +56 -0
- lmcache/tools/controller_benchmark/handlers/p2p_lookup.py +47 -0
- lmcache/tools/controller_benchmark/handlers/register.py +56 -0
- lmcache/tools/mp_status_viewer/__init__.py +1 -0
- lmcache/tools/mp_status_viewer/__main__.py +95 -0
- lmcache/usage_context.py +417 -0
- lmcache/utils.py +665 -0
- lmcache/v1/__init__.py +2 -0
- lmcache/v1/api_server/__init__.py +2 -0
- lmcache/v1/api_server/__main__.py +537 -0
- lmcache/v1/basic_check.py +112 -0
- lmcache/v1/cache_controller/__init__.py +9 -0
- lmcache/v1/cache_controller/commands/__init__.py +15 -0
- lmcache/v1/cache_controller/commands/base.py +35 -0
- lmcache/v1/cache_controller/commands/full_sync.py +49 -0
- lmcache/v1/cache_controller/config.py +176 -0
- lmcache/v1/cache_controller/controller_manager.py +535 -0
- lmcache/v1/cache_controller/controllers/__init__.py +11 -0
- lmcache/v1/cache_controller/controllers/full_sync_tracker.py +473 -0
- lmcache/v1/cache_controller/controllers/kv_controller.py +439 -0
- lmcache/v1/cache_controller/controllers/registration_controller.py +282 -0
- lmcache/v1/cache_controller/executor.py +463 -0
- lmcache/v1/cache_controller/frontend/static/css/style.css +201 -0
- lmcache/v1/cache_controller/frontend/static/img/logo.png +0 -0
- lmcache/v1/cache_controller/frontend/static/index.html +234 -0
- lmcache/v1/cache_controller/frontend/static/js/controller_app.js +660 -0
- lmcache/v1/cache_controller/full_sync_sender.py +475 -0
- lmcache/v1/cache_controller/locks.py +149 -0
- lmcache/v1/cache_controller/message.py +828 -0
- lmcache/v1/cache_controller/observability.py +208 -0
- lmcache/v1/cache_controller/utils.py +679 -0
- lmcache/v1/cache_controller/worker.py +665 -0
- lmcache/v1/cache_engine.py +2058 -0
- lmcache/v1/cache_interface.py +19 -0
- lmcache/v1/check/__init__.py +74 -0
- lmcache/v1/check/check_mode_gen.py +86 -0
- lmcache/v1/check/check_mode_test_l2_adapter.py +284 -0
- lmcache/v1/check/check_mode_test_remote.py +155 -0
- lmcache/v1/check/check_mode_test_storage_manager.py +142 -0
- lmcache/v1/check/utils.py +571 -0
- lmcache/v1/compute/__init__.py +2 -0
- lmcache/v1/compute/attention/__init__.py +0 -0
- lmcache/v1/compute/attention/abstract.py +39 -0
- lmcache/v1/compute/attention/flash_attn.py +129 -0
- lmcache/v1/compute/attention/flash_infer_sparse.py +284 -0
- lmcache/v1/compute/attention/metadata.py +85 -0
- lmcache/v1/compute/attention/utils.py +14 -0
- lmcache/v1/compute/blend/__init__.py +7 -0
- lmcache/v1/compute/blend/blender.py +168 -0
- lmcache/v1/compute/blend/metadata.py +34 -0
- lmcache/v1/compute/blend/utils.py +63 -0
- lmcache/v1/compute/models/__init__.py +0 -0
- lmcache/v1/compute/models/base.py +141 -0
- lmcache/v1/compute/models/llama.py +9 -0
- lmcache/v1/compute/models/qwen3.py +24 -0
- lmcache/v1/compute/models/utils.py +68 -0
- lmcache/v1/compute/positional_encoding.py +199 -0
- lmcache/v1/config.py +848 -0
- lmcache/v1/config_base.py +848 -0
- lmcache/v1/distributed/api.py +248 -0
- lmcache/v1/distributed/config.py +321 -0
- lmcache/v1/distributed/error.py +64 -0
- lmcache/v1/distributed/eviction.py +192 -0
- lmcache/v1/distributed/eviction_policy/__init__.py +21 -0
- lmcache/v1/distributed/eviction_policy/factory.py +27 -0
- lmcache/v1/distributed/eviction_policy/lru.py +244 -0
- lmcache/v1/distributed/eviction_policy/noop.py +50 -0
- lmcache/v1/distributed/internal_api.py +170 -0
- lmcache/v1/distributed/l1_manager.py +835 -0
- lmcache/v1/distributed/l2_adapters/__init__.py +67 -0
- lmcache/v1/distributed/l2_adapters/base.py +360 -0
- lmcache/v1/distributed/l2_adapters/config.py +385 -0
- lmcache/v1/distributed/l2_adapters/factory.py +205 -0
- lmcache/v1/distributed/l2_adapters/fs_l2_adapter.py +747 -0
- lmcache/v1/distributed/l2_adapters/fs_native_l2_adapter.py +167 -0
- lmcache/v1/distributed/l2_adapters/mock_l2_adapter.py +516 -0
- lmcache/v1/distributed/l2_adapters/mooncake_store_l2_adapter.py +135 -0
- lmcache/v1/distributed/l2_adapters/native_connector_l2_adapter.py +468 -0
- lmcache/v1/distributed/l2_adapters/native_plugin_l2_adapter.py +199 -0
- lmcache/v1/distributed/l2_adapters/nixl_store_dynamic_l2_adapter.py +831 -0
- lmcache/v1/distributed/l2_adapters/nixl_store_l2_adapter.py +983 -0
- lmcache/v1/distributed/l2_adapters/plugin_l2_adapter.py +210 -0
- lmcache/v1/distributed/l2_adapters/resp_l2_adapter.py +176 -0
- lmcache/v1/distributed/memory_manager.py +179 -0
- lmcache/v1/distributed/storage_controller.py +39 -0
- lmcache/v1/distributed/storage_controllers/__init__.py +43 -0
- lmcache/v1/distributed/storage_controllers/eviction_controller.py +242 -0
- lmcache/v1/distributed/storage_controllers/prefetch_controller.py +830 -0
- lmcache/v1/distributed/storage_controllers/prefetch_policy.py +193 -0
- lmcache/v1/distributed/storage_controllers/store_controller.py +452 -0
- lmcache/v1/distributed/storage_controllers/store_policy.py +213 -0
- lmcache/v1/distributed/storage_manager.py +532 -0
- lmcache/v1/event_manager.py +145 -0
- lmcache/v1/exceptions/__init__.py +16 -0
- lmcache/v1/gpu_connector/__init__.py +126 -0
- lmcache/v1/gpu_connector/gpu_connectors.py +1906 -0
- lmcache/v1/gpu_connector/gpu_ops.py +85 -0
- lmcache/v1/gpu_connector/hpu_connector.py +326 -0
- lmcache/v1/gpu_connector/mock_gpu_connector.py +67 -0
- lmcache/v1/gpu_connector/utils.py +890 -0
- lmcache/v1/gpu_connector/xpu_connectors.py +916 -0
- lmcache/v1/health_monitor/__init__.py +1 -0
- lmcache/v1/health_monitor/base.py +587 -0
- lmcache/v1/health_monitor/checks/__init__.py +1 -0
- lmcache/v1/health_monitor/checks/remote_backend_check.py +304 -0
- lmcache/v1/health_monitor/constants.py +36 -0
- lmcache/v1/internal_api_server/__init__.py +0 -0
- lmcache/v1/internal_api_server/api_registry.py +59 -0
- lmcache/v1/internal_api_server/api_server.py +120 -0
- lmcache/v1/internal_api_server/common/__init__.py +1 -0
- lmcache/v1/internal_api_server/common/env_api.py +22 -0
- lmcache/v1/internal_api_server/common/loglevel_api.py +57 -0
- lmcache/v1/internal_api_server/common/metrics_api.py +29 -0
- lmcache/v1/internal_api_server/common/periodic_thread_api.py +138 -0
- lmcache/v1/internal_api_server/common/run_script_api.py +73 -0
- lmcache/v1/internal_api_server/common/thread_api.py +63 -0
- lmcache/v1/internal_api_server/controller/__init__.py +1 -0
- lmcache/v1/internal_api_server/controller/key_stats_api.py +81 -0
- lmcache/v1/internal_api_server/controller/worker_info_api.py +136 -0
- lmcache/v1/internal_api_server/utils.py +43 -0
- lmcache/v1/internal_api_server/vllm/__init__.py +1 -0
- lmcache/v1/internal_api_server/vllm/backend_api.py +221 -0
- lmcache/v1/internal_api_server/vllm/bypass_api.py +204 -0
- lmcache/v1/internal_api_server/vllm/cache_api.py +895 -0
- lmcache/v1/internal_api_server/vllm/chunk_statistics_api.py +141 -0
- lmcache/v1/internal_api_server/vllm/conf_api.py +147 -0
- lmcache/v1/internal_api_server/vllm/freeze_api.py +172 -0
- lmcache/v1/internal_api_server/vllm/hot_cache_api.py +184 -0
- lmcache/v1/internal_api_server/vllm/inference_api.py +65 -0
- lmcache/v1/internal_api_server/vllm/load_fs_chunks_api.py +320 -0
- lmcache/v1/internal_api_server/vllm/lookup_api.py +145 -0
- lmcache/v1/internal_api_server/vllm/version_api.py +25 -0
- lmcache/v1/kv_layer_groups.py +267 -0
- lmcache/v1/lazy_memory_allocator.py +284 -0
- lmcache/v1/lookup_client/__init__.py +25 -0
- lmcache/v1/lookup_client/abstract_client.py +77 -0
- lmcache/v1/lookup_client/async_lookup_message.py +50 -0
- lmcache/v1/lookup_client/chunk_statistics_lookup_client.py +200 -0
- lmcache/v1/lookup_client/factory.py +251 -0
- lmcache/v1/lookup_client/hit_limit_lookup_client.py +86 -0
- lmcache/v1/lookup_client/lmcache_async_lookup_client.py +407 -0
- lmcache/v1/lookup_client/lmcache_lookup_client.py +285 -0
- lmcache/v1/lookup_client/lmcache_lookup_client_bypass.py +99 -0
- lmcache/v1/lookup_client/mooncake_lookup_client.py +87 -0
- lmcache/v1/lookup_client/record_strategies/__init__.py +77 -0
- lmcache/v1/lookup_client/record_strategies/base.py +327 -0
- lmcache/v1/lookup_client/record_strategies/file_hash.py +130 -0
- lmcache/v1/lookup_client/record_strategies/memory_bloom_filter.py +81 -0
- lmcache/v1/manager.py +539 -0
- lmcache/v1/memory_management.py +2619 -0
- lmcache/v1/metadata.py +114 -0
- lmcache/v1/mp_observability/AGENTS.override.md +21 -0
- lmcache/v1/mp_observability/README.md +204 -0
- lmcache/v1/mp_observability/config.py +340 -0
- lmcache/v1/mp_observability/event.py +100 -0
- lmcache/v1/mp_observability/event_bus.py +313 -0
- lmcache/v1/mp_observability/otel_init.py +145 -0
- lmcache/v1/mp_observability/subscribers/__init__.py +28 -0
- lmcache/v1/mp_observability/subscribers/logging/__init__.py +19 -0
- lmcache/v1/mp_observability/subscribers/logging/l1.py +56 -0
- lmcache/v1/mp_observability/subscribers/logging/l2.py +73 -0
- lmcache/v1/mp_observability/subscribers/logging/lookup_hash.py +209 -0
- lmcache/v1/mp_observability/subscribers/logging/mp_server.py +90 -0
- lmcache/v1/mp_observability/subscribers/logging/sm.py +59 -0
- lmcache/v1/mp_observability/subscribers/metrics/__init__.py +20 -0
- lmcache/v1/mp_observability/subscribers/metrics/l0_lifecycle.py +290 -0
- lmcache/v1/mp_observability/subscribers/metrics/l1.py +55 -0
- lmcache/v1/mp_observability/subscribers/metrics/l1_lifecycle.py +166 -0
- lmcache/v1/mp_observability/subscribers/metrics/l2.py +121 -0
- lmcache/v1/mp_observability/subscribers/metrics/sm.py +69 -0
- lmcache/v1/mp_observability/subscribers/tracing/__init__.py +12 -0
- lmcache/v1/mp_observability/subscribers/tracing/mp_server.py +333 -0
- lmcache/v1/mp_observability/subscribers/tracing/span_registry.py +148 -0
- lmcache/v1/mp_observability/trace/__init__.py +50 -0
- lmcache/v1/mp_observability/trace/codecs.py +255 -0
- lmcache/v1/mp_observability/trace/decorator.py +147 -0
- lmcache/v1/mp_observability/trace/format.py +132 -0
- lmcache/v1/mp_observability/trace/lifecycle.py +83 -0
- lmcache/v1/mp_observability/trace/reader.py +167 -0
- lmcache/v1/mp_observability/trace/recorder.py +300 -0
- lmcache/v1/multiprocess/__init__.py +0 -0
- lmcache/v1/multiprocess/affinity_pool.py +102 -0
- lmcache/v1/multiprocess/blend_server_v2.py +891 -0
- lmcache/v1/multiprocess/config.py +253 -0
- lmcache/v1/multiprocess/custom_types.py +281 -0
- lmcache/v1/multiprocess/futures.py +194 -0
- lmcache/v1/multiprocess/gpu_context.py +511 -0
- lmcache/v1/multiprocess/http_server.py +235 -0
- lmcache/v1/multiprocess/mp_runtime_plugin_launcher.py +130 -0
- lmcache/v1/multiprocess/mq.py +732 -0
- lmcache/v1/multiprocess/protocol.py +86 -0
- lmcache/v1/multiprocess/protocols/README.md +213 -0
- lmcache/v1/multiprocess/protocols/__init__.py +127 -0
- lmcache/v1/multiprocess/protocols/base.py +89 -0
- lmcache/v1/multiprocess/protocols/blend.py +109 -0
- lmcache/v1/multiprocess/protocols/blend_v2.py +57 -0
- lmcache/v1/multiprocess/protocols/controller.py +53 -0
- lmcache/v1/multiprocess/protocols/debug.py +34 -0
- lmcache/v1/multiprocess/protocols/engine.py +146 -0
- lmcache/v1/multiprocess/protocols/observability.py +39 -0
- lmcache/v1/multiprocess/server.py +1134 -0
- lmcache/v1/multiprocess/session.py +190 -0
- lmcache/v1/multiprocess/token_hasher.py +441 -0
- lmcache/v1/offload_server/__init__.py +17 -0
- lmcache/v1/offload_server/abstract_server.py +37 -0
- lmcache/v1/offload_server/message.py +30 -0
- lmcache/v1/offload_server/zmq_server.py +122 -0
- lmcache/v1/periodic_thread.py +579 -0
- lmcache/v1/pin_monitor.py +246 -0
- lmcache/v1/plugin/__init__.py +0 -0
- lmcache/v1/plugin/runtime_plugin_launcher.py +211 -0
- lmcache/v1/protocol.py +317 -0
- lmcache/v1/rpc/__init__.py +17 -0
- lmcache/v1/rpc/transport.py +105 -0
- lmcache/v1/rpc/zmq_transport.py +213 -0
- lmcache/v1/rpc_utils.py +165 -0
- lmcache/v1/server/__init__.py +2 -0
- lmcache/v1/server/__main__.py +170 -0
- lmcache/v1/server/storage_backend/__init__.py +21 -0
- lmcache/v1/server/storage_backend/abstract_backend.py +80 -0
- lmcache/v1/server/storage_backend/local_backend.py +75 -0
- lmcache/v1/server/utils.py +21 -0
- lmcache/v1/standalone/__init__.py +1 -0
- lmcache/v1/standalone/__main__.py +583 -0
- lmcache/v1/standalone/manager.py +80 -0
- lmcache/v1/standalone/standalone_service_factory.py +86 -0
- lmcache/v1/storage_backend/__init__.py +313 -0
- lmcache/v1/storage_backend/abstract_backend.py +445 -0
- lmcache/v1/storage_backend/audit_backend.py +233 -0
- lmcache/v1/storage_backend/batched_message_sender.py +222 -0
- lmcache/v1/storage_backend/cache_policy/__init__.py +45 -0
- lmcache/v1/storage_backend/cache_policy/base_policy.py +87 -0
- lmcache/v1/storage_backend/cache_policy/fifo.py +58 -0
- lmcache/v1/storage_backend/cache_policy/lfu.py +105 -0
- lmcache/v1/storage_backend/cache_policy/lru.py +81 -0
- lmcache/v1/storage_backend/cache_policy/mru.py +61 -0
- lmcache/v1/storage_backend/connector/__init__.py +443 -0
- lmcache/v1/storage_backend/connector/audit_adapter.py +77 -0
- lmcache/v1/storage_backend/connector/audit_connector.py +320 -0
- lmcache/v1/storage_backend/connector/base_connector.py +379 -0
- lmcache/v1/storage_backend/connector/blackhole_adapter.py +21 -0
- lmcache/v1/storage_backend/connector/blackhole_connector.py +37 -0
- lmcache/v1/storage_backend/connector/eic_adapter.py +31 -0
- lmcache/v1/storage_backend/connector/eic_connector.py +757 -0
- lmcache/v1/storage_backend/connector/external_adapter.py +79 -0
- lmcache/v1/storage_backend/connector/fs_adapter.py +51 -0
- lmcache/v1/storage_backend/connector/fs_connector.py +403 -0
- lmcache/v1/storage_backend/connector/infinistore_adapter.py +56 -0
- lmcache/v1/storage_backend/connector/infinistore_connector.py +177 -0
- lmcache/v1/storage_backend/connector/instrumented_connector.py +219 -0
- lmcache/v1/storage_backend/connector/lm_adapter.py +31 -0
- lmcache/v1/storage_backend/connector/lm_connector.py +176 -0
- lmcache/v1/storage_backend/connector/mock_adapter.py +57 -0
- lmcache/v1/storage_backend/connector/mock_connector.py +349 -0
- lmcache/v1/storage_backend/connector/mooncakestore_adapter.py +43 -0
- lmcache/v1/storage_backend/connector/mooncakestore_connector.py +614 -0
- lmcache/v1/storage_backend/connector/redis_adapter.py +181 -0
- lmcache/v1/storage_backend/connector/redis_connector.py +828 -0
- lmcache/v1/storage_backend/connector/s3_adapter.py +59 -0
- lmcache/v1/storage_backend/connector/s3_connector.py +699 -0
- lmcache/v1/storage_backend/connector/sagemaker_hyperpod_adapter.py +233 -0
- lmcache/v1/storage_backend/connector/sagemaker_hyperpod_connector.py +987 -0
- lmcache/v1/storage_backend/connector/valkey_adapter.py +114 -0
- lmcache/v1/storage_backend/connector/valkey_connector.py +627 -0
- lmcache/v1/storage_backend/gds_backend.py +1199 -0
- lmcache/v1/storage_backend/job_executor/__init__.py +0 -0
- lmcache/v1/storage_backend/job_executor/base_executor.py +34 -0
- lmcache/v1/storage_backend/job_executor/pq_executor.py +235 -0
- lmcache/v1/storage_backend/local_cpu_backend.py +810 -0
- lmcache/v1/storage_backend/local_disk_backend.py +656 -0
- lmcache/v1/storage_backend/maru_backend.py +734 -0
- lmcache/v1/storage_backend/naive_serde/__init__.py +50 -0
- lmcache/v1/storage_backend/naive_serde/cachegen_basics.py +133 -0
- lmcache/v1/storage_backend/naive_serde/cachegen_decoder.py +135 -0
- lmcache/v1/storage_backend/naive_serde/cachegen_encoder.py +83 -0
- lmcache/v1/storage_backend/naive_serde/kivi_serde.py +22 -0
- lmcache/v1/storage_backend/naive_serde/naive_serde.py +18 -0
- lmcache/v1/storage_backend/naive_serde/serde.py +37 -0
- lmcache/v1/storage_backend/native_clients/connector_client_base.py +165 -0
- lmcache/v1/storage_backend/native_clients/resp_client.py +35 -0
- lmcache/v1/storage_backend/nixl_storage_backend.py +1400 -0
- lmcache/v1/storage_backend/p2p_backend.py +788 -0
- lmcache/v1/storage_backend/path_sharder.py +117 -0
- lmcache/v1/storage_backend/pd_backend.py +646 -0
- lmcache/v1/storage_backend/plugins/dax_backend.py +1443 -0
- lmcache/v1/storage_backend/plugins/rust_raw_block_backend.py +1361 -0
- lmcache/v1/storage_backend/remote_backend.py +624 -0
- lmcache/v1/storage_backend/resp_client.py +227 -0
- lmcache/v1/storage_backend/storage_backend_listener.py +19 -0
- lmcache/v1/storage_backend/storage_manager.py +1352 -0
- lmcache/v1/system_detection.py +110 -0
- lmcache/v1/token_database.py +551 -0
- lmcache/v1/transfer_channel/__init__.py +83 -0
- lmcache/v1/transfer_channel/abstract.py +285 -0
- lmcache/v1/transfer_channel/mock_memory_channel.py +156 -0
- lmcache/v1/transfer_channel/nixl_channel.py +639 -0
- lmcache/v1/transfer_channel/py_socket_channel.py +260 -0
- lmcache/v1/transfer_channel/transfer_utils.py +63 -0
- lmcache/v1/utils/__init__.py +1 -0
- lmcache/v1/utils/bloom_filter.py +109 -0
- lmcache/v1/utils/cache_utils.py +125 -0
- lmcache_cli-0.4.5.dev0.dist-info/METADATA +185 -0
- lmcache_cli-0.4.5.dev0.dist-info/RECORD +399 -0
- lmcache_cli-0.4.5.dev0.dist-info/WHEEL +5 -0
- lmcache_cli-0.4.5.dev0.dist-info/entry_points.txt +2 -0
- lmcache_cli-0.4.5.dev0.dist-info/licenses/LICENSE +201 -0
- lmcache_cli-0.4.5.dev0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,660 @@
|
|
|
1
|
+
"""Core benchmark class for LMCache Controller ZMQ testing"""
|
|
2
|
+
|
|
3
|
+
# SPDX-License-Identifier: Apache-2.0
|
|
4
|
+
|
|
5
|
+
# Standard
|
|
6
|
+
from dataclasses import dataclass, field
|
|
7
|
+
from typing import Any, Dict, List, Optional, Tuple
|
|
8
|
+
import asyncio
|
|
9
|
+
import random
|
|
10
|
+
import statistics
|
|
11
|
+
import time
|
|
12
|
+
|
|
13
|
+
# Third Party
|
|
14
|
+
import msgspec
|
|
15
|
+
import psutil
|
|
16
|
+
import zmq
|
|
17
|
+
import zmq.asyncio
|
|
18
|
+
|
|
19
|
+
# First Party
|
|
20
|
+
from lmcache.logging import init_logger
|
|
21
|
+
from lmcache.v1.cache_controller.message import (
|
|
22
|
+
DeRegisterMsg,
|
|
23
|
+
RegisterMsg,
|
|
24
|
+
RegisterRetMsg,
|
|
25
|
+
)
|
|
26
|
+
from lmcache.v1.cache_controller.utils import KVChunkInfo
|
|
27
|
+
from lmcache.v1.rpc_utils import (
|
|
28
|
+
close_zmq_socket,
|
|
29
|
+
get_zmq_context,
|
|
30
|
+
get_zmq_socket,
|
|
31
|
+
get_zmq_socket_with_timeout,
|
|
32
|
+
)
|
|
33
|
+
|
|
34
|
+
# Local
|
|
35
|
+
from .config import ZMQBenchmarkConfig
|
|
36
|
+
from .constants import (
|
|
37
|
+
DEFAULT_BATCH_SEND_SIZE,
|
|
38
|
+
DEFAULT_OP_DISTRIBUTION_BASE,
|
|
39
|
+
DEFAULT_RECV_TIMEOUT_MS,
|
|
40
|
+
DEFAULT_SEND_HWM,
|
|
41
|
+
DEFAULT_SEND_TIMEOUT_MS,
|
|
42
|
+
)
|
|
43
|
+
from .handlers import OPERATION_HANDLERS
|
|
44
|
+
from .handlers.base import SocketType
|
|
45
|
+
|
|
46
|
+
logger = init_logger(__name__)
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
@dataclass
|
|
50
|
+
class TestData:
|
|
51
|
+
"""Test data for benchmark operations"""
|
|
52
|
+
|
|
53
|
+
instances: List[str]
|
|
54
|
+
workers: List[int]
|
|
55
|
+
locations: List[str]
|
|
56
|
+
keys: List[int]
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
@dataclass
|
|
60
|
+
class OperationStats:
|
|
61
|
+
"""Statistics for a single operation type"""
|
|
62
|
+
|
|
63
|
+
qps: float = 0.0 # messages per second
|
|
64
|
+
rps: float = 0.0 # requests per second
|
|
65
|
+
avg_latency: float = 0.0
|
|
66
|
+
min_latency: float = 0.0
|
|
67
|
+
max_latency: float = 0.0
|
|
68
|
+
p95_latency: float = 0.0
|
|
69
|
+
errors: int = 0
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
@dataclass
|
|
73
|
+
class BenchmarkResults:
|
|
74
|
+
"""Overall benchmark results"""
|
|
75
|
+
|
|
76
|
+
total_requests: int = 0
|
|
77
|
+
total_messages: int = 0
|
|
78
|
+
total_time: float = 0.0
|
|
79
|
+
overall_rps: float = 0.0 # requests per second
|
|
80
|
+
overall_qps: float = 0.0 # messages per second
|
|
81
|
+
operations: Dict[str, OperationStats] = field(default_factory=dict)
|
|
82
|
+
memory_usage: List[float] = field(default_factory=list)
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
class ZMQControllerBenchmark:
|
|
86
|
+
"""Benchmark class for LMCache Controller via ZMQ"""
|
|
87
|
+
|
|
88
|
+
def __init__(self, config: ZMQBenchmarkConfig):
|
|
89
|
+
self.config = config
|
|
90
|
+
self.context: Optional[zmq.asyncio.Context] = None
|
|
91
|
+
self.push_socket: Optional[Any] = None
|
|
92
|
+
self.req_socket: Optional[Any] = None
|
|
93
|
+
self.heartbeat_socket: Optional[Any] = None
|
|
94
|
+
self.heartbeat_url: Optional[str] = None
|
|
95
|
+
self.results = BenchmarkResults()
|
|
96
|
+
self.running = False
|
|
97
|
+
# Track sequence numbers per KVChunkInfo (instance_id, worker_id, location)
|
|
98
|
+
self.sequence_numbers: Dict[KVChunkInfo, int] = {}
|
|
99
|
+
|
|
100
|
+
# Track registered workers for cleanup
|
|
101
|
+
self.registered_workers: List[Tuple[str, int, str, int]] = []
|
|
102
|
+
|
|
103
|
+
async def setup(self):
|
|
104
|
+
"""Setup ZMQ sockets"""
|
|
105
|
+
self.context = get_zmq_context(use_asyncio=True)
|
|
106
|
+
self.push_socket = get_zmq_socket(
|
|
107
|
+
self.context,
|
|
108
|
+
self.config.controller_pull_url,
|
|
109
|
+
protocol="tcp",
|
|
110
|
+
role=zmq.PUSH,
|
|
111
|
+
bind_or_connect="connect",
|
|
112
|
+
)
|
|
113
|
+
# Set send timeout to avoid blocking indefinitely when controller is down
|
|
114
|
+
# SNDTIMEO: timeout in milliseconds, 0 means non-blocking
|
|
115
|
+
self.push_socket.setsockopt(zmq.SNDTIMEO, DEFAULT_SEND_TIMEOUT_MS)
|
|
116
|
+
# SNDHWM: high watermark for outbound messages
|
|
117
|
+
self.push_socket.setsockopt(zmq.SNDHWM, DEFAULT_SEND_HWM)
|
|
118
|
+
logger.info(
|
|
119
|
+
"Connected to controller PULL socket at %s",
|
|
120
|
+
self.config.controller_pull_url,
|
|
121
|
+
)
|
|
122
|
+
|
|
123
|
+
# Setup DEALER socket for request-reply operations (e.g., P2P lookup)
|
|
124
|
+
if self.config.controller_reply_url:
|
|
125
|
+
self.req_socket = get_zmq_socket_with_timeout(
|
|
126
|
+
self.context,
|
|
127
|
+
self.config.controller_reply_url,
|
|
128
|
+
protocol="tcp",
|
|
129
|
+
role=zmq.DEALER,
|
|
130
|
+
bind_or_connect="connect",
|
|
131
|
+
recv_timeout_ms=DEFAULT_RECV_TIMEOUT_MS,
|
|
132
|
+
send_timeout_ms=DEFAULT_SEND_TIMEOUT_MS,
|
|
133
|
+
)
|
|
134
|
+
logger.info(
|
|
135
|
+
"Connected to controller ROUTER socket at tcp://%s",
|
|
136
|
+
self.config.controller_reply_url,
|
|
137
|
+
)
|
|
138
|
+
logger.info(
|
|
139
|
+
"DEALER socket type: %d, last_endpoint: %s",
|
|
140
|
+
self.req_socket.get(zmq.TYPE),
|
|
141
|
+
self.req_socket.get_string(zmq.LAST_ENDPOINT, encoding="utf-8"),
|
|
142
|
+
)
|
|
143
|
+
# Give ZMQ time to establish the connection
|
|
144
|
+
time.sleep(0.1)
|
|
145
|
+
|
|
146
|
+
# Setup heartbeat DEALER socket if configured
|
|
147
|
+
if self.config.controller_heartbeat_url:
|
|
148
|
+
self.heartbeat_url = self.config.controller_heartbeat_url
|
|
149
|
+
self._setup_heartbeat_socket()
|
|
150
|
+
logger.info(
|
|
151
|
+
"Connected to heartbeat ROUTER socket at tcp://%s",
|
|
152
|
+
self.config.controller_heartbeat_url,
|
|
153
|
+
)
|
|
154
|
+
|
|
155
|
+
def cleanup(self):
|
|
156
|
+
"""Cleanup ZMQ sockets"""
|
|
157
|
+
if self.push_socket:
|
|
158
|
+
close_zmq_socket(self.push_socket)
|
|
159
|
+
if self.req_socket:
|
|
160
|
+
close_zmq_socket(self.req_socket)
|
|
161
|
+
if self.heartbeat_socket:
|
|
162
|
+
close_zmq_socket(self.heartbeat_socket)
|
|
163
|
+
logger.info("ZMQ sockets closed")
|
|
164
|
+
|
|
165
|
+
def generate_test_data(self) -> TestData:
|
|
166
|
+
"""Generate test data based on configuration
|
|
167
|
+
|
|
168
|
+
Each process gets a unique range of instance IDs to avoid conflicts.
|
|
169
|
+
Format: instance_p{process_id}_{instance_index}
|
|
170
|
+
"""
|
|
171
|
+
process_id = self.config.process_id
|
|
172
|
+
return TestData(
|
|
173
|
+
instances=[
|
|
174
|
+
"instance_p%d_%d" % (process_id, i)
|
|
175
|
+
for i in range(self.config.num_instances)
|
|
176
|
+
],
|
|
177
|
+
workers=list(range(self.config.num_workers)),
|
|
178
|
+
locations=["location_%d" % i for i in range(self.config.num_locations)],
|
|
179
|
+
keys=list(range(self.config.num_keys)),
|
|
180
|
+
)
|
|
181
|
+
|
|
182
|
+
def get_next_sequence_number(
|
|
183
|
+
self, instance_id: str, worker_id: int, location: str
|
|
184
|
+
) -> int:
|
|
185
|
+
"""
|
|
186
|
+
Get monotonically increasing sequence number for specific
|
|
187
|
+
instance-worker-location
|
|
188
|
+
"""
|
|
189
|
+
key = KVChunkInfo(instance_id, worker_id, location)
|
|
190
|
+
if key not in self.sequence_numbers:
|
|
191
|
+
self.sequence_numbers[key] = 0
|
|
192
|
+
seq = self.sequence_numbers[key]
|
|
193
|
+
self.sequence_numbers[key] += 1
|
|
194
|
+
return seq
|
|
195
|
+
|
|
196
|
+
async def send_messages(self, messages: List[Any]) -> float:
|
|
197
|
+
"""Send multiple messages via ZMQ PUSH socket
|
|
198
|
+
|
|
199
|
+
Args:
|
|
200
|
+
messages: List of messages to send
|
|
201
|
+
|
|
202
|
+
Returns:
|
|
203
|
+
Time taken to send all messages
|
|
204
|
+
|
|
205
|
+
Raises:
|
|
206
|
+
RuntimeError: If socket not initialized or send timeout
|
|
207
|
+
"""
|
|
208
|
+
if self.push_socket is None:
|
|
209
|
+
raise RuntimeError("Socket not initialized. Call setup() first.")
|
|
210
|
+
start_time = time.time()
|
|
211
|
+
encoded_msgs = [msgspec.msgpack.encode(msg) for msg in messages]
|
|
212
|
+
try:
|
|
213
|
+
await self.push_socket.send_multipart(encoded_msgs)
|
|
214
|
+
except zmq.Again as e:
|
|
215
|
+
raise RuntimeError(
|
|
216
|
+
"Send timeout - Controller may not be running at %s"
|
|
217
|
+
% self.config.controller_pull_url
|
|
218
|
+
) from e
|
|
219
|
+
return time.time() - start_time
|
|
220
|
+
|
|
221
|
+
async def send_request(self, message: Any) -> Tuple[float, Any]:
|
|
222
|
+
"""Send a message via ZMQ DEALER socket and wait for reply
|
|
223
|
+
|
|
224
|
+
Args:
|
|
225
|
+
message: Message to send
|
|
226
|
+
|
|
227
|
+
Returns:
|
|
228
|
+
Tuple of (time taken, response)
|
|
229
|
+
|
|
230
|
+
Raises:
|
|
231
|
+
RuntimeError: If socket not initialized or timeout
|
|
232
|
+
zmq.ZMQError: If DEALER socket error occurs
|
|
233
|
+
"""
|
|
234
|
+
if self.req_socket is None:
|
|
235
|
+
raise RuntimeError(
|
|
236
|
+
"DEALER socket not initialized. "
|
|
237
|
+
"Ensure controller_reply_url is configured."
|
|
238
|
+
)
|
|
239
|
+
start_time = time.time()
|
|
240
|
+
encoded_msg = msgspec.msgpack.encode(message)
|
|
241
|
+
logger.debug(
|
|
242
|
+
"Sending request to %s, message type: %s, size: %d bytes",
|
|
243
|
+
self.config.controller_reply_url,
|
|
244
|
+
type(message).__name__,
|
|
245
|
+
len(encoded_msg),
|
|
246
|
+
)
|
|
247
|
+
|
|
248
|
+
try:
|
|
249
|
+
# DEALER socket: send [empty_frame, payload]
|
|
250
|
+
await self.req_socket.send_multipart([b"", encoded_msg])
|
|
251
|
+
frames = await self.req_socket.recv_multipart()
|
|
252
|
+
# DEALER receives: [empty_frame, payload]
|
|
253
|
+
response = frames[-1]
|
|
254
|
+
logger.debug("Response received, size: %d bytes", len(response))
|
|
255
|
+
return time.time() - start_time, response
|
|
256
|
+
except zmq.Again as e:
|
|
257
|
+
logger.error("Request timeout after waiting for response")
|
|
258
|
+
raise RuntimeError(
|
|
259
|
+
"Request timeout - Controller may not be running at %s"
|
|
260
|
+
% self.config.controller_reply_url
|
|
261
|
+
) from e
|
|
262
|
+
except zmq.ZMQError as e:
|
|
263
|
+
logger.error("ZMQ error: %s", e)
|
|
264
|
+
# Re-raise other ZMQ errors
|
|
265
|
+
raise
|
|
266
|
+
|
|
267
|
+
async def register_workers(self, test_data: TestData):
|
|
268
|
+
"""Pre-register all workers before benchmark using DEALER-ROUTER mode"""
|
|
269
|
+
if not self.config.register_first:
|
|
270
|
+
return
|
|
271
|
+
|
|
272
|
+
if self.req_socket is None:
|
|
273
|
+
logger.warning(
|
|
274
|
+
"DEALER socket not initialized, skipping worker registration"
|
|
275
|
+
)
|
|
276
|
+
return
|
|
277
|
+
|
|
278
|
+
logger.info("Pre-registering workers via REQ-REP...")
|
|
279
|
+
for instance in test_data.instances:
|
|
280
|
+
for worker in test_data.workers:
|
|
281
|
+
ip = "192.168.1.%d" % (worker + 1)
|
|
282
|
+
port = 10000 + worker
|
|
283
|
+
peer_port = 20000 + worker
|
|
284
|
+
peer_init_url = "tcp://%s:%d" % (ip, peer_port)
|
|
285
|
+
msg = RegisterMsg(
|
|
286
|
+
instance_id=instance,
|
|
287
|
+
worker_id=worker,
|
|
288
|
+
ip=ip,
|
|
289
|
+
port=port,
|
|
290
|
+
peer_init_url=peer_init_url,
|
|
291
|
+
)
|
|
292
|
+
try:
|
|
293
|
+
_, response = await self.send_request(msg)
|
|
294
|
+
self.registered_workers.append((instance, worker, ip, port))
|
|
295
|
+
# Extract heartbeat_url from first successful registration
|
|
296
|
+
# (only if not already configured)
|
|
297
|
+
if self.heartbeat_url is None and response:
|
|
298
|
+
self._process_register_response(response)
|
|
299
|
+
except (RuntimeError, zmq.ZMQError) as e:
|
|
300
|
+
logger.error(
|
|
301
|
+
"Failed to register worker %s-%d: %s", instance, worker, e
|
|
302
|
+
)
|
|
303
|
+
|
|
304
|
+
logger.info("Registered %d workers", len(self.registered_workers))
|
|
305
|
+
# Setup heartbeat socket after getting heartbeat_url (if not already setup)
|
|
306
|
+
if self.heartbeat_url and self.heartbeat_socket is None:
|
|
307
|
+
self._setup_heartbeat_socket()
|
|
308
|
+
await asyncio.sleep(0.5)
|
|
309
|
+
|
|
310
|
+
def _process_register_response(self, response: bytes):
|
|
311
|
+
"""Process RegisterRetMsg to extract heartbeat_url"""
|
|
312
|
+
try:
|
|
313
|
+
ret_msg = msgspec.msgpack.decode(response, type=RegisterRetMsg)
|
|
314
|
+
if ret_msg.extra_config and "heartbeat_url" in ret_msg.extra_config:
|
|
315
|
+
raw_url = ret_msg.extra_config["heartbeat_url"]
|
|
316
|
+
# Strip tcp:// prefix if present (get_zmq_socket adds it)
|
|
317
|
+
if raw_url.startswith("tcp://"):
|
|
318
|
+
raw_url = raw_url[6:]
|
|
319
|
+
# If benchmark connects to localhost but controller returns
|
|
320
|
+
# a different IP, use localhost for heartbeat as well
|
|
321
|
+
raw_url = self._normalize_heartbeat_url(raw_url)
|
|
322
|
+
self.heartbeat_url = raw_url
|
|
323
|
+
logger.info("Got heartbeat_url from register: %s", self.heartbeat_url)
|
|
324
|
+
except msgspec.DecodeError as e:
|
|
325
|
+
logger.warning("Failed to decode RegisterRetMsg: %s", e)
|
|
326
|
+
|
|
327
|
+
def _normalize_heartbeat_url(self, heartbeat_url: str) -> str:
|
|
328
|
+
"""Normalize heartbeat URL based on controller connection.
|
|
329
|
+
|
|
330
|
+
If benchmark connects to controller via localhost (127.0.0.1),
|
|
331
|
+
but heartbeat_url contains a different IP (e.g., from get_ip()),
|
|
332
|
+
replace it with 127.0.0.1 to ensure connectivity.
|
|
333
|
+
|
|
334
|
+
Args:
|
|
335
|
+
heartbeat_url: The heartbeat URL from controller (e.g., "10.0.0.1:7557")
|
|
336
|
+
|
|
337
|
+
Returns:
|
|
338
|
+
Normalized URL (e.g., "127.0.0.1:7557" if connecting locally)
|
|
339
|
+
"""
|
|
340
|
+
if ":" not in heartbeat_url:
|
|
341
|
+
return heartbeat_url
|
|
342
|
+
|
|
343
|
+
hb_host, hb_port = heartbeat_url.rsplit(":", 1)
|
|
344
|
+
|
|
345
|
+
# Check if we're connecting to controller via localhost
|
|
346
|
+
controller_host = self.config.controller_pull_url.split(":")[0]
|
|
347
|
+
if controller_host in ("127.0.0.1", "localhost"):
|
|
348
|
+
# If controller returns a non-localhost IP, use localhost instead
|
|
349
|
+
if hb_host not in ("127.0.0.1", "localhost"):
|
|
350
|
+
logger.info(
|
|
351
|
+
"Controller returned heartbeat IP %s, "
|
|
352
|
+
"but we're connecting locally. Using 127.0.0.1 instead.",
|
|
353
|
+
hb_host,
|
|
354
|
+
)
|
|
355
|
+
return "127.0.0.1:%s" % hb_port
|
|
356
|
+
|
|
357
|
+
return heartbeat_url
|
|
358
|
+
|
|
359
|
+
def _setup_heartbeat_socket(self):
|
|
360
|
+
"""Setup heartbeat DEALER socket after getting heartbeat_url from register"""
|
|
361
|
+
if not self.heartbeat_url or not self.context:
|
|
362
|
+
logger.warning(
|
|
363
|
+
"Cannot setup heartbeat socket: heartbeat_url=%s, context=%s",
|
|
364
|
+
self.heartbeat_url,
|
|
365
|
+
self.context is not None,
|
|
366
|
+
)
|
|
367
|
+
return
|
|
368
|
+
if self.heartbeat_socket is not None:
|
|
369
|
+
return # Already setup
|
|
370
|
+
logger.info(
|
|
371
|
+
"Setting up heartbeat DEALER socket to %s, "
|
|
372
|
+
"recv_timeout=%dms, send_timeout=%dms",
|
|
373
|
+
self.heartbeat_url,
|
|
374
|
+
DEFAULT_RECV_TIMEOUT_MS,
|
|
375
|
+
DEFAULT_SEND_TIMEOUT_MS,
|
|
376
|
+
)
|
|
377
|
+
self.heartbeat_socket = get_zmq_socket_with_timeout(
|
|
378
|
+
self.context,
|
|
379
|
+
self.heartbeat_url,
|
|
380
|
+
protocol="tcp",
|
|
381
|
+
role=zmq.DEALER,
|
|
382
|
+
bind_or_connect="connect",
|
|
383
|
+
recv_timeout_ms=DEFAULT_RECV_TIMEOUT_MS,
|
|
384
|
+
send_timeout_ms=DEFAULT_SEND_TIMEOUT_MS,
|
|
385
|
+
)
|
|
386
|
+
logger.info("Heartbeat socket created successfully")
|
|
387
|
+
|
|
388
|
+
async def send_heartbeat(self, message: Any) -> Tuple[float, Any]:
|
|
389
|
+
"""Send heartbeat via dedicated heartbeat DEALER socket
|
|
390
|
+
|
|
391
|
+
Args:
|
|
392
|
+
message: HeartbeatMsg to send
|
|
393
|
+
|
|
394
|
+
Returns:
|
|
395
|
+
Tuple of (time taken, response)
|
|
396
|
+
"""
|
|
397
|
+
if self.heartbeat_socket is None:
|
|
398
|
+
raise RuntimeError(
|
|
399
|
+
"Heartbeat socket not initialized. "
|
|
400
|
+
"heartbeat_url=%s. Register first to get heartbeat_url."
|
|
401
|
+
% self.heartbeat_url
|
|
402
|
+
)
|
|
403
|
+
start_time = time.time()
|
|
404
|
+
encoded_msg = msgspec.msgpack.encode(message)
|
|
405
|
+
try:
|
|
406
|
+
# DEALER socket: send [empty_frame, payload]
|
|
407
|
+
await self.heartbeat_socket.send_multipart([b"", encoded_msg])
|
|
408
|
+
frames = await self.heartbeat_socket.recv_multipart()
|
|
409
|
+
response = frames[-1]
|
|
410
|
+
return time.time() - start_time, response
|
|
411
|
+
except zmq.Again as e:
|
|
412
|
+
raise RuntimeError(
|
|
413
|
+
"Heartbeat timeout waiting for response from %s" % self.heartbeat_url
|
|
414
|
+
) from e
|
|
415
|
+
|
|
416
|
+
async def deregister_workers(self):
|
|
417
|
+
"""Deregister all workers after benchmark"""
|
|
418
|
+
if not self.registered_workers:
|
|
419
|
+
return
|
|
420
|
+
|
|
421
|
+
logger.info("Deregistering workers...")
|
|
422
|
+
messages = []
|
|
423
|
+
for instance, worker, ip, port in self.registered_workers:
|
|
424
|
+
msg = DeRegisterMsg(
|
|
425
|
+
instance_id=instance,
|
|
426
|
+
worker_id=worker,
|
|
427
|
+
ip=ip,
|
|
428
|
+
port=port,
|
|
429
|
+
)
|
|
430
|
+
messages.append(msg)
|
|
431
|
+
|
|
432
|
+
# Send in batches
|
|
433
|
+
for i in range(0, len(messages), DEFAULT_BATCH_SEND_SIZE):
|
|
434
|
+
batch = messages[i : i + DEFAULT_BATCH_SEND_SIZE]
|
|
435
|
+
await self.send_messages(batch)
|
|
436
|
+
|
|
437
|
+
logger.info("Deregistered %d workers", len(messages))
|
|
438
|
+
self.registered_workers.clear()
|
|
439
|
+
|
|
440
|
+
def _build_operation_distribution(self) -> List[str]:
|
|
441
|
+
"""Build operation distribution list based on percentages"""
|
|
442
|
+
operations = []
|
|
443
|
+
for op_name, percentage in self.config.operations.items():
|
|
444
|
+
count = int(DEFAULT_OP_DISTRIBUTION_BASE * percentage / 100)
|
|
445
|
+
operations.extend([op_name] * count)
|
|
446
|
+
random.shuffle(operations)
|
|
447
|
+
return operations
|
|
448
|
+
|
|
449
|
+
async def _execute_operation(
|
|
450
|
+
self, op_name: str, test_data: TestData
|
|
451
|
+
) -> Tuple[int, int, float, Optional[Exception]]:
|
|
452
|
+
"""Execute a single operation
|
|
453
|
+
|
|
454
|
+
Returns:
|
|
455
|
+
Tuple of (message_count, request_count, latency, error)
|
|
456
|
+
"""
|
|
457
|
+
handler = OPERATION_HANDLERS.get(op_name)
|
|
458
|
+
if not handler:
|
|
459
|
+
logger.warning("Unknown operation: %s", op_name)
|
|
460
|
+
return 0, 0, 0.0, ValueError("Unknown operation")
|
|
461
|
+
|
|
462
|
+
try:
|
|
463
|
+
msg = handler.create_message(self, test_data)
|
|
464
|
+
socket_type = handler.socket_type
|
|
465
|
+
|
|
466
|
+
if socket_type == SocketType.HEARTBEAT:
|
|
467
|
+
latency, _ = await self.send_heartbeat(msg)
|
|
468
|
+
elif socket_type == SocketType.DEALER:
|
|
469
|
+
latency, _ = await self.send_request(msg)
|
|
470
|
+
else: # SocketType.PUSH
|
|
471
|
+
msg_start = time.time()
|
|
472
|
+
await self.send_messages([msg])
|
|
473
|
+
latency = time.time() - msg_start
|
|
474
|
+
return handler.get_message_count(self), 1, latency, None
|
|
475
|
+
except Exception as e:
|
|
476
|
+
logger.error("Error in %s: %s", op_name, e)
|
|
477
|
+
return 0, 0, 0.0, e
|
|
478
|
+
|
|
479
|
+
async def run_benchmark(self):
|
|
480
|
+
"""Run the main benchmark"""
|
|
481
|
+
await self.setup()
|
|
482
|
+
|
|
483
|
+
try:
|
|
484
|
+
test_data = self.generate_test_data()
|
|
485
|
+
|
|
486
|
+
# Pre-register workers
|
|
487
|
+
await self.register_workers(test_data)
|
|
488
|
+
|
|
489
|
+
# Build operation distribution
|
|
490
|
+
operations = self._build_operation_distribution()
|
|
491
|
+
|
|
492
|
+
# Initialize tracking
|
|
493
|
+
latencies: Dict[str, List[float]] = {
|
|
494
|
+
op: [] for op in self.config.operations.keys()
|
|
495
|
+
}
|
|
496
|
+
errors: Dict[str, int] = {op: 0 for op in self.config.operations.keys()}
|
|
497
|
+
message_counts: Dict[str, int] = {
|
|
498
|
+
op: 0 for op in self.config.operations.keys()
|
|
499
|
+
}
|
|
500
|
+
request_counts: Dict[str, int] = {
|
|
501
|
+
op: 0 for op in self.config.operations.keys()
|
|
502
|
+
}
|
|
503
|
+
total_messages = 0
|
|
504
|
+
total_requests = 0
|
|
505
|
+
|
|
506
|
+
# Start monitoring
|
|
507
|
+
self.running = True
|
|
508
|
+
monitoring_task = asyncio.create_task(self.monitor_system())
|
|
509
|
+
|
|
510
|
+
start_time = time.time()
|
|
511
|
+
op_index = 0
|
|
512
|
+
|
|
513
|
+
logger.info("Starting benchmark for %d seconds...", self.config.duration)
|
|
514
|
+
|
|
515
|
+
while time.time() - start_time < self.config.duration:
|
|
516
|
+
# Get next operation
|
|
517
|
+
op_name = operations[op_index % len(operations)]
|
|
518
|
+
op_index += 1
|
|
519
|
+
|
|
520
|
+
msg_count, req_count, latency, error = await self._execute_operation(
|
|
521
|
+
op_name, test_data
|
|
522
|
+
)
|
|
523
|
+
total_messages += msg_count
|
|
524
|
+
total_requests += req_count
|
|
525
|
+
if error:
|
|
526
|
+
errors[op_name] += 1
|
|
527
|
+
else:
|
|
528
|
+
latencies[op_name].append(latency)
|
|
529
|
+
message_counts[op_name] += msg_count
|
|
530
|
+
request_counts[op_name] += req_count
|
|
531
|
+
|
|
532
|
+
# Small yield to prevent blocking
|
|
533
|
+
if op_index % 100 == 0:
|
|
534
|
+
await asyncio.sleep(0)
|
|
535
|
+
|
|
536
|
+
# Stop monitoring
|
|
537
|
+
self.running = False
|
|
538
|
+
monitoring_task.cancel()
|
|
539
|
+
try:
|
|
540
|
+
await monitoring_task
|
|
541
|
+
except asyncio.CancelledError:
|
|
542
|
+
pass
|
|
543
|
+
|
|
544
|
+
# Calculate results
|
|
545
|
+
total_time = time.time() - start_time
|
|
546
|
+
overall_qps = total_messages / total_time if total_time > 0 else 0
|
|
547
|
+
overall_rps = total_requests / total_time if total_time > 0 else 0
|
|
548
|
+
|
|
549
|
+
self.results.total_messages = total_messages
|
|
550
|
+
self.results.total_requests = total_requests
|
|
551
|
+
self.results.total_time = total_time
|
|
552
|
+
self.results.overall_qps = overall_qps
|
|
553
|
+
self.results.overall_rps = overall_rps
|
|
554
|
+
|
|
555
|
+
# Per-operation stats
|
|
556
|
+
for op_name in self.config.operations.keys():
|
|
557
|
+
if latencies[op_name]:
|
|
558
|
+
op_qps = (
|
|
559
|
+
message_counts[op_name] / total_time if total_time > 0 else 0
|
|
560
|
+
)
|
|
561
|
+
op_rps = (
|
|
562
|
+
request_counts[op_name] / total_time if total_time > 0 else 0
|
|
563
|
+
)
|
|
564
|
+
avg_latency = statistics.mean(latencies[op_name])
|
|
565
|
+
|
|
566
|
+
self.results.operations[op_name] = OperationStats(
|
|
567
|
+
qps=op_qps,
|
|
568
|
+
rps=op_rps,
|
|
569
|
+
avg_latency=avg_latency,
|
|
570
|
+
min_latency=min(latencies[op_name]),
|
|
571
|
+
max_latency=max(latencies[op_name]),
|
|
572
|
+
p95_latency=(
|
|
573
|
+
statistics.quantiles(latencies[op_name], n=20)[18]
|
|
574
|
+
if len(latencies[op_name]) >= 20
|
|
575
|
+
else max(latencies[op_name])
|
|
576
|
+
),
|
|
577
|
+
errors=errors[op_name],
|
|
578
|
+
)
|
|
579
|
+
|
|
580
|
+
# Deregister workers
|
|
581
|
+
await self.deregister_workers()
|
|
582
|
+
|
|
583
|
+
finally:
|
|
584
|
+
self.cleanup()
|
|
585
|
+
|
|
586
|
+
async def monitor_system(self):
|
|
587
|
+
"""Monitor system metrics during benchmark"""
|
|
588
|
+
while self.running:
|
|
589
|
+
try:
|
|
590
|
+
memory_usage = psutil.virtual_memory().percent
|
|
591
|
+
self.results.memory_usage.append(memory_usage)
|
|
592
|
+
except Exception as e:
|
|
593
|
+
logger.warning("Failed to get memory usage: %s", e)
|
|
594
|
+
await asyncio.sleep(1)
|
|
595
|
+
|
|
596
|
+
def print_results(self):
|
|
597
|
+
"""Print benchmark results"""
|
|
598
|
+
print("\n" + "=" * 80)
|
|
599
|
+
if self.config.num_processes > 1:
|
|
600
|
+
print(
|
|
601
|
+
"LMCache Controller ZMQ Benchmark Results (Process %d/%d)"
|
|
602
|
+
% (self.config.process_id + 1, self.config.num_processes)
|
|
603
|
+
)
|
|
604
|
+
else:
|
|
605
|
+
print("LMCache Controller ZMQ Benchmark Results")
|
|
606
|
+
print("=" * 80)
|
|
607
|
+
|
|
608
|
+
print("\nConfiguration:")
|
|
609
|
+
print(" Controller URL: %s" % self.config.controller_pull_url)
|
|
610
|
+
print(" Duration: %d seconds" % self.config.duration)
|
|
611
|
+
print(" Batch Size: %d" % self.config.batch_size)
|
|
612
|
+
print(" Operations: %s" % self.config.operations)
|
|
613
|
+
print(
|
|
614
|
+
" Instances: %d, Workers: %d, Locations: %d, Keys: %d"
|
|
615
|
+
% (
|
|
616
|
+
self.config.num_instances,
|
|
617
|
+
self.config.num_workers,
|
|
618
|
+
self.config.num_locations,
|
|
619
|
+
self.config.num_keys,
|
|
620
|
+
)
|
|
621
|
+
)
|
|
622
|
+
|
|
623
|
+
print("\nOverall Performance:")
|
|
624
|
+
print(" Total Requests: %d" % self.results.total_requests)
|
|
625
|
+
print(" Total Messages: %d" % self.results.total_messages)
|
|
626
|
+
print(" Total Time: %.2fs" % self.results.total_time)
|
|
627
|
+
print(" Overall RPS (Requests/sec): %.2f" % self.results.overall_rps)
|
|
628
|
+
print(" Overall QPS (Messages/sec): %.2f" % self.results.overall_qps)
|
|
629
|
+
|
|
630
|
+
print("\nPer-Operation Performance:")
|
|
631
|
+
for op_name in self.config.operations.keys():
|
|
632
|
+
if op_name in self.results.operations:
|
|
633
|
+
stats = self.results.operations[op_name]
|
|
634
|
+
print(" %s:" % op_name)
|
|
635
|
+
print(" RPS (Requests/sec): %.2f" % stats.rps)
|
|
636
|
+
print(" QPS (Messages/sec): %.2f" % stats.qps)
|
|
637
|
+
print(
|
|
638
|
+
" Latency - Avg: %.3fms, Min: %.3fms, Max: %.3fms, P95: %.3fms"
|
|
639
|
+
% (
|
|
640
|
+
stats.avg_latency * 1000,
|
|
641
|
+
stats.min_latency * 1000,
|
|
642
|
+
stats.max_latency * 1000,
|
|
643
|
+
stats.p95_latency * 1000,
|
|
644
|
+
)
|
|
645
|
+
)
|
|
646
|
+
print(" Errors: %d" % stats.errors)
|
|
647
|
+
|
|
648
|
+
print("\nSystem Metrics:")
|
|
649
|
+
if self.results.memory_usage:
|
|
650
|
+
avg_memory = statistics.mean(self.results.memory_usage)
|
|
651
|
+
max_memory = max(self.results.memory_usage)
|
|
652
|
+
print(
|
|
653
|
+
" Memory Usage - Avg: %.1f%%, Max: %.1f%%" % (avg_memory, max_memory)
|
|
654
|
+
)
|
|
655
|
+
|
|
656
|
+
print("=" * 80)
|
|
657
|
+
|
|
658
|
+
def get_results(self) -> BenchmarkResults:
|
|
659
|
+
"""Return benchmark results for aggregation"""
|
|
660
|
+
return self.results
|
|
@@ -0,0 +1,44 @@
|
|
|
1
|
+
"""Configuration for LMCache Controller ZMQ Benchmark"""
|
|
2
|
+
|
|
3
|
+
# SPDX-License-Identifier: Apache-2.0
|
|
4
|
+
|
|
5
|
+
# Standard
|
|
6
|
+
from dataclasses import dataclass, field
|
|
7
|
+
from typing import Dict, Optional
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
@dataclass
|
|
11
|
+
class ZMQBenchmarkConfig:
|
|
12
|
+
"""Configuration for ZMQ benchmark parameters"""
|
|
13
|
+
|
|
14
|
+
controller_pull_url: str
|
|
15
|
+
controller_reply_url: Optional[str]
|
|
16
|
+
duration: int
|
|
17
|
+
batch_size: int
|
|
18
|
+
num_instances: int
|
|
19
|
+
num_workers: int
|
|
20
|
+
num_locations: int
|
|
21
|
+
num_keys: int
|
|
22
|
+
controller_heartbeat_url: Optional[str] = None
|
|
23
|
+
num_hashes: int = 100
|
|
24
|
+
operations: Dict[str, float] = field(default_factory=dict)
|
|
25
|
+
heartbeat_interval: float = 1.0
|
|
26
|
+
register_first: bool = True
|
|
27
|
+
# Multi-process settings
|
|
28
|
+
num_processes: int = 1
|
|
29
|
+
process_id: int = 0
|
|
30
|
+
|
|
31
|
+
def __post_init__(self):
|
|
32
|
+
if not self.operations:
|
|
33
|
+
# Default: 70% admit, 25% evict, 5% heartbeat
|
|
34
|
+
self.operations = {
|
|
35
|
+
"admit": 70.0,
|
|
36
|
+
"evict": 25.0,
|
|
37
|
+
"heartbeat": 5.0,
|
|
38
|
+
}
|
|
39
|
+
# Validate operation percentages sum to 100
|
|
40
|
+
total_percentage = sum(self.operations.values())
|
|
41
|
+
if abs(total_percentage - 100.0) > 0.01:
|
|
42
|
+
raise ValueError(
|
|
43
|
+
"Operation percentages must sum to 100, got: %s" % total_percentage
|
|
44
|
+
)
|