lmcache-cli 0.4.5.dev0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- lmcache/__init__.py +84 -0
- lmcache/_version.py +24 -0
- lmcache/cli/__init__.py +1 -0
- lmcache/cli/commands/__init__.py +34 -0
- lmcache/cli/commands/base.py +157 -0
- lmcache/cli/commands/bench/__init__.py +557 -0
- lmcache/cli/commands/bench/engine_bench/__init__.py +1 -0
- lmcache/cli/commands/bench/engine_bench/config.py +245 -0
- lmcache/cli/commands/bench/engine_bench/interactive/__init__.py +274 -0
- lmcache/cli/commands/bench/engine_bench/interactive/config.json +10 -0
- lmcache/cli/commands/bench/engine_bench/interactive/schema.py +352 -0
- lmcache/cli/commands/bench/engine_bench/interactive/state.py +327 -0
- lmcache/cli/commands/bench/engine_bench/interactive/terminal.py +291 -0
- lmcache/cli/commands/bench/engine_bench/progress.py +145 -0
- lmcache/cli/commands/bench/engine_bench/request_sender.py +232 -0
- lmcache/cli/commands/bench/engine_bench/stats.py +275 -0
- lmcache/cli/commands/bench/engine_bench/workloads/__init__.py +153 -0
- lmcache/cli/commands/bench/engine_bench/workloads/base.py +122 -0
- lmcache/cli/commands/bench/engine_bench/workloads/long_doc_permutator.py +435 -0
- lmcache/cli/commands/bench/engine_bench/workloads/long_doc_qa.py +281 -0
- lmcache/cli/commands/bench/engine_bench/workloads/multi_round_chat.py +337 -0
- lmcache/cli/commands/bench/engine_bench/workloads/random_prefill.py +178 -0
- lmcache/cli/commands/describe.py +310 -0
- lmcache/cli/commands/kvcache.py +133 -0
- lmcache/cli/commands/mock.py +75 -0
- lmcache/cli/commands/ping.py +113 -0
- lmcache/cli/commands/query/__init__.py +155 -0
- lmcache/cli/commands/query/prompt.py +134 -0
- lmcache/cli/commands/query/request.py +357 -0
- lmcache/cli/commands/server.py +99 -0
- lmcache/cli/commands/tool/__init__.py +63 -0
- lmcache/cli/commands/tool/cache_simulator.py +113 -0
- lmcache/cli/commands/trace/__init__.py +505 -0
- lmcache/cli/commands/trace/dispatch.py +249 -0
- lmcache/cli/commands/trace/driver.py +372 -0
- lmcache/cli/commands/trace/stats.py +289 -0
- lmcache/cli/documents/lmcache.txt +11 -0
- lmcache/cli/main.py +42 -0
- lmcache/cli/metrics/__init__.py +29 -0
- lmcache/cli/metrics/formatter.py +171 -0
- lmcache/cli/metrics/handler.py +94 -0
- lmcache/cli/metrics/metrics.py +161 -0
- lmcache/cli/metrics/section.py +77 -0
- lmcache/connections.py +173 -0
- lmcache/integration/__init__.py +2 -0
- lmcache/integration/base_service_factory.py +165 -0
- lmcache/integration/request_telemetry/__init__.py +1 -0
- lmcache/integration/request_telemetry/base.py +51 -0
- lmcache/integration/request_telemetry/factory.py +113 -0
- lmcache/integration/request_telemetry/fastapi.py +109 -0
- lmcache/integration/request_telemetry/noop.py +35 -0
- lmcache/integration/sglang/__init__.py +2 -0
- lmcache/integration/sglang/sglang_adapter.py +326 -0
- lmcache/integration/sglang/utils.py +39 -0
- lmcache/integration/vllm/__init__.py +1 -0
- lmcache/integration/vllm/lmcache_connector_v1.py +213 -0
- lmcache/integration/vllm/lmcache_connector_v1_085.py +150 -0
- lmcache/integration/vllm/lmcache_mp_connector_0180.py +1072 -0
- lmcache/integration/vllm/tests/test_mm_hash_utils.py +112 -0
- lmcache/integration/vllm/utils.py +433 -0
- lmcache/integration/vllm/vllm_multi_process_adapter.py +1090 -0
- lmcache/integration/vllm/vllm_service_factory.py +339 -0
- lmcache/integration/vllm/vllm_v1_adapter.py +1713 -0
- lmcache/logging.py +107 -0
- lmcache/native_storage_ops.pyi +230 -0
- lmcache/non_cuda_equivalents.py +1424 -0
- lmcache/observability.py +1958 -0
- lmcache/storage_backend/serde/__init__.py +1 -0
- lmcache/storage_backend/serde/cachegen_basics.py +210 -0
- lmcache/storage_backend/serde/cachegen_decoder.py +207 -0
- lmcache/storage_backend/serde/cachegen_encoder.py +394 -0
- lmcache/storage_backend/serde/serde.py +75 -0
- lmcache/tools/__init__.py +1 -0
- lmcache/tools/cache_simulator/README.md +392 -0
- lmcache/tools/cache_simulator/__init__.py +1 -0
- lmcache/tools/cache_simulator/docs/simulate_example.png +0 -0
- lmcache/tools/cache_simulator/docs/sweep_example.png +0 -0
- lmcache/tools/cache_simulator/gen_bench_dataset.py +360 -0
- lmcache/tools/cache_simulator/lru_cache.py +124 -0
- lmcache/tools/cache_simulator/plot_hit_rate.py +231 -0
- lmcache/tools/cache_simulator/simulator.py +795 -0
- lmcache/tools/controller_benchmark/README.md +161 -0
- lmcache/tools/controller_benchmark/__init__.py +1 -0
- lmcache/tools/controller_benchmark/__main__.py +331 -0
- lmcache/tools/controller_benchmark/benchmark.py +660 -0
- lmcache/tools/controller_benchmark/config.py +44 -0
- lmcache/tools/controller_benchmark/constants.py +10 -0
- lmcache/tools/controller_benchmark/handlers/__init__.py +46 -0
- lmcache/tools/controller_benchmark/handlers/admit.py +52 -0
- lmcache/tools/controller_benchmark/handlers/base.py +47 -0
- lmcache/tools/controller_benchmark/handlers/deregister.py +49 -0
- lmcache/tools/controller_benchmark/handlers/evict.py +52 -0
- lmcache/tools/controller_benchmark/handlers/heartbeat.py +56 -0
- lmcache/tools/controller_benchmark/handlers/p2p_lookup.py +47 -0
- lmcache/tools/controller_benchmark/handlers/register.py +56 -0
- lmcache/tools/mp_status_viewer/__init__.py +1 -0
- lmcache/tools/mp_status_viewer/__main__.py +95 -0
- lmcache/usage_context.py +417 -0
- lmcache/utils.py +665 -0
- lmcache/v1/__init__.py +2 -0
- lmcache/v1/api_server/__init__.py +2 -0
- lmcache/v1/api_server/__main__.py +537 -0
- lmcache/v1/basic_check.py +112 -0
- lmcache/v1/cache_controller/__init__.py +9 -0
- lmcache/v1/cache_controller/commands/__init__.py +15 -0
- lmcache/v1/cache_controller/commands/base.py +35 -0
- lmcache/v1/cache_controller/commands/full_sync.py +49 -0
- lmcache/v1/cache_controller/config.py +176 -0
- lmcache/v1/cache_controller/controller_manager.py +535 -0
- lmcache/v1/cache_controller/controllers/__init__.py +11 -0
- lmcache/v1/cache_controller/controllers/full_sync_tracker.py +473 -0
- lmcache/v1/cache_controller/controllers/kv_controller.py +439 -0
- lmcache/v1/cache_controller/controllers/registration_controller.py +282 -0
- lmcache/v1/cache_controller/executor.py +463 -0
- lmcache/v1/cache_controller/frontend/static/css/style.css +201 -0
- lmcache/v1/cache_controller/frontend/static/img/logo.png +0 -0
- lmcache/v1/cache_controller/frontend/static/index.html +234 -0
- lmcache/v1/cache_controller/frontend/static/js/controller_app.js +660 -0
- lmcache/v1/cache_controller/full_sync_sender.py +475 -0
- lmcache/v1/cache_controller/locks.py +149 -0
- lmcache/v1/cache_controller/message.py +828 -0
- lmcache/v1/cache_controller/observability.py +208 -0
- lmcache/v1/cache_controller/utils.py +679 -0
- lmcache/v1/cache_controller/worker.py +665 -0
- lmcache/v1/cache_engine.py +2058 -0
- lmcache/v1/cache_interface.py +19 -0
- lmcache/v1/check/__init__.py +74 -0
- lmcache/v1/check/check_mode_gen.py +86 -0
- lmcache/v1/check/check_mode_test_l2_adapter.py +284 -0
- lmcache/v1/check/check_mode_test_remote.py +155 -0
- lmcache/v1/check/check_mode_test_storage_manager.py +142 -0
- lmcache/v1/check/utils.py +571 -0
- lmcache/v1/compute/__init__.py +2 -0
- lmcache/v1/compute/attention/__init__.py +0 -0
- lmcache/v1/compute/attention/abstract.py +39 -0
- lmcache/v1/compute/attention/flash_attn.py +129 -0
- lmcache/v1/compute/attention/flash_infer_sparse.py +284 -0
- lmcache/v1/compute/attention/metadata.py +85 -0
- lmcache/v1/compute/attention/utils.py +14 -0
- lmcache/v1/compute/blend/__init__.py +7 -0
- lmcache/v1/compute/blend/blender.py +168 -0
- lmcache/v1/compute/blend/metadata.py +34 -0
- lmcache/v1/compute/blend/utils.py +63 -0
- lmcache/v1/compute/models/__init__.py +0 -0
- lmcache/v1/compute/models/base.py +141 -0
- lmcache/v1/compute/models/llama.py +9 -0
- lmcache/v1/compute/models/qwen3.py +24 -0
- lmcache/v1/compute/models/utils.py +68 -0
- lmcache/v1/compute/positional_encoding.py +199 -0
- lmcache/v1/config.py +848 -0
- lmcache/v1/config_base.py +848 -0
- lmcache/v1/distributed/api.py +248 -0
- lmcache/v1/distributed/config.py +321 -0
- lmcache/v1/distributed/error.py +64 -0
- lmcache/v1/distributed/eviction.py +192 -0
- lmcache/v1/distributed/eviction_policy/__init__.py +21 -0
- lmcache/v1/distributed/eviction_policy/factory.py +27 -0
- lmcache/v1/distributed/eviction_policy/lru.py +244 -0
- lmcache/v1/distributed/eviction_policy/noop.py +50 -0
- lmcache/v1/distributed/internal_api.py +170 -0
- lmcache/v1/distributed/l1_manager.py +835 -0
- lmcache/v1/distributed/l2_adapters/__init__.py +67 -0
- lmcache/v1/distributed/l2_adapters/base.py +360 -0
- lmcache/v1/distributed/l2_adapters/config.py +385 -0
- lmcache/v1/distributed/l2_adapters/factory.py +205 -0
- lmcache/v1/distributed/l2_adapters/fs_l2_adapter.py +747 -0
- lmcache/v1/distributed/l2_adapters/fs_native_l2_adapter.py +167 -0
- lmcache/v1/distributed/l2_adapters/mock_l2_adapter.py +516 -0
- lmcache/v1/distributed/l2_adapters/mooncake_store_l2_adapter.py +135 -0
- lmcache/v1/distributed/l2_adapters/native_connector_l2_adapter.py +468 -0
- lmcache/v1/distributed/l2_adapters/native_plugin_l2_adapter.py +199 -0
- lmcache/v1/distributed/l2_adapters/nixl_store_dynamic_l2_adapter.py +831 -0
- lmcache/v1/distributed/l2_adapters/nixl_store_l2_adapter.py +983 -0
- lmcache/v1/distributed/l2_adapters/plugin_l2_adapter.py +210 -0
- lmcache/v1/distributed/l2_adapters/resp_l2_adapter.py +176 -0
- lmcache/v1/distributed/memory_manager.py +179 -0
- lmcache/v1/distributed/storage_controller.py +39 -0
- lmcache/v1/distributed/storage_controllers/__init__.py +43 -0
- lmcache/v1/distributed/storage_controllers/eviction_controller.py +242 -0
- lmcache/v1/distributed/storage_controllers/prefetch_controller.py +830 -0
- lmcache/v1/distributed/storage_controllers/prefetch_policy.py +193 -0
- lmcache/v1/distributed/storage_controllers/store_controller.py +452 -0
- lmcache/v1/distributed/storage_controllers/store_policy.py +213 -0
- lmcache/v1/distributed/storage_manager.py +532 -0
- lmcache/v1/event_manager.py +145 -0
- lmcache/v1/exceptions/__init__.py +16 -0
- lmcache/v1/gpu_connector/__init__.py +126 -0
- lmcache/v1/gpu_connector/gpu_connectors.py +1906 -0
- lmcache/v1/gpu_connector/gpu_ops.py +85 -0
- lmcache/v1/gpu_connector/hpu_connector.py +326 -0
- lmcache/v1/gpu_connector/mock_gpu_connector.py +67 -0
- lmcache/v1/gpu_connector/utils.py +890 -0
- lmcache/v1/gpu_connector/xpu_connectors.py +916 -0
- lmcache/v1/health_monitor/__init__.py +1 -0
- lmcache/v1/health_monitor/base.py +587 -0
- lmcache/v1/health_monitor/checks/__init__.py +1 -0
- lmcache/v1/health_monitor/checks/remote_backend_check.py +304 -0
- lmcache/v1/health_monitor/constants.py +36 -0
- lmcache/v1/internal_api_server/__init__.py +0 -0
- lmcache/v1/internal_api_server/api_registry.py +59 -0
- lmcache/v1/internal_api_server/api_server.py +120 -0
- lmcache/v1/internal_api_server/common/__init__.py +1 -0
- lmcache/v1/internal_api_server/common/env_api.py +22 -0
- lmcache/v1/internal_api_server/common/loglevel_api.py +57 -0
- lmcache/v1/internal_api_server/common/metrics_api.py +29 -0
- lmcache/v1/internal_api_server/common/periodic_thread_api.py +138 -0
- lmcache/v1/internal_api_server/common/run_script_api.py +73 -0
- lmcache/v1/internal_api_server/common/thread_api.py +63 -0
- lmcache/v1/internal_api_server/controller/__init__.py +1 -0
- lmcache/v1/internal_api_server/controller/key_stats_api.py +81 -0
- lmcache/v1/internal_api_server/controller/worker_info_api.py +136 -0
- lmcache/v1/internal_api_server/utils.py +43 -0
- lmcache/v1/internal_api_server/vllm/__init__.py +1 -0
- lmcache/v1/internal_api_server/vllm/backend_api.py +221 -0
- lmcache/v1/internal_api_server/vllm/bypass_api.py +204 -0
- lmcache/v1/internal_api_server/vllm/cache_api.py +895 -0
- lmcache/v1/internal_api_server/vllm/chunk_statistics_api.py +141 -0
- lmcache/v1/internal_api_server/vllm/conf_api.py +147 -0
- lmcache/v1/internal_api_server/vllm/freeze_api.py +172 -0
- lmcache/v1/internal_api_server/vllm/hot_cache_api.py +184 -0
- lmcache/v1/internal_api_server/vllm/inference_api.py +65 -0
- lmcache/v1/internal_api_server/vllm/load_fs_chunks_api.py +320 -0
- lmcache/v1/internal_api_server/vllm/lookup_api.py +145 -0
- lmcache/v1/internal_api_server/vllm/version_api.py +25 -0
- lmcache/v1/kv_layer_groups.py +267 -0
- lmcache/v1/lazy_memory_allocator.py +284 -0
- lmcache/v1/lookup_client/__init__.py +25 -0
- lmcache/v1/lookup_client/abstract_client.py +77 -0
- lmcache/v1/lookup_client/async_lookup_message.py +50 -0
- lmcache/v1/lookup_client/chunk_statistics_lookup_client.py +200 -0
- lmcache/v1/lookup_client/factory.py +251 -0
- lmcache/v1/lookup_client/hit_limit_lookup_client.py +86 -0
- lmcache/v1/lookup_client/lmcache_async_lookup_client.py +407 -0
- lmcache/v1/lookup_client/lmcache_lookup_client.py +285 -0
- lmcache/v1/lookup_client/lmcache_lookup_client_bypass.py +99 -0
- lmcache/v1/lookup_client/mooncake_lookup_client.py +87 -0
- lmcache/v1/lookup_client/record_strategies/__init__.py +77 -0
- lmcache/v1/lookup_client/record_strategies/base.py +327 -0
- lmcache/v1/lookup_client/record_strategies/file_hash.py +130 -0
- lmcache/v1/lookup_client/record_strategies/memory_bloom_filter.py +81 -0
- lmcache/v1/manager.py +539 -0
- lmcache/v1/memory_management.py +2619 -0
- lmcache/v1/metadata.py +114 -0
- lmcache/v1/mp_observability/AGENTS.override.md +21 -0
- lmcache/v1/mp_observability/README.md +204 -0
- lmcache/v1/mp_observability/config.py +340 -0
- lmcache/v1/mp_observability/event.py +100 -0
- lmcache/v1/mp_observability/event_bus.py +313 -0
- lmcache/v1/mp_observability/otel_init.py +145 -0
- lmcache/v1/mp_observability/subscribers/__init__.py +28 -0
- lmcache/v1/mp_observability/subscribers/logging/__init__.py +19 -0
- lmcache/v1/mp_observability/subscribers/logging/l1.py +56 -0
- lmcache/v1/mp_observability/subscribers/logging/l2.py +73 -0
- lmcache/v1/mp_observability/subscribers/logging/lookup_hash.py +209 -0
- lmcache/v1/mp_observability/subscribers/logging/mp_server.py +90 -0
- lmcache/v1/mp_observability/subscribers/logging/sm.py +59 -0
- lmcache/v1/mp_observability/subscribers/metrics/__init__.py +20 -0
- lmcache/v1/mp_observability/subscribers/metrics/l0_lifecycle.py +290 -0
- lmcache/v1/mp_observability/subscribers/metrics/l1.py +55 -0
- lmcache/v1/mp_observability/subscribers/metrics/l1_lifecycle.py +166 -0
- lmcache/v1/mp_observability/subscribers/metrics/l2.py +121 -0
- lmcache/v1/mp_observability/subscribers/metrics/sm.py +69 -0
- lmcache/v1/mp_observability/subscribers/tracing/__init__.py +12 -0
- lmcache/v1/mp_observability/subscribers/tracing/mp_server.py +333 -0
- lmcache/v1/mp_observability/subscribers/tracing/span_registry.py +148 -0
- lmcache/v1/mp_observability/trace/__init__.py +50 -0
- lmcache/v1/mp_observability/trace/codecs.py +255 -0
- lmcache/v1/mp_observability/trace/decorator.py +147 -0
- lmcache/v1/mp_observability/trace/format.py +132 -0
- lmcache/v1/mp_observability/trace/lifecycle.py +83 -0
- lmcache/v1/mp_observability/trace/reader.py +167 -0
- lmcache/v1/mp_observability/trace/recorder.py +300 -0
- lmcache/v1/multiprocess/__init__.py +0 -0
- lmcache/v1/multiprocess/affinity_pool.py +102 -0
- lmcache/v1/multiprocess/blend_server_v2.py +891 -0
- lmcache/v1/multiprocess/config.py +253 -0
- lmcache/v1/multiprocess/custom_types.py +281 -0
- lmcache/v1/multiprocess/futures.py +194 -0
- lmcache/v1/multiprocess/gpu_context.py +511 -0
- lmcache/v1/multiprocess/http_server.py +235 -0
- lmcache/v1/multiprocess/mp_runtime_plugin_launcher.py +130 -0
- lmcache/v1/multiprocess/mq.py +732 -0
- lmcache/v1/multiprocess/protocol.py +86 -0
- lmcache/v1/multiprocess/protocols/README.md +213 -0
- lmcache/v1/multiprocess/protocols/__init__.py +127 -0
- lmcache/v1/multiprocess/protocols/base.py +89 -0
- lmcache/v1/multiprocess/protocols/blend.py +109 -0
- lmcache/v1/multiprocess/protocols/blend_v2.py +57 -0
- lmcache/v1/multiprocess/protocols/controller.py +53 -0
- lmcache/v1/multiprocess/protocols/debug.py +34 -0
- lmcache/v1/multiprocess/protocols/engine.py +146 -0
- lmcache/v1/multiprocess/protocols/observability.py +39 -0
- lmcache/v1/multiprocess/server.py +1134 -0
- lmcache/v1/multiprocess/session.py +190 -0
- lmcache/v1/multiprocess/token_hasher.py +441 -0
- lmcache/v1/offload_server/__init__.py +17 -0
- lmcache/v1/offload_server/abstract_server.py +37 -0
- lmcache/v1/offload_server/message.py +30 -0
- lmcache/v1/offload_server/zmq_server.py +122 -0
- lmcache/v1/periodic_thread.py +579 -0
- lmcache/v1/pin_monitor.py +246 -0
- lmcache/v1/plugin/__init__.py +0 -0
- lmcache/v1/plugin/runtime_plugin_launcher.py +211 -0
- lmcache/v1/protocol.py +317 -0
- lmcache/v1/rpc/__init__.py +17 -0
- lmcache/v1/rpc/transport.py +105 -0
- lmcache/v1/rpc/zmq_transport.py +213 -0
- lmcache/v1/rpc_utils.py +165 -0
- lmcache/v1/server/__init__.py +2 -0
- lmcache/v1/server/__main__.py +170 -0
- lmcache/v1/server/storage_backend/__init__.py +21 -0
- lmcache/v1/server/storage_backend/abstract_backend.py +80 -0
- lmcache/v1/server/storage_backend/local_backend.py +75 -0
- lmcache/v1/server/utils.py +21 -0
- lmcache/v1/standalone/__init__.py +1 -0
- lmcache/v1/standalone/__main__.py +583 -0
- lmcache/v1/standalone/manager.py +80 -0
- lmcache/v1/standalone/standalone_service_factory.py +86 -0
- lmcache/v1/storage_backend/__init__.py +313 -0
- lmcache/v1/storage_backend/abstract_backend.py +445 -0
- lmcache/v1/storage_backend/audit_backend.py +233 -0
- lmcache/v1/storage_backend/batched_message_sender.py +222 -0
- lmcache/v1/storage_backend/cache_policy/__init__.py +45 -0
- lmcache/v1/storage_backend/cache_policy/base_policy.py +87 -0
- lmcache/v1/storage_backend/cache_policy/fifo.py +58 -0
- lmcache/v1/storage_backend/cache_policy/lfu.py +105 -0
- lmcache/v1/storage_backend/cache_policy/lru.py +81 -0
- lmcache/v1/storage_backend/cache_policy/mru.py +61 -0
- lmcache/v1/storage_backend/connector/__init__.py +443 -0
- lmcache/v1/storage_backend/connector/audit_adapter.py +77 -0
- lmcache/v1/storage_backend/connector/audit_connector.py +320 -0
- lmcache/v1/storage_backend/connector/base_connector.py +379 -0
- lmcache/v1/storage_backend/connector/blackhole_adapter.py +21 -0
- lmcache/v1/storage_backend/connector/blackhole_connector.py +37 -0
- lmcache/v1/storage_backend/connector/eic_adapter.py +31 -0
- lmcache/v1/storage_backend/connector/eic_connector.py +757 -0
- lmcache/v1/storage_backend/connector/external_adapter.py +79 -0
- lmcache/v1/storage_backend/connector/fs_adapter.py +51 -0
- lmcache/v1/storage_backend/connector/fs_connector.py +403 -0
- lmcache/v1/storage_backend/connector/infinistore_adapter.py +56 -0
- lmcache/v1/storage_backend/connector/infinistore_connector.py +177 -0
- lmcache/v1/storage_backend/connector/instrumented_connector.py +219 -0
- lmcache/v1/storage_backend/connector/lm_adapter.py +31 -0
- lmcache/v1/storage_backend/connector/lm_connector.py +176 -0
- lmcache/v1/storage_backend/connector/mock_adapter.py +57 -0
- lmcache/v1/storage_backend/connector/mock_connector.py +349 -0
- lmcache/v1/storage_backend/connector/mooncakestore_adapter.py +43 -0
- lmcache/v1/storage_backend/connector/mooncakestore_connector.py +614 -0
- lmcache/v1/storage_backend/connector/redis_adapter.py +181 -0
- lmcache/v1/storage_backend/connector/redis_connector.py +828 -0
- lmcache/v1/storage_backend/connector/s3_adapter.py +59 -0
- lmcache/v1/storage_backend/connector/s3_connector.py +699 -0
- lmcache/v1/storage_backend/connector/sagemaker_hyperpod_adapter.py +233 -0
- lmcache/v1/storage_backend/connector/sagemaker_hyperpod_connector.py +987 -0
- lmcache/v1/storage_backend/connector/valkey_adapter.py +114 -0
- lmcache/v1/storage_backend/connector/valkey_connector.py +627 -0
- lmcache/v1/storage_backend/gds_backend.py +1199 -0
- lmcache/v1/storage_backend/job_executor/__init__.py +0 -0
- lmcache/v1/storage_backend/job_executor/base_executor.py +34 -0
- lmcache/v1/storage_backend/job_executor/pq_executor.py +235 -0
- lmcache/v1/storage_backend/local_cpu_backend.py +810 -0
- lmcache/v1/storage_backend/local_disk_backend.py +656 -0
- lmcache/v1/storage_backend/maru_backend.py +734 -0
- lmcache/v1/storage_backend/naive_serde/__init__.py +50 -0
- lmcache/v1/storage_backend/naive_serde/cachegen_basics.py +133 -0
- lmcache/v1/storage_backend/naive_serde/cachegen_decoder.py +135 -0
- lmcache/v1/storage_backend/naive_serde/cachegen_encoder.py +83 -0
- lmcache/v1/storage_backend/naive_serde/kivi_serde.py +22 -0
- lmcache/v1/storage_backend/naive_serde/naive_serde.py +18 -0
- lmcache/v1/storage_backend/naive_serde/serde.py +37 -0
- lmcache/v1/storage_backend/native_clients/connector_client_base.py +165 -0
- lmcache/v1/storage_backend/native_clients/resp_client.py +35 -0
- lmcache/v1/storage_backend/nixl_storage_backend.py +1400 -0
- lmcache/v1/storage_backend/p2p_backend.py +788 -0
- lmcache/v1/storage_backend/path_sharder.py +117 -0
- lmcache/v1/storage_backend/pd_backend.py +646 -0
- lmcache/v1/storage_backend/plugins/dax_backend.py +1443 -0
- lmcache/v1/storage_backend/plugins/rust_raw_block_backend.py +1361 -0
- lmcache/v1/storage_backend/remote_backend.py +624 -0
- lmcache/v1/storage_backend/resp_client.py +227 -0
- lmcache/v1/storage_backend/storage_backend_listener.py +19 -0
- lmcache/v1/storage_backend/storage_manager.py +1352 -0
- lmcache/v1/system_detection.py +110 -0
- lmcache/v1/token_database.py +551 -0
- lmcache/v1/transfer_channel/__init__.py +83 -0
- lmcache/v1/transfer_channel/abstract.py +285 -0
- lmcache/v1/transfer_channel/mock_memory_channel.py +156 -0
- lmcache/v1/transfer_channel/nixl_channel.py +639 -0
- lmcache/v1/transfer_channel/py_socket_channel.py +260 -0
- lmcache/v1/transfer_channel/transfer_utils.py +63 -0
- lmcache/v1/utils/__init__.py +1 -0
- lmcache/v1/utils/bloom_filter.py +109 -0
- lmcache/v1/utils/cache_utils.py +125 -0
- lmcache_cli-0.4.5.dev0.dist-info/METADATA +185 -0
- lmcache_cli-0.4.5.dev0.dist-info/RECORD +399 -0
- lmcache_cli-0.4.5.dev0.dist-info/WHEEL +5 -0
- lmcache_cli-0.4.5.dev0.dist-info/entry_points.txt +2 -0
- lmcache_cli-0.4.5.dev0.dist-info/licenses/LICENSE +201 -0
- lmcache_cli-0.4.5.dev0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,245 @@
|
|
|
1
|
+
# SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
"""Configuration dataclasses and CLI-arg parsing for ``lmcache bench engine``."""
|
|
3
|
+
|
|
4
|
+
# Standard
|
|
5
|
+
from dataclasses import dataclass
|
|
6
|
+
import argparse
|
|
7
|
+
import json
|
|
8
|
+
import os
|
|
9
|
+
import urllib.error
|
|
10
|
+
import urllib.request
|
|
11
|
+
|
|
12
|
+
# Third Party
|
|
13
|
+
from openai import OpenAI
|
|
14
|
+
|
|
15
|
+
# First Party
|
|
16
|
+
from lmcache.logging import init_logger
|
|
17
|
+
|
|
18
|
+
logger = init_logger(__name__)
|
|
19
|
+
|
|
20
|
+
_GB = 1024**3
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
@dataclass
|
|
24
|
+
class EngineBenchConfig:
|
|
25
|
+
"""Top-level config produced from CLI args, interactive mode, or saved config.
|
|
26
|
+
|
|
27
|
+
Contains only general benchmark parameters. Workload-specific configs
|
|
28
|
+
(e.g., ``LongDocQAConfig``) live in their respective workload modules
|
|
29
|
+
and are resolved by the workload factory.
|
|
30
|
+
"""
|
|
31
|
+
|
|
32
|
+
engine_url: str
|
|
33
|
+
model: str
|
|
34
|
+
workload: str
|
|
35
|
+
kv_cache_volume_gb: float
|
|
36
|
+
tokens_per_gb_kvcache: int
|
|
37
|
+
seed: int
|
|
38
|
+
output_dir: str
|
|
39
|
+
export_csv: bool
|
|
40
|
+
export_json: bool
|
|
41
|
+
quiet: bool
|
|
42
|
+
|
|
43
|
+
def __post_init__(self) -> None:
|
|
44
|
+
if not self.engine_url:
|
|
45
|
+
raise ValueError("engine_url must be non-empty")
|
|
46
|
+
if self.kv_cache_volume_gb <= 0:
|
|
47
|
+
raise ValueError(
|
|
48
|
+
f"kv_cache_volume_gb must be positive, got {self.kv_cache_volume_gb}"
|
|
49
|
+
)
|
|
50
|
+
if self.tokens_per_gb_kvcache <= 0:
|
|
51
|
+
raise ValueError(
|
|
52
|
+
f"tokens_per_gb_kvcache must be positive, "
|
|
53
|
+
f"got {self.tokens_per_gb_kvcache}"
|
|
54
|
+
)
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def auto_detect_model(engine_url: str) -> str:
|
|
58
|
+
"""Fetch the first model ID from the engine's ``/v1/models`` endpoint.
|
|
59
|
+
|
|
60
|
+
Args:
|
|
61
|
+
engine_url: Base URL of the inference engine (e.g.,
|
|
62
|
+
``http://localhost:8000``).
|
|
63
|
+
|
|
64
|
+
Returns:
|
|
65
|
+
The model ID string.
|
|
66
|
+
|
|
67
|
+
Raises:
|
|
68
|
+
RuntimeError: If the engine is unreachable or returns no models.
|
|
69
|
+
"""
|
|
70
|
+
base_url = engine_url.rstrip("/")
|
|
71
|
+
if not base_url.startswith(("http://", "https://")):
|
|
72
|
+
base_url = f"http://{base_url}"
|
|
73
|
+
if not base_url.endswith("/v1"):
|
|
74
|
+
base_url += "/v1"
|
|
75
|
+
|
|
76
|
+
api_key = os.getenv("OPENAI_API_KEY", "sk-dummy")
|
|
77
|
+
logger.debug("Auto-detecting model from %s/models", base_url)
|
|
78
|
+
|
|
79
|
+
try:
|
|
80
|
+
client = OpenAI(base_url=base_url, api_key=api_key)
|
|
81
|
+
models = client.models.list()
|
|
82
|
+
except Exception as e:
|
|
83
|
+
raise RuntimeError(f"Failed to fetch models from {base_url}/models: {e}") from e
|
|
84
|
+
|
|
85
|
+
if not models.data:
|
|
86
|
+
raise RuntimeError(
|
|
87
|
+
f"No models returned by {base_url}/models; pass --model explicitly."
|
|
88
|
+
)
|
|
89
|
+
|
|
90
|
+
model_id = models.data[0].id
|
|
91
|
+
logger.debug("Auto-detected model: %s", model_id)
|
|
92
|
+
return model_id
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
def _fetch_lmcache_status(lmcache_url: str) -> dict:
|
|
96
|
+
"""Fetch ``/api/status`` from the LMCache HTTP server.
|
|
97
|
+
|
|
98
|
+
Returns:
|
|
99
|
+
Parsed JSON response.
|
|
100
|
+
|
|
101
|
+
Raises:
|
|
102
|
+
RuntimeError: If the server is unreachable.
|
|
103
|
+
"""
|
|
104
|
+
url = lmcache_url.rstrip("/")
|
|
105
|
+
if not url.startswith(("http://", "https://")):
|
|
106
|
+
url = f"http://{url}"
|
|
107
|
+
status_url = f"{url}/api/status"
|
|
108
|
+
|
|
109
|
+
logger.debug("Fetching LMCache status from %s", status_url)
|
|
110
|
+
|
|
111
|
+
try:
|
|
112
|
+
req = urllib.request.Request(status_url)
|
|
113
|
+
with urllib.request.urlopen(req, timeout=10) as resp:
|
|
114
|
+
return json.loads(resp.read().decode())
|
|
115
|
+
except (urllib.error.URLError, OSError) as e:
|
|
116
|
+
raise RuntimeError(
|
|
117
|
+
f"Cannot connect to LMCache server at {status_url}: {e}"
|
|
118
|
+
) from e
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
def _find_model_meta(
|
|
122
|
+
gpu_meta: dict,
|
|
123
|
+
model_name: str,
|
|
124
|
+
) -> dict:
|
|
125
|
+
"""Find the GPU metadata entry matching *model_name*.
|
|
126
|
+
|
|
127
|
+
Args:
|
|
128
|
+
gpu_meta: The ``gpu_context_meta`` dict from ``/api/status``.
|
|
129
|
+
model_name: Model name to match.
|
|
130
|
+
|
|
131
|
+
Returns:
|
|
132
|
+
The matching GPU metadata dict.
|
|
133
|
+
|
|
134
|
+
Raises:
|
|
135
|
+
RuntimeError: If no entry matches *model_name*.
|
|
136
|
+
"""
|
|
137
|
+
for meta in gpu_meta.values():
|
|
138
|
+
if meta.get("model_name") == model_name:
|
|
139
|
+
return meta
|
|
140
|
+
|
|
141
|
+
available = sorted({m.get("model_name", "?") for m in gpu_meta.values()})
|
|
142
|
+
raise RuntimeError(
|
|
143
|
+
f"Model {model_name!r} not found on LMCache server. "
|
|
144
|
+
f"Available: {', '.join(available)}"
|
|
145
|
+
)
|
|
146
|
+
|
|
147
|
+
|
|
148
|
+
def resolve_tokens_per_gb(lmcache_url: str, model_name: str) -> int:
|
|
149
|
+
"""Query the LMCache server and compute tokens per GB of KV cache.
|
|
150
|
+
|
|
151
|
+
Fetches ``/api/status``, finds the model entry matching
|
|
152
|
+
*model_name*, and computes::
|
|
153
|
+
|
|
154
|
+
global_bytes_per_token = cache_size_per_token * world_size
|
|
155
|
+
tokens_per_gb = (1024**3) // global_bytes_per_token
|
|
156
|
+
|
|
157
|
+
``cache_size_per_token`` is rank-local, so it must be multiplied
|
|
158
|
+
by ``world_size`` for tensor-parallel models.
|
|
159
|
+
|
|
160
|
+
Args:
|
|
161
|
+
lmcache_url: URL of the LMCache HTTP server.
|
|
162
|
+
model_name: Model name to look up (must match a model served
|
|
163
|
+
by the LMCache server).
|
|
164
|
+
|
|
165
|
+
Returns:
|
|
166
|
+
tokens_per_gb_kvcache value.
|
|
167
|
+
|
|
168
|
+
Raises:
|
|
169
|
+
RuntimeError: If the server is unreachable, the model is not
|
|
170
|
+
found, or the layout is missing required fields.
|
|
171
|
+
"""
|
|
172
|
+
data = _fetch_lmcache_status(lmcache_url)
|
|
173
|
+
|
|
174
|
+
gpu_meta = data.get("gpu_context_meta", {})
|
|
175
|
+
if not gpu_meta:
|
|
176
|
+
raise RuntimeError(
|
|
177
|
+
"No model info returned by LMCache server; "
|
|
178
|
+
"is the server running with a model loaded?"
|
|
179
|
+
)
|
|
180
|
+
|
|
181
|
+
meta = _find_model_meta(gpu_meta, model_name)
|
|
182
|
+
layout = meta.get("kv_cache_layout")
|
|
183
|
+
if not layout:
|
|
184
|
+
raise RuntimeError(f"No kv_cache_layout for model {model_name!r}")
|
|
185
|
+
|
|
186
|
+
cache_size_per_token = layout.get("cache_size_per_token")
|
|
187
|
+
if cache_size_per_token is None:
|
|
188
|
+
raise RuntimeError(
|
|
189
|
+
f"cache_size_per_token not available for model "
|
|
190
|
+
f"{model_name!r}; is the LMCache server up to date?"
|
|
191
|
+
)
|
|
192
|
+
|
|
193
|
+
world_size = meta.get("world_size", 1)
|
|
194
|
+
global_bytes_per_token = cache_size_per_token * world_size
|
|
195
|
+
tokens_per_gb = _GB // global_bytes_per_token
|
|
196
|
+
|
|
197
|
+
logger.info(
|
|
198
|
+
"Resolved from LMCache: model=%s, "
|
|
199
|
+
"cache_size_per_token=%d bytes (rank-local), "
|
|
200
|
+
"world_size=%d -> %d bytes/token (global) -> %d tokens/GB",
|
|
201
|
+
model_name,
|
|
202
|
+
cache_size_per_token,
|
|
203
|
+
world_size,
|
|
204
|
+
global_bytes_per_token,
|
|
205
|
+
tokens_per_gb,
|
|
206
|
+
)
|
|
207
|
+
return tokens_per_gb
|
|
208
|
+
|
|
209
|
+
|
|
210
|
+
def parse_args_to_config(args: argparse.Namespace) -> EngineBenchConfig:
|
|
211
|
+
"""Convert parsed CLI arguments into a fully-resolved EngineBenchConfig.
|
|
212
|
+
|
|
213
|
+
Handles model auto-detection and tokens-per-GB resolution from the
|
|
214
|
+
LMCache server when ``--lmcache-url`` is provided.
|
|
215
|
+
|
|
216
|
+
Args:
|
|
217
|
+
args: Parsed argparse Namespace from the bench engine subcommand.
|
|
218
|
+
|
|
219
|
+
Returns:
|
|
220
|
+
A fully-resolved EngineBenchConfig.
|
|
221
|
+
"""
|
|
222
|
+
model = args.model if args.model else auto_detect_model(args.engine_url)
|
|
223
|
+
|
|
224
|
+
tokens_per_gb = args.tokens_per_gb_kvcache
|
|
225
|
+
if tokens_per_gb is None:
|
|
226
|
+
lmcache_url = getattr(args, "lmcache_url", None)
|
|
227
|
+
if lmcache_url is not None:
|
|
228
|
+
tokens_per_gb = resolve_tokens_per_gb(lmcache_url, model)
|
|
229
|
+
else:
|
|
230
|
+
raise ValueError(
|
|
231
|
+
"--tokens-per-gb-kvcache is required when --lmcache-url is not set"
|
|
232
|
+
)
|
|
233
|
+
|
|
234
|
+
return EngineBenchConfig(
|
|
235
|
+
engine_url=args.engine_url,
|
|
236
|
+
model=model,
|
|
237
|
+
workload=args.workload,
|
|
238
|
+
kv_cache_volume_gb=args.kv_cache_volume,
|
|
239
|
+
tokens_per_gb_kvcache=tokens_per_gb,
|
|
240
|
+
seed=args.seed,
|
|
241
|
+
output_dir=args.output_dir,
|
|
242
|
+
export_csv=not args.no_csv,
|
|
243
|
+
export_json=args.json,
|
|
244
|
+
quiet=args.quiet,
|
|
245
|
+
)
|
|
@@ -0,0 +1,274 @@
|
|
|
1
|
+
# SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
"""Interactive configuration flow for ``lmcache bench engine``.
|
|
3
|
+
|
|
4
|
+
Entry point: ``run_interactive(args)`` — walks the user through
|
|
5
|
+
missing configuration, offers gates for optional settings, and
|
|
6
|
+
returns a complete ``argparse.Namespace`` ready for the orchestrator.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
# Standard
|
|
10
|
+
import argparse
|
|
11
|
+
import sys
|
|
12
|
+
|
|
13
|
+
# First Party
|
|
14
|
+
from lmcache.cli.commands.bench.engine_bench.interactive.schema import (
|
|
15
|
+
ConfigItem,
|
|
16
|
+
)
|
|
17
|
+
from lmcache.cli.commands.bench.engine_bench.interactive.state import (
|
|
18
|
+
InteractiveState,
|
|
19
|
+
)
|
|
20
|
+
from lmcache.cli.commands.bench.engine_bench.interactive.terminal import (
|
|
21
|
+
BOLD,
|
|
22
|
+
CYAN,
|
|
23
|
+
RESET,
|
|
24
|
+
YELLOW,
|
|
25
|
+
prompt_bool,
|
|
26
|
+
prompt_choice,
|
|
27
|
+
prompt_number,
|
|
28
|
+
prompt_text,
|
|
29
|
+
)
|
|
30
|
+
|
|
31
|
+
__all__ = ["run_interactive"]
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
# ---------------------------------------------------------------------------
|
|
35
|
+
# Prompt dispatcher
|
|
36
|
+
# ---------------------------------------------------------------------------
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def _prompt_for_item(item: ConfigItem) -> object:
|
|
40
|
+
"""Prompt the user for a single config item based on its type."""
|
|
41
|
+
if item.input_type == "text":
|
|
42
|
+
return prompt_text(
|
|
43
|
+
item.display_name,
|
|
44
|
+
item.description,
|
|
45
|
+
default=item.default if item.default is not None else "",
|
|
46
|
+
)
|
|
47
|
+
if item.input_type == "int":
|
|
48
|
+
return prompt_number(
|
|
49
|
+
item.display_name,
|
|
50
|
+
item.description,
|
|
51
|
+
default=item.default,
|
|
52
|
+
number_type=int,
|
|
53
|
+
)
|
|
54
|
+
if item.input_type == "float":
|
|
55
|
+
return prompt_number(
|
|
56
|
+
item.display_name,
|
|
57
|
+
item.description,
|
|
58
|
+
default=item.default,
|
|
59
|
+
number_type=float,
|
|
60
|
+
)
|
|
61
|
+
if item.input_type == "bool":
|
|
62
|
+
return prompt_bool(
|
|
63
|
+
item.display_name,
|
|
64
|
+
item.description,
|
|
65
|
+
default=bool(item.default) if item.default is not None else True,
|
|
66
|
+
)
|
|
67
|
+
if item.input_type == "choice":
|
|
68
|
+
return prompt_choice(
|
|
69
|
+
item.display_name,
|
|
70
|
+
item.description,
|
|
71
|
+
choices=item.choices,
|
|
72
|
+
default=item.default if item.default is not None else "",
|
|
73
|
+
)
|
|
74
|
+
raise ValueError(f"Unknown input_type {item.input_type!r} for {item.key}")
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
# ---------------------------------------------------------------------------
|
|
78
|
+
# Gate prompt
|
|
79
|
+
# ---------------------------------------------------------------------------
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def _prompt_gate(section_name: str, detail: str) -> bool:
|
|
83
|
+
"""Ask the user whether to configure a section or skip with defaults.
|
|
84
|
+
|
|
85
|
+
Returns True if the user wants to configure.
|
|
86
|
+
"""
|
|
87
|
+
return (
|
|
88
|
+
prompt_choice(
|
|
89
|
+
section_name,
|
|
90
|
+
f"Would you like to configure {detail}?\n"
|
|
91
|
+
f" Defaults will be used if you skip.",
|
|
92
|
+
choices=[
|
|
93
|
+
("use defaults", "Skip, use defaults"),
|
|
94
|
+
("configure", "Yes, configure"),
|
|
95
|
+
],
|
|
96
|
+
default="use defaults",
|
|
97
|
+
)
|
|
98
|
+
== "configure"
|
|
99
|
+
)
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
# ---------------------------------------------------------------------------
|
|
103
|
+
# Summary
|
|
104
|
+
# ---------------------------------------------------------------------------
|
|
105
|
+
|
|
106
|
+
|
|
107
|
+
def _print_summary(state: InteractiveState) -> None:
|
|
108
|
+
"""Print a formatted configuration summary."""
|
|
109
|
+
print()
|
|
110
|
+
print(f"{BOLD}{'─' * 50}{RESET}")
|
|
111
|
+
print(f"{BOLD} Configuration Summary{RESET}")
|
|
112
|
+
print(f"{BOLD}{'─' * 50}{RESET}")
|
|
113
|
+
for label, value in state.summary_lines():
|
|
114
|
+
padding = max(0, 26 - len(label))
|
|
115
|
+
print(f" {label}:{' ' * padding}{CYAN}{value}{RESET}")
|
|
116
|
+
print(f"{BOLD}{'─' * 50}{RESET}")
|
|
117
|
+
|
|
118
|
+
|
|
119
|
+
# ---------------------------------------------------------------------------
|
|
120
|
+
# Action prompt
|
|
121
|
+
# ---------------------------------------------------------------------------
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
def _prompt_action() -> str:
|
|
125
|
+
"""Ask the user to start the benchmark or export config.
|
|
126
|
+
|
|
127
|
+
Returns ``"start"`` or ``"export"``.
|
|
128
|
+
"""
|
|
129
|
+
return prompt_choice(
|
|
130
|
+
"What would you like to do?",
|
|
131
|
+
"",
|
|
132
|
+
choices=[
|
|
133
|
+
("start", "Start benchmark"),
|
|
134
|
+
("export", "Export configuration for later use and exit"),
|
|
135
|
+
],
|
|
136
|
+
default="start",
|
|
137
|
+
)
|
|
138
|
+
|
|
139
|
+
|
|
140
|
+
def _resolve_before_export(state: InteractiveState) -> None:
|
|
141
|
+
"""Resolve tokens_per_gb_kvcache and model before exporting.
|
|
142
|
+
|
|
143
|
+
If the user provided an LMCache URL, query the server to get
|
|
144
|
+
``tokens_per_gb_kvcache`` so the exported config is standalone.
|
|
145
|
+
If the model is empty and an engine URL is available, auto-detect it.
|
|
146
|
+
"""
|
|
147
|
+
# First Party
|
|
148
|
+
from lmcache.cli.commands.bench.engine_bench.config import (
|
|
149
|
+
auto_detect_model,
|
|
150
|
+
resolve_tokens_per_gb,
|
|
151
|
+
)
|
|
152
|
+
|
|
153
|
+
engine_url = state.get("engine_url", "")
|
|
154
|
+
model = state.get("model", "")
|
|
155
|
+
|
|
156
|
+
# Auto-detect model if empty
|
|
157
|
+
if not model and engine_url:
|
|
158
|
+
try:
|
|
159
|
+
model = auto_detect_model(engine_url)
|
|
160
|
+
state.set("model", model)
|
|
161
|
+
except RuntimeError as e:
|
|
162
|
+
print(f" {YELLOW}Warning: could not auto-detect model: {e}{RESET}")
|
|
163
|
+
|
|
164
|
+
# Resolve tokens_per_gb from LMCache if needed
|
|
165
|
+
lmcache_url = state.get("lmcache_url", "")
|
|
166
|
+
if lmcache_url and not state.is_set("tokens_per_gb_kvcache"):
|
|
167
|
+
try:
|
|
168
|
+
tokens = resolve_tokens_per_gb(lmcache_url, model)
|
|
169
|
+
state.set("tokens_per_gb_kvcache", tokens)
|
|
170
|
+
except RuntimeError as e:
|
|
171
|
+
print(
|
|
172
|
+
f" {YELLOW}Warning: could not resolve "
|
|
173
|
+
f"tokens_per_gb from LMCache: {e}{RESET}"
|
|
174
|
+
)
|
|
175
|
+
|
|
176
|
+
|
|
177
|
+
def _handle_export(state: InteractiveState) -> None:
|
|
178
|
+
"""Prompt for filename, resolve values, save JSON, and exit."""
|
|
179
|
+
_resolve_before_export(state)
|
|
180
|
+
filename = prompt_text(
|
|
181
|
+
"Export filename",
|
|
182
|
+
"",
|
|
183
|
+
default="bench_config.json",
|
|
184
|
+
)
|
|
185
|
+
state.save_json(filename)
|
|
186
|
+
print()
|
|
187
|
+
print(f" {CYAN}Saved to {filename}{RESET}")
|
|
188
|
+
print(
|
|
189
|
+
f" {BOLD}Replay with:{RESET} "
|
|
190
|
+
f"{CYAN}lmcache bench engine "
|
|
191
|
+
f"--engine-url <URL> --config {filename}{RESET}"
|
|
192
|
+
)
|
|
193
|
+
print()
|
|
194
|
+
sys.exit(0)
|
|
195
|
+
|
|
196
|
+
|
|
197
|
+
# ---------------------------------------------------------------------------
|
|
198
|
+
# Main entry point
|
|
199
|
+
# ---------------------------------------------------------------------------
|
|
200
|
+
|
|
201
|
+
|
|
202
|
+
def run_interactive(args: argparse.Namespace) -> argparse.Namespace:
|
|
203
|
+
"""Run the interactive configuration flow.
|
|
204
|
+
|
|
205
|
+
Walks the user through missing required items, offers gates for
|
|
206
|
+
general and workload-specific settings, shows a summary, and
|
|
207
|
+
returns a complete ``argparse.Namespace``.
|
|
208
|
+
|
|
209
|
+
Args:
|
|
210
|
+
args: Partially-populated CLI args (some may be None).
|
|
211
|
+
|
|
212
|
+
Returns:
|
|
213
|
+
A fully-populated ``argparse.Namespace`` ready for the
|
|
214
|
+
benchmark orchestrator.
|
|
215
|
+
"""
|
|
216
|
+
state = InteractiveState.from_cli_args(args)
|
|
217
|
+
|
|
218
|
+
print()
|
|
219
|
+
print(f"{BOLD}{'═' * 50}{RESET}")
|
|
220
|
+
print(f"{BOLD} lmcache bench engine — Interactive Setup{RESET}")
|
|
221
|
+
print(f"{BOLD}{'═' * 50}{RESET}")
|
|
222
|
+
|
|
223
|
+
# ── Phase 1: Required items ───────────────────────────────────────
|
|
224
|
+
# Walk through missing required items one by one. Re-evaluate the
|
|
225
|
+
# list after each prompt because setting one value (e.g., has_lmcache)
|
|
226
|
+
# can make new items eligible (e.g., lmcache_url) or skip others
|
|
227
|
+
# (e.g., tokens_per_gb_kvcache).
|
|
228
|
+
while True:
|
|
229
|
+
missing = state.get_missing_required()
|
|
230
|
+
if not missing:
|
|
231
|
+
break
|
|
232
|
+
item = missing[0]
|
|
233
|
+
value = _prompt_for_item(item)
|
|
234
|
+
state.set(item.key, value)
|
|
235
|
+
|
|
236
|
+
# ── Phase 2: General settings gate ────────────────────────────────
|
|
237
|
+
if state.has_unconfigured_general():
|
|
238
|
+
if _prompt_gate(
|
|
239
|
+
"General settings",
|
|
240
|
+
"general settings (model, KV cache volume, etc.)",
|
|
241
|
+
):
|
|
242
|
+
for item in state.get_general_items():
|
|
243
|
+
value = _prompt_for_item(item)
|
|
244
|
+
state.set(item.key, value)
|
|
245
|
+
|
|
246
|
+
# ── Phase 3: Workload settings gate (always shown) ────────────────
|
|
247
|
+
if state.has_workload_items():
|
|
248
|
+
workload = state.get("workload", "workload")
|
|
249
|
+
if _prompt_gate(
|
|
250
|
+
f"Workload settings ({workload})",
|
|
251
|
+
"workload-specific settings",
|
|
252
|
+
):
|
|
253
|
+
for item in state.get_workload_items():
|
|
254
|
+
value = _prompt_for_item(item)
|
|
255
|
+
state.set(item.key, value)
|
|
256
|
+
|
|
257
|
+
# Fill all remaining defaults
|
|
258
|
+
state.fill_defaults()
|
|
259
|
+
|
|
260
|
+
# ── Phase 4: Summary + action ─────────────────────────────────────
|
|
261
|
+
_print_summary(state)
|
|
262
|
+
action = _prompt_action()
|
|
263
|
+
|
|
264
|
+
if action == "export":
|
|
265
|
+
_handle_export(state)
|
|
266
|
+
|
|
267
|
+
# Carry over output settings from original CLI args
|
|
268
|
+
ns = state.to_namespace()
|
|
269
|
+
for attr in ("output_dir", "seed", "no_csv", "json", "quiet", "format", "output"):
|
|
270
|
+
cli_val = getattr(args, attr, None)
|
|
271
|
+
if cli_val is not None:
|
|
272
|
+
setattr(ns, attr, cli_val)
|
|
273
|
+
|
|
274
|
+
return ns
|