lmcache-cli 0.4.5.dev0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- lmcache/__init__.py +84 -0
- lmcache/_version.py +24 -0
- lmcache/cli/__init__.py +1 -0
- lmcache/cli/commands/__init__.py +34 -0
- lmcache/cli/commands/base.py +157 -0
- lmcache/cli/commands/bench/__init__.py +557 -0
- lmcache/cli/commands/bench/engine_bench/__init__.py +1 -0
- lmcache/cli/commands/bench/engine_bench/config.py +245 -0
- lmcache/cli/commands/bench/engine_bench/interactive/__init__.py +274 -0
- lmcache/cli/commands/bench/engine_bench/interactive/config.json +10 -0
- lmcache/cli/commands/bench/engine_bench/interactive/schema.py +352 -0
- lmcache/cli/commands/bench/engine_bench/interactive/state.py +327 -0
- lmcache/cli/commands/bench/engine_bench/interactive/terminal.py +291 -0
- lmcache/cli/commands/bench/engine_bench/progress.py +145 -0
- lmcache/cli/commands/bench/engine_bench/request_sender.py +232 -0
- lmcache/cli/commands/bench/engine_bench/stats.py +275 -0
- lmcache/cli/commands/bench/engine_bench/workloads/__init__.py +153 -0
- lmcache/cli/commands/bench/engine_bench/workloads/base.py +122 -0
- lmcache/cli/commands/bench/engine_bench/workloads/long_doc_permutator.py +435 -0
- lmcache/cli/commands/bench/engine_bench/workloads/long_doc_qa.py +281 -0
- lmcache/cli/commands/bench/engine_bench/workloads/multi_round_chat.py +337 -0
- lmcache/cli/commands/bench/engine_bench/workloads/random_prefill.py +178 -0
- lmcache/cli/commands/describe.py +310 -0
- lmcache/cli/commands/kvcache.py +133 -0
- lmcache/cli/commands/mock.py +75 -0
- lmcache/cli/commands/ping.py +113 -0
- lmcache/cli/commands/query/__init__.py +155 -0
- lmcache/cli/commands/query/prompt.py +134 -0
- lmcache/cli/commands/query/request.py +357 -0
- lmcache/cli/commands/server.py +99 -0
- lmcache/cli/commands/tool/__init__.py +63 -0
- lmcache/cli/commands/tool/cache_simulator.py +113 -0
- lmcache/cli/commands/trace/__init__.py +505 -0
- lmcache/cli/commands/trace/dispatch.py +249 -0
- lmcache/cli/commands/trace/driver.py +372 -0
- lmcache/cli/commands/trace/stats.py +289 -0
- lmcache/cli/documents/lmcache.txt +11 -0
- lmcache/cli/main.py +42 -0
- lmcache/cli/metrics/__init__.py +29 -0
- lmcache/cli/metrics/formatter.py +171 -0
- lmcache/cli/metrics/handler.py +94 -0
- lmcache/cli/metrics/metrics.py +161 -0
- lmcache/cli/metrics/section.py +77 -0
- lmcache/connections.py +173 -0
- lmcache/integration/__init__.py +2 -0
- lmcache/integration/base_service_factory.py +165 -0
- lmcache/integration/request_telemetry/__init__.py +1 -0
- lmcache/integration/request_telemetry/base.py +51 -0
- lmcache/integration/request_telemetry/factory.py +113 -0
- lmcache/integration/request_telemetry/fastapi.py +109 -0
- lmcache/integration/request_telemetry/noop.py +35 -0
- lmcache/integration/sglang/__init__.py +2 -0
- lmcache/integration/sglang/sglang_adapter.py +326 -0
- lmcache/integration/sglang/utils.py +39 -0
- lmcache/integration/vllm/__init__.py +1 -0
- lmcache/integration/vllm/lmcache_connector_v1.py +213 -0
- lmcache/integration/vllm/lmcache_connector_v1_085.py +150 -0
- lmcache/integration/vllm/lmcache_mp_connector_0180.py +1072 -0
- lmcache/integration/vllm/tests/test_mm_hash_utils.py +112 -0
- lmcache/integration/vllm/utils.py +433 -0
- lmcache/integration/vllm/vllm_multi_process_adapter.py +1090 -0
- lmcache/integration/vllm/vllm_service_factory.py +339 -0
- lmcache/integration/vllm/vllm_v1_adapter.py +1713 -0
- lmcache/logging.py +107 -0
- lmcache/native_storage_ops.pyi +230 -0
- lmcache/non_cuda_equivalents.py +1424 -0
- lmcache/observability.py +1958 -0
- lmcache/storage_backend/serde/__init__.py +1 -0
- lmcache/storage_backend/serde/cachegen_basics.py +210 -0
- lmcache/storage_backend/serde/cachegen_decoder.py +207 -0
- lmcache/storage_backend/serde/cachegen_encoder.py +394 -0
- lmcache/storage_backend/serde/serde.py +75 -0
- lmcache/tools/__init__.py +1 -0
- lmcache/tools/cache_simulator/README.md +392 -0
- lmcache/tools/cache_simulator/__init__.py +1 -0
- lmcache/tools/cache_simulator/docs/simulate_example.png +0 -0
- lmcache/tools/cache_simulator/docs/sweep_example.png +0 -0
- lmcache/tools/cache_simulator/gen_bench_dataset.py +360 -0
- lmcache/tools/cache_simulator/lru_cache.py +124 -0
- lmcache/tools/cache_simulator/plot_hit_rate.py +231 -0
- lmcache/tools/cache_simulator/simulator.py +795 -0
- lmcache/tools/controller_benchmark/README.md +161 -0
- lmcache/tools/controller_benchmark/__init__.py +1 -0
- lmcache/tools/controller_benchmark/__main__.py +331 -0
- lmcache/tools/controller_benchmark/benchmark.py +660 -0
- lmcache/tools/controller_benchmark/config.py +44 -0
- lmcache/tools/controller_benchmark/constants.py +10 -0
- lmcache/tools/controller_benchmark/handlers/__init__.py +46 -0
- lmcache/tools/controller_benchmark/handlers/admit.py +52 -0
- lmcache/tools/controller_benchmark/handlers/base.py +47 -0
- lmcache/tools/controller_benchmark/handlers/deregister.py +49 -0
- lmcache/tools/controller_benchmark/handlers/evict.py +52 -0
- lmcache/tools/controller_benchmark/handlers/heartbeat.py +56 -0
- lmcache/tools/controller_benchmark/handlers/p2p_lookup.py +47 -0
- lmcache/tools/controller_benchmark/handlers/register.py +56 -0
- lmcache/tools/mp_status_viewer/__init__.py +1 -0
- lmcache/tools/mp_status_viewer/__main__.py +95 -0
- lmcache/usage_context.py +417 -0
- lmcache/utils.py +665 -0
- lmcache/v1/__init__.py +2 -0
- lmcache/v1/api_server/__init__.py +2 -0
- lmcache/v1/api_server/__main__.py +537 -0
- lmcache/v1/basic_check.py +112 -0
- lmcache/v1/cache_controller/__init__.py +9 -0
- lmcache/v1/cache_controller/commands/__init__.py +15 -0
- lmcache/v1/cache_controller/commands/base.py +35 -0
- lmcache/v1/cache_controller/commands/full_sync.py +49 -0
- lmcache/v1/cache_controller/config.py +176 -0
- lmcache/v1/cache_controller/controller_manager.py +535 -0
- lmcache/v1/cache_controller/controllers/__init__.py +11 -0
- lmcache/v1/cache_controller/controllers/full_sync_tracker.py +473 -0
- lmcache/v1/cache_controller/controllers/kv_controller.py +439 -0
- lmcache/v1/cache_controller/controllers/registration_controller.py +282 -0
- lmcache/v1/cache_controller/executor.py +463 -0
- lmcache/v1/cache_controller/frontend/static/css/style.css +201 -0
- lmcache/v1/cache_controller/frontend/static/img/logo.png +0 -0
- lmcache/v1/cache_controller/frontend/static/index.html +234 -0
- lmcache/v1/cache_controller/frontend/static/js/controller_app.js +660 -0
- lmcache/v1/cache_controller/full_sync_sender.py +475 -0
- lmcache/v1/cache_controller/locks.py +149 -0
- lmcache/v1/cache_controller/message.py +828 -0
- lmcache/v1/cache_controller/observability.py +208 -0
- lmcache/v1/cache_controller/utils.py +679 -0
- lmcache/v1/cache_controller/worker.py +665 -0
- lmcache/v1/cache_engine.py +2058 -0
- lmcache/v1/cache_interface.py +19 -0
- lmcache/v1/check/__init__.py +74 -0
- lmcache/v1/check/check_mode_gen.py +86 -0
- lmcache/v1/check/check_mode_test_l2_adapter.py +284 -0
- lmcache/v1/check/check_mode_test_remote.py +155 -0
- lmcache/v1/check/check_mode_test_storage_manager.py +142 -0
- lmcache/v1/check/utils.py +571 -0
- lmcache/v1/compute/__init__.py +2 -0
- lmcache/v1/compute/attention/__init__.py +0 -0
- lmcache/v1/compute/attention/abstract.py +39 -0
- lmcache/v1/compute/attention/flash_attn.py +129 -0
- lmcache/v1/compute/attention/flash_infer_sparse.py +284 -0
- lmcache/v1/compute/attention/metadata.py +85 -0
- lmcache/v1/compute/attention/utils.py +14 -0
- lmcache/v1/compute/blend/__init__.py +7 -0
- lmcache/v1/compute/blend/blender.py +168 -0
- lmcache/v1/compute/blend/metadata.py +34 -0
- lmcache/v1/compute/blend/utils.py +63 -0
- lmcache/v1/compute/models/__init__.py +0 -0
- lmcache/v1/compute/models/base.py +141 -0
- lmcache/v1/compute/models/llama.py +9 -0
- lmcache/v1/compute/models/qwen3.py +24 -0
- lmcache/v1/compute/models/utils.py +68 -0
- lmcache/v1/compute/positional_encoding.py +199 -0
- lmcache/v1/config.py +848 -0
- lmcache/v1/config_base.py +848 -0
- lmcache/v1/distributed/api.py +248 -0
- lmcache/v1/distributed/config.py +321 -0
- lmcache/v1/distributed/error.py +64 -0
- lmcache/v1/distributed/eviction.py +192 -0
- lmcache/v1/distributed/eviction_policy/__init__.py +21 -0
- lmcache/v1/distributed/eviction_policy/factory.py +27 -0
- lmcache/v1/distributed/eviction_policy/lru.py +244 -0
- lmcache/v1/distributed/eviction_policy/noop.py +50 -0
- lmcache/v1/distributed/internal_api.py +170 -0
- lmcache/v1/distributed/l1_manager.py +835 -0
- lmcache/v1/distributed/l2_adapters/__init__.py +67 -0
- lmcache/v1/distributed/l2_adapters/base.py +360 -0
- lmcache/v1/distributed/l2_adapters/config.py +385 -0
- lmcache/v1/distributed/l2_adapters/factory.py +205 -0
- lmcache/v1/distributed/l2_adapters/fs_l2_adapter.py +747 -0
- lmcache/v1/distributed/l2_adapters/fs_native_l2_adapter.py +167 -0
- lmcache/v1/distributed/l2_adapters/mock_l2_adapter.py +516 -0
- lmcache/v1/distributed/l2_adapters/mooncake_store_l2_adapter.py +135 -0
- lmcache/v1/distributed/l2_adapters/native_connector_l2_adapter.py +468 -0
- lmcache/v1/distributed/l2_adapters/native_plugin_l2_adapter.py +199 -0
- lmcache/v1/distributed/l2_adapters/nixl_store_dynamic_l2_adapter.py +831 -0
- lmcache/v1/distributed/l2_adapters/nixl_store_l2_adapter.py +983 -0
- lmcache/v1/distributed/l2_adapters/plugin_l2_adapter.py +210 -0
- lmcache/v1/distributed/l2_adapters/resp_l2_adapter.py +176 -0
- lmcache/v1/distributed/memory_manager.py +179 -0
- lmcache/v1/distributed/storage_controller.py +39 -0
- lmcache/v1/distributed/storage_controllers/__init__.py +43 -0
- lmcache/v1/distributed/storage_controllers/eviction_controller.py +242 -0
- lmcache/v1/distributed/storage_controllers/prefetch_controller.py +830 -0
- lmcache/v1/distributed/storage_controllers/prefetch_policy.py +193 -0
- lmcache/v1/distributed/storage_controllers/store_controller.py +452 -0
- lmcache/v1/distributed/storage_controllers/store_policy.py +213 -0
- lmcache/v1/distributed/storage_manager.py +532 -0
- lmcache/v1/event_manager.py +145 -0
- lmcache/v1/exceptions/__init__.py +16 -0
- lmcache/v1/gpu_connector/__init__.py +126 -0
- lmcache/v1/gpu_connector/gpu_connectors.py +1906 -0
- lmcache/v1/gpu_connector/gpu_ops.py +85 -0
- lmcache/v1/gpu_connector/hpu_connector.py +326 -0
- lmcache/v1/gpu_connector/mock_gpu_connector.py +67 -0
- lmcache/v1/gpu_connector/utils.py +890 -0
- lmcache/v1/gpu_connector/xpu_connectors.py +916 -0
- lmcache/v1/health_monitor/__init__.py +1 -0
- lmcache/v1/health_monitor/base.py +587 -0
- lmcache/v1/health_monitor/checks/__init__.py +1 -0
- lmcache/v1/health_monitor/checks/remote_backend_check.py +304 -0
- lmcache/v1/health_monitor/constants.py +36 -0
- lmcache/v1/internal_api_server/__init__.py +0 -0
- lmcache/v1/internal_api_server/api_registry.py +59 -0
- lmcache/v1/internal_api_server/api_server.py +120 -0
- lmcache/v1/internal_api_server/common/__init__.py +1 -0
- lmcache/v1/internal_api_server/common/env_api.py +22 -0
- lmcache/v1/internal_api_server/common/loglevel_api.py +57 -0
- lmcache/v1/internal_api_server/common/metrics_api.py +29 -0
- lmcache/v1/internal_api_server/common/periodic_thread_api.py +138 -0
- lmcache/v1/internal_api_server/common/run_script_api.py +73 -0
- lmcache/v1/internal_api_server/common/thread_api.py +63 -0
- lmcache/v1/internal_api_server/controller/__init__.py +1 -0
- lmcache/v1/internal_api_server/controller/key_stats_api.py +81 -0
- lmcache/v1/internal_api_server/controller/worker_info_api.py +136 -0
- lmcache/v1/internal_api_server/utils.py +43 -0
- lmcache/v1/internal_api_server/vllm/__init__.py +1 -0
- lmcache/v1/internal_api_server/vllm/backend_api.py +221 -0
- lmcache/v1/internal_api_server/vllm/bypass_api.py +204 -0
- lmcache/v1/internal_api_server/vllm/cache_api.py +895 -0
- lmcache/v1/internal_api_server/vllm/chunk_statistics_api.py +141 -0
- lmcache/v1/internal_api_server/vllm/conf_api.py +147 -0
- lmcache/v1/internal_api_server/vllm/freeze_api.py +172 -0
- lmcache/v1/internal_api_server/vllm/hot_cache_api.py +184 -0
- lmcache/v1/internal_api_server/vllm/inference_api.py +65 -0
- lmcache/v1/internal_api_server/vllm/load_fs_chunks_api.py +320 -0
- lmcache/v1/internal_api_server/vllm/lookup_api.py +145 -0
- lmcache/v1/internal_api_server/vllm/version_api.py +25 -0
- lmcache/v1/kv_layer_groups.py +267 -0
- lmcache/v1/lazy_memory_allocator.py +284 -0
- lmcache/v1/lookup_client/__init__.py +25 -0
- lmcache/v1/lookup_client/abstract_client.py +77 -0
- lmcache/v1/lookup_client/async_lookup_message.py +50 -0
- lmcache/v1/lookup_client/chunk_statistics_lookup_client.py +200 -0
- lmcache/v1/lookup_client/factory.py +251 -0
- lmcache/v1/lookup_client/hit_limit_lookup_client.py +86 -0
- lmcache/v1/lookup_client/lmcache_async_lookup_client.py +407 -0
- lmcache/v1/lookup_client/lmcache_lookup_client.py +285 -0
- lmcache/v1/lookup_client/lmcache_lookup_client_bypass.py +99 -0
- lmcache/v1/lookup_client/mooncake_lookup_client.py +87 -0
- lmcache/v1/lookup_client/record_strategies/__init__.py +77 -0
- lmcache/v1/lookup_client/record_strategies/base.py +327 -0
- lmcache/v1/lookup_client/record_strategies/file_hash.py +130 -0
- lmcache/v1/lookup_client/record_strategies/memory_bloom_filter.py +81 -0
- lmcache/v1/manager.py +539 -0
- lmcache/v1/memory_management.py +2619 -0
- lmcache/v1/metadata.py +114 -0
- lmcache/v1/mp_observability/AGENTS.override.md +21 -0
- lmcache/v1/mp_observability/README.md +204 -0
- lmcache/v1/mp_observability/config.py +340 -0
- lmcache/v1/mp_observability/event.py +100 -0
- lmcache/v1/mp_observability/event_bus.py +313 -0
- lmcache/v1/mp_observability/otel_init.py +145 -0
- lmcache/v1/mp_observability/subscribers/__init__.py +28 -0
- lmcache/v1/mp_observability/subscribers/logging/__init__.py +19 -0
- lmcache/v1/mp_observability/subscribers/logging/l1.py +56 -0
- lmcache/v1/mp_observability/subscribers/logging/l2.py +73 -0
- lmcache/v1/mp_observability/subscribers/logging/lookup_hash.py +209 -0
- lmcache/v1/mp_observability/subscribers/logging/mp_server.py +90 -0
- lmcache/v1/mp_observability/subscribers/logging/sm.py +59 -0
- lmcache/v1/mp_observability/subscribers/metrics/__init__.py +20 -0
- lmcache/v1/mp_observability/subscribers/metrics/l0_lifecycle.py +290 -0
- lmcache/v1/mp_observability/subscribers/metrics/l1.py +55 -0
- lmcache/v1/mp_observability/subscribers/metrics/l1_lifecycle.py +166 -0
- lmcache/v1/mp_observability/subscribers/metrics/l2.py +121 -0
- lmcache/v1/mp_observability/subscribers/metrics/sm.py +69 -0
- lmcache/v1/mp_observability/subscribers/tracing/__init__.py +12 -0
- lmcache/v1/mp_observability/subscribers/tracing/mp_server.py +333 -0
- lmcache/v1/mp_observability/subscribers/tracing/span_registry.py +148 -0
- lmcache/v1/mp_observability/trace/__init__.py +50 -0
- lmcache/v1/mp_observability/trace/codecs.py +255 -0
- lmcache/v1/mp_observability/trace/decorator.py +147 -0
- lmcache/v1/mp_observability/trace/format.py +132 -0
- lmcache/v1/mp_observability/trace/lifecycle.py +83 -0
- lmcache/v1/mp_observability/trace/reader.py +167 -0
- lmcache/v1/mp_observability/trace/recorder.py +300 -0
- lmcache/v1/multiprocess/__init__.py +0 -0
- lmcache/v1/multiprocess/affinity_pool.py +102 -0
- lmcache/v1/multiprocess/blend_server_v2.py +891 -0
- lmcache/v1/multiprocess/config.py +253 -0
- lmcache/v1/multiprocess/custom_types.py +281 -0
- lmcache/v1/multiprocess/futures.py +194 -0
- lmcache/v1/multiprocess/gpu_context.py +511 -0
- lmcache/v1/multiprocess/http_server.py +235 -0
- lmcache/v1/multiprocess/mp_runtime_plugin_launcher.py +130 -0
- lmcache/v1/multiprocess/mq.py +732 -0
- lmcache/v1/multiprocess/protocol.py +86 -0
- lmcache/v1/multiprocess/protocols/README.md +213 -0
- lmcache/v1/multiprocess/protocols/__init__.py +127 -0
- lmcache/v1/multiprocess/protocols/base.py +89 -0
- lmcache/v1/multiprocess/protocols/blend.py +109 -0
- lmcache/v1/multiprocess/protocols/blend_v2.py +57 -0
- lmcache/v1/multiprocess/protocols/controller.py +53 -0
- lmcache/v1/multiprocess/protocols/debug.py +34 -0
- lmcache/v1/multiprocess/protocols/engine.py +146 -0
- lmcache/v1/multiprocess/protocols/observability.py +39 -0
- lmcache/v1/multiprocess/server.py +1134 -0
- lmcache/v1/multiprocess/session.py +190 -0
- lmcache/v1/multiprocess/token_hasher.py +441 -0
- lmcache/v1/offload_server/__init__.py +17 -0
- lmcache/v1/offload_server/abstract_server.py +37 -0
- lmcache/v1/offload_server/message.py +30 -0
- lmcache/v1/offload_server/zmq_server.py +122 -0
- lmcache/v1/periodic_thread.py +579 -0
- lmcache/v1/pin_monitor.py +246 -0
- lmcache/v1/plugin/__init__.py +0 -0
- lmcache/v1/plugin/runtime_plugin_launcher.py +211 -0
- lmcache/v1/protocol.py +317 -0
- lmcache/v1/rpc/__init__.py +17 -0
- lmcache/v1/rpc/transport.py +105 -0
- lmcache/v1/rpc/zmq_transport.py +213 -0
- lmcache/v1/rpc_utils.py +165 -0
- lmcache/v1/server/__init__.py +2 -0
- lmcache/v1/server/__main__.py +170 -0
- lmcache/v1/server/storage_backend/__init__.py +21 -0
- lmcache/v1/server/storage_backend/abstract_backend.py +80 -0
- lmcache/v1/server/storage_backend/local_backend.py +75 -0
- lmcache/v1/server/utils.py +21 -0
- lmcache/v1/standalone/__init__.py +1 -0
- lmcache/v1/standalone/__main__.py +583 -0
- lmcache/v1/standalone/manager.py +80 -0
- lmcache/v1/standalone/standalone_service_factory.py +86 -0
- lmcache/v1/storage_backend/__init__.py +313 -0
- lmcache/v1/storage_backend/abstract_backend.py +445 -0
- lmcache/v1/storage_backend/audit_backend.py +233 -0
- lmcache/v1/storage_backend/batched_message_sender.py +222 -0
- lmcache/v1/storage_backend/cache_policy/__init__.py +45 -0
- lmcache/v1/storage_backend/cache_policy/base_policy.py +87 -0
- lmcache/v1/storage_backend/cache_policy/fifo.py +58 -0
- lmcache/v1/storage_backend/cache_policy/lfu.py +105 -0
- lmcache/v1/storage_backend/cache_policy/lru.py +81 -0
- lmcache/v1/storage_backend/cache_policy/mru.py +61 -0
- lmcache/v1/storage_backend/connector/__init__.py +443 -0
- lmcache/v1/storage_backend/connector/audit_adapter.py +77 -0
- lmcache/v1/storage_backend/connector/audit_connector.py +320 -0
- lmcache/v1/storage_backend/connector/base_connector.py +379 -0
- lmcache/v1/storage_backend/connector/blackhole_adapter.py +21 -0
- lmcache/v1/storage_backend/connector/blackhole_connector.py +37 -0
- lmcache/v1/storage_backend/connector/eic_adapter.py +31 -0
- lmcache/v1/storage_backend/connector/eic_connector.py +757 -0
- lmcache/v1/storage_backend/connector/external_adapter.py +79 -0
- lmcache/v1/storage_backend/connector/fs_adapter.py +51 -0
- lmcache/v1/storage_backend/connector/fs_connector.py +403 -0
- lmcache/v1/storage_backend/connector/infinistore_adapter.py +56 -0
- lmcache/v1/storage_backend/connector/infinistore_connector.py +177 -0
- lmcache/v1/storage_backend/connector/instrumented_connector.py +219 -0
- lmcache/v1/storage_backend/connector/lm_adapter.py +31 -0
- lmcache/v1/storage_backend/connector/lm_connector.py +176 -0
- lmcache/v1/storage_backend/connector/mock_adapter.py +57 -0
- lmcache/v1/storage_backend/connector/mock_connector.py +349 -0
- lmcache/v1/storage_backend/connector/mooncakestore_adapter.py +43 -0
- lmcache/v1/storage_backend/connector/mooncakestore_connector.py +614 -0
- lmcache/v1/storage_backend/connector/redis_adapter.py +181 -0
- lmcache/v1/storage_backend/connector/redis_connector.py +828 -0
- lmcache/v1/storage_backend/connector/s3_adapter.py +59 -0
- lmcache/v1/storage_backend/connector/s3_connector.py +699 -0
- lmcache/v1/storage_backend/connector/sagemaker_hyperpod_adapter.py +233 -0
- lmcache/v1/storage_backend/connector/sagemaker_hyperpod_connector.py +987 -0
- lmcache/v1/storage_backend/connector/valkey_adapter.py +114 -0
- lmcache/v1/storage_backend/connector/valkey_connector.py +627 -0
- lmcache/v1/storage_backend/gds_backend.py +1199 -0
- lmcache/v1/storage_backend/job_executor/__init__.py +0 -0
- lmcache/v1/storage_backend/job_executor/base_executor.py +34 -0
- lmcache/v1/storage_backend/job_executor/pq_executor.py +235 -0
- lmcache/v1/storage_backend/local_cpu_backend.py +810 -0
- lmcache/v1/storage_backend/local_disk_backend.py +656 -0
- lmcache/v1/storage_backend/maru_backend.py +734 -0
- lmcache/v1/storage_backend/naive_serde/__init__.py +50 -0
- lmcache/v1/storage_backend/naive_serde/cachegen_basics.py +133 -0
- lmcache/v1/storage_backend/naive_serde/cachegen_decoder.py +135 -0
- lmcache/v1/storage_backend/naive_serde/cachegen_encoder.py +83 -0
- lmcache/v1/storage_backend/naive_serde/kivi_serde.py +22 -0
- lmcache/v1/storage_backend/naive_serde/naive_serde.py +18 -0
- lmcache/v1/storage_backend/naive_serde/serde.py +37 -0
- lmcache/v1/storage_backend/native_clients/connector_client_base.py +165 -0
- lmcache/v1/storage_backend/native_clients/resp_client.py +35 -0
- lmcache/v1/storage_backend/nixl_storage_backend.py +1400 -0
- lmcache/v1/storage_backend/p2p_backend.py +788 -0
- lmcache/v1/storage_backend/path_sharder.py +117 -0
- lmcache/v1/storage_backend/pd_backend.py +646 -0
- lmcache/v1/storage_backend/plugins/dax_backend.py +1443 -0
- lmcache/v1/storage_backend/plugins/rust_raw_block_backend.py +1361 -0
- lmcache/v1/storage_backend/remote_backend.py +624 -0
- lmcache/v1/storage_backend/resp_client.py +227 -0
- lmcache/v1/storage_backend/storage_backend_listener.py +19 -0
- lmcache/v1/storage_backend/storage_manager.py +1352 -0
- lmcache/v1/system_detection.py +110 -0
- lmcache/v1/token_database.py +551 -0
- lmcache/v1/transfer_channel/__init__.py +83 -0
- lmcache/v1/transfer_channel/abstract.py +285 -0
- lmcache/v1/transfer_channel/mock_memory_channel.py +156 -0
- lmcache/v1/transfer_channel/nixl_channel.py +639 -0
- lmcache/v1/transfer_channel/py_socket_channel.py +260 -0
- lmcache/v1/transfer_channel/transfer_utils.py +63 -0
- lmcache/v1/utils/__init__.py +1 -0
- lmcache/v1/utils/bloom_filter.py +109 -0
- lmcache/v1/utils/cache_utils.py +125 -0
- lmcache_cli-0.4.5.dev0.dist-info/METADATA +185 -0
- lmcache_cli-0.4.5.dev0.dist-info/RECORD +399 -0
- lmcache_cli-0.4.5.dev0.dist-info/WHEEL +5 -0
- lmcache_cli-0.4.5.dev0.dist-info/entry_points.txt +2 -0
- lmcache_cli-0.4.5.dev0.dist-info/licenses/LICENSE +201 -0
- lmcache_cli-0.4.5.dev0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,352 @@
|
|
|
1
|
+
# SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
"""Centralized config item definitions for interactive configuration.
|
|
3
|
+
|
|
4
|
+
Each ``ConfigItem`` declaratively describes one configurable parameter:
|
|
5
|
+
its key, display name, description, input type, default, and when it
|
|
6
|
+
should be shown. The ``ALL_ITEMS`` list is the single source of truth
|
|
7
|
+
for descriptions, ordering, and defaults.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
# Standard
|
|
11
|
+
from collections.abc import Callable
|
|
12
|
+
from dataclasses import dataclass, field
|
|
13
|
+
from typing import Any
|
|
14
|
+
|
|
15
|
+
# ---------------------------------------------------------------------------
|
|
16
|
+
# Phases
|
|
17
|
+
# ---------------------------------------------------------------------------
|
|
18
|
+
|
|
19
|
+
PHASE_REQUIRED = 1
|
|
20
|
+
PHASE_GENERAL = 2
|
|
21
|
+
PHASE_WORKLOAD = 3
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
# ---------------------------------------------------------------------------
|
|
25
|
+
# ConfigItem
|
|
26
|
+
# ---------------------------------------------------------------------------
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
@dataclass
|
|
30
|
+
class ConfigItem:
|
|
31
|
+
"""Declarative description of a single configurable parameter.
|
|
32
|
+
|
|
33
|
+
Attributes:
|
|
34
|
+
key: State dict key (matches argparse attr name, e.g., ``"engine_url"``).
|
|
35
|
+
display_name: Heading shown in the prompt.
|
|
36
|
+
description: One-sentence explanation shown below the heading.
|
|
37
|
+
input_type: One of ``"text"``, ``"int"``, ``"float"``, ``"bool"``,
|
|
38
|
+
``"choice"``.
|
|
39
|
+
default: Default value. ``None`` means required (no default).
|
|
40
|
+
required: If True, this item must have a value before the benchmark
|
|
41
|
+
can start.
|
|
42
|
+
choices: For ``"choice"`` type — list of ``(value, description)`` tuples.
|
|
43
|
+
condition: Callable ``(state_dict) -> bool`` that determines whether
|
|
44
|
+
this item should be shown. ``None`` means always shown.
|
|
45
|
+
phase: Which interactive phase this item belongs to.
|
|
46
|
+
"""
|
|
47
|
+
|
|
48
|
+
key: str
|
|
49
|
+
display_name: str
|
|
50
|
+
description: str
|
|
51
|
+
input_type: str # "text", "int", "float", "bool", "choice"
|
|
52
|
+
default: Any = None
|
|
53
|
+
required: bool = False
|
|
54
|
+
choices: list[tuple[str, str]] = field(default_factory=list)
|
|
55
|
+
condition: Callable[[dict[str, Any]], bool] | None = None
|
|
56
|
+
phase: int = PHASE_GENERAL
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
# ---------------------------------------------------------------------------
|
|
60
|
+
# Condition helpers
|
|
61
|
+
# ---------------------------------------------------------------------------
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def _has_lmcache(state: dict[str, Any]) -> bool:
|
|
65
|
+
"""Show this item only when the user said they have LMCache."""
|
|
66
|
+
return bool(state.get("has_lmcache"))
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def _no_lmcache_url(state: dict[str, Any]) -> bool:
|
|
70
|
+
"""Show this item only when lmcache_url is not set."""
|
|
71
|
+
return not state.get("lmcache_url")
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def _workload_is(name: str) -> Callable[[dict[str, Any]], bool]:
|
|
75
|
+
"""Return a condition that checks the workload value."""
|
|
76
|
+
|
|
77
|
+
def check(state: dict[str, Any]) -> bool:
|
|
78
|
+
return state.get("workload") == name
|
|
79
|
+
|
|
80
|
+
return check
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
# ---------------------------------------------------------------------------
|
|
84
|
+
# ALL_ITEMS — the centralized registry
|
|
85
|
+
# ---------------------------------------------------------------------------
|
|
86
|
+
|
|
87
|
+
ALL_ITEMS: list[ConfigItem] = [
|
|
88
|
+
# ── Phase 1: Required ─────────────────────────────────────────────
|
|
89
|
+
ConfigItem(
|
|
90
|
+
key="engine_url",
|
|
91
|
+
display_name="Engine URL",
|
|
92
|
+
description=(
|
|
93
|
+
"URL of the inference engine. "
|
|
94
|
+
"Set OPENAI_API_KEY env var if authentication is needed."
|
|
95
|
+
),
|
|
96
|
+
input_type="text",
|
|
97
|
+
default="http://localhost:8000",
|
|
98
|
+
required=True,
|
|
99
|
+
phase=PHASE_REQUIRED,
|
|
100
|
+
),
|
|
101
|
+
ConfigItem(
|
|
102
|
+
key="workload",
|
|
103
|
+
display_name="Workload",
|
|
104
|
+
description="The type of benchmark workload to run.",
|
|
105
|
+
input_type="choice",
|
|
106
|
+
default=None,
|
|
107
|
+
required=True,
|
|
108
|
+
choices=[
|
|
109
|
+
(
|
|
110
|
+
"long-doc-permutator",
|
|
111
|
+
"Query the same set of long documents with different orders",
|
|
112
|
+
),
|
|
113
|
+
("long-doc-qa", "Repeated Q&A over long documents (tests KV cache reuse)"),
|
|
114
|
+
("multi-round-chat", "Multi-turn chat with stateful sessions"),
|
|
115
|
+
("random-prefill", "Prefill-only requests fired simultaneously"),
|
|
116
|
+
],
|
|
117
|
+
phase=PHASE_REQUIRED,
|
|
118
|
+
),
|
|
119
|
+
ConfigItem(
|
|
120
|
+
key="has_lmcache",
|
|
121
|
+
display_name="LMCache Server",
|
|
122
|
+
description=(
|
|
123
|
+
"Do you have a running LMCache server? "
|
|
124
|
+
"It can auto-detect KV cache size information."
|
|
125
|
+
),
|
|
126
|
+
input_type="bool",
|
|
127
|
+
default=True,
|
|
128
|
+
required=False,
|
|
129
|
+
phase=PHASE_REQUIRED,
|
|
130
|
+
),
|
|
131
|
+
ConfigItem(
|
|
132
|
+
key="lmcache_url",
|
|
133
|
+
display_name="LMCache Server URL",
|
|
134
|
+
description="URL of the running LMCache HTTP server.",
|
|
135
|
+
input_type="text",
|
|
136
|
+
default="http://localhost:8080",
|
|
137
|
+
required=False,
|
|
138
|
+
condition=_has_lmcache,
|
|
139
|
+
phase=PHASE_REQUIRED,
|
|
140
|
+
),
|
|
141
|
+
ConfigItem(
|
|
142
|
+
key="tokens_per_gb_kvcache",
|
|
143
|
+
display_name="Tokens per GB KV cache",
|
|
144
|
+
description=(
|
|
145
|
+
"How many tokens fit in 1 GB of KV cache for your model.\n"
|
|
146
|
+
" If using vLLM, look for these lines in the startup log:\n"
|
|
147
|
+
' "Available KV cache memory: XX.XX GiB"\n'
|
|
148
|
+
' "GPU KV cache size: XXX,XXX tokens"\n'
|
|
149
|
+
" Then compute: tokens_per_gb = "
|
|
150
|
+
"GPU_KV_cache_tokens / Available_KV_cache_GiB"
|
|
151
|
+
),
|
|
152
|
+
input_type="int",
|
|
153
|
+
default=None,
|
|
154
|
+
required=True,
|
|
155
|
+
condition=_no_lmcache_url,
|
|
156
|
+
phase=PHASE_REQUIRED,
|
|
157
|
+
),
|
|
158
|
+
# ── Phase 2: General ──────────────────────────────────────────────
|
|
159
|
+
ConfigItem(
|
|
160
|
+
key="model",
|
|
161
|
+
display_name="Model name",
|
|
162
|
+
description=(
|
|
163
|
+
"The model served by the engine. "
|
|
164
|
+
"Leave empty to auto-detect from the engine."
|
|
165
|
+
),
|
|
166
|
+
input_type="text",
|
|
167
|
+
default="",
|
|
168
|
+
phase=PHASE_GENERAL,
|
|
169
|
+
),
|
|
170
|
+
ConfigItem(
|
|
171
|
+
key="kv_cache_volume",
|
|
172
|
+
display_name="KV cache volume (GB)",
|
|
173
|
+
description="Target active KV cache size for the benchmark.",
|
|
174
|
+
input_type="float",
|
|
175
|
+
default=100.0,
|
|
176
|
+
phase=PHASE_GENERAL,
|
|
177
|
+
),
|
|
178
|
+
# ── Phase 3: long-doc-permutator ─────────────────────────────────
|
|
179
|
+
ConfigItem(
|
|
180
|
+
key="ldp_num_contexts",
|
|
181
|
+
display_name="Number of contexts",
|
|
182
|
+
description="Number of unique context documents to generate.",
|
|
183
|
+
input_type="int",
|
|
184
|
+
default=5,
|
|
185
|
+
condition=_workload_is("long-doc-permutator"),
|
|
186
|
+
phase=PHASE_WORKLOAD,
|
|
187
|
+
),
|
|
188
|
+
ConfigItem(
|
|
189
|
+
key="ldp_context_length",
|
|
190
|
+
display_name="Context length (tokens)",
|
|
191
|
+
description="Token length of each context document.",
|
|
192
|
+
input_type="int",
|
|
193
|
+
default=5000,
|
|
194
|
+
condition=_workload_is("long-doc-permutator"),
|
|
195
|
+
phase=PHASE_WORKLOAD,
|
|
196
|
+
),
|
|
197
|
+
ConfigItem(
|
|
198
|
+
key="ldp_system_prompt_length",
|
|
199
|
+
display_name="System prompt length (tokens)",
|
|
200
|
+
description="Token length of the shared system prompt. Use 0 for none.",
|
|
201
|
+
input_type="int",
|
|
202
|
+
default=1000,
|
|
203
|
+
condition=_workload_is("long-doc-permutator"),
|
|
204
|
+
phase=PHASE_WORKLOAD,
|
|
205
|
+
),
|
|
206
|
+
ConfigItem(
|
|
207
|
+
key="ldp_num_permutations",
|
|
208
|
+
display_name="Number of permutations",
|
|
209
|
+
description="Distinct permutations to send. Capped at N! (N = num_contexts).",
|
|
210
|
+
input_type="int",
|
|
211
|
+
default=10,
|
|
212
|
+
condition=_workload_is("long-doc-permutator"),
|
|
213
|
+
phase=PHASE_WORKLOAD,
|
|
214
|
+
),
|
|
215
|
+
ConfigItem(
|
|
216
|
+
key="ldp_num_inflight_requests",
|
|
217
|
+
display_name="Max inflight requests",
|
|
218
|
+
description="Maximum concurrent in-flight requests.",
|
|
219
|
+
input_type="int",
|
|
220
|
+
default=1,
|
|
221
|
+
condition=_workload_is("long-doc-permutator"),
|
|
222
|
+
phase=PHASE_WORKLOAD,
|
|
223
|
+
),
|
|
224
|
+
# ── Phase 3: long-doc-qa ──────────────────────────────────────────
|
|
225
|
+
ConfigItem(
|
|
226
|
+
key="ldqa_document_length",
|
|
227
|
+
display_name="Document length (tokens)",
|
|
228
|
+
description="Token length of each synthetic document.",
|
|
229
|
+
input_type="int",
|
|
230
|
+
default=10000,
|
|
231
|
+
condition=_workload_is("long-doc-qa"),
|
|
232
|
+
phase=PHASE_WORKLOAD,
|
|
233
|
+
),
|
|
234
|
+
ConfigItem(
|
|
235
|
+
key="ldqa_query_per_document",
|
|
236
|
+
display_name="Queries per document",
|
|
237
|
+
description="Number of questions asked per document.",
|
|
238
|
+
input_type="int",
|
|
239
|
+
default=2,
|
|
240
|
+
condition=_workload_is("long-doc-qa"),
|
|
241
|
+
phase=PHASE_WORKLOAD,
|
|
242
|
+
),
|
|
243
|
+
ConfigItem(
|
|
244
|
+
key="ldqa_shuffle_policy",
|
|
245
|
+
display_name="Shuffle policy",
|
|
246
|
+
description="How benchmark requests are ordered.",
|
|
247
|
+
input_type="choice",
|
|
248
|
+
default="random",
|
|
249
|
+
choices=[
|
|
250
|
+
("random", "Shuffle all (doc, query) pairs randomly"),
|
|
251
|
+
("tile", "Process queries round by round across all documents"),
|
|
252
|
+
],
|
|
253
|
+
condition=_workload_is("long-doc-qa"),
|
|
254
|
+
phase=PHASE_WORKLOAD,
|
|
255
|
+
),
|
|
256
|
+
ConfigItem(
|
|
257
|
+
key="ldqa_num_inflight_requests",
|
|
258
|
+
display_name="Max inflight requests",
|
|
259
|
+
description="Maximum concurrent in-flight requests.",
|
|
260
|
+
input_type="int",
|
|
261
|
+
default=3,
|
|
262
|
+
condition=_workload_is("long-doc-qa"),
|
|
263
|
+
phase=PHASE_WORKLOAD,
|
|
264
|
+
),
|
|
265
|
+
# ── Phase 3: multi-round-chat ─────────────────────────────────────
|
|
266
|
+
ConfigItem(
|
|
267
|
+
key="mrc_shared_prompt_length",
|
|
268
|
+
display_name="System prompt length (tokens)",
|
|
269
|
+
description="Token length of the system prompt per session.",
|
|
270
|
+
input_type="int",
|
|
271
|
+
default=2000,
|
|
272
|
+
condition=_workload_is("multi-round-chat"),
|
|
273
|
+
phase=PHASE_WORKLOAD,
|
|
274
|
+
),
|
|
275
|
+
ConfigItem(
|
|
276
|
+
key="mrc_chat_history_length",
|
|
277
|
+
display_name="Chat history length (tokens)",
|
|
278
|
+
description="Token length of pre-filled conversation history.",
|
|
279
|
+
input_type="int",
|
|
280
|
+
default=10000,
|
|
281
|
+
condition=_workload_is("multi-round-chat"),
|
|
282
|
+
phase=PHASE_WORKLOAD,
|
|
283
|
+
),
|
|
284
|
+
ConfigItem(
|
|
285
|
+
key="mrc_user_input_length",
|
|
286
|
+
display_name="User input length (tokens)",
|
|
287
|
+
description="Tokens per user query in each round.",
|
|
288
|
+
input_type="int",
|
|
289
|
+
default=50,
|
|
290
|
+
condition=_workload_is("multi-round-chat"),
|
|
291
|
+
phase=PHASE_WORKLOAD,
|
|
292
|
+
),
|
|
293
|
+
ConfigItem(
|
|
294
|
+
key="mrc_output_length",
|
|
295
|
+
display_name="Output length (tokens)",
|
|
296
|
+
description="Max tokens to generate per response.",
|
|
297
|
+
input_type="int",
|
|
298
|
+
default=200,
|
|
299
|
+
condition=_workload_is("multi-round-chat"),
|
|
300
|
+
phase=PHASE_WORKLOAD,
|
|
301
|
+
),
|
|
302
|
+
ConfigItem(
|
|
303
|
+
key="mrc_qps",
|
|
304
|
+
display_name="Queries per second",
|
|
305
|
+
description="Target request dispatch rate.",
|
|
306
|
+
input_type="float",
|
|
307
|
+
default=1.0,
|
|
308
|
+
condition=_workload_is("multi-round-chat"),
|
|
309
|
+
phase=PHASE_WORKLOAD,
|
|
310
|
+
),
|
|
311
|
+
ConfigItem(
|
|
312
|
+
key="mrc_duration",
|
|
313
|
+
display_name="Duration (seconds)",
|
|
314
|
+
description="How long the benchmark runs.",
|
|
315
|
+
input_type="float",
|
|
316
|
+
default=60.0,
|
|
317
|
+
condition=_workload_is("multi-round-chat"),
|
|
318
|
+
phase=PHASE_WORKLOAD,
|
|
319
|
+
),
|
|
320
|
+
# ── Phase 3: random-prefill ───────────────────────────────────────
|
|
321
|
+
ConfigItem(
|
|
322
|
+
key="rp_request_length",
|
|
323
|
+
display_name="Request length (tokens)",
|
|
324
|
+
description="Token length of each prefill request.",
|
|
325
|
+
input_type="int",
|
|
326
|
+
default=10000,
|
|
327
|
+
condition=_workload_is("random-prefill"),
|
|
328
|
+
phase=PHASE_WORKLOAD,
|
|
329
|
+
),
|
|
330
|
+
ConfigItem(
|
|
331
|
+
key="rp_num_requests",
|
|
332
|
+
display_name="Number of requests",
|
|
333
|
+
description="Total prefill requests to fire simultaneously.",
|
|
334
|
+
input_type="int",
|
|
335
|
+
default=50,
|
|
336
|
+
condition=_workload_is("random-prefill"),
|
|
337
|
+
phase=PHASE_WORKLOAD,
|
|
338
|
+
),
|
|
339
|
+
]
|
|
340
|
+
|
|
341
|
+
|
|
342
|
+
def get_items_by_phase(phase: int) -> list[ConfigItem]:
|
|
343
|
+
"""Return all items belonging to a given phase."""
|
|
344
|
+
return [item for item in ALL_ITEMS if item.phase == phase]
|
|
345
|
+
|
|
346
|
+
|
|
347
|
+
def get_item(key: str) -> ConfigItem:
|
|
348
|
+
"""Look up a ConfigItem by key. Raises KeyError if not found."""
|
|
349
|
+
for item in ALL_ITEMS:
|
|
350
|
+
if item.key == key:
|
|
351
|
+
return item
|
|
352
|
+
raise KeyError(f"No ConfigItem with key {key!r}")
|
|
@@ -0,0 +1,327 @@
|
|
|
1
|
+
# SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
"""Intermediate state tracker for interactive configuration.
|
|
3
|
+
|
|
4
|
+
``InteractiveState`` holds the partially-configured benchmark parameters,
|
|
5
|
+
can be initialized from CLI args or a saved JSON file, and can be
|
|
6
|
+
converted to the ``argparse.Namespace`` that ``_bench_engine()`` expects.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
# Standard
|
|
10
|
+
from typing import Any
|
|
11
|
+
import argparse
|
|
12
|
+
import json
|
|
13
|
+
|
|
14
|
+
# First Party
|
|
15
|
+
from lmcache.cli.commands.bench.engine_bench.interactive.schema import (
|
|
16
|
+
ALL_ITEMS,
|
|
17
|
+
PHASE_GENERAL,
|
|
18
|
+
PHASE_REQUIRED,
|
|
19
|
+
PHASE_WORKLOAD,
|
|
20
|
+
ConfigItem,
|
|
21
|
+
)
|
|
22
|
+
|
|
23
|
+
# Keys that exist on argparse.Namespace but are NOT part of the interactive
|
|
24
|
+
# config item registry (operational flags, handled separately).
|
|
25
|
+
_OUTPUT_KEYS = ("output_dir", "seed", "no_csv", "export_csv", "json", "quiet")
|
|
26
|
+
|
|
27
|
+
# Keys used only during the interactive flow, never serialized or
|
|
28
|
+
# passed to the orchestrator.
|
|
29
|
+
_INTERACTIVE_ONLY_KEYS = {"has_lmcache"}
|
|
30
|
+
|
|
31
|
+
# Keys excluded from exported JSON configs. These are either
|
|
32
|
+
# environment-specific (engine_url, lmcache_url) or interactive-only.
|
|
33
|
+
_EXPORT_EXCLUDED_KEYS = _INTERACTIVE_ONLY_KEYS | {"engine_url", "lmcache_url"}
|
|
34
|
+
|
|
35
|
+
# Mapping from ConfigItem.key to the argparse attribute name when they differ.
|
|
36
|
+
# Most keys match directly; these are the exceptions.
|
|
37
|
+
_KEY_TO_ATTR: dict[str, str] = {
|
|
38
|
+
"kv_cache_volume": "kv_cache_volume",
|
|
39
|
+
"tokens_per_gb_kvcache": "tokens_per_gb_kvcache",
|
|
40
|
+
}
|
|
41
|
+
|
|
42
|
+
# argparse attribute names where the CLI default is None (meaning "not set"),
|
|
43
|
+
# versus attributes where a non-None argparse default is the real default
|
|
44
|
+
# (e.g., kv_cache_volume defaults to 100.0).
|
|
45
|
+
_ARGPARSE_NONE_MEANS_UNSET = {
|
|
46
|
+
"engine_url",
|
|
47
|
+
"workload",
|
|
48
|
+
"model",
|
|
49
|
+
"lmcache_url",
|
|
50
|
+
"tokens_per_gb_kvcache",
|
|
51
|
+
}
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
class InteractiveState:
|
|
55
|
+
"""Tracks which config items have been set and their values.
|
|
56
|
+
|
|
57
|
+
Keys present in ``_values`` are considered "set". Missing keys are
|
|
58
|
+
"unset" and will either be prompted for or filled with defaults.
|
|
59
|
+
"""
|
|
60
|
+
|
|
61
|
+
def __init__(self) -> None:
|
|
62
|
+
self._values: dict[str, Any] = {}
|
|
63
|
+
|
|
64
|
+
# ------------------------------------------------------------------
|
|
65
|
+
# Basic accessors
|
|
66
|
+
# ------------------------------------------------------------------
|
|
67
|
+
|
|
68
|
+
def is_set(self, key: str) -> bool:
|
|
69
|
+
return key in self._values
|
|
70
|
+
|
|
71
|
+
def get(self, key: str, default: Any = None) -> Any:
|
|
72
|
+
return self._values.get(key, default)
|
|
73
|
+
|
|
74
|
+
def set(self, key: str, value: Any) -> None:
|
|
75
|
+
self._values[key] = value
|
|
76
|
+
|
|
77
|
+
@property
|
|
78
|
+
def values(self) -> dict[str, Any]:
|
|
79
|
+
return dict(self._values)
|
|
80
|
+
|
|
81
|
+
# ------------------------------------------------------------------
|
|
82
|
+
# Readiness checks
|
|
83
|
+
# ------------------------------------------------------------------
|
|
84
|
+
|
|
85
|
+
def is_ready(self) -> bool:
|
|
86
|
+
"""True when all required items (whose conditions are met) have values."""
|
|
87
|
+
for item in ALL_ITEMS:
|
|
88
|
+
if not item.required:
|
|
89
|
+
continue
|
|
90
|
+
if not self._condition_met(item):
|
|
91
|
+
continue
|
|
92
|
+
if not self.is_set(item.key):
|
|
93
|
+
return False
|
|
94
|
+
return True
|
|
95
|
+
|
|
96
|
+
def get_missing_required(self) -> list[ConfigItem]:
|
|
97
|
+
"""Return phase-1 items that still need user input.
|
|
98
|
+
|
|
99
|
+
Includes both required items that are unset, and non-required
|
|
100
|
+
phase-1 items (like ``lmcache_url``) that are relevant because
|
|
101
|
+
a downstream required item (``tokens_per_gb_kvcache``) is unset.
|
|
102
|
+
"""
|
|
103
|
+
missing: list[ConfigItem] = []
|
|
104
|
+
for item in ALL_ITEMS:
|
|
105
|
+
if item.phase != PHASE_REQUIRED:
|
|
106
|
+
continue
|
|
107
|
+
if not self._condition_met(item):
|
|
108
|
+
continue
|
|
109
|
+
if self.is_set(item.key):
|
|
110
|
+
continue
|
|
111
|
+
if item.required:
|
|
112
|
+
missing.append(item)
|
|
113
|
+
elif item.key in ("has_lmcache", "lmcache_url") and not self.is_set(
|
|
114
|
+
"tokens_per_gb_kvcache"
|
|
115
|
+
):
|
|
116
|
+
# Only ask about LMCache when tokens_per_gb is needed
|
|
117
|
+
missing.append(item)
|
|
118
|
+
return missing
|
|
119
|
+
|
|
120
|
+
def get_general_items(self) -> list[ConfigItem]:
|
|
121
|
+
"""Return phase-2 (general) items that are not yet set."""
|
|
122
|
+
return [
|
|
123
|
+
item
|
|
124
|
+
for item in ALL_ITEMS
|
|
125
|
+
if item.phase == PHASE_GENERAL
|
|
126
|
+
and not self.is_set(item.key)
|
|
127
|
+
and self._condition_met(item)
|
|
128
|
+
]
|
|
129
|
+
|
|
130
|
+
def get_workload_items(self) -> list[ConfigItem]:
|
|
131
|
+
"""Return phase-3 (workload-specific) items whose conditions are met."""
|
|
132
|
+
return [
|
|
133
|
+
item
|
|
134
|
+
for item in ALL_ITEMS
|
|
135
|
+
if item.phase == PHASE_WORKLOAD and self._condition_met(item)
|
|
136
|
+
]
|
|
137
|
+
|
|
138
|
+
def has_unconfigured_general(self) -> bool:
|
|
139
|
+
"""True if there are general items the user hasn't explicitly set."""
|
|
140
|
+
return len(self.get_general_items()) > 0
|
|
141
|
+
|
|
142
|
+
def has_workload_items(self) -> bool:
|
|
143
|
+
"""True if there are workload-specific items to configure."""
|
|
144
|
+
return len(self.get_workload_items()) > 0
|
|
145
|
+
|
|
146
|
+
def workload_items_all_default(self) -> bool:
|
|
147
|
+
"""True if no workload-specific items have been explicitly set."""
|
|
148
|
+
for item in ALL_ITEMS:
|
|
149
|
+
if item.phase != PHASE_WORKLOAD:
|
|
150
|
+
continue
|
|
151
|
+
if not self._condition_met(item):
|
|
152
|
+
continue
|
|
153
|
+
if self.is_set(item.key):
|
|
154
|
+
return False
|
|
155
|
+
return True
|
|
156
|
+
|
|
157
|
+
# ------------------------------------------------------------------
|
|
158
|
+
# Defaults
|
|
159
|
+
# ------------------------------------------------------------------
|
|
160
|
+
|
|
161
|
+
def fill_defaults(self) -> None:
|
|
162
|
+
"""Set all unset items (whose conditions are met) to their defaults."""
|
|
163
|
+
for item in ALL_ITEMS:
|
|
164
|
+
if self.is_set(item.key):
|
|
165
|
+
continue
|
|
166
|
+
if not self._condition_met(item):
|
|
167
|
+
continue
|
|
168
|
+
if item.default is not None:
|
|
169
|
+
self._values[item.key] = item.default
|
|
170
|
+
|
|
171
|
+
# ------------------------------------------------------------------
|
|
172
|
+
# Conversion: CLI args ↔ InteractiveState
|
|
173
|
+
# ------------------------------------------------------------------
|
|
174
|
+
|
|
175
|
+
@classmethod
|
|
176
|
+
def from_cli_args(cls, args: argparse.Namespace) -> "InteractiveState":
|
|
177
|
+
"""Build state from parsed CLI arguments.
|
|
178
|
+
|
|
179
|
+
Only sets values that the user explicitly provided (i.e., not the
|
|
180
|
+
argparse default). This lets us distinguish "user set
|
|
181
|
+
``--kv-cache-volume 100``" from "user didn't touch it".
|
|
182
|
+
"""
|
|
183
|
+
state = cls()
|
|
184
|
+
for item in ALL_ITEMS:
|
|
185
|
+
if item.key in _INTERACTIVE_ONLY_KEYS:
|
|
186
|
+
continue
|
|
187
|
+
attr = _KEY_TO_ATTR.get(item.key, item.key)
|
|
188
|
+
value = getattr(args, attr, None)
|
|
189
|
+
if value is None:
|
|
190
|
+
continue
|
|
191
|
+
# For keys where argparse default is None, any non-None value
|
|
192
|
+
# means the user set it.
|
|
193
|
+
if attr in _ARGPARSE_NONE_MEANS_UNSET:
|
|
194
|
+
state._values[item.key] = value
|
|
195
|
+
continue
|
|
196
|
+
# For keys with real argparse defaults, we can't easily tell
|
|
197
|
+
# if the user typed --kv-cache-volume 100 vs it being the
|
|
198
|
+
# default. We mark it as "set" only if it differs from the
|
|
199
|
+
# schema default. This is imperfect but good enough — the
|
|
200
|
+
# worst case is we re-prompt for a value the user explicitly
|
|
201
|
+
# set to the default.
|
|
202
|
+
if item.default is not None and value == item.default:
|
|
203
|
+
continue
|
|
204
|
+
state._values[item.key] = value
|
|
205
|
+
|
|
206
|
+
# Derive has_lmcache from lmcache_url if provided via CLI
|
|
207
|
+
if state.is_set("lmcache_url"):
|
|
208
|
+
state._values["has_lmcache"] = True
|
|
209
|
+
|
|
210
|
+
return state
|
|
211
|
+
|
|
212
|
+
def to_namespace(self) -> argparse.Namespace:
|
|
213
|
+
"""Convert to an ``argparse.Namespace`` compatible with ``_bench_engine``.
|
|
214
|
+
|
|
215
|
+
Fills defaults for any unset items, then builds the namespace
|
|
216
|
+
with the attribute names that ``parse_args_to_config()`` and
|
|
217
|
+
``create_workload()`` expect.
|
|
218
|
+
"""
|
|
219
|
+
self.fill_defaults()
|
|
220
|
+
ns = argparse.Namespace()
|
|
221
|
+
|
|
222
|
+
# Map state keys to namespace attributes
|
|
223
|
+
for item in ALL_ITEMS:
|
|
224
|
+
if item.key in _INTERACTIVE_ONLY_KEYS:
|
|
225
|
+
continue
|
|
226
|
+
attr = _KEY_TO_ATTR.get(item.key, item.key)
|
|
227
|
+
# Only fall back to schema default if the item's condition is met.
|
|
228
|
+
# This prevents lmcache_url's default from leaking when
|
|
229
|
+
# has_lmcache is not set.
|
|
230
|
+
if item.key in self._values:
|
|
231
|
+
value = self._values[item.key]
|
|
232
|
+
elif self._condition_met(item):
|
|
233
|
+
value = item.default
|
|
234
|
+
else:
|
|
235
|
+
value = None
|
|
236
|
+
setattr(ns, attr, value)
|
|
237
|
+
|
|
238
|
+
# Output settings (not in the interactive registry)
|
|
239
|
+
ns.output_dir = self._values.get("output_dir", ".")
|
|
240
|
+
ns.seed = self._values.get("seed", 42)
|
|
241
|
+
ns.no_csv = self._values.get("no_csv", False)
|
|
242
|
+
ns.json = self._values.get("export_json", False)
|
|
243
|
+
ns.quiet = self._values.get("quiet", False)
|
|
244
|
+
ns.bench_target = "engine"
|
|
245
|
+
|
|
246
|
+
# Ensure format/output attrs exist for create_metrics
|
|
247
|
+
if not hasattr(ns, "format"):
|
|
248
|
+
ns.format = None
|
|
249
|
+
if not hasattr(ns, "output"):
|
|
250
|
+
ns.output = None
|
|
251
|
+
|
|
252
|
+
return ns
|
|
253
|
+
|
|
254
|
+
# ------------------------------------------------------------------
|
|
255
|
+
# Conversion: JSON ↔ InteractiveState
|
|
256
|
+
# ------------------------------------------------------------------
|
|
257
|
+
|
|
258
|
+
def to_json(self) -> dict[str, Any]:
|
|
259
|
+
"""Serialize to a JSON-compatible dict for config export.
|
|
260
|
+
|
|
261
|
+
Excludes environment-specific keys (``engine_url``,
|
|
262
|
+
``lmcache_url``) and interactive-only keys so the exported
|
|
263
|
+
config is portable and works without an LMCache server.
|
|
264
|
+
"""
|
|
265
|
+
self.fill_defaults()
|
|
266
|
+
return {k: v for k, v in self._values.items() if k not in _EXPORT_EXCLUDED_KEYS}
|
|
267
|
+
|
|
268
|
+
@classmethod
|
|
269
|
+
def from_json(cls, data: dict[str, Any]) -> "InteractiveState":
|
|
270
|
+
"""Load from a saved config JSON dict."""
|
|
271
|
+
state = cls()
|
|
272
|
+
for key, value in data.items():
|
|
273
|
+
state._values[key] = value
|
|
274
|
+
return state
|
|
275
|
+
|
|
276
|
+
def save_json(self, path: str) -> None:
|
|
277
|
+
"""Export the current state to a JSON file."""
|
|
278
|
+
with open(path, "w") as f:
|
|
279
|
+
json.dump(self.to_json(), f, indent=2)
|
|
280
|
+
f.write("\n")
|
|
281
|
+
|
|
282
|
+
@classmethod
|
|
283
|
+
def load_json(cls, path: str) -> "InteractiveState":
|
|
284
|
+
"""Load state from a JSON config file."""
|
|
285
|
+
with open(path) as f:
|
|
286
|
+
data = json.load(f)
|
|
287
|
+
return cls.from_json(data)
|
|
288
|
+
|
|
289
|
+
def merge_cli_args(self, args: argparse.Namespace) -> None:
|
|
290
|
+
"""Merge CLI args on top of existing state (CLI args win)."""
|
|
291
|
+
cli_state = InteractiveState.from_cli_args(args)
|
|
292
|
+
self._values.update(cli_state._values)
|
|
293
|
+
|
|
294
|
+
# ------------------------------------------------------------------
|
|
295
|
+
# Summary
|
|
296
|
+
# ------------------------------------------------------------------
|
|
297
|
+
|
|
298
|
+
def summary_lines(self) -> list[tuple[str, str]]:
|
|
299
|
+
"""Return ``(label, value_str)`` pairs for the config summary."""
|
|
300
|
+
lines: list[tuple[str, str]] = []
|
|
301
|
+
for item in ALL_ITEMS:
|
|
302
|
+
if item.key in _INTERACTIVE_ONLY_KEYS:
|
|
303
|
+
continue
|
|
304
|
+
if not self._condition_met(item):
|
|
305
|
+
continue
|
|
306
|
+
value = self._values.get(item.key, item.default)
|
|
307
|
+
if value is None or value == "":
|
|
308
|
+
if item.key == "model":
|
|
309
|
+
display = "(auto-detect)"
|
|
310
|
+
elif item.key == "lmcache_url":
|
|
311
|
+
continue # skip empty lmcache_url
|
|
312
|
+
else:
|
|
313
|
+
display = "(not set)"
|
|
314
|
+
else:
|
|
315
|
+
display = str(value)
|
|
316
|
+
lines.append((item.display_name, display))
|
|
317
|
+
return lines
|
|
318
|
+
|
|
319
|
+
# ------------------------------------------------------------------
|
|
320
|
+
# Internal
|
|
321
|
+
# ------------------------------------------------------------------
|
|
322
|
+
|
|
323
|
+
def _condition_met(self, item: ConfigItem) -> bool:
|
|
324
|
+
"""Check whether an item's condition is satisfied."""
|
|
325
|
+
if item.condition is None:
|
|
326
|
+
return True
|
|
327
|
+
return item.condition(self._values)
|