lmcache-cli 0.4.5.dev0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- lmcache/__init__.py +84 -0
- lmcache/_version.py +24 -0
- lmcache/cli/__init__.py +1 -0
- lmcache/cli/commands/__init__.py +34 -0
- lmcache/cli/commands/base.py +157 -0
- lmcache/cli/commands/bench/__init__.py +557 -0
- lmcache/cli/commands/bench/engine_bench/__init__.py +1 -0
- lmcache/cli/commands/bench/engine_bench/config.py +245 -0
- lmcache/cli/commands/bench/engine_bench/interactive/__init__.py +274 -0
- lmcache/cli/commands/bench/engine_bench/interactive/config.json +10 -0
- lmcache/cli/commands/bench/engine_bench/interactive/schema.py +352 -0
- lmcache/cli/commands/bench/engine_bench/interactive/state.py +327 -0
- lmcache/cli/commands/bench/engine_bench/interactive/terminal.py +291 -0
- lmcache/cli/commands/bench/engine_bench/progress.py +145 -0
- lmcache/cli/commands/bench/engine_bench/request_sender.py +232 -0
- lmcache/cli/commands/bench/engine_bench/stats.py +275 -0
- lmcache/cli/commands/bench/engine_bench/workloads/__init__.py +153 -0
- lmcache/cli/commands/bench/engine_bench/workloads/base.py +122 -0
- lmcache/cli/commands/bench/engine_bench/workloads/long_doc_permutator.py +435 -0
- lmcache/cli/commands/bench/engine_bench/workloads/long_doc_qa.py +281 -0
- lmcache/cli/commands/bench/engine_bench/workloads/multi_round_chat.py +337 -0
- lmcache/cli/commands/bench/engine_bench/workloads/random_prefill.py +178 -0
- lmcache/cli/commands/describe.py +310 -0
- lmcache/cli/commands/kvcache.py +133 -0
- lmcache/cli/commands/mock.py +75 -0
- lmcache/cli/commands/ping.py +113 -0
- lmcache/cli/commands/query/__init__.py +155 -0
- lmcache/cli/commands/query/prompt.py +134 -0
- lmcache/cli/commands/query/request.py +357 -0
- lmcache/cli/commands/server.py +99 -0
- lmcache/cli/commands/tool/__init__.py +63 -0
- lmcache/cli/commands/tool/cache_simulator.py +113 -0
- lmcache/cli/commands/trace/__init__.py +505 -0
- lmcache/cli/commands/trace/dispatch.py +249 -0
- lmcache/cli/commands/trace/driver.py +372 -0
- lmcache/cli/commands/trace/stats.py +289 -0
- lmcache/cli/documents/lmcache.txt +11 -0
- lmcache/cli/main.py +42 -0
- lmcache/cli/metrics/__init__.py +29 -0
- lmcache/cli/metrics/formatter.py +171 -0
- lmcache/cli/metrics/handler.py +94 -0
- lmcache/cli/metrics/metrics.py +161 -0
- lmcache/cli/metrics/section.py +77 -0
- lmcache/connections.py +173 -0
- lmcache/integration/__init__.py +2 -0
- lmcache/integration/base_service_factory.py +165 -0
- lmcache/integration/request_telemetry/__init__.py +1 -0
- lmcache/integration/request_telemetry/base.py +51 -0
- lmcache/integration/request_telemetry/factory.py +113 -0
- lmcache/integration/request_telemetry/fastapi.py +109 -0
- lmcache/integration/request_telemetry/noop.py +35 -0
- lmcache/integration/sglang/__init__.py +2 -0
- lmcache/integration/sglang/sglang_adapter.py +326 -0
- lmcache/integration/sglang/utils.py +39 -0
- lmcache/integration/vllm/__init__.py +1 -0
- lmcache/integration/vllm/lmcache_connector_v1.py +213 -0
- lmcache/integration/vllm/lmcache_connector_v1_085.py +150 -0
- lmcache/integration/vllm/lmcache_mp_connector_0180.py +1072 -0
- lmcache/integration/vllm/tests/test_mm_hash_utils.py +112 -0
- lmcache/integration/vllm/utils.py +433 -0
- lmcache/integration/vllm/vllm_multi_process_adapter.py +1090 -0
- lmcache/integration/vllm/vllm_service_factory.py +339 -0
- lmcache/integration/vllm/vllm_v1_adapter.py +1713 -0
- lmcache/logging.py +107 -0
- lmcache/native_storage_ops.pyi +230 -0
- lmcache/non_cuda_equivalents.py +1424 -0
- lmcache/observability.py +1958 -0
- lmcache/storage_backend/serde/__init__.py +1 -0
- lmcache/storage_backend/serde/cachegen_basics.py +210 -0
- lmcache/storage_backend/serde/cachegen_decoder.py +207 -0
- lmcache/storage_backend/serde/cachegen_encoder.py +394 -0
- lmcache/storage_backend/serde/serde.py +75 -0
- lmcache/tools/__init__.py +1 -0
- lmcache/tools/cache_simulator/README.md +392 -0
- lmcache/tools/cache_simulator/__init__.py +1 -0
- lmcache/tools/cache_simulator/docs/simulate_example.png +0 -0
- lmcache/tools/cache_simulator/docs/sweep_example.png +0 -0
- lmcache/tools/cache_simulator/gen_bench_dataset.py +360 -0
- lmcache/tools/cache_simulator/lru_cache.py +124 -0
- lmcache/tools/cache_simulator/plot_hit_rate.py +231 -0
- lmcache/tools/cache_simulator/simulator.py +795 -0
- lmcache/tools/controller_benchmark/README.md +161 -0
- lmcache/tools/controller_benchmark/__init__.py +1 -0
- lmcache/tools/controller_benchmark/__main__.py +331 -0
- lmcache/tools/controller_benchmark/benchmark.py +660 -0
- lmcache/tools/controller_benchmark/config.py +44 -0
- lmcache/tools/controller_benchmark/constants.py +10 -0
- lmcache/tools/controller_benchmark/handlers/__init__.py +46 -0
- lmcache/tools/controller_benchmark/handlers/admit.py +52 -0
- lmcache/tools/controller_benchmark/handlers/base.py +47 -0
- lmcache/tools/controller_benchmark/handlers/deregister.py +49 -0
- lmcache/tools/controller_benchmark/handlers/evict.py +52 -0
- lmcache/tools/controller_benchmark/handlers/heartbeat.py +56 -0
- lmcache/tools/controller_benchmark/handlers/p2p_lookup.py +47 -0
- lmcache/tools/controller_benchmark/handlers/register.py +56 -0
- lmcache/tools/mp_status_viewer/__init__.py +1 -0
- lmcache/tools/mp_status_viewer/__main__.py +95 -0
- lmcache/usage_context.py +417 -0
- lmcache/utils.py +665 -0
- lmcache/v1/__init__.py +2 -0
- lmcache/v1/api_server/__init__.py +2 -0
- lmcache/v1/api_server/__main__.py +537 -0
- lmcache/v1/basic_check.py +112 -0
- lmcache/v1/cache_controller/__init__.py +9 -0
- lmcache/v1/cache_controller/commands/__init__.py +15 -0
- lmcache/v1/cache_controller/commands/base.py +35 -0
- lmcache/v1/cache_controller/commands/full_sync.py +49 -0
- lmcache/v1/cache_controller/config.py +176 -0
- lmcache/v1/cache_controller/controller_manager.py +535 -0
- lmcache/v1/cache_controller/controllers/__init__.py +11 -0
- lmcache/v1/cache_controller/controllers/full_sync_tracker.py +473 -0
- lmcache/v1/cache_controller/controllers/kv_controller.py +439 -0
- lmcache/v1/cache_controller/controllers/registration_controller.py +282 -0
- lmcache/v1/cache_controller/executor.py +463 -0
- lmcache/v1/cache_controller/frontend/static/css/style.css +201 -0
- lmcache/v1/cache_controller/frontend/static/img/logo.png +0 -0
- lmcache/v1/cache_controller/frontend/static/index.html +234 -0
- lmcache/v1/cache_controller/frontend/static/js/controller_app.js +660 -0
- lmcache/v1/cache_controller/full_sync_sender.py +475 -0
- lmcache/v1/cache_controller/locks.py +149 -0
- lmcache/v1/cache_controller/message.py +828 -0
- lmcache/v1/cache_controller/observability.py +208 -0
- lmcache/v1/cache_controller/utils.py +679 -0
- lmcache/v1/cache_controller/worker.py +665 -0
- lmcache/v1/cache_engine.py +2058 -0
- lmcache/v1/cache_interface.py +19 -0
- lmcache/v1/check/__init__.py +74 -0
- lmcache/v1/check/check_mode_gen.py +86 -0
- lmcache/v1/check/check_mode_test_l2_adapter.py +284 -0
- lmcache/v1/check/check_mode_test_remote.py +155 -0
- lmcache/v1/check/check_mode_test_storage_manager.py +142 -0
- lmcache/v1/check/utils.py +571 -0
- lmcache/v1/compute/__init__.py +2 -0
- lmcache/v1/compute/attention/__init__.py +0 -0
- lmcache/v1/compute/attention/abstract.py +39 -0
- lmcache/v1/compute/attention/flash_attn.py +129 -0
- lmcache/v1/compute/attention/flash_infer_sparse.py +284 -0
- lmcache/v1/compute/attention/metadata.py +85 -0
- lmcache/v1/compute/attention/utils.py +14 -0
- lmcache/v1/compute/blend/__init__.py +7 -0
- lmcache/v1/compute/blend/blender.py +168 -0
- lmcache/v1/compute/blend/metadata.py +34 -0
- lmcache/v1/compute/blend/utils.py +63 -0
- lmcache/v1/compute/models/__init__.py +0 -0
- lmcache/v1/compute/models/base.py +141 -0
- lmcache/v1/compute/models/llama.py +9 -0
- lmcache/v1/compute/models/qwen3.py +24 -0
- lmcache/v1/compute/models/utils.py +68 -0
- lmcache/v1/compute/positional_encoding.py +199 -0
- lmcache/v1/config.py +848 -0
- lmcache/v1/config_base.py +848 -0
- lmcache/v1/distributed/api.py +248 -0
- lmcache/v1/distributed/config.py +321 -0
- lmcache/v1/distributed/error.py +64 -0
- lmcache/v1/distributed/eviction.py +192 -0
- lmcache/v1/distributed/eviction_policy/__init__.py +21 -0
- lmcache/v1/distributed/eviction_policy/factory.py +27 -0
- lmcache/v1/distributed/eviction_policy/lru.py +244 -0
- lmcache/v1/distributed/eviction_policy/noop.py +50 -0
- lmcache/v1/distributed/internal_api.py +170 -0
- lmcache/v1/distributed/l1_manager.py +835 -0
- lmcache/v1/distributed/l2_adapters/__init__.py +67 -0
- lmcache/v1/distributed/l2_adapters/base.py +360 -0
- lmcache/v1/distributed/l2_adapters/config.py +385 -0
- lmcache/v1/distributed/l2_adapters/factory.py +205 -0
- lmcache/v1/distributed/l2_adapters/fs_l2_adapter.py +747 -0
- lmcache/v1/distributed/l2_adapters/fs_native_l2_adapter.py +167 -0
- lmcache/v1/distributed/l2_adapters/mock_l2_adapter.py +516 -0
- lmcache/v1/distributed/l2_adapters/mooncake_store_l2_adapter.py +135 -0
- lmcache/v1/distributed/l2_adapters/native_connector_l2_adapter.py +468 -0
- lmcache/v1/distributed/l2_adapters/native_plugin_l2_adapter.py +199 -0
- lmcache/v1/distributed/l2_adapters/nixl_store_dynamic_l2_adapter.py +831 -0
- lmcache/v1/distributed/l2_adapters/nixl_store_l2_adapter.py +983 -0
- lmcache/v1/distributed/l2_adapters/plugin_l2_adapter.py +210 -0
- lmcache/v1/distributed/l2_adapters/resp_l2_adapter.py +176 -0
- lmcache/v1/distributed/memory_manager.py +179 -0
- lmcache/v1/distributed/storage_controller.py +39 -0
- lmcache/v1/distributed/storage_controllers/__init__.py +43 -0
- lmcache/v1/distributed/storage_controllers/eviction_controller.py +242 -0
- lmcache/v1/distributed/storage_controllers/prefetch_controller.py +830 -0
- lmcache/v1/distributed/storage_controllers/prefetch_policy.py +193 -0
- lmcache/v1/distributed/storage_controllers/store_controller.py +452 -0
- lmcache/v1/distributed/storage_controllers/store_policy.py +213 -0
- lmcache/v1/distributed/storage_manager.py +532 -0
- lmcache/v1/event_manager.py +145 -0
- lmcache/v1/exceptions/__init__.py +16 -0
- lmcache/v1/gpu_connector/__init__.py +126 -0
- lmcache/v1/gpu_connector/gpu_connectors.py +1906 -0
- lmcache/v1/gpu_connector/gpu_ops.py +85 -0
- lmcache/v1/gpu_connector/hpu_connector.py +326 -0
- lmcache/v1/gpu_connector/mock_gpu_connector.py +67 -0
- lmcache/v1/gpu_connector/utils.py +890 -0
- lmcache/v1/gpu_connector/xpu_connectors.py +916 -0
- lmcache/v1/health_monitor/__init__.py +1 -0
- lmcache/v1/health_monitor/base.py +587 -0
- lmcache/v1/health_monitor/checks/__init__.py +1 -0
- lmcache/v1/health_monitor/checks/remote_backend_check.py +304 -0
- lmcache/v1/health_monitor/constants.py +36 -0
- lmcache/v1/internal_api_server/__init__.py +0 -0
- lmcache/v1/internal_api_server/api_registry.py +59 -0
- lmcache/v1/internal_api_server/api_server.py +120 -0
- lmcache/v1/internal_api_server/common/__init__.py +1 -0
- lmcache/v1/internal_api_server/common/env_api.py +22 -0
- lmcache/v1/internal_api_server/common/loglevel_api.py +57 -0
- lmcache/v1/internal_api_server/common/metrics_api.py +29 -0
- lmcache/v1/internal_api_server/common/periodic_thread_api.py +138 -0
- lmcache/v1/internal_api_server/common/run_script_api.py +73 -0
- lmcache/v1/internal_api_server/common/thread_api.py +63 -0
- lmcache/v1/internal_api_server/controller/__init__.py +1 -0
- lmcache/v1/internal_api_server/controller/key_stats_api.py +81 -0
- lmcache/v1/internal_api_server/controller/worker_info_api.py +136 -0
- lmcache/v1/internal_api_server/utils.py +43 -0
- lmcache/v1/internal_api_server/vllm/__init__.py +1 -0
- lmcache/v1/internal_api_server/vllm/backend_api.py +221 -0
- lmcache/v1/internal_api_server/vllm/bypass_api.py +204 -0
- lmcache/v1/internal_api_server/vllm/cache_api.py +895 -0
- lmcache/v1/internal_api_server/vllm/chunk_statistics_api.py +141 -0
- lmcache/v1/internal_api_server/vllm/conf_api.py +147 -0
- lmcache/v1/internal_api_server/vllm/freeze_api.py +172 -0
- lmcache/v1/internal_api_server/vllm/hot_cache_api.py +184 -0
- lmcache/v1/internal_api_server/vllm/inference_api.py +65 -0
- lmcache/v1/internal_api_server/vllm/load_fs_chunks_api.py +320 -0
- lmcache/v1/internal_api_server/vllm/lookup_api.py +145 -0
- lmcache/v1/internal_api_server/vllm/version_api.py +25 -0
- lmcache/v1/kv_layer_groups.py +267 -0
- lmcache/v1/lazy_memory_allocator.py +284 -0
- lmcache/v1/lookup_client/__init__.py +25 -0
- lmcache/v1/lookup_client/abstract_client.py +77 -0
- lmcache/v1/lookup_client/async_lookup_message.py +50 -0
- lmcache/v1/lookup_client/chunk_statistics_lookup_client.py +200 -0
- lmcache/v1/lookup_client/factory.py +251 -0
- lmcache/v1/lookup_client/hit_limit_lookup_client.py +86 -0
- lmcache/v1/lookup_client/lmcache_async_lookup_client.py +407 -0
- lmcache/v1/lookup_client/lmcache_lookup_client.py +285 -0
- lmcache/v1/lookup_client/lmcache_lookup_client_bypass.py +99 -0
- lmcache/v1/lookup_client/mooncake_lookup_client.py +87 -0
- lmcache/v1/lookup_client/record_strategies/__init__.py +77 -0
- lmcache/v1/lookup_client/record_strategies/base.py +327 -0
- lmcache/v1/lookup_client/record_strategies/file_hash.py +130 -0
- lmcache/v1/lookup_client/record_strategies/memory_bloom_filter.py +81 -0
- lmcache/v1/manager.py +539 -0
- lmcache/v1/memory_management.py +2619 -0
- lmcache/v1/metadata.py +114 -0
- lmcache/v1/mp_observability/AGENTS.override.md +21 -0
- lmcache/v1/mp_observability/README.md +204 -0
- lmcache/v1/mp_observability/config.py +340 -0
- lmcache/v1/mp_observability/event.py +100 -0
- lmcache/v1/mp_observability/event_bus.py +313 -0
- lmcache/v1/mp_observability/otel_init.py +145 -0
- lmcache/v1/mp_observability/subscribers/__init__.py +28 -0
- lmcache/v1/mp_observability/subscribers/logging/__init__.py +19 -0
- lmcache/v1/mp_observability/subscribers/logging/l1.py +56 -0
- lmcache/v1/mp_observability/subscribers/logging/l2.py +73 -0
- lmcache/v1/mp_observability/subscribers/logging/lookup_hash.py +209 -0
- lmcache/v1/mp_observability/subscribers/logging/mp_server.py +90 -0
- lmcache/v1/mp_observability/subscribers/logging/sm.py +59 -0
- lmcache/v1/mp_observability/subscribers/metrics/__init__.py +20 -0
- lmcache/v1/mp_observability/subscribers/metrics/l0_lifecycle.py +290 -0
- lmcache/v1/mp_observability/subscribers/metrics/l1.py +55 -0
- lmcache/v1/mp_observability/subscribers/metrics/l1_lifecycle.py +166 -0
- lmcache/v1/mp_observability/subscribers/metrics/l2.py +121 -0
- lmcache/v1/mp_observability/subscribers/metrics/sm.py +69 -0
- lmcache/v1/mp_observability/subscribers/tracing/__init__.py +12 -0
- lmcache/v1/mp_observability/subscribers/tracing/mp_server.py +333 -0
- lmcache/v1/mp_observability/subscribers/tracing/span_registry.py +148 -0
- lmcache/v1/mp_observability/trace/__init__.py +50 -0
- lmcache/v1/mp_observability/trace/codecs.py +255 -0
- lmcache/v1/mp_observability/trace/decorator.py +147 -0
- lmcache/v1/mp_observability/trace/format.py +132 -0
- lmcache/v1/mp_observability/trace/lifecycle.py +83 -0
- lmcache/v1/mp_observability/trace/reader.py +167 -0
- lmcache/v1/mp_observability/trace/recorder.py +300 -0
- lmcache/v1/multiprocess/__init__.py +0 -0
- lmcache/v1/multiprocess/affinity_pool.py +102 -0
- lmcache/v1/multiprocess/blend_server_v2.py +891 -0
- lmcache/v1/multiprocess/config.py +253 -0
- lmcache/v1/multiprocess/custom_types.py +281 -0
- lmcache/v1/multiprocess/futures.py +194 -0
- lmcache/v1/multiprocess/gpu_context.py +511 -0
- lmcache/v1/multiprocess/http_server.py +235 -0
- lmcache/v1/multiprocess/mp_runtime_plugin_launcher.py +130 -0
- lmcache/v1/multiprocess/mq.py +732 -0
- lmcache/v1/multiprocess/protocol.py +86 -0
- lmcache/v1/multiprocess/protocols/README.md +213 -0
- lmcache/v1/multiprocess/protocols/__init__.py +127 -0
- lmcache/v1/multiprocess/protocols/base.py +89 -0
- lmcache/v1/multiprocess/protocols/blend.py +109 -0
- lmcache/v1/multiprocess/protocols/blend_v2.py +57 -0
- lmcache/v1/multiprocess/protocols/controller.py +53 -0
- lmcache/v1/multiprocess/protocols/debug.py +34 -0
- lmcache/v1/multiprocess/protocols/engine.py +146 -0
- lmcache/v1/multiprocess/protocols/observability.py +39 -0
- lmcache/v1/multiprocess/server.py +1134 -0
- lmcache/v1/multiprocess/session.py +190 -0
- lmcache/v1/multiprocess/token_hasher.py +441 -0
- lmcache/v1/offload_server/__init__.py +17 -0
- lmcache/v1/offload_server/abstract_server.py +37 -0
- lmcache/v1/offload_server/message.py +30 -0
- lmcache/v1/offload_server/zmq_server.py +122 -0
- lmcache/v1/periodic_thread.py +579 -0
- lmcache/v1/pin_monitor.py +246 -0
- lmcache/v1/plugin/__init__.py +0 -0
- lmcache/v1/plugin/runtime_plugin_launcher.py +211 -0
- lmcache/v1/protocol.py +317 -0
- lmcache/v1/rpc/__init__.py +17 -0
- lmcache/v1/rpc/transport.py +105 -0
- lmcache/v1/rpc/zmq_transport.py +213 -0
- lmcache/v1/rpc_utils.py +165 -0
- lmcache/v1/server/__init__.py +2 -0
- lmcache/v1/server/__main__.py +170 -0
- lmcache/v1/server/storage_backend/__init__.py +21 -0
- lmcache/v1/server/storage_backend/abstract_backend.py +80 -0
- lmcache/v1/server/storage_backend/local_backend.py +75 -0
- lmcache/v1/server/utils.py +21 -0
- lmcache/v1/standalone/__init__.py +1 -0
- lmcache/v1/standalone/__main__.py +583 -0
- lmcache/v1/standalone/manager.py +80 -0
- lmcache/v1/standalone/standalone_service_factory.py +86 -0
- lmcache/v1/storage_backend/__init__.py +313 -0
- lmcache/v1/storage_backend/abstract_backend.py +445 -0
- lmcache/v1/storage_backend/audit_backend.py +233 -0
- lmcache/v1/storage_backend/batched_message_sender.py +222 -0
- lmcache/v1/storage_backend/cache_policy/__init__.py +45 -0
- lmcache/v1/storage_backend/cache_policy/base_policy.py +87 -0
- lmcache/v1/storage_backend/cache_policy/fifo.py +58 -0
- lmcache/v1/storage_backend/cache_policy/lfu.py +105 -0
- lmcache/v1/storage_backend/cache_policy/lru.py +81 -0
- lmcache/v1/storage_backend/cache_policy/mru.py +61 -0
- lmcache/v1/storage_backend/connector/__init__.py +443 -0
- lmcache/v1/storage_backend/connector/audit_adapter.py +77 -0
- lmcache/v1/storage_backend/connector/audit_connector.py +320 -0
- lmcache/v1/storage_backend/connector/base_connector.py +379 -0
- lmcache/v1/storage_backend/connector/blackhole_adapter.py +21 -0
- lmcache/v1/storage_backend/connector/blackhole_connector.py +37 -0
- lmcache/v1/storage_backend/connector/eic_adapter.py +31 -0
- lmcache/v1/storage_backend/connector/eic_connector.py +757 -0
- lmcache/v1/storage_backend/connector/external_adapter.py +79 -0
- lmcache/v1/storage_backend/connector/fs_adapter.py +51 -0
- lmcache/v1/storage_backend/connector/fs_connector.py +403 -0
- lmcache/v1/storage_backend/connector/infinistore_adapter.py +56 -0
- lmcache/v1/storage_backend/connector/infinistore_connector.py +177 -0
- lmcache/v1/storage_backend/connector/instrumented_connector.py +219 -0
- lmcache/v1/storage_backend/connector/lm_adapter.py +31 -0
- lmcache/v1/storage_backend/connector/lm_connector.py +176 -0
- lmcache/v1/storage_backend/connector/mock_adapter.py +57 -0
- lmcache/v1/storage_backend/connector/mock_connector.py +349 -0
- lmcache/v1/storage_backend/connector/mooncakestore_adapter.py +43 -0
- lmcache/v1/storage_backend/connector/mooncakestore_connector.py +614 -0
- lmcache/v1/storage_backend/connector/redis_adapter.py +181 -0
- lmcache/v1/storage_backend/connector/redis_connector.py +828 -0
- lmcache/v1/storage_backend/connector/s3_adapter.py +59 -0
- lmcache/v1/storage_backend/connector/s3_connector.py +699 -0
- lmcache/v1/storage_backend/connector/sagemaker_hyperpod_adapter.py +233 -0
- lmcache/v1/storage_backend/connector/sagemaker_hyperpod_connector.py +987 -0
- lmcache/v1/storage_backend/connector/valkey_adapter.py +114 -0
- lmcache/v1/storage_backend/connector/valkey_connector.py +627 -0
- lmcache/v1/storage_backend/gds_backend.py +1199 -0
- lmcache/v1/storage_backend/job_executor/__init__.py +0 -0
- lmcache/v1/storage_backend/job_executor/base_executor.py +34 -0
- lmcache/v1/storage_backend/job_executor/pq_executor.py +235 -0
- lmcache/v1/storage_backend/local_cpu_backend.py +810 -0
- lmcache/v1/storage_backend/local_disk_backend.py +656 -0
- lmcache/v1/storage_backend/maru_backend.py +734 -0
- lmcache/v1/storage_backend/naive_serde/__init__.py +50 -0
- lmcache/v1/storage_backend/naive_serde/cachegen_basics.py +133 -0
- lmcache/v1/storage_backend/naive_serde/cachegen_decoder.py +135 -0
- lmcache/v1/storage_backend/naive_serde/cachegen_encoder.py +83 -0
- lmcache/v1/storage_backend/naive_serde/kivi_serde.py +22 -0
- lmcache/v1/storage_backend/naive_serde/naive_serde.py +18 -0
- lmcache/v1/storage_backend/naive_serde/serde.py +37 -0
- lmcache/v1/storage_backend/native_clients/connector_client_base.py +165 -0
- lmcache/v1/storage_backend/native_clients/resp_client.py +35 -0
- lmcache/v1/storage_backend/nixl_storage_backend.py +1400 -0
- lmcache/v1/storage_backend/p2p_backend.py +788 -0
- lmcache/v1/storage_backend/path_sharder.py +117 -0
- lmcache/v1/storage_backend/pd_backend.py +646 -0
- lmcache/v1/storage_backend/plugins/dax_backend.py +1443 -0
- lmcache/v1/storage_backend/plugins/rust_raw_block_backend.py +1361 -0
- lmcache/v1/storage_backend/remote_backend.py +624 -0
- lmcache/v1/storage_backend/resp_client.py +227 -0
- lmcache/v1/storage_backend/storage_backend_listener.py +19 -0
- lmcache/v1/storage_backend/storage_manager.py +1352 -0
- lmcache/v1/system_detection.py +110 -0
- lmcache/v1/token_database.py +551 -0
- lmcache/v1/transfer_channel/__init__.py +83 -0
- lmcache/v1/transfer_channel/abstract.py +285 -0
- lmcache/v1/transfer_channel/mock_memory_channel.py +156 -0
- lmcache/v1/transfer_channel/nixl_channel.py +639 -0
- lmcache/v1/transfer_channel/py_socket_channel.py +260 -0
- lmcache/v1/transfer_channel/transfer_utils.py +63 -0
- lmcache/v1/utils/__init__.py +1 -0
- lmcache/v1/utils/bloom_filter.py +109 -0
- lmcache/v1/utils/cache_utils.py +125 -0
- lmcache_cli-0.4.5.dev0.dist-info/METADATA +185 -0
- lmcache_cli-0.4.5.dev0.dist-info/RECORD +399 -0
- lmcache_cli-0.4.5.dev0.dist-info/WHEEL +5 -0
- lmcache_cli-0.4.5.dev0.dist-info/entry_points.txt +2 -0
- lmcache_cli-0.4.5.dev0.dist-info/licenses/LICENSE +201 -0
- lmcache_cli-0.4.5.dev0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,178 @@
|
|
|
1
|
+
# SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
"""Random-prefill workload for ``lmcache bench engine``."""
|
|
3
|
+
|
|
4
|
+
# Standard
|
|
5
|
+
from dataclasses import dataclass
|
|
6
|
+
import asyncio
|
|
7
|
+
|
|
8
|
+
# First Party
|
|
9
|
+
from lmcache.cli.commands.bench.engine_bench.progress import ProgressMonitor
|
|
10
|
+
from lmcache.cli.commands.bench.engine_bench.request_sender import (
|
|
11
|
+
RequestSender,
|
|
12
|
+
)
|
|
13
|
+
from lmcache.cli.commands.bench.engine_bench.stats import StatsCollector
|
|
14
|
+
from lmcache.cli.commands.bench.engine_bench.workloads.base import BaseWorkload
|
|
15
|
+
from lmcache.logging import init_logger
|
|
16
|
+
|
|
17
|
+
logger = init_logger(__name__)
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
@dataclass
|
|
21
|
+
class RandomPrefillConfig:
|
|
22
|
+
"""Workload-specific config for the random-prefill workload."""
|
|
23
|
+
|
|
24
|
+
request_length: int = 10000
|
|
25
|
+
num_requests: int = 50
|
|
26
|
+
|
|
27
|
+
def __post_init__(self) -> None:
|
|
28
|
+
if self.request_length <= 0:
|
|
29
|
+
raise ValueError(
|
|
30
|
+
f"request_length must be positive, got {self.request_length}"
|
|
31
|
+
)
|
|
32
|
+
if self.num_requests < 1:
|
|
33
|
+
raise ValueError(f"num_requests must be >= 1, got {self.num_requests}")
|
|
34
|
+
|
|
35
|
+
@classmethod
|
|
36
|
+
def resolve(
|
|
37
|
+
cls,
|
|
38
|
+
request_length: int = 10000,
|
|
39
|
+
num_requests: int = 50,
|
|
40
|
+
) -> "RandomPrefillConfig":
|
|
41
|
+
"""Create a config from CLI args.
|
|
42
|
+
|
|
43
|
+
Unlike other workloads, random-prefill does not use the KV cache
|
|
44
|
+
budget to compute request count — the user specifies it directly.
|
|
45
|
+
"""
|
|
46
|
+
return cls(
|
|
47
|
+
request_length=request_length,
|
|
48
|
+
num_requests=num_requests,
|
|
49
|
+
)
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
class RandomPrefillWorkload(BaseWorkload):
|
|
53
|
+
"""Workload that tests prefill speed by firing all requests at once.
|
|
54
|
+
|
|
55
|
+
Generates synthetic prompts of ``request_length`` tokens and dispatches
|
|
56
|
+
all ``num_requests`` simultaneously with ``max_tokens=1``. There is
|
|
57
|
+
no warmup phase.
|
|
58
|
+
"""
|
|
59
|
+
|
|
60
|
+
def __init__(
|
|
61
|
+
self,
|
|
62
|
+
config: RandomPrefillConfig,
|
|
63
|
+
request_sender: RequestSender,
|
|
64
|
+
stats_collector: StatsCollector,
|
|
65
|
+
progress_monitor: ProgressMonitor,
|
|
66
|
+
seed: int = 42,
|
|
67
|
+
) -> None:
|
|
68
|
+
super().__init__(request_sender, stats_collector, progress_monitor)
|
|
69
|
+
self._config = config
|
|
70
|
+
self._seed = seed
|
|
71
|
+
|
|
72
|
+
self._prompts = self._generate_prompts()
|
|
73
|
+
self._dispatched = False
|
|
74
|
+
self._pending_tasks: set[asyncio.Task] = set()
|
|
75
|
+
|
|
76
|
+
def log_config(self) -> None:
|
|
77
|
+
"""Log key workload config before the benchmark starts."""
|
|
78
|
+
c = self._config
|
|
79
|
+
B = "\033[1m" # bold
|
|
80
|
+
C = "\033[96m" # cyan
|
|
81
|
+
Y = "\033[93m" # yellow
|
|
82
|
+
R = "\033[0m" # reset
|
|
83
|
+
print(
|
|
84
|
+
f"{B}{'═' * 50}{R}\n"
|
|
85
|
+
f"{B} Workload: {C}random-prefill{R}\n"
|
|
86
|
+
f"{B}{'─' * 50}{R}\n"
|
|
87
|
+
f" Requests: {Y}{c.num_requests}{R}\n"
|
|
88
|
+
f" Request length: {Y}{c.request_length}{R} tokens\n"
|
|
89
|
+
f" Max output: {Y}1{R} token\n"
|
|
90
|
+
f"{B}{'═' * 50}{R}"
|
|
91
|
+
)
|
|
92
|
+
|
|
93
|
+
# ------------------------------------------------------------------
|
|
94
|
+
# Prompt generation
|
|
95
|
+
# ------------------------------------------------------------------
|
|
96
|
+
|
|
97
|
+
def _generate_prompts(self) -> list[str]:
|
|
98
|
+
"""Generate synthetic prompts of approximately ``request_length`` tokens."""
|
|
99
|
+
prompts: list[str] = []
|
|
100
|
+
for i in range(self._config.num_requests):
|
|
101
|
+
prefix = f"Request {i}: "
|
|
102
|
+
body = " ".join(["hi"] * max(self._config.request_length - 10, 1))
|
|
103
|
+
prompts.append(prefix + body)
|
|
104
|
+
logger.debug(
|
|
105
|
+
"Generated %d prompts of ~%d tokens each",
|
|
106
|
+
len(prompts),
|
|
107
|
+
self._config.request_length,
|
|
108
|
+
)
|
|
109
|
+
return prompts
|
|
110
|
+
|
|
111
|
+
# ------------------------------------------------------------------
|
|
112
|
+
# Warmup
|
|
113
|
+
# ------------------------------------------------------------------
|
|
114
|
+
|
|
115
|
+
async def warmup(self) -> None:
|
|
116
|
+
"""No warmup for random-prefill."""
|
|
117
|
+
|
|
118
|
+
# ------------------------------------------------------------------
|
|
119
|
+
# Benchmark dispatch
|
|
120
|
+
# ------------------------------------------------------------------
|
|
121
|
+
|
|
122
|
+
async def step(self, time_offset: float) -> float:
|
|
123
|
+
"""Dispatch all requests at once on the first call.
|
|
124
|
+
|
|
125
|
+
Returns:
|
|
126
|
+
0.0 while tasks are pending, -1.0 when all done.
|
|
127
|
+
"""
|
|
128
|
+
if not self._dispatched:
|
|
129
|
+
self._dispatched = True
|
|
130
|
+
for i, prompt in enumerate(self._prompts):
|
|
131
|
+
request_id = f"prefill_{i}"
|
|
132
|
+
messages = [{"role": "user", "content": prompt}]
|
|
133
|
+
self._progress_monitor.on_request_sent(request_id)
|
|
134
|
+
|
|
135
|
+
task = asyncio.create_task(
|
|
136
|
+
self._dispatch(request_id, messages),
|
|
137
|
+
)
|
|
138
|
+
self._pending_tasks.add(task)
|
|
139
|
+
task.add_done_callback(self._on_task_done)
|
|
140
|
+
|
|
141
|
+
self._progress_monitor.log_message(
|
|
142
|
+
f"Dispatched all {self._config.num_requests} requests"
|
|
143
|
+
)
|
|
144
|
+
return 0.0
|
|
145
|
+
|
|
146
|
+
# Wait for pending tasks
|
|
147
|
+
if self._pending_tasks:
|
|
148
|
+
await asyncio.wait(
|
|
149
|
+
self._pending_tasks,
|
|
150
|
+
return_when=asyncio.FIRST_COMPLETED,
|
|
151
|
+
)
|
|
152
|
+
return 0.0
|
|
153
|
+
|
|
154
|
+
return -1.0
|
|
155
|
+
|
|
156
|
+
async def _dispatch(
|
|
157
|
+
self,
|
|
158
|
+
request_id: str,
|
|
159
|
+
messages: list[dict[str, str]],
|
|
160
|
+
) -> None:
|
|
161
|
+
"""Send a single prefill request with max_tokens=1."""
|
|
162
|
+
await self._request_sender.send_request(
|
|
163
|
+
request_id,
|
|
164
|
+
messages,
|
|
165
|
+
max_tokens=1,
|
|
166
|
+
)
|
|
167
|
+
|
|
168
|
+
def _on_task_done(self, task: asyncio.Task) -> None:
|
|
169
|
+
"""Clean up completed tasks and log unexpected errors."""
|
|
170
|
+
self._pending_tasks.discard(task)
|
|
171
|
+
if not task.cancelled():
|
|
172
|
+
exc = task.exception()
|
|
173
|
+
if exc is not None:
|
|
174
|
+
self._progress_monitor.log_message(f"Dispatch task failed: {exc}")
|
|
175
|
+
|
|
176
|
+
def on_request_finished(self, request_id: str, output: str) -> None:
|
|
177
|
+
"""No-op — this workload is stateless."""
|
|
178
|
+
self._progress_monitor.log_message(f"Request {request_id} finished")
|
|
@@ -0,0 +1,310 @@
|
|
|
1
|
+
# SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
"""``lmcache describe`` — show detailed status of a running LMCache service.
|
|
3
|
+
|
|
4
|
+
Usage::
|
|
5
|
+
|
|
6
|
+
lmcache describe kvcache --url http://localhost:8000
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
# Standard
|
|
10
|
+
import argparse
|
|
11
|
+
import json
|
|
12
|
+
import sys
|
|
13
|
+
import urllib.error
|
|
14
|
+
import urllib.request
|
|
15
|
+
|
|
16
|
+
# First Party
|
|
17
|
+
from lmcache.cli.commands.base import BaseCommand
|
|
18
|
+
from lmcache.cli.metrics import Metrics
|
|
19
|
+
|
|
20
|
+
# -------------------------------------------------------------------
|
|
21
|
+
# Shared helpers
|
|
22
|
+
# -------------------------------------------------------------------
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
class DescribeError(Exception):
|
|
26
|
+
"""Raised when the describe command cannot fetch or parse status data."""
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def normalize_url(url: str) -> str:
|
|
30
|
+
"""Ensure *url* has an ``http://`` or ``https://`` scheme."""
|
|
31
|
+
if not url.startswith(("http://", "https://")):
|
|
32
|
+
url = f"http://{url}"
|
|
33
|
+
return url.rstrip("/")
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def fetch_json(url: str, timeout: int = 10) -> dict:
|
|
37
|
+
"""GET *url* and return the parsed JSON body.
|
|
38
|
+
|
|
39
|
+
Raises:
|
|
40
|
+
DescribeError: On network/HTTP errors.
|
|
41
|
+
"""
|
|
42
|
+
req = urllib.request.Request(url)
|
|
43
|
+
try:
|
|
44
|
+
with urllib.request.urlopen(req, timeout=timeout) as resp:
|
|
45
|
+
return json.loads(resp.read().decode())
|
|
46
|
+
except urllib.error.HTTPError as exc:
|
|
47
|
+
if exc.code == 503:
|
|
48
|
+
body = exc.read().decode()
|
|
49
|
+
try:
|
|
50
|
+
detail = json.loads(body).get("error", body)
|
|
51
|
+
except (json.JSONDecodeError, AttributeError):
|
|
52
|
+
detail = body
|
|
53
|
+
raise DescribeError(f"Server unhealthy: {detail}") from exc
|
|
54
|
+
raise DescribeError(f"HTTP {exc.code} from {url}: {exc.reason}") from exc
|
|
55
|
+
except urllib.error.URLError as exc:
|
|
56
|
+
raise DescribeError(f"Cannot connect to {url}: {exc.reason}") from exc
|
|
57
|
+
except OSError as exc:
|
|
58
|
+
raise DescribeError(f"Cannot connect to {url}: {exc}") from exc
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
def fmt_bytes(n: int) -> str:
|
|
62
|
+
"""Format a byte count as a human-readable string."""
|
|
63
|
+
if n >= 1024**3:
|
|
64
|
+
return f"{n / 1024**3:.2f} GB"
|
|
65
|
+
if n >= 1024**2:
|
|
66
|
+
return f"{n / 1024**2:.2f} MB"
|
|
67
|
+
if n >= 1024:
|
|
68
|
+
return f"{n / 1024:.2f} KB"
|
|
69
|
+
return f"{n} B"
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def fmt_health(is_healthy: object) -> str | None:
|
|
73
|
+
"""Format a boolean health flag as ``'OK'`` / ``'UNHEALTHY'``."""
|
|
74
|
+
if is_healthy is None:
|
|
75
|
+
return None
|
|
76
|
+
return "OK" if is_healthy else "UNHEALTHY"
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
def safe_get(data: dict, *keys, default=None): # type: ignore[type-arg]
|
|
80
|
+
"""Walk nested dicts by *keys*, returning *default* on any miss."""
|
|
81
|
+
cur: object = data
|
|
82
|
+
for key in keys:
|
|
83
|
+
if not isinstance(cur, dict):
|
|
84
|
+
return default
|
|
85
|
+
cur = cur.get(key)
|
|
86
|
+
if cur is None:
|
|
87
|
+
return default
|
|
88
|
+
return cur
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
# -------------------------------------------------------------------
|
|
92
|
+
# KVCache describer
|
|
93
|
+
# -------------------------------------------------------------------
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
class KVCacheDescriber:
|
|
97
|
+
"""Builds the ``describe kvcache`` output from a ``/api/status`` response.
|
|
98
|
+
|
|
99
|
+
Each ``add_*`` method populates one logical section. The orchestrating
|
|
100
|
+
:meth:`describe` calls them in order and emits the result. Adding a
|
|
101
|
+
new section is a one-method change — no other code needs to know
|
|
102
|
+
about it.
|
|
103
|
+
"""
|
|
104
|
+
|
|
105
|
+
def __init__(self, metrics: Metrics, data: dict, base_url: str) -> None:
|
|
106
|
+
self.metrics = metrics
|
|
107
|
+
self.data = data
|
|
108
|
+
self.base_url = base_url
|
|
109
|
+
|
|
110
|
+
def describe(self) -> None:
|
|
111
|
+
"""Run all section builders and emit."""
|
|
112
|
+
self.add_overview()
|
|
113
|
+
self.add_l1_storage()
|
|
114
|
+
self.add_models()
|
|
115
|
+
self.add_l2_adapters()
|
|
116
|
+
self.metrics.emit()
|
|
117
|
+
|
|
118
|
+
# -- sections --------------------------------------------------------
|
|
119
|
+
|
|
120
|
+
def add_overview(self) -> None:
|
|
121
|
+
"""Top-level engine overview."""
|
|
122
|
+
self.metrics.add("health", "Health", fmt_health(self.data.get("is_healthy")))
|
|
123
|
+
self.metrics.add("url", "URL", self.base_url)
|
|
124
|
+
self.metrics.add("engine_type", "Engine type", self.data.get("engine_type"))
|
|
125
|
+
self.metrics.add("chunk_size", "Chunk size", self.data.get("chunk_size"))
|
|
126
|
+
|
|
127
|
+
def add_l1_storage(self) -> None:
|
|
128
|
+
"""L1 cache capacity, usage, eviction, and object count."""
|
|
129
|
+
total_bytes = safe_get(
|
|
130
|
+
self.data, "storage_manager", "l1_manager", "memory_total_bytes"
|
|
131
|
+
)
|
|
132
|
+
if total_bytes is not None:
|
|
133
|
+
self.metrics.add(
|
|
134
|
+
"l1_capacity_gb",
|
|
135
|
+
"L1 capacity (GB)",
|
|
136
|
+
round(total_bytes / (1024**3), 2),
|
|
137
|
+
)
|
|
138
|
+
else:
|
|
139
|
+
self.metrics.add("l1_capacity_gb", "L1 capacity (GB)", None)
|
|
140
|
+
|
|
141
|
+
used_bytes = safe_get(
|
|
142
|
+
self.data, "storage_manager", "l1_manager", "memory_used_bytes"
|
|
143
|
+
)
|
|
144
|
+
usage_ratio = safe_get(
|
|
145
|
+
self.data, "storage_manager", "l1_manager", "memory_usage_ratio"
|
|
146
|
+
)
|
|
147
|
+
if used_bytes is not None and usage_ratio is not None:
|
|
148
|
+
gb = used_bytes / (1024**3)
|
|
149
|
+
pct = usage_ratio * 100
|
|
150
|
+
self.metrics.add("l1_used_gb", "L1 used (GB)", f"{gb:.2f} ({pct:.1f}%)")
|
|
151
|
+
else:
|
|
152
|
+
self.metrics.add("l1_used_gb", "L1 used (GB)", None)
|
|
153
|
+
|
|
154
|
+
self.metrics.add(
|
|
155
|
+
"eviction_policy",
|
|
156
|
+
"Eviction policy",
|
|
157
|
+
safe_get(
|
|
158
|
+
self.data,
|
|
159
|
+
"storage_manager",
|
|
160
|
+
"eviction_controller",
|
|
161
|
+
"eviction_policy",
|
|
162
|
+
),
|
|
163
|
+
)
|
|
164
|
+
self.metrics.add(
|
|
165
|
+
"cached_objects",
|
|
166
|
+
"Cached objects",
|
|
167
|
+
safe_get(self.data, "storage_manager", "l1_manager", "total_object_count"),
|
|
168
|
+
)
|
|
169
|
+
self.metrics.add(
|
|
170
|
+
"active_sessions", "Active sessions", self.data.get("active_sessions")
|
|
171
|
+
)
|
|
172
|
+
|
|
173
|
+
def add_models(self) -> None:
|
|
174
|
+
"""Per-model KV cache layout sections."""
|
|
175
|
+
gpu_meta = self.data.get("gpu_context_meta", {})
|
|
176
|
+
if not gpu_meta:
|
|
177
|
+
return
|
|
178
|
+
|
|
179
|
+
# Deduplicate by (model_name, world_size) — multiple GPU IDs
|
|
180
|
+
# may share the same model.
|
|
181
|
+
seen: dict[tuple[str, int], dict] = {}
|
|
182
|
+
for gpu_id, meta in gpu_meta.items():
|
|
183
|
+
key = (meta["model_name"], meta["world_size"])
|
|
184
|
+
if key not in seen:
|
|
185
|
+
seen[key] = {
|
|
186
|
+
"gpu_ids": [],
|
|
187
|
+
"layout": meta.get("kv_cache_layout"),
|
|
188
|
+
}
|
|
189
|
+
seen[key]["gpu_ids"].append(gpu_id)
|
|
190
|
+
|
|
191
|
+
for idx, ((model_name, world_size), info) in enumerate(seen.items()):
|
|
192
|
+
section_key = f"model_{idx}"
|
|
193
|
+
self.metrics.add_list_section("models", section_key, f"Model: {model_name}")
|
|
194
|
+
sec = self.metrics[section_key]
|
|
195
|
+
sec.add("model", "Model", model_name)
|
|
196
|
+
sec.add("world_size", "World size", world_size)
|
|
197
|
+
sec.add("gpu_ids", "GPU IDs", ", ".join(info["gpu_ids"]))
|
|
198
|
+
|
|
199
|
+
layout = info.get("layout")
|
|
200
|
+
if not layout:
|
|
201
|
+
continue
|
|
202
|
+
sec.add(
|
|
203
|
+
"attention_backend",
|
|
204
|
+
"Attention backend",
|
|
205
|
+
layout.get("attention_backend"),
|
|
206
|
+
)
|
|
207
|
+
sec.add("gpu_kv_shape", "GPU KV shape", layout.get("gpu_kv_shape"))
|
|
208
|
+
sec.add(
|
|
209
|
+
"gpu_kv_concrete_shape",
|
|
210
|
+
"GPU KV tensor shape",
|
|
211
|
+
layout.get("gpu_kv_concrete_shape"),
|
|
212
|
+
)
|
|
213
|
+
sec.add("num_layers", "Num layers", layout["num_layers"])
|
|
214
|
+
sec.add("block_size", "Block size", layout["block_size"])
|
|
215
|
+
sec.add("hidden_dim_sizes", "Hidden dim sizes", layout["hidden_dim_sizes"])
|
|
216
|
+
sec.add("dtype", "Dtype", layout["dtype"])
|
|
217
|
+
sec.add("is_mla", "MLA", layout["is_mla"])
|
|
218
|
+
sec.add("num_blocks", "Num blocks", layout["num_blocks"])
|
|
219
|
+
sec.add(
|
|
220
|
+
"cache_size_per_token",
|
|
221
|
+
"Cache size per token (bytes)",
|
|
222
|
+
layout["cache_size_per_token"],
|
|
223
|
+
)
|
|
224
|
+
|
|
225
|
+
def add_l2_adapters(self) -> None:
|
|
226
|
+
"""L2 adapter sections."""
|
|
227
|
+
l2_adapters = safe_get(self.data, "storage_manager", "l2_adapters") or []
|
|
228
|
+
for idx, adapter in enumerate(l2_adapters):
|
|
229
|
+
adapter_type = adapter.get("type", "Unknown")
|
|
230
|
+
section_key = f"l2_{idx}"
|
|
231
|
+
self.metrics.add_list_section(
|
|
232
|
+
"l2_adapters", section_key, f"L2: {adapter_type}"
|
|
233
|
+
)
|
|
234
|
+
sec = self.metrics[section_key]
|
|
235
|
+
sec.add("type", "Type", adapter_type)
|
|
236
|
+
sec.add("health", "Health", fmt_health(adapter.get("is_healthy")))
|
|
237
|
+
|
|
238
|
+
if "backend" in adapter:
|
|
239
|
+
sec.add("backend", "Backend", adapter["backend"])
|
|
240
|
+
if "base_path" in adapter:
|
|
241
|
+
sec.add("base_path", "Base path", adapter["base_path"])
|
|
242
|
+
if "stored_object_count" in adapter:
|
|
243
|
+
sec.add(
|
|
244
|
+
"stored_object_count",
|
|
245
|
+
"Stored objects",
|
|
246
|
+
adapter["stored_object_count"],
|
|
247
|
+
)
|
|
248
|
+
|
|
249
|
+
cap = adapter.get("max_capacity_bytes")
|
|
250
|
+
used = adapter.get("current_size_bytes")
|
|
251
|
+
if cap is not None and used is not None:
|
|
252
|
+
pct = used / cap * 100 if cap > 0 else 0.0
|
|
253
|
+
sec.add(
|
|
254
|
+
"used",
|
|
255
|
+
"Used",
|
|
256
|
+
f"{fmt_bytes(used)} / {fmt_bytes(cap)} ({pct:.1f}%)",
|
|
257
|
+
)
|
|
258
|
+
|
|
259
|
+
pool_size = adapter.get("pool_size")
|
|
260
|
+
pool_free = adapter.get("pool_free_slots")
|
|
261
|
+
if pool_size is not None and pool_free is not None:
|
|
262
|
+
pool_used = pool_size - pool_free
|
|
263
|
+
pct = pool_used / pool_size * 100 if pool_size > 0 else 0.0
|
|
264
|
+
sec.add(
|
|
265
|
+
"pool_used",
|
|
266
|
+
"Pool used",
|
|
267
|
+
f"{pool_used} / {pool_size} ({pct:.1f}%)",
|
|
268
|
+
)
|
|
269
|
+
|
|
270
|
+
|
|
271
|
+
# -------------------------------------------------------------------
|
|
272
|
+
# Command
|
|
273
|
+
# -------------------------------------------------------------------
|
|
274
|
+
|
|
275
|
+
|
|
276
|
+
class DescribeCommand(BaseCommand):
|
|
277
|
+
"""Show detailed status of a running LMCache service."""
|
|
278
|
+
|
|
279
|
+
def name(self) -> str:
|
|
280
|
+
return "describe"
|
|
281
|
+
|
|
282
|
+
def help(self) -> str:
|
|
283
|
+
return "Show detailed status of a running LMCache service."
|
|
284
|
+
|
|
285
|
+
def add_arguments(self, parser: argparse.ArgumentParser) -> None:
|
|
286
|
+
parser.add_argument(
|
|
287
|
+
"target",
|
|
288
|
+
choices=["kvcache"],
|
|
289
|
+
help="What to describe.",
|
|
290
|
+
)
|
|
291
|
+
parser.add_argument(
|
|
292
|
+
"--url",
|
|
293
|
+
help="LMCache HTTP server URL (default to http://localhost:8080).",
|
|
294
|
+
default="http://localhost:8080",
|
|
295
|
+
)
|
|
296
|
+
|
|
297
|
+
def execute(self, args: argparse.Namespace) -> None:
|
|
298
|
+
if args.target == "kvcache":
|
|
299
|
+
self._describe_kvcache(args)
|
|
300
|
+
|
|
301
|
+
def _describe_kvcache(self, args: argparse.Namespace) -> None:
|
|
302
|
+
base_url = normalize_url(args.url)
|
|
303
|
+
try:
|
|
304
|
+
data = fetch_json(f"{base_url}/api/status")
|
|
305
|
+
except DescribeError as exc:
|
|
306
|
+
print(str(exc), file=sys.stderr)
|
|
307
|
+
sys.exit(1)
|
|
308
|
+
|
|
309
|
+
metrics = self.create_metrics("LMCache KV Cache Service", args, width=50)
|
|
310
|
+
KVCacheDescriber(metrics, data, base_url).describe()
|
|
@@ -0,0 +1,133 @@
|
|
|
1
|
+
# SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
"""``lmcache kvcache`` — KV cache management.
|
|
3
|
+
|
|
4
|
+
Sub-commands:
|
|
5
|
+
clear Clear all cached KV data
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
# Standard
|
|
9
|
+
from typing import Any, Optional
|
|
10
|
+
import argparse
|
|
11
|
+
import json
|
|
12
|
+
import sys
|
|
13
|
+
import urllib.error
|
|
14
|
+
import urllib.request
|
|
15
|
+
|
|
16
|
+
# First Party
|
|
17
|
+
from lmcache.cli.commands.base import BaseCommand
|
|
18
|
+
from lmcache.logging import init_logger
|
|
19
|
+
|
|
20
|
+
logger = init_logger(__name__)
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def _http_request(
|
|
24
|
+
method: str, url: str, data: Optional[dict[str, Any]] = None
|
|
25
|
+
) -> dict[str, Any]:
|
|
26
|
+
"""Send an HTTP request and return the parsed JSON response.
|
|
27
|
+
|
|
28
|
+
Args:
|
|
29
|
+
method: HTTP method (GET, POST, PUT, DELETE).
|
|
30
|
+
url: Full URL to request.
|
|
31
|
+
data: Optional JSON body to send.
|
|
32
|
+
|
|
33
|
+
Returns:
|
|
34
|
+
Parsed JSON response as a dict.
|
|
35
|
+
|
|
36
|
+
Raises:
|
|
37
|
+
SystemExit: On connection error or non-2xx HTTP response.
|
|
38
|
+
"""
|
|
39
|
+
body = None
|
|
40
|
+
headers: dict[str, str] = {}
|
|
41
|
+
if data is not None:
|
|
42
|
+
body = json.dumps(data).encode()
|
|
43
|
+
headers["Content-Type"] = "application/json"
|
|
44
|
+
|
|
45
|
+
req = urllib.request.Request(url, data=body, headers=headers, method=method)
|
|
46
|
+
try:
|
|
47
|
+
with urllib.request.urlopen(req, timeout=10) as resp:
|
|
48
|
+
return json.loads(resp.read().decode())
|
|
49
|
+
except urllib.error.HTTPError as e:
|
|
50
|
+
try:
|
|
51
|
+
error_body = json.loads(e.read().decode())
|
|
52
|
+
msg = error_body.get("message") or error_body.get("error") or str(e)
|
|
53
|
+
except (json.JSONDecodeError, ValueError, OSError):
|
|
54
|
+
msg = str(e)
|
|
55
|
+
logger.error("Server error: %s", msg)
|
|
56
|
+
sys.exit(1)
|
|
57
|
+
except urllib.error.URLError as e:
|
|
58
|
+
logger.error("Cannot reach %s — is the server running? (%s)", url, e.reason)
|
|
59
|
+
sys.exit(1)
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
class KVCacheCommand(BaseCommand):
|
|
63
|
+
"""Manage KV cache state.
|
|
64
|
+
|
|
65
|
+
This command provides sub-commands for managing KV cache data on the
|
|
66
|
+
MP HTTP server. Currently supports clearing all L1 cache. Future
|
|
67
|
+
sub-commands (pin, compress, info) are defined in the design doc.
|
|
68
|
+
"""
|
|
69
|
+
|
|
70
|
+
def name(self) -> str:
|
|
71
|
+
"""Return the subcommand name.
|
|
72
|
+
|
|
73
|
+
Returns:
|
|
74
|
+
The string ``"kvcache"``.
|
|
75
|
+
"""
|
|
76
|
+
return "kvcache"
|
|
77
|
+
|
|
78
|
+
def help(self) -> str:
|
|
79
|
+
"""Return short help text.
|
|
80
|
+
|
|
81
|
+
Returns:
|
|
82
|
+
Help string shown by ``lmcache -h``.
|
|
83
|
+
"""
|
|
84
|
+
return "Manage KV cache state."
|
|
85
|
+
|
|
86
|
+
def add_arguments(self, parser: argparse.ArgumentParser) -> None:
|
|
87
|
+
"""Add kvcache-specific sub-commands and their arguments.
|
|
88
|
+
|
|
89
|
+
Args:
|
|
90
|
+
parser: The ``ArgumentParser`` for this subcommand.
|
|
91
|
+
"""
|
|
92
|
+
sub = parser.add_subparsers(dest="action")
|
|
93
|
+
|
|
94
|
+
# clear
|
|
95
|
+
p_clear = sub.add_parser("clear", help="Clear all cached KV data in L1 (CPU).")
|
|
96
|
+
p_clear.add_argument(
|
|
97
|
+
"--url",
|
|
98
|
+
type=str,
|
|
99
|
+
required=True,
|
|
100
|
+
help="Target MP HTTP endpoint (e.g. http://localhost:8000).",
|
|
101
|
+
)
|
|
102
|
+
|
|
103
|
+
def execute(self, args: argparse.Namespace) -> None:
|
|
104
|
+
"""Dispatch to the appropriate sub-command handler.
|
|
105
|
+
|
|
106
|
+
Args:
|
|
107
|
+
args: Parsed CLI arguments. Must contain ``action`` indicating
|
|
108
|
+
which sub-command was invoked.
|
|
109
|
+
"""
|
|
110
|
+
action = getattr(args, "action", None)
|
|
111
|
+
if action is None:
|
|
112
|
+
logger.error("No sub-command specified. Run: lmcache kvcache -h")
|
|
113
|
+
sys.exit(1)
|
|
114
|
+
|
|
115
|
+
dispatch = {
|
|
116
|
+
"clear": self._clear,
|
|
117
|
+
}
|
|
118
|
+
dispatch[action](args)
|
|
119
|
+
|
|
120
|
+
def _clear(self, args: argparse.Namespace) -> None:
|
|
121
|
+
"""Clear all cached KV data via the MP HTTP server."""
|
|
122
|
+
url = args.url.rstrip("/")
|
|
123
|
+
|
|
124
|
+
# MP HTTP server endpoint: POST /api/clear-cache
|
|
125
|
+
_http_request("POST", f"{url}/api/clear-cache")
|
|
126
|
+
|
|
127
|
+
quiet = getattr(args, "quiet", False)
|
|
128
|
+
if quiet:
|
|
129
|
+
return
|
|
130
|
+
|
|
131
|
+
metrics = self.create_metrics("KV Cache Clear", args)
|
|
132
|
+
metrics.add("status", "Status", "OK")
|
|
133
|
+
metrics.emit()
|
|
@@ -0,0 +1,75 @@
|
|
|
1
|
+
# SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
"""``lmcache mock`` — example command that demonstrates the CLI framework.
|
|
3
|
+
|
|
4
|
+
This command does not connect to any server. It generates fake metrics
|
|
5
|
+
to exercise argument parsing, the ``Metrics`` logger, and both terminal
|
|
6
|
+
and JSON output paths.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
# Standard
|
|
10
|
+
import argparse
|
|
11
|
+
|
|
12
|
+
# First Party
|
|
13
|
+
from lmcache.cli.commands.base import BaseCommand
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
class MockCommand(BaseCommand):
|
|
17
|
+
"""Mock command used as a reference implementation for new commands."""
|
|
18
|
+
|
|
19
|
+
def name(self) -> str:
|
|
20
|
+
"""Return the subcommand name.
|
|
21
|
+
|
|
22
|
+
Returns:
|
|
23
|
+
The string ``"mock"``.
|
|
24
|
+
"""
|
|
25
|
+
return "mock"
|
|
26
|
+
|
|
27
|
+
def help(self) -> str:
|
|
28
|
+
"""Return short help text.
|
|
29
|
+
|
|
30
|
+
Returns:
|
|
31
|
+
Help string shown by ``lmcache -h``.
|
|
32
|
+
"""
|
|
33
|
+
return "Run a mock command (example/test)."
|
|
34
|
+
|
|
35
|
+
def add_arguments(self, parser: argparse.ArgumentParser) -> None:
|
|
36
|
+
"""Add mock-specific arguments.
|
|
37
|
+
|
|
38
|
+
Args:
|
|
39
|
+
parser: The ``ArgumentParser`` for this subcommand.
|
|
40
|
+
"""
|
|
41
|
+
parser.add_argument(
|
|
42
|
+
"--name",
|
|
43
|
+
type=str,
|
|
44
|
+
default="default",
|
|
45
|
+
help="Name tag for this mock run.",
|
|
46
|
+
)
|
|
47
|
+
parser.add_argument(
|
|
48
|
+
"--num-items",
|
|
49
|
+
type=int,
|
|
50
|
+
default=10,
|
|
51
|
+
help="Number of fake items to process.",
|
|
52
|
+
)
|
|
53
|
+
|
|
54
|
+
def execute(self, args: argparse.Namespace) -> None:
|
|
55
|
+
"""Execute the mock command.
|
|
56
|
+
|
|
57
|
+
Args:
|
|
58
|
+
args: Parsed CLI arguments containing ``name``, ``num_items``,
|
|
59
|
+
and optionally ``output``.
|
|
60
|
+
"""
|
|
61
|
+
metrics = self.create_metrics("Mock Result", args, width=40)
|
|
62
|
+
|
|
63
|
+
metrics.add_section("input", "Input Parameters")
|
|
64
|
+
metrics["input"].add("name", "Name", args.name)
|
|
65
|
+
metrics["input"].add("num_items", "Num items", args.num_items)
|
|
66
|
+
|
|
67
|
+
metrics.add_section("mock", "Mock Metrics")
|
|
68
|
+
metrics["mock"].add("items_processed", "Items processed", 42)
|
|
69
|
+
metrics["mock"].add("total_time_ms", "Total time (ms)", 12.34)
|
|
70
|
+
metrics["mock"].add("throughput", "Throughput (items/s)", 3403.73)
|
|
71
|
+
|
|
72
|
+
metrics.add_section("validation", "Validation")
|
|
73
|
+
metrics["validation"].add("status", "Status", "OK")
|
|
74
|
+
|
|
75
|
+
metrics.emit()
|