lmcache-cli 0.4.5.dev0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- lmcache/__init__.py +84 -0
- lmcache/_version.py +24 -0
- lmcache/cli/__init__.py +1 -0
- lmcache/cli/commands/__init__.py +34 -0
- lmcache/cli/commands/base.py +157 -0
- lmcache/cli/commands/bench/__init__.py +557 -0
- lmcache/cli/commands/bench/engine_bench/__init__.py +1 -0
- lmcache/cli/commands/bench/engine_bench/config.py +245 -0
- lmcache/cli/commands/bench/engine_bench/interactive/__init__.py +274 -0
- lmcache/cli/commands/bench/engine_bench/interactive/config.json +10 -0
- lmcache/cli/commands/bench/engine_bench/interactive/schema.py +352 -0
- lmcache/cli/commands/bench/engine_bench/interactive/state.py +327 -0
- lmcache/cli/commands/bench/engine_bench/interactive/terminal.py +291 -0
- lmcache/cli/commands/bench/engine_bench/progress.py +145 -0
- lmcache/cli/commands/bench/engine_bench/request_sender.py +232 -0
- lmcache/cli/commands/bench/engine_bench/stats.py +275 -0
- lmcache/cli/commands/bench/engine_bench/workloads/__init__.py +153 -0
- lmcache/cli/commands/bench/engine_bench/workloads/base.py +122 -0
- lmcache/cli/commands/bench/engine_bench/workloads/long_doc_permutator.py +435 -0
- lmcache/cli/commands/bench/engine_bench/workloads/long_doc_qa.py +281 -0
- lmcache/cli/commands/bench/engine_bench/workloads/multi_round_chat.py +337 -0
- lmcache/cli/commands/bench/engine_bench/workloads/random_prefill.py +178 -0
- lmcache/cli/commands/describe.py +310 -0
- lmcache/cli/commands/kvcache.py +133 -0
- lmcache/cli/commands/mock.py +75 -0
- lmcache/cli/commands/ping.py +113 -0
- lmcache/cli/commands/query/__init__.py +155 -0
- lmcache/cli/commands/query/prompt.py +134 -0
- lmcache/cli/commands/query/request.py +357 -0
- lmcache/cli/commands/server.py +99 -0
- lmcache/cli/commands/tool/__init__.py +63 -0
- lmcache/cli/commands/tool/cache_simulator.py +113 -0
- lmcache/cli/commands/trace/__init__.py +505 -0
- lmcache/cli/commands/trace/dispatch.py +249 -0
- lmcache/cli/commands/trace/driver.py +372 -0
- lmcache/cli/commands/trace/stats.py +289 -0
- lmcache/cli/documents/lmcache.txt +11 -0
- lmcache/cli/main.py +42 -0
- lmcache/cli/metrics/__init__.py +29 -0
- lmcache/cli/metrics/formatter.py +171 -0
- lmcache/cli/metrics/handler.py +94 -0
- lmcache/cli/metrics/metrics.py +161 -0
- lmcache/cli/metrics/section.py +77 -0
- lmcache/connections.py +173 -0
- lmcache/integration/__init__.py +2 -0
- lmcache/integration/base_service_factory.py +165 -0
- lmcache/integration/request_telemetry/__init__.py +1 -0
- lmcache/integration/request_telemetry/base.py +51 -0
- lmcache/integration/request_telemetry/factory.py +113 -0
- lmcache/integration/request_telemetry/fastapi.py +109 -0
- lmcache/integration/request_telemetry/noop.py +35 -0
- lmcache/integration/sglang/__init__.py +2 -0
- lmcache/integration/sglang/sglang_adapter.py +326 -0
- lmcache/integration/sglang/utils.py +39 -0
- lmcache/integration/vllm/__init__.py +1 -0
- lmcache/integration/vllm/lmcache_connector_v1.py +213 -0
- lmcache/integration/vllm/lmcache_connector_v1_085.py +150 -0
- lmcache/integration/vllm/lmcache_mp_connector_0180.py +1072 -0
- lmcache/integration/vllm/tests/test_mm_hash_utils.py +112 -0
- lmcache/integration/vllm/utils.py +433 -0
- lmcache/integration/vllm/vllm_multi_process_adapter.py +1090 -0
- lmcache/integration/vllm/vllm_service_factory.py +339 -0
- lmcache/integration/vllm/vllm_v1_adapter.py +1713 -0
- lmcache/logging.py +107 -0
- lmcache/native_storage_ops.pyi +230 -0
- lmcache/non_cuda_equivalents.py +1424 -0
- lmcache/observability.py +1958 -0
- lmcache/storage_backend/serde/__init__.py +1 -0
- lmcache/storage_backend/serde/cachegen_basics.py +210 -0
- lmcache/storage_backend/serde/cachegen_decoder.py +207 -0
- lmcache/storage_backend/serde/cachegen_encoder.py +394 -0
- lmcache/storage_backend/serde/serde.py +75 -0
- lmcache/tools/__init__.py +1 -0
- lmcache/tools/cache_simulator/README.md +392 -0
- lmcache/tools/cache_simulator/__init__.py +1 -0
- lmcache/tools/cache_simulator/docs/simulate_example.png +0 -0
- lmcache/tools/cache_simulator/docs/sweep_example.png +0 -0
- lmcache/tools/cache_simulator/gen_bench_dataset.py +360 -0
- lmcache/tools/cache_simulator/lru_cache.py +124 -0
- lmcache/tools/cache_simulator/plot_hit_rate.py +231 -0
- lmcache/tools/cache_simulator/simulator.py +795 -0
- lmcache/tools/controller_benchmark/README.md +161 -0
- lmcache/tools/controller_benchmark/__init__.py +1 -0
- lmcache/tools/controller_benchmark/__main__.py +331 -0
- lmcache/tools/controller_benchmark/benchmark.py +660 -0
- lmcache/tools/controller_benchmark/config.py +44 -0
- lmcache/tools/controller_benchmark/constants.py +10 -0
- lmcache/tools/controller_benchmark/handlers/__init__.py +46 -0
- lmcache/tools/controller_benchmark/handlers/admit.py +52 -0
- lmcache/tools/controller_benchmark/handlers/base.py +47 -0
- lmcache/tools/controller_benchmark/handlers/deregister.py +49 -0
- lmcache/tools/controller_benchmark/handlers/evict.py +52 -0
- lmcache/tools/controller_benchmark/handlers/heartbeat.py +56 -0
- lmcache/tools/controller_benchmark/handlers/p2p_lookup.py +47 -0
- lmcache/tools/controller_benchmark/handlers/register.py +56 -0
- lmcache/tools/mp_status_viewer/__init__.py +1 -0
- lmcache/tools/mp_status_viewer/__main__.py +95 -0
- lmcache/usage_context.py +417 -0
- lmcache/utils.py +665 -0
- lmcache/v1/__init__.py +2 -0
- lmcache/v1/api_server/__init__.py +2 -0
- lmcache/v1/api_server/__main__.py +537 -0
- lmcache/v1/basic_check.py +112 -0
- lmcache/v1/cache_controller/__init__.py +9 -0
- lmcache/v1/cache_controller/commands/__init__.py +15 -0
- lmcache/v1/cache_controller/commands/base.py +35 -0
- lmcache/v1/cache_controller/commands/full_sync.py +49 -0
- lmcache/v1/cache_controller/config.py +176 -0
- lmcache/v1/cache_controller/controller_manager.py +535 -0
- lmcache/v1/cache_controller/controllers/__init__.py +11 -0
- lmcache/v1/cache_controller/controllers/full_sync_tracker.py +473 -0
- lmcache/v1/cache_controller/controllers/kv_controller.py +439 -0
- lmcache/v1/cache_controller/controllers/registration_controller.py +282 -0
- lmcache/v1/cache_controller/executor.py +463 -0
- lmcache/v1/cache_controller/frontend/static/css/style.css +201 -0
- lmcache/v1/cache_controller/frontend/static/img/logo.png +0 -0
- lmcache/v1/cache_controller/frontend/static/index.html +234 -0
- lmcache/v1/cache_controller/frontend/static/js/controller_app.js +660 -0
- lmcache/v1/cache_controller/full_sync_sender.py +475 -0
- lmcache/v1/cache_controller/locks.py +149 -0
- lmcache/v1/cache_controller/message.py +828 -0
- lmcache/v1/cache_controller/observability.py +208 -0
- lmcache/v1/cache_controller/utils.py +679 -0
- lmcache/v1/cache_controller/worker.py +665 -0
- lmcache/v1/cache_engine.py +2058 -0
- lmcache/v1/cache_interface.py +19 -0
- lmcache/v1/check/__init__.py +74 -0
- lmcache/v1/check/check_mode_gen.py +86 -0
- lmcache/v1/check/check_mode_test_l2_adapter.py +284 -0
- lmcache/v1/check/check_mode_test_remote.py +155 -0
- lmcache/v1/check/check_mode_test_storage_manager.py +142 -0
- lmcache/v1/check/utils.py +571 -0
- lmcache/v1/compute/__init__.py +2 -0
- lmcache/v1/compute/attention/__init__.py +0 -0
- lmcache/v1/compute/attention/abstract.py +39 -0
- lmcache/v1/compute/attention/flash_attn.py +129 -0
- lmcache/v1/compute/attention/flash_infer_sparse.py +284 -0
- lmcache/v1/compute/attention/metadata.py +85 -0
- lmcache/v1/compute/attention/utils.py +14 -0
- lmcache/v1/compute/blend/__init__.py +7 -0
- lmcache/v1/compute/blend/blender.py +168 -0
- lmcache/v1/compute/blend/metadata.py +34 -0
- lmcache/v1/compute/blend/utils.py +63 -0
- lmcache/v1/compute/models/__init__.py +0 -0
- lmcache/v1/compute/models/base.py +141 -0
- lmcache/v1/compute/models/llama.py +9 -0
- lmcache/v1/compute/models/qwen3.py +24 -0
- lmcache/v1/compute/models/utils.py +68 -0
- lmcache/v1/compute/positional_encoding.py +199 -0
- lmcache/v1/config.py +848 -0
- lmcache/v1/config_base.py +848 -0
- lmcache/v1/distributed/api.py +248 -0
- lmcache/v1/distributed/config.py +321 -0
- lmcache/v1/distributed/error.py +64 -0
- lmcache/v1/distributed/eviction.py +192 -0
- lmcache/v1/distributed/eviction_policy/__init__.py +21 -0
- lmcache/v1/distributed/eviction_policy/factory.py +27 -0
- lmcache/v1/distributed/eviction_policy/lru.py +244 -0
- lmcache/v1/distributed/eviction_policy/noop.py +50 -0
- lmcache/v1/distributed/internal_api.py +170 -0
- lmcache/v1/distributed/l1_manager.py +835 -0
- lmcache/v1/distributed/l2_adapters/__init__.py +67 -0
- lmcache/v1/distributed/l2_adapters/base.py +360 -0
- lmcache/v1/distributed/l2_adapters/config.py +385 -0
- lmcache/v1/distributed/l2_adapters/factory.py +205 -0
- lmcache/v1/distributed/l2_adapters/fs_l2_adapter.py +747 -0
- lmcache/v1/distributed/l2_adapters/fs_native_l2_adapter.py +167 -0
- lmcache/v1/distributed/l2_adapters/mock_l2_adapter.py +516 -0
- lmcache/v1/distributed/l2_adapters/mooncake_store_l2_adapter.py +135 -0
- lmcache/v1/distributed/l2_adapters/native_connector_l2_adapter.py +468 -0
- lmcache/v1/distributed/l2_adapters/native_plugin_l2_adapter.py +199 -0
- lmcache/v1/distributed/l2_adapters/nixl_store_dynamic_l2_adapter.py +831 -0
- lmcache/v1/distributed/l2_adapters/nixl_store_l2_adapter.py +983 -0
- lmcache/v1/distributed/l2_adapters/plugin_l2_adapter.py +210 -0
- lmcache/v1/distributed/l2_adapters/resp_l2_adapter.py +176 -0
- lmcache/v1/distributed/memory_manager.py +179 -0
- lmcache/v1/distributed/storage_controller.py +39 -0
- lmcache/v1/distributed/storage_controllers/__init__.py +43 -0
- lmcache/v1/distributed/storage_controllers/eviction_controller.py +242 -0
- lmcache/v1/distributed/storage_controllers/prefetch_controller.py +830 -0
- lmcache/v1/distributed/storage_controllers/prefetch_policy.py +193 -0
- lmcache/v1/distributed/storage_controllers/store_controller.py +452 -0
- lmcache/v1/distributed/storage_controllers/store_policy.py +213 -0
- lmcache/v1/distributed/storage_manager.py +532 -0
- lmcache/v1/event_manager.py +145 -0
- lmcache/v1/exceptions/__init__.py +16 -0
- lmcache/v1/gpu_connector/__init__.py +126 -0
- lmcache/v1/gpu_connector/gpu_connectors.py +1906 -0
- lmcache/v1/gpu_connector/gpu_ops.py +85 -0
- lmcache/v1/gpu_connector/hpu_connector.py +326 -0
- lmcache/v1/gpu_connector/mock_gpu_connector.py +67 -0
- lmcache/v1/gpu_connector/utils.py +890 -0
- lmcache/v1/gpu_connector/xpu_connectors.py +916 -0
- lmcache/v1/health_monitor/__init__.py +1 -0
- lmcache/v1/health_monitor/base.py +587 -0
- lmcache/v1/health_monitor/checks/__init__.py +1 -0
- lmcache/v1/health_monitor/checks/remote_backend_check.py +304 -0
- lmcache/v1/health_monitor/constants.py +36 -0
- lmcache/v1/internal_api_server/__init__.py +0 -0
- lmcache/v1/internal_api_server/api_registry.py +59 -0
- lmcache/v1/internal_api_server/api_server.py +120 -0
- lmcache/v1/internal_api_server/common/__init__.py +1 -0
- lmcache/v1/internal_api_server/common/env_api.py +22 -0
- lmcache/v1/internal_api_server/common/loglevel_api.py +57 -0
- lmcache/v1/internal_api_server/common/metrics_api.py +29 -0
- lmcache/v1/internal_api_server/common/periodic_thread_api.py +138 -0
- lmcache/v1/internal_api_server/common/run_script_api.py +73 -0
- lmcache/v1/internal_api_server/common/thread_api.py +63 -0
- lmcache/v1/internal_api_server/controller/__init__.py +1 -0
- lmcache/v1/internal_api_server/controller/key_stats_api.py +81 -0
- lmcache/v1/internal_api_server/controller/worker_info_api.py +136 -0
- lmcache/v1/internal_api_server/utils.py +43 -0
- lmcache/v1/internal_api_server/vllm/__init__.py +1 -0
- lmcache/v1/internal_api_server/vllm/backend_api.py +221 -0
- lmcache/v1/internal_api_server/vllm/bypass_api.py +204 -0
- lmcache/v1/internal_api_server/vllm/cache_api.py +895 -0
- lmcache/v1/internal_api_server/vllm/chunk_statistics_api.py +141 -0
- lmcache/v1/internal_api_server/vllm/conf_api.py +147 -0
- lmcache/v1/internal_api_server/vllm/freeze_api.py +172 -0
- lmcache/v1/internal_api_server/vllm/hot_cache_api.py +184 -0
- lmcache/v1/internal_api_server/vllm/inference_api.py +65 -0
- lmcache/v1/internal_api_server/vllm/load_fs_chunks_api.py +320 -0
- lmcache/v1/internal_api_server/vllm/lookup_api.py +145 -0
- lmcache/v1/internal_api_server/vllm/version_api.py +25 -0
- lmcache/v1/kv_layer_groups.py +267 -0
- lmcache/v1/lazy_memory_allocator.py +284 -0
- lmcache/v1/lookup_client/__init__.py +25 -0
- lmcache/v1/lookup_client/abstract_client.py +77 -0
- lmcache/v1/lookup_client/async_lookup_message.py +50 -0
- lmcache/v1/lookup_client/chunk_statistics_lookup_client.py +200 -0
- lmcache/v1/lookup_client/factory.py +251 -0
- lmcache/v1/lookup_client/hit_limit_lookup_client.py +86 -0
- lmcache/v1/lookup_client/lmcache_async_lookup_client.py +407 -0
- lmcache/v1/lookup_client/lmcache_lookup_client.py +285 -0
- lmcache/v1/lookup_client/lmcache_lookup_client_bypass.py +99 -0
- lmcache/v1/lookup_client/mooncake_lookup_client.py +87 -0
- lmcache/v1/lookup_client/record_strategies/__init__.py +77 -0
- lmcache/v1/lookup_client/record_strategies/base.py +327 -0
- lmcache/v1/lookup_client/record_strategies/file_hash.py +130 -0
- lmcache/v1/lookup_client/record_strategies/memory_bloom_filter.py +81 -0
- lmcache/v1/manager.py +539 -0
- lmcache/v1/memory_management.py +2619 -0
- lmcache/v1/metadata.py +114 -0
- lmcache/v1/mp_observability/AGENTS.override.md +21 -0
- lmcache/v1/mp_observability/README.md +204 -0
- lmcache/v1/mp_observability/config.py +340 -0
- lmcache/v1/mp_observability/event.py +100 -0
- lmcache/v1/mp_observability/event_bus.py +313 -0
- lmcache/v1/mp_observability/otel_init.py +145 -0
- lmcache/v1/mp_observability/subscribers/__init__.py +28 -0
- lmcache/v1/mp_observability/subscribers/logging/__init__.py +19 -0
- lmcache/v1/mp_observability/subscribers/logging/l1.py +56 -0
- lmcache/v1/mp_observability/subscribers/logging/l2.py +73 -0
- lmcache/v1/mp_observability/subscribers/logging/lookup_hash.py +209 -0
- lmcache/v1/mp_observability/subscribers/logging/mp_server.py +90 -0
- lmcache/v1/mp_observability/subscribers/logging/sm.py +59 -0
- lmcache/v1/mp_observability/subscribers/metrics/__init__.py +20 -0
- lmcache/v1/mp_observability/subscribers/metrics/l0_lifecycle.py +290 -0
- lmcache/v1/mp_observability/subscribers/metrics/l1.py +55 -0
- lmcache/v1/mp_observability/subscribers/metrics/l1_lifecycle.py +166 -0
- lmcache/v1/mp_observability/subscribers/metrics/l2.py +121 -0
- lmcache/v1/mp_observability/subscribers/metrics/sm.py +69 -0
- lmcache/v1/mp_observability/subscribers/tracing/__init__.py +12 -0
- lmcache/v1/mp_observability/subscribers/tracing/mp_server.py +333 -0
- lmcache/v1/mp_observability/subscribers/tracing/span_registry.py +148 -0
- lmcache/v1/mp_observability/trace/__init__.py +50 -0
- lmcache/v1/mp_observability/trace/codecs.py +255 -0
- lmcache/v1/mp_observability/trace/decorator.py +147 -0
- lmcache/v1/mp_observability/trace/format.py +132 -0
- lmcache/v1/mp_observability/trace/lifecycle.py +83 -0
- lmcache/v1/mp_observability/trace/reader.py +167 -0
- lmcache/v1/mp_observability/trace/recorder.py +300 -0
- lmcache/v1/multiprocess/__init__.py +0 -0
- lmcache/v1/multiprocess/affinity_pool.py +102 -0
- lmcache/v1/multiprocess/blend_server_v2.py +891 -0
- lmcache/v1/multiprocess/config.py +253 -0
- lmcache/v1/multiprocess/custom_types.py +281 -0
- lmcache/v1/multiprocess/futures.py +194 -0
- lmcache/v1/multiprocess/gpu_context.py +511 -0
- lmcache/v1/multiprocess/http_server.py +235 -0
- lmcache/v1/multiprocess/mp_runtime_plugin_launcher.py +130 -0
- lmcache/v1/multiprocess/mq.py +732 -0
- lmcache/v1/multiprocess/protocol.py +86 -0
- lmcache/v1/multiprocess/protocols/README.md +213 -0
- lmcache/v1/multiprocess/protocols/__init__.py +127 -0
- lmcache/v1/multiprocess/protocols/base.py +89 -0
- lmcache/v1/multiprocess/protocols/blend.py +109 -0
- lmcache/v1/multiprocess/protocols/blend_v2.py +57 -0
- lmcache/v1/multiprocess/protocols/controller.py +53 -0
- lmcache/v1/multiprocess/protocols/debug.py +34 -0
- lmcache/v1/multiprocess/protocols/engine.py +146 -0
- lmcache/v1/multiprocess/protocols/observability.py +39 -0
- lmcache/v1/multiprocess/server.py +1134 -0
- lmcache/v1/multiprocess/session.py +190 -0
- lmcache/v1/multiprocess/token_hasher.py +441 -0
- lmcache/v1/offload_server/__init__.py +17 -0
- lmcache/v1/offload_server/abstract_server.py +37 -0
- lmcache/v1/offload_server/message.py +30 -0
- lmcache/v1/offload_server/zmq_server.py +122 -0
- lmcache/v1/periodic_thread.py +579 -0
- lmcache/v1/pin_monitor.py +246 -0
- lmcache/v1/plugin/__init__.py +0 -0
- lmcache/v1/plugin/runtime_plugin_launcher.py +211 -0
- lmcache/v1/protocol.py +317 -0
- lmcache/v1/rpc/__init__.py +17 -0
- lmcache/v1/rpc/transport.py +105 -0
- lmcache/v1/rpc/zmq_transport.py +213 -0
- lmcache/v1/rpc_utils.py +165 -0
- lmcache/v1/server/__init__.py +2 -0
- lmcache/v1/server/__main__.py +170 -0
- lmcache/v1/server/storage_backend/__init__.py +21 -0
- lmcache/v1/server/storage_backend/abstract_backend.py +80 -0
- lmcache/v1/server/storage_backend/local_backend.py +75 -0
- lmcache/v1/server/utils.py +21 -0
- lmcache/v1/standalone/__init__.py +1 -0
- lmcache/v1/standalone/__main__.py +583 -0
- lmcache/v1/standalone/manager.py +80 -0
- lmcache/v1/standalone/standalone_service_factory.py +86 -0
- lmcache/v1/storage_backend/__init__.py +313 -0
- lmcache/v1/storage_backend/abstract_backend.py +445 -0
- lmcache/v1/storage_backend/audit_backend.py +233 -0
- lmcache/v1/storage_backend/batched_message_sender.py +222 -0
- lmcache/v1/storage_backend/cache_policy/__init__.py +45 -0
- lmcache/v1/storage_backend/cache_policy/base_policy.py +87 -0
- lmcache/v1/storage_backend/cache_policy/fifo.py +58 -0
- lmcache/v1/storage_backend/cache_policy/lfu.py +105 -0
- lmcache/v1/storage_backend/cache_policy/lru.py +81 -0
- lmcache/v1/storage_backend/cache_policy/mru.py +61 -0
- lmcache/v1/storage_backend/connector/__init__.py +443 -0
- lmcache/v1/storage_backend/connector/audit_adapter.py +77 -0
- lmcache/v1/storage_backend/connector/audit_connector.py +320 -0
- lmcache/v1/storage_backend/connector/base_connector.py +379 -0
- lmcache/v1/storage_backend/connector/blackhole_adapter.py +21 -0
- lmcache/v1/storage_backend/connector/blackhole_connector.py +37 -0
- lmcache/v1/storage_backend/connector/eic_adapter.py +31 -0
- lmcache/v1/storage_backend/connector/eic_connector.py +757 -0
- lmcache/v1/storage_backend/connector/external_adapter.py +79 -0
- lmcache/v1/storage_backend/connector/fs_adapter.py +51 -0
- lmcache/v1/storage_backend/connector/fs_connector.py +403 -0
- lmcache/v1/storage_backend/connector/infinistore_adapter.py +56 -0
- lmcache/v1/storage_backend/connector/infinistore_connector.py +177 -0
- lmcache/v1/storage_backend/connector/instrumented_connector.py +219 -0
- lmcache/v1/storage_backend/connector/lm_adapter.py +31 -0
- lmcache/v1/storage_backend/connector/lm_connector.py +176 -0
- lmcache/v1/storage_backend/connector/mock_adapter.py +57 -0
- lmcache/v1/storage_backend/connector/mock_connector.py +349 -0
- lmcache/v1/storage_backend/connector/mooncakestore_adapter.py +43 -0
- lmcache/v1/storage_backend/connector/mooncakestore_connector.py +614 -0
- lmcache/v1/storage_backend/connector/redis_adapter.py +181 -0
- lmcache/v1/storage_backend/connector/redis_connector.py +828 -0
- lmcache/v1/storage_backend/connector/s3_adapter.py +59 -0
- lmcache/v1/storage_backend/connector/s3_connector.py +699 -0
- lmcache/v1/storage_backend/connector/sagemaker_hyperpod_adapter.py +233 -0
- lmcache/v1/storage_backend/connector/sagemaker_hyperpod_connector.py +987 -0
- lmcache/v1/storage_backend/connector/valkey_adapter.py +114 -0
- lmcache/v1/storage_backend/connector/valkey_connector.py +627 -0
- lmcache/v1/storage_backend/gds_backend.py +1199 -0
- lmcache/v1/storage_backend/job_executor/__init__.py +0 -0
- lmcache/v1/storage_backend/job_executor/base_executor.py +34 -0
- lmcache/v1/storage_backend/job_executor/pq_executor.py +235 -0
- lmcache/v1/storage_backend/local_cpu_backend.py +810 -0
- lmcache/v1/storage_backend/local_disk_backend.py +656 -0
- lmcache/v1/storage_backend/maru_backend.py +734 -0
- lmcache/v1/storage_backend/naive_serde/__init__.py +50 -0
- lmcache/v1/storage_backend/naive_serde/cachegen_basics.py +133 -0
- lmcache/v1/storage_backend/naive_serde/cachegen_decoder.py +135 -0
- lmcache/v1/storage_backend/naive_serde/cachegen_encoder.py +83 -0
- lmcache/v1/storage_backend/naive_serde/kivi_serde.py +22 -0
- lmcache/v1/storage_backend/naive_serde/naive_serde.py +18 -0
- lmcache/v1/storage_backend/naive_serde/serde.py +37 -0
- lmcache/v1/storage_backend/native_clients/connector_client_base.py +165 -0
- lmcache/v1/storage_backend/native_clients/resp_client.py +35 -0
- lmcache/v1/storage_backend/nixl_storage_backend.py +1400 -0
- lmcache/v1/storage_backend/p2p_backend.py +788 -0
- lmcache/v1/storage_backend/path_sharder.py +117 -0
- lmcache/v1/storage_backend/pd_backend.py +646 -0
- lmcache/v1/storage_backend/plugins/dax_backend.py +1443 -0
- lmcache/v1/storage_backend/plugins/rust_raw_block_backend.py +1361 -0
- lmcache/v1/storage_backend/remote_backend.py +624 -0
- lmcache/v1/storage_backend/resp_client.py +227 -0
- lmcache/v1/storage_backend/storage_backend_listener.py +19 -0
- lmcache/v1/storage_backend/storage_manager.py +1352 -0
- lmcache/v1/system_detection.py +110 -0
- lmcache/v1/token_database.py +551 -0
- lmcache/v1/transfer_channel/__init__.py +83 -0
- lmcache/v1/transfer_channel/abstract.py +285 -0
- lmcache/v1/transfer_channel/mock_memory_channel.py +156 -0
- lmcache/v1/transfer_channel/nixl_channel.py +639 -0
- lmcache/v1/transfer_channel/py_socket_channel.py +260 -0
- lmcache/v1/transfer_channel/transfer_utils.py +63 -0
- lmcache/v1/utils/__init__.py +1 -0
- lmcache/v1/utils/bloom_filter.py +109 -0
- lmcache/v1/utils/cache_utils.py +125 -0
- lmcache_cli-0.4.5.dev0.dist-info/METADATA +185 -0
- lmcache_cli-0.4.5.dev0.dist-info/RECORD +399 -0
- lmcache_cli-0.4.5.dev0.dist-info/WHEEL +5 -0
- lmcache_cli-0.4.5.dev0.dist-info/entry_points.txt +2 -0
- lmcache_cli-0.4.5.dev0.dist-info/licenses/LICENSE +201 -0
- lmcache_cli-0.4.5.dev0.dist-info/top_level.txt +1 -0
lmcache/v1/metadata.py
ADDED
|
@@ -0,0 +1,114 @@
|
|
|
1
|
+
# SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
# Standard
|
|
3
|
+
from dataclasses import dataclass, field
|
|
4
|
+
from typing import Optional
|
|
5
|
+
|
|
6
|
+
# Third Party
|
|
7
|
+
import torch
|
|
8
|
+
|
|
9
|
+
# First Party
|
|
10
|
+
from lmcache.logging import init_logger
|
|
11
|
+
from lmcache.v1.kv_layer_groups import KVLayerGroupsManager
|
|
12
|
+
|
|
13
|
+
logger = init_logger(__name__)
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
@dataclass
|
|
17
|
+
class LMCacheMetadata:
|
|
18
|
+
"""
|
|
19
|
+
LMCacheMetadata should be extracted from the northbound
|
|
20
|
+
serving engine configuration and wrap the extraction of
|
|
21
|
+
attributes (e.g. model name, tp rank, etc.)
|
|
22
|
+
"""
|
|
23
|
+
|
|
24
|
+
"""name of the LLM model"""
|
|
25
|
+
model_name: str
|
|
26
|
+
""" global world size when running under a distributed setting
|
|
27
|
+
(total number of workers)"""
|
|
28
|
+
world_size: int
|
|
29
|
+
""" host world size (workers on active localhost)
|
|
30
|
+
This information can be useful for multi-node
|
|
31
|
+
deployment. Will be the same as world_size
|
|
32
|
+
in single-node deployments.
|
|
33
|
+
"""
|
|
34
|
+
local_world_size: int
|
|
35
|
+
""" worker id when running under a distributed setting """
|
|
36
|
+
worker_id: int
|
|
37
|
+
""" host worker id (a gpu bound worker id on active localhost)
|
|
38
|
+
This information can be useful for multi-node deployment.
|
|
39
|
+
Will be the same as worker_id in single-node deployments.
|
|
40
|
+
"""
|
|
41
|
+
local_worker_id: int
|
|
42
|
+
""" the data type of kv tensors """
|
|
43
|
+
# (Deprecated) Will be replaced by kv_layer_groups_manager in the future
|
|
44
|
+
kv_dtype: torch.dtype
|
|
45
|
+
""" the shape of kv tensors """
|
|
46
|
+
# (Deprecated) Will be replaced by kv_layer_groups_manager in the future
|
|
47
|
+
""" (num_layer, 2, chunk_size, num_kv_head, head_size) """
|
|
48
|
+
kv_shape: tuple[int, int, int, int, int]
|
|
49
|
+
""" whether use MLA"""
|
|
50
|
+
use_mla: bool = False
|
|
51
|
+
""" the role of the current instance (e.g., 'scheduler', 'worker') """
|
|
52
|
+
role: Optional[str] = None
|
|
53
|
+
""" the first rank of the distributed setting """
|
|
54
|
+
# TODO(baoloongmao): first_rank should be configurable
|
|
55
|
+
first_rank = 0
|
|
56
|
+
served_model_name: Optional[str] = None
|
|
57
|
+
"""chunk size"""
|
|
58
|
+
chunk_size: int = 256
|
|
59
|
+
""" Manager for groups of layers with identical KV cache structure """
|
|
60
|
+
kv_layer_groups_manager: KVLayerGroupsManager = field(
|
|
61
|
+
default_factory=KVLayerGroupsManager
|
|
62
|
+
)
|
|
63
|
+
""" engine_id for RPC path (used by lookup client/server) """
|
|
64
|
+
engine_id: Optional[str] = None
|
|
65
|
+
""" extra config from kv_connector (e.g., lmcache_rpc_port) """
|
|
66
|
+
kv_connector_extra_config: Optional[dict] = None
|
|
67
|
+
|
|
68
|
+
def is_first_rank(self) -> bool:
|
|
69
|
+
"""Check if the current worker is the first rank"""
|
|
70
|
+
return self.worker_id == self.first_rank
|
|
71
|
+
|
|
72
|
+
# TODO(chunxiaozheng): some uts do not `build_kv_layer_groups`
|
|
73
|
+
def get_dtypes(self) -> list[torch.dtype]:
|
|
74
|
+
if self.kv_layer_groups_manager.kv_layer_groups:
|
|
75
|
+
return [
|
|
76
|
+
group.dtype for group in self.kv_layer_groups_manager.kv_layer_groups
|
|
77
|
+
]
|
|
78
|
+
return [self.kv_dtype]
|
|
79
|
+
|
|
80
|
+
def get_shapes(self, num_tokens: Optional[int] = None) -> list[torch.Size]:
|
|
81
|
+
"""Get the shapes of the KV cache in LMCache"""
|
|
82
|
+
if num_tokens is None:
|
|
83
|
+
num_tokens = self.chunk_size
|
|
84
|
+
if self.kv_layer_groups_manager.kv_layer_groups:
|
|
85
|
+
shapes = []
|
|
86
|
+
kv_size = 1 if self.use_mla else 2
|
|
87
|
+
for group in self.kv_layer_groups_manager.kv_layer_groups:
|
|
88
|
+
shapes.append(
|
|
89
|
+
torch.Size(
|
|
90
|
+
[
|
|
91
|
+
kv_size,
|
|
92
|
+
group.num_layers,
|
|
93
|
+
num_tokens,
|
|
94
|
+
group.hidden_dim_size,
|
|
95
|
+
]
|
|
96
|
+
)
|
|
97
|
+
)
|
|
98
|
+
return shapes
|
|
99
|
+
else:
|
|
100
|
+
return [
|
|
101
|
+
torch.Size(
|
|
102
|
+
[
|
|
103
|
+
self.kv_shape[1],
|
|
104
|
+
self.kv_shape[0],
|
|
105
|
+
num_tokens,
|
|
106
|
+
self.kv_shape[3] * self.kv_shape[4],
|
|
107
|
+
]
|
|
108
|
+
)
|
|
109
|
+
]
|
|
110
|
+
|
|
111
|
+
def get_num_groups(self) -> int:
|
|
112
|
+
if self.kv_layer_groups_manager.kv_layer_groups:
|
|
113
|
+
return self.kv_layer_groups_manager.num_groups
|
|
114
|
+
return 1
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
# MP Observability — Agent Instructions
|
|
2
|
+
|
|
3
|
+
When working in `lmcache/v1/mp_observability/`:
|
|
4
|
+
|
|
5
|
+
Design docs live in `docs/design/v1/mp_observability/` — update them whenever
|
|
6
|
+
the contracts below change.
|
|
7
|
+
|
|
8
|
+
1. **New `EventType`** — after adding an entry to `event.py`, update the
|
|
9
|
+
metadata contract table in `docs/design/v1/mp_observability/EVENTS.md`
|
|
10
|
+
with the new type's metadata keys and types.
|
|
11
|
+
|
|
12
|
+
2. **New metrics subscriber** — after adding counters/histograms, update the
|
|
13
|
+
metrics table in `docs/design/v1/mp_observability/METRICS.md` with the
|
|
14
|
+
metric name, type, and description.
|
|
15
|
+
|
|
16
|
+
3. **New subscriber class** — follow the step-by-step guide in `README.md`
|
|
17
|
+
(co-located with this file; "How to Add a New Event and Subscriber") and
|
|
18
|
+
the design rules in `docs/design/v1/mp_observability/event-bus.md`.
|
|
19
|
+
|
|
20
|
+
4. **CLI args** — if you add or change observability CLI flags in `config.py`,
|
|
21
|
+
update `docs/source/mp/observability.rst` to match.
|
|
@@ -0,0 +1,204 @@
|
|
|
1
|
+
# MP Observability
|
|
2
|
+
|
|
3
|
+
Event-driven observability for LMCache's multiprocess (MP) mode, built on
|
|
4
|
+
[OpenTelemetry](https://opentelemetry.io/).
|
|
5
|
+
|
|
6
|
+
For metrics, see [METRICS.md](../../../docs/design/v1/mp_observability/METRICS.md).
|
|
7
|
+
For event metadata contracts, see
|
|
8
|
+
[EVENTS.md](../../../docs/design/v1/mp_observability/EVENTS.md).
|
|
9
|
+
For design rationale, see
|
|
10
|
+
[event-bus.md](../../../docs/design/v1/mp_observability/event-bus.md).
|
|
11
|
+
For the trace recording subsystem (`lmcache trace`), see
|
|
12
|
+
[trace.md](../../../docs/design/v1/mp_observability/trace.md).
|
|
13
|
+
|
|
14
|
+
---
|
|
15
|
+
|
|
16
|
+
## Architecture
|
|
17
|
+
|
|
18
|
+
```
|
|
19
|
+
Producers (L1Manager, StorageManager, MPCacheEngine)
|
|
20
|
+
│
|
|
21
|
+
│ event_bus.publish(Event(...))
|
|
22
|
+
▼
|
|
23
|
+
EventBus (async queue + drain thread)
|
|
24
|
+
│
|
|
25
|
+
├──► L1MetricsSubscriber → OTel counter.add(...)
|
|
26
|
+
├──► SMMetricsSubscriber → OTel counter.add(...)
|
|
27
|
+
├──► L1LoggingSubscriber → logger.debug(...)
|
|
28
|
+
├──► SMLoggingSubscriber → logger.debug(...)
|
|
29
|
+
├──► MPServerLoggingSubscriber → logger.debug(...)
|
|
30
|
+
└──► MPServerTracingSubscriber → OTel span start/end
|
|
31
|
+
|
|
32
|
+
OTel SDK (configured at startup)
|
|
33
|
+
│
|
|
34
|
+
├──► OTLP push (production) → OTel collector → Prometheus / Grafana / etc.
|
|
35
|
+
└──► Prometheus pull (dev/debug) → /metrics on configured port
|
|
36
|
+
```
|
|
37
|
+
|
|
38
|
+
---
|
|
39
|
+
|
|
40
|
+
## Configuration
|
|
41
|
+
|
|
42
|
+
All observability behaviour is controlled by `ObservabilityConfig`
|
|
43
|
+
(defined in `config.py`). When running the LMCache MP mode server from the
|
|
44
|
+
CLI, pass the flags below; when embedding programmatically, construct an
|
|
45
|
+
`ObservabilityConfig` directly.
|
|
46
|
+
|
|
47
|
+
### CLI flags
|
|
48
|
+
|
|
49
|
+
| Flag | Default | Description |
|
|
50
|
+
|---|---|---|
|
|
51
|
+
| `--disable-observability` | off | Disable the EventBus entirely. No events are published or consumed. |
|
|
52
|
+
| `--disable-metrics` | off | Skip registering metrics subscribers (OTel counters). |
|
|
53
|
+
| `--disable-logging` | off | Skip registering logging subscribers. |
|
|
54
|
+
| `--enable-tracing` | off | Register tracing subscribers (OTel spans). Disabled by default. **Requires `--otlp-endpoint`.** |
|
|
55
|
+
| `--event-bus-queue-size N` | `10000` | Maximum number of events in the EventBus queue before tail-drop. |
|
|
56
|
+
| `--otlp-endpoint URL` | *(none)* | OTLP gRPC endpoint (e.g. `http://localhost:4317`). When set, metrics and traces are pushed to an OTel collector. When unset, metrics fall back to Prometheus pull mode. |
|
|
57
|
+
| `--prometheus-port PORT` | `9090` | Port for the Prometheus `/metrics` endpoint. Only used when `--otlp-endpoint` is not set. |
|
|
58
|
+
|
|
59
|
+
### `ObservabilityConfig` fields
|
|
60
|
+
|
|
61
|
+
| Field | Type | Default | Description |
|
|
62
|
+
|---|---|---|---|
|
|
63
|
+
| `enabled` | `bool` | `True` | Master switch for the EventBus. |
|
|
64
|
+
| `max_queue_size` | `int` | `10000` | Maximum events in the EventBus queue before tail-drop. |
|
|
65
|
+
| `metrics_enabled` | `bool` | `True` | Register metrics subscribers (OTel counters / histograms). |
|
|
66
|
+
| `logging_enabled` | `bool` | `True` | Register logging subscribers. |
|
|
67
|
+
| `tracing_enabled` | `bool` | `False` | Register tracing subscribers (OTel spans). |
|
|
68
|
+
| `otlp_endpoint` | `str \| None` | `None` | OTLP gRPC endpoint. When set, metrics and traces are pushed. When `None`, metrics use Prometheus pull fallback. |
|
|
69
|
+
| `prometheus_port` | `int` | `9090` | Port for the Prometheus `/metrics` endpoint (pull fallback only). |
|
|
70
|
+
|
|
71
|
+
### Metrics export modes
|
|
72
|
+
|
|
73
|
+
| `otlp_endpoint` | Mode | How to query |
|
|
74
|
+
|---|---|---|
|
|
75
|
+
| `http://host:4317` | OTLP push | Query the OTel collector's Prometheus exporter |
|
|
76
|
+
| `None` | Prometheus pull fallback | `curl http://localhost:<prometheus-port>/metrics` |
|
|
77
|
+
|
|
78
|
+
> **Note:** OTel counters only appear on `/metrics` after the first increment.
|
|
79
|
+
> If you see only Python runtime metrics, trigger a store/retrieve first.
|
|
80
|
+
|
|
81
|
+
### Tracing
|
|
82
|
+
|
|
83
|
+
Tracing is opt-in (`--enable-tracing`). When enabled, `MPServerTracingSubscriber`
|
|
84
|
+
creates OTel spans from MP server START/END event pairs (store, retrieve,
|
|
85
|
+
lookup/prefetch). Trace export requires an OTLP endpoint — there is no local
|
|
86
|
+
fallback. `--enable-tracing` **requires** `--otlp-endpoint`; the server will
|
|
87
|
+
raise a `ValueError` at startup if the endpoint is missing.
|
|
88
|
+
|
|
89
|
+
---
|
|
90
|
+
|
|
91
|
+
## How to Add a New Event and Subscriber
|
|
92
|
+
|
|
93
|
+
### Step 1 — Define the event type
|
|
94
|
+
|
|
95
|
+
Add a new member to `EventType` in `event.py`:
|
|
96
|
+
|
|
97
|
+
```python
|
|
98
|
+
class EventType(Enum):
|
|
99
|
+
# ... existing events ...
|
|
100
|
+
|
|
101
|
+
# My new component events
|
|
102
|
+
MY_COMPONENT_OPERATION = "my_component.operation"
|
|
103
|
+
```
|
|
104
|
+
|
|
105
|
+
### Step 2 — Publish the event from the producer
|
|
106
|
+
|
|
107
|
+
In your component (e.g., a manager class), publish to the EventBus:
|
|
108
|
+
|
|
109
|
+
```python
|
|
110
|
+
from lmcache.v1.mp_observability.event import Event, EventType
|
|
111
|
+
from lmcache.v1.mp_observability.event_bus import get_event_bus
|
|
112
|
+
|
|
113
|
+
class MyComponent:
|
|
114
|
+
def __init__(self):
|
|
115
|
+
self._event_bus = get_event_bus()
|
|
116
|
+
|
|
117
|
+
def do_operation(self, keys):
|
|
118
|
+
# ... business logic ...
|
|
119
|
+
|
|
120
|
+
self._event_bus.publish(Event(
|
|
121
|
+
event_type=EventType.MY_COMPONENT_OPERATION,
|
|
122
|
+
metadata={"keys": keys},
|
|
123
|
+
))
|
|
124
|
+
```
|
|
125
|
+
|
|
126
|
+
### Step 3 — Create a subscriber
|
|
127
|
+
|
|
128
|
+
Create a file under the appropriate `subscribers/` subdirectory:
|
|
129
|
+
|
|
130
|
+
- `subscribers/metrics/` for OTel counters / histograms
|
|
131
|
+
- `subscribers/logging/` for debug log output
|
|
132
|
+
- `subscribers/tracing/` for OTel spans
|
|
133
|
+
|
|
134
|
+
Example metrics subscriber (`subscribers/metrics/my_component.py`):
|
|
135
|
+
|
|
136
|
+
```python
|
|
137
|
+
from opentelemetry import metrics
|
|
138
|
+
from lmcache.v1.mp_observability.event import Event, EventType
|
|
139
|
+
from lmcache.v1.mp_observability.event_bus import EventCallback, EventSubscriber
|
|
140
|
+
|
|
141
|
+
|
|
142
|
+
class MyComponentMetricsSubscriber(EventSubscriber):
|
|
143
|
+
def __init__(self):
|
|
144
|
+
meter = metrics.get_meter("lmcache.my_component")
|
|
145
|
+
self._op_counter = meter.create_counter(
|
|
146
|
+
"lmcache_mp.my_component_operations",
|
|
147
|
+
description="Total operations on my component",
|
|
148
|
+
)
|
|
149
|
+
|
|
150
|
+
def get_subscriptions(self) -> dict[EventType, EventCallback]:
|
|
151
|
+
return {
|
|
152
|
+
EventType.MY_COMPONENT_OPERATION: self._on_operation,
|
|
153
|
+
}
|
|
154
|
+
|
|
155
|
+
def _on_operation(self, event: Event) -> None:
|
|
156
|
+
self._op_counter.add(len(event.metadata["keys"]))
|
|
157
|
+
```
|
|
158
|
+
|
|
159
|
+
### Step 4 — Export from `__init__.py`
|
|
160
|
+
|
|
161
|
+
Add the subscriber to the corresponding `__init__.py` so it can be
|
|
162
|
+
imported from the package:
|
|
163
|
+
|
|
164
|
+
```python
|
|
165
|
+
# subscribers/metrics/__init__.py
|
|
166
|
+
from lmcache.v1.mp_observability.subscribers.metrics.my_component import (
|
|
167
|
+
MyComponentMetricsSubscriber,
|
|
168
|
+
)
|
|
169
|
+
```
|
|
170
|
+
|
|
171
|
+
### Step 5 — Register the subscriber at startup
|
|
172
|
+
|
|
173
|
+
In the server startup function (e.g., `run_cache_server()` in `server.py`),
|
|
174
|
+
register conditionally based on `ObservabilityConfig`:
|
|
175
|
+
|
|
176
|
+
```python
|
|
177
|
+
if obs_config.metrics_enabled:
|
|
178
|
+
from lmcache.v1.mp_observability.subscribers.metrics import (
|
|
179
|
+
MyComponentMetricsSubscriber,
|
|
180
|
+
)
|
|
181
|
+
bus.register_subscriber(MyComponentMetricsSubscriber())
|
|
182
|
+
```
|
|
183
|
+
|
|
184
|
+
### Step 6 — Document the metadata contract
|
|
185
|
+
|
|
186
|
+
Add a row to the metadata contracts table in
|
|
187
|
+
[EVENTS.md](../../../docs/design/v1/mp_observability/EVENTS.md) so
|
|
188
|
+
subscribers can rely on the schema:
|
|
189
|
+
|
|
190
|
+
```markdown
|
|
191
|
+
| `MY_COMPONENT_OPERATION` | `keys` | `list[ObjectKey]` |
|
|
192
|
+
```
|
|
193
|
+
|
|
194
|
+
---
|
|
195
|
+
|
|
196
|
+
## Design rules
|
|
197
|
+
|
|
198
|
+
| Rule | Reason |
|
|
199
|
+
|---|---|
|
|
200
|
+
| Create meters and counters in `__init__()`, not at module level | `MeterProvider` must be set before `get_meter()` is called. Module-level calls happen at import time, before setup. |
|
|
201
|
+
| Prefix OTel metric names with `lmcache_mp.` | Keeps the MP namespace separate from `lmcache.` (the single-process engine namespace). |
|
|
202
|
+
| Use `metadata: dict[str, Any]` for event payloads | Flexible, no coupling between producers and subscribers. See metadata contracts in [EVENTS.md](../../../docs/design/v1/mp_observability/EVENTS.md). |
|
|
203
|
+
| Separate metrics, logging, and tracing subscribers | Single responsibility. Can enable/disable independently via config. |
|
|
204
|
+
| Store `self._event_bus = get_event_bus()` in `__init__` | Avoids calling the singleton getter on every publish. |
|
|
@@ -0,0 +1,340 @@
|
|
|
1
|
+
# SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
|
|
3
|
+
"""
|
|
4
|
+
Configuration for the MP-mode observability stack.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
# Future
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
# Standard
|
|
11
|
+
from dataclasses import dataclass, field
|
|
12
|
+
from typing import TYPE_CHECKING
|
|
13
|
+
import argparse
|
|
14
|
+
|
|
15
|
+
if TYPE_CHECKING:
|
|
16
|
+
# First Party
|
|
17
|
+
from lmcache.v1.mp_observability.event_bus import EventBus
|
|
18
|
+
|
|
19
|
+
# First Party
|
|
20
|
+
from lmcache.v1.mp_observability.subscribers.logging.lookup_hash import (
|
|
21
|
+
LookupHashLogConfig,
|
|
22
|
+
)
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
@dataclass
|
|
26
|
+
class ObservabilityConfig:
|
|
27
|
+
"""Unified configuration for the EventBus-based observability system.
|
|
28
|
+
|
|
29
|
+
Controls the EventBus, OTel metrics/tracing pipelines, and subscriber
|
|
30
|
+
registration.
|
|
31
|
+
"""
|
|
32
|
+
|
|
33
|
+
enabled: bool = True
|
|
34
|
+
"""Master switch for the EventBus."""
|
|
35
|
+
|
|
36
|
+
max_queue_size: int = 10_000
|
|
37
|
+
"""Maximum events in the EventBus queue before tail-drop."""
|
|
38
|
+
|
|
39
|
+
metrics_enabled: bool = True
|
|
40
|
+
"""Register metrics subscribers (OTel counters / histograms)."""
|
|
41
|
+
|
|
42
|
+
logging_enabled: bool = True
|
|
43
|
+
"""Register logging subscribers."""
|
|
44
|
+
|
|
45
|
+
tracing_enabled: bool = False
|
|
46
|
+
"""Register span subscribers (OTel traces)."""
|
|
47
|
+
|
|
48
|
+
otlp_endpoint: str | None = None
|
|
49
|
+
"""OTLP gRPC endpoint (e.g. ``http://localhost:4317``). When set,
|
|
50
|
+
metrics and traces are pushed to an OTel collector. When ``None``,
|
|
51
|
+
metrics fall back to an in-process Prometheus ``/metrics`` endpoint."""
|
|
52
|
+
|
|
53
|
+
prometheus_port: int = 9090
|
|
54
|
+
"""Port for the Prometheus /metrics endpoint. Only used when
|
|
55
|
+
``otlp_endpoint`` is ``None`` (Prometheus pull fallback)."""
|
|
56
|
+
|
|
57
|
+
metrics_sample_rate: float = 0.01
|
|
58
|
+
"""Fraction of chunks/blocks to track for lifecycle histograms (0, 1.0].
|
|
59
|
+
Counters always count all events regardless of this setting."""
|
|
60
|
+
|
|
61
|
+
lookup_hash_log: LookupHashLogConfig = field(default_factory=LookupHashLogConfig)
|
|
62
|
+
"""Configuration for lookup hash file logging. Disabled by default
|
|
63
|
+
(empty ``output_dir``)."""
|
|
64
|
+
|
|
65
|
+
trace_level: str | None = None
|
|
66
|
+
"""If set, enables trace recording at the given level. Currently
|
|
67
|
+
only ``"storage"`` is supported. See
|
|
68
|
+
:mod:`lmcache.v1.mp_observability.trace` for details."""
|
|
69
|
+
|
|
70
|
+
trace_output: str | None = None
|
|
71
|
+
"""Path to write the trace file. When :attr:`trace_level` is set
|
|
72
|
+
but this is ``None``, a timestamped path under ``$TMPDIR`` is
|
|
73
|
+
minted and logged at INFO."""
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
DEFAULT_OBSERVABILITY_CONFIG = ObservabilityConfig(enabled=False)
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
def add_observability_args(
|
|
80
|
+
parser: argparse.ArgumentParser,
|
|
81
|
+
) -> argparse.ArgumentParser:
|
|
82
|
+
"""Add observability configuration arguments to an existing parser.
|
|
83
|
+
|
|
84
|
+
Args:
|
|
85
|
+
parser: The argument parser to add arguments to.
|
|
86
|
+
|
|
87
|
+
Returns:
|
|
88
|
+
The same parser with observability arguments added.
|
|
89
|
+
"""
|
|
90
|
+
group = parser.add_argument_group(
|
|
91
|
+
"Observability", "Configuration for metrics, logging, and tracing"
|
|
92
|
+
)
|
|
93
|
+
group.add_argument(
|
|
94
|
+
"--disable-observability",
|
|
95
|
+
action="store_true",
|
|
96
|
+
default=False,
|
|
97
|
+
help="Disable the observability EventBus entirely.",
|
|
98
|
+
)
|
|
99
|
+
group.add_argument(
|
|
100
|
+
"--disable-metrics",
|
|
101
|
+
action="store_true",
|
|
102
|
+
default=False,
|
|
103
|
+
help="Disable metrics subscribers (OTel counters).",
|
|
104
|
+
)
|
|
105
|
+
group.add_argument(
|
|
106
|
+
"--disable-logging",
|
|
107
|
+
action="store_true",
|
|
108
|
+
default=False,
|
|
109
|
+
help="Disable logging subscribers.",
|
|
110
|
+
)
|
|
111
|
+
group.add_argument(
|
|
112
|
+
"--enable-tracing",
|
|
113
|
+
action="store_true",
|
|
114
|
+
default=False,
|
|
115
|
+
help="Enable span subscribers (OTel traces). Disabled by default.",
|
|
116
|
+
)
|
|
117
|
+
group.add_argument(
|
|
118
|
+
"--otlp-endpoint",
|
|
119
|
+
type=str,
|
|
120
|
+
default=None,
|
|
121
|
+
help=(
|
|
122
|
+
"OTLP gRPC endpoint (e.g. http://localhost:4317). "
|
|
123
|
+
"When set, metrics/traces are pushed to an OTel collector. "
|
|
124
|
+
"When unset, falls back to Prometheus pull mode."
|
|
125
|
+
),
|
|
126
|
+
)
|
|
127
|
+
group.add_argument(
|
|
128
|
+
"--event-bus-queue-size",
|
|
129
|
+
type=int,
|
|
130
|
+
default=10_000,
|
|
131
|
+
help=(
|
|
132
|
+
"Maximum number of events in the EventBus queue before "
|
|
133
|
+
"tail-drop. Default is 10000."
|
|
134
|
+
),
|
|
135
|
+
)
|
|
136
|
+
group.add_argument(
|
|
137
|
+
"--prometheus-port",
|
|
138
|
+
type=int,
|
|
139
|
+
default=9090,
|
|
140
|
+
help=(
|
|
141
|
+
"Port for the Prometheus /metrics endpoint. "
|
|
142
|
+
"Only used when --otlp-endpoint is not set. Default is 9090."
|
|
143
|
+
),
|
|
144
|
+
)
|
|
145
|
+
group.add_argument(
|
|
146
|
+
"--metrics-sample-rate",
|
|
147
|
+
type=float,
|
|
148
|
+
default=0.01,
|
|
149
|
+
help=(
|
|
150
|
+
"Fraction of chunks/blocks to track for lifecycle histograms "
|
|
151
|
+
"(0, 1.0]. Counters always count all events. Default is 0.01 (1%%)."
|
|
152
|
+
),
|
|
153
|
+
)
|
|
154
|
+
|
|
155
|
+
# Lookup hash logging config
|
|
156
|
+
log_group = parser.add_argument_group(
|
|
157
|
+
"Lookup Hash Logging",
|
|
158
|
+
"Configuration for lookup hash file logging (offline analysis)",
|
|
159
|
+
)
|
|
160
|
+
log_group.add_argument(
|
|
161
|
+
"--lookup-hash-log-dir",
|
|
162
|
+
type=str,
|
|
163
|
+
default="",
|
|
164
|
+
help="Directory to write lookup hash JSONL files for offline analysis. "
|
|
165
|
+
"Empty string (default) disables logging.",
|
|
166
|
+
)
|
|
167
|
+
log_group.add_argument(
|
|
168
|
+
"--lookup-hash-log-rotation-interval",
|
|
169
|
+
type=int,
|
|
170
|
+
default=6 * 3600,
|
|
171
|
+
help="Time interval in seconds before rotating to a new log file. "
|
|
172
|
+
"Default is 21600 (6 hours).",
|
|
173
|
+
)
|
|
174
|
+
log_group.add_argument(
|
|
175
|
+
"--lookup-hash-log-rotation-max-size",
|
|
176
|
+
type=int,
|
|
177
|
+
default=100 * 1024 * 1024,
|
|
178
|
+
help="Max file size in bytes before rotating even if the time "
|
|
179
|
+
"interval has not elapsed. Default is 100MB (104857600).",
|
|
180
|
+
)
|
|
181
|
+
log_group.add_argument(
|
|
182
|
+
"--lookup-hash-log-max-files",
|
|
183
|
+
type=int,
|
|
184
|
+
default=100,
|
|
185
|
+
help="Max number of lookup hash log files to keep. "
|
|
186
|
+
"Oldest files are deleted when this limit is exceeded. Default is 100.",
|
|
187
|
+
)
|
|
188
|
+
|
|
189
|
+
trace_group = parser.add_argument_group(
|
|
190
|
+
"Trace Recording",
|
|
191
|
+
"Capture LMCache operations to a binary trace file for replay "
|
|
192
|
+
"(see `lmcache trace`).",
|
|
193
|
+
)
|
|
194
|
+
trace_group.add_argument(
|
|
195
|
+
"--trace-level",
|
|
196
|
+
type=str,
|
|
197
|
+
choices=["storage"],
|
|
198
|
+
default=None,
|
|
199
|
+
help="Enable trace recording at the given level. Currently only "
|
|
200
|
+
"'storage' is supported (records StorageManager public-API calls).",
|
|
201
|
+
)
|
|
202
|
+
trace_group.add_argument(
|
|
203
|
+
"--trace-output",
|
|
204
|
+
type=str,
|
|
205
|
+
default=None,
|
|
206
|
+
help="Path to write the trace file. Defaults to a timestamped "
|
|
207
|
+
"file under $TMPDIR when --trace-level is set without an explicit "
|
|
208
|
+
"output path.",
|
|
209
|
+
)
|
|
210
|
+
|
|
211
|
+
return parser
|
|
212
|
+
|
|
213
|
+
|
|
214
|
+
def parse_args_to_observability_config(
|
|
215
|
+
args: argparse.Namespace,
|
|
216
|
+
) -> ObservabilityConfig:
|
|
217
|
+
"""Convert parsed command line arguments to an ObservabilityConfig.
|
|
218
|
+
|
|
219
|
+
Args:
|
|
220
|
+
args: Parsed arguments from the argument parser.
|
|
221
|
+
|
|
222
|
+
Returns:
|
|
223
|
+
The configuration object.
|
|
224
|
+
"""
|
|
225
|
+
config = ObservabilityConfig(
|
|
226
|
+
enabled=not args.disable_observability,
|
|
227
|
+
max_queue_size=args.event_bus_queue_size,
|
|
228
|
+
metrics_enabled=not args.disable_metrics,
|
|
229
|
+
logging_enabled=not args.disable_logging,
|
|
230
|
+
tracing_enabled=args.enable_tracing,
|
|
231
|
+
otlp_endpoint=args.otlp_endpoint,
|
|
232
|
+
prometheus_port=args.prometheus_port,
|
|
233
|
+
metrics_sample_rate=args.metrics_sample_rate,
|
|
234
|
+
lookup_hash_log=LookupHashLogConfig(
|
|
235
|
+
output_dir=args.lookup_hash_log_dir,
|
|
236
|
+
rotation_interval_sec=args.lookup_hash_log_rotation_interval,
|
|
237
|
+
rotation_max_size=args.lookup_hash_log_rotation_max_size,
|
|
238
|
+
max_files=args.lookup_hash_log_max_files,
|
|
239
|
+
),
|
|
240
|
+
trace_level=args.trace_level,
|
|
241
|
+
trace_output=args.trace_output,
|
|
242
|
+
)
|
|
243
|
+
|
|
244
|
+
if config.tracing_enabled and config.otlp_endpoint is None:
|
|
245
|
+
raise ValueError(
|
|
246
|
+
"--enable-tracing requires --otlp-endpoint to be set. "
|
|
247
|
+
"Tracing needs an OTLP gRPC endpoint to export spans."
|
|
248
|
+
)
|
|
249
|
+
|
|
250
|
+
return config
|
|
251
|
+
|
|
252
|
+
|
|
253
|
+
def init_observability(obs_config: ObservabilityConfig) -> EventBus:
|
|
254
|
+
"""Initialize OTel providers, EventBus, and register subscribers.
|
|
255
|
+
|
|
256
|
+
This is the single entry-point that every MP server calls at startup.
|
|
257
|
+
Returns a **started** EventBus.
|
|
258
|
+
"""
|
|
259
|
+
# First Party
|
|
260
|
+
from lmcache.v1.mp_observability.event_bus import (
|
|
261
|
+
EventBusConfig,
|
|
262
|
+
init_event_bus,
|
|
263
|
+
)
|
|
264
|
+
|
|
265
|
+
# Set up OTel providers BEFORE creating subscribers so that
|
|
266
|
+
# module-level get_meter()/get_tracer() calls bind to the real provider
|
|
267
|
+
if obs_config.enabled and obs_config.metrics_enabled:
|
|
268
|
+
# First Party
|
|
269
|
+
from lmcache.v1.mp_observability.otel_init import init_otel_metrics
|
|
270
|
+
|
|
271
|
+
init_otel_metrics(
|
|
272
|
+
otlp_endpoint=obs_config.otlp_endpoint,
|
|
273
|
+
prometheus_port=obs_config.prometheus_port,
|
|
274
|
+
)
|
|
275
|
+
|
|
276
|
+
if obs_config.enabled and obs_config.tracing_enabled:
|
|
277
|
+
# First Party
|
|
278
|
+
from lmcache.v1.mp_observability.otel_init import init_otel_tracing
|
|
279
|
+
|
|
280
|
+
init_otel_tracing(otlp_endpoint=obs_config.otlp_endpoint)
|
|
281
|
+
|
|
282
|
+
bus = init_event_bus(
|
|
283
|
+
EventBusConfig(
|
|
284
|
+
enabled=obs_config.enabled,
|
|
285
|
+
max_queue_size=obs_config.max_queue_size,
|
|
286
|
+
)
|
|
287
|
+
)
|
|
288
|
+
|
|
289
|
+
if obs_config.metrics_enabled:
|
|
290
|
+
# First Party
|
|
291
|
+
from lmcache.v1.mp_observability.subscribers.metrics import (
|
|
292
|
+
L0LifecycleSubscriber,
|
|
293
|
+
L1LifecycleSubscriber,
|
|
294
|
+
L1MetricsSubscriber,
|
|
295
|
+
L2MetricsSubscriber,
|
|
296
|
+
SMMetricsSubscriber,
|
|
297
|
+
)
|
|
298
|
+
|
|
299
|
+
sample_rate = obs_config.metrics_sample_rate
|
|
300
|
+
bus.register_subscriber(L0LifecycleSubscriber(sample_rate=sample_rate))
|
|
301
|
+
bus.register_subscriber(L1MetricsSubscriber())
|
|
302
|
+
bus.register_subscriber(L1LifecycleSubscriber(sample_rate=sample_rate))
|
|
303
|
+
bus.register_subscriber(L2MetricsSubscriber())
|
|
304
|
+
bus.register_subscriber(SMMetricsSubscriber())
|
|
305
|
+
|
|
306
|
+
if obs_config.logging_enabled:
|
|
307
|
+
# First Party
|
|
308
|
+
from lmcache.v1.mp_observability.subscribers.logging import (
|
|
309
|
+
L1LoggingSubscriber,
|
|
310
|
+
L2LoggingSubscriber,
|
|
311
|
+
MPServerLoggingSubscriber,
|
|
312
|
+
SMLoggingSubscriber,
|
|
313
|
+
)
|
|
314
|
+
|
|
315
|
+
bus.register_subscriber(MPServerLoggingSubscriber())
|
|
316
|
+
bus.register_subscriber(L1LoggingSubscriber())
|
|
317
|
+
bus.register_subscriber(L2LoggingSubscriber())
|
|
318
|
+
bus.register_subscriber(SMLoggingSubscriber())
|
|
319
|
+
|
|
320
|
+
if obs_config.tracing_enabled:
|
|
321
|
+
# First Party
|
|
322
|
+
from lmcache.v1.mp_observability.subscribers.tracing import (
|
|
323
|
+
MPServerTracingSubscriber,
|
|
324
|
+
get_span_registry,
|
|
325
|
+
)
|
|
326
|
+
|
|
327
|
+
bus.register_subscriber(MPServerTracingSubscriber(get_span_registry()))
|
|
328
|
+
|
|
329
|
+
# Lookup hash file logging (independent of the logging_enabled flag —
|
|
330
|
+
# it has its own enable gate via output_dir).
|
|
331
|
+
if obs_config.lookup_hash_log.enabled:
|
|
332
|
+
# First Party
|
|
333
|
+
from lmcache.v1.mp_observability.subscribers.logging.lookup_hash import (
|
|
334
|
+
LookupHashLoggingSubscriber,
|
|
335
|
+
)
|
|
336
|
+
|
|
337
|
+
bus.register_subscriber(LookupHashLoggingSubscriber(obs_config.lookup_hash_log))
|
|
338
|
+
|
|
339
|
+
bus.start()
|
|
340
|
+
return bus
|