lmcache-cli 0.4.5.dev0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- lmcache/__init__.py +84 -0
- lmcache/_version.py +24 -0
- lmcache/cli/__init__.py +1 -0
- lmcache/cli/commands/__init__.py +34 -0
- lmcache/cli/commands/base.py +157 -0
- lmcache/cli/commands/bench/__init__.py +557 -0
- lmcache/cli/commands/bench/engine_bench/__init__.py +1 -0
- lmcache/cli/commands/bench/engine_bench/config.py +245 -0
- lmcache/cli/commands/bench/engine_bench/interactive/__init__.py +274 -0
- lmcache/cli/commands/bench/engine_bench/interactive/config.json +10 -0
- lmcache/cli/commands/bench/engine_bench/interactive/schema.py +352 -0
- lmcache/cli/commands/bench/engine_bench/interactive/state.py +327 -0
- lmcache/cli/commands/bench/engine_bench/interactive/terminal.py +291 -0
- lmcache/cli/commands/bench/engine_bench/progress.py +145 -0
- lmcache/cli/commands/bench/engine_bench/request_sender.py +232 -0
- lmcache/cli/commands/bench/engine_bench/stats.py +275 -0
- lmcache/cli/commands/bench/engine_bench/workloads/__init__.py +153 -0
- lmcache/cli/commands/bench/engine_bench/workloads/base.py +122 -0
- lmcache/cli/commands/bench/engine_bench/workloads/long_doc_permutator.py +435 -0
- lmcache/cli/commands/bench/engine_bench/workloads/long_doc_qa.py +281 -0
- lmcache/cli/commands/bench/engine_bench/workloads/multi_round_chat.py +337 -0
- lmcache/cli/commands/bench/engine_bench/workloads/random_prefill.py +178 -0
- lmcache/cli/commands/describe.py +310 -0
- lmcache/cli/commands/kvcache.py +133 -0
- lmcache/cli/commands/mock.py +75 -0
- lmcache/cli/commands/ping.py +113 -0
- lmcache/cli/commands/query/__init__.py +155 -0
- lmcache/cli/commands/query/prompt.py +134 -0
- lmcache/cli/commands/query/request.py +357 -0
- lmcache/cli/commands/server.py +99 -0
- lmcache/cli/commands/tool/__init__.py +63 -0
- lmcache/cli/commands/tool/cache_simulator.py +113 -0
- lmcache/cli/commands/trace/__init__.py +505 -0
- lmcache/cli/commands/trace/dispatch.py +249 -0
- lmcache/cli/commands/trace/driver.py +372 -0
- lmcache/cli/commands/trace/stats.py +289 -0
- lmcache/cli/documents/lmcache.txt +11 -0
- lmcache/cli/main.py +42 -0
- lmcache/cli/metrics/__init__.py +29 -0
- lmcache/cli/metrics/formatter.py +171 -0
- lmcache/cli/metrics/handler.py +94 -0
- lmcache/cli/metrics/metrics.py +161 -0
- lmcache/cli/metrics/section.py +77 -0
- lmcache/connections.py +173 -0
- lmcache/integration/__init__.py +2 -0
- lmcache/integration/base_service_factory.py +165 -0
- lmcache/integration/request_telemetry/__init__.py +1 -0
- lmcache/integration/request_telemetry/base.py +51 -0
- lmcache/integration/request_telemetry/factory.py +113 -0
- lmcache/integration/request_telemetry/fastapi.py +109 -0
- lmcache/integration/request_telemetry/noop.py +35 -0
- lmcache/integration/sglang/__init__.py +2 -0
- lmcache/integration/sglang/sglang_adapter.py +326 -0
- lmcache/integration/sglang/utils.py +39 -0
- lmcache/integration/vllm/__init__.py +1 -0
- lmcache/integration/vllm/lmcache_connector_v1.py +213 -0
- lmcache/integration/vllm/lmcache_connector_v1_085.py +150 -0
- lmcache/integration/vllm/lmcache_mp_connector_0180.py +1072 -0
- lmcache/integration/vllm/tests/test_mm_hash_utils.py +112 -0
- lmcache/integration/vllm/utils.py +433 -0
- lmcache/integration/vllm/vllm_multi_process_adapter.py +1090 -0
- lmcache/integration/vllm/vllm_service_factory.py +339 -0
- lmcache/integration/vllm/vllm_v1_adapter.py +1713 -0
- lmcache/logging.py +107 -0
- lmcache/native_storage_ops.pyi +230 -0
- lmcache/non_cuda_equivalents.py +1424 -0
- lmcache/observability.py +1958 -0
- lmcache/storage_backend/serde/__init__.py +1 -0
- lmcache/storage_backend/serde/cachegen_basics.py +210 -0
- lmcache/storage_backend/serde/cachegen_decoder.py +207 -0
- lmcache/storage_backend/serde/cachegen_encoder.py +394 -0
- lmcache/storage_backend/serde/serde.py +75 -0
- lmcache/tools/__init__.py +1 -0
- lmcache/tools/cache_simulator/README.md +392 -0
- lmcache/tools/cache_simulator/__init__.py +1 -0
- lmcache/tools/cache_simulator/docs/simulate_example.png +0 -0
- lmcache/tools/cache_simulator/docs/sweep_example.png +0 -0
- lmcache/tools/cache_simulator/gen_bench_dataset.py +360 -0
- lmcache/tools/cache_simulator/lru_cache.py +124 -0
- lmcache/tools/cache_simulator/plot_hit_rate.py +231 -0
- lmcache/tools/cache_simulator/simulator.py +795 -0
- lmcache/tools/controller_benchmark/README.md +161 -0
- lmcache/tools/controller_benchmark/__init__.py +1 -0
- lmcache/tools/controller_benchmark/__main__.py +331 -0
- lmcache/tools/controller_benchmark/benchmark.py +660 -0
- lmcache/tools/controller_benchmark/config.py +44 -0
- lmcache/tools/controller_benchmark/constants.py +10 -0
- lmcache/tools/controller_benchmark/handlers/__init__.py +46 -0
- lmcache/tools/controller_benchmark/handlers/admit.py +52 -0
- lmcache/tools/controller_benchmark/handlers/base.py +47 -0
- lmcache/tools/controller_benchmark/handlers/deregister.py +49 -0
- lmcache/tools/controller_benchmark/handlers/evict.py +52 -0
- lmcache/tools/controller_benchmark/handlers/heartbeat.py +56 -0
- lmcache/tools/controller_benchmark/handlers/p2p_lookup.py +47 -0
- lmcache/tools/controller_benchmark/handlers/register.py +56 -0
- lmcache/tools/mp_status_viewer/__init__.py +1 -0
- lmcache/tools/mp_status_viewer/__main__.py +95 -0
- lmcache/usage_context.py +417 -0
- lmcache/utils.py +665 -0
- lmcache/v1/__init__.py +2 -0
- lmcache/v1/api_server/__init__.py +2 -0
- lmcache/v1/api_server/__main__.py +537 -0
- lmcache/v1/basic_check.py +112 -0
- lmcache/v1/cache_controller/__init__.py +9 -0
- lmcache/v1/cache_controller/commands/__init__.py +15 -0
- lmcache/v1/cache_controller/commands/base.py +35 -0
- lmcache/v1/cache_controller/commands/full_sync.py +49 -0
- lmcache/v1/cache_controller/config.py +176 -0
- lmcache/v1/cache_controller/controller_manager.py +535 -0
- lmcache/v1/cache_controller/controllers/__init__.py +11 -0
- lmcache/v1/cache_controller/controllers/full_sync_tracker.py +473 -0
- lmcache/v1/cache_controller/controllers/kv_controller.py +439 -0
- lmcache/v1/cache_controller/controllers/registration_controller.py +282 -0
- lmcache/v1/cache_controller/executor.py +463 -0
- lmcache/v1/cache_controller/frontend/static/css/style.css +201 -0
- lmcache/v1/cache_controller/frontend/static/img/logo.png +0 -0
- lmcache/v1/cache_controller/frontend/static/index.html +234 -0
- lmcache/v1/cache_controller/frontend/static/js/controller_app.js +660 -0
- lmcache/v1/cache_controller/full_sync_sender.py +475 -0
- lmcache/v1/cache_controller/locks.py +149 -0
- lmcache/v1/cache_controller/message.py +828 -0
- lmcache/v1/cache_controller/observability.py +208 -0
- lmcache/v1/cache_controller/utils.py +679 -0
- lmcache/v1/cache_controller/worker.py +665 -0
- lmcache/v1/cache_engine.py +2058 -0
- lmcache/v1/cache_interface.py +19 -0
- lmcache/v1/check/__init__.py +74 -0
- lmcache/v1/check/check_mode_gen.py +86 -0
- lmcache/v1/check/check_mode_test_l2_adapter.py +284 -0
- lmcache/v1/check/check_mode_test_remote.py +155 -0
- lmcache/v1/check/check_mode_test_storage_manager.py +142 -0
- lmcache/v1/check/utils.py +571 -0
- lmcache/v1/compute/__init__.py +2 -0
- lmcache/v1/compute/attention/__init__.py +0 -0
- lmcache/v1/compute/attention/abstract.py +39 -0
- lmcache/v1/compute/attention/flash_attn.py +129 -0
- lmcache/v1/compute/attention/flash_infer_sparse.py +284 -0
- lmcache/v1/compute/attention/metadata.py +85 -0
- lmcache/v1/compute/attention/utils.py +14 -0
- lmcache/v1/compute/blend/__init__.py +7 -0
- lmcache/v1/compute/blend/blender.py +168 -0
- lmcache/v1/compute/blend/metadata.py +34 -0
- lmcache/v1/compute/blend/utils.py +63 -0
- lmcache/v1/compute/models/__init__.py +0 -0
- lmcache/v1/compute/models/base.py +141 -0
- lmcache/v1/compute/models/llama.py +9 -0
- lmcache/v1/compute/models/qwen3.py +24 -0
- lmcache/v1/compute/models/utils.py +68 -0
- lmcache/v1/compute/positional_encoding.py +199 -0
- lmcache/v1/config.py +848 -0
- lmcache/v1/config_base.py +848 -0
- lmcache/v1/distributed/api.py +248 -0
- lmcache/v1/distributed/config.py +321 -0
- lmcache/v1/distributed/error.py +64 -0
- lmcache/v1/distributed/eviction.py +192 -0
- lmcache/v1/distributed/eviction_policy/__init__.py +21 -0
- lmcache/v1/distributed/eviction_policy/factory.py +27 -0
- lmcache/v1/distributed/eviction_policy/lru.py +244 -0
- lmcache/v1/distributed/eviction_policy/noop.py +50 -0
- lmcache/v1/distributed/internal_api.py +170 -0
- lmcache/v1/distributed/l1_manager.py +835 -0
- lmcache/v1/distributed/l2_adapters/__init__.py +67 -0
- lmcache/v1/distributed/l2_adapters/base.py +360 -0
- lmcache/v1/distributed/l2_adapters/config.py +385 -0
- lmcache/v1/distributed/l2_adapters/factory.py +205 -0
- lmcache/v1/distributed/l2_adapters/fs_l2_adapter.py +747 -0
- lmcache/v1/distributed/l2_adapters/fs_native_l2_adapter.py +167 -0
- lmcache/v1/distributed/l2_adapters/mock_l2_adapter.py +516 -0
- lmcache/v1/distributed/l2_adapters/mooncake_store_l2_adapter.py +135 -0
- lmcache/v1/distributed/l2_adapters/native_connector_l2_adapter.py +468 -0
- lmcache/v1/distributed/l2_adapters/native_plugin_l2_adapter.py +199 -0
- lmcache/v1/distributed/l2_adapters/nixl_store_dynamic_l2_adapter.py +831 -0
- lmcache/v1/distributed/l2_adapters/nixl_store_l2_adapter.py +983 -0
- lmcache/v1/distributed/l2_adapters/plugin_l2_adapter.py +210 -0
- lmcache/v1/distributed/l2_adapters/resp_l2_adapter.py +176 -0
- lmcache/v1/distributed/memory_manager.py +179 -0
- lmcache/v1/distributed/storage_controller.py +39 -0
- lmcache/v1/distributed/storage_controllers/__init__.py +43 -0
- lmcache/v1/distributed/storage_controllers/eviction_controller.py +242 -0
- lmcache/v1/distributed/storage_controllers/prefetch_controller.py +830 -0
- lmcache/v1/distributed/storage_controllers/prefetch_policy.py +193 -0
- lmcache/v1/distributed/storage_controllers/store_controller.py +452 -0
- lmcache/v1/distributed/storage_controllers/store_policy.py +213 -0
- lmcache/v1/distributed/storage_manager.py +532 -0
- lmcache/v1/event_manager.py +145 -0
- lmcache/v1/exceptions/__init__.py +16 -0
- lmcache/v1/gpu_connector/__init__.py +126 -0
- lmcache/v1/gpu_connector/gpu_connectors.py +1906 -0
- lmcache/v1/gpu_connector/gpu_ops.py +85 -0
- lmcache/v1/gpu_connector/hpu_connector.py +326 -0
- lmcache/v1/gpu_connector/mock_gpu_connector.py +67 -0
- lmcache/v1/gpu_connector/utils.py +890 -0
- lmcache/v1/gpu_connector/xpu_connectors.py +916 -0
- lmcache/v1/health_monitor/__init__.py +1 -0
- lmcache/v1/health_monitor/base.py +587 -0
- lmcache/v1/health_monitor/checks/__init__.py +1 -0
- lmcache/v1/health_monitor/checks/remote_backend_check.py +304 -0
- lmcache/v1/health_monitor/constants.py +36 -0
- lmcache/v1/internal_api_server/__init__.py +0 -0
- lmcache/v1/internal_api_server/api_registry.py +59 -0
- lmcache/v1/internal_api_server/api_server.py +120 -0
- lmcache/v1/internal_api_server/common/__init__.py +1 -0
- lmcache/v1/internal_api_server/common/env_api.py +22 -0
- lmcache/v1/internal_api_server/common/loglevel_api.py +57 -0
- lmcache/v1/internal_api_server/common/metrics_api.py +29 -0
- lmcache/v1/internal_api_server/common/periodic_thread_api.py +138 -0
- lmcache/v1/internal_api_server/common/run_script_api.py +73 -0
- lmcache/v1/internal_api_server/common/thread_api.py +63 -0
- lmcache/v1/internal_api_server/controller/__init__.py +1 -0
- lmcache/v1/internal_api_server/controller/key_stats_api.py +81 -0
- lmcache/v1/internal_api_server/controller/worker_info_api.py +136 -0
- lmcache/v1/internal_api_server/utils.py +43 -0
- lmcache/v1/internal_api_server/vllm/__init__.py +1 -0
- lmcache/v1/internal_api_server/vllm/backend_api.py +221 -0
- lmcache/v1/internal_api_server/vllm/bypass_api.py +204 -0
- lmcache/v1/internal_api_server/vllm/cache_api.py +895 -0
- lmcache/v1/internal_api_server/vllm/chunk_statistics_api.py +141 -0
- lmcache/v1/internal_api_server/vllm/conf_api.py +147 -0
- lmcache/v1/internal_api_server/vllm/freeze_api.py +172 -0
- lmcache/v1/internal_api_server/vllm/hot_cache_api.py +184 -0
- lmcache/v1/internal_api_server/vllm/inference_api.py +65 -0
- lmcache/v1/internal_api_server/vllm/load_fs_chunks_api.py +320 -0
- lmcache/v1/internal_api_server/vllm/lookup_api.py +145 -0
- lmcache/v1/internal_api_server/vllm/version_api.py +25 -0
- lmcache/v1/kv_layer_groups.py +267 -0
- lmcache/v1/lazy_memory_allocator.py +284 -0
- lmcache/v1/lookup_client/__init__.py +25 -0
- lmcache/v1/lookup_client/abstract_client.py +77 -0
- lmcache/v1/lookup_client/async_lookup_message.py +50 -0
- lmcache/v1/lookup_client/chunk_statistics_lookup_client.py +200 -0
- lmcache/v1/lookup_client/factory.py +251 -0
- lmcache/v1/lookup_client/hit_limit_lookup_client.py +86 -0
- lmcache/v1/lookup_client/lmcache_async_lookup_client.py +407 -0
- lmcache/v1/lookup_client/lmcache_lookup_client.py +285 -0
- lmcache/v1/lookup_client/lmcache_lookup_client_bypass.py +99 -0
- lmcache/v1/lookup_client/mooncake_lookup_client.py +87 -0
- lmcache/v1/lookup_client/record_strategies/__init__.py +77 -0
- lmcache/v1/lookup_client/record_strategies/base.py +327 -0
- lmcache/v1/lookup_client/record_strategies/file_hash.py +130 -0
- lmcache/v1/lookup_client/record_strategies/memory_bloom_filter.py +81 -0
- lmcache/v1/manager.py +539 -0
- lmcache/v1/memory_management.py +2619 -0
- lmcache/v1/metadata.py +114 -0
- lmcache/v1/mp_observability/AGENTS.override.md +21 -0
- lmcache/v1/mp_observability/README.md +204 -0
- lmcache/v1/mp_observability/config.py +340 -0
- lmcache/v1/mp_observability/event.py +100 -0
- lmcache/v1/mp_observability/event_bus.py +313 -0
- lmcache/v1/mp_observability/otel_init.py +145 -0
- lmcache/v1/mp_observability/subscribers/__init__.py +28 -0
- lmcache/v1/mp_observability/subscribers/logging/__init__.py +19 -0
- lmcache/v1/mp_observability/subscribers/logging/l1.py +56 -0
- lmcache/v1/mp_observability/subscribers/logging/l2.py +73 -0
- lmcache/v1/mp_observability/subscribers/logging/lookup_hash.py +209 -0
- lmcache/v1/mp_observability/subscribers/logging/mp_server.py +90 -0
- lmcache/v1/mp_observability/subscribers/logging/sm.py +59 -0
- lmcache/v1/mp_observability/subscribers/metrics/__init__.py +20 -0
- lmcache/v1/mp_observability/subscribers/metrics/l0_lifecycle.py +290 -0
- lmcache/v1/mp_observability/subscribers/metrics/l1.py +55 -0
- lmcache/v1/mp_observability/subscribers/metrics/l1_lifecycle.py +166 -0
- lmcache/v1/mp_observability/subscribers/metrics/l2.py +121 -0
- lmcache/v1/mp_observability/subscribers/metrics/sm.py +69 -0
- lmcache/v1/mp_observability/subscribers/tracing/__init__.py +12 -0
- lmcache/v1/mp_observability/subscribers/tracing/mp_server.py +333 -0
- lmcache/v1/mp_observability/subscribers/tracing/span_registry.py +148 -0
- lmcache/v1/mp_observability/trace/__init__.py +50 -0
- lmcache/v1/mp_observability/trace/codecs.py +255 -0
- lmcache/v1/mp_observability/trace/decorator.py +147 -0
- lmcache/v1/mp_observability/trace/format.py +132 -0
- lmcache/v1/mp_observability/trace/lifecycle.py +83 -0
- lmcache/v1/mp_observability/trace/reader.py +167 -0
- lmcache/v1/mp_observability/trace/recorder.py +300 -0
- lmcache/v1/multiprocess/__init__.py +0 -0
- lmcache/v1/multiprocess/affinity_pool.py +102 -0
- lmcache/v1/multiprocess/blend_server_v2.py +891 -0
- lmcache/v1/multiprocess/config.py +253 -0
- lmcache/v1/multiprocess/custom_types.py +281 -0
- lmcache/v1/multiprocess/futures.py +194 -0
- lmcache/v1/multiprocess/gpu_context.py +511 -0
- lmcache/v1/multiprocess/http_server.py +235 -0
- lmcache/v1/multiprocess/mp_runtime_plugin_launcher.py +130 -0
- lmcache/v1/multiprocess/mq.py +732 -0
- lmcache/v1/multiprocess/protocol.py +86 -0
- lmcache/v1/multiprocess/protocols/README.md +213 -0
- lmcache/v1/multiprocess/protocols/__init__.py +127 -0
- lmcache/v1/multiprocess/protocols/base.py +89 -0
- lmcache/v1/multiprocess/protocols/blend.py +109 -0
- lmcache/v1/multiprocess/protocols/blend_v2.py +57 -0
- lmcache/v1/multiprocess/protocols/controller.py +53 -0
- lmcache/v1/multiprocess/protocols/debug.py +34 -0
- lmcache/v1/multiprocess/protocols/engine.py +146 -0
- lmcache/v1/multiprocess/protocols/observability.py +39 -0
- lmcache/v1/multiprocess/server.py +1134 -0
- lmcache/v1/multiprocess/session.py +190 -0
- lmcache/v1/multiprocess/token_hasher.py +441 -0
- lmcache/v1/offload_server/__init__.py +17 -0
- lmcache/v1/offload_server/abstract_server.py +37 -0
- lmcache/v1/offload_server/message.py +30 -0
- lmcache/v1/offload_server/zmq_server.py +122 -0
- lmcache/v1/periodic_thread.py +579 -0
- lmcache/v1/pin_monitor.py +246 -0
- lmcache/v1/plugin/__init__.py +0 -0
- lmcache/v1/plugin/runtime_plugin_launcher.py +211 -0
- lmcache/v1/protocol.py +317 -0
- lmcache/v1/rpc/__init__.py +17 -0
- lmcache/v1/rpc/transport.py +105 -0
- lmcache/v1/rpc/zmq_transport.py +213 -0
- lmcache/v1/rpc_utils.py +165 -0
- lmcache/v1/server/__init__.py +2 -0
- lmcache/v1/server/__main__.py +170 -0
- lmcache/v1/server/storage_backend/__init__.py +21 -0
- lmcache/v1/server/storage_backend/abstract_backend.py +80 -0
- lmcache/v1/server/storage_backend/local_backend.py +75 -0
- lmcache/v1/server/utils.py +21 -0
- lmcache/v1/standalone/__init__.py +1 -0
- lmcache/v1/standalone/__main__.py +583 -0
- lmcache/v1/standalone/manager.py +80 -0
- lmcache/v1/standalone/standalone_service_factory.py +86 -0
- lmcache/v1/storage_backend/__init__.py +313 -0
- lmcache/v1/storage_backend/abstract_backend.py +445 -0
- lmcache/v1/storage_backend/audit_backend.py +233 -0
- lmcache/v1/storage_backend/batched_message_sender.py +222 -0
- lmcache/v1/storage_backend/cache_policy/__init__.py +45 -0
- lmcache/v1/storage_backend/cache_policy/base_policy.py +87 -0
- lmcache/v1/storage_backend/cache_policy/fifo.py +58 -0
- lmcache/v1/storage_backend/cache_policy/lfu.py +105 -0
- lmcache/v1/storage_backend/cache_policy/lru.py +81 -0
- lmcache/v1/storage_backend/cache_policy/mru.py +61 -0
- lmcache/v1/storage_backend/connector/__init__.py +443 -0
- lmcache/v1/storage_backend/connector/audit_adapter.py +77 -0
- lmcache/v1/storage_backend/connector/audit_connector.py +320 -0
- lmcache/v1/storage_backend/connector/base_connector.py +379 -0
- lmcache/v1/storage_backend/connector/blackhole_adapter.py +21 -0
- lmcache/v1/storage_backend/connector/blackhole_connector.py +37 -0
- lmcache/v1/storage_backend/connector/eic_adapter.py +31 -0
- lmcache/v1/storage_backend/connector/eic_connector.py +757 -0
- lmcache/v1/storage_backend/connector/external_adapter.py +79 -0
- lmcache/v1/storage_backend/connector/fs_adapter.py +51 -0
- lmcache/v1/storage_backend/connector/fs_connector.py +403 -0
- lmcache/v1/storage_backend/connector/infinistore_adapter.py +56 -0
- lmcache/v1/storage_backend/connector/infinistore_connector.py +177 -0
- lmcache/v1/storage_backend/connector/instrumented_connector.py +219 -0
- lmcache/v1/storage_backend/connector/lm_adapter.py +31 -0
- lmcache/v1/storage_backend/connector/lm_connector.py +176 -0
- lmcache/v1/storage_backend/connector/mock_adapter.py +57 -0
- lmcache/v1/storage_backend/connector/mock_connector.py +349 -0
- lmcache/v1/storage_backend/connector/mooncakestore_adapter.py +43 -0
- lmcache/v1/storage_backend/connector/mooncakestore_connector.py +614 -0
- lmcache/v1/storage_backend/connector/redis_adapter.py +181 -0
- lmcache/v1/storage_backend/connector/redis_connector.py +828 -0
- lmcache/v1/storage_backend/connector/s3_adapter.py +59 -0
- lmcache/v1/storage_backend/connector/s3_connector.py +699 -0
- lmcache/v1/storage_backend/connector/sagemaker_hyperpod_adapter.py +233 -0
- lmcache/v1/storage_backend/connector/sagemaker_hyperpod_connector.py +987 -0
- lmcache/v1/storage_backend/connector/valkey_adapter.py +114 -0
- lmcache/v1/storage_backend/connector/valkey_connector.py +627 -0
- lmcache/v1/storage_backend/gds_backend.py +1199 -0
- lmcache/v1/storage_backend/job_executor/__init__.py +0 -0
- lmcache/v1/storage_backend/job_executor/base_executor.py +34 -0
- lmcache/v1/storage_backend/job_executor/pq_executor.py +235 -0
- lmcache/v1/storage_backend/local_cpu_backend.py +810 -0
- lmcache/v1/storage_backend/local_disk_backend.py +656 -0
- lmcache/v1/storage_backend/maru_backend.py +734 -0
- lmcache/v1/storage_backend/naive_serde/__init__.py +50 -0
- lmcache/v1/storage_backend/naive_serde/cachegen_basics.py +133 -0
- lmcache/v1/storage_backend/naive_serde/cachegen_decoder.py +135 -0
- lmcache/v1/storage_backend/naive_serde/cachegen_encoder.py +83 -0
- lmcache/v1/storage_backend/naive_serde/kivi_serde.py +22 -0
- lmcache/v1/storage_backend/naive_serde/naive_serde.py +18 -0
- lmcache/v1/storage_backend/naive_serde/serde.py +37 -0
- lmcache/v1/storage_backend/native_clients/connector_client_base.py +165 -0
- lmcache/v1/storage_backend/native_clients/resp_client.py +35 -0
- lmcache/v1/storage_backend/nixl_storage_backend.py +1400 -0
- lmcache/v1/storage_backend/p2p_backend.py +788 -0
- lmcache/v1/storage_backend/path_sharder.py +117 -0
- lmcache/v1/storage_backend/pd_backend.py +646 -0
- lmcache/v1/storage_backend/plugins/dax_backend.py +1443 -0
- lmcache/v1/storage_backend/plugins/rust_raw_block_backend.py +1361 -0
- lmcache/v1/storage_backend/remote_backend.py +624 -0
- lmcache/v1/storage_backend/resp_client.py +227 -0
- lmcache/v1/storage_backend/storage_backend_listener.py +19 -0
- lmcache/v1/storage_backend/storage_manager.py +1352 -0
- lmcache/v1/system_detection.py +110 -0
- lmcache/v1/token_database.py +551 -0
- lmcache/v1/transfer_channel/__init__.py +83 -0
- lmcache/v1/transfer_channel/abstract.py +285 -0
- lmcache/v1/transfer_channel/mock_memory_channel.py +156 -0
- lmcache/v1/transfer_channel/nixl_channel.py +639 -0
- lmcache/v1/transfer_channel/py_socket_channel.py +260 -0
- lmcache/v1/transfer_channel/transfer_utils.py +63 -0
- lmcache/v1/utils/__init__.py +1 -0
- lmcache/v1/utils/bloom_filter.py +109 -0
- lmcache/v1/utils/cache_utils.py +125 -0
- lmcache_cli-0.4.5.dev0.dist-info/METADATA +185 -0
- lmcache_cli-0.4.5.dev0.dist-info/RECORD +399 -0
- lmcache_cli-0.4.5.dev0.dist-info/WHEEL +5 -0
- lmcache_cli-0.4.5.dev0.dist-info/entry_points.txt +2 -0
- lmcache_cli-0.4.5.dev0.dist-info/licenses/LICENSE +201 -0
- lmcache_cli-0.4.5.dev0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,190 @@
|
|
|
1
|
+
# SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
"""
|
|
3
|
+
Session and SessionManager for tracking per-request state
|
|
4
|
+
in the multiprocess cache server.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
# Standard
|
|
8
|
+
from dataclasses import dataclass, field
|
|
9
|
+
from typing import Any, Optional, overload
|
|
10
|
+
import threading
|
|
11
|
+
import time
|
|
12
|
+
|
|
13
|
+
# First Party
|
|
14
|
+
from lmcache.logging import init_logger
|
|
15
|
+
from lmcache.v1.multiprocess.custom_types import IPCCacheEngineKey
|
|
16
|
+
from lmcache.v1.multiprocess.token_hasher import TokenHasher
|
|
17
|
+
|
|
18
|
+
logger = init_logger(__name__)
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
@dataclass
|
|
22
|
+
class Session:
|
|
23
|
+
"""Tracks accumulated token IDs and computed chunk hashes for a request.
|
|
24
|
+
|
|
25
|
+
Thread-safe: all public methods are protected by an internal lock
|
|
26
|
+
to allow concurrent access from multiple TP worker threads.
|
|
27
|
+
"""
|
|
28
|
+
|
|
29
|
+
request_id: str
|
|
30
|
+
hasher: TokenHasher
|
|
31
|
+
token_ids: list[int] = field(default_factory=list)
|
|
32
|
+
chunk_hashes: list = field(default_factory=list)
|
|
33
|
+
last_prefix_hash: Any = None
|
|
34
|
+
num_chunks_processed: int = 0
|
|
35
|
+
created_at: float = field(default_factory=time.time)
|
|
36
|
+
lookup_ipc_key: Optional[IPCCacheEngineKey] = None
|
|
37
|
+
_lock: threading.Lock = field(default_factory=threading.Lock, repr=False)
|
|
38
|
+
|
|
39
|
+
def set_tokens(self, full_token_ids: list[int]) -> None:
|
|
40
|
+
"""Update the token sequence (idempotent, replaces not extends).
|
|
41
|
+
|
|
42
|
+
Args:
|
|
43
|
+
full_token_ids: Complete token sequence.
|
|
44
|
+
"""
|
|
45
|
+
with self._lock:
|
|
46
|
+
self.token_ids = full_token_ids
|
|
47
|
+
|
|
48
|
+
@overload
|
|
49
|
+
def get_hashes(self, start: int, end: int) -> list: ...
|
|
50
|
+
|
|
51
|
+
@overload
|
|
52
|
+
def get_hashes(self, start: int) -> list: ...
|
|
53
|
+
|
|
54
|
+
def get_hashes(self, start: int, end: int | None = None) -> list:
|
|
55
|
+
"""Compute and return chunk hashes for the [start, end) token range.
|
|
56
|
+
|
|
57
|
+
Internally computes rolling hashes up to end_chunk, skipping
|
|
58
|
+
already-computed chunks.
|
|
59
|
+
|
|
60
|
+
Two calling conventions are supported (declared via ``@overload``)::
|
|
61
|
+
|
|
62
|
+
get_hashes(start, end) # explicit end
|
|
63
|
+
get_hashes(start) # end = last full-chunk boundary
|
|
64
|
+
|
|
65
|
+
Args:
|
|
66
|
+
start: Start token index (must be aligned to chunk_size).
|
|
67
|
+
end: End token index (must be aligned to chunk_size).
|
|
68
|
+
When omitted (``None``), automatically set to the last
|
|
69
|
+
full-chunk boundary of the current token sequence.
|
|
70
|
+
|
|
71
|
+
Returns:
|
|
72
|
+
List of hash values for chunks in [start_chunk, end_chunk).
|
|
73
|
+
"""
|
|
74
|
+
chunk_size = self.hasher.chunk_size
|
|
75
|
+
assert start % chunk_size == 0, (
|
|
76
|
+
f"start ({start}) must be a multiple of chunk_size ({chunk_size})"
|
|
77
|
+
)
|
|
78
|
+
start_chunk = start // chunk_size
|
|
79
|
+
|
|
80
|
+
with self._lock:
|
|
81
|
+
if end is None:
|
|
82
|
+
# No explicit end: use the last full-chunk boundary.
|
|
83
|
+
# Lock must be held here because `self.token_ids` may be
|
|
84
|
+
# concurrently replaced by `set_tokens` from another thread.
|
|
85
|
+
end = len(self.token_ids) - (len(self.token_ids) % chunk_size)
|
|
86
|
+
assert end % chunk_size == 0, (
|
|
87
|
+
f"end ({end}) must be a multiple of chunk_size ({chunk_size})"
|
|
88
|
+
)
|
|
89
|
+
end_chunk = end // chunk_size
|
|
90
|
+
self._compute_hash(end_chunk)
|
|
91
|
+
return self.chunk_hashes[start_chunk:end_chunk]
|
|
92
|
+
|
|
93
|
+
def _compute_hash(self, end_chunk: int) -> None:
|
|
94
|
+
"""Compute rolling hashes up to end_chunk.
|
|
95
|
+
|
|
96
|
+
Uses cached state to skip already-computed chunks.
|
|
97
|
+
|
|
98
|
+
Args:
|
|
99
|
+
end_chunk: Compute hashes up to (but not including) this chunk.
|
|
100
|
+
"""
|
|
101
|
+
chunk_size = self.hasher.chunk_size
|
|
102
|
+
|
|
103
|
+
while self.num_chunks_processed < end_chunk:
|
|
104
|
+
cs = self.num_chunks_processed * chunk_size
|
|
105
|
+
ce = cs + chunk_size
|
|
106
|
+
chunk = self.token_ids[cs:ce]
|
|
107
|
+
|
|
108
|
+
prefix = (
|
|
109
|
+
self.last_prefix_hash
|
|
110
|
+
if self.last_prefix_hash is not None
|
|
111
|
+
else self.hasher.none_hash
|
|
112
|
+
)
|
|
113
|
+
h = self.hasher.hash_tokens(chunk, prefix)
|
|
114
|
+
self.last_prefix_hash = h
|
|
115
|
+
self.chunk_hashes.append(h)
|
|
116
|
+
self.num_chunks_processed += 1
|
|
117
|
+
|
|
118
|
+
|
|
119
|
+
class SessionManager:
|
|
120
|
+
"""Thread-safe manager for per-request sessions."""
|
|
121
|
+
|
|
122
|
+
DEFAULT_SESSION_TTL = 600 # 10 minutes
|
|
123
|
+
|
|
124
|
+
def __init__(self, hasher: TokenHasher, ttl: float = DEFAULT_SESSION_TTL):
|
|
125
|
+
self._hasher = hasher
|
|
126
|
+
self._ttl = ttl
|
|
127
|
+
self._sessions: dict[str, Session] = {}
|
|
128
|
+
self._lock = threading.Lock()
|
|
129
|
+
|
|
130
|
+
def get_or_create(self, request_id: str) -> Session:
|
|
131
|
+
"""Get existing session or create a new one.
|
|
132
|
+
|
|
133
|
+
Args:
|
|
134
|
+
request_id: Unique request identifier.
|
|
135
|
+
|
|
136
|
+
Returns:
|
|
137
|
+
The Session for this request_id.
|
|
138
|
+
"""
|
|
139
|
+
with self._lock:
|
|
140
|
+
if request_id not in self._sessions:
|
|
141
|
+
self._sessions[request_id] = Session(
|
|
142
|
+
request_id=request_id, hasher=self._hasher
|
|
143
|
+
)
|
|
144
|
+
logger.debug("Created session for request_id=%s", request_id)
|
|
145
|
+
return self._sessions[request_id]
|
|
146
|
+
|
|
147
|
+
def remove(self, request_id: str) -> Optional[Session]:
|
|
148
|
+
"""Remove a session by request_id.
|
|
149
|
+
|
|
150
|
+
Args:
|
|
151
|
+
request_id: Unique request identifier.
|
|
152
|
+
|
|
153
|
+
Returns:
|
|
154
|
+
The removed session, or None if no session was found.
|
|
155
|
+
"""
|
|
156
|
+
with self._lock:
|
|
157
|
+
if request_id in self._sessions:
|
|
158
|
+
session = self._sessions[request_id]
|
|
159
|
+
del self._sessions[request_id]
|
|
160
|
+
logger.debug("Removed session for request_id=%s", request_id)
|
|
161
|
+
return session
|
|
162
|
+
return None
|
|
163
|
+
|
|
164
|
+
def cleanup_expired(self) -> int:
|
|
165
|
+
"""Remove sessions that have exceeded their TTL.
|
|
166
|
+
|
|
167
|
+
Returns:
|
|
168
|
+
Number of sessions removed.
|
|
169
|
+
"""
|
|
170
|
+
now = time.time()
|
|
171
|
+
expired = []
|
|
172
|
+
with self._lock:
|
|
173
|
+
for rid, session in self._sessions.items():
|
|
174
|
+
if now - session.created_at > self._ttl:
|
|
175
|
+
expired.append(rid)
|
|
176
|
+
for rid in expired:
|
|
177
|
+
del self._sessions[rid]
|
|
178
|
+
|
|
179
|
+
if expired:
|
|
180
|
+
logger.info("Cleaned up %d expired sessions", len(expired))
|
|
181
|
+
return len(expired)
|
|
182
|
+
|
|
183
|
+
def active_count(self) -> int:
|
|
184
|
+
"""Return the number of active sessions.
|
|
185
|
+
|
|
186
|
+
Returns:
|
|
187
|
+
Number of currently tracked sessions.
|
|
188
|
+
"""
|
|
189
|
+
with self._lock:
|
|
190
|
+
return len(self._sessions)
|
|
@@ -0,0 +1,441 @@
|
|
|
1
|
+
# SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
"""
|
|
3
|
+
TokenHasher: Standalone hash computation for the multiprocess server.
|
|
4
|
+
|
|
5
|
+
Hash function loading logic is adapted from token_database.py to avoid
|
|
6
|
+
coupling with TokenDatabase's config/metadata dependencies.
|
|
7
|
+
|
|
8
|
+
vLLM compatibility notes:
|
|
9
|
+
- PR#20511: Introduced kv_cache_utils.init_none_hash()
|
|
10
|
+
- PR#23673: Renamed sha256_cbor_64bit to sha256_cbor
|
|
11
|
+
- PR#27151: Moved hash functions to vllm.utils.hashing module
|
|
12
|
+
"""
|
|
13
|
+
|
|
14
|
+
# Standard
|
|
15
|
+
from typing import Any, Callable
|
|
16
|
+
import os
|
|
17
|
+
|
|
18
|
+
# Third Party
|
|
19
|
+
from numba import njit
|
|
20
|
+
import numpy as np
|
|
21
|
+
|
|
22
|
+
# First Party
|
|
23
|
+
from lmcache.logging import init_logger
|
|
24
|
+
|
|
25
|
+
logger = init_logger(__name__)
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def _make_blake3_hash_func() -> Callable:
|
|
29
|
+
"""Create a blake3-based hash function compatible with the
|
|
30
|
+
(prefix_hash, tuple(tokens), None) calling convention."""
|
|
31
|
+
# Standard
|
|
32
|
+
import struct
|
|
33
|
+
|
|
34
|
+
# Third Party
|
|
35
|
+
import blake3 as _blake3
|
|
36
|
+
|
|
37
|
+
def blake3_hash(args):
|
|
38
|
+
prefix_hash, tokens, _ = args
|
|
39
|
+
h = _blake3.blake3()
|
|
40
|
+
# Serialize prefix hash
|
|
41
|
+
if isinstance(prefix_hash, bytes):
|
|
42
|
+
h.update(prefix_hash)
|
|
43
|
+
elif isinstance(prefix_hash, int):
|
|
44
|
+
h.update(prefix_hash.to_bytes(8, byteorder="big", signed=True))
|
|
45
|
+
else:
|
|
46
|
+
h.update(bytes(prefix_hash))
|
|
47
|
+
# Serialize token IDs in one batch
|
|
48
|
+
h.update(struct.pack(f">{len(tokens)}I", *tokens))
|
|
49
|
+
return h.digest() # 32 bytes
|
|
50
|
+
|
|
51
|
+
return blake3_hash
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
class TokenHasher:
|
|
55
|
+
"""Computes rolling prefix hashes for token chunks.
|
|
56
|
+
|
|
57
|
+
This class encapsulates the hash function loading and hash computation
|
|
58
|
+
logic needed by the multiprocess server to convert token IDs into
|
|
59
|
+
chunk hashes compatible with IPCCacheEngineKey (hash mode).
|
|
60
|
+
"""
|
|
61
|
+
|
|
62
|
+
def __init__(self, chunk_size: int = 256, hash_algorithm: str = "blake3"):
|
|
63
|
+
self.chunk_size = chunk_size
|
|
64
|
+
self.hash_algorithm_name = hash_algorithm
|
|
65
|
+
self.hash_func = self._get_hash_func(hash_algorithm)
|
|
66
|
+
self.none_hash = self._init_none_hash()
|
|
67
|
+
logger.info(
|
|
68
|
+
"TokenHasher initialized: chunk_size=%d, hash_algorithm=%s",
|
|
69
|
+
chunk_size,
|
|
70
|
+
hash_algorithm,
|
|
71
|
+
)
|
|
72
|
+
|
|
73
|
+
def _get_hash_func(self, hash_algorithm: str) -> Callable:
|
|
74
|
+
"""Load hash function with vLLM version compatibility.
|
|
75
|
+
|
|
76
|
+
Adapted from TokenDatabase._get_vllm_hash_func (token_database.py:97-168).
|
|
77
|
+
"""
|
|
78
|
+
if hash_algorithm == "blake3":
|
|
79
|
+
logger.info("Using blake3 hash function")
|
|
80
|
+
return _make_blake3_hash_func()
|
|
81
|
+
|
|
82
|
+
# Try get_hash_fn_by_name from both locations (PR#27151)
|
|
83
|
+
for module_path in ["vllm.utils.hashing", "vllm.utils"]:
|
|
84
|
+
try:
|
|
85
|
+
module = __import__(module_path, fromlist=["get_hash_fn_by_name"])
|
|
86
|
+
get_hash_fn_by_name = module.get_hash_fn_by_name
|
|
87
|
+
return self._try_get_hash(
|
|
88
|
+
get_hash_fn_by_name, hash_algorithm, module_path
|
|
89
|
+
)
|
|
90
|
+
except (ImportError, AttributeError, ValueError):
|
|
91
|
+
continue
|
|
92
|
+
|
|
93
|
+
# Try direct imports as fallback (for older vLLM versions)
|
|
94
|
+
func_names = (
|
|
95
|
+
["sha256_cbor", "sha256_cbor_64bit"]
|
|
96
|
+
if hash_algorithm in ("sha256_cbor", "sha256_cbor_64bit")
|
|
97
|
+
else [hash_algorithm]
|
|
98
|
+
)
|
|
99
|
+
for module_path in ["vllm.utils.hashing", "vllm.utils"]:
|
|
100
|
+
for func_name in func_names:
|
|
101
|
+
try:
|
|
102
|
+
module = __import__(module_path, fromlist=[func_name])
|
|
103
|
+
hash_func = getattr(module, func_name)
|
|
104
|
+
logger.info(
|
|
105
|
+
"Loaded '%s' from %s (direct import)", func_name, module_path
|
|
106
|
+
)
|
|
107
|
+
return hash_func
|
|
108
|
+
except (ImportError, AttributeError):
|
|
109
|
+
continue
|
|
110
|
+
|
|
111
|
+
# Fallback to builtin hash
|
|
112
|
+
logger.warning(
|
|
113
|
+
"Could not load '%s' from vLLM. Using builtin hash. "
|
|
114
|
+
"This may cause inconsistencies in distributed caching.",
|
|
115
|
+
hash_algorithm,
|
|
116
|
+
)
|
|
117
|
+
|
|
118
|
+
# Check PYTHONHASHSEED when using builtin hash
|
|
119
|
+
if os.getenv("PYTHONHASHSEED") is None:
|
|
120
|
+
logger.warning(
|
|
121
|
+
"Using builtin hash without PYTHONHASHSEED set. "
|
|
122
|
+
"For production environments (non-testing scenarios), you MUST set "
|
|
123
|
+
"PYTHONHASHSEED to ensure consistent hashing across processes. "
|
|
124
|
+
"Example: export PYTHONHASHSEED=0"
|
|
125
|
+
)
|
|
126
|
+
|
|
127
|
+
return hash
|
|
128
|
+
|
|
129
|
+
def _try_get_hash(
|
|
130
|
+
self, get_hash_fn_by_name: Callable, hash_algorithm: str, module_name: str
|
|
131
|
+
) -> Callable:
|
|
132
|
+
"""Try to get hash function, handling sha256_cbor_64bit rename.
|
|
133
|
+
|
|
134
|
+
Adapted from TokenDatabase._try_get_hash (token_database.py:152-168).
|
|
135
|
+
"""
|
|
136
|
+
# Handle sha256_cbor_64bit -> sha256_cbor rename (PR#23673)
|
|
137
|
+
names_to_try = (
|
|
138
|
+
["sha256_cbor", "sha256_cbor_64bit"]
|
|
139
|
+
if hash_algorithm in ("sha256_cbor", "sha256_cbor_64bit")
|
|
140
|
+
else [hash_algorithm]
|
|
141
|
+
)
|
|
142
|
+
|
|
143
|
+
for name in names_to_try:
|
|
144
|
+
try:
|
|
145
|
+
hash_func = get_hash_fn_by_name(name)
|
|
146
|
+
logger.info("Loaded '%s' from %s", name, module_name)
|
|
147
|
+
return hash_func
|
|
148
|
+
except ValueError:
|
|
149
|
+
continue
|
|
150
|
+
raise ValueError(f"Hash function '{hash_algorithm}' not found in {module_name}")
|
|
151
|
+
|
|
152
|
+
def _init_none_hash(self) -> Any:
|
|
153
|
+
"""Initialize NONE_HASH.
|
|
154
|
+
|
|
155
|
+
Adapted from TokenDatabase.__init__ (token_database.py:64-82).
|
|
156
|
+
"""
|
|
157
|
+
try:
|
|
158
|
+
# Third Party
|
|
159
|
+
from vllm.v1.core import kv_cache_utils
|
|
160
|
+
|
|
161
|
+
if hasattr(kv_cache_utils, "init_none_hash"):
|
|
162
|
+
kv_cache_utils.init_none_hash(self.hash_func)
|
|
163
|
+
none_hash = kv_cache_utils.NONE_HASH
|
|
164
|
+
logger.info("Initialized NONE_HASH=%s from vLLM", none_hash)
|
|
165
|
+
return none_hash
|
|
166
|
+
except (ImportError, AttributeError, ValueError):
|
|
167
|
+
pass
|
|
168
|
+
|
|
169
|
+
# Fallback: compute none_hash using our hash function
|
|
170
|
+
none_hash = self.hash_func((0, (0,), None))
|
|
171
|
+
logger.info("Computed NONE_HASH=%s using hash function", none_hash)
|
|
172
|
+
return none_hash
|
|
173
|
+
|
|
174
|
+
def hash_tokens(self, tokens: list[int], prefix_hash: Any = None) -> Any:
|
|
175
|
+
"""Hash one chunk with rolling prefix.
|
|
176
|
+
|
|
177
|
+
Returns int or bytes depending on hash_func.
|
|
178
|
+
"""
|
|
179
|
+
if prefix_hash is None:
|
|
180
|
+
prefix_hash = self.none_hash
|
|
181
|
+
return self.hash_func((prefix_hash, tuple(tokens), None))
|
|
182
|
+
|
|
183
|
+
def compute_chunk_hashes(
|
|
184
|
+
self,
|
|
185
|
+
token_ids: list[int],
|
|
186
|
+
prefix_hash: Any = None,
|
|
187
|
+
start: int = 0,
|
|
188
|
+
end: int | None = None,
|
|
189
|
+
) -> list[bytes]:
|
|
190
|
+
"""Compute rolling prefix hashes for complete chunks in a token range.
|
|
191
|
+
|
|
192
|
+
The rolling hash is always computed from the beginning of
|
|
193
|
+
``token_ids`` (since each chunk's hash depends on all previous
|
|
194
|
+
chunks), but only hashes for chunks within ``[start, end)`` are
|
|
195
|
+
returned, and hashing stops at ``end`` to avoid unnecessary work.
|
|
196
|
+
|
|
197
|
+
``start`` and ``end`` are token-level indices and must be
|
|
198
|
+
multiples of ``chunk_size``. Partial chunks are discarded.
|
|
199
|
+
|
|
200
|
+
Args:
|
|
201
|
+
token_ids: Full token sequence.
|
|
202
|
+
prefix_hash: Optional initial prefix hash (defaults to none_hash).
|
|
203
|
+
start: Token-level start index (must be chunk-aligned).
|
|
204
|
+
Chunks before this index are computed but not returned.
|
|
205
|
+
end: Token-level end index (must be chunk-aligned). When
|
|
206
|
+
provided, hashing stops at this index.
|
|
207
|
+
|
|
208
|
+
Returns:
|
|
209
|
+
List of ``bytes`` hash values for chunks in ``[start, end)``.
|
|
210
|
+
"""
|
|
211
|
+
hashes: list[bytes] = []
|
|
212
|
+
prefix_hash = self.none_hash if prefix_hash is None else prefix_hash
|
|
213
|
+
effective_len = min(len(token_ids), end) if end is not None else len(token_ids)
|
|
214
|
+
num_complete = effective_len - effective_len % self.chunk_size
|
|
215
|
+
for i in range(0, num_complete, self.chunk_size):
|
|
216
|
+
prefix_hash = self.hash_tokens(
|
|
217
|
+
token_ids[i : i + self.chunk_size], prefix_hash
|
|
218
|
+
)
|
|
219
|
+
if i >= start:
|
|
220
|
+
hashes.append(self.hash_to_bytes(prefix_hash))
|
|
221
|
+
return hashes
|
|
222
|
+
|
|
223
|
+
@staticmethod
|
|
224
|
+
def hash_to_bytes(hash_val: Any) -> bytes:
|
|
225
|
+
"""Convert hash value to bytes for ObjectKey.chunk_hash.
|
|
226
|
+
|
|
227
|
+
Handles both bytes (sha256_cbor) and int (builtin hash) return types.
|
|
228
|
+
"""
|
|
229
|
+
if isinstance(hash_val, bytes):
|
|
230
|
+
return hash_val # sha256_cbor already returns bytes
|
|
231
|
+
return hash_val.to_bytes(8, byteorder="big", signed=True)
|
|
232
|
+
|
|
233
|
+
|
|
234
|
+
### Functions for fast rolling/chunk hash and dict lookup
|
|
235
|
+
|
|
236
|
+
|
|
237
|
+
@njit(cache=True)
|
|
238
|
+
def rolling_hash_windows_numba(
|
|
239
|
+
arr_u64: np.ndarray, k: int, base: np.uint64
|
|
240
|
+
) -> np.ndarray:
|
|
241
|
+
"""
|
|
242
|
+
Compute rolling polynomial hashes over a uint64 array.
|
|
243
|
+
|
|
244
|
+
This function computes a polynomial rolling hash over a sliding window
|
|
245
|
+
of size `k` across the input array `arr_u64`. Arithmetic is performed
|
|
246
|
+
in uint64 with natural overflow, which is equivalent to computing
|
|
247
|
+
modulo 2^64.
|
|
248
|
+
|
|
249
|
+
Hash definition for a window [x0, x1, ..., x_{k-1}]:
|
|
250
|
+
|
|
251
|
+
H = x0 * base^(k-1) + x1 * base^(k-2) + ... + x_{k-1}
|
|
252
|
+
|
|
253
|
+
For each subsequent window the hash is updated in O(1):
|
|
254
|
+
|
|
255
|
+
H_new = (H - x_old * base^(k-1)) * base + x_new
|
|
256
|
+
|
|
257
|
+
Parameters
|
|
258
|
+
----------
|
|
259
|
+
arr_u64 : np.ndarray[np.uint64]
|
|
260
|
+
Input array of integers encoded as uint64 values.
|
|
261
|
+
|
|
262
|
+
k : int
|
|
263
|
+
Sliding window size.
|
|
264
|
+
|
|
265
|
+
base : np.uint64
|
|
266
|
+
Base of the polynomial hash. Typically a random odd 64-bit number.
|
|
267
|
+
|
|
268
|
+
Returns
|
|
269
|
+
-------
|
|
270
|
+
np.ndarray[np.uint64]
|
|
271
|
+
Array of rolling hash values of length:
|
|
272
|
+
|
|
273
|
+
len(arr_u64) - k + 1
|
|
274
|
+
|
|
275
|
+
Each element corresponds to the hash of one window.
|
|
276
|
+
"""
|
|
277
|
+
n = arr_u64.shape[0]
|
|
278
|
+
out = np.empty(n - k + 1, dtype=np.uint64)
|
|
279
|
+
|
|
280
|
+
power = np.uint64(1)
|
|
281
|
+
for _ in range(k - 1):
|
|
282
|
+
power = power * base # uint64 overflow = mod 2^64
|
|
283
|
+
|
|
284
|
+
h = np.uint64(0)
|
|
285
|
+
for i in range(k):
|
|
286
|
+
h = h * base + arr_u64[i]
|
|
287
|
+
out[0] = h
|
|
288
|
+
|
|
289
|
+
j = 1
|
|
290
|
+
for i in range(k, n):
|
|
291
|
+
old = arr_u64[i - k]
|
|
292
|
+
new = arr_u64[i]
|
|
293
|
+
h = h - old * power
|
|
294
|
+
h = h * base + new
|
|
295
|
+
out[j] = h
|
|
296
|
+
j += 1
|
|
297
|
+
|
|
298
|
+
return out
|
|
299
|
+
|
|
300
|
+
|
|
301
|
+
@njit(cache=True)
|
|
302
|
+
def chunk_hash_windows_numba(arr_u64, k, base):
|
|
303
|
+
"""Compute polynomial hashes over non-overlapping (chunked) windows.
|
|
304
|
+
|
|
305
|
+
Unlike the rolling-hash variant, each window's hash is computed
|
|
306
|
+
independently from scratch, which is efficient when stride equals the
|
|
307
|
+
window size (i.e., windows do not overlap).
|
|
308
|
+
|
|
309
|
+
The hash for a window starting at position ``s`` is:
|
|
310
|
+
|
|
311
|
+
h = arr[s]*base^(k-1) + arr[s+1]*base^(k-2) + ... + arr[s+k-1]
|
|
312
|
+
|
|
313
|
+
computed with natural ``uint64`` overflow (mod 2^64).
|
|
314
|
+
|
|
315
|
+
Parameters
|
|
316
|
+
----------
|
|
317
|
+
arr_u64 : np.ndarray[np.uint64]
|
|
318
|
+
1-D array of token values cast to ``uint64``.
|
|
319
|
+
k : int
|
|
320
|
+
Window (chunk) size.
|
|
321
|
+
base : np.uint64
|
|
322
|
+
Base of the polynomial hash.
|
|
323
|
+
|
|
324
|
+
Returns
|
|
325
|
+
-------
|
|
326
|
+
np.ndarray[np.uint64]
|
|
327
|
+
Array of length ``len(arr_u64) // k`` containing one hash per
|
|
328
|
+
non-overlapping chunk. Trailing tokens that do not fill a
|
|
329
|
+
complete chunk are ignored.
|
|
330
|
+
"""
|
|
331
|
+
n = arr_u64.shape[0]
|
|
332
|
+
num_windows = n // k
|
|
333
|
+
out = np.empty(num_windows, dtype=np.uint64)
|
|
334
|
+
|
|
335
|
+
for w in range(num_windows):
|
|
336
|
+
h = np.uint64(0)
|
|
337
|
+
start = w * k
|
|
338
|
+
# Compute fresh hash for this block
|
|
339
|
+
for i in range(start, start + k):
|
|
340
|
+
h = h * base + arr_u64[i]
|
|
341
|
+
out[w] = h
|
|
342
|
+
return out
|
|
343
|
+
|
|
344
|
+
|
|
345
|
+
@njit(cache=True)
|
|
346
|
+
def update_table_id_numba(
|
|
347
|
+
hashes_u64: np.ndarray,
|
|
348
|
+
table_id_i64: np.ndarray,
|
|
349
|
+
vals_to_update: np.ndarray,
|
|
350
|
+
):
|
|
351
|
+
"""
|
|
352
|
+
Update the direct-address table with new ID values for given hashes.
|
|
353
|
+
|
|
354
|
+
For each hash in `hashes_u64`, compute the index as:
|
|
355
|
+
|
|
356
|
+
idx = hash & (table_id_i64.size - 1)
|
|
357
|
+
|
|
358
|
+
and update `table_id_i64[idx]` with the corresponding value from
|
|
359
|
+
`vals_to_update`.
|
|
360
|
+
|
|
361
|
+
Parameters
|
|
362
|
+
----------
|
|
363
|
+
hashes_u64 : np.ndarray[np.uint64]
|
|
364
|
+
Array of hash values to update.
|
|
365
|
+
|
|
366
|
+
table_id_i64 : np.ndarray[np.int64]
|
|
367
|
+
Direct-address lookup table mapping index → ID. This array is
|
|
368
|
+
modified in-place.
|
|
369
|
+
|
|
370
|
+
vals_to_update : np.ndarray[np.int64]
|
|
371
|
+
Array of new ID values to write into the table. Must have the same
|
|
372
|
+
length as `hashes_u64`.
|
|
373
|
+
"""
|
|
374
|
+
n = hashes_u64.shape[0]
|
|
375
|
+
m = table_id_i64.shape[0]
|
|
376
|
+
|
|
377
|
+
for i in range(n):
|
|
378
|
+
idx = hashes_u64[i] & (m - 1) # Assuming m is a power of 2
|
|
379
|
+
table_id_i64[idx] = vals_to_update[i]
|
|
380
|
+
|
|
381
|
+
|
|
382
|
+
@njit(cache=True)
|
|
383
|
+
def unique_hits_direct_id_numba(
|
|
384
|
+
hashes_u64: np.ndarray, table_id_i64: np.ndarray, mask_u64: np.uint64, num_ids: int
|
|
385
|
+
) -> np.ndarray:
|
|
386
|
+
"""
|
|
387
|
+
Perform direct-address lookup with deduplication of results.
|
|
388
|
+
|
|
389
|
+
This function looks up each hash in a direct-address table using
|
|
390
|
+
the lower bits of the hash:
|
|
391
|
+
|
|
392
|
+
idx = hash & mask
|
|
393
|
+
|
|
394
|
+
The lookup table maps each index to an integer ID.
|
|
395
|
+
|
|
396
|
+
The function returns **unique IDs only**, meaning that if the same
|
|
397
|
+
ID appears multiple times across the hash stream it will be returned
|
|
398
|
+
only once.
|
|
399
|
+
|
|
400
|
+
Parameters
|
|
401
|
+
----------
|
|
402
|
+
hashes_u64 : np.ndarray[np.uint64]
|
|
403
|
+
Array of rolling hash values.
|
|
404
|
+
|
|
405
|
+
table_id_i64 : np.ndarray[np.int64]
|
|
406
|
+
Direct-address lookup table mapping index → ID.
|
|
407
|
+
Values of -1 represent "no entry".
|
|
408
|
+
|
|
409
|
+
mask_u64 : np.uint64
|
|
410
|
+
Bitmask used to compute the index:
|
|
411
|
+
|
|
412
|
+
idx = hash & mask_u64
|
|
413
|
+
|
|
414
|
+
Typically mask = (2^bits - 1).
|
|
415
|
+
|
|
416
|
+
num_ids : int
|
|
417
|
+
Maximum possible ID value + 1. This determines the size of the
|
|
418
|
+
internal `seen` array used for deduplication.
|
|
419
|
+
|
|
420
|
+
Returns
|
|
421
|
+
-------
|
|
422
|
+
np.ndarray[np.int64]
|
|
423
|
+
Array containing the unique IDs encountered in the lookup stream.
|
|
424
|
+
Length ≤ len(hashes_u64).
|
|
425
|
+
"""
|
|
426
|
+
|
|
427
|
+
# TODO(Jiayi): These allocations can be avoided by pre-allocations
|
|
428
|
+
seen = np.zeros(num_ids, dtype=np.uint8) # 1 byte per possible id
|
|
429
|
+
out = np.empty(hashes_u64.shape[0], dtype=np.int64)
|
|
430
|
+
|
|
431
|
+
m = 0
|
|
432
|
+
for i in range(hashes_u64.shape[0]):
|
|
433
|
+
idx = hashes_u64[i] & mask_u64
|
|
434
|
+
hit = table_id_i64[idx]
|
|
435
|
+
|
|
436
|
+
if hit != -1 and seen[hit] == 0:
|
|
437
|
+
seen[hit] = 1
|
|
438
|
+
out[m] = hit
|
|
439
|
+
m += 1
|
|
440
|
+
|
|
441
|
+
return out[:m]
|
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
# SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
# First Party
|
|
3
|
+
from lmcache.v1.lookup_client.abstract_client import LookupClientInterface
|
|
4
|
+
from lmcache.v1.lookup_client.factory import LookupClientFactory
|
|
5
|
+
from lmcache.v1.lookup_client.lmcache_lookup_client import (
|
|
6
|
+
LMCacheLookupClient,
|
|
7
|
+
LMCacheLookupServer,
|
|
8
|
+
)
|
|
9
|
+
from lmcache.v1.lookup_client.mooncake_lookup_client import MooncakeLookupClient
|
|
10
|
+
|
|
11
|
+
__all__ = [
|
|
12
|
+
"LookupClientInterface",
|
|
13
|
+
"LookupClientFactory",
|
|
14
|
+
"MooncakeLookupClient",
|
|
15
|
+
"LMCacheLookupClient",
|
|
16
|
+
"LMCacheLookupServer",
|
|
17
|
+
]
|
|
@@ -0,0 +1,37 @@
|
|
|
1
|
+
# SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
# Standard
|
|
3
|
+
from typing import TYPE_CHECKING, List
|
|
4
|
+
import abc
|
|
5
|
+
|
|
6
|
+
if TYPE_CHECKING:
|
|
7
|
+
# Third Party
|
|
8
|
+
pass
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
class OffloadServerInterface(metaclass=abc.ABCMeta):
|
|
12
|
+
"""Abstract interface for offload server."""
|
|
13
|
+
|
|
14
|
+
@abc.abstractmethod
|
|
15
|
+
def offload(
|
|
16
|
+
self,
|
|
17
|
+
hashes: List[int],
|
|
18
|
+
slot_mapping: List[int],
|
|
19
|
+
offsets: List[int],
|
|
20
|
+
) -> bool:
|
|
21
|
+
"""
|
|
22
|
+
Perform offload for the given hashes and block IDs.
|
|
23
|
+
|
|
24
|
+
Args:
|
|
25
|
+
hashes: The hashes to offload.
|
|
26
|
+
slot_mapping: The slot ids to offload.
|
|
27
|
+
offsets: Number of tokens in each block.
|
|
28
|
+
|
|
29
|
+
Returns:
|
|
30
|
+
Whether the offload was successful.
|
|
31
|
+
"""
|
|
32
|
+
raise NotImplementedError
|
|
33
|
+
|
|
34
|
+
@abc.abstractmethod
|
|
35
|
+
def close(self) -> None:
|
|
36
|
+
"""Close the offload server and clean up resources."""
|
|
37
|
+
raise NotImplementedError
|