lmcache-cli 0.4.5.dev0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (399) hide show
  1. lmcache/__init__.py +84 -0
  2. lmcache/_version.py +24 -0
  3. lmcache/cli/__init__.py +1 -0
  4. lmcache/cli/commands/__init__.py +34 -0
  5. lmcache/cli/commands/base.py +157 -0
  6. lmcache/cli/commands/bench/__init__.py +557 -0
  7. lmcache/cli/commands/bench/engine_bench/__init__.py +1 -0
  8. lmcache/cli/commands/bench/engine_bench/config.py +245 -0
  9. lmcache/cli/commands/bench/engine_bench/interactive/__init__.py +274 -0
  10. lmcache/cli/commands/bench/engine_bench/interactive/config.json +10 -0
  11. lmcache/cli/commands/bench/engine_bench/interactive/schema.py +352 -0
  12. lmcache/cli/commands/bench/engine_bench/interactive/state.py +327 -0
  13. lmcache/cli/commands/bench/engine_bench/interactive/terminal.py +291 -0
  14. lmcache/cli/commands/bench/engine_bench/progress.py +145 -0
  15. lmcache/cli/commands/bench/engine_bench/request_sender.py +232 -0
  16. lmcache/cli/commands/bench/engine_bench/stats.py +275 -0
  17. lmcache/cli/commands/bench/engine_bench/workloads/__init__.py +153 -0
  18. lmcache/cli/commands/bench/engine_bench/workloads/base.py +122 -0
  19. lmcache/cli/commands/bench/engine_bench/workloads/long_doc_permutator.py +435 -0
  20. lmcache/cli/commands/bench/engine_bench/workloads/long_doc_qa.py +281 -0
  21. lmcache/cli/commands/bench/engine_bench/workloads/multi_round_chat.py +337 -0
  22. lmcache/cli/commands/bench/engine_bench/workloads/random_prefill.py +178 -0
  23. lmcache/cli/commands/describe.py +310 -0
  24. lmcache/cli/commands/kvcache.py +133 -0
  25. lmcache/cli/commands/mock.py +75 -0
  26. lmcache/cli/commands/ping.py +113 -0
  27. lmcache/cli/commands/query/__init__.py +155 -0
  28. lmcache/cli/commands/query/prompt.py +134 -0
  29. lmcache/cli/commands/query/request.py +357 -0
  30. lmcache/cli/commands/server.py +99 -0
  31. lmcache/cli/commands/tool/__init__.py +63 -0
  32. lmcache/cli/commands/tool/cache_simulator.py +113 -0
  33. lmcache/cli/commands/trace/__init__.py +505 -0
  34. lmcache/cli/commands/trace/dispatch.py +249 -0
  35. lmcache/cli/commands/trace/driver.py +372 -0
  36. lmcache/cli/commands/trace/stats.py +289 -0
  37. lmcache/cli/documents/lmcache.txt +11 -0
  38. lmcache/cli/main.py +42 -0
  39. lmcache/cli/metrics/__init__.py +29 -0
  40. lmcache/cli/metrics/formatter.py +171 -0
  41. lmcache/cli/metrics/handler.py +94 -0
  42. lmcache/cli/metrics/metrics.py +161 -0
  43. lmcache/cli/metrics/section.py +77 -0
  44. lmcache/connections.py +173 -0
  45. lmcache/integration/__init__.py +2 -0
  46. lmcache/integration/base_service_factory.py +165 -0
  47. lmcache/integration/request_telemetry/__init__.py +1 -0
  48. lmcache/integration/request_telemetry/base.py +51 -0
  49. lmcache/integration/request_telemetry/factory.py +113 -0
  50. lmcache/integration/request_telemetry/fastapi.py +109 -0
  51. lmcache/integration/request_telemetry/noop.py +35 -0
  52. lmcache/integration/sglang/__init__.py +2 -0
  53. lmcache/integration/sglang/sglang_adapter.py +326 -0
  54. lmcache/integration/sglang/utils.py +39 -0
  55. lmcache/integration/vllm/__init__.py +1 -0
  56. lmcache/integration/vllm/lmcache_connector_v1.py +213 -0
  57. lmcache/integration/vllm/lmcache_connector_v1_085.py +150 -0
  58. lmcache/integration/vllm/lmcache_mp_connector_0180.py +1072 -0
  59. lmcache/integration/vllm/tests/test_mm_hash_utils.py +112 -0
  60. lmcache/integration/vllm/utils.py +433 -0
  61. lmcache/integration/vllm/vllm_multi_process_adapter.py +1090 -0
  62. lmcache/integration/vllm/vllm_service_factory.py +339 -0
  63. lmcache/integration/vllm/vllm_v1_adapter.py +1713 -0
  64. lmcache/logging.py +107 -0
  65. lmcache/native_storage_ops.pyi +230 -0
  66. lmcache/non_cuda_equivalents.py +1424 -0
  67. lmcache/observability.py +1958 -0
  68. lmcache/storage_backend/serde/__init__.py +1 -0
  69. lmcache/storage_backend/serde/cachegen_basics.py +210 -0
  70. lmcache/storage_backend/serde/cachegen_decoder.py +207 -0
  71. lmcache/storage_backend/serde/cachegen_encoder.py +394 -0
  72. lmcache/storage_backend/serde/serde.py +75 -0
  73. lmcache/tools/__init__.py +1 -0
  74. lmcache/tools/cache_simulator/README.md +392 -0
  75. lmcache/tools/cache_simulator/__init__.py +1 -0
  76. lmcache/tools/cache_simulator/docs/simulate_example.png +0 -0
  77. lmcache/tools/cache_simulator/docs/sweep_example.png +0 -0
  78. lmcache/tools/cache_simulator/gen_bench_dataset.py +360 -0
  79. lmcache/tools/cache_simulator/lru_cache.py +124 -0
  80. lmcache/tools/cache_simulator/plot_hit_rate.py +231 -0
  81. lmcache/tools/cache_simulator/simulator.py +795 -0
  82. lmcache/tools/controller_benchmark/README.md +161 -0
  83. lmcache/tools/controller_benchmark/__init__.py +1 -0
  84. lmcache/tools/controller_benchmark/__main__.py +331 -0
  85. lmcache/tools/controller_benchmark/benchmark.py +660 -0
  86. lmcache/tools/controller_benchmark/config.py +44 -0
  87. lmcache/tools/controller_benchmark/constants.py +10 -0
  88. lmcache/tools/controller_benchmark/handlers/__init__.py +46 -0
  89. lmcache/tools/controller_benchmark/handlers/admit.py +52 -0
  90. lmcache/tools/controller_benchmark/handlers/base.py +47 -0
  91. lmcache/tools/controller_benchmark/handlers/deregister.py +49 -0
  92. lmcache/tools/controller_benchmark/handlers/evict.py +52 -0
  93. lmcache/tools/controller_benchmark/handlers/heartbeat.py +56 -0
  94. lmcache/tools/controller_benchmark/handlers/p2p_lookup.py +47 -0
  95. lmcache/tools/controller_benchmark/handlers/register.py +56 -0
  96. lmcache/tools/mp_status_viewer/__init__.py +1 -0
  97. lmcache/tools/mp_status_viewer/__main__.py +95 -0
  98. lmcache/usage_context.py +417 -0
  99. lmcache/utils.py +665 -0
  100. lmcache/v1/__init__.py +2 -0
  101. lmcache/v1/api_server/__init__.py +2 -0
  102. lmcache/v1/api_server/__main__.py +537 -0
  103. lmcache/v1/basic_check.py +112 -0
  104. lmcache/v1/cache_controller/__init__.py +9 -0
  105. lmcache/v1/cache_controller/commands/__init__.py +15 -0
  106. lmcache/v1/cache_controller/commands/base.py +35 -0
  107. lmcache/v1/cache_controller/commands/full_sync.py +49 -0
  108. lmcache/v1/cache_controller/config.py +176 -0
  109. lmcache/v1/cache_controller/controller_manager.py +535 -0
  110. lmcache/v1/cache_controller/controllers/__init__.py +11 -0
  111. lmcache/v1/cache_controller/controllers/full_sync_tracker.py +473 -0
  112. lmcache/v1/cache_controller/controllers/kv_controller.py +439 -0
  113. lmcache/v1/cache_controller/controllers/registration_controller.py +282 -0
  114. lmcache/v1/cache_controller/executor.py +463 -0
  115. lmcache/v1/cache_controller/frontend/static/css/style.css +201 -0
  116. lmcache/v1/cache_controller/frontend/static/img/logo.png +0 -0
  117. lmcache/v1/cache_controller/frontend/static/index.html +234 -0
  118. lmcache/v1/cache_controller/frontend/static/js/controller_app.js +660 -0
  119. lmcache/v1/cache_controller/full_sync_sender.py +475 -0
  120. lmcache/v1/cache_controller/locks.py +149 -0
  121. lmcache/v1/cache_controller/message.py +828 -0
  122. lmcache/v1/cache_controller/observability.py +208 -0
  123. lmcache/v1/cache_controller/utils.py +679 -0
  124. lmcache/v1/cache_controller/worker.py +665 -0
  125. lmcache/v1/cache_engine.py +2058 -0
  126. lmcache/v1/cache_interface.py +19 -0
  127. lmcache/v1/check/__init__.py +74 -0
  128. lmcache/v1/check/check_mode_gen.py +86 -0
  129. lmcache/v1/check/check_mode_test_l2_adapter.py +284 -0
  130. lmcache/v1/check/check_mode_test_remote.py +155 -0
  131. lmcache/v1/check/check_mode_test_storage_manager.py +142 -0
  132. lmcache/v1/check/utils.py +571 -0
  133. lmcache/v1/compute/__init__.py +2 -0
  134. lmcache/v1/compute/attention/__init__.py +0 -0
  135. lmcache/v1/compute/attention/abstract.py +39 -0
  136. lmcache/v1/compute/attention/flash_attn.py +129 -0
  137. lmcache/v1/compute/attention/flash_infer_sparse.py +284 -0
  138. lmcache/v1/compute/attention/metadata.py +85 -0
  139. lmcache/v1/compute/attention/utils.py +14 -0
  140. lmcache/v1/compute/blend/__init__.py +7 -0
  141. lmcache/v1/compute/blend/blender.py +168 -0
  142. lmcache/v1/compute/blend/metadata.py +34 -0
  143. lmcache/v1/compute/blend/utils.py +63 -0
  144. lmcache/v1/compute/models/__init__.py +0 -0
  145. lmcache/v1/compute/models/base.py +141 -0
  146. lmcache/v1/compute/models/llama.py +9 -0
  147. lmcache/v1/compute/models/qwen3.py +24 -0
  148. lmcache/v1/compute/models/utils.py +68 -0
  149. lmcache/v1/compute/positional_encoding.py +199 -0
  150. lmcache/v1/config.py +848 -0
  151. lmcache/v1/config_base.py +848 -0
  152. lmcache/v1/distributed/api.py +248 -0
  153. lmcache/v1/distributed/config.py +321 -0
  154. lmcache/v1/distributed/error.py +64 -0
  155. lmcache/v1/distributed/eviction.py +192 -0
  156. lmcache/v1/distributed/eviction_policy/__init__.py +21 -0
  157. lmcache/v1/distributed/eviction_policy/factory.py +27 -0
  158. lmcache/v1/distributed/eviction_policy/lru.py +244 -0
  159. lmcache/v1/distributed/eviction_policy/noop.py +50 -0
  160. lmcache/v1/distributed/internal_api.py +170 -0
  161. lmcache/v1/distributed/l1_manager.py +835 -0
  162. lmcache/v1/distributed/l2_adapters/__init__.py +67 -0
  163. lmcache/v1/distributed/l2_adapters/base.py +360 -0
  164. lmcache/v1/distributed/l2_adapters/config.py +385 -0
  165. lmcache/v1/distributed/l2_adapters/factory.py +205 -0
  166. lmcache/v1/distributed/l2_adapters/fs_l2_adapter.py +747 -0
  167. lmcache/v1/distributed/l2_adapters/fs_native_l2_adapter.py +167 -0
  168. lmcache/v1/distributed/l2_adapters/mock_l2_adapter.py +516 -0
  169. lmcache/v1/distributed/l2_adapters/mooncake_store_l2_adapter.py +135 -0
  170. lmcache/v1/distributed/l2_adapters/native_connector_l2_adapter.py +468 -0
  171. lmcache/v1/distributed/l2_adapters/native_plugin_l2_adapter.py +199 -0
  172. lmcache/v1/distributed/l2_adapters/nixl_store_dynamic_l2_adapter.py +831 -0
  173. lmcache/v1/distributed/l2_adapters/nixl_store_l2_adapter.py +983 -0
  174. lmcache/v1/distributed/l2_adapters/plugin_l2_adapter.py +210 -0
  175. lmcache/v1/distributed/l2_adapters/resp_l2_adapter.py +176 -0
  176. lmcache/v1/distributed/memory_manager.py +179 -0
  177. lmcache/v1/distributed/storage_controller.py +39 -0
  178. lmcache/v1/distributed/storage_controllers/__init__.py +43 -0
  179. lmcache/v1/distributed/storage_controllers/eviction_controller.py +242 -0
  180. lmcache/v1/distributed/storage_controllers/prefetch_controller.py +830 -0
  181. lmcache/v1/distributed/storage_controllers/prefetch_policy.py +193 -0
  182. lmcache/v1/distributed/storage_controllers/store_controller.py +452 -0
  183. lmcache/v1/distributed/storage_controllers/store_policy.py +213 -0
  184. lmcache/v1/distributed/storage_manager.py +532 -0
  185. lmcache/v1/event_manager.py +145 -0
  186. lmcache/v1/exceptions/__init__.py +16 -0
  187. lmcache/v1/gpu_connector/__init__.py +126 -0
  188. lmcache/v1/gpu_connector/gpu_connectors.py +1906 -0
  189. lmcache/v1/gpu_connector/gpu_ops.py +85 -0
  190. lmcache/v1/gpu_connector/hpu_connector.py +326 -0
  191. lmcache/v1/gpu_connector/mock_gpu_connector.py +67 -0
  192. lmcache/v1/gpu_connector/utils.py +890 -0
  193. lmcache/v1/gpu_connector/xpu_connectors.py +916 -0
  194. lmcache/v1/health_monitor/__init__.py +1 -0
  195. lmcache/v1/health_monitor/base.py +587 -0
  196. lmcache/v1/health_monitor/checks/__init__.py +1 -0
  197. lmcache/v1/health_monitor/checks/remote_backend_check.py +304 -0
  198. lmcache/v1/health_monitor/constants.py +36 -0
  199. lmcache/v1/internal_api_server/__init__.py +0 -0
  200. lmcache/v1/internal_api_server/api_registry.py +59 -0
  201. lmcache/v1/internal_api_server/api_server.py +120 -0
  202. lmcache/v1/internal_api_server/common/__init__.py +1 -0
  203. lmcache/v1/internal_api_server/common/env_api.py +22 -0
  204. lmcache/v1/internal_api_server/common/loglevel_api.py +57 -0
  205. lmcache/v1/internal_api_server/common/metrics_api.py +29 -0
  206. lmcache/v1/internal_api_server/common/periodic_thread_api.py +138 -0
  207. lmcache/v1/internal_api_server/common/run_script_api.py +73 -0
  208. lmcache/v1/internal_api_server/common/thread_api.py +63 -0
  209. lmcache/v1/internal_api_server/controller/__init__.py +1 -0
  210. lmcache/v1/internal_api_server/controller/key_stats_api.py +81 -0
  211. lmcache/v1/internal_api_server/controller/worker_info_api.py +136 -0
  212. lmcache/v1/internal_api_server/utils.py +43 -0
  213. lmcache/v1/internal_api_server/vllm/__init__.py +1 -0
  214. lmcache/v1/internal_api_server/vllm/backend_api.py +221 -0
  215. lmcache/v1/internal_api_server/vllm/bypass_api.py +204 -0
  216. lmcache/v1/internal_api_server/vllm/cache_api.py +895 -0
  217. lmcache/v1/internal_api_server/vllm/chunk_statistics_api.py +141 -0
  218. lmcache/v1/internal_api_server/vllm/conf_api.py +147 -0
  219. lmcache/v1/internal_api_server/vllm/freeze_api.py +172 -0
  220. lmcache/v1/internal_api_server/vllm/hot_cache_api.py +184 -0
  221. lmcache/v1/internal_api_server/vllm/inference_api.py +65 -0
  222. lmcache/v1/internal_api_server/vllm/load_fs_chunks_api.py +320 -0
  223. lmcache/v1/internal_api_server/vllm/lookup_api.py +145 -0
  224. lmcache/v1/internal_api_server/vllm/version_api.py +25 -0
  225. lmcache/v1/kv_layer_groups.py +267 -0
  226. lmcache/v1/lazy_memory_allocator.py +284 -0
  227. lmcache/v1/lookup_client/__init__.py +25 -0
  228. lmcache/v1/lookup_client/abstract_client.py +77 -0
  229. lmcache/v1/lookup_client/async_lookup_message.py +50 -0
  230. lmcache/v1/lookup_client/chunk_statistics_lookup_client.py +200 -0
  231. lmcache/v1/lookup_client/factory.py +251 -0
  232. lmcache/v1/lookup_client/hit_limit_lookup_client.py +86 -0
  233. lmcache/v1/lookup_client/lmcache_async_lookup_client.py +407 -0
  234. lmcache/v1/lookup_client/lmcache_lookup_client.py +285 -0
  235. lmcache/v1/lookup_client/lmcache_lookup_client_bypass.py +99 -0
  236. lmcache/v1/lookup_client/mooncake_lookup_client.py +87 -0
  237. lmcache/v1/lookup_client/record_strategies/__init__.py +77 -0
  238. lmcache/v1/lookup_client/record_strategies/base.py +327 -0
  239. lmcache/v1/lookup_client/record_strategies/file_hash.py +130 -0
  240. lmcache/v1/lookup_client/record_strategies/memory_bloom_filter.py +81 -0
  241. lmcache/v1/manager.py +539 -0
  242. lmcache/v1/memory_management.py +2619 -0
  243. lmcache/v1/metadata.py +114 -0
  244. lmcache/v1/mp_observability/AGENTS.override.md +21 -0
  245. lmcache/v1/mp_observability/README.md +204 -0
  246. lmcache/v1/mp_observability/config.py +340 -0
  247. lmcache/v1/mp_observability/event.py +100 -0
  248. lmcache/v1/mp_observability/event_bus.py +313 -0
  249. lmcache/v1/mp_observability/otel_init.py +145 -0
  250. lmcache/v1/mp_observability/subscribers/__init__.py +28 -0
  251. lmcache/v1/mp_observability/subscribers/logging/__init__.py +19 -0
  252. lmcache/v1/mp_observability/subscribers/logging/l1.py +56 -0
  253. lmcache/v1/mp_observability/subscribers/logging/l2.py +73 -0
  254. lmcache/v1/mp_observability/subscribers/logging/lookup_hash.py +209 -0
  255. lmcache/v1/mp_observability/subscribers/logging/mp_server.py +90 -0
  256. lmcache/v1/mp_observability/subscribers/logging/sm.py +59 -0
  257. lmcache/v1/mp_observability/subscribers/metrics/__init__.py +20 -0
  258. lmcache/v1/mp_observability/subscribers/metrics/l0_lifecycle.py +290 -0
  259. lmcache/v1/mp_observability/subscribers/metrics/l1.py +55 -0
  260. lmcache/v1/mp_observability/subscribers/metrics/l1_lifecycle.py +166 -0
  261. lmcache/v1/mp_observability/subscribers/metrics/l2.py +121 -0
  262. lmcache/v1/mp_observability/subscribers/metrics/sm.py +69 -0
  263. lmcache/v1/mp_observability/subscribers/tracing/__init__.py +12 -0
  264. lmcache/v1/mp_observability/subscribers/tracing/mp_server.py +333 -0
  265. lmcache/v1/mp_observability/subscribers/tracing/span_registry.py +148 -0
  266. lmcache/v1/mp_observability/trace/__init__.py +50 -0
  267. lmcache/v1/mp_observability/trace/codecs.py +255 -0
  268. lmcache/v1/mp_observability/trace/decorator.py +147 -0
  269. lmcache/v1/mp_observability/trace/format.py +132 -0
  270. lmcache/v1/mp_observability/trace/lifecycle.py +83 -0
  271. lmcache/v1/mp_observability/trace/reader.py +167 -0
  272. lmcache/v1/mp_observability/trace/recorder.py +300 -0
  273. lmcache/v1/multiprocess/__init__.py +0 -0
  274. lmcache/v1/multiprocess/affinity_pool.py +102 -0
  275. lmcache/v1/multiprocess/blend_server_v2.py +891 -0
  276. lmcache/v1/multiprocess/config.py +253 -0
  277. lmcache/v1/multiprocess/custom_types.py +281 -0
  278. lmcache/v1/multiprocess/futures.py +194 -0
  279. lmcache/v1/multiprocess/gpu_context.py +511 -0
  280. lmcache/v1/multiprocess/http_server.py +235 -0
  281. lmcache/v1/multiprocess/mp_runtime_plugin_launcher.py +130 -0
  282. lmcache/v1/multiprocess/mq.py +732 -0
  283. lmcache/v1/multiprocess/protocol.py +86 -0
  284. lmcache/v1/multiprocess/protocols/README.md +213 -0
  285. lmcache/v1/multiprocess/protocols/__init__.py +127 -0
  286. lmcache/v1/multiprocess/protocols/base.py +89 -0
  287. lmcache/v1/multiprocess/protocols/blend.py +109 -0
  288. lmcache/v1/multiprocess/protocols/blend_v2.py +57 -0
  289. lmcache/v1/multiprocess/protocols/controller.py +53 -0
  290. lmcache/v1/multiprocess/protocols/debug.py +34 -0
  291. lmcache/v1/multiprocess/protocols/engine.py +146 -0
  292. lmcache/v1/multiprocess/protocols/observability.py +39 -0
  293. lmcache/v1/multiprocess/server.py +1134 -0
  294. lmcache/v1/multiprocess/session.py +190 -0
  295. lmcache/v1/multiprocess/token_hasher.py +441 -0
  296. lmcache/v1/offload_server/__init__.py +17 -0
  297. lmcache/v1/offload_server/abstract_server.py +37 -0
  298. lmcache/v1/offload_server/message.py +30 -0
  299. lmcache/v1/offload_server/zmq_server.py +122 -0
  300. lmcache/v1/periodic_thread.py +579 -0
  301. lmcache/v1/pin_monitor.py +246 -0
  302. lmcache/v1/plugin/__init__.py +0 -0
  303. lmcache/v1/plugin/runtime_plugin_launcher.py +211 -0
  304. lmcache/v1/protocol.py +317 -0
  305. lmcache/v1/rpc/__init__.py +17 -0
  306. lmcache/v1/rpc/transport.py +105 -0
  307. lmcache/v1/rpc/zmq_transport.py +213 -0
  308. lmcache/v1/rpc_utils.py +165 -0
  309. lmcache/v1/server/__init__.py +2 -0
  310. lmcache/v1/server/__main__.py +170 -0
  311. lmcache/v1/server/storage_backend/__init__.py +21 -0
  312. lmcache/v1/server/storage_backend/abstract_backend.py +80 -0
  313. lmcache/v1/server/storage_backend/local_backend.py +75 -0
  314. lmcache/v1/server/utils.py +21 -0
  315. lmcache/v1/standalone/__init__.py +1 -0
  316. lmcache/v1/standalone/__main__.py +583 -0
  317. lmcache/v1/standalone/manager.py +80 -0
  318. lmcache/v1/standalone/standalone_service_factory.py +86 -0
  319. lmcache/v1/storage_backend/__init__.py +313 -0
  320. lmcache/v1/storage_backend/abstract_backend.py +445 -0
  321. lmcache/v1/storage_backend/audit_backend.py +233 -0
  322. lmcache/v1/storage_backend/batched_message_sender.py +222 -0
  323. lmcache/v1/storage_backend/cache_policy/__init__.py +45 -0
  324. lmcache/v1/storage_backend/cache_policy/base_policy.py +87 -0
  325. lmcache/v1/storage_backend/cache_policy/fifo.py +58 -0
  326. lmcache/v1/storage_backend/cache_policy/lfu.py +105 -0
  327. lmcache/v1/storage_backend/cache_policy/lru.py +81 -0
  328. lmcache/v1/storage_backend/cache_policy/mru.py +61 -0
  329. lmcache/v1/storage_backend/connector/__init__.py +443 -0
  330. lmcache/v1/storage_backend/connector/audit_adapter.py +77 -0
  331. lmcache/v1/storage_backend/connector/audit_connector.py +320 -0
  332. lmcache/v1/storage_backend/connector/base_connector.py +379 -0
  333. lmcache/v1/storage_backend/connector/blackhole_adapter.py +21 -0
  334. lmcache/v1/storage_backend/connector/blackhole_connector.py +37 -0
  335. lmcache/v1/storage_backend/connector/eic_adapter.py +31 -0
  336. lmcache/v1/storage_backend/connector/eic_connector.py +757 -0
  337. lmcache/v1/storage_backend/connector/external_adapter.py +79 -0
  338. lmcache/v1/storage_backend/connector/fs_adapter.py +51 -0
  339. lmcache/v1/storage_backend/connector/fs_connector.py +403 -0
  340. lmcache/v1/storage_backend/connector/infinistore_adapter.py +56 -0
  341. lmcache/v1/storage_backend/connector/infinistore_connector.py +177 -0
  342. lmcache/v1/storage_backend/connector/instrumented_connector.py +219 -0
  343. lmcache/v1/storage_backend/connector/lm_adapter.py +31 -0
  344. lmcache/v1/storage_backend/connector/lm_connector.py +176 -0
  345. lmcache/v1/storage_backend/connector/mock_adapter.py +57 -0
  346. lmcache/v1/storage_backend/connector/mock_connector.py +349 -0
  347. lmcache/v1/storage_backend/connector/mooncakestore_adapter.py +43 -0
  348. lmcache/v1/storage_backend/connector/mooncakestore_connector.py +614 -0
  349. lmcache/v1/storage_backend/connector/redis_adapter.py +181 -0
  350. lmcache/v1/storage_backend/connector/redis_connector.py +828 -0
  351. lmcache/v1/storage_backend/connector/s3_adapter.py +59 -0
  352. lmcache/v1/storage_backend/connector/s3_connector.py +699 -0
  353. lmcache/v1/storage_backend/connector/sagemaker_hyperpod_adapter.py +233 -0
  354. lmcache/v1/storage_backend/connector/sagemaker_hyperpod_connector.py +987 -0
  355. lmcache/v1/storage_backend/connector/valkey_adapter.py +114 -0
  356. lmcache/v1/storage_backend/connector/valkey_connector.py +627 -0
  357. lmcache/v1/storage_backend/gds_backend.py +1199 -0
  358. lmcache/v1/storage_backend/job_executor/__init__.py +0 -0
  359. lmcache/v1/storage_backend/job_executor/base_executor.py +34 -0
  360. lmcache/v1/storage_backend/job_executor/pq_executor.py +235 -0
  361. lmcache/v1/storage_backend/local_cpu_backend.py +810 -0
  362. lmcache/v1/storage_backend/local_disk_backend.py +656 -0
  363. lmcache/v1/storage_backend/maru_backend.py +734 -0
  364. lmcache/v1/storage_backend/naive_serde/__init__.py +50 -0
  365. lmcache/v1/storage_backend/naive_serde/cachegen_basics.py +133 -0
  366. lmcache/v1/storage_backend/naive_serde/cachegen_decoder.py +135 -0
  367. lmcache/v1/storage_backend/naive_serde/cachegen_encoder.py +83 -0
  368. lmcache/v1/storage_backend/naive_serde/kivi_serde.py +22 -0
  369. lmcache/v1/storage_backend/naive_serde/naive_serde.py +18 -0
  370. lmcache/v1/storage_backend/naive_serde/serde.py +37 -0
  371. lmcache/v1/storage_backend/native_clients/connector_client_base.py +165 -0
  372. lmcache/v1/storage_backend/native_clients/resp_client.py +35 -0
  373. lmcache/v1/storage_backend/nixl_storage_backend.py +1400 -0
  374. lmcache/v1/storage_backend/p2p_backend.py +788 -0
  375. lmcache/v1/storage_backend/path_sharder.py +117 -0
  376. lmcache/v1/storage_backend/pd_backend.py +646 -0
  377. lmcache/v1/storage_backend/plugins/dax_backend.py +1443 -0
  378. lmcache/v1/storage_backend/plugins/rust_raw_block_backend.py +1361 -0
  379. lmcache/v1/storage_backend/remote_backend.py +624 -0
  380. lmcache/v1/storage_backend/resp_client.py +227 -0
  381. lmcache/v1/storage_backend/storage_backend_listener.py +19 -0
  382. lmcache/v1/storage_backend/storage_manager.py +1352 -0
  383. lmcache/v1/system_detection.py +110 -0
  384. lmcache/v1/token_database.py +551 -0
  385. lmcache/v1/transfer_channel/__init__.py +83 -0
  386. lmcache/v1/transfer_channel/abstract.py +285 -0
  387. lmcache/v1/transfer_channel/mock_memory_channel.py +156 -0
  388. lmcache/v1/transfer_channel/nixl_channel.py +639 -0
  389. lmcache/v1/transfer_channel/py_socket_channel.py +260 -0
  390. lmcache/v1/transfer_channel/transfer_utils.py +63 -0
  391. lmcache/v1/utils/__init__.py +1 -0
  392. lmcache/v1/utils/bloom_filter.py +109 -0
  393. lmcache/v1/utils/cache_utils.py +125 -0
  394. lmcache_cli-0.4.5.dev0.dist-info/METADATA +185 -0
  395. lmcache_cli-0.4.5.dev0.dist-info/RECORD +399 -0
  396. lmcache_cli-0.4.5.dev0.dist-info/WHEEL +5 -0
  397. lmcache_cli-0.4.5.dev0.dist-info/entry_points.txt +2 -0
  398. lmcache_cli-0.4.5.dev0.dist-info/licenses/LICENSE +201 -0
  399. lmcache_cli-0.4.5.dev0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,190 @@
1
+ # SPDX-License-Identifier: Apache-2.0
2
+ """
3
+ Session and SessionManager for tracking per-request state
4
+ in the multiprocess cache server.
5
+ """
6
+
7
+ # Standard
8
+ from dataclasses import dataclass, field
9
+ from typing import Any, Optional, overload
10
+ import threading
11
+ import time
12
+
13
+ # First Party
14
+ from lmcache.logging import init_logger
15
+ from lmcache.v1.multiprocess.custom_types import IPCCacheEngineKey
16
+ from lmcache.v1.multiprocess.token_hasher import TokenHasher
17
+
18
+ logger = init_logger(__name__)
19
+
20
+
21
+ @dataclass
22
+ class Session:
23
+ """Tracks accumulated token IDs and computed chunk hashes for a request.
24
+
25
+ Thread-safe: all public methods are protected by an internal lock
26
+ to allow concurrent access from multiple TP worker threads.
27
+ """
28
+
29
+ request_id: str
30
+ hasher: TokenHasher
31
+ token_ids: list[int] = field(default_factory=list)
32
+ chunk_hashes: list = field(default_factory=list)
33
+ last_prefix_hash: Any = None
34
+ num_chunks_processed: int = 0
35
+ created_at: float = field(default_factory=time.time)
36
+ lookup_ipc_key: Optional[IPCCacheEngineKey] = None
37
+ _lock: threading.Lock = field(default_factory=threading.Lock, repr=False)
38
+
39
+ def set_tokens(self, full_token_ids: list[int]) -> None:
40
+ """Update the token sequence (idempotent, replaces not extends).
41
+
42
+ Args:
43
+ full_token_ids: Complete token sequence.
44
+ """
45
+ with self._lock:
46
+ self.token_ids = full_token_ids
47
+
48
+ @overload
49
+ def get_hashes(self, start: int, end: int) -> list: ...
50
+
51
+ @overload
52
+ def get_hashes(self, start: int) -> list: ...
53
+
54
+ def get_hashes(self, start: int, end: int | None = None) -> list:
55
+ """Compute and return chunk hashes for the [start, end) token range.
56
+
57
+ Internally computes rolling hashes up to end_chunk, skipping
58
+ already-computed chunks.
59
+
60
+ Two calling conventions are supported (declared via ``@overload``)::
61
+
62
+ get_hashes(start, end) # explicit end
63
+ get_hashes(start) # end = last full-chunk boundary
64
+
65
+ Args:
66
+ start: Start token index (must be aligned to chunk_size).
67
+ end: End token index (must be aligned to chunk_size).
68
+ When omitted (``None``), automatically set to the last
69
+ full-chunk boundary of the current token sequence.
70
+
71
+ Returns:
72
+ List of hash values for chunks in [start_chunk, end_chunk).
73
+ """
74
+ chunk_size = self.hasher.chunk_size
75
+ assert start % chunk_size == 0, (
76
+ f"start ({start}) must be a multiple of chunk_size ({chunk_size})"
77
+ )
78
+ start_chunk = start // chunk_size
79
+
80
+ with self._lock:
81
+ if end is None:
82
+ # No explicit end: use the last full-chunk boundary.
83
+ # Lock must be held here because `self.token_ids` may be
84
+ # concurrently replaced by `set_tokens` from another thread.
85
+ end = len(self.token_ids) - (len(self.token_ids) % chunk_size)
86
+ assert end % chunk_size == 0, (
87
+ f"end ({end}) must be a multiple of chunk_size ({chunk_size})"
88
+ )
89
+ end_chunk = end // chunk_size
90
+ self._compute_hash(end_chunk)
91
+ return self.chunk_hashes[start_chunk:end_chunk]
92
+
93
+ def _compute_hash(self, end_chunk: int) -> None:
94
+ """Compute rolling hashes up to end_chunk.
95
+
96
+ Uses cached state to skip already-computed chunks.
97
+
98
+ Args:
99
+ end_chunk: Compute hashes up to (but not including) this chunk.
100
+ """
101
+ chunk_size = self.hasher.chunk_size
102
+
103
+ while self.num_chunks_processed < end_chunk:
104
+ cs = self.num_chunks_processed * chunk_size
105
+ ce = cs + chunk_size
106
+ chunk = self.token_ids[cs:ce]
107
+
108
+ prefix = (
109
+ self.last_prefix_hash
110
+ if self.last_prefix_hash is not None
111
+ else self.hasher.none_hash
112
+ )
113
+ h = self.hasher.hash_tokens(chunk, prefix)
114
+ self.last_prefix_hash = h
115
+ self.chunk_hashes.append(h)
116
+ self.num_chunks_processed += 1
117
+
118
+
119
+ class SessionManager:
120
+ """Thread-safe manager for per-request sessions."""
121
+
122
+ DEFAULT_SESSION_TTL = 600 # 10 minutes
123
+
124
+ def __init__(self, hasher: TokenHasher, ttl: float = DEFAULT_SESSION_TTL):
125
+ self._hasher = hasher
126
+ self._ttl = ttl
127
+ self._sessions: dict[str, Session] = {}
128
+ self._lock = threading.Lock()
129
+
130
+ def get_or_create(self, request_id: str) -> Session:
131
+ """Get existing session or create a new one.
132
+
133
+ Args:
134
+ request_id: Unique request identifier.
135
+
136
+ Returns:
137
+ The Session for this request_id.
138
+ """
139
+ with self._lock:
140
+ if request_id not in self._sessions:
141
+ self._sessions[request_id] = Session(
142
+ request_id=request_id, hasher=self._hasher
143
+ )
144
+ logger.debug("Created session for request_id=%s", request_id)
145
+ return self._sessions[request_id]
146
+
147
+ def remove(self, request_id: str) -> Optional[Session]:
148
+ """Remove a session by request_id.
149
+
150
+ Args:
151
+ request_id: Unique request identifier.
152
+
153
+ Returns:
154
+ The removed session, or None if no session was found.
155
+ """
156
+ with self._lock:
157
+ if request_id in self._sessions:
158
+ session = self._sessions[request_id]
159
+ del self._sessions[request_id]
160
+ logger.debug("Removed session for request_id=%s", request_id)
161
+ return session
162
+ return None
163
+
164
+ def cleanup_expired(self) -> int:
165
+ """Remove sessions that have exceeded their TTL.
166
+
167
+ Returns:
168
+ Number of sessions removed.
169
+ """
170
+ now = time.time()
171
+ expired = []
172
+ with self._lock:
173
+ for rid, session in self._sessions.items():
174
+ if now - session.created_at > self._ttl:
175
+ expired.append(rid)
176
+ for rid in expired:
177
+ del self._sessions[rid]
178
+
179
+ if expired:
180
+ logger.info("Cleaned up %d expired sessions", len(expired))
181
+ return len(expired)
182
+
183
+ def active_count(self) -> int:
184
+ """Return the number of active sessions.
185
+
186
+ Returns:
187
+ Number of currently tracked sessions.
188
+ """
189
+ with self._lock:
190
+ return len(self._sessions)
@@ -0,0 +1,441 @@
1
+ # SPDX-License-Identifier: Apache-2.0
2
+ """
3
+ TokenHasher: Standalone hash computation for the multiprocess server.
4
+
5
+ Hash function loading logic is adapted from token_database.py to avoid
6
+ coupling with TokenDatabase's config/metadata dependencies.
7
+
8
+ vLLM compatibility notes:
9
+ - PR#20511: Introduced kv_cache_utils.init_none_hash()
10
+ - PR#23673: Renamed sha256_cbor_64bit to sha256_cbor
11
+ - PR#27151: Moved hash functions to vllm.utils.hashing module
12
+ """
13
+
14
+ # Standard
15
+ from typing import Any, Callable
16
+ import os
17
+
18
+ # Third Party
19
+ from numba import njit
20
+ import numpy as np
21
+
22
+ # First Party
23
+ from lmcache.logging import init_logger
24
+
25
+ logger = init_logger(__name__)
26
+
27
+
28
+ def _make_blake3_hash_func() -> Callable:
29
+ """Create a blake3-based hash function compatible with the
30
+ (prefix_hash, tuple(tokens), None) calling convention."""
31
+ # Standard
32
+ import struct
33
+
34
+ # Third Party
35
+ import blake3 as _blake3
36
+
37
+ def blake3_hash(args):
38
+ prefix_hash, tokens, _ = args
39
+ h = _blake3.blake3()
40
+ # Serialize prefix hash
41
+ if isinstance(prefix_hash, bytes):
42
+ h.update(prefix_hash)
43
+ elif isinstance(prefix_hash, int):
44
+ h.update(prefix_hash.to_bytes(8, byteorder="big", signed=True))
45
+ else:
46
+ h.update(bytes(prefix_hash))
47
+ # Serialize token IDs in one batch
48
+ h.update(struct.pack(f">{len(tokens)}I", *tokens))
49
+ return h.digest() # 32 bytes
50
+
51
+ return blake3_hash
52
+
53
+
54
+ class TokenHasher:
55
+ """Computes rolling prefix hashes for token chunks.
56
+
57
+ This class encapsulates the hash function loading and hash computation
58
+ logic needed by the multiprocess server to convert token IDs into
59
+ chunk hashes compatible with IPCCacheEngineKey (hash mode).
60
+ """
61
+
62
+ def __init__(self, chunk_size: int = 256, hash_algorithm: str = "blake3"):
63
+ self.chunk_size = chunk_size
64
+ self.hash_algorithm_name = hash_algorithm
65
+ self.hash_func = self._get_hash_func(hash_algorithm)
66
+ self.none_hash = self._init_none_hash()
67
+ logger.info(
68
+ "TokenHasher initialized: chunk_size=%d, hash_algorithm=%s",
69
+ chunk_size,
70
+ hash_algorithm,
71
+ )
72
+
73
+ def _get_hash_func(self, hash_algorithm: str) -> Callable:
74
+ """Load hash function with vLLM version compatibility.
75
+
76
+ Adapted from TokenDatabase._get_vllm_hash_func (token_database.py:97-168).
77
+ """
78
+ if hash_algorithm == "blake3":
79
+ logger.info("Using blake3 hash function")
80
+ return _make_blake3_hash_func()
81
+
82
+ # Try get_hash_fn_by_name from both locations (PR#27151)
83
+ for module_path in ["vllm.utils.hashing", "vllm.utils"]:
84
+ try:
85
+ module = __import__(module_path, fromlist=["get_hash_fn_by_name"])
86
+ get_hash_fn_by_name = module.get_hash_fn_by_name
87
+ return self._try_get_hash(
88
+ get_hash_fn_by_name, hash_algorithm, module_path
89
+ )
90
+ except (ImportError, AttributeError, ValueError):
91
+ continue
92
+
93
+ # Try direct imports as fallback (for older vLLM versions)
94
+ func_names = (
95
+ ["sha256_cbor", "sha256_cbor_64bit"]
96
+ if hash_algorithm in ("sha256_cbor", "sha256_cbor_64bit")
97
+ else [hash_algorithm]
98
+ )
99
+ for module_path in ["vllm.utils.hashing", "vllm.utils"]:
100
+ for func_name in func_names:
101
+ try:
102
+ module = __import__(module_path, fromlist=[func_name])
103
+ hash_func = getattr(module, func_name)
104
+ logger.info(
105
+ "Loaded '%s' from %s (direct import)", func_name, module_path
106
+ )
107
+ return hash_func
108
+ except (ImportError, AttributeError):
109
+ continue
110
+
111
+ # Fallback to builtin hash
112
+ logger.warning(
113
+ "Could not load '%s' from vLLM. Using builtin hash. "
114
+ "This may cause inconsistencies in distributed caching.",
115
+ hash_algorithm,
116
+ )
117
+
118
+ # Check PYTHONHASHSEED when using builtin hash
119
+ if os.getenv("PYTHONHASHSEED") is None:
120
+ logger.warning(
121
+ "Using builtin hash without PYTHONHASHSEED set. "
122
+ "For production environments (non-testing scenarios), you MUST set "
123
+ "PYTHONHASHSEED to ensure consistent hashing across processes. "
124
+ "Example: export PYTHONHASHSEED=0"
125
+ )
126
+
127
+ return hash
128
+
129
+ def _try_get_hash(
130
+ self, get_hash_fn_by_name: Callable, hash_algorithm: str, module_name: str
131
+ ) -> Callable:
132
+ """Try to get hash function, handling sha256_cbor_64bit rename.
133
+
134
+ Adapted from TokenDatabase._try_get_hash (token_database.py:152-168).
135
+ """
136
+ # Handle sha256_cbor_64bit -> sha256_cbor rename (PR#23673)
137
+ names_to_try = (
138
+ ["sha256_cbor", "sha256_cbor_64bit"]
139
+ if hash_algorithm in ("sha256_cbor", "sha256_cbor_64bit")
140
+ else [hash_algorithm]
141
+ )
142
+
143
+ for name in names_to_try:
144
+ try:
145
+ hash_func = get_hash_fn_by_name(name)
146
+ logger.info("Loaded '%s' from %s", name, module_name)
147
+ return hash_func
148
+ except ValueError:
149
+ continue
150
+ raise ValueError(f"Hash function '{hash_algorithm}' not found in {module_name}")
151
+
152
+ def _init_none_hash(self) -> Any:
153
+ """Initialize NONE_HASH.
154
+
155
+ Adapted from TokenDatabase.__init__ (token_database.py:64-82).
156
+ """
157
+ try:
158
+ # Third Party
159
+ from vllm.v1.core import kv_cache_utils
160
+
161
+ if hasattr(kv_cache_utils, "init_none_hash"):
162
+ kv_cache_utils.init_none_hash(self.hash_func)
163
+ none_hash = kv_cache_utils.NONE_HASH
164
+ logger.info("Initialized NONE_HASH=%s from vLLM", none_hash)
165
+ return none_hash
166
+ except (ImportError, AttributeError, ValueError):
167
+ pass
168
+
169
+ # Fallback: compute none_hash using our hash function
170
+ none_hash = self.hash_func((0, (0,), None))
171
+ logger.info("Computed NONE_HASH=%s using hash function", none_hash)
172
+ return none_hash
173
+
174
+ def hash_tokens(self, tokens: list[int], prefix_hash: Any = None) -> Any:
175
+ """Hash one chunk with rolling prefix.
176
+
177
+ Returns int or bytes depending on hash_func.
178
+ """
179
+ if prefix_hash is None:
180
+ prefix_hash = self.none_hash
181
+ return self.hash_func((prefix_hash, tuple(tokens), None))
182
+
183
+ def compute_chunk_hashes(
184
+ self,
185
+ token_ids: list[int],
186
+ prefix_hash: Any = None,
187
+ start: int = 0,
188
+ end: int | None = None,
189
+ ) -> list[bytes]:
190
+ """Compute rolling prefix hashes for complete chunks in a token range.
191
+
192
+ The rolling hash is always computed from the beginning of
193
+ ``token_ids`` (since each chunk's hash depends on all previous
194
+ chunks), but only hashes for chunks within ``[start, end)`` are
195
+ returned, and hashing stops at ``end`` to avoid unnecessary work.
196
+
197
+ ``start`` and ``end`` are token-level indices and must be
198
+ multiples of ``chunk_size``. Partial chunks are discarded.
199
+
200
+ Args:
201
+ token_ids: Full token sequence.
202
+ prefix_hash: Optional initial prefix hash (defaults to none_hash).
203
+ start: Token-level start index (must be chunk-aligned).
204
+ Chunks before this index are computed but not returned.
205
+ end: Token-level end index (must be chunk-aligned). When
206
+ provided, hashing stops at this index.
207
+
208
+ Returns:
209
+ List of ``bytes`` hash values for chunks in ``[start, end)``.
210
+ """
211
+ hashes: list[bytes] = []
212
+ prefix_hash = self.none_hash if prefix_hash is None else prefix_hash
213
+ effective_len = min(len(token_ids), end) if end is not None else len(token_ids)
214
+ num_complete = effective_len - effective_len % self.chunk_size
215
+ for i in range(0, num_complete, self.chunk_size):
216
+ prefix_hash = self.hash_tokens(
217
+ token_ids[i : i + self.chunk_size], prefix_hash
218
+ )
219
+ if i >= start:
220
+ hashes.append(self.hash_to_bytes(prefix_hash))
221
+ return hashes
222
+
223
+ @staticmethod
224
+ def hash_to_bytes(hash_val: Any) -> bytes:
225
+ """Convert hash value to bytes for ObjectKey.chunk_hash.
226
+
227
+ Handles both bytes (sha256_cbor) and int (builtin hash) return types.
228
+ """
229
+ if isinstance(hash_val, bytes):
230
+ return hash_val # sha256_cbor already returns bytes
231
+ return hash_val.to_bytes(8, byteorder="big", signed=True)
232
+
233
+
234
+ ### Functions for fast rolling/chunk hash and dict lookup
235
+
236
+
237
+ @njit(cache=True)
238
+ def rolling_hash_windows_numba(
239
+ arr_u64: np.ndarray, k: int, base: np.uint64
240
+ ) -> np.ndarray:
241
+ """
242
+ Compute rolling polynomial hashes over a uint64 array.
243
+
244
+ This function computes a polynomial rolling hash over a sliding window
245
+ of size `k` across the input array `arr_u64`. Arithmetic is performed
246
+ in uint64 with natural overflow, which is equivalent to computing
247
+ modulo 2^64.
248
+
249
+ Hash definition for a window [x0, x1, ..., x_{k-1}]:
250
+
251
+ H = x0 * base^(k-1) + x1 * base^(k-2) + ... + x_{k-1}
252
+
253
+ For each subsequent window the hash is updated in O(1):
254
+
255
+ H_new = (H - x_old * base^(k-1)) * base + x_new
256
+
257
+ Parameters
258
+ ----------
259
+ arr_u64 : np.ndarray[np.uint64]
260
+ Input array of integers encoded as uint64 values.
261
+
262
+ k : int
263
+ Sliding window size.
264
+
265
+ base : np.uint64
266
+ Base of the polynomial hash. Typically a random odd 64-bit number.
267
+
268
+ Returns
269
+ -------
270
+ np.ndarray[np.uint64]
271
+ Array of rolling hash values of length:
272
+
273
+ len(arr_u64) - k + 1
274
+
275
+ Each element corresponds to the hash of one window.
276
+ """
277
+ n = arr_u64.shape[0]
278
+ out = np.empty(n - k + 1, dtype=np.uint64)
279
+
280
+ power = np.uint64(1)
281
+ for _ in range(k - 1):
282
+ power = power * base # uint64 overflow = mod 2^64
283
+
284
+ h = np.uint64(0)
285
+ for i in range(k):
286
+ h = h * base + arr_u64[i]
287
+ out[0] = h
288
+
289
+ j = 1
290
+ for i in range(k, n):
291
+ old = arr_u64[i - k]
292
+ new = arr_u64[i]
293
+ h = h - old * power
294
+ h = h * base + new
295
+ out[j] = h
296
+ j += 1
297
+
298
+ return out
299
+
300
+
301
+ @njit(cache=True)
302
+ def chunk_hash_windows_numba(arr_u64, k, base):
303
+ """Compute polynomial hashes over non-overlapping (chunked) windows.
304
+
305
+ Unlike the rolling-hash variant, each window's hash is computed
306
+ independently from scratch, which is efficient when stride equals the
307
+ window size (i.e., windows do not overlap).
308
+
309
+ The hash for a window starting at position ``s`` is:
310
+
311
+ h = arr[s]*base^(k-1) + arr[s+1]*base^(k-2) + ... + arr[s+k-1]
312
+
313
+ computed with natural ``uint64`` overflow (mod 2^64).
314
+
315
+ Parameters
316
+ ----------
317
+ arr_u64 : np.ndarray[np.uint64]
318
+ 1-D array of token values cast to ``uint64``.
319
+ k : int
320
+ Window (chunk) size.
321
+ base : np.uint64
322
+ Base of the polynomial hash.
323
+
324
+ Returns
325
+ -------
326
+ np.ndarray[np.uint64]
327
+ Array of length ``len(arr_u64) // k`` containing one hash per
328
+ non-overlapping chunk. Trailing tokens that do not fill a
329
+ complete chunk are ignored.
330
+ """
331
+ n = arr_u64.shape[0]
332
+ num_windows = n // k
333
+ out = np.empty(num_windows, dtype=np.uint64)
334
+
335
+ for w in range(num_windows):
336
+ h = np.uint64(0)
337
+ start = w * k
338
+ # Compute fresh hash for this block
339
+ for i in range(start, start + k):
340
+ h = h * base + arr_u64[i]
341
+ out[w] = h
342
+ return out
343
+
344
+
345
+ @njit(cache=True)
346
+ def update_table_id_numba(
347
+ hashes_u64: np.ndarray,
348
+ table_id_i64: np.ndarray,
349
+ vals_to_update: np.ndarray,
350
+ ):
351
+ """
352
+ Update the direct-address table with new ID values for given hashes.
353
+
354
+ For each hash in `hashes_u64`, compute the index as:
355
+
356
+ idx = hash & (table_id_i64.size - 1)
357
+
358
+ and update `table_id_i64[idx]` with the corresponding value from
359
+ `vals_to_update`.
360
+
361
+ Parameters
362
+ ----------
363
+ hashes_u64 : np.ndarray[np.uint64]
364
+ Array of hash values to update.
365
+
366
+ table_id_i64 : np.ndarray[np.int64]
367
+ Direct-address lookup table mapping index → ID. This array is
368
+ modified in-place.
369
+
370
+ vals_to_update : np.ndarray[np.int64]
371
+ Array of new ID values to write into the table. Must have the same
372
+ length as `hashes_u64`.
373
+ """
374
+ n = hashes_u64.shape[0]
375
+ m = table_id_i64.shape[0]
376
+
377
+ for i in range(n):
378
+ idx = hashes_u64[i] & (m - 1) # Assuming m is a power of 2
379
+ table_id_i64[idx] = vals_to_update[i]
380
+
381
+
382
+ @njit(cache=True)
383
+ def unique_hits_direct_id_numba(
384
+ hashes_u64: np.ndarray, table_id_i64: np.ndarray, mask_u64: np.uint64, num_ids: int
385
+ ) -> np.ndarray:
386
+ """
387
+ Perform direct-address lookup with deduplication of results.
388
+
389
+ This function looks up each hash in a direct-address table using
390
+ the lower bits of the hash:
391
+
392
+ idx = hash & mask
393
+
394
+ The lookup table maps each index to an integer ID.
395
+
396
+ The function returns **unique IDs only**, meaning that if the same
397
+ ID appears multiple times across the hash stream it will be returned
398
+ only once.
399
+
400
+ Parameters
401
+ ----------
402
+ hashes_u64 : np.ndarray[np.uint64]
403
+ Array of rolling hash values.
404
+
405
+ table_id_i64 : np.ndarray[np.int64]
406
+ Direct-address lookup table mapping index → ID.
407
+ Values of -1 represent "no entry".
408
+
409
+ mask_u64 : np.uint64
410
+ Bitmask used to compute the index:
411
+
412
+ idx = hash & mask_u64
413
+
414
+ Typically mask = (2^bits - 1).
415
+
416
+ num_ids : int
417
+ Maximum possible ID value + 1. This determines the size of the
418
+ internal `seen` array used for deduplication.
419
+
420
+ Returns
421
+ -------
422
+ np.ndarray[np.int64]
423
+ Array containing the unique IDs encountered in the lookup stream.
424
+ Length ≤ len(hashes_u64).
425
+ """
426
+
427
+ # TODO(Jiayi): These allocations can be avoided by pre-allocations
428
+ seen = np.zeros(num_ids, dtype=np.uint8) # 1 byte per possible id
429
+ out = np.empty(hashes_u64.shape[0], dtype=np.int64)
430
+
431
+ m = 0
432
+ for i in range(hashes_u64.shape[0]):
433
+ idx = hashes_u64[i] & mask_u64
434
+ hit = table_id_i64[idx]
435
+
436
+ if hit != -1 and seen[hit] == 0:
437
+ seen[hit] = 1
438
+ out[m] = hit
439
+ m += 1
440
+
441
+ return out[:m]
@@ -0,0 +1,17 @@
1
+ # SPDX-License-Identifier: Apache-2.0
2
+ # First Party
3
+ from lmcache.v1.lookup_client.abstract_client import LookupClientInterface
4
+ from lmcache.v1.lookup_client.factory import LookupClientFactory
5
+ from lmcache.v1.lookup_client.lmcache_lookup_client import (
6
+ LMCacheLookupClient,
7
+ LMCacheLookupServer,
8
+ )
9
+ from lmcache.v1.lookup_client.mooncake_lookup_client import MooncakeLookupClient
10
+
11
+ __all__ = [
12
+ "LookupClientInterface",
13
+ "LookupClientFactory",
14
+ "MooncakeLookupClient",
15
+ "LMCacheLookupClient",
16
+ "LMCacheLookupServer",
17
+ ]
@@ -0,0 +1,37 @@
1
+ # SPDX-License-Identifier: Apache-2.0
2
+ # Standard
3
+ from typing import TYPE_CHECKING, List
4
+ import abc
5
+
6
+ if TYPE_CHECKING:
7
+ # Third Party
8
+ pass
9
+
10
+
11
+ class OffloadServerInterface(metaclass=abc.ABCMeta):
12
+ """Abstract interface for offload server."""
13
+
14
+ @abc.abstractmethod
15
+ def offload(
16
+ self,
17
+ hashes: List[int],
18
+ slot_mapping: List[int],
19
+ offsets: List[int],
20
+ ) -> bool:
21
+ """
22
+ Perform offload for the given hashes and block IDs.
23
+
24
+ Args:
25
+ hashes: The hashes to offload.
26
+ slot_mapping: The slot ids to offload.
27
+ offsets: Number of tokens in each block.
28
+
29
+ Returns:
30
+ Whether the offload was successful.
31
+ """
32
+ raise NotImplementedError
33
+
34
+ @abc.abstractmethod
35
+ def close(self) -> None:
36
+ """Close the offload server and clean up resources."""
37
+ raise NotImplementedError