lmcache-cli 0.4.5.dev0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (399) hide show
  1. lmcache/__init__.py +84 -0
  2. lmcache/_version.py +24 -0
  3. lmcache/cli/__init__.py +1 -0
  4. lmcache/cli/commands/__init__.py +34 -0
  5. lmcache/cli/commands/base.py +157 -0
  6. lmcache/cli/commands/bench/__init__.py +557 -0
  7. lmcache/cli/commands/bench/engine_bench/__init__.py +1 -0
  8. lmcache/cli/commands/bench/engine_bench/config.py +245 -0
  9. lmcache/cli/commands/bench/engine_bench/interactive/__init__.py +274 -0
  10. lmcache/cli/commands/bench/engine_bench/interactive/config.json +10 -0
  11. lmcache/cli/commands/bench/engine_bench/interactive/schema.py +352 -0
  12. lmcache/cli/commands/bench/engine_bench/interactive/state.py +327 -0
  13. lmcache/cli/commands/bench/engine_bench/interactive/terminal.py +291 -0
  14. lmcache/cli/commands/bench/engine_bench/progress.py +145 -0
  15. lmcache/cli/commands/bench/engine_bench/request_sender.py +232 -0
  16. lmcache/cli/commands/bench/engine_bench/stats.py +275 -0
  17. lmcache/cli/commands/bench/engine_bench/workloads/__init__.py +153 -0
  18. lmcache/cli/commands/bench/engine_bench/workloads/base.py +122 -0
  19. lmcache/cli/commands/bench/engine_bench/workloads/long_doc_permutator.py +435 -0
  20. lmcache/cli/commands/bench/engine_bench/workloads/long_doc_qa.py +281 -0
  21. lmcache/cli/commands/bench/engine_bench/workloads/multi_round_chat.py +337 -0
  22. lmcache/cli/commands/bench/engine_bench/workloads/random_prefill.py +178 -0
  23. lmcache/cli/commands/describe.py +310 -0
  24. lmcache/cli/commands/kvcache.py +133 -0
  25. lmcache/cli/commands/mock.py +75 -0
  26. lmcache/cli/commands/ping.py +113 -0
  27. lmcache/cli/commands/query/__init__.py +155 -0
  28. lmcache/cli/commands/query/prompt.py +134 -0
  29. lmcache/cli/commands/query/request.py +357 -0
  30. lmcache/cli/commands/server.py +99 -0
  31. lmcache/cli/commands/tool/__init__.py +63 -0
  32. lmcache/cli/commands/tool/cache_simulator.py +113 -0
  33. lmcache/cli/commands/trace/__init__.py +505 -0
  34. lmcache/cli/commands/trace/dispatch.py +249 -0
  35. lmcache/cli/commands/trace/driver.py +372 -0
  36. lmcache/cli/commands/trace/stats.py +289 -0
  37. lmcache/cli/documents/lmcache.txt +11 -0
  38. lmcache/cli/main.py +42 -0
  39. lmcache/cli/metrics/__init__.py +29 -0
  40. lmcache/cli/metrics/formatter.py +171 -0
  41. lmcache/cli/metrics/handler.py +94 -0
  42. lmcache/cli/metrics/metrics.py +161 -0
  43. lmcache/cli/metrics/section.py +77 -0
  44. lmcache/connections.py +173 -0
  45. lmcache/integration/__init__.py +2 -0
  46. lmcache/integration/base_service_factory.py +165 -0
  47. lmcache/integration/request_telemetry/__init__.py +1 -0
  48. lmcache/integration/request_telemetry/base.py +51 -0
  49. lmcache/integration/request_telemetry/factory.py +113 -0
  50. lmcache/integration/request_telemetry/fastapi.py +109 -0
  51. lmcache/integration/request_telemetry/noop.py +35 -0
  52. lmcache/integration/sglang/__init__.py +2 -0
  53. lmcache/integration/sglang/sglang_adapter.py +326 -0
  54. lmcache/integration/sglang/utils.py +39 -0
  55. lmcache/integration/vllm/__init__.py +1 -0
  56. lmcache/integration/vllm/lmcache_connector_v1.py +213 -0
  57. lmcache/integration/vllm/lmcache_connector_v1_085.py +150 -0
  58. lmcache/integration/vllm/lmcache_mp_connector_0180.py +1072 -0
  59. lmcache/integration/vllm/tests/test_mm_hash_utils.py +112 -0
  60. lmcache/integration/vllm/utils.py +433 -0
  61. lmcache/integration/vllm/vllm_multi_process_adapter.py +1090 -0
  62. lmcache/integration/vllm/vllm_service_factory.py +339 -0
  63. lmcache/integration/vllm/vllm_v1_adapter.py +1713 -0
  64. lmcache/logging.py +107 -0
  65. lmcache/native_storage_ops.pyi +230 -0
  66. lmcache/non_cuda_equivalents.py +1424 -0
  67. lmcache/observability.py +1958 -0
  68. lmcache/storage_backend/serde/__init__.py +1 -0
  69. lmcache/storage_backend/serde/cachegen_basics.py +210 -0
  70. lmcache/storage_backend/serde/cachegen_decoder.py +207 -0
  71. lmcache/storage_backend/serde/cachegen_encoder.py +394 -0
  72. lmcache/storage_backend/serde/serde.py +75 -0
  73. lmcache/tools/__init__.py +1 -0
  74. lmcache/tools/cache_simulator/README.md +392 -0
  75. lmcache/tools/cache_simulator/__init__.py +1 -0
  76. lmcache/tools/cache_simulator/docs/simulate_example.png +0 -0
  77. lmcache/tools/cache_simulator/docs/sweep_example.png +0 -0
  78. lmcache/tools/cache_simulator/gen_bench_dataset.py +360 -0
  79. lmcache/tools/cache_simulator/lru_cache.py +124 -0
  80. lmcache/tools/cache_simulator/plot_hit_rate.py +231 -0
  81. lmcache/tools/cache_simulator/simulator.py +795 -0
  82. lmcache/tools/controller_benchmark/README.md +161 -0
  83. lmcache/tools/controller_benchmark/__init__.py +1 -0
  84. lmcache/tools/controller_benchmark/__main__.py +331 -0
  85. lmcache/tools/controller_benchmark/benchmark.py +660 -0
  86. lmcache/tools/controller_benchmark/config.py +44 -0
  87. lmcache/tools/controller_benchmark/constants.py +10 -0
  88. lmcache/tools/controller_benchmark/handlers/__init__.py +46 -0
  89. lmcache/tools/controller_benchmark/handlers/admit.py +52 -0
  90. lmcache/tools/controller_benchmark/handlers/base.py +47 -0
  91. lmcache/tools/controller_benchmark/handlers/deregister.py +49 -0
  92. lmcache/tools/controller_benchmark/handlers/evict.py +52 -0
  93. lmcache/tools/controller_benchmark/handlers/heartbeat.py +56 -0
  94. lmcache/tools/controller_benchmark/handlers/p2p_lookup.py +47 -0
  95. lmcache/tools/controller_benchmark/handlers/register.py +56 -0
  96. lmcache/tools/mp_status_viewer/__init__.py +1 -0
  97. lmcache/tools/mp_status_viewer/__main__.py +95 -0
  98. lmcache/usage_context.py +417 -0
  99. lmcache/utils.py +665 -0
  100. lmcache/v1/__init__.py +2 -0
  101. lmcache/v1/api_server/__init__.py +2 -0
  102. lmcache/v1/api_server/__main__.py +537 -0
  103. lmcache/v1/basic_check.py +112 -0
  104. lmcache/v1/cache_controller/__init__.py +9 -0
  105. lmcache/v1/cache_controller/commands/__init__.py +15 -0
  106. lmcache/v1/cache_controller/commands/base.py +35 -0
  107. lmcache/v1/cache_controller/commands/full_sync.py +49 -0
  108. lmcache/v1/cache_controller/config.py +176 -0
  109. lmcache/v1/cache_controller/controller_manager.py +535 -0
  110. lmcache/v1/cache_controller/controllers/__init__.py +11 -0
  111. lmcache/v1/cache_controller/controllers/full_sync_tracker.py +473 -0
  112. lmcache/v1/cache_controller/controllers/kv_controller.py +439 -0
  113. lmcache/v1/cache_controller/controllers/registration_controller.py +282 -0
  114. lmcache/v1/cache_controller/executor.py +463 -0
  115. lmcache/v1/cache_controller/frontend/static/css/style.css +201 -0
  116. lmcache/v1/cache_controller/frontend/static/img/logo.png +0 -0
  117. lmcache/v1/cache_controller/frontend/static/index.html +234 -0
  118. lmcache/v1/cache_controller/frontend/static/js/controller_app.js +660 -0
  119. lmcache/v1/cache_controller/full_sync_sender.py +475 -0
  120. lmcache/v1/cache_controller/locks.py +149 -0
  121. lmcache/v1/cache_controller/message.py +828 -0
  122. lmcache/v1/cache_controller/observability.py +208 -0
  123. lmcache/v1/cache_controller/utils.py +679 -0
  124. lmcache/v1/cache_controller/worker.py +665 -0
  125. lmcache/v1/cache_engine.py +2058 -0
  126. lmcache/v1/cache_interface.py +19 -0
  127. lmcache/v1/check/__init__.py +74 -0
  128. lmcache/v1/check/check_mode_gen.py +86 -0
  129. lmcache/v1/check/check_mode_test_l2_adapter.py +284 -0
  130. lmcache/v1/check/check_mode_test_remote.py +155 -0
  131. lmcache/v1/check/check_mode_test_storage_manager.py +142 -0
  132. lmcache/v1/check/utils.py +571 -0
  133. lmcache/v1/compute/__init__.py +2 -0
  134. lmcache/v1/compute/attention/__init__.py +0 -0
  135. lmcache/v1/compute/attention/abstract.py +39 -0
  136. lmcache/v1/compute/attention/flash_attn.py +129 -0
  137. lmcache/v1/compute/attention/flash_infer_sparse.py +284 -0
  138. lmcache/v1/compute/attention/metadata.py +85 -0
  139. lmcache/v1/compute/attention/utils.py +14 -0
  140. lmcache/v1/compute/blend/__init__.py +7 -0
  141. lmcache/v1/compute/blend/blender.py +168 -0
  142. lmcache/v1/compute/blend/metadata.py +34 -0
  143. lmcache/v1/compute/blend/utils.py +63 -0
  144. lmcache/v1/compute/models/__init__.py +0 -0
  145. lmcache/v1/compute/models/base.py +141 -0
  146. lmcache/v1/compute/models/llama.py +9 -0
  147. lmcache/v1/compute/models/qwen3.py +24 -0
  148. lmcache/v1/compute/models/utils.py +68 -0
  149. lmcache/v1/compute/positional_encoding.py +199 -0
  150. lmcache/v1/config.py +848 -0
  151. lmcache/v1/config_base.py +848 -0
  152. lmcache/v1/distributed/api.py +248 -0
  153. lmcache/v1/distributed/config.py +321 -0
  154. lmcache/v1/distributed/error.py +64 -0
  155. lmcache/v1/distributed/eviction.py +192 -0
  156. lmcache/v1/distributed/eviction_policy/__init__.py +21 -0
  157. lmcache/v1/distributed/eviction_policy/factory.py +27 -0
  158. lmcache/v1/distributed/eviction_policy/lru.py +244 -0
  159. lmcache/v1/distributed/eviction_policy/noop.py +50 -0
  160. lmcache/v1/distributed/internal_api.py +170 -0
  161. lmcache/v1/distributed/l1_manager.py +835 -0
  162. lmcache/v1/distributed/l2_adapters/__init__.py +67 -0
  163. lmcache/v1/distributed/l2_adapters/base.py +360 -0
  164. lmcache/v1/distributed/l2_adapters/config.py +385 -0
  165. lmcache/v1/distributed/l2_adapters/factory.py +205 -0
  166. lmcache/v1/distributed/l2_adapters/fs_l2_adapter.py +747 -0
  167. lmcache/v1/distributed/l2_adapters/fs_native_l2_adapter.py +167 -0
  168. lmcache/v1/distributed/l2_adapters/mock_l2_adapter.py +516 -0
  169. lmcache/v1/distributed/l2_adapters/mooncake_store_l2_adapter.py +135 -0
  170. lmcache/v1/distributed/l2_adapters/native_connector_l2_adapter.py +468 -0
  171. lmcache/v1/distributed/l2_adapters/native_plugin_l2_adapter.py +199 -0
  172. lmcache/v1/distributed/l2_adapters/nixl_store_dynamic_l2_adapter.py +831 -0
  173. lmcache/v1/distributed/l2_adapters/nixl_store_l2_adapter.py +983 -0
  174. lmcache/v1/distributed/l2_adapters/plugin_l2_adapter.py +210 -0
  175. lmcache/v1/distributed/l2_adapters/resp_l2_adapter.py +176 -0
  176. lmcache/v1/distributed/memory_manager.py +179 -0
  177. lmcache/v1/distributed/storage_controller.py +39 -0
  178. lmcache/v1/distributed/storage_controllers/__init__.py +43 -0
  179. lmcache/v1/distributed/storage_controllers/eviction_controller.py +242 -0
  180. lmcache/v1/distributed/storage_controllers/prefetch_controller.py +830 -0
  181. lmcache/v1/distributed/storage_controllers/prefetch_policy.py +193 -0
  182. lmcache/v1/distributed/storage_controllers/store_controller.py +452 -0
  183. lmcache/v1/distributed/storage_controllers/store_policy.py +213 -0
  184. lmcache/v1/distributed/storage_manager.py +532 -0
  185. lmcache/v1/event_manager.py +145 -0
  186. lmcache/v1/exceptions/__init__.py +16 -0
  187. lmcache/v1/gpu_connector/__init__.py +126 -0
  188. lmcache/v1/gpu_connector/gpu_connectors.py +1906 -0
  189. lmcache/v1/gpu_connector/gpu_ops.py +85 -0
  190. lmcache/v1/gpu_connector/hpu_connector.py +326 -0
  191. lmcache/v1/gpu_connector/mock_gpu_connector.py +67 -0
  192. lmcache/v1/gpu_connector/utils.py +890 -0
  193. lmcache/v1/gpu_connector/xpu_connectors.py +916 -0
  194. lmcache/v1/health_monitor/__init__.py +1 -0
  195. lmcache/v1/health_monitor/base.py +587 -0
  196. lmcache/v1/health_monitor/checks/__init__.py +1 -0
  197. lmcache/v1/health_monitor/checks/remote_backend_check.py +304 -0
  198. lmcache/v1/health_monitor/constants.py +36 -0
  199. lmcache/v1/internal_api_server/__init__.py +0 -0
  200. lmcache/v1/internal_api_server/api_registry.py +59 -0
  201. lmcache/v1/internal_api_server/api_server.py +120 -0
  202. lmcache/v1/internal_api_server/common/__init__.py +1 -0
  203. lmcache/v1/internal_api_server/common/env_api.py +22 -0
  204. lmcache/v1/internal_api_server/common/loglevel_api.py +57 -0
  205. lmcache/v1/internal_api_server/common/metrics_api.py +29 -0
  206. lmcache/v1/internal_api_server/common/periodic_thread_api.py +138 -0
  207. lmcache/v1/internal_api_server/common/run_script_api.py +73 -0
  208. lmcache/v1/internal_api_server/common/thread_api.py +63 -0
  209. lmcache/v1/internal_api_server/controller/__init__.py +1 -0
  210. lmcache/v1/internal_api_server/controller/key_stats_api.py +81 -0
  211. lmcache/v1/internal_api_server/controller/worker_info_api.py +136 -0
  212. lmcache/v1/internal_api_server/utils.py +43 -0
  213. lmcache/v1/internal_api_server/vllm/__init__.py +1 -0
  214. lmcache/v1/internal_api_server/vllm/backend_api.py +221 -0
  215. lmcache/v1/internal_api_server/vllm/bypass_api.py +204 -0
  216. lmcache/v1/internal_api_server/vllm/cache_api.py +895 -0
  217. lmcache/v1/internal_api_server/vllm/chunk_statistics_api.py +141 -0
  218. lmcache/v1/internal_api_server/vllm/conf_api.py +147 -0
  219. lmcache/v1/internal_api_server/vllm/freeze_api.py +172 -0
  220. lmcache/v1/internal_api_server/vllm/hot_cache_api.py +184 -0
  221. lmcache/v1/internal_api_server/vllm/inference_api.py +65 -0
  222. lmcache/v1/internal_api_server/vllm/load_fs_chunks_api.py +320 -0
  223. lmcache/v1/internal_api_server/vllm/lookup_api.py +145 -0
  224. lmcache/v1/internal_api_server/vllm/version_api.py +25 -0
  225. lmcache/v1/kv_layer_groups.py +267 -0
  226. lmcache/v1/lazy_memory_allocator.py +284 -0
  227. lmcache/v1/lookup_client/__init__.py +25 -0
  228. lmcache/v1/lookup_client/abstract_client.py +77 -0
  229. lmcache/v1/lookup_client/async_lookup_message.py +50 -0
  230. lmcache/v1/lookup_client/chunk_statistics_lookup_client.py +200 -0
  231. lmcache/v1/lookup_client/factory.py +251 -0
  232. lmcache/v1/lookup_client/hit_limit_lookup_client.py +86 -0
  233. lmcache/v1/lookup_client/lmcache_async_lookup_client.py +407 -0
  234. lmcache/v1/lookup_client/lmcache_lookup_client.py +285 -0
  235. lmcache/v1/lookup_client/lmcache_lookup_client_bypass.py +99 -0
  236. lmcache/v1/lookup_client/mooncake_lookup_client.py +87 -0
  237. lmcache/v1/lookup_client/record_strategies/__init__.py +77 -0
  238. lmcache/v1/lookup_client/record_strategies/base.py +327 -0
  239. lmcache/v1/lookup_client/record_strategies/file_hash.py +130 -0
  240. lmcache/v1/lookup_client/record_strategies/memory_bloom_filter.py +81 -0
  241. lmcache/v1/manager.py +539 -0
  242. lmcache/v1/memory_management.py +2619 -0
  243. lmcache/v1/metadata.py +114 -0
  244. lmcache/v1/mp_observability/AGENTS.override.md +21 -0
  245. lmcache/v1/mp_observability/README.md +204 -0
  246. lmcache/v1/mp_observability/config.py +340 -0
  247. lmcache/v1/mp_observability/event.py +100 -0
  248. lmcache/v1/mp_observability/event_bus.py +313 -0
  249. lmcache/v1/mp_observability/otel_init.py +145 -0
  250. lmcache/v1/mp_observability/subscribers/__init__.py +28 -0
  251. lmcache/v1/mp_observability/subscribers/logging/__init__.py +19 -0
  252. lmcache/v1/mp_observability/subscribers/logging/l1.py +56 -0
  253. lmcache/v1/mp_observability/subscribers/logging/l2.py +73 -0
  254. lmcache/v1/mp_observability/subscribers/logging/lookup_hash.py +209 -0
  255. lmcache/v1/mp_observability/subscribers/logging/mp_server.py +90 -0
  256. lmcache/v1/mp_observability/subscribers/logging/sm.py +59 -0
  257. lmcache/v1/mp_observability/subscribers/metrics/__init__.py +20 -0
  258. lmcache/v1/mp_observability/subscribers/metrics/l0_lifecycle.py +290 -0
  259. lmcache/v1/mp_observability/subscribers/metrics/l1.py +55 -0
  260. lmcache/v1/mp_observability/subscribers/metrics/l1_lifecycle.py +166 -0
  261. lmcache/v1/mp_observability/subscribers/metrics/l2.py +121 -0
  262. lmcache/v1/mp_observability/subscribers/metrics/sm.py +69 -0
  263. lmcache/v1/mp_observability/subscribers/tracing/__init__.py +12 -0
  264. lmcache/v1/mp_observability/subscribers/tracing/mp_server.py +333 -0
  265. lmcache/v1/mp_observability/subscribers/tracing/span_registry.py +148 -0
  266. lmcache/v1/mp_observability/trace/__init__.py +50 -0
  267. lmcache/v1/mp_observability/trace/codecs.py +255 -0
  268. lmcache/v1/mp_observability/trace/decorator.py +147 -0
  269. lmcache/v1/mp_observability/trace/format.py +132 -0
  270. lmcache/v1/mp_observability/trace/lifecycle.py +83 -0
  271. lmcache/v1/mp_observability/trace/reader.py +167 -0
  272. lmcache/v1/mp_observability/trace/recorder.py +300 -0
  273. lmcache/v1/multiprocess/__init__.py +0 -0
  274. lmcache/v1/multiprocess/affinity_pool.py +102 -0
  275. lmcache/v1/multiprocess/blend_server_v2.py +891 -0
  276. lmcache/v1/multiprocess/config.py +253 -0
  277. lmcache/v1/multiprocess/custom_types.py +281 -0
  278. lmcache/v1/multiprocess/futures.py +194 -0
  279. lmcache/v1/multiprocess/gpu_context.py +511 -0
  280. lmcache/v1/multiprocess/http_server.py +235 -0
  281. lmcache/v1/multiprocess/mp_runtime_plugin_launcher.py +130 -0
  282. lmcache/v1/multiprocess/mq.py +732 -0
  283. lmcache/v1/multiprocess/protocol.py +86 -0
  284. lmcache/v1/multiprocess/protocols/README.md +213 -0
  285. lmcache/v1/multiprocess/protocols/__init__.py +127 -0
  286. lmcache/v1/multiprocess/protocols/base.py +89 -0
  287. lmcache/v1/multiprocess/protocols/blend.py +109 -0
  288. lmcache/v1/multiprocess/protocols/blend_v2.py +57 -0
  289. lmcache/v1/multiprocess/protocols/controller.py +53 -0
  290. lmcache/v1/multiprocess/protocols/debug.py +34 -0
  291. lmcache/v1/multiprocess/protocols/engine.py +146 -0
  292. lmcache/v1/multiprocess/protocols/observability.py +39 -0
  293. lmcache/v1/multiprocess/server.py +1134 -0
  294. lmcache/v1/multiprocess/session.py +190 -0
  295. lmcache/v1/multiprocess/token_hasher.py +441 -0
  296. lmcache/v1/offload_server/__init__.py +17 -0
  297. lmcache/v1/offload_server/abstract_server.py +37 -0
  298. lmcache/v1/offload_server/message.py +30 -0
  299. lmcache/v1/offload_server/zmq_server.py +122 -0
  300. lmcache/v1/periodic_thread.py +579 -0
  301. lmcache/v1/pin_monitor.py +246 -0
  302. lmcache/v1/plugin/__init__.py +0 -0
  303. lmcache/v1/plugin/runtime_plugin_launcher.py +211 -0
  304. lmcache/v1/protocol.py +317 -0
  305. lmcache/v1/rpc/__init__.py +17 -0
  306. lmcache/v1/rpc/transport.py +105 -0
  307. lmcache/v1/rpc/zmq_transport.py +213 -0
  308. lmcache/v1/rpc_utils.py +165 -0
  309. lmcache/v1/server/__init__.py +2 -0
  310. lmcache/v1/server/__main__.py +170 -0
  311. lmcache/v1/server/storage_backend/__init__.py +21 -0
  312. lmcache/v1/server/storage_backend/abstract_backend.py +80 -0
  313. lmcache/v1/server/storage_backend/local_backend.py +75 -0
  314. lmcache/v1/server/utils.py +21 -0
  315. lmcache/v1/standalone/__init__.py +1 -0
  316. lmcache/v1/standalone/__main__.py +583 -0
  317. lmcache/v1/standalone/manager.py +80 -0
  318. lmcache/v1/standalone/standalone_service_factory.py +86 -0
  319. lmcache/v1/storage_backend/__init__.py +313 -0
  320. lmcache/v1/storage_backend/abstract_backend.py +445 -0
  321. lmcache/v1/storage_backend/audit_backend.py +233 -0
  322. lmcache/v1/storage_backend/batched_message_sender.py +222 -0
  323. lmcache/v1/storage_backend/cache_policy/__init__.py +45 -0
  324. lmcache/v1/storage_backend/cache_policy/base_policy.py +87 -0
  325. lmcache/v1/storage_backend/cache_policy/fifo.py +58 -0
  326. lmcache/v1/storage_backend/cache_policy/lfu.py +105 -0
  327. lmcache/v1/storage_backend/cache_policy/lru.py +81 -0
  328. lmcache/v1/storage_backend/cache_policy/mru.py +61 -0
  329. lmcache/v1/storage_backend/connector/__init__.py +443 -0
  330. lmcache/v1/storage_backend/connector/audit_adapter.py +77 -0
  331. lmcache/v1/storage_backend/connector/audit_connector.py +320 -0
  332. lmcache/v1/storage_backend/connector/base_connector.py +379 -0
  333. lmcache/v1/storage_backend/connector/blackhole_adapter.py +21 -0
  334. lmcache/v1/storage_backend/connector/blackhole_connector.py +37 -0
  335. lmcache/v1/storage_backend/connector/eic_adapter.py +31 -0
  336. lmcache/v1/storage_backend/connector/eic_connector.py +757 -0
  337. lmcache/v1/storage_backend/connector/external_adapter.py +79 -0
  338. lmcache/v1/storage_backend/connector/fs_adapter.py +51 -0
  339. lmcache/v1/storage_backend/connector/fs_connector.py +403 -0
  340. lmcache/v1/storage_backend/connector/infinistore_adapter.py +56 -0
  341. lmcache/v1/storage_backend/connector/infinistore_connector.py +177 -0
  342. lmcache/v1/storage_backend/connector/instrumented_connector.py +219 -0
  343. lmcache/v1/storage_backend/connector/lm_adapter.py +31 -0
  344. lmcache/v1/storage_backend/connector/lm_connector.py +176 -0
  345. lmcache/v1/storage_backend/connector/mock_adapter.py +57 -0
  346. lmcache/v1/storage_backend/connector/mock_connector.py +349 -0
  347. lmcache/v1/storage_backend/connector/mooncakestore_adapter.py +43 -0
  348. lmcache/v1/storage_backend/connector/mooncakestore_connector.py +614 -0
  349. lmcache/v1/storage_backend/connector/redis_adapter.py +181 -0
  350. lmcache/v1/storage_backend/connector/redis_connector.py +828 -0
  351. lmcache/v1/storage_backend/connector/s3_adapter.py +59 -0
  352. lmcache/v1/storage_backend/connector/s3_connector.py +699 -0
  353. lmcache/v1/storage_backend/connector/sagemaker_hyperpod_adapter.py +233 -0
  354. lmcache/v1/storage_backend/connector/sagemaker_hyperpod_connector.py +987 -0
  355. lmcache/v1/storage_backend/connector/valkey_adapter.py +114 -0
  356. lmcache/v1/storage_backend/connector/valkey_connector.py +627 -0
  357. lmcache/v1/storage_backend/gds_backend.py +1199 -0
  358. lmcache/v1/storage_backend/job_executor/__init__.py +0 -0
  359. lmcache/v1/storage_backend/job_executor/base_executor.py +34 -0
  360. lmcache/v1/storage_backend/job_executor/pq_executor.py +235 -0
  361. lmcache/v1/storage_backend/local_cpu_backend.py +810 -0
  362. lmcache/v1/storage_backend/local_disk_backend.py +656 -0
  363. lmcache/v1/storage_backend/maru_backend.py +734 -0
  364. lmcache/v1/storage_backend/naive_serde/__init__.py +50 -0
  365. lmcache/v1/storage_backend/naive_serde/cachegen_basics.py +133 -0
  366. lmcache/v1/storage_backend/naive_serde/cachegen_decoder.py +135 -0
  367. lmcache/v1/storage_backend/naive_serde/cachegen_encoder.py +83 -0
  368. lmcache/v1/storage_backend/naive_serde/kivi_serde.py +22 -0
  369. lmcache/v1/storage_backend/naive_serde/naive_serde.py +18 -0
  370. lmcache/v1/storage_backend/naive_serde/serde.py +37 -0
  371. lmcache/v1/storage_backend/native_clients/connector_client_base.py +165 -0
  372. lmcache/v1/storage_backend/native_clients/resp_client.py +35 -0
  373. lmcache/v1/storage_backend/nixl_storage_backend.py +1400 -0
  374. lmcache/v1/storage_backend/p2p_backend.py +788 -0
  375. lmcache/v1/storage_backend/path_sharder.py +117 -0
  376. lmcache/v1/storage_backend/pd_backend.py +646 -0
  377. lmcache/v1/storage_backend/plugins/dax_backend.py +1443 -0
  378. lmcache/v1/storage_backend/plugins/rust_raw_block_backend.py +1361 -0
  379. lmcache/v1/storage_backend/remote_backend.py +624 -0
  380. lmcache/v1/storage_backend/resp_client.py +227 -0
  381. lmcache/v1/storage_backend/storage_backend_listener.py +19 -0
  382. lmcache/v1/storage_backend/storage_manager.py +1352 -0
  383. lmcache/v1/system_detection.py +110 -0
  384. lmcache/v1/token_database.py +551 -0
  385. lmcache/v1/transfer_channel/__init__.py +83 -0
  386. lmcache/v1/transfer_channel/abstract.py +285 -0
  387. lmcache/v1/transfer_channel/mock_memory_channel.py +156 -0
  388. lmcache/v1/transfer_channel/nixl_channel.py +639 -0
  389. lmcache/v1/transfer_channel/py_socket_channel.py +260 -0
  390. lmcache/v1/transfer_channel/transfer_utils.py +63 -0
  391. lmcache/v1/utils/__init__.py +1 -0
  392. lmcache/v1/utils/bloom_filter.py +109 -0
  393. lmcache/v1/utils/cache_utils.py +125 -0
  394. lmcache_cli-0.4.5.dev0.dist-info/METADATA +185 -0
  395. lmcache_cli-0.4.5.dev0.dist-info/RECORD +399 -0
  396. lmcache_cli-0.4.5.dev0.dist-info/WHEEL +5 -0
  397. lmcache_cli-0.4.5.dev0.dist-info/entry_points.txt +2 -0
  398. lmcache_cli-0.4.5.dev0.dist-info/licenses/LICENSE +201 -0
  399. lmcache_cli-0.4.5.dev0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,177 @@
1
+ # SPDX-License-Identifier: Apache-2.0
2
+ # Standard
3
+ from functools import reduce
4
+ from typing import List, Optional, Union, no_type_check
5
+ import asyncio
6
+ import ctypes
7
+ import operator
8
+
9
+ # Third Party
10
+ import infinistore
11
+ import torch
12
+
13
+ # First Party
14
+ from lmcache.logging import init_logger
15
+ from lmcache.utils import CacheEngineKey
16
+ from lmcache.v1.memory_management import MemoryObj
17
+
18
+ # reuse
19
+ from lmcache.v1.protocol import RemoteMetadata
20
+ from lmcache.v1.storage_backend.connector.base_connector import RemoteConnector
21
+ from lmcache.v1.storage_backend.local_cpu_backend import LocalCPUBackend
22
+
23
+ logger = init_logger(__name__)
24
+
25
+ MAX_BUFFER_SIZE = 40 << 20 # 40MB
26
+ METADATA_BYTES_LEN = 28
27
+ MAX_BUFFER_CNT = 16
28
+
29
+
30
+ def _get_ptr(mv: Union[bytearray, memoryview]) -> int:
31
+ return ctypes.addressof(ctypes.c_char.from_buffer(mv))
32
+
33
+
34
+ class InfinistoreConnector(RemoteConnector):
35
+ def __init__(
36
+ self,
37
+ host: str,
38
+ port: int,
39
+ dev_name: str,
40
+ link_type: str,
41
+ loop: asyncio.AbstractEventLoop,
42
+ memory_allocator: LocalCPUBackend,
43
+ ):
44
+ # initialize base class, which includes some common attributes
45
+ super().__init__(memory_allocator.config, memory_allocator.metadata)
46
+
47
+ config = infinistore.ClientConfig(
48
+ host_addr=host,
49
+ service_port=port,
50
+ log_level="info",
51
+ connection_type=infinistore.TYPE_RDMA,
52
+ ib_port=1,
53
+ link_type=link_type,
54
+ dev_name=dev_name,
55
+ )
56
+
57
+ self.rdma_conn = infinistore.InfinityConnection(config)
58
+
59
+ self.loop = loop
60
+ self.rdma_conn.connect()
61
+
62
+ self.send_buffers = []
63
+ self.recv_buffers = []
64
+ self.send_queue: asyncio.Queue[int] = asyncio.Queue(maxsize=MAX_BUFFER_CNT)
65
+ self.recv_queue: asyncio.Queue[int] = asyncio.Queue(maxsize=MAX_BUFFER_CNT)
66
+
67
+ self.buffer_size = MAX_BUFFER_SIZE
68
+ self.memory_allocator = memory_allocator
69
+
70
+ for i in range(MAX_BUFFER_CNT):
71
+ send_buffer = bytearray(self.buffer_size)
72
+ self.rdma_conn.register_mr(_get_ptr(send_buffer), self.buffer_size)
73
+ self.send_buffers.append(send_buffer)
74
+ self.send_queue.put_nowait(i)
75
+
76
+ recv_buffer = bytearray(self.buffer_size)
77
+ self.rdma_conn.register_mr(_get_ptr(recv_buffer), self.buffer_size)
78
+ self.recv_buffers.append(recv_buffer)
79
+ self.recv_queue.put_nowait(i)
80
+
81
+ async def exists(self, key: CacheEngineKey) -> bool:
82
+ def blocking_io():
83
+ return self.rdma_conn.check_exist(key.to_string())
84
+
85
+ return await self.loop.run_in_executor(None, blocking_io)
86
+
87
+ def exists_sync(self, key: CacheEngineKey) -> bool:
88
+ return self.rdma_conn.check_exist(key.to_string())
89
+
90
+ async def get(self, key: CacheEngineKey) -> Optional[MemoryObj]:
91
+ key_str = key.to_string()
92
+
93
+ buf_idx = await self.recv_queue.get()
94
+ buffer = self.recv_buffers[buf_idx]
95
+ try:
96
+ await self.rdma_conn.rdma_read_cache_async(
97
+ [(key_str, 0)], self.buffer_size, _get_ptr(buffer)
98
+ )
99
+ except Exception as e:
100
+ logger.warning(f"get failed: {e}")
101
+ self.recv_queue.put_nowait(buf_idx)
102
+ return None
103
+
104
+ metadata = RemoteMetadata.deserialize(buffer)
105
+
106
+ num_elements = reduce(operator.mul, metadata.shapes[0])
107
+ assert len(metadata.dtypes) == 1
108
+ temp_tensor = torch.frombuffer(
109
+ buffer,
110
+ dtype=metadata.dtypes[0],
111
+ offset=METADATA_BYTES_LEN,
112
+ count=num_elements,
113
+ ).reshape(metadata.shapes[0])
114
+
115
+ memory_obj = self.memory_allocator.allocate(
116
+ metadata.shapes[0],
117
+ metadata.dtypes[0],
118
+ metadata.fmt,
119
+ )
120
+
121
+ assert memory_obj is not None
122
+ assert memory_obj.tensor is not None
123
+
124
+ # deep copy to pinned memory
125
+ # and hot cache will reference this memory obj
126
+ memory_obj.tensor.copy_(temp_tensor)
127
+
128
+ logger.debug(f"get key: {key_str} done, {memory_obj.get_shape()}")
129
+ self.recv_queue.put_nowait(buf_idx)
130
+
131
+ return memory_obj
132
+
133
+ async def put(self, key: CacheEngineKey, memory_obj: MemoryObj):
134
+ key_str = key.to_string()
135
+
136
+ kv_bytes = memory_obj.byte_array
137
+ kv_shapes = memory_obj.get_shapes()
138
+ kv_dtypes = memory_obj.get_dtypes()
139
+ memory_format = memory_obj.get_memory_format()
140
+
141
+ buf_idx = await self.send_queue.get()
142
+ buffer = self.send_buffers[buf_idx]
143
+
144
+ RemoteMetadata(
145
+ len(kv_bytes), kv_shapes, kv_dtypes, memory_format
146
+ ).serialize_into(buffer)
147
+
148
+ buffer[METADATA_BYTES_LEN : METADATA_BYTES_LEN + len(kv_bytes)] = kv_bytes
149
+
150
+ size = memory_obj.get_physical_size()
151
+
152
+ if size + METADATA_BYTES_LEN > self.buffer_size:
153
+ raise ValueError(
154
+ f"Value size ({size + METADATA_BYTES_LEN} bytes)"
155
+ f"exceeds the maximum allowed size"
156
+ f"({self.buffer_size} bytes). Please decrease chunk_size."
157
+ )
158
+ try:
159
+ await self.rdma_conn.rdma_write_cache_async(
160
+ [(key_str, 0)], METADATA_BYTES_LEN + size, _get_ptr(buffer)
161
+ )
162
+ except Exception as e:
163
+ logger.warning(f"exception happens in rdma_write_cache_async kv_bytes {e}")
164
+ return
165
+ finally:
166
+ self.send_queue.put_nowait(buf_idx)
167
+
168
+ logger.debug(f"put key: {key.to_string()}, {memory_obj.get_shape()}")
169
+
170
+ # TODO
171
+ @no_type_check
172
+ async def list(self) -> List[str]:
173
+ pass
174
+
175
+ async def close(self):
176
+ self.rdma_conn.close()
177
+ logger.info("Closed the infinistore connection")
@@ -0,0 +1,219 @@
1
+ # SPDX-License-Identifier: Apache-2.0
2
+ # Standard
3
+ from typing import List, Optional
4
+ import time
5
+
6
+ # First Party
7
+ from lmcache.logging import init_logger
8
+ from lmcache.observability import LMCStatsMonitor
9
+ from lmcache.utils import CacheEngineKey
10
+ from lmcache.v1.memory_management import MemoryObj
11
+ from lmcache.v1.storage_backend.connector.base_connector import RemoteConnector
12
+
13
+ logger = init_logger(__name__)
14
+
15
+
16
+ class InstrumentedRemoteConnector(RemoteConnector):
17
+ """
18
+ A connector that instruments the underlying connector with
19
+ metrics collection and logging capabilities.
20
+ """
21
+
22
+ def __init__(self, connector: RemoteConnector):
23
+ self._connector = connector
24
+ self._stats_monitor = LMCStatsMonitor.GetOrCreate()
25
+ self.name = self.__repr__()
26
+
27
+ async def put(self, key: CacheEngineKey, memory_obj: MemoryObj) -> None:
28
+ obj_size = memory_obj.get_size()
29
+ begin = time.perf_counter()
30
+ try:
31
+ await self._connector.put(key, memory_obj)
32
+ finally:
33
+ # Ensure reference count is decreased even if exception occurs
34
+ memory_obj.ref_count_down()
35
+
36
+ end = time.perf_counter()
37
+ self._stats_monitor.update_interval_remote_time_to_put((end - begin) * 1000)
38
+ self._stats_monitor.update_interval_remote_write_metrics(obj_size)
39
+ logger.debug(
40
+ "[%s]Bytes offloaded: %.3f MBytes in %.3f ms",
41
+ self.name,
42
+ obj_size / 1e6,
43
+ (end - begin) * 1000,
44
+ )
45
+
46
+ async def get(self, key: CacheEngineKey) -> Optional[MemoryObj]:
47
+ begin = time.perf_counter()
48
+ memory_obj = await self._connector.get(key)
49
+ end = time.perf_counter()
50
+ duration = end - begin
51
+
52
+ retrieve_stats = self._stats_monitor.get_current_retrieve_stats()
53
+ if (
54
+ retrieve_stats is not None
55
+ and "remote_backend_individual_get_stats" in retrieve_stats.detailed_metrics
56
+ ):
57
+ retrieve_stats.detailed_metrics["remote_backend_individual_get_stats"][
58
+ key
59
+ ] = {"instrumented_connector_get_time": duration}
60
+
61
+ if memory_obj is not None:
62
+ obj_size = memory_obj.get_size()
63
+ self._stats_monitor.update_interval_remote_read_metrics(obj_size)
64
+ logger.debug(
65
+ "[%s]Bytes loaded: %.3f MBytes in %.3f ms",
66
+ self.name,
67
+ obj_size / 1e6,
68
+ duration * 1000,
69
+ )
70
+ return memory_obj
71
+
72
+ # Delegate all other methods to the underlying connector
73
+ async def exists(self, key: CacheEngineKey) -> bool:
74
+ return await self._connector.exists(key)
75
+
76
+ def exists_sync(self, key: CacheEngineKey) -> bool:
77
+ return self._connector.exists_sync(key)
78
+
79
+ async def list(self) -> List[str]:
80
+ return await self._connector.list()
81
+
82
+ async def close(self) -> None:
83
+ await self._connector.close()
84
+
85
+ def getWrappedConnector(self) -> RemoteConnector:
86
+ return self._connector
87
+
88
+ def support_ping(self) -> bool:
89
+ return self._connector.support_ping()
90
+
91
+ async def ping(self) -> int:
92
+ return await self._connector.ping()
93
+
94
+ def support_batched_put(self) -> bool:
95
+ return self._connector.support_batched_put()
96
+
97
+ def support_batched_get(self) -> bool:
98
+ return self._connector.support_batched_get()
99
+
100
+ def support_batched_async_contains(self) -> bool:
101
+ return self._connector.support_batched_async_contains()
102
+
103
+ async def batched_async_contains(
104
+ self,
105
+ lookup_id: str,
106
+ keys: List[CacheEngineKey],
107
+ pin: bool = False,
108
+ ) -> int:
109
+ return await self._connector.batched_async_contains(lookup_id, keys, pin)
110
+
111
+ def support_batched_get_non_blocking(self) -> bool:
112
+ return self._connector.support_batched_get_non_blocking()
113
+
114
+ async def batched_get_non_blocking(
115
+ self,
116
+ lookup_id: str,
117
+ keys: List[CacheEngineKey],
118
+ ) -> List[MemoryObj]:
119
+ begin = time.perf_counter()
120
+ memory_objs = await self._connector.batched_get_non_blocking(lookup_id, keys)
121
+ end = time.perf_counter()
122
+ duration = end - begin
123
+
124
+ total_size = sum(
125
+ memory_obj.get_size()
126
+ for memory_obj in memory_objs
127
+ if memory_obj is not None
128
+ )
129
+ if total_size > 0:
130
+ self._stats_monitor.update_interval_remote_read_metrics(total_size)
131
+ logger.debug(
132
+ "[%s]Bytes loaded: %.3f MBytes in %.3f ms",
133
+ self.name,
134
+ total_size / 1e6,
135
+ duration * 1000,
136
+ )
137
+ return memory_objs
138
+
139
+ async def batched_get(
140
+ self, keys: List[CacheEngineKey]
141
+ ) -> List[Optional[MemoryObj]]:
142
+ begin = time.perf_counter()
143
+ memory_objs = await self._connector.batched_get(keys)
144
+ end = time.perf_counter()
145
+ duration = end - begin
146
+ self._stats_monitor.update_interval_remote_time_to_get(duration * 1000)
147
+
148
+ retrieve_stats = self._stats_monitor.get_current_retrieve_stats()
149
+ if retrieve_stats is not None:
150
+ retrieve_stats.detailed_metrics[
151
+ "instrumented_connector_batched_get_time"
152
+ ] = (
153
+ retrieve_stats.detailed_metrics.get(
154
+ "instrumented_connector_batched_get_time", 0.0
155
+ )
156
+ + duration
157
+ )
158
+
159
+ total_size = sum(
160
+ memory_obj.get_size()
161
+ for memory_obj in memory_objs
162
+ if memory_obj is not None
163
+ )
164
+ if total_size > 0:
165
+ self._stats_monitor.update_interval_remote_read_metrics(total_size)
166
+ logger.debug(
167
+ "[%s]Bytes loaded: %.3f MBytes in %.3f ms",
168
+ self.name,
169
+ total_size / 1e6,
170
+ duration * 1000,
171
+ )
172
+ return memory_objs
173
+
174
+ async def batched_put(
175
+ self, keys: List[CacheEngineKey], memory_objs: List[MemoryObj]
176
+ ):
177
+ total_size = sum(
178
+ memory_obj.get_size()
179
+ for memory_obj in memory_objs
180
+ if memory_obj is not None
181
+ )
182
+ begin = time.perf_counter()
183
+ try:
184
+ await self._connector.batched_put(keys, memory_objs)
185
+ except Exception as e:
186
+ logger.warning(f"batched put error: {e}")
187
+ finally:
188
+ for memory_obj in memory_objs:
189
+ memory_obj.ref_count_down()
190
+
191
+ end = time.perf_counter()
192
+ self._stats_monitor.update_interval_remote_time_to_put((end - begin) * 1000)
193
+ self._stats_monitor.update_interval_remote_write_metrics(total_size)
194
+ logger.debug(
195
+ "[%s]Bytes offloaded: %.3f MBytes in %.3f ms",
196
+ self.name,
197
+ total_size / 1e6,
198
+ (end - begin) * 1000,
199
+ )
200
+
201
+ def remove_sync(self, key: CacheEngineKey) -> bool:
202
+ return self._connector.remove_sync(key)
203
+
204
+ def batched_contains(self, keys: List[CacheEngineKey]) -> int:
205
+ return self._connector.batched_contains(keys)
206
+
207
+ def support_batched_contains(self) -> bool:
208
+ return self._connector.support_batched_contains()
209
+
210
+ def reshape_partial_chunk(
211
+ self, memory_obj: MemoryObj, bytes_read: int
212
+ ) -> MemoryObj:
213
+ return self._connector.reshape_partial_chunk(memory_obj, bytes_read)
214
+
215
+ def post_init(self):
216
+ return self._connector.post_init()
217
+
218
+ def __repr__(self) -> str:
219
+ return f"InstrumentedRemoteConnector({self._connector})"
@@ -0,0 +1,31 @@
1
+ # SPDX-License-Identifier: Apache-2.0
2
+ # First Party
3
+ from lmcache.logging import init_logger
4
+ from lmcache.v1.storage_backend.connector import (
5
+ ConnectorAdapter,
6
+ ConnectorContext,
7
+ parse_remote_url,
8
+ )
9
+ from lmcache.v1.storage_backend.connector.base_connector import RemoteConnector
10
+
11
+ logger = init_logger(__name__)
12
+
13
+
14
+ class LMServerConnectorAdapter(ConnectorAdapter):
15
+ """Adapter for LM Server connectors."""
16
+
17
+ def __init__(self) -> None:
18
+ super().__init__("lm://")
19
+
20
+ def create_connector(self, context: ConnectorContext) -> RemoteConnector:
21
+ # Local
22
+ from .lm_connector import LMCServerConnector
23
+
24
+ logger.info(f"Creating LM Server connector for URL: {context.url}")
25
+ parse_url = parse_remote_url(context.url)
26
+ return LMCServerConnector(
27
+ host=parse_url.host,
28
+ port=parse_url.port,
29
+ loop=context.loop,
30
+ local_cpu_backend=context.local_cpu_backend,
31
+ )
@@ -0,0 +1,176 @@
1
+ # SPDX-License-Identifier: Apache-2.0
2
+ # Standard
3
+ from typing import List, Optional, no_type_check
4
+ import asyncio
5
+ import socket
6
+
7
+ # Third Party
8
+ import torch
9
+
10
+ # First Party
11
+ from lmcache.logging import init_logger
12
+ from lmcache.utils import CacheEngineKey, _lmcache_nvtx_annotate
13
+ from lmcache.v1.memory_management import MemoryFormat, MemoryObj
14
+ from lmcache.v1.protocol import (
15
+ ClientCommand,
16
+ ClientMetaMessage,
17
+ ServerMetaMessage,
18
+ ServerReturnCode,
19
+ )
20
+ from lmcache.v1.storage_backend.connector.base_connector import RemoteConnector
21
+ from lmcache.v1.storage_backend.local_cpu_backend import LocalCPUBackend
22
+
23
+ logger = init_logger(__name__)
24
+
25
+
26
+ # TODO: performance optimization for this class, consider using C/C++/Rust
27
+ # for communication + deserialization
28
+ class LMCServerConnector(RemoteConnector):
29
+ def __init__(
30
+ self,
31
+ host: str,
32
+ port: int,
33
+ loop: asyncio.AbstractEventLoop,
34
+ local_cpu_backend: LocalCPUBackend,
35
+ ):
36
+ # NOTE(Jiayi): According to Python documentation:
37
+ # https://docs.python.org/3/library/asyncio-eventloop.html
38
+ # In general, protocol implementations that use transport-based APIs
39
+ # such as loop.create_connection() and loop.create_server() are faster
40
+ # than implementations that work with sockets.
41
+ # However, we use socket here as we need to use the socket.recv_into()
42
+ # to reduce memory copy.
43
+
44
+ # initialize base class, which includes some common attributes
45
+ super().__init__(local_cpu_backend.config, local_cpu_backend.metadata)
46
+
47
+ self.client_socket = socket.socket(socket.AF_INET, socket.SOCK_STREAM)
48
+ self.client_socket.connect((host, port))
49
+ # loop.sock_recv_into(sock, buf)
50
+
51
+ self.loop = loop
52
+ self.local_cpu_backend = local_cpu_backend
53
+
54
+ self.async_socket_lock = asyncio.Lock()
55
+
56
+ # TODO(Jiayi): This should be an async function
57
+ def receive_all(self, meta: ServerMetaMessage) -> Optional[MemoryObj]:
58
+ received = 0
59
+ n = meta.length
60
+
61
+ # TODO(Jiayi): Format will be used once we support
62
+ # compressed memory format
63
+ memory_obj = self.local_cpu_backend.allocate(
64
+ meta.shape,
65
+ meta.dtype,
66
+ meta.fmt,
67
+ )
68
+ if memory_obj is None:
69
+ logger.warning("Failed to allocate memory during remote receive")
70
+ return None
71
+
72
+ buffer = memory_obj.byte_array
73
+ view = memoryview(buffer)
74
+
75
+ while received < n:
76
+ num_bytes = self.client_socket.recv_into(view[received:], n - received)
77
+ if num_bytes == 0:
78
+ return None
79
+ received += num_bytes
80
+
81
+ return memory_obj
82
+
83
+ async def exists(self, key: CacheEngineKey) -> bool:
84
+ # logger.debug("Call to exists()!")
85
+
86
+ async with self.async_socket_lock:
87
+ self.client_socket.sendall(
88
+ ClientMetaMessage(
89
+ ClientCommand.EXIST,
90
+ key,
91
+ 0,
92
+ MemoryFormat(1),
93
+ torch.float16,
94
+ torch.Size([0, 0, 0, 0]),
95
+ ).serialize()
96
+ )
97
+
98
+ response = self.client_socket.recv(ServerMetaMessage.packlength())
99
+
100
+ return ServerMetaMessage.deserialize(response).code == ServerReturnCode.SUCCESS
101
+
102
+ def exists_sync(self, key: CacheEngineKey) -> bool:
103
+ future = asyncio.run_coroutine_threadsafe(self.exists(key), self.loop)
104
+ try:
105
+ res = future.result()
106
+ return res
107
+ except Exception as e:
108
+ logger.warning(f"lm connector failed in exists: {e}")
109
+ return False
110
+
111
+ async def put(
112
+ self,
113
+ key: CacheEngineKey,
114
+ memory_obj: MemoryObj,
115
+ ):
116
+ # logger.debug("Async call to put()!")
117
+
118
+ kv_bytes = memory_obj.byte_array
119
+ kv_shape = memory_obj.get_shape()
120
+ kv_dtype = memory_obj.get_dtype()
121
+ memory_format = memory_obj.get_memory_format()
122
+
123
+ async with self.async_socket_lock:
124
+ await self.loop.sock_sendall(
125
+ self.client_socket,
126
+ ClientMetaMessage(
127
+ ClientCommand.PUT,
128
+ key,
129
+ len(kv_bytes),
130
+ memory_format,
131
+ kv_dtype,
132
+ kv_shape,
133
+ ).serialize(),
134
+ )
135
+
136
+ await self.loop.sock_sendall(self.client_socket, kv_bytes)
137
+
138
+ # TODO(Jiayi): This should be an async function
139
+ @_lmcache_nvtx_annotate
140
+ async def get(self, key: CacheEngineKey) -> Optional[MemoryObj]:
141
+ # NOTE(Jiayi): Not using any await in the following as
142
+ # we don't want to yield control to other tasks which could
143
+ # sacrifice the performance loading to trade the performance of
144
+ # saving
145
+ async with self.async_socket_lock:
146
+ self.client_socket.sendall(
147
+ ClientMetaMessage(
148
+ ClientCommand.GET,
149
+ key,
150
+ 0,
151
+ MemoryFormat(1),
152
+ torch.float16,
153
+ torch.Size([0, 0, 0, 0]),
154
+ ).serialize()
155
+ )
156
+
157
+ data = self.client_socket.recv(ServerMetaMessage.packlength())
158
+
159
+ meta = ServerMetaMessage.deserialize(data)
160
+ if meta.code != ServerReturnCode.SUCCESS:
161
+ return None
162
+
163
+ async with self.async_socket_lock:
164
+ memory_obj = self.receive_all(meta)
165
+
166
+ return memory_obj
167
+
168
+ # TODO
169
+ @no_type_check
170
+ async def list(self) -> List[str]:
171
+ pass
172
+
173
+ async def close(self):
174
+ async with self.async_socket_lock:
175
+ self.client_socket.close()
176
+ logger.info("Closed the lmserver connection")
@@ -0,0 +1,57 @@
1
+ # SPDX-License-Identifier: Apache-2.0
2
+ # Standard
3
+ from urllib.parse import parse_qs, urlparse
4
+
5
+ # First Party
6
+ from lmcache.logging import init_logger
7
+ from lmcache.v1.storage_backend.connector import (
8
+ ConnectorAdapter,
9
+ ConnectorContext,
10
+ )
11
+ from lmcache.v1.storage_backend.connector.base_connector import RemoteConnector
12
+
13
+ logger = init_logger(__name__)
14
+
15
+
16
+ class MockConnectorAdapter(ConnectorAdapter):
17
+ """Adapter for Mock Connector"""
18
+
19
+ def __init__(self) -> None:
20
+ super().__init__("mock://")
21
+
22
+ def create_connector(self, context: ConnectorContext) -> RemoteConnector:
23
+ # Local import to avoid circular dependencies
24
+ # Local
25
+ from .mock_connector import MockConnector
26
+
27
+ logger.info(f"Creating Mock connector for URL: {context.url}")
28
+
29
+ parsed = urlparse(context.url)
30
+ # capacity is provided as the netloc in URLs like: mock://100/?...
31
+ if not parsed.netloc:
32
+ raise ValueError(
33
+ "mock connector requires capacity in GB as netloc, e.g. mock://100/?..."
34
+ )
35
+ try:
36
+ capacity_gb = int(parsed.netloc)
37
+ except ValueError as e:
38
+ raise ValueError(
39
+ f"Invalid capacity '{parsed.netloc}' for",
40
+ " mock connector; must be an integer (GB).",
41
+ ) from e
42
+
43
+ params = parse_qs(parsed.query) if parsed.query else {}
44
+ # Defaults
45
+ peeking_latency_ms = float(params.get("peeking_latency", ["1"])[0])
46
+ read_throughput_gbps = float(params.get("read_throughput", ["2"])[0])
47
+ write_throughput_gbps = float(params.get("write_throughput", ["2"])[0])
48
+
49
+ return MockConnector(
50
+ url=context.url,
51
+ loop=context.loop,
52
+ local_cpu_backend=context.local_cpu_backend,
53
+ capacity=capacity_gb,
54
+ read_throughput=read_throughput_gbps,
55
+ write_throughput=write_throughput_gbps,
56
+ peeking_latency=peeking_latency_ms,
57
+ )