lmcache-cli 0.4.5.dev0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (399) hide show
  1. lmcache/__init__.py +84 -0
  2. lmcache/_version.py +24 -0
  3. lmcache/cli/__init__.py +1 -0
  4. lmcache/cli/commands/__init__.py +34 -0
  5. lmcache/cli/commands/base.py +157 -0
  6. lmcache/cli/commands/bench/__init__.py +557 -0
  7. lmcache/cli/commands/bench/engine_bench/__init__.py +1 -0
  8. lmcache/cli/commands/bench/engine_bench/config.py +245 -0
  9. lmcache/cli/commands/bench/engine_bench/interactive/__init__.py +274 -0
  10. lmcache/cli/commands/bench/engine_bench/interactive/config.json +10 -0
  11. lmcache/cli/commands/bench/engine_bench/interactive/schema.py +352 -0
  12. lmcache/cli/commands/bench/engine_bench/interactive/state.py +327 -0
  13. lmcache/cli/commands/bench/engine_bench/interactive/terminal.py +291 -0
  14. lmcache/cli/commands/bench/engine_bench/progress.py +145 -0
  15. lmcache/cli/commands/bench/engine_bench/request_sender.py +232 -0
  16. lmcache/cli/commands/bench/engine_bench/stats.py +275 -0
  17. lmcache/cli/commands/bench/engine_bench/workloads/__init__.py +153 -0
  18. lmcache/cli/commands/bench/engine_bench/workloads/base.py +122 -0
  19. lmcache/cli/commands/bench/engine_bench/workloads/long_doc_permutator.py +435 -0
  20. lmcache/cli/commands/bench/engine_bench/workloads/long_doc_qa.py +281 -0
  21. lmcache/cli/commands/bench/engine_bench/workloads/multi_round_chat.py +337 -0
  22. lmcache/cli/commands/bench/engine_bench/workloads/random_prefill.py +178 -0
  23. lmcache/cli/commands/describe.py +310 -0
  24. lmcache/cli/commands/kvcache.py +133 -0
  25. lmcache/cli/commands/mock.py +75 -0
  26. lmcache/cli/commands/ping.py +113 -0
  27. lmcache/cli/commands/query/__init__.py +155 -0
  28. lmcache/cli/commands/query/prompt.py +134 -0
  29. lmcache/cli/commands/query/request.py +357 -0
  30. lmcache/cli/commands/server.py +99 -0
  31. lmcache/cli/commands/tool/__init__.py +63 -0
  32. lmcache/cli/commands/tool/cache_simulator.py +113 -0
  33. lmcache/cli/commands/trace/__init__.py +505 -0
  34. lmcache/cli/commands/trace/dispatch.py +249 -0
  35. lmcache/cli/commands/trace/driver.py +372 -0
  36. lmcache/cli/commands/trace/stats.py +289 -0
  37. lmcache/cli/documents/lmcache.txt +11 -0
  38. lmcache/cli/main.py +42 -0
  39. lmcache/cli/metrics/__init__.py +29 -0
  40. lmcache/cli/metrics/formatter.py +171 -0
  41. lmcache/cli/metrics/handler.py +94 -0
  42. lmcache/cli/metrics/metrics.py +161 -0
  43. lmcache/cli/metrics/section.py +77 -0
  44. lmcache/connections.py +173 -0
  45. lmcache/integration/__init__.py +2 -0
  46. lmcache/integration/base_service_factory.py +165 -0
  47. lmcache/integration/request_telemetry/__init__.py +1 -0
  48. lmcache/integration/request_telemetry/base.py +51 -0
  49. lmcache/integration/request_telemetry/factory.py +113 -0
  50. lmcache/integration/request_telemetry/fastapi.py +109 -0
  51. lmcache/integration/request_telemetry/noop.py +35 -0
  52. lmcache/integration/sglang/__init__.py +2 -0
  53. lmcache/integration/sglang/sglang_adapter.py +326 -0
  54. lmcache/integration/sglang/utils.py +39 -0
  55. lmcache/integration/vllm/__init__.py +1 -0
  56. lmcache/integration/vllm/lmcache_connector_v1.py +213 -0
  57. lmcache/integration/vllm/lmcache_connector_v1_085.py +150 -0
  58. lmcache/integration/vllm/lmcache_mp_connector_0180.py +1072 -0
  59. lmcache/integration/vllm/tests/test_mm_hash_utils.py +112 -0
  60. lmcache/integration/vllm/utils.py +433 -0
  61. lmcache/integration/vllm/vllm_multi_process_adapter.py +1090 -0
  62. lmcache/integration/vllm/vllm_service_factory.py +339 -0
  63. lmcache/integration/vllm/vllm_v1_adapter.py +1713 -0
  64. lmcache/logging.py +107 -0
  65. lmcache/native_storage_ops.pyi +230 -0
  66. lmcache/non_cuda_equivalents.py +1424 -0
  67. lmcache/observability.py +1958 -0
  68. lmcache/storage_backend/serde/__init__.py +1 -0
  69. lmcache/storage_backend/serde/cachegen_basics.py +210 -0
  70. lmcache/storage_backend/serde/cachegen_decoder.py +207 -0
  71. lmcache/storage_backend/serde/cachegen_encoder.py +394 -0
  72. lmcache/storage_backend/serde/serde.py +75 -0
  73. lmcache/tools/__init__.py +1 -0
  74. lmcache/tools/cache_simulator/README.md +392 -0
  75. lmcache/tools/cache_simulator/__init__.py +1 -0
  76. lmcache/tools/cache_simulator/docs/simulate_example.png +0 -0
  77. lmcache/tools/cache_simulator/docs/sweep_example.png +0 -0
  78. lmcache/tools/cache_simulator/gen_bench_dataset.py +360 -0
  79. lmcache/tools/cache_simulator/lru_cache.py +124 -0
  80. lmcache/tools/cache_simulator/plot_hit_rate.py +231 -0
  81. lmcache/tools/cache_simulator/simulator.py +795 -0
  82. lmcache/tools/controller_benchmark/README.md +161 -0
  83. lmcache/tools/controller_benchmark/__init__.py +1 -0
  84. lmcache/tools/controller_benchmark/__main__.py +331 -0
  85. lmcache/tools/controller_benchmark/benchmark.py +660 -0
  86. lmcache/tools/controller_benchmark/config.py +44 -0
  87. lmcache/tools/controller_benchmark/constants.py +10 -0
  88. lmcache/tools/controller_benchmark/handlers/__init__.py +46 -0
  89. lmcache/tools/controller_benchmark/handlers/admit.py +52 -0
  90. lmcache/tools/controller_benchmark/handlers/base.py +47 -0
  91. lmcache/tools/controller_benchmark/handlers/deregister.py +49 -0
  92. lmcache/tools/controller_benchmark/handlers/evict.py +52 -0
  93. lmcache/tools/controller_benchmark/handlers/heartbeat.py +56 -0
  94. lmcache/tools/controller_benchmark/handlers/p2p_lookup.py +47 -0
  95. lmcache/tools/controller_benchmark/handlers/register.py +56 -0
  96. lmcache/tools/mp_status_viewer/__init__.py +1 -0
  97. lmcache/tools/mp_status_viewer/__main__.py +95 -0
  98. lmcache/usage_context.py +417 -0
  99. lmcache/utils.py +665 -0
  100. lmcache/v1/__init__.py +2 -0
  101. lmcache/v1/api_server/__init__.py +2 -0
  102. lmcache/v1/api_server/__main__.py +537 -0
  103. lmcache/v1/basic_check.py +112 -0
  104. lmcache/v1/cache_controller/__init__.py +9 -0
  105. lmcache/v1/cache_controller/commands/__init__.py +15 -0
  106. lmcache/v1/cache_controller/commands/base.py +35 -0
  107. lmcache/v1/cache_controller/commands/full_sync.py +49 -0
  108. lmcache/v1/cache_controller/config.py +176 -0
  109. lmcache/v1/cache_controller/controller_manager.py +535 -0
  110. lmcache/v1/cache_controller/controllers/__init__.py +11 -0
  111. lmcache/v1/cache_controller/controllers/full_sync_tracker.py +473 -0
  112. lmcache/v1/cache_controller/controllers/kv_controller.py +439 -0
  113. lmcache/v1/cache_controller/controllers/registration_controller.py +282 -0
  114. lmcache/v1/cache_controller/executor.py +463 -0
  115. lmcache/v1/cache_controller/frontend/static/css/style.css +201 -0
  116. lmcache/v1/cache_controller/frontend/static/img/logo.png +0 -0
  117. lmcache/v1/cache_controller/frontend/static/index.html +234 -0
  118. lmcache/v1/cache_controller/frontend/static/js/controller_app.js +660 -0
  119. lmcache/v1/cache_controller/full_sync_sender.py +475 -0
  120. lmcache/v1/cache_controller/locks.py +149 -0
  121. lmcache/v1/cache_controller/message.py +828 -0
  122. lmcache/v1/cache_controller/observability.py +208 -0
  123. lmcache/v1/cache_controller/utils.py +679 -0
  124. lmcache/v1/cache_controller/worker.py +665 -0
  125. lmcache/v1/cache_engine.py +2058 -0
  126. lmcache/v1/cache_interface.py +19 -0
  127. lmcache/v1/check/__init__.py +74 -0
  128. lmcache/v1/check/check_mode_gen.py +86 -0
  129. lmcache/v1/check/check_mode_test_l2_adapter.py +284 -0
  130. lmcache/v1/check/check_mode_test_remote.py +155 -0
  131. lmcache/v1/check/check_mode_test_storage_manager.py +142 -0
  132. lmcache/v1/check/utils.py +571 -0
  133. lmcache/v1/compute/__init__.py +2 -0
  134. lmcache/v1/compute/attention/__init__.py +0 -0
  135. lmcache/v1/compute/attention/abstract.py +39 -0
  136. lmcache/v1/compute/attention/flash_attn.py +129 -0
  137. lmcache/v1/compute/attention/flash_infer_sparse.py +284 -0
  138. lmcache/v1/compute/attention/metadata.py +85 -0
  139. lmcache/v1/compute/attention/utils.py +14 -0
  140. lmcache/v1/compute/blend/__init__.py +7 -0
  141. lmcache/v1/compute/blend/blender.py +168 -0
  142. lmcache/v1/compute/blend/metadata.py +34 -0
  143. lmcache/v1/compute/blend/utils.py +63 -0
  144. lmcache/v1/compute/models/__init__.py +0 -0
  145. lmcache/v1/compute/models/base.py +141 -0
  146. lmcache/v1/compute/models/llama.py +9 -0
  147. lmcache/v1/compute/models/qwen3.py +24 -0
  148. lmcache/v1/compute/models/utils.py +68 -0
  149. lmcache/v1/compute/positional_encoding.py +199 -0
  150. lmcache/v1/config.py +848 -0
  151. lmcache/v1/config_base.py +848 -0
  152. lmcache/v1/distributed/api.py +248 -0
  153. lmcache/v1/distributed/config.py +321 -0
  154. lmcache/v1/distributed/error.py +64 -0
  155. lmcache/v1/distributed/eviction.py +192 -0
  156. lmcache/v1/distributed/eviction_policy/__init__.py +21 -0
  157. lmcache/v1/distributed/eviction_policy/factory.py +27 -0
  158. lmcache/v1/distributed/eviction_policy/lru.py +244 -0
  159. lmcache/v1/distributed/eviction_policy/noop.py +50 -0
  160. lmcache/v1/distributed/internal_api.py +170 -0
  161. lmcache/v1/distributed/l1_manager.py +835 -0
  162. lmcache/v1/distributed/l2_adapters/__init__.py +67 -0
  163. lmcache/v1/distributed/l2_adapters/base.py +360 -0
  164. lmcache/v1/distributed/l2_adapters/config.py +385 -0
  165. lmcache/v1/distributed/l2_adapters/factory.py +205 -0
  166. lmcache/v1/distributed/l2_adapters/fs_l2_adapter.py +747 -0
  167. lmcache/v1/distributed/l2_adapters/fs_native_l2_adapter.py +167 -0
  168. lmcache/v1/distributed/l2_adapters/mock_l2_adapter.py +516 -0
  169. lmcache/v1/distributed/l2_adapters/mooncake_store_l2_adapter.py +135 -0
  170. lmcache/v1/distributed/l2_adapters/native_connector_l2_adapter.py +468 -0
  171. lmcache/v1/distributed/l2_adapters/native_plugin_l2_adapter.py +199 -0
  172. lmcache/v1/distributed/l2_adapters/nixl_store_dynamic_l2_adapter.py +831 -0
  173. lmcache/v1/distributed/l2_adapters/nixl_store_l2_adapter.py +983 -0
  174. lmcache/v1/distributed/l2_adapters/plugin_l2_adapter.py +210 -0
  175. lmcache/v1/distributed/l2_adapters/resp_l2_adapter.py +176 -0
  176. lmcache/v1/distributed/memory_manager.py +179 -0
  177. lmcache/v1/distributed/storage_controller.py +39 -0
  178. lmcache/v1/distributed/storage_controllers/__init__.py +43 -0
  179. lmcache/v1/distributed/storage_controllers/eviction_controller.py +242 -0
  180. lmcache/v1/distributed/storage_controllers/prefetch_controller.py +830 -0
  181. lmcache/v1/distributed/storage_controllers/prefetch_policy.py +193 -0
  182. lmcache/v1/distributed/storage_controllers/store_controller.py +452 -0
  183. lmcache/v1/distributed/storage_controllers/store_policy.py +213 -0
  184. lmcache/v1/distributed/storage_manager.py +532 -0
  185. lmcache/v1/event_manager.py +145 -0
  186. lmcache/v1/exceptions/__init__.py +16 -0
  187. lmcache/v1/gpu_connector/__init__.py +126 -0
  188. lmcache/v1/gpu_connector/gpu_connectors.py +1906 -0
  189. lmcache/v1/gpu_connector/gpu_ops.py +85 -0
  190. lmcache/v1/gpu_connector/hpu_connector.py +326 -0
  191. lmcache/v1/gpu_connector/mock_gpu_connector.py +67 -0
  192. lmcache/v1/gpu_connector/utils.py +890 -0
  193. lmcache/v1/gpu_connector/xpu_connectors.py +916 -0
  194. lmcache/v1/health_monitor/__init__.py +1 -0
  195. lmcache/v1/health_monitor/base.py +587 -0
  196. lmcache/v1/health_monitor/checks/__init__.py +1 -0
  197. lmcache/v1/health_monitor/checks/remote_backend_check.py +304 -0
  198. lmcache/v1/health_monitor/constants.py +36 -0
  199. lmcache/v1/internal_api_server/__init__.py +0 -0
  200. lmcache/v1/internal_api_server/api_registry.py +59 -0
  201. lmcache/v1/internal_api_server/api_server.py +120 -0
  202. lmcache/v1/internal_api_server/common/__init__.py +1 -0
  203. lmcache/v1/internal_api_server/common/env_api.py +22 -0
  204. lmcache/v1/internal_api_server/common/loglevel_api.py +57 -0
  205. lmcache/v1/internal_api_server/common/metrics_api.py +29 -0
  206. lmcache/v1/internal_api_server/common/periodic_thread_api.py +138 -0
  207. lmcache/v1/internal_api_server/common/run_script_api.py +73 -0
  208. lmcache/v1/internal_api_server/common/thread_api.py +63 -0
  209. lmcache/v1/internal_api_server/controller/__init__.py +1 -0
  210. lmcache/v1/internal_api_server/controller/key_stats_api.py +81 -0
  211. lmcache/v1/internal_api_server/controller/worker_info_api.py +136 -0
  212. lmcache/v1/internal_api_server/utils.py +43 -0
  213. lmcache/v1/internal_api_server/vllm/__init__.py +1 -0
  214. lmcache/v1/internal_api_server/vllm/backend_api.py +221 -0
  215. lmcache/v1/internal_api_server/vllm/bypass_api.py +204 -0
  216. lmcache/v1/internal_api_server/vllm/cache_api.py +895 -0
  217. lmcache/v1/internal_api_server/vllm/chunk_statistics_api.py +141 -0
  218. lmcache/v1/internal_api_server/vllm/conf_api.py +147 -0
  219. lmcache/v1/internal_api_server/vllm/freeze_api.py +172 -0
  220. lmcache/v1/internal_api_server/vllm/hot_cache_api.py +184 -0
  221. lmcache/v1/internal_api_server/vllm/inference_api.py +65 -0
  222. lmcache/v1/internal_api_server/vllm/load_fs_chunks_api.py +320 -0
  223. lmcache/v1/internal_api_server/vllm/lookup_api.py +145 -0
  224. lmcache/v1/internal_api_server/vllm/version_api.py +25 -0
  225. lmcache/v1/kv_layer_groups.py +267 -0
  226. lmcache/v1/lazy_memory_allocator.py +284 -0
  227. lmcache/v1/lookup_client/__init__.py +25 -0
  228. lmcache/v1/lookup_client/abstract_client.py +77 -0
  229. lmcache/v1/lookup_client/async_lookup_message.py +50 -0
  230. lmcache/v1/lookup_client/chunk_statistics_lookup_client.py +200 -0
  231. lmcache/v1/lookup_client/factory.py +251 -0
  232. lmcache/v1/lookup_client/hit_limit_lookup_client.py +86 -0
  233. lmcache/v1/lookup_client/lmcache_async_lookup_client.py +407 -0
  234. lmcache/v1/lookup_client/lmcache_lookup_client.py +285 -0
  235. lmcache/v1/lookup_client/lmcache_lookup_client_bypass.py +99 -0
  236. lmcache/v1/lookup_client/mooncake_lookup_client.py +87 -0
  237. lmcache/v1/lookup_client/record_strategies/__init__.py +77 -0
  238. lmcache/v1/lookup_client/record_strategies/base.py +327 -0
  239. lmcache/v1/lookup_client/record_strategies/file_hash.py +130 -0
  240. lmcache/v1/lookup_client/record_strategies/memory_bloom_filter.py +81 -0
  241. lmcache/v1/manager.py +539 -0
  242. lmcache/v1/memory_management.py +2619 -0
  243. lmcache/v1/metadata.py +114 -0
  244. lmcache/v1/mp_observability/AGENTS.override.md +21 -0
  245. lmcache/v1/mp_observability/README.md +204 -0
  246. lmcache/v1/mp_observability/config.py +340 -0
  247. lmcache/v1/mp_observability/event.py +100 -0
  248. lmcache/v1/mp_observability/event_bus.py +313 -0
  249. lmcache/v1/mp_observability/otel_init.py +145 -0
  250. lmcache/v1/mp_observability/subscribers/__init__.py +28 -0
  251. lmcache/v1/mp_observability/subscribers/logging/__init__.py +19 -0
  252. lmcache/v1/mp_observability/subscribers/logging/l1.py +56 -0
  253. lmcache/v1/mp_observability/subscribers/logging/l2.py +73 -0
  254. lmcache/v1/mp_observability/subscribers/logging/lookup_hash.py +209 -0
  255. lmcache/v1/mp_observability/subscribers/logging/mp_server.py +90 -0
  256. lmcache/v1/mp_observability/subscribers/logging/sm.py +59 -0
  257. lmcache/v1/mp_observability/subscribers/metrics/__init__.py +20 -0
  258. lmcache/v1/mp_observability/subscribers/metrics/l0_lifecycle.py +290 -0
  259. lmcache/v1/mp_observability/subscribers/metrics/l1.py +55 -0
  260. lmcache/v1/mp_observability/subscribers/metrics/l1_lifecycle.py +166 -0
  261. lmcache/v1/mp_observability/subscribers/metrics/l2.py +121 -0
  262. lmcache/v1/mp_observability/subscribers/metrics/sm.py +69 -0
  263. lmcache/v1/mp_observability/subscribers/tracing/__init__.py +12 -0
  264. lmcache/v1/mp_observability/subscribers/tracing/mp_server.py +333 -0
  265. lmcache/v1/mp_observability/subscribers/tracing/span_registry.py +148 -0
  266. lmcache/v1/mp_observability/trace/__init__.py +50 -0
  267. lmcache/v1/mp_observability/trace/codecs.py +255 -0
  268. lmcache/v1/mp_observability/trace/decorator.py +147 -0
  269. lmcache/v1/mp_observability/trace/format.py +132 -0
  270. lmcache/v1/mp_observability/trace/lifecycle.py +83 -0
  271. lmcache/v1/mp_observability/trace/reader.py +167 -0
  272. lmcache/v1/mp_observability/trace/recorder.py +300 -0
  273. lmcache/v1/multiprocess/__init__.py +0 -0
  274. lmcache/v1/multiprocess/affinity_pool.py +102 -0
  275. lmcache/v1/multiprocess/blend_server_v2.py +891 -0
  276. lmcache/v1/multiprocess/config.py +253 -0
  277. lmcache/v1/multiprocess/custom_types.py +281 -0
  278. lmcache/v1/multiprocess/futures.py +194 -0
  279. lmcache/v1/multiprocess/gpu_context.py +511 -0
  280. lmcache/v1/multiprocess/http_server.py +235 -0
  281. lmcache/v1/multiprocess/mp_runtime_plugin_launcher.py +130 -0
  282. lmcache/v1/multiprocess/mq.py +732 -0
  283. lmcache/v1/multiprocess/protocol.py +86 -0
  284. lmcache/v1/multiprocess/protocols/README.md +213 -0
  285. lmcache/v1/multiprocess/protocols/__init__.py +127 -0
  286. lmcache/v1/multiprocess/protocols/base.py +89 -0
  287. lmcache/v1/multiprocess/protocols/blend.py +109 -0
  288. lmcache/v1/multiprocess/protocols/blend_v2.py +57 -0
  289. lmcache/v1/multiprocess/protocols/controller.py +53 -0
  290. lmcache/v1/multiprocess/protocols/debug.py +34 -0
  291. lmcache/v1/multiprocess/protocols/engine.py +146 -0
  292. lmcache/v1/multiprocess/protocols/observability.py +39 -0
  293. lmcache/v1/multiprocess/server.py +1134 -0
  294. lmcache/v1/multiprocess/session.py +190 -0
  295. lmcache/v1/multiprocess/token_hasher.py +441 -0
  296. lmcache/v1/offload_server/__init__.py +17 -0
  297. lmcache/v1/offload_server/abstract_server.py +37 -0
  298. lmcache/v1/offload_server/message.py +30 -0
  299. lmcache/v1/offload_server/zmq_server.py +122 -0
  300. lmcache/v1/periodic_thread.py +579 -0
  301. lmcache/v1/pin_monitor.py +246 -0
  302. lmcache/v1/plugin/__init__.py +0 -0
  303. lmcache/v1/plugin/runtime_plugin_launcher.py +211 -0
  304. lmcache/v1/protocol.py +317 -0
  305. lmcache/v1/rpc/__init__.py +17 -0
  306. lmcache/v1/rpc/transport.py +105 -0
  307. lmcache/v1/rpc/zmq_transport.py +213 -0
  308. lmcache/v1/rpc_utils.py +165 -0
  309. lmcache/v1/server/__init__.py +2 -0
  310. lmcache/v1/server/__main__.py +170 -0
  311. lmcache/v1/server/storage_backend/__init__.py +21 -0
  312. lmcache/v1/server/storage_backend/abstract_backend.py +80 -0
  313. lmcache/v1/server/storage_backend/local_backend.py +75 -0
  314. lmcache/v1/server/utils.py +21 -0
  315. lmcache/v1/standalone/__init__.py +1 -0
  316. lmcache/v1/standalone/__main__.py +583 -0
  317. lmcache/v1/standalone/manager.py +80 -0
  318. lmcache/v1/standalone/standalone_service_factory.py +86 -0
  319. lmcache/v1/storage_backend/__init__.py +313 -0
  320. lmcache/v1/storage_backend/abstract_backend.py +445 -0
  321. lmcache/v1/storage_backend/audit_backend.py +233 -0
  322. lmcache/v1/storage_backend/batched_message_sender.py +222 -0
  323. lmcache/v1/storage_backend/cache_policy/__init__.py +45 -0
  324. lmcache/v1/storage_backend/cache_policy/base_policy.py +87 -0
  325. lmcache/v1/storage_backend/cache_policy/fifo.py +58 -0
  326. lmcache/v1/storage_backend/cache_policy/lfu.py +105 -0
  327. lmcache/v1/storage_backend/cache_policy/lru.py +81 -0
  328. lmcache/v1/storage_backend/cache_policy/mru.py +61 -0
  329. lmcache/v1/storage_backend/connector/__init__.py +443 -0
  330. lmcache/v1/storage_backend/connector/audit_adapter.py +77 -0
  331. lmcache/v1/storage_backend/connector/audit_connector.py +320 -0
  332. lmcache/v1/storage_backend/connector/base_connector.py +379 -0
  333. lmcache/v1/storage_backend/connector/blackhole_adapter.py +21 -0
  334. lmcache/v1/storage_backend/connector/blackhole_connector.py +37 -0
  335. lmcache/v1/storage_backend/connector/eic_adapter.py +31 -0
  336. lmcache/v1/storage_backend/connector/eic_connector.py +757 -0
  337. lmcache/v1/storage_backend/connector/external_adapter.py +79 -0
  338. lmcache/v1/storage_backend/connector/fs_adapter.py +51 -0
  339. lmcache/v1/storage_backend/connector/fs_connector.py +403 -0
  340. lmcache/v1/storage_backend/connector/infinistore_adapter.py +56 -0
  341. lmcache/v1/storage_backend/connector/infinistore_connector.py +177 -0
  342. lmcache/v1/storage_backend/connector/instrumented_connector.py +219 -0
  343. lmcache/v1/storage_backend/connector/lm_adapter.py +31 -0
  344. lmcache/v1/storage_backend/connector/lm_connector.py +176 -0
  345. lmcache/v1/storage_backend/connector/mock_adapter.py +57 -0
  346. lmcache/v1/storage_backend/connector/mock_connector.py +349 -0
  347. lmcache/v1/storage_backend/connector/mooncakestore_adapter.py +43 -0
  348. lmcache/v1/storage_backend/connector/mooncakestore_connector.py +614 -0
  349. lmcache/v1/storage_backend/connector/redis_adapter.py +181 -0
  350. lmcache/v1/storage_backend/connector/redis_connector.py +828 -0
  351. lmcache/v1/storage_backend/connector/s3_adapter.py +59 -0
  352. lmcache/v1/storage_backend/connector/s3_connector.py +699 -0
  353. lmcache/v1/storage_backend/connector/sagemaker_hyperpod_adapter.py +233 -0
  354. lmcache/v1/storage_backend/connector/sagemaker_hyperpod_connector.py +987 -0
  355. lmcache/v1/storage_backend/connector/valkey_adapter.py +114 -0
  356. lmcache/v1/storage_backend/connector/valkey_connector.py +627 -0
  357. lmcache/v1/storage_backend/gds_backend.py +1199 -0
  358. lmcache/v1/storage_backend/job_executor/__init__.py +0 -0
  359. lmcache/v1/storage_backend/job_executor/base_executor.py +34 -0
  360. lmcache/v1/storage_backend/job_executor/pq_executor.py +235 -0
  361. lmcache/v1/storage_backend/local_cpu_backend.py +810 -0
  362. lmcache/v1/storage_backend/local_disk_backend.py +656 -0
  363. lmcache/v1/storage_backend/maru_backend.py +734 -0
  364. lmcache/v1/storage_backend/naive_serde/__init__.py +50 -0
  365. lmcache/v1/storage_backend/naive_serde/cachegen_basics.py +133 -0
  366. lmcache/v1/storage_backend/naive_serde/cachegen_decoder.py +135 -0
  367. lmcache/v1/storage_backend/naive_serde/cachegen_encoder.py +83 -0
  368. lmcache/v1/storage_backend/naive_serde/kivi_serde.py +22 -0
  369. lmcache/v1/storage_backend/naive_serde/naive_serde.py +18 -0
  370. lmcache/v1/storage_backend/naive_serde/serde.py +37 -0
  371. lmcache/v1/storage_backend/native_clients/connector_client_base.py +165 -0
  372. lmcache/v1/storage_backend/native_clients/resp_client.py +35 -0
  373. lmcache/v1/storage_backend/nixl_storage_backend.py +1400 -0
  374. lmcache/v1/storage_backend/p2p_backend.py +788 -0
  375. lmcache/v1/storage_backend/path_sharder.py +117 -0
  376. lmcache/v1/storage_backend/pd_backend.py +646 -0
  377. lmcache/v1/storage_backend/plugins/dax_backend.py +1443 -0
  378. lmcache/v1/storage_backend/plugins/rust_raw_block_backend.py +1361 -0
  379. lmcache/v1/storage_backend/remote_backend.py +624 -0
  380. lmcache/v1/storage_backend/resp_client.py +227 -0
  381. lmcache/v1/storage_backend/storage_backend_listener.py +19 -0
  382. lmcache/v1/storage_backend/storage_manager.py +1352 -0
  383. lmcache/v1/system_detection.py +110 -0
  384. lmcache/v1/token_database.py +551 -0
  385. lmcache/v1/transfer_channel/__init__.py +83 -0
  386. lmcache/v1/transfer_channel/abstract.py +285 -0
  387. lmcache/v1/transfer_channel/mock_memory_channel.py +156 -0
  388. lmcache/v1/transfer_channel/nixl_channel.py +639 -0
  389. lmcache/v1/transfer_channel/py_socket_channel.py +260 -0
  390. lmcache/v1/transfer_channel/transfer_utils.py +63 -0
  391. lmcache/v1/utils/__init__.py +1 -0
  392. lmcache/v1/utils/bloom_filter.py +109 -0
  393. lmcache/v1/utils/cache_utils.py +125 -0
  394. lmcache_cli-0.4.5.dev0.dist-info/METADATA +185 -0
  395. lmcache_cli-0.4.5.dev0.dist-info/RECORD +399 -0
  396. lmcache_cli-0.4.5.dev0.dist-info/WHEEL +5 -0
  397. lmcache_cli-0.4.5.dev0.dist-info/entry_points.txt +2 -0
  398. lmcache_cli-0.4.5.dev0.dist-info/licenses/LICENSE +201 -0
  399. lmcache_cli-0.4.5.dev0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,352 @@
1
+ # SPDX-License-Identifier: Apache-2.0
2
+ """Centralized config item definitions for interactive configuration.
3
+
4
+ Each ``ConfigItem`` declaratively describes one configurable parameter:
5
+ its key, display name, description, input type, default, and when it
6
+ should be shown. The ``ALL_ITEMS`` list is the single source of truth
7
+ for descriptions, ordering, and defaults.
8
+ """
9
+
10
+ # Standard
11
+ from collections.abc import Callable
12
+ from dataclasses import dataclass, field
13
+ from typing import Any
14
+
15
+ # ---------------------------------------------------------------------------
16
+ # Phases
17
+ # ---------------------------------------------------------------------------
18
+
19
+ PHASE_REQUIRED = 1
20
+ PHASE_GENERAL = 2
21
+ PHASE_WORKLOAD = 3
22
+
23
+
24
+ # ---------------------------------------------------------------------------
25
+ # ConfigItem
26
+ # ---------------------------------------------------------------------------
27
+
28
+
29
+ @dataclass
30
+ class ConfigItem:
31
+ """Declarative description of a single configurable parameter.
32
+
33
+ Attributes:
34
+ key: State dict key (matches argparse attr name, e.g., ``"engine_url"``).
35
+ display_name: Heading shown in the prompt.
36
+ description: One-sentence explanation shown below the heading.
37
+ input_type: One of ``"text"``, ``"int"``, ``"float"``, ``"bool"``,
38
+ ``"choice"``.
39
+ default: Default value. ``None`` means required (no default).
40
+ required: If True, this item must have a value before the benchmark
41
+ can start.
42
+ choices: For ``"choice"`` type — list of ``(value, description)`` tuples.
43
+ condition: Callable ``(state_dict) -> bool`` that determines whether
44
+ this item should be shown. ``None`` means always shown.
45
+ phase: Which interactive phase this item belongs to.
46
+ """
47
+
48
+ key: str
49
+ display_name: str
50
+ description: str
51
+ input_type: str # "text", "int", "float", "bool", "choice"
52
+ default: Any = None
53
+ required: bool = False
54
+ choices: list[tuple[str, str]] = field(default_factory=list)
55
+ condition: Callable[[dict[str, Any]], bool] | None = None
56
+ phase: int = PHASE_GENERAL
57
+
58
+
59
+ # ---------------------------------------------------------------------------
60
+ # Condition helpers
61
+ # ---------------------------------------------------------------------------
62
+
63
+
64
+ def _has_lmcache(state: dict[str, Any]) -> bool:
65
+ """Show this item only when the user said they have LMCache."""
66
+ return bool(state.get("has_lmcache"))
67
+
68
+
69
+ def _no_lmcache_url(state: dict[str, Any]) -> bool:
70
+ """Show this item only when lmcache_url is not set."""
71
+ return not state.get("lmcache_url")
72
+
73
+
74
+ def _workload_is(name: str) -> Callable[[dict[str, Any]], bool]:
75
+ """Return a condition that checks the workload value."""
76
+
77
+ def check(state: dict[str, Any]) -> bool:
78
+ return state.get("workload") == name
79
+
80
+ return check
81
+
82
+
83
+ # ---------------------------------------------------------------------------
84
+ # ALL_ITEMS — the centralized registry
85
+ # ---------------------------------------------------------------------------
86
+
87
+ ALL_ITEMS: list[ConfigItem] = [
88
+ # ── Phase 1: Required ─────────────────────────────────────────────
89
+ ConfigItem(
90
+ key="engine_url",
91
+ display_name="Engine URL",
92
+ description=(
93
+ "URL of the inference engine. "
94
+ "Set OPENAI_API_KEY env var if authentication is needed."
95
+ ),
96
+ input_type="text",
97
+ default="http://localhost:8000",
98
+ required=True,
99
+ phase=PHASE_REQUIRED,
100
+ ),
101
+ ConfigItem(
102
+ key="workload",
103
+ display_name="Workload",
104
+ description="The type of benchmark workload to run.",
105
+ input_type="choice",
106
+ default=None,
107
+ required=True,
108
+ choices=[
109
+ (
110
+ "long-doc-permutator",
111
+ "Query the same set of long documents with different orders",
112
+ ),
113
+ ("long-doc-qa", "Repeated Q&A over long documents (tests KV cache reuse)"),
114
+ ("multi-round-chat", "Multi-turn chat with stateful sessions"),
115
+ ("random-prefill", "Prefill-only requests fired simultaneously"),
116
+ ],
117
+ phase=PHASE_REQUIRED,
118
+ ),
119
+ ConfigItem(
120
+ key="has_lmcache",
121
+ display_name="LMCache Server",
122
+ description=(
123
+ "Do you have a running LMCache server? "
124
+ "It can auto-detect KV cache size information."
125
+ ),
126
+ input_type="bool",
127
+ default=True,
128
+ required=False,
129
+ phase=PHASE_REQUIRED,
130
+ ),
131
+ ConfigItem(
132
+ key="lmcache_url",
133
+ display_name="LMCache Server URL",
134
+ description="URL of the running LMCache HTTP server.",
135
+ input_type="text",
136
+ default="http://localhost:8080",
137
+ required=False,
138
+ condition=_has_lmcache,
139
+ phase=PHASE_REQUIRED,
140
+ ),
141
+ ConfigItem(
142
+ key="tokens_per_gb_kvcache",
143
+ display_name="Tokens per GB KV cache",
144
+ description=(
145
+ "How many tokens fit in 1 GB of KV cache for your model.\n"
146
+ " If using vLLM, look for these lines in the startup log:\n"
147
+ ' "Available KV cache memory: XX.XX GiB"\n'
148
+ ' "GPU KV cache size: XXX,XXX tokens"\n'
149
+ " Then compute: tokens_per_gb = "
150
+ "GPU_KV_cache_tokens / Available_KV_cache_GiB"
151
+ ),
152
+ input_type="int",
153
+ default=None,
154
+ required=True,
155
+ condition=_no_lmcache_url,
156
+ phase=PHASE_REQUIRED,
157
+ ),
158
+ # ── Phase 2: General ──────────────────────────────────────────────
159
+ ConfigItem(
160
+ key="model",
161
+ display_name="Model name",
162
+ description=(
163
+ "The model served by the engine. "
164
+ "Leave empty to auto-detect from the engine."
165
+ ),
166
+ input_type="text",
167
+ default="",
168
+ phase=PHASE_GENERAL,
169
+ ),
170
+ ConfigItem(
171
+ key="kv_cache_volume",
172
+ display_name="KV cache volume (GB)",
173
+ description="Target active KV cache size for the benchmark.",
174
+ input_type="float",
175
+ default=100.0,
176
+ phase=PHASE_GENERAL,
177
+ ),
178
+ # ── Phase 3: long-doc-permutator ─────────────────────────────────
179
+ ConfigItem(
180
+ key="ldp_num_contexts",
181
+ display_name="Number of contexts",
182
+ description="Number of unique context documents to generate.",
183
+ input_type="int",
184
+ default=5,
185
+ condition=_workload_is("long-doc-permutator"),
186
+ phase=PHASE_WORKLOAD,
187
+ ),
188
+ ConfigItem(
189
+ key="ldp_context_length",
190
+ display_name="Context length (tokens)",
191
+ description="Token length of each context document.",
192
+ input_type="int",
193
+ default=5000,
194
+ condition=_workload_is("long-doc-permutator"),
195
+ phase=PHASE_WORKLOAD,
196
+ ),
197
+ ConfigItem(
198
+ key="ldp_system_prompt_length",
199
+ display_name="System prompt length (tokens)",
200
+ description="Token length of the shared system prompt. Use 0 for none.",
201
+ input_type="int",
202
+ default=1000,
203
+ condition=_workload_is("long-doc-permutator"),
204
+ phase=PHASE_WORKLOAD,
205
+ ),
206
+ ConfigItem(
207
+ key="ldp_num_permutations",
208
+ display_name="Number of permutations",
209
+ description="Distinct permutations to send. Capped at N! (N = num_contexts).",
210
+ input_type="int",
211
+ default=10,
212
+ condition=_workload_is("long-doc-permutator"),
213
+ phase=PHASE_WORKLOAD,
214
+ ),
215
+ ConfigItem(
216
+ key="ldp_num_inflight_requests",
217
+ display_name="Max inflight requests",
218
+ description="Maximum concurrent in-flight requests.",
219
+ input_type="int",
220
+ default=1,
221
+ condition=_workload_is("long-doc-permutator"),
222
+ phase=PHASE_WORKLOAD,
223
+ ),
224
+ # ── Phase 3: long-doc-qa ──────────────────────────────────────────
225
+ ConfigItem(
226
+ key="ldqa_document_length",
227
+ display_name="Document length (tokens)",
228
+ description="Token length of each synthetic document.",
229
+ input_type="int",
230
+ default=10000,
231
+ condition=_workload_is("long-doc-qa"),
232
+ phase=PHASE_WORKLOAD,
233
+ ),
234
+ ConfigItem(
235
+ key="ldqa_query_per_document",
236
+ display_name="Queries per document",
237
+ description="Number of questions asked per document.",
238
+ input_type="int",
239
+ default=2,
240
+ condition=_workload_is("long-doc-qa"),
241
+ phase=PHASE_WORKLOAD,
242
+ ),
243
+ ConfigItem(
244
+ key="ldqa_shuffle_policy",
245
+ display_name="Shuffle policy",
246
+ description="How benchmark requests are ordered.",
247
+ input_type="choice",
248
+ default="random",
249
+ choices=[
250
+ ("random", "Shuffle all (doc, query) pairs randomly"),
251
+ ("tile", "Process queries round by round across all documents"),
252
+ ],
253
+ condition=_workload_is("long-doc-qa"),
254
+ phase=PHASE_WORKLOAD,
255
+ ),
256
+ ConfigItem(
257
+ key="ldqa_num_inflight_requests",
258
+ display_name="Max inflight requests",
259
+ description="Maximum concurrent in-flight requests.",
260
+ input_type="int",
261
+ default=3,
262
+ condition=_workload_is("long-doc-qa"),
263
+ phase=PHASE_WORKLOAD,
264
+ ),
265
+ # ── Phase 3: multi-round-chat ─────────────────────────────────────
266
+ ConfigItem(
267
+ key="mrc_shared_prompt_length",
268
+ display_name="System prompt length (tokens)",
269
+ description="Token length of the system prompt per session.",
270
+ input_type="int",
271
+ default=2000,
272
+ condition=_workload_is("multi-round-chat"),
273
+ phase=PHASE_WORKLOAD,
274
+ ),
275
+ ConfigItem(
276
+ key="mrc_chat_history_length",
277
+ display_name="Chat history length (tokens)",
278
+ description="Token length of pre-filled conversation history.",
279
+ input_type="int",
280
+ default=10000,
281
+ condition=_workload_is("multi-round-chat"),
282
+ phase=PHASE_WORKLOAD,
283
+ ),
284
+ ConfigItem(
285
+ key="mrc_user_input_length",
286
+ display_name="User input length (tokens)",
287
+ description="Tokens per user query in each round.",
288
+ input_type="int",
289
+ default=50,
290
+ condition=_workload_is("multi-round-chat"),
291
+ phase=PHASE_WORKLOAD,
292
+ ),
293
+ ConfigItem(
294
+ key="mrc_output_length",
295
+ display_name="Output length (tokens)",
296
+ description="Max tokens to generate per response.",
297
+ input_type="int",
298
+ default=200,
299
+ condition=_workload_is("multi-round-chat"),
300
+ phase=PHASE_WORKLOAD,
301
+ ),
302
+ ConfigItem(
303
+ key="mrc_qps",
304
+ display_name="Queries per second",
305
+ description="Target request dispatch rate.",
306
+ input_type="float",
307
+ default=1.0,
308
+ condition=_workload_is("multi-round-chat"),
309
+ phase=PHASE_WORKLOAD,
310
+ ),
311
+ ConfigItem(
312
+ key="mrc_duration",
313
+ display_name="Duration (seconds)",
314
+ description="How long the benchmark runs.",
315
+ input_type="float",
316
+ default=60.0,
317
+ condition=_workload_is("multi-round-chat"),
318
+ phase=PHASE_WORKLOAD,
319
+ ),
320
+ # ── Phase 3: random-prefill ───────────────────────────────────────
321
+ ConfigItem(
322
+ key="rp_request_length",
323
+ display_name="Request length (tokens)",
324
+ description="Token length of each prefill request.",
325
+ input_type="int",
326
+ default=10000,
327
+ condition=_workload_is("random-prefill"),
328
+ phase=PHASE_WORKLOAD,
329
+ ),
330
+ ConfigItem(
331
+ key="rp_num_requests",
332
+ display_name="Number of requests",
333
+ description="Total prefill requests to fire simultaneously.",
334
+ input_type="int",
335
+ default=50,
336
+ condition=_workload_is("random-prefill"),
337
+ phase=PHASE_WORKLOAD,
338
+ ),
339
+ ]
340
+
341
+
342
+ def get_items_by_phase(phase: int) -> list[ConfigItem]:
343
+ """Return all items belonging to a given phase."""
344
+ return [item for item in ALL_ITEMS if item.phase == phase]
345
+
346
+
347
+ def get_item(key: str) -> ConfigItem:
348
+ """Look up a ConfigItem by key. Raises KeyError if not found."""
349
+ for item in ALL_ITEMS:
350
+ if item.key == key:
351
+ return item
352
+ raise KeyError(f"No ConfigItem with key {key!r}")
@@ -0,0 +1,327 @@
1
+ # SPDX-License-Identifier: Apache-2.0
2
+ """Intermediate state tracker for interactive configuration.
3
+
4
+ ``InteractiveState`` holds the partially-configured benchmark parameters,
5
+ can be initialized from CLI args or a saved JSON file, and can be
6
+ converted to the ``argparse.Namespace`` that ``_bench_engine()`` expects.
7
+ """
8
+
9
+ # Standard
10
+ from typing import Any
11
+ import argparse
12
+ import json
13
+
14
+ # First Party
15
+ from lmcache.cli.commands.bench.engine_bench.interactive.schema import (
16
+ ALL_ITEMS,
17
+ PHASE_GENERAL,
18
+ PHASE_REQUIRED,
19
+ PHASE_WORKLOAD,
20
+ ConfigItem,
21
+ )
22
+
23
+ # Keys that exist on argparse.Namespace but are NOT part of the interactive
24
+ # config item registry (operational flags, handled separately).
25
+ _OUTPUT_KEYS = ("output_dir", "seed", "no_csv", "export_csv", "json", "quiet")
26
+
27
+ # Keys used only during the interactive flow, never serialized or
28
+ # passed to the orchestrator.
29
+ _INTERACTIVE_ONLY_KEYS = {"has_lmcache"}
30
+
31
+ # Keys excluded from exported JSON configs. These are either
32
+ # environment-specific (engine_url, lmcache_url) or interactive-only.
33
+ _EXPORT_EXCLUDED_KEYS = _INTERACTIVE_ONLY_KEYS | {"engine_url", "lmcache_url"}
34
+
35
+ # Mapping from ConfigItem.key to the argparse attribute name when they differ.
36
+ # Most keys match directly; these are the exceptions.
37
+ _KEY_TO_ATTR: dict[str, str] = {
38
+ "kv_cache_volume": "kv_cache_volume",
39
+ "tokens_per_gb_kvcache": "tokens_per_gb_kvcache",
40
+ }
41
+
42
+ # argparse attribute names where the CLI default is None (meaning "not set"),
43
+ # versus attributes where a non-None argparse default is the real default
44
+ # (e.g., kv_cache_volume defaults to 100.0).
45
+ _ARGPARSE_NONE_MEANS_UNSET = {
46
+ "engine_url",
47
+ "workload",
48
+ "model",
49
+ "lmcache_url",
50
+ "tokens_per_gb_kvcache",
51
+ }
52
+
53
+
54
+ class InteractiveState:
55
+ """Tracks which config items have been set and their values.
56
+
57
+ Keys present in ``_values`` are considered "set". Missing keys are
58
+ "unset" and will either be prompted for or filled with defaults.
59
+ """
60
+
61
+ def __init__(self) -> None:
62
+ self._values: dict[str, Any] = {}
63
+
64
+ # ------------------------------------------------------------------
65
+ # Basic accessors
66
+ # ------------------------------------------------------------------
67
+
68
+ def is_set(self, key: str) -> bool:
69
+ return key in self._values
70
+
71
+ def get(self, key: str, default: Any = None) -> Any:
72
+ return self._values.get(key, default)
73
+
74
+ def set(self, key: str, value: Any) -> None:
75
+ self._values[key] = value
76
+
77
+ @property
78
+ def values(self) -> dict[str, Any]:
79
+ return dict(self._values)
80
+
81
+ # ------------------------------------------------------------------
82
+ # Readiness checks
83
+ # ------------------------------------------------------------------
84
+
85
+ def is_ready(self) -> bool:
86
+ """True when all required items (whose conditions are met) have values."""
87
+ for item in ALL_ITEMS:
88
+ if not item.required:
89
+ continue
90
+ if not self._condition_met(item):
91
+ continue
92
+ if not self.is_set(item.key):
93
+ return False
94
+ return True
95
+
96
+ def get_missing_required(self) -> list[ConfigItem]:
97
+ """Return phase-1 items that still need user input.
98
+
99
+ Includes both required items that are unset, and non-required
100
+ phase-1 items (like ``lmcache_url``) that are relevant because
101
+ a downstream required item (``tokens_per_gb_kvcache``) is unset.
102
+ """
103
+ missing: list[ConfigItem] = []
104
+ for item in ALL_ITEMS:
105
+ if item.phase != PHASE_REQUIRED:
106
+ continue
107
+ if not self._condition_met(item):
108
+ continue
109
+ if self.is_set(item.key):
110
+ continue
111
+ if item.required:
112
+ missing.append(item)
113
+ elif item.key in ("has_lmcache", "lmcache_url") and not self.is_set(
114
+ "tokens_per_gb_kvcache"
115
+ ):
116
+ # Only ask about LMCache when tokens_per_gb is needed
117
+ missing.append(item)
118
+ return missing
119
+
120
+ def get_general_items(self) -> list[ConfigItem]:
121
+ """Return phase-2 (general) items that are not yet set."""
122
+ return [
123
+ item
124
+ for item in ALL_ITEMS
125
+ if item.phase == PHASE_GENERAL
126
+ and not self.is_set(item.key)
127
+ and self._condition_met(item)
128
+ ]
129
+
130
+ def get_workload_items(self) -> list[ConfigItem]:
131
+ """Return phase-3 (workload-specific) items whose conditions are met."""
132
+ return [
133
+ item
134
+ for item in ALL_ITEMS
135
+ if item.phase == PHASE_WORKLOAD and self._condition_met(item)
136
+ ]
137
+
138
+ def has_unconfigured_general(self) -> bool:
139
+ """True if there are general items the user hasn't explicitly set."""
140
+ return len(self.get_general_items()) > 0
141
+
142
+ def has_workload_items(self) -> bool:
143
+ """True if there are workload-specific items to configure."""
144
+ return len(self.get_workload_items()) > 0
145
+
146
+ def workload_items_all_default(self) -> bool:
147
+ """True if no workload-specific items have been explicitly set."""
148
+ for item in ALL_ITEMS:
149
+ if item.phase != PHASE_WORKLOAD:
150
+ continue
151
+ if not self._condition_met(item):
152
+ continue
153
+ if self.is_set(item.key):
154
+ return False
155
+ return True
156
+
157
+ # ------------------------------------------------------------------
158
+ # Defaults
159
+ # ------------------------------------------------------------------
160
+
161
+ def fill_defaults(self) -> None:
162
+ """Set all unset items (whose conditions are met) to their defaults."""
163
+ for item in ALL_ITEMS:
164
+ if self.is_set(item.key):
165
+ continue
166
+ if not self._condition_met(item):
167
+ continue
168
+ if item.default is not None:
169
+ self._values[item.key] = item.default
170
+
171
+ # ------------------------------------------------------------------
172
+ # Conversion: CLI args ↔ InteractiveState
173
+ # ------------------------------------------------------------------
174
+
175
+ @classmethod
176
+ def from_cli_args(cls, args: argparse.Namespace) -> "InteractiveState":
177
+ """Build state from parsed CLI arguments.
178
+
179
+ Only sets values that the user explicitly provided (i.e., not the
180
+ argparse default). This lets us distinguish "user set
181
+ ``--kv-cache-volume 100``" from "user didn't touch it".
182
+ """
183
+ state = cls()
184
+ for item in ALL_ITEMS:
185
+ if item.key in _INTERACTIVE_ONLY_KEYS:
186
+ continue
187
+ attr = _KEY_TO_ATTR.get(item.key, item.key)
188
+ value = getattr(args, attr, None)
189
+ if value is None:
190
+ continue
191
+ # For keys where argparse default is None, any non-None value
192
+ # means the user set it.
193
+ if attr in _ARGPARSE_NONE_MEANS_UNSET:
194
+ state._values[item.key] = value
195
+ continue
196
+ # For keys with real argparse defaults, we can't easily tell
197
+ # if the user typed --kv-cache-volume 100 vs it being the
198
+ # default. We mark it as "set" only if it differs from the
199
+ # schema default. This is imperfect but good enough — the
200
+ # worst case is we re-prompt for a value the user explicitly
201
+ # set to the default.
202
+ if item.default is not None and value == item.default:
203
+ continue
204
+ state._values[item.key] = value
205
+
206
+ # Derive has_lmcache from lmcache_url if provided via CLI
207
+ if state.is_set("lmcache_url"):
208
+ state._values["has_lmcache"] = True
209
+
210
+ return state
211
+
212
+ def to_namespace(self) -> argparse.Namespace:
213
+ """Convert to an ``argparse.Namespace`` compatible with ``_bench_engine``.
214
+
215
+ Fills defaults for any unset items, then builds the namespace
216
+ with the attribute names that ``parse_args_to_config()`` and
217
+ ``create_workload()`` expect.
218
+ """
219
+ self.fill_defaults()
220
+ ns = argparse.Namespace()
221
+
222
+ # Map state keys to namespace attributes
223
+ for item in ALL_ITEMS:
224
+ if item.key in _INTERACTIVE_ONLY_KEYS:
225
+ continue
226
+ attr = _KEY_TO_ATTR.get(item.key, item.key)
227
+ # Only fall back to schema default if the item's condition is met.
228
+ # This prevents lmcache_url's default from leaking when
229
+ # has_lmcache is not set.
230
+ if item.key in self._values:
231
+ value = self._values[item.key]
232
+ elif self._condition_met(item):
233
+ value = item.default
234
+ else:
235
+ value = None
236
+ setattr(ns, attr, value)
237
+
238
+ # Output settings (not in the interactive registry)
239
+ ns.output_dir = self._values.get("output_dir", ".")
240
+ ns.seed = self._values.get("seed", 42)
241
+ ns.no_csv = self._values.get("no_csv", False)
242
+ ns.json = self._values.get("export_json", False)
243
+ ns.quiet = self._values.get("quiet", False)
244
+ ns.bench_target = "engine"
245
+
246
+ # Ensure format/output attrs exist for create_metrics
247
+ if not hasattr(ns, "format"):
248
+ ns.format = None
249
+ if not hasattr(ns, "output"):
250
+ ns.output = None
251
+
252
+ return ns
253
+
254
+ # ------------------------------------------------------------------
255
+ # Conversion: JSON ↔ InteractiveState
256
+ # ------------------------------------------------------------------
257
+
258
+ def to_json(self) -> dict[str, Any]:
259
+ """Serialize to a JSON-compatible dict for config export.
260
+
261
+ Excludes environment-specific keys (``engine_url``,
262
+ ``lmcache_url``) and interactive-only keys so the exported
263
+ config is portable and works without an LMCache server.
264
+ """
265
+ self.fill_defaults()
266
+ return {k: v for k, v in self._values.items() if k not in _EXPORT_EXCLUDED_KEYS}
267
+
268
+ @classmethod
269
+ def from_json(cls, data: dict[str, Any]) -> "InteractiveState":
270
+ """Load from a saved config JSON dict."""
271
+ state = cls()
272
+ for key, value in data.items():
273
+ state._values[key] = value
274
+ return state
275
+
276
+ def save_json(self, path: str) -> None:
277
+ """Export the current state to a JSON file."""
278
+ with open(path, "w") as f:
279
+ json.dump(self.to_json(), f, indent=2)
280
+ f.write("\n")
281
+
282
+ @classmethod
283
+ def load_json(cls, path: str) -> "InteractiveState":
284
+ """Load state from a JSON config file."""
285
+ with open(path) as f:
286
+ data = json.load(f)
287
+ return cls.from_json(data)
288
+
289
+ def merge_cli_args(self, args: argparse.Namespace) -> None:
290
+ """Merge CLI args on top of existing state (CLI args win)."""
291
+ cli_state = InteractiveState.from_cli_args(args)
292
+ self._values.update(cli_state._values)
293
+
294
+ # ------------------------------------------------------------------
295
+ # Summary
296
+ # ------------------------------------------------------------------
297
+
298
+ def summary_lines(self) -> list[tuple[str, str]]:
299
+ """Return ``(label, value_str)`` pairs for the config summary."""
300
+ lines: list[tuple[str, str]] = []
301
+ for item in ALL_ITEMS:
302
+ if item.key in _INTERACTIVE_ONLY_KEYS:
303
+ continue
304
+ if not self._condition_met(item):
305
+ continue
306
+ value = self._values.get(item.key, item.default)
307
+ if value is None or value == "":
308
+ if item.key == "model":
309
+ display = "(auto-detect)"
310
+ elif item.key == "lmcache_url":
311
+ continue # skip empty lmcache_url
312
+ else:
313
+ display = "(not set)"
314
+ else:
315
+ display = str(value)
316
+ lines.append((item.display_name, display))
317
+ return lines
318
+
319
+ # ------------------------------------------------------------------
320
+ # Internal
321
+ # ------------------------------------------------------------------
322
+
323
+ def _condition_met(self, item: ConfigItem) -> bool:
324
+ """Check whether an item's condition is satisfied."""
325
+ if item.condition is None:
326
+ return True
327
+ return item.condition(self._values)