levers9 0.1.265__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (113) hide show
  1. __init__.py +0 -0
  2. levers9/__init__.py +77 -0
  3. levers9/__main__.py +3 -0
  4. levers9/abstractions/__init__.py +0 -0
  5. levers9/abstractions/base/__init__.py +77 -0
  6. levers9/abstractions/base/capacity.py +503 -0
  7. levers9/abstractions/base/container.py +150 -0
  8. levers9/abstractions/base/runner.py +819 -0
  9. levers9/abstractions/base/utils.py +85 -0
  10. levers9/abstractions/endpoint.py +648 -0
  11. levers9/abstractions/experimental/__init__.py +5 -0
  12. levers9/abstractions/experimental/bot/__init__.py +0 -0
  13. levers9/abstractions/experimental/bot/bot.py +402 -0
  14. levers9/abstractions/experimental/bot/marker.py +50 -0
  15. levers9/abstractions/experimental/bot/types.py +243 -0
  16. levers9/abstractions/experimental/signal.py +110 -0
  17. levers9/abstractions/function.py +445 -0
  18. levers9/abstractions/image.py +912 -0
  19. levers9/abstractions/integrations/__init__.py +4 -0
  20. levers9/abstractions/integrations/fastmcp.py +217 -0
  21. levers9/abstractions/integrations/vllm.py +477 -0
  22. levers9/abstractions/map.py +135 -0
  23. levers9/abstractions/mixins.py +214 -0
  24. levers9/abstractions/output.py +350 -0
  25. levers9/abstractions/pod.py +509 -0
  26. levers9/abstractions/queue.py +118 -0
  27. levers9/abstractions/sandbox.py +4270 -0
  28. levers9/abstractions/service.py +374 -0
  29. levers9/abstractions/shell.py +601 -0
  30. levers9/abstractions/taskqueue.py +303 -0
  31. levers9/abstractions/volume.py +159 -0
  32. levers9/aio.py +10 -0
  33. levers9/channel.py +325 -0
  34. levers9/cli/__init__.py +0 -0
  35. levers9/cli/config.py +207 -0
  36. levers9/cli/container.py +205 -0
  37. levers9/cli/database.py +1075 -0
  38. levers9/cli/deployment.py +638 -0
  39. levers9/cli/dev.py +72 -0
  40. levers9/cli/disk.py +263 -0
  41. levers9/cli/extraclick.py +585 -0
  42. levers9/cli/llm.py +269 -0
  43. levers9/cli/machine.py +643 -0
  44. levers9/cli/machine_format.py +187 -0
  45. levers9/cli/main.py +135 -0
  46. levers9/cli/pool.py +863 -0
  47. levers9/cli/run.py +153 -0
  48. levers9/cli/secret.py +144 -0
  49. levers9/cli/serve.py +67 -0
  50. levers9/cli/shell.py +80 -0
  51. levers9/cli/task.py +151 -0
  52. levers9/cli/token.py +194 -0
  53. levers9/cli/volume.py +485 -0
  54. levers9/cli/worker.py +216 -0
  55. levers9/cli/worker_management.py +55 -0
  56. levers9/client/__init__.py +50 -0
  57. levers9/client/client.py +168 -0
  58. levers9/client/deployment.py +72 -0
  59. levers9/client/task.py +127 -0
  60. levers9/clients/__init__.py +0 -0
  61. levers9/clients/bot/__init__.py +159 -0
  62. levers9/clients/disk/__init__.py +139 -0
  63. levers9/clients/endpoint/__init__.py +46 -0
  64. levers9/clients/function/__init__.py +140 -0
  65. levers9/clients/gateway/__init__.py +2545 -0
  66. levers9/clients/google/__init__.py +0 -0
  67. levers9/clients/google/api/__init__.py +379 -0
  68. levers9/clients/image/__init__.py +108 -0
  69. levers9/clients/map/__init__.py +119 -0
  70. levers9/clients/output/__init__.py +112 -0
  71. levers9/clients/pod/__init__.py +597 -0
  72. levers9/clients/secret/__init__.py +140 -0
  73. levers9/clients/shell/__init__.py +72 -0
  74. levers9/clients/signal/__init__.py +84 -0
  75. levers9/clients/simplequeue/__init__.py +115 -0
  76. levers9/clients/taskqueue/__init__.py +162 -0
  77. levers9/clients/types/__init__.py +340 -0
  78. levers9/clients/volume/__init__.py +363 -0
  79. levers9/config.py +262 -0
  80. levers9/env.py +94 -0
  81. levers9/exceptions.py +173 -0
  82. levers9/integrations.py +3 -0
  83. levers9/logging.py +175 -0
  84. levers9/middleware.py +382 -0
  85. levers9/multipart.py +785 -0
  86. levers9/runner/__init__.py +0 -0
  87. levers9/runner/bot/__init__.py +0 -0
  88. levers9/runner/bot/transition.py +254 -0
  89. levers9/runner/common.py +709 -0
  90. levers9/runner/container.py +98 -0
  91. levers9/runner/endpoint.py +331 -0
  92. levers9/runner/function.py +353 -0
  93. levers9/runner/serve.py +129 -0
  94. levers9/runner/taskqueue.py +416 -0
  95. levers9/schema.py +566 -0
  96. levers9/sync.py +301 -0
  97. levers9/terminal.py +520 -0
  98. levers9/type.py +447 -0
  99. levers9/utils.py +131 -0
  100. levers9/vendor/pathspec/__init__.py +76 -0
  101. levers9/vendor/pathspec/_meta.py +58 -0
  102. levers9/vendor/pathspec/gitignore.py +157 -0
  103. levers9/vendor/pathspec/pathspec.py +394 -0
  104. levers9/vendor/pathspec/pattern.py +213 -0
  105. levers9/vendor/pathspec/patterns/__init__.py +11 -0
  106. levers9/vendor/pathspec/patterns/gitwildmatch.py +421 -0
  107. levers9/vendor/pathspec/py.typed +1 -0
  108. levers9/vendor/pathspec/util.py +792 -0
  109. levers9-0.1.265.dist-info/METADATA +27 -0
  110. levers9-0.1.265.dist-info/RECORD +113 -0
  111. levers9-0.1.265.dist-info/WHEEL +5 -0
  112. levers9-0.1.265.dist-info/entry_points.txt +2 -0
  113. levers9-0.1.265.dist-info/top_level.txt +2 -0
__init__.py ADDED
File without changes
levers9/__init__.py ADDED
@@ -0,0 +1,77 @@
1
+ from importlib import import_module
2
+
3
+
4
+ _EXPORTS = {
5
+ "Map": (".abstractions.map", "Map"),
6
+ "Image": (".abstractions.image", "Image"),
7
+ "Queue": (".abstractions.queue", "SimpleQueue"),
8
+ "Volume": (".abstractions.volume", "Volume"),
9
+ "CloudBucket": (".abstractions.volume", "CloudBucket"),
10
+ "CloudBucketConfig": (".abstractions.volume", "CloudBucketConfig"),
11
+ "task_queue": (".abstractions.taskqueue", "TaskQueue"),
12
+ "function": (".abstractions.function", "Function"),
13
+ "endpoint": (".abstractions.endpoint", "Endpoint"),
14
+ "asgi": (".abstractions.endpoint", "ASGI"),
15
+ "realtime": (".abstractions.endpoint", "RealtimeASGI"),
16
+ "Container": (".abstractions.base.container", "Container"),
17
+ "env": (".env", None),
18
+ "GpuType": (".type", "GpuType"),
19
+ "DatabaseServingConfig": (".type", "DatabaseServingConfig"),
20
+ "DurableDisk": (".type", "DurableDisk"),
21
+ "LLMConfig": (".type", "LLMConfig"),
22
+ "LLMTokenPressureAutoscaler": (".type", "LLMTokenPressureAutoscaler"),
23
+ "Pool": (".type", "Pool"),
24
+ "PythonVersion": (".type", "PythonVersion"),
25
+ "Output": (".abstractions.output", "Output"),
26
+ "QueueDepthAutoscaler": (".type", "QueueDepthAutoscaler"),
27
+ "ServingConfig": (".type", "ServingConfig"),
28
+ "experimental": (".abstractions.experimental", None),
29
+ "integrations": (".abstractions.integrations", None),
30
+ "schedule": (".abstractions.function", "Schedule"),
31
+ "TaskPolicy": (".type", "TaskPolicy"),
32
+ "Bot": (".abstractions.experimental.bot.bot", "Bot"),
33
+ "BotLocation": (".abstractions.experimental.bot.bot", "BotLocation"),
34
+ "BotEventType": (".abstractions.experimental.bot.bot", "BotEventType"),
35
+ "BotContext": (".abstractions.experimental.bot.types", "BotContext"),
36
+ "Pod": (".abstractions.pod", "Pod"),
37
+ "Service": (".abstractions.service", "Service"),
38
+ "PricingPolicy": (".type", "PricingPolicy"),
39
+ "PricingPolicyCostModel": (".type", "PricingPolicyCostModel"),
40
+ "Client": (".client.client", "Client"),
41
+ "Task": (".client.task", "Task"),
42
+ "Deployment": (".client.deployment", "Deployment"),
43
+ "schema": (".schema", None),
44
+ "Sandbox": (".abstractions.sandbox", "Sandbox"),
45
+ "SandboxInstance": (".abstractions.sandbox", "SandboxInstance"),
46
+ "SandboxProcess": (".abstractions.sandbox", "SandboxProcess"),
47
+ "SandboxProcessStream": (".abstractions.sandbox", "SandboxProcessStream"),
48
+ "SandboxProcessManager": (".abstractions.sandbox", "SandboxProcessManager"),
49
+ "SandboxProcessResponse": (".abstractions.sandbox", "SandboxProcessResponse"),
50
+ "SandboxConnectionError": (".abstractions.sandbox", "SandboxConnectionError"),
51
+ "SandboxProcessError": (".abstractions.sandbox", "SandboxProcessError"),
52
+ "SandboxFileSystemError": (".abstractions.sandbox", "SandboxFileSystemError"),
53
+ "SandboxFilePosition": (".abstractions.sandbox", "SandboxFilePosition"),
54
+ "SandboxFileSearchRange": (".abstractions.sandbox", "SandboxFileSearchRange"),
55
+ "SandboxFileSearchMatch": (".abstractions.sandbox", "SandboxFileSearchMatch"),
56
+ "SandboxFileInfo": (".abstractions.sandbox", "SandboxFileInfo"),
57
+ "SandboxFileSystem": (".abstractions.sandbox", "SandboxFileSystem"),
58
+ "SandboxFileSearchResult": (".abstractions.sandbox", "SandboxFileSearchResult"),
59
+ }
60
+
61
+ __all__ = list(_EXPORTS)
62
+
63
+
64
+ def __getattr__(name):
65
+ try:
66
+ module_name, attribute = _EXPORTS[name]
67
+ except KeyError as error:
68
+ raise AttributeError(f"module {__name__!r} has no attribute {name!r}") from error
69
+
70
+ module = import_module(module_name, __name__)
71
+ value = module if attribute is None else getattr(module, attribute)
72
+ globals()[name] = value
73
+ return value
74
+
75
+
76
+ def __dir__():
77
+ return sorted((*globals(), *__all__))
levers9/__main__.py ADDED
@@ -0,0 +1,3 @@
1
+ from .cli import main
2
+
3
+ main.start()
File without changes
@@ -0,0 +1,77 @@
1
+ import os
2
+
3
+ os.environ["GRPC_VERBOSITY"] = os.getenv("GRPC_VERBOSITY") or "NONE"
4
+
5
+ import sys
6
+ from abc import ABC
7
+ from typing import Optional
8
+
9
+ from ...channel import Channel
10
+ from ...channel import get_channel as _get_channel
11
+ from ...config import ConfigContext, get_config_context, set_settings
12
+
13
+ # Global channel
14
+ _channel: Optional[Channel] = None
15
+
16
+
17
+ def set_channel(
18
+ channel: Optional[Channel] = None,
19
+ context: Optional[ConfigContext] = None,
20
+ ) -> None:
21
+ """
22
+ Sets the channel globally for the SDK.
23
+
24
+ Use this before importing any abstraction to control which
25
+ gateway to connect to. When you provide a channel, it should already be
26
+ authenticated. When you provide a context, this will authenticate for you.
27
+ If neither is provided, this uses the default context and will create a
28
+ channel. If there is no default context (or config file), then we will
29
+ prompt the user for it.
30
+
31
+ Args:
32
+ channel: gRPC channel. Defaults to None.
33
+ context: Config context that defines the channel credentials. Defaults to None.
34
+ """
35
+ global _channel
36
+
37
+ if channel:
38
+ _channel = channel
39
+ return
40
+
41
+ if context:
42
+ _channel = _get_channel(context)
43
+ return
44
+
45
+ context = get_config_context()
46
+ _channel = _get_channel(context)
47
+
48
+
49
+ def unset_channel():
50
+ global _channel
51
+ _channel = None
52
+
53
+
54
+ def get_channel() -> Channel:
55
+ global _channel
56
+
57
+ if not _channel:
58
+ set_channel()
59
+
60
+ return _channel # type: ignore
61
+
62
+
63
+ class BaseAbstraction(ABC):
64
+ @property
65
+ def channel(self) -> Channel:
66
+ return get_channel()
67
+
68
+ def __init_subclass__(cls, /, **kwargs):
69
+ """
70
+ Dynamically load settings depending on if this library is being used
71
+ by levers9 or beam.
72
+ """
73
+ if "beam" in sys.modules:
74
+ # Settings will be configured in SDKSettings.__post_init__
75
+ set_settings()
76
+
77
+ super().__init_subclass__(**kwargs)
@@ -0,0 +1,503 @@
1
+ """
2
+ Interactive capacity flow.
3
+
4
+ When stub creation reports that no serverless pool supports the requested GPU
5
+ (a guaranteed scheduling blackhole), this module lets the user launch
6
+ on-demand hardware from the compute marketplace and routes the workload onto
7
+ it — or fails fast with an actionable error in headless environments.
8
+ """
9
+
10
+ import time
11
+ from typing import TYPE_CHECKING, List, Optional
12
+
13
+ from ... import terminal
14
+ from ...clients.gateway import (
15
+ GetOrCreateStubRequest,
16
+ GetOrCreateStubResponse,
17
+ ListPoolOffersRequest,
18
+ ListPrivatePoolsRequest,
19
+ PoolConfig,
20
+ PoolOffer,
21
+ )
22
+
23
+ if TYPE_CHECKING:
24
+ from .runner import RunnerAbstraction
25
+
26
+ CAPACITY_STATUS_AVAILABLE = "available"
27
+ CAPACITY_STATUS_LOW = "low"
28
+ CAPACITY_STATUS_NONE = "none"
29
+
30
+ DEFAULT_ONDEMAND_TTL = "1h"
31
+ DEFAULT_ONDEMAND_NODES = 1
32
+ # "Indefinite" reservations: the control plane requires a TTL and spend cap
33
+ # on every reservation, so manual-spindown mode reserves a 30-day window the
34
+ # user releases early with 'levers9 machine release' (or extends with
35
+ # 'levers9 pool extend').
36
+ INDEFINITE_TTL = "720h"
37
+ MAX_OFFER_CHOICES = 10
38
+ PROVISION_TIMEOUT_S = 15 * 60
39
+ PROVISION_POLL_INTERVAL_S = 5
40
+
41
+
42
+ def requested_gpus(stub_request: GetOrCreateStubRequest) -> List[str]:
43
+ return [g for g in (stub_request.gpu or "").split(",") if g and g != "NO_GPU"]
44
+
45
+
46
+ def cli_name() -> str:
47
+ """The installed CLI executable name ("levers9" or "beam"), for command hints."""
48
+ from ...config import get_settings
49
+
50
+ return (get_settings().name or "levers9").lower()
51
+
52
+
53
+ def no_capacity_hint(gpus: List[str]) -> str:
54
+ gpu = gpus[0] if gpus else "<gpu>"
55
+ return (
56
+ f"run '{cli_name()} machine reserve --gpu {gpu}' to get hardware, or pick a different GPU"
57
+ )
58
+
59
+
60
+ def no_offers_hint() -> str:
61
+ cli = cli_name()
62
+ return (
63
+ f"browse available GPUs with '{cli} machine list', "
64
+ f"or attach your own machine with '{cli} pool join'"
65
+ )
66
+
67
+
68
+ def credits_url() -> Optional[str]:
69
+ """
70
+ The dashboard credits page, derived from the SDK's API host so the link
71
+ follows the environment (app.beam.cloud -> platform.beam.cloud,
72
+ app.stage.beam.cloud -> platform.stage.beam.cloud). Self-hosted installs
73
+ have no credits page, so hosts without the app. prefix return None.
74
+ """
75
+ from ...config import get_settings
76
+
77
+ host = (get_settings().api_host or "").split(":")[0]
78
+ if host.startswith("app."):
79
+ return f"https://platform.{host[len('app.') :]}/settings/credits"
80
+ return None
81
+
82
+
83
+ def credit_error_hint(message: str) -> Optional[str]:
84
+ """A purchase link for credit-related failures; None for everything else."""
85
+ url = credits_url()
86
+ if url and "credit" in (message or "").lower():
87
+ return f"purchase credits at {url}"
88
+ return None
89
+
90
+
91
+ def launch_failure_hint(
92
+ err_msg: str,
93
+ error_code: str = "",
94
+ required_cents: int = 0,
95
+ available_cents: int = 0,
96
+ ) -> Optional[str]:
97
+ """Actionable hint for a failed capacity launch; currently credit-focused."""
98
+ if "credit" not in (error_code or "") and not credit_error_hint(err_msg):
99
+ return None
100
+
101
+ parts = []
102
+ if required_cents:
103
+ parts.append(
104
+ f"requires ${required_cents / 100:.2f} in credit, you have ${available_cents / 100:.2f}"
105
+ )
106
+ if url := credits_url():
107
+ parts.append(f"purchase credits at {url}")
108
+ return " — ".join(parts) or None
109
+
110
+
111
+ def offer_hardware_label(offer: PoolOffer, nodes: int = 1) -> str:
112
+ """Human label for an offer's hardware, e.g. '2 nodes of 4x A6000'."""
113
+ gpu_count = offer.gpu_count or 1
114
+ hardware = (
115
+ f"{gpu_count}x {offer.gpu}"
116
+ if offer.gpu
117
+ else (offer.display_name or offer.instance_type or "CPU node")
118
+ )
119
+ if nodes > 1:
120
+ hardware = f"{nodes} nodes of {hardware}"
121
+ return hardware
122
+
123
+
124
+ def no_offers_error(gpu_label: str) -> None:
125
+ terminal.error(
126
+ f"No on-demand offers currently available for {gpu_label}.",
127
+ exit=False,
128
+ hint=no_offers_hint(),
129
+ )
130
+
131
+
132
+ def _runner_interactive(runner: "RunnerAbstraction") -> bool:
133
+ return terminal.is_interactive() and not getattr(runner, "headless", False)
134
+
135
+
136
+ def handle_capacity_verdict(
137
+ runner: "RunnerAbstraction",
138
+ stub_request: GetOrCreateStubRequest,
139
+ stub_response: GetOrCreateStubResponse,
140
+ stub_type: str,
141
+ ) -> Optional[GetOrCreateStubResponse]:
142
+ """
143
+ Reacts to the capacity verdict on a stub creation response. Returns the
144
+ response to continue with, or None when the workload should not proceed.
145
+ """
146
+ if not stub_response.ok:
147
+ return stub_response
148
+
149
+ # An explicit pool config is the user's placement choice; don't second-guess it.
150
+ if runner.pool_config is not None:
151
+ return stub_response
152
+
153
+ if stub_response.matched_private_pool:
154
+ return _attach_to_private_pool(runner, stub_request, stub_response)
155
+
156
+ if stub_response.capacity_status != CAPACITY_STATUS_NONE:
157
+ return stub_response
158
+
159
+ gpus = requested_gpus(stub_request)
160
+ gpu_label = ", ".join(gpus) if gpus else "the requested GPU"
161
+
162
+ if stub_type.endswith("/deployment"):
163
+ return _handle_deployment_capacity(runner, stub_request, stub_response, gpus, gpu_label)
164
+
165
+ if _runner_interactive(runner):
166
+ return _run_interactive_capacity_flow(
167
+ runner,
168
+ stub_request,
169
+ gpus,
170
+ gpu_label,
171
+ wait_for_ready=stub_type != "pod/run",
172
+ )
173
+
174
+ terminal.error(
175
+ f"No compute capacity supports {gpu_label}, so this workload can never be scheduled.",
176
+ exit=False,
177
+ hint=no_capacity_hint(gpus),
178
+ )
179
+ return None
180
+
181
+
182
+ def _handle_deployment_capacity(
183
+ runner: "RunnerAbstraction",
184
+ stub_request: GetOrCreateStubRequest,
185
+ stub_response: GetOrCreateStubResponse,
186
+ gpus: List[str],
187
+ gpu_label: str,
188
+ ) -> Optional[GetOrCreateStubResponse]:
189
+ """
190
+ Deployments are long-lived and capacity is dynamic. Interactively the
191
+ best path is to reserve on-demand hardware and pin the deployment to
192
+ that pool; alternatively the user can deploy anyway (requests fail fast
193
+ with a clear reason until capacity is attached, then start succeeding
194
+ with no redeploy). Headless deploys warn and proceed — CI never blocks.
195
+ """
196
+ terminal.warn(f"No compute capacity currently supports {gpu_label}.")
197
+
198
+ if not _runner_interactive(runner):
199
+ terminal.detail(f" hint: {no_capacity_hint(gpus)}")
200
+ return stub_response
201
+
202
+ try:
203
+ choice = terminal.select(
204
+ "How do you want to proceed?",
205
+ [
206
+ terminal.SelectOption(
207
+ label="Reserve on-demand hardware",
208
+ value="reserve",
209
+ description="pins this deployment to the reserved machine",
210
+ ),
211
+ terminal.SelectOption(
212
+ label="Deploy anyway",
213
+ value="deploy",
214
+ description="requests fail until capacity is attached",
215
+ ),
216
+ terminal.SelectOption(label="Cancel", value="cancel"),
217
+ ],
218
+ )
219
+ except KeyboardInterrupt:
220
+ choice = "cancel"
221
+
222
+ if choice == "reserve":
223
+ return _run_interactive_capacity_flow(runner, stub_request, gpus, gpu_label, announce=False)
224
+ if choice == "cancel":
225
+ # Cancelling is not a failure.
226
+ terminal.detail("Deployment cancelled.")
227
+ raise SystemExit(0)
228
+ return stub_response
229
+
230
+
231
+ def _attach_to_private_pool(
232
+ runner: "RunnerAbstraction",
233
+ stub_request: GetOrCreateStubRequest,
234
+ stub_response: GetOrCreateStubResponse,
235
+ ) -> GetOrCreateStubResponse:
236
+ pool_name = stub_response.matched_private_pool
237
+ pool = PoolConfig(name=pool_name, selector=pool_name)
238
+
239
+ terminal.print(
240
+ f"[bold {terminal.BRAND_COLOR}]=>[/bold {terminal.BRAND_COLOR}] "
241
+ f"No serverless capacity for this GPU — using your pool [bold]{pool_name}[/bold]"
242
+ )
243
+
244
+ stub_request.pool = pool
245
+ response = runner.gateway_stub.get_or_create_stub(stub_request)
246
+ if response.ok:
247
+ runner.pool_config = pool
248
+ return response
249
+
250
+
251
+ def _run_interactive_capacity_flow(
252
+ runner: "RunnerAbstraction",
253
+ stub_request: GetOrCreateStubRequest,
254
+ gpus: List[str],
255
+ gpu_label: str,
256
+ announce: bool = True,
257
+ wait_for_ready: bool = True,
258
+ ) -> Optional[GetOrCreateStubResponse]:
259
+ if announce:
260
+ terminal.warn(f"No serverless capacity for {gpu_label}.")
261
+
262
+ offers = fetch_offers(runner.gateway_stub, gpus)
263
+ if offers is None:
264
+ return None
265
+ if not offers:
266
+ no_offers_error(gpu_label)
267
+ return None
268
+
269
+ try:
270
+ offer = terminal.select(
271
+ "Launch on-demand hardware to run this workload?",
272
+ offer_options(offers),
273
+ )
274
+ if offer is None:
275
+ return None
276
+
277
+ hourly = offer.hourly_cost_micros / 1_000_000
278
+ ttl = select_ttl(hourly)
279
+ if not terminal.confirm(
280
+ f"Launch {offer_hardware_label(offer)} for ~${hourly:.2f}/hr ({ttl_label(ttl)})?",
281
+ default=True,
282
+ ):
283
+ return None
284
+ except KeyboardInterrupt:
285
+ terminal.detail("Cancelled.")
286
+ return None
287
+
288
+ pool = pool_config_for_offer(offer, ttl=ttl)
289
+ stub_request.pool = pool
290
+
291
+ response = runner.gateway_stub.get_or_create_stub(stub_request)
292
+ if not response.ok:
293
+ return response
294
+ runner.pool_config = pool
295
+
296
+ if wait_for_ready and not wait_for_pool_ready(runner.gateway_stub, pool.name):
297
+ return None
298
+ return response
299
+
300
+
301
+ def fetch_offers(
302
+ gateway_stub, gpus: List[str], limit: int = MAX_OFFER_CHOICES
303
+ ) -> Optional[List[PoolOffer]]:
304
+ """Fetch on-demand offers sorted by price; None indicates a fetch error."""
305
+ with terminal.progress("Finding available hardware..."):
306
+ res = gateway_stub.list_pool_offers(ListPoolOffersRequest(pool=PoolConfig(gpu=gpus)))
307
+ if not res.ok:
308
+ terminal.error(f"Failed to list hardware offers: {res.err_msg}", exit=False)
309
+ return None
310
+
311
+ offers = list(res.offers)
312
+ offers.sort(key=lambda o: o.hourly_cost_micros)
313
+ if limit > 0:
314
+ offers = offers[:limit]
315
+ return offers
316
+
317
+
318
+ def _offer_columns(offer: PoolOffer) -> tuple:
319
+ hourly = offer.hourly_cost_micros / 1_000_000
320
+ region = offer.region_display_name or offer.region or "any region"
321
+ hardware = offer_hardware_label(offer)
322
+ details = []
323
+ if offer.cpu_millicores:
324
+ details.append(f"{offer.cpu_millicores // 1000} vCPU")
325
+ if offer.memory_mb:
326
+ details.append(f"{offer.memory_mb // 1024}GB RAM")
327
+ if offer.reliability:
328
+ details.append(f"{offer.reliability * 100:.0f}% reliability")
329
+ return hardware, region, f"${hourly:.2f}/hr", ", ".join(details)
330
+
331
+
332
+ def offer_options(offers: List[PoolOffer]) -> List[terminal.SelectOption]:
333
+ """
334
+ Build select options with the hardware/region/price columns padded to
335
+ equal widths across all offers, so the picker reads like a table.
336
+ """
337
+ rows = [_offer_columns(o) for o in offers]
338
+ widths = [max(len(row[i]) for row in rows) for i in range(3)] if rows else [0, 0, 0]
339
+ return [
340
+ terminal.SelectOption(
341
+ label=f"{hardware:<{widths[0]}} {region:<{widths[1]}} {price:>{widths[2]}}",
342
+ value=offer,
343
+ description=details,
344
+ )
345
+ for offer, (hardware, region, price, details) in zip(offers, rows)
346
+ ]
347
+
348
+
349
+ def pool_config_for_offer(
350
+ offer: PoolOffer,
351
+ name: str = "",
352
+ nodes: int = DEFAULT_ONDEMAND_NODES,
353
+ ttl: str = DEFAULT_ONDEMAND_TTL,
354
+ max_spend: float = 0.0,
355
+ ) -> PoolConfig:
356
+ hourly = offer.hourly_cost_micros / 1_000_000
357
+ if max_spend <= 0:
358
+ # Cover the TTL with headroom so billing reconciliation never kills
359
+ # the node mid-session; the TTL is what actually bounds the spend.
360
+ # Long/indefinite reservations get a slimmer buffer so the cap stays
361
+ # a sane guardrail rather than a blank check.
362
+ hours = max(1.0, ttl_hours(ttl))
363
+ buffer = 2.0 if hours <= 24 else 1.25
364
+ max_spend = max(1.0, round(hourly * nodes * hours * buffer, 2))
365
+ pool_name = _sanitize_pool_name(name or f"ondemand-{offer.gpu or 'cpu'}")
366
+ return PoolConfig(
367
+ name=pool_name,
368
+ selector=pool_name,
369
+ gpu=[offer.gpu] if offer.gpu else [],
370
+ nodes=nodes,
371
+ ttl=ttl,
372
+ max_spend=max_spend,
373
+ providers=[offer.provider] if offer.provider else [],
374
+ offer_id=offer.id,
375
+ )
376
+
377
+
378
+ def ttl_hours(ttl: str) -> float:
379
+ value = (ttl or "").strip().lower()
380
+ try:
381
+ if value.endswith("h"):
382
+ return float(value[:-1])
383
+ if value.endswith("m"):
384
+ return float(value[:-1]) / 60
385
+ if value.endswith("d"):
386
+ return float(value[:-1]) * 24
387
+ return float(value)
388
+ except ValueError:
389
+ return 1.0
390
+
391
+
392
+ def ttl_label(ttl: str) -> str:
393
+ if ttl == INDEFINITE_TTL:
394
+ return "runs until you release it (30d cap)"
395
+ return f"expires in {ttl}"
396
+
397
+
398
+ def valid_ttl(ttl: str) -> bool:
399
+ value = (ttl or "").strip().lower()
400
+ if len(value) < 2 or value[-1] not in "mhd":
401
+ return False
402
+ try:
403
+ return float(value[:-1]) > 0
404
+ except ValueError:
405
+ return False
406
+
407
+
408
+ TTL_CHOICES = ["1h", "2h", "4h", "8h", "24h"]
409
+
410
+
411
+ def select_ttl(hourly: float, nodes: int = 1, default: str = DEFAULT_ONDEMAND_TTL) -> str:
412
+ """
413
+ Interactive duration picker with the projected cost per choice. Falls
414
+ back to the default duration when the terminal isn't interactive.
415
+ """
416
+ if not terminal.is_interactive():
417
+ return default
418
+
419
+ options = [
420
+ terminal.SelectOption(
421
+ label="until I stop it",
422
+ value=INDEFINITE_TTL,
423
+ description=f"~${hourly * nodes:.2f}/hr until '{cli_name()} machine release' (30d cap)",
424
+ )
425
+ ]
426
+ for ttl in TTL_CHOICES:
427
+ cost = hourly * nodes * ttl_hours(ttl)
428
+ options.append(terminal.SelectOption(label=f"{ttl:<12} ~${cost:.2f}", value=ttl))
429
+ options.append(terminal.SelectOption(label="other", value="", description="custom duration"))
430
+
431
+ choice = terminal.select("How long do you need it?", options)
432
+ if choice:
433
+ return choice
434
+
435
+ while True:
436
+ raw = str(terminal.prompt(text="Duration (e.g. 45m, 6h, 2d)", default=default))
437
+ if valid_ttl(raw):
438
+ return raw
439
+ terminal.warn("Use a number with a unit: 45m, 6h, or 2d.")
440
+
441
+
442
+ def _sanitize_pool_name(value: str) -> str:
443
+ out = "".join(c if (c.isalnum() or c in "-._") else "-" for c in value.lower())
444
+ return out.strip("-._") or "ondemand"
445
+
446
+
447
+ def wait_for_pool_ready(
448
+ gateway_stub,
449
+ pool_name: str,
450
+ timeout_s: int = PROVISION_TIMEOUT_S,
451
+ ) -> bool:
452
+ """
453
+ Block with live progress until the pool has at least one ready machine.
454
+ Milestones (reservation → booting → ready) print with elapsed times as
455
+ they are observed.
456
+ """
457
+ start = time.monotonic()
458
+ milestones_seen = set()
459
+
460
+ def milestone(name: str, label: str):
461
+ if name in milestones_seen:
462
+ return
463
+ milestones_seen.add(name)
464
+ terminal.print(
465
+ f"[bold green]✓[/bold green] {label} [dim]({time.monotonic() - start:.0f}s)[/dim]"
466
+ )
467
+
468
+ try:
469
+ with terminal.progress("Provisioning node...") as status:
470
+ while time.monotonic() - start < timeout_s:
471
+ pool = _find_pool(gateway_stub, pool_name)
472
+ if pool is not None:
473
+ if pool.reservations or pool.reserved_nodes > 0:
474
+ milestone("reserved", "Node reserved")
475
+ status.update("Waiting for instance to boot...")
476
+ if pool.machine_count > 0:
477
+ milestone("joined", "Machine online, agent joined")
478
+ status.update("Waiting for machine to become ready...")
479
+ if pool.ready_machine_count > 0:
480
+ milestone("ready", "Machine ready")
481
+ return True
482
+ time.sleep(PROVISION_POLL_INTERVAL_S)
483
+ except KeyboardInterrupt:
484
+ terminal.warn(f"Interrupted; pool '{pool_name}' keeps provisioning in the background.")
485
+ terminal.detail(f" hint: check progress with '{cli_name()} pool machines {pool_name}'")
486
+ return False
487
+
488
+ terminal.error(
489
+ f"Timed out waiting for pool '{pool_name}' to become ready; it may still be provisioning.",
490
+ exit=False,
491
+ hint=f"check progress with '{cli_name()} pool machines {pool_name}' and re-run once a machine is ready",
492
+ )
493
+ return False
494
+
495
+
496
+ def _find_pool(gateway_stub, pool_name: str):
497
+ res = gateway_stub.list_private_pools(ListPrivatePoolsRequest())
498
+ if not res.ok:
499
+ return None
500
+ for pool in res.pools:
501
+ if pool.name == pool_name or pool.selector == pool_name:
502
+ return pool
503
+ return None