levers9 0.1.265__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- __init__.py +0 -0
- levers9/__init__.py +77 -0
- levers9/__main__.py +3 -0
- levers9/abstractions/__init__.py +0 -0
- levers9/abstractions/base/__init__.py +77 -0
- levers9/abstractions/base/capacity.py +503 -0
- levers9/abstractions/base/container.py +150 -0
- levers9/abstractions/base/runner.py +819 -0
- levers9/abstractions/base/utils.py +85 -0
- levers9/abstractions/endpoint.py +648 -0
- levers9/abstractions/experimental/__init__.py +5 -0
- levers9/abstractions/experimental/bot/__init__.py +0 -0
- levers9/abstractions/experimental/bot/bot.py +402 -0
- levers9/abstractions/experimental/bot/marker.py +50 -0
- levers9/abstractions/experimental/bot/types.py +243 -0
- levers9/abstractions/experimental/signal.py +110 -0
- levers9/abstractions/function.py +445 -0
- levers9/abstractions/image.py +912 -0
- levers9/abstractions/integrations/__init__.py +4 -0
- levers9/abstractions/integrations/fastmcp.py +217 -0
- levers9/abstractions/integrations/vllm.py +477 -0
- levers9/abstractions/map.py +135 -0
- levers9/abstractions/mixins.py +214 -0
- levers9/abstractions/output.py +350 -0
- levers9/abstractions/pod.py +509 -0
- levers9/abstractions/queue.py +118 -0
- levers9/abstractions/sandbox.py +4270 -0
- levers9/abstractions/service.py +374 -0
- levers9/abstractions/shell.py +601 -0
- levers9/abstractions/taskqueue.py +303 -0
- levers9/abstractions/volume.py +159 -0
- levers9/aio.py +10 -0
- levers9/channel.py +325 -0
- levers9/cli/__init__.py +0 -0
- levers9/cli/config.py +207 -0
- levers9/cli/container.py +205 -0
- levers9/cli/database.py +1075 -0
- levers9/cli/deployment.py +638 -0
- levers9/cli/dev.py +72 -0
- levers9/cli/disk.py +263 -0
- levers9/cli/extraclick.py +585 -0
- levers9/cli/llm.py +269 -0
- levers9/cli/machine.py +643 -0
- levers9/cli/machine_format.py +187 -0
- levers9/cli/main.py +135 -0
- levers9/cli/pool.py +863 -0
- levers9/cli/run.py +153 -0
- levers9/cli/secret.py +144 -0
- levers9/cli/serve.py +67 -0
- levers9/cli/shell.py +80 -0
- levers9/cli/task.py +151 -0
- levers9/cli/token.py +194 -0
- levers9/cli/volume.py +485 -0
- levers9/cli/worker.py +216 -0
- levers9/cli/worker_management.py +55 -0
- levers9/client/__init__.py +50 -0
- levers9/client/client.py +168 -0
- levers9/client/deployment.py +72 -0
- levers9/client/task.py +127 -0
- levers9/clients/__init__.py +0 -0
- levers9/clients/bot/__init__.py +159 -0
- levers9/clients/disk/__init__.py +139 -0
- levers9/clients/endpoint/__init__.py +46 -0
- levers9/clients/function/__init__.py +140 -0
- levers9/clients/gateway/__init__.py +2545 -0
- levers9/clients/google/__init__.py +0 -0
- levers9/clients/google/api/__init__.py +379 -0
- levers9/clients/image/__init__.py +108 -0
- levers9/clients/map/__init__.py +119 -0
- levers9/clients/output/__init__.py +112 -0
- levers9/clients/pod/__init__.py +597 -0
- levers9/clients/secret/__init__.py +140 -0
- levers9/clients/shell/__init__.py +72 -0
- levers9/clients/signal/__init__.py +84 -0
- levers9/clients/simplequeue/__init__.py +115 -0
- levers9/clients/taskqueue/__init__.py +162 -0
- levers9/clients/types/__init__.py +340 -0
- levers9/clients/volume/__init__.py +363 -0
- levers9/config.py +262 -0
- levers9/env.py +94 -0
- levers9/exceptions.py +173 -0
- levers9/integrations.py +3 -0
- levers9/logging.py +175 -0
- levers9/middleware.py +382 -0
- levers9/multipart.py +785 -0
- levers9/runner/__init__.py +0 -0
- levers9/runner/bot/__init__.py +0 -0
- levers9/runner/bot/transition.py +254 -0
- levers9/runner/common.py +709 -0
- levers9/runner/container.py +98 -0
- levers9/runner/endpoint.py +331 -0
- levers9/runner/function.py +353 -0
- levers9/runner/serve.py +129 -0
- levers9/runner/taskqueue.py +416 -0
- levers9/schema.py +566 -0
- levers9/sync.py +301 -0
- levers9/terminal.py +520 -0
- levers9/type.py +447 -0
- levers9/utils.py +131 -0
- levers9/vendor/pathspec/__init__.py +76 -0
- levers9/vendor/pathspec/_meta.py +58 -0
- levers9/vendor/pathspec/gitignore.py +157 -0
- levers9/vendor/pathspec/pathspec.py +394 -0
- levers9/vendor/pathspec/pattern.py +213 -0
- levers9/vendor/pathspec/patterns/__init__.py +11 -0
- levers9/vendor/pathspec/patterns/gitwildmatch.py +421 -0
- levers9/vendor/pathspec/py.typed +1 -0
- levers9/vendor/pathspec/util.py +792 -0
- levers9-0.1.265.dist-info/METADATA +27 -0
- levers9-0.1.265.dist-info/RECORD +113 -0
- levers9-0.1.265.dist-info/WHEEL +5 -0
- levers9-0.1.265.dist-info/entry_points.txt +2 -0
- levers9-0.1.265.dist-info/top_level.txt +2 -0
__init__.py
ADDED
|
File without changes
|
levers9/__init__.py
ADDED
|
@@ -0,0 +1,77 @@
|
|
|
1
|
+
from importlib import import_module
|
|
2
|
+
|
|
3
|
+
|
|
4
|
+
_EXPORTS = {
|
|
5
|
+
"Map": (".abstractions.map", "Map"),
|
|
6
|
+
"Image": (".abstractions.image", "Image"),
|
|
7
|
+
"Queue": (".abstractions.queue", "SimpleQueue"),
|
|
8
|
+
"Volume": (".abstractions.volume", "Volume"),
|
|
9
|
+
"CloudBucket": (".abstractions.volume", "CloudBucket"),
|
|
10
|
+
"CloudBucketConfig": (".abstractions.volume", "CloudBucketConfig"),
|
|
11
|
+
"task_queue": (".abstractions.taskqueue", "TaskQueue"),
|
|
12
|
+
"function": (".abstractions.function", "Function"),
|
|
13
|
+
"endpoint": (".abstractions.endpoint", "Endpoint"),
|
|
14
|
+
"asgi": (".abstractions.endpoint", "ASGI"),
|
|
15
|
+
"realtime": (".abstractions.endpoint", "RealtimeASGI"),
|
|
16
|
+
"Container": (".abstractions.base.container", "Container"),
|
|
17
|
+
"env": (".env", None),
|
|
18
|
+
"GpuType": (".type", "GpuType"),
|
|
19
|
+
"DatabaseServingConfig": (".type", "DatabaseServingConfig"),
|
|
20
|
+
"DurableDisk": (".type", "DurableDisk"),
|
|
21
|
+
"LLMConfig": (".type", "LLMConfig"),
|
|
22
|
+
"LLMTokenPressureAutoscaler": (".type", "LLMTokenPressureAutoscaler"),
|
|
23
|
+
"Pool": (".type", "Pool"),
|
|
24
|
+
"PythonVersion": (".type", "PythonVersion"),
|
|
25
|
+
"Output": (".abstractions.output", "Output"),
|
|
26
|
+
"QueueDepthAutoscaler": (".type", "QueueDepthAutoscaler"),
|
|
27
|
+
"ServingConfig": (".type", "ServingConfig"),
|
|
28
|
+
"experimental": (".abstractions.experimental", None),
|
|
29
|
+
"integrations": (".abstractions.integrations", None),
|
|
30
|
+
"schedule": (".abstractions.function", "Schedule"),
|
|
31
|
+
"TaskPolicy": (".type", "TaskPolicy"),
|
|
32
|
+
"Bot": (".abstractions.experimental.bot.bot", "Bot"),
|
|
33
|
+
"BotLocation": (".abstractions.experimental.bot.bot", "BotLocation"),
|
|
34
|
+
"BotEventType": (".abstractions.experimental.bot.bot", "BotEventType"),
|
|
35
|
+
"BotContext": (".abstractions.experimental.bot.types", "BotContext"),
|
|
36
|
+
"Pod": (".abstractions.pod", "Pod"),
|
|
37
|
+
"Service": (".abstractions.service", "Service"),
|
|
38
|
+
"PricingPolicy": (".type", "PricingPolicy"),
|
|
39
|
+
"PricingPolicyCostModel": (".type", "PricingPolicyCostModel"),
|
|
40
|
+
"Client": (".client.client", "Client"),
|
|
41
|
+
"Task": (".client.task", "Task"),
|
|
42
|
+
"Deployment": (".client.deployment", "Deployment"),
|
|
43
|
+
"schema": (".schema", None),
|
|
44
|
+
"Sandbox": (".abstractions.sandbox", "Sandbox"),
|
|
45
|
+
"SandboxInstance": (".abstractions.sandbox", "SandboxInstance"),
|
|
46
|
+
"SandboxProcess": (".abstractions.sandbox", "SandboxProcess"),
|
|
47
|
+
"SandboxProcessStream": (".abstractions.sandbox", "SandboxProcessStream"),
|
|
48
|
+
"SandboxProcessManager": (".abstractions.sandbox", "SandboxProcessManager"),
|
|
49
|
+
"SandboxProcessResponse": (".abstractions.sandbox", "SandboxProcessResponse"),
|
|
50
|
+
"SandboxConnectionError": (".abstractions.sandbox", "SandboxConnectionError"),
|
|
51
|
+
"SandboxProcessError": (".abstractions.sandbox", "SandboxProcessError"),
|
|
52
|
+
"SandboxFileSystemError": (".abstractions.sandbox", "SandboxFileSystemError"),
|
|
53
|
+
"SandboxFilePosition": (".abstractions.sandbox", "SandboxFilePosition"),
|
|
54
|
+
"SandboxFileSearchRange": (".abstractions.sandbox", "SandboxFileSearchRange"),
|
|
55
|
+
"SandboxFileSearchMatch": (".abstractions.sandbox", "SandboxFileSearchMatch"),
|
|
56
|
+
"SandboxFileInfo": (".abstractions.sandbox", "SandboxFileInfo"),
|
|
57
|
+
"SandboxFileSystem": (".abstractions.sandbox", "SandboxFileSystem"),
|
|
58
|
+
"SandboxFileSearchResult": (".abstractions.sandbox", "SandboxFileSearchResult"),
|
|
59
|
+
}
|
|
60
|
+
|
|
61
|
+
__all__ = list(_EXPORTS)
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def __getattr__(name):
|
|
65
|
+
try:
|
|
66
|
+
module_name, attribute = _EXPORTS[name]
|
|
67
|
+
except KeyError as error:
|
|
68
|
+
raise AttributeError(f"module {__name__!r} has no attribute {name!r}") from error
|
|
69
|
+
|
|
70
|
+
module = import_module(module_name, __name__)
|
|
71
|
+
value = module if attribute is None else getattr(module, attribute)
|
|
72
|
+
globals()[name] = value
|
|
73
|
+
return value
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
def __dir__():
|
|
77
|
+
return sorted((*globals(), *__all__))
|
levers9/__main__.py
ADDED
|
File without changes
|
|
@@ -0,0 +1,77 @@
|
|
|
1
|
+
import os
|
|
2
|
+
|
|
3
|
+
os.environ["GRPC_VERBOSITY"] = os.getenv("GRPC_VERBOSITY") or "NONE"
|
|
4
|
+
|
|
5
|
+
import sys
|
|
6
|
+
from abc import ABC
|
|
7
|
+
from typing import Optional
|
|
8
|
+
|
|
9
|
+
from ...channel import Channel
|
|
10
|
+
from ...channel import get_channel as _get_channel
|
|
11
|
+
from ...config import ConfigContext, get_config_context, set_settings
|
|
12
|
+
|
|
13
|
+
# Global channel
|
|
14
|
+
_channel: Optional[Channel] = None
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def set_channel(
|
|
18
|
+
channel: Optional[Channel] = None,
|
|
19
|
+
context: Optional[ConfigContext] = None,
|
|
20
|
+
) -> None:
|
|
21
|
+
"""
|
|
22
|
+
Sets the channel globally for the SDK.
|
|
23
|
+
|
|
24
|
+
Use this before importing any abstraction to control which
|
|
25
|
+
gateway to connect to. When you provide a channel, it should already be
|
|
26
|
+
authenticated. When you provide a context, this will authenticate for you.
|
|
27
|
+
If neither is provided, this uses the default context and will create a
|
|
28
|
+
channel. If there is no default context (or config file), then we will
|
|
29
|
+
prompt the user for it.
|
|
30
|
+
|
|
31
|
+
Args:
|
|
32
|
+
channel: gRPC channel. Defaults to None.
|
|
33
|
+
context: Config context that defines the channel credentials. Defaults to None.
|
|
34
|
+
"""
|
|
35
|
+
global _channel
|
|
36
|
+
|
|
37
|
+
if channel:
|
|
38
|
+
_channel = channel
|
|
39
|
+
return
|
|
40
|
+
|
|
41
|
+
if context:
|
|
42
|
+
_channel = _get_channel(context)
|
|
43
|
+
return
|
|
44
|
+
|
|
45
|
+
context = get_config_context()
|
|
46
|
+
_channel = _get_channel(context)
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def unset_channel():
|
|
50
|
+
global _channel
|
|
51
|
+
_channel = None
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def get_channel() -> Channel:
|
|
55
|
+
global _channel
|
|
56
|
+
|
|
57
|
+
if not _channel:
|
|
58
|
+
set_channel()
|
|
59
|
+
|
|
60
|
+
return _channel # type: ignore
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
class BaseAbstraction(ABC):
|
|
64
|
+
@property
|
|
65
|
+
def channel(self) -> Channel:
|
|
66
|
+
return get_channel()
|
|
67
|
+
|
|
68
|
+
def __init_subclass__(cls, /, **kwargs):
|
|
69
|
+
"""
|
|
70
|
+
Dynamically load settings depending on if this library is being used
|
|
71
|
+
by levers9 or beam.
|
|
72
|
+
"""
|
|
73
|
+
if "beam" in sys.modules:
|
|
74
|
+
# Settings will be configured in SDKSettings.__post_init__
|
|
75
|
+
set_settings()
|
|
76
|
+
|
|
77
|
+
super().__init_subclass__(**kwargs)
|
|
@@ -0,0 +1,503 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Interactive capacity flow.
|
|
3
|
+
|
|
4
|
+
When stub creation reports that no serverless pool supports the requested GPU
|
|
5
|
+
(a guaranteed scheduling blackhole), this module lets the user launch
|
|
6
|
+
on-demand hardware from the compute marketplace and routes the workload onto
|
|
7
|
+
it — or fails fast with an actionable error in headless environments.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
import time
|
|
11
|
+
from typing import TYPE_CHECKING, List, Optional
|
|
12
|
+
|
|
13
|
+
from ... import terminal
|
|
14
|
+
from ...clients.gateway import (
|
|
15
|
+
GetOrCreateStubRequest,
|
|
16
|
+
GetOrCreateStubResponse,
|
|
17
|
+
ListPoolOffersRequest,
|
|
18
|
+
ListPrivatePoolsRequest,
|
|
19
|
+
PoolConfig,
|
|
20
|
+
PoolOffer,
|
|
21
|
+
)
|
|
22
|
+
|
|
23
|
+
if TYPE_CHECKING:
|
|
24
|
+
from .runner import RunnerAbstraction
|
|
25
|
+
|
|
26
|
+
CAPACITY_STATUS_AVAILABLE = "available"
|
|
27
|
+
CAPACITY_STATUS_LOW = "low"
|
|
28
|
+
CAPACITY_STATUS_NONE = "none"
|
|
29
|
+
|
|
30
|
+
DEFAULT_ONDEMAND_TTL = "1h"
|
|
31
|
+
DEFAULT_ONDEMAND_NODES = 1
|
|
32
|
+
# "Indefinite" reservations: the control plane requires a TTL and spend cap
|
|
33
|
+
# on every reservation, so manual-spindown mode reserves a 30-day window the
|
|
34
|
+
# user releases early with 'levers9 machine release' (or extends with
|
|
35
|
+
# 'levers9 pool extend').
|
|
36
|
+
INDEFINITE_TTL = "720h"
|
|
37
|
+
MAX_OFFER_CHOICES = 10
|
|
38
|
+
PROVISION_TIMEOUT_S = 15 * 60
|
|
39
|
+
PROVISION_POLL_INTERVAL_S = 5
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def requested_gpus(stub_request: GetOrCreateStubRequest) -> List[str]:
|
|
43
|
+
return [g for g in (stub_request.gpu or "").split(",") if g and g != "NO_GPU"]
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def cli_name() -> str:
|
|
47
|
+
"""The installed CLI executable name ("levers9" or "beam"), for command hints."""
|
|
48
|
+
from ...config import get_settings
|
|
49
|
+
|
|
50
|
+
return (get_settings().name or "levers9").lower()
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
def no_capacity_hint(gpus: List[str]) -> str:
|
|
54
|
+
gpu = gpus[0] if gpus else "<gpu>"
|
|
55
|
+
return (
|
|
56
|
+
f"run '{cli_name()} machine reserve --gpu {gpu}' to get hardware, or pick a different GPU"
|
|
57
|
+
)
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def no_offers_hint() -> str:
|
|
61
|
+
cli = cli_name()
|
|
62
|
+
return (
|
|
63
|
+
f"browse available GPUs with '{cli} machine list', "
|
|
64
|
+
f"or attach your own machine with '{cli} pool join'"
|
|
65
|
+
)
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def credits_url() -> Optional[str]:
|
|
69
|
+
"""
|
|
70
|
+
The dashboard credits page, derived from the SDK's API host so the link
|
|
71
|
+
follows the environment (app.beam.cloud -> platform.beam.cloud,
|
|
72
|
+
app.stage.beam.cloud -> platform.stage.beam.cloud). Self-hosted installs
|
|
73
|
+
have no credits page, so hosts without the app. prefix return None.
|
|
74
|
+
"""
|
|
75
|
+
from ...config import get_settings
|
|
76
|
+
|
|
77
|
+
host = (get_settings().api_host or "").split(":")[0]
|
|
78
|
+
if host.startswith("app."):
|
|
79
|
+
return f"https://platform.{host[len('app.') :]}/settings/credits"
|
|
80
|
+
return None
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
def credit_error_hint(message: str) -> Optional[str]:
|
|
84
|
+
"""A purchase link for credit-related failures; None for everything else."""
|
|
85
|
+
url = credits_url()
|
|
86
|
+
if url and "credit" in (message or "").lower():
|
|
87
|
+
return f"purchase credits at {url}"
|
|
88
|
+
return None
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
def launch_failure_hint(
|
|
92
|
+
err_msg: str,
|
|
93
|
+
error_code: str = "",
|
|
94
|
+
required_cents: int = 0,
|
|
95
|
+
available_cents: int = 0,
|
|
96
|
+
) -> Optional[str]:
|
|
97
|
+
"""Actionable hint for a failed capacity launch; currently credit-focused."""
|
|
98
|
+
if "credit" not in (error_code or "") and not credit_error_hint(err_msg):
|
|
99
|
+
return None
|
|
100
|
+
|
|
101
|
+
parts = []
|
|
102
|
+
if required_cents:
|
|
103
|
+
parts.append(
|
|
104
|
+
f"requires ${required_cents / 100:.2f} in credit, you have ${available_cents / 100:.2f}"
|
|
105
|
+
)
|
|
106
|
+
if url := credits_url():
|
|
107
|
+
parts.append(f"purchase credits at {url}")
|
|
108
|
+
return " — ".join(parts) or None
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
def offer_hardware_label(offer: PoolOffer, nodes: int = 1) -> str:
|
|
112
|
+
"""Human label for an offer's hardware, e.g. '2 nodes of 4x A6000'."""
|
|
113
|
+
gpu_count = offer.gpu_count or 1
|
|
114
|
+
hardware = (
|
|
115
|
+
f"{gpu_count}x {offer.gpu}"
|
|
116
|
+
if offer.gpu
|
|
117
|
+
else (offer.display_name or offer.instance_type or "CPU node")
|
|
118
|
+
)
|
|
119
|
+
if nodes > 1:
|
|
120
|
+
hardware = f"{nodes} nodes of {hardware}"
|
|
121
|
+
return hardware
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
def no_offers_error(gpu_label: str) -> None:
|
|
125
|
+
terminal.error(
|
|
126
|
+
f"No on-demand offers currently available for {gpu_label}.",
|
|
127
|
+
exit=False,
|
|
128
|
+
hint=no_offers_hint(),
|
|
129
|
+
)
|
|
130
|
+
|
|
131
|
+
|
|
132
|
+
def _runner_interactive(runner: "RunnerAbstraction") -> bool:
|
|
133
|
+
return terminal.is_interactive() and not getattr(runner, "headless", False)
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
def handle_capacity_verdict(
|
|
137
|
+
runner: "RunnerAbstraction",
|
|
138
|
+
stub_request: GetOrCreateStubRequest,
|
|
139
|
+
stub_response: GetOrCreateStubResponse,
|
|
140
|
+
stub_type: str,
|
|
141
|
+
) -> Optional[GetOrCreateStubResponse]:
|
|
142
|
+
"""
|
|
143
|
+
Reacts to the capacity verdict on a stub creation response. Returns the
|
|
144
|
+
response to continue with, or None when the workload should not proceed.
|
|
145
|
+
"""
|
|
146
|
+
if not stub_response.ok:
|
|
147
|
+
return stub_response
|
|
148
|
+
|
|
149
|
+
# An explicit pool config is the user's placement choice; don't second-guess it.
|
|
150
|
+
if runner.pool_config is not None:
|
|
151
|
+
return stub_response
|
|
152
|
+
|
|
153
|
+
if stub_response.matched_private_pool:
|
|
154
|
+
return _attach_to_private_pool(runner, stub_request, stub_response)
|
|
155
|
+
|
|
156
|
+
if stub_response.capacity_status != CAPACITY_STATUS_NONE:
|
|
157
|
+
return stub_response
|
|
158
|
+
|
|
159
|
+
gpus = requested_gpus(stub_request)
|
|
160
|
+
gpu_label = ", ".join(gpus) if gpus else "the requested GPU"
|
|
161
|
+
|
|
162
|
+
if stub_type.endswith("/deployment"):
|
|
163
|
+
return _handle_deployment_capacity(runner, stub_request, stub_response, gpus, gpu_label)
|
|
164
|
+
|
|
165
|
+
if _runner_interactive(runner):
|
|
166
|
+
return _run_interactive_capacity_flow(
|
|
167
|
+
runner,
|
|
168
|
+
stub_request,
|
|
169
|
+
gpus,
|
|
170
|
+
gpu_label,
|
|
171
|
+
wait_for_ready=stub_type != "pod/run",
|
|
172
|
+
)
|
|
173
|
+
|
|
174
|
+
terminal.error(
|
|
175
|
+
f"No compute capacity supports {gpu_label}, so this workload can never be scheduled.",
|
|
176
|
+
exit=False,
|
|
177
|
+
hint=no_capacity_hint(gpus),
|
|
178
|
+
)
|
|
179
|
+
return None
|
|
180
|
+
|
|
181
|
+
|
|
182
|
+
def _handle_deployment_capacity(
|
|
183
|
+
runner: "RunnerAbstraction",
|
|
184
|
+
stub_request: GetOrCreateStubRequest,
|
|
185
|
+
stub_response: GetOrCreateStubResponse,
|
|
186
|
+
gpus: List[str],
|
|
187
|
+
gpu_label: str,
|
|
188
|
+
) -> Optional[GetOrCreateStubResponse]:
|
|
189
|
+
"""
|
|
190
|
+
Deployments are long-lived and capacity is dynamic. Interactively the
|
|
191
|
+
best path is to reserve on-demand hardware and pin the deployment to
|
|
192
|
+
that pool; alternatively the user can deploy anyway (requests fail fast
|
|
193
|
+
with a clear reason until capacity is attached, then start succeeding
|
|
194
|
+
with no redeploy). Headless deploys warn and proceed — CI never blocks.
|
|
195
|
+
"""
|
|
196
|
+
terminal.warn(f"No compute capacity currently supports {gpu_label}.")
|
|
197
|
+
|
|
198
|
+
if not _runner_interactive(runner):
|
|
199
|
+
terminal.detail(f" hint: {no_capacity_hint(gpus)}")
|
|
200
|
+
return stub_response
|
|
201
|
+
|
|
202
|
+
try:
|
|
203
|
+
choice = terminal.select(
|
|
204
|
+
"How do you want to proceed?",
|
|
205
|
+
[
|
|
206
|
+
terminal.SelectOption(
|
|
207
|
+
label="Reserve on-demand hardware",
|
|
208
|
+
value="reserve",
|
|
209
|
+
description="pins this deployment to the reserved machine",
|
|
210
|
+
),
|
|
211
|
+
terminal.SelectOption(
|
|
212
|
+
label="Deploy anyway",
|
|
213
|
+
value="deploy",
|
|
214
|
+
description="requests fail until capacity is attached",
|
|
215
|
+
),
|
|
216
|
+
terminal.SelectOption(label="Cancel", value="cancel"),
|
|
217
|
+
],
|
|
218
|
+
)
|
|
219
|
+
except KeyboardInterrupt:
|
|
220
|
+
choice = "cancel"
|
|
221
|
+
|
|
222
|
+
if choice == "reserve":
|
|
223
|
+
return _run_interactive_capacity_flow(runner, stub_request, gpus, gpu_label, announce=False)
|
|
224
|
+
if choice == "cancel":
|
|
225
|
+
# Cancelling is not a failure.
|
|
226
|
+
terminal.detail("Deployment cancelled.")
|
|
227
|
+
raise SystemExit(0)
|
|
228
|
+
return stub_response
|
|
229
|
+
|
|
230
|
+
|
|
231
|
+
def _attach_to_private_pool(
|
|
232
|
+
runner: "RunnerAbstraction",
|
|
233
|
+
stub_request: GetOrCreateStubRequest,
|
|
234
|
+
stub_response: GetOrCreateStubResponse,
|
|
235
|
+
) -> GetOrCreateStubResponse:
|
|
236
|
+
pool_name = stub_response.matched_private_pool
|
|
237
|
+
pool = PoolConfig(name=pool_name, selector=pool_name)
|
|
238
|
+
|
|
239
|
+
terminal.print(
|
|
240
|
+
f"[bold {terminal.BRAND_COLOR}]=>[/bold {terminal.BRAND_COLOR}] "
|
|
241
|
+
f"No serverless capacity for this GPU — using your pool [bold]{pool_name}[/bold]"
|
|
242
|
+
)
|
|
243
|
+
|
|
244
|
+
stub_request.pool = pool
|
|
245
|
+
response = runner.gateway_stub.get_or_create_stub(stub_request)
|
|
246
|
+
if response.ok:
|
|
247
|
+
runner.pool_config = pool
|
|
248
|
+
return response
|
|
249
|
+
|
|
250
|
+
|
|
251
|
+
def _run_interactive_capacity_flow(
|
|
252
|
+
runner: "RunnerAbstraction",
|
|
253
|
+
stub_request: GetOrCreateStubRequest,
|
|
254
|
+
gpus: List[str],
|
|
255
|
+
gpu_label: str,
|
|
256
|
+
announce: bool = True,
|
|
257
|
+
wait_for_ready: bool = True,
|
|
258
|
+
) -> Optional[GetOrCreateStubResponse]:
|
|
259
|
+
if announce:
|
|
260
|
+
terminal.warn(f"No serverless capacity for {gpu_label}.")
|
|
261
|
+
|
|
262
|
+
offers = fetch_offers(runner.gateway_stub, gpus)
|
|
263
|
+
if offers is None:
|
|
264
|
+
return None
|
|
265
|
+
if not offers:
|
|
266
|
+
no_offers_error(gpu_label)
|
|
267
|
+
return None
|
|
268
|
+
|
|
269
|
+
try:
|
|
270
|
+
offer = terminal.select(
|
|
271
|
+
"Launch on-demand hardware to run this workload?",
|
|
272
|
+
offer_options(offers),
|
|
273
|
+
)
|
|
274
|
+
if offer is None:
|
|
275
|
+
return None
|
|
276
|
+
|
|
277
|
+
hourly = offer.hourly_cost_micros / 1_000_000
|
|
278
|
+
ttl = select_ttl(hourly)
|
|
279
|
+
if not terminal.confirm(
|
|
280
|
+
f"Launch {offer_hardware_label(offer)} for ~${hourly:.2f}/hr ({ttl_label(ttl)})?",
|
|
281
|
+
default=True,
|
|
282
|
+
):
|
|
283
|
+
return None
|
|
284
|
+
except KeyboardInterrupt:
|
|
285
|
+
terminal.detail("Cancelled.")
|
|
286
|
+
return None
|
|
287
|
+
|
|
288
|
+
pool = pool_config_for_offer(offer, ttl=ttl)
|
|
289
|
+
stub_request.pool = pool
|
|
290
|
+
|
|
291
|
+
response = runner.gateway_stub.get_or_create_stub(stub_request)
|
|
292
|
+
if not response.ok:
|
|
293
|
+
return response
|
|
294
|
+
runner.pool_config = pool
|
|
295
|
+
|
|
296
|
+
if wait_for_ready and not wait_for_pool_ready(runner.gateway_stub, pool.name):
|
|
297
|
+
return None
|
|
298
|
+
return response
|
|
299
|
+
|
|
300
|
+
|
|
301
|
+
def fetch_offers(
|
|
302
|
+
gateway_stub, gpus: List[str], limit: int = MAX_OFFER_CHOICES
|
|
303
|
+
) -> Optional[List[PoolOffer]]:
|
|
304
|
+
"""Fetch on-demand offers sorted by price; None indicates a fetch error."""
|
|
305
|
+
with terminal.progress("Finding available hardware..."):
|
|
306
|
+
res = gateway_stub.list_pool_offers(ListPoolOffersRequest(pool=PoolConfig(gpu=gpus)))
|
|
307
|
+
if not res.ok:
|
|
308
|
+
terminal.error(f"Failed to list hardware offers: {res.err_msg}", exit=False)
|
|
309
|
+
return None
|
|
310
|
+
|
|
311
|
+
offers = list(res.offers)
|
|
312
|
+
offers.sort(key=lambda o: o.hourly_cost_micros)
|
|
313
|
+
if limit > 0:
|
|
314
|
+
offers = offers[:limit]
|
|
315
|
+
return offers
|
|
316
|
+
|
|
317
|
+
|
|
318
|
+
def _offer_columns(offer: PoolOffer) -> tuple:
|
|
319
|
+
hourly = offer.hourly_cost_micros / 1_000_000
|
|
320
|
+
region = offer.region_display_name or offer.region or "any region"
|
|
321
|
+
hardware = offer_hardware_label(offer)
|
|
322
|
+
details = []
|
|
323
|
+
if offer.cpu_millicores:
|
|
324
|
+
details.append(f"{offer.cpu_millicores // 1000} vCPU")
|
|
325
|
+
if offer.memory_mb:
|
|
326
|
+
details.append(f"{offer.memory_mb // 1024}GB RAM")
|
|
327
|
+
if offer.reliability:
|
|
328
|
+
details.append(f"{offer.reliability * 100:.0f}% reliability")
|
|
329
|
+
return hardware, region, f"${hourly:.2f}/hr", ", ".join(details)
|
|
330
|
+
|
|
331
|
+
|
|
332
|
+
def offer_options(offers: List[PoolOffer]) -> List[terminal.SelectOption]:
|
|
333
|
+
"""
|
|
334
|
+
Build select options with the hardware/region/price columns padded to
|
|
335
|
+
equal widths across all offers, so the picker reads like a table.
|
|
336
|
+
"""
|
|
337
|
+
rows = [_offer_columns(o) for o in offers]
|
|
338
|
+
widths = [max(len(row[i]) for row in rows) for i in range(3)] if rows else [0, 0, 0]
|
|
339
|
+
return [
|
|
340
|
+
terminal.SelectOption(
|
|
341
|
+
label=f"{hardware:<{widths[0]}} {region:<{widths[1]}} {price:>{widths[2]}}",
|
|
342
|
+
value=offer,
|
|
343
|
+
description=details,
|
|
344
|
+
)
|
|
345
|
+
for offer, (hardware, region, price, details) in zip(offers, rows)
|
|
346
|
+
]
|
|
347
|
+
|
|
348
|
+
|
|
349
|
+
def pool_config_for_offer(
|
|
350
|
+
offer: PoolOffer,
|
|
351
|
+
name: str = "",
|
|
352
|
+
nodes: int = DEFAULT_ONDEMAND_NODES,
|
|
353
|
+
ttl: str = DEFAULT_ONDEMAND_TTL,
|
|
354
|
+
max_spend: float = 0.0,
|
|
355
|
+
) -> PoolConfig:
|
|
356
|
+
hourly = offer.hourly_cost_micros / 1_000_000
|
|
357
|
+
if max_spend <= 0:
|
|
358
|
+
# Cover the TTL with headroom so billing reconciliation never kills
|
|
359
|
+
# the node mid-session; the TTL is what actually bounds the spend.
|
|
360
|
+
# Long/indefinite reservations get a slimmer buffer so the cap stays
|
|
361
|
+
# a sane guardrail rather than a blank check.
|
|
362
|
+
hours = max(1.0, ttl_hours(ttl))
|
|
363
|
+
buffer = 2.0 if hours <= 24 else 1.25
|
|
364
|
+
max_spend = max(1.0, round(hourly * nodes * hours * buffer, 2))
|
|
365
|
+
pool_name = _sanitize_pool_name(name or f"ondemand-{offer.gpu or 'cpu'}")
|
|
366
|
+
return PoolConfig(
|
|
367
|
+
name=pool_name,
|
|
368
|
+
selector=pool_name,
|
|
369
|
+
gpu=[offer.gpu] if offer.gpu else [],
|
|
370
|
+
nodes=nodes,
|
|
371
|
+
ttl=ttl,
|
|
372
|
+
max_spend=max_spend,
|
|
373
|
+
providers=[offer.provider] if offer.provider else [],
|
|
374
|
+
offer_id=offer.id,
|
|
375
|
+
)
|
|
376
|
+
|
|
377
|
+
|
|
378
|
+
def ttl_hours(ttl: str) -> float:
|
|
379
|
+
value = (ttl or "").strip().lower()
|
|
380
|
+
try:
|
|
381
|
+
if value.endswith("h"):
|
|
382
|
+
return float(value[:-1])
|
|
383
|
+
if value.endswith("m"):
|
|
384
|
+
return float(value[:-1]) / 60
|
|
385
|
+
if value.endswith("d"):
|
|
386
|
+
return float(value[:-1]) * 24
|
|
387
|
+
return float(value)
|
|
388
|
+
except ValueError:
|
|
389
|
+
return 1.0
|
|
390
|
+
|
|
391
|
+
|
|
392
|
+
def ttl_label(ttl: str) -> str:
|
|
393
|
+
if ttl == INDEFINITE_TTL:
|
|
394
|
+
return "runs until you release it (30d cap)"
|
|
395
|
+
return f"expires in {ttl}"
|
|
396
|
+
|
|
397
|
+
|
|
398
|
+
def valid_ttl(ttl: str) -> bool:
|
|
399
|
+
value = (ttl or "").strip().lower()
|
|
400
|
+
if len(value) < 2 or value[-1] not in "mhd":
|
|
401
|
+
return False
|
|
402
|
+
try:
|
|
403
|
+
return float(value[:-1]) > 0
|
|
404
|
+
except ValueError:
|
|
405
|
+
return False
|
|
406
|
+
|
|
407
|
+
|
|
408
|
+
TTL_CHOICES = ["1h", "2h", "4h", "8h", "24h"]
|
|
409
|
+
|
|
410
|
+
|
|
411
|
+
def select_ttl(hourly: float, nodes: int = 1, default: str = DEFAULT_ONDEMAND_TTL) -> str:
|
|
412
|
+
"""
|
|
413
|
+
Interactive duration picker with the projected cost per choice. Falls
|
|
414
|
+
back to the default duration when the terminal isn't interactive.
|
|
415
|
+
"""
|
|
416
|
+
if not terminal.is_interactive():
|
|
417
|
+
return default
|
|
418
|
+
|
|
419
|
+
options = [
|
|
420
|
+
terminal.SelectOption(
|
|
421
|
+
label="until I stop it",
|
|
422
|
+
value=INDEFINITE_TTL,
|
|
423
|
+
description=f"~${hourly * nodes:.2f}/hr until '{cli_name()} machine release' (30d cap)",
|
|
424
|
+
)
|
|
425
|
+
]
|
|
426
|
+
for ttl in TTL_CHOICES:
|
|
427
|
+
cost = hourly * nodes * ttl_hours(ttl)
|
|
428
|
+
options.append(terminal.SelectOption(label=f"{ttl:<12} ~${cost:.2f}", value=ttl))
|
|
429
|
+
options.append(terminal.SelectOption(label="other", value="", description="custom duration"))
|
|
430
|
+
|
|
431
|
+
choice = terminal.select("How long do you need it?", options)
|
|
432
|
+
if choice:
|
|
433
|
+
return choice
|
|
434
|
+
|
|
435
|
+
while True:
|
|
436
|
+
raw = str(terminal.prompt(text="Duration (e.g. 45m, 6h, 2d)", default=default))
|
|
437
|
+
if valid_ttl(raw):
|
|
438
|
+
return raw
|
|
439
|
+
terminal.warn("Use a number with a unit: 45m, 6h, or 2d.")
|
|
440
|
+
|
|
441
|
+
|
|
442
|
+
def _sanitize_pool_name(value: str) -> str:
|
|
443
|
+
out = "".join(c if (c.isalnum() or c in "-._") else "-" for c in value.lower())
|
|
444
|
+
return out.strip("-._") or "ondemand"
|
|
445
|
+
|
|
446
|
+
|
|
447
|
+
def wait_for_pool_ready(
|
|
448
|
+
gateway_stub,
|
|
449
|
+
pool_name: str,
|
|
450
|
+
timeout_s: int = PROVISION_TIMEOUT_S,
|
|
451
|
+
) -> bool:
|
|
452
|
+
"""
|
|
453
|
+
Block with live progress until the pool has at least one ready machine.
|
|
454
|
+
Milestones (reservation → booting → ready) print with elapsed times as
|
|
455
|
+
they are observed.
|
|
456
|
+
"""
|
|
457
|
+
start = time.monotonic()
|
|
458
|
+
milestones_seen = set()
|
|
459
|
+
|
|
460
|
+
def milestone(name: str, label: str):
|
|
461
|
+
if name in milestones_seen:
|
|
462
|
+
return
|
|
463
|
+
milestones_seen.add(name)
|
|
464
|
+
terminal.print(
|
|
465
|
+
f"[bold green]✓[/bold green] {label} [dim]({time.monotonic() - start:.0f}s)[/dim]"
|
|
466
|
+
)
|
|
467
|
+
|
|
468
|
+
try:
|
|
469
|
+
with terminal.progress("Provisioning node...") as status:
|
|
470
|
+
while time.monotonic() - start < timeout_s:
|
|
471
|
+
pool = _find_pool(gateway_stub, pool_name)
|
|
472
|
+
if pool is not None:
|
|
473
|
+
if pool.reservations or pool.reserved_nodes > 0:
|
|
474
|
+
milestone("reserved", "Node reserved")
|
|
475
|
+
status.update("Waiting for instance to boot...")
|
|
476
|
+
if pool.machine_count > 0:
|
|
477
|
+
milestone("joined", "Machine online, agent joined")
|
|
478
|
+
status.update("Waiting for machine to become ready...")
|
|
479
|
+
if pool.ready_machine_count > 0:
|
|
480
|
+
milestone("ready", "Machine ready")
|
|
481
|
+
return True
|
|
482
|
+
time.sleep(PROVISION_POLL_INTERVAL_S)
|
|
483
|
+
except KeyboardInterrupt:
|
|
484
|
+
terminal.warn(f"Interrupted; pool '{pool_name}' keeps provisioning in the background.")
|
|
485
|
+
terminal.detail(f" hint: check progress with '{cli_name()} pool machines {pool_name}'")
|
|
486
|
+
return False
|
|
487
|
+
|
|
488
|
+
terminal.error(
|
|
489
|
+
f"Timed out waiting for pool '{pool_name}' to become ready; it may still be provisioning.",
|
|
490
|
+
exit=False,
|
|
491
|
+
hint=f"check progress with '{cli_name()} pool machines {pool_name}' and re-run once a machine is ready",
|
|
492
|
+
)
|
|
493
|
+
return False
|
|
494
|
+
|
|
495
|
+
|
|
496
|
+
def _find_pool(gateway_stub, pool_name: str):
|
|
497
|
+
res = gateway_stub.list_private_pools(ListPrivatePoolsRequest())
|
|
498
|
+
if not res.ok:
|
|
499
|
+
return None
|
|
500
|
+
for pool in res.pools:
|
|
501
|
+
if pool.name == pool_name or pool.selector == pool_name:
|
|
502
|
+
return pool
|
|
503
|
+
return None
|