fcloud-sdk 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- fcloud/SKILL.md +1185 -0
- fcloud/__init__.py +85 -0
- fcloud/__main__.py +150 -0
- fcloud/_direct_bridge.py +1645 -0
- fcloud/_legacy_env.py +14 -0
- fcloud/cli/__init__.py +23 -0
- fcloud/cli/attach.py +180 -0
- fcloud/cli/common.py +382 -0
- fcloud/cli/console.py +227 -0
- fcloud/cli/context.py +161 -0
- fcloud/cli/exec_cmd.py +201 -0
- fcloud/cli/files.py +280 -0
- fcloud/cli/hardware.py +83 -0
- fcloud/cli/help.py +150 -0
- fcloud/cli/interactive.py +104 -0
- fcloud/cli/job.py +300 -0
- fcloud/cli/main.py +140 -0
- fcloud/cli/migration.py +276 -0
- fcloud/cli/mount.py +150 -0
- fcloud/cli/output.py +17 -0
- fcloud/cli/processes.py +360 -0
- fcloud/cli/registry.py +28 -0
- fcloud/cli/run.py +318 -0
- fcloud/cli/sessions.py +592 -0
- fcloud/cli/setup.py +44 -0
- fcloud/cli/ssh.py +255 -0
- fcloud/cli/sweep.py +578 -0
- fcloud/cli/sweep_harvest.py +187 -0
- fcloud/cli/sweep_watch.py +104 -0
- fcloud/cli/volume.py +317 -0
- fcloud/cli/wait.py +314 -0
- fcloud/cli_args.py +348 -0
- fcloud/client.py +632 -0
- fcloud/client_projects.py +52 -0
- fcloud/client_sessions.py +111 -0
- fcloud/client_volumes.py +80 -0
- fcloud/config.py +404 -0
- fcloud/direct.py +535 -0
- fcloud/errors.py +102 -0
- fcloud/fileset.py +155 -0
- fcloud/image.py +483 -0
- fcloud/job.py +249 -0
- fcloud/providers/__init__.py +5 -0
- fcloud/providers/requests_http.py +59 -0
- fcloud/py.typed +0 -0
- fcloud/session.py +442 -0
- fcloud/setup_cmd.py +213 -0
- fcloud/shell.py +558 -0
- fcloud/sweeps.py +557 -0
- fcloud/tunnel.py +251 -0
- fcloud/types.py +302 -0
- fcloud/v2_connect.py +928 -0
- fcloud/version.py +33 -0
- fcloud/volume_wait.py +131 -0
- fcloud/volumes.py +204 -0
- fcloud_sdk-0.1.0.dist-info/METADATA +234 -0
- fcloud_sdk-0.1.0.dist-info/RECORD +63 -0
- fcloud_sdk-0.1.0.dist-info/WHEEL +5 -0
- fcloud_sdk-0.1.0.dist-info/entry_points.txt +3 -0
- fcloud_sdk-0.1.0.dist-info/licenses/LICENSE +202 -0
- fcloud_sdk-0.1.0.dist-info/licenses/NOTICE +4 -0
- fcloud_sdk-0.1.0.dist-info/top_level.txt +2 -0
- foom/__init__.py +36 -0
fcloud/_legacy_env.py
ADDED
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
"""Map legacy ``FOOM_*`` environment variables onto their ``FCLOUD_*`` names.
|
|
2
|
+
|
|
3
|
+
Imported first by the package so module-level reads (queue timeout, quiet
|
|
4
|
+
mode, ...) already see the new names. ``FOOM_API_KEY``/``FOOM_URL`` are left
|
|
5
|
+
to :mod:`fcloud.config`, whose .env-vs-export ranking must know which
|
|
6
|
+
spelling a value arrived under.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
import os
|
|
10
|
+
|
|
11
|
+
# ponytail: drop once nobody exports FOOM_* any more.
|
|
12
|
+
for _k in [k for k in os.environ if k.startswith("FOOM_")]:
|
|
13
|
+
if _k not in ("FOOM_API_KEY", "FOOM_URL"):
|
|
14
|
+
os.environ.setdefault("FCLOUD_" + _k[len("FOOM_"):], os.environ[_k])
|
fcloud/cli/__init__.py
ADDED
|
@@ -0,0 +1,23 @@
|
|
|
1
|
+
"""fcloud CLI — GPU compute from the command line.
|
|
2
|
+
|
|
3
|
+
Importing this package registers every command (each module below calls
|
|
4
|
+
`registry.cmd` at import time); `fcloud.cli.main.main` dispatches them.
|
|
5
|
+
Command modules mirror the `fcloud help` groups.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from fcloud.cli import ( # noqa: F401 — imported for their registration side effects
|
|
9
|
+
exec_cmd,
|
|
10
|
+
files,
|
|
11
|
+
hardware,
|
|
12
|
+
interactive,
|
|
13
|
+
job,
|
|
14
|
+
mount,
|
|
15
|
+
processes,
|
|
16
|
+
run,
|
|
17
|
+
sessions,
|
|
18
|
+
setup,
|
|
19
|
+
ssh,
|
|
20
|
+
sweep,
|
|
21
|
+
volume,
|
|
22
|
+
wait,
|
|
23
|
+
)
|
fcloud/cli/attach.py
ADDED
|
@@ -0,0 +1,180 @@
|
|
|
1
|
+
"""Attaching to an existing session, honestly and with retries.
|
|
2
|
+
|
|
3
|
+
Terminal-state vocabulary, the attach retry schedule, the "session
|
|
4
|
+
unreachable" exit and the Session handle every attach-style verb uses.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
import sys
|
|
8
|
+
import time
|
|
9
|
+
|
|
10
|
+
from fcloud.cli import output
|
|
11
|
+
from fcloud.client import Client, PaymentOverdueError
|
|
12
|
+
from fcloud.errors import VolumeConflictError, VolumeNotFoundError
|
|
13
|
+
|
|
14
|
+
# Ride a restore for up to ~10 min — matches the dispatcher's
|
|
15
|
+
# restorePhaseDeadline, so the client's patience tracks the server's, instead
|
|
16
|
+
# of the old 300s reconnect budget vs 3600s run-timeout mismatch.
|
|
17
|
+
_RESUME_BUDGET_S = 600.0
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
# Authoritative session states that mean "do NOT resume": the session was
|
|
21
|
+
# deliberately ended (stop), suspended, or died — or never existed at all
|
|
22
|
+
# ("not_found", the dispatcher's 404). Reconnecting these would resurrect a
|
|
23
|
+
# host the user/admin/billing already tore down.
|
|
24
|
+
_TERMINAL_SESSION_STATES = {"closed", "closing", "dead", "failed", "not_found"}
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
# States in which a session can never run its processes again. "closing" is
|
|
28
|
+
# excluded: the host is still finalizing and its last process-index flush
|
|
29
|
+
# (which records real exits) may not have landed yet — treating it as ended
|
|
30
|
+
# would misreport a gracefully-stopped process as lost.
|
|
31
|
+
_ENDED_SESSION_STATES = _TERMINAL_SESSION_STATES - {"closing"}
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
# A fresh attach can fail transiently while the session is still binding to its
|
|
35
|
+
# just-booted host (the setup window) or during a momentary control-plane blip.
|
|
36
|
+
# Retry the attach on this short schedule before surfacing a failure — it
|
|
37
|
+
# absorbs the first-exec-after-provision race without hanging on a dead session.
|
|
38
|
+
_ATTACH_RETRY_DELAYS_S = (2.0, 5.0, 10.0)
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def _attach_with_retry(
|
|
42
|
+
client: Client, sid: str, volume_mounts, image_spec: str, sku: str,
|
|
43
|
+
):
|
|
44
|
+
"""Attach to a session, retrying a transient failure in the binding window.
|
|
45
|
+
|
|
46
|
+
Attach is idempotent (it only opens a data-plane WS), so retrying is safe —
|
|
47
|
+
unlike run(), which the caller issues exactly once. Bails out immediately if
|
|
48
|
+
the dispatcher says the session is terminal (closed/dead/failed): retrying
|
|
49
|
+
those would just wait out the schedule on a session that is never coming
|
|
50
|
+
back. The final failure re-raises so the caller can report the true cause.
|
|
51
|
+
"""
|
|
52
|
+
last_exc: Exception | None = None
|
|
53
|
+
for delay in (0.0, *_ATTACH_RETRY_DELAYS_S):
|
|
54
|
+
if delay:
|
|
55
|
+
time.sleep(delay)
|
|
56
|
+
try:
|
|
57
|
+
return client.attach_session(
|
|
58
|
+
sid, volume_mounts=volume_mounts, image_spec=image_spec, sku=sku)
|
|
59
|
+
except PaymentOverdueError:
|
|
60
|
+
raise # A billing suspension won't self-heal; surface it verbatim.
|
|
61
|
+
except (VolumeNotFoundError, VolumeConflictError):
|
|
62
|
+
# A bad --volume (typo'd name, mount conflict) is a user error;
|
|
63
|
+
# re-running the whole attach+resume schedule can't fix it.
|
|
64
|
+
raise
|
|
65
|
+
except Exception as exc:
|
|
66
|
+
last_exc = exc
|
|
67
|
+
if _authoritative_session_status(client, sid) in _TERMINAL_SESSION_STATES:
|
|
68
|
+
break
|
|
69
|
+
assert last_exc is not None
|
|
70
|
+
raise last_exc
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def _exit_session_unreachable(client: Client, sid: str, exc: Exception) -> None:
|
|
74
|
+
"""Report an unreachable session honestly, then exit 75 (EX_TEMPFAIL).
|
|
75
|
+
|
|
76
|
+
The old code blamed every attach/run failure on a host migration ("being
|
|
77
|
+
restored on a new host"), which was usually false — the common case is a
|
|
78
|
+
transient blip while the session binds to its freshly-booted host. Consult
|
|
79
|
+
the dispatcher for the real status and describe *that*: a terminal session
|
|
80
|
+
is reported as ended, everything else as a retryable "not reachable yet".
|
|
81
|
+
Exit 75 either way so a polling caller retries, preserving prior behavior.
|
|
82
|
+
"""
|
|
83
|
+
if isinstance(exc, (VolumeNotFoundError, VolumeConflictError)):
|
|
84
|
+
# The session is fine — the --volume request was wrong. Blaming
|
|
85
|
+
# session connectivity ("retry shortly") sends agents into a
|
|
86
|
+
# forever-retry loop on a permanent user error.
|
|
87
|
+
sys.stderr.write(f"Error: {exc}\n")
|
|
88
|
+
if output._JSON_OUTPUT:
|
|
89
|
+
output._out({"error": "volume", "detail": str(exc)})
|
|
90
|
+
sys.exit(2)
|
|
91
|
+
status = _authoritative_session_status(client, sid)
|
|
92
|
+
if status == "not_found":
|
|
93
|
+
# The dispatcher has no record of this session — say so instead of
|
|
94
|
+
# inventing a lifecycle ("has ended") for something that never lived.
|
|
95
|
+
msg = (f"session {sid} not found — check the ID; "
|
|
96
|
+
f"fcloud sessions --all lists your sessions")
|
|
97
|
+
sys.stderr.write(f"\033[33m⏳ {msg}\033[0m\n")
|
|
98
|
+
if output._JSON_OUTPUT:
|
|
99
|
+
output._out({"error": "not_found", "detail": str(exc)})
|
|
100
|
+
sys.exit(1) # permanent — retrying can never succeed
|
|
101
|
+
if status in _TERMINAL_SESSION_STATES:
|
|
102
|
+
# An ended EPHEMERAL job can never come back (no persisted
|
|
103
|
+
# workspace) — a retry loop must stop, and the useful pointers are
|
|
104
|
+
# its durable outputs, not the session listing.
|
|
105
|
+
ephemeral = False
|
|
106
|
+
try:
|
|
107
|
+
ephemeral = client.describe_session(sid).ephemeral
|
|
108
|
+
except Exception: # noqa: BLE001 — the hint is best-effort
|
|
109
|
+
pass
|
|
110
|
+
if ephemeral:
|
|
111
|
+
msg = (f"session {sid} was an ephemeral job and has ended "
|
|
112
|
+
f"(status: {status}); it cannot be resumed. Durable "
|
|
113
|
+
f"outputs: volumes (fcloud volume files <name>) and logs "
|
|
114
|
+
f"(fcloud job logs {sid})")
|
|
115
|
+
sys.stderr.write(f"\033[33m⏳ {msg}\033[0m\n")
|
|
116
|
+
if output._JSON_OUTPUT:
|
|
117
|
+
output._out({"error": "ended", "detail": str(exc)})
|
|
118
|
+
sys.exit(1) # permanent — retrying can never succeed
|
|
119
|
+
if status == "closed":
|
|
120
|
+
# Raw dispatcher status is "closed", but the user-facing warmth
|
|
121
|
+
# vocabulary (fcloud sessions) calls this "cold" — and unlike
|
|
122
|
+
# dead/failed it is not permanent: the session resumes on use.
|
|
123
|
+
detail, msg = "ended", (
|
|
124
|
+
f"session {sid} is cold (stopped, $0 — files kept); it comes "
|
|
125
|
+
f"back online the next time you run something on it. "
|
|
126
|
+
f"Inspect it with: fcloud sessions --all")
|
|
127
|
+
else:
|
|
128
|
+
detail, msg = "ended", (
|
|
129
|
+
f"session {sid} has ended (status: {status}); it is not "
|
|
130
|
+
f"reachable. Inspect it with: fcloud sessions --all")
|
|
131
|
+
else:
|
|
132
|
+
shown = f" (status: {status})" if status else ""
|
|
133
|
+
detail, msg = "unreachable", (
|
|
134
|
+
f"session {sid} is not reachable yet{shown} — it may be finishing "
|
|
135
|
+
"setup on its host, or its host may be migrating. Retry shortly.")
|
|
136
|
+
sys.stderr.write(f"\033[33m⏳ {msg}\033[0m\n")
|
|
137
|
+
if output._JSON_OUTPUT:
|
|
138
|
+
output._out({"error": detail, "detail": str(exc)})
|
|
139
|
+
sys.exit(75) # EX_TEMPFAIL
|
|
140
|
+
|
|
141
|
+
|
|
142
|
+
def _authoritative_session_status(client: Client, sid: str) -> str:
|
|
143
|
+
"""Dispatcher-truth session status, used to gate a migration reconnect.
|
|
144
|
+
|
|
145
|
+
Returns the status string, "unreachable" for a transient 503, or "dead"
|
|
146
|
+
for a 404/410 Gone. Never raises — a gating read must not crash the run.
|
|
147
|
+
An indeterminate read returns "" (treated as non-terminal: prefer riding a
|
|
148
|
+
likely migration over abandoning it; the dispatcher GET rarely flaps).
|
|
149
|
+
The logic lives on Client.session_status so SDK users get the same map.
|
|
150
|
+
"""
|
|
151
|
+
return client.session_status(sid)
|
|
152
|
+
|
|
153
|
+
|
|
154
|
+
def _get_session_handle(client: Client, session_id: str, volume_mounts=None):
|
|
155
|
+
"""Attach to an existing session and return a Session handle.
|
|
156
|
+
|
|
157
|
+
Returns (session, detach_fn). Call detach_fn() when done to
|
|
158
|
+
drop the WS without closing the session. ``volume_mounts`` attaches
|
|
159
|
+
volumes at the same time (spawn parity with exec's --volume); it
|
|
160
|
+
defaults to None so the other verbs attach unchanged.
|
|
161
|
+
|
|
162
|
+
Attaches with the same retry schedule as `exec`, so verbs like
|
|
163
|
+
`spawn`/`upload` ride the setup-binding blip and the seconds-scale
|
|
164
|
+
auto-close window instead of dying on a raw dispatcher error. A
|
|
165
|
+
session that still can't be reached is reported honestly (ended vs
|
|
166
|
+
retryable) and exits 75 — never a bare `HTTP 410` or
|
|
167
|
+
`session_not_requeueable`.
|
|
168
|
+
|
|
169
|
+
``volume_mounts`` are applied on attach the same way `exec` applies
|
|
170
|
+
them, so `spawn --volume` mounts a volume for the process it starts.
|
|
171
|
+
"""
|
|
172
|
+
try:
|
|
173
|
+
session = _attach_with_retry(client, session_id, volume_mounts, "", "")
|
|
174
|
+
except PaymentOverdueError:
|
|
175
|
+
# Billing suspension is not "unreachable" — let main()'s dedicated
|
|
176
|
+
# handler explain the suspended account and exit 3, not 75.
|
|
177
|
+
raise
|
|
178
|
+
except Exception as exc:
|
|
179
|
+
_exit_session_unreachable(client, session_id, exc)
|
|
180
|
+
return session, lambda: client.detach_session(session_id)
|
fcloud/cli/common.py
ADDED
|
@@ -0,0 +1,382 @@
|
|
|
1
|
+
"""Shared argv helpers: flag extraction, unit parsing, session-id rules.
|
|
2
|
+
|
|
3
|
+
Pure functions over `list[str]` argv slices; no network, no config.
|
|
4
|
+
"""
|
|
5
|
+
|
|
6
|
+
import os
|
|
7
|
+
import sys
|
|
8
|
+
|
|
9
|
+
from fcloud.config import dotenv_value
|
|
10
|
+
from fcloud.fileset import UploadOptions
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
def _extract_flag(args: list[str], flag: str) -> tuple[str, list[str]]:
|
|
14
|
+
"""Pull a --flag VALUE from args, return (value, remaining_args).
|
|
15
|
+
|
|
16
|
+
Rejects two easy mistakes with a precise error instead of letting a
|
|
17
|
+
malformed value flow downstream as a mystery `session_not_found` (or a
|
|
18
|
+
dropped command): the flag repeated, or its value being flag-shaped —
|
|
19
|
+
e.g. `--on --on s-abc` binds `--on`'s value to the literal "--on" and
|
|
20
|
+
silently drops the real id.
|
|
21
|
+
"""
|
|
22
|
+
if args.count(flag) > 1:
|
|
23
|
+
raise ValueError(f"{flag} specified more than once")
|
|
24
|
+
value = ""
|
|
25
|
+
rest = []
|
|
26
|
+
i = 0
|
|
27
|
+
while i < len(args):
|
|
28
|
+
if args[i] == flag:
|
|
29
|
+
if i + 1 >= len(args):
|
|
30
|
+
raise ValueError(f"{flag} expects a value")
|
|
31
|
+
candidate = args[i + 1]
|
|
32
|
+
if candidate.startswith("--"):
|
|
33
|
+
raise ValueError(
|
|
34
|
+
f"{flag} expects a value, but got '{candidate}' "
|
|
35
|
+
"(looks like a flag)")
|
|
36
|
+
value = candidate
|
|
37
|
+
i += 2
|
|
38
|
+
else:
|
|
39
|
+
rest.append(args[i])
|
|
40
|
+
i += 1
|
|
41
|
+
return value, rest
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def _extract_bool_flag(args: list[str], flag: str) -> tuple[bool, list[str]]:
|
|
45
|
+
"""Pull a standalone --flag out of args, return (present, remaining)."""
|
|
46
|
+
present = flag in args
|
|
47
|
+
return present, [a for a in args if a != flag]
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def _split_trailing_args(args: list[str]) -> tuple[list[str], list[str]]:
|
|
51
|
+
if "--" not in args:
|
|
52
|
+
return args, []
|
|
53
|
+
split = args.index("--")
|
|
54
|
+
return args[:split], args[split + 1:]
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def _looks_like_session_id(value: str) -> bool:
|
|
58
|
+
return value.startswith("s-")
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
def _consume_positional_session(args: list[str]) -> tuple[str, list[str]]:
|
|
62
|
+
if args and _looks_like_session_id(args[0]):
|
|
63
|
+
return args[0], args[1:]
|
|
64
|
+
return "", args
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
def _resolve_session_and_rest(args: list[str]) -> tuple[str, list[str]]:
|
|
68
|
+
"""Resolve a session id from either --on SID or a leading positional SID,
|
|
69
|
+
returning (sid, remaining_args). Mirrors how exec/run/shell accept the
|
|
70
|
+
session, so upload/download take it the same way (consistency papercut)."""
|
|
71
|
+
on_sid, rest = _extract_on(args)
|
|
72
|
+
if not on_sid:
|
|
73
|
+
on_sid, rest = _consume_positional_session(rest)
|
|
74
|
+
return on_sid, rest
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def _extract_sku(args: list[str]) -> tuple[str, list[str]]:
|
|
78
|
+
return _extract_flag(args, "--sku")
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
def _extract_on(args: list[str]) -> tuple[str, list[str]]:
|
|
82
|
+
return _extract_flag(args, "--on")
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
def _extract_wait(args: list[str]) -> tuple[int, list[str]]:
|
|
86
|
+
value, rest = _extract_flag(args, "--wait")
|
|
87
|
+
if not value:
|
|
88
|
+
value, rest = _extract_flag(rest, "--timeout")
|
|
89
|
+
return _parse_duration_seconds(value or "30s"), rest
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
def _parse_duration_seconds(value: str) -> int:
|
|
93
|
+
value = value.strip().lower()
|
|
94
|
+
if value in ("0", "none", "forever"):
|
|
95
|
+
return 0
|
|
96
|
+
multipliers = {
|
|
97
|
+
"s": 1,
|
|
98
|
+
"m": 60,
|
|
99
|
+
"h": 60 * 60,
|
|
100
|
+
"d": 24 * 60 * 60,
|
|
101
|
+
}
|
|
102
|
+
suffix = value[-1]
|
|
103
|
+
if suffix in multipliers:
|
|
104
|
+
return int(float(value[:-1]) * multipliers[suffix])
|
|
105
|
+
return int(float(value))
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
def _parse_byte_count(value: str) -> int:
|
|
109
|
+
value = value.strip().lower()
|
|
110
|
+
multipliers = {
|
|
111
|
+
"b": 1,
|
|
112
|
+
"k": 1024,
|
|
113
|
+
"kb": 1024,
|
|
114
|
+
"m": 1024 * 1024,
|
|
115
|
+
"mb": 1024 * 1024,
|
|
116
|
+
"g": 1024 * 1024 * 1024,
|
|
117
|
+
"gb": 1024 * 1024 * 1024,
|
|
118
|
+
}
|
|
119
|
+
for suffix in ("gb", "kb", "mb", "g", "k", "m", "b"):
|
|
120
|
+
if value.endswith(suffix):
|
|
121
|
+
return int(float(value[: -len(suffix)]) * multipliers[suffix])
|
|
122
|
+
return int(float(value))
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
def _parse_keep_warm(spec: str) -> int:
|
|
126
|
+
"""Parse --keep-warm SECONDS into a non-negative int."""
|
|
127
|
+
try:
|
|
128
|
+
seconds = int(spec)
|
|
129
|
+
except ValueError:
|
|
130
|
+
raise ValueError(f"--keep-warm expects a number of seconds, got {spec!r}") from None
|
|
131
|
+
if seconds < 0:
|
|
132
|
+
raise ValueError("--keep-warm seconds must be >= 0")
|
|
133
|
+
return seconds
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
# Hosts default to a 256 GiB root volume; the largest satisfiable request
|
|
137
|
+
# (a 10 TiB volume + headroom) stays well under 16 TiB. Anything above that
|
|
138
|
+
# would queue a session no provider can ever place.
|
|
139
|
+
_MIN_DISK_GB_MAX = 16384
|
|
140
|
+
|
|
141
|
+
|
|
142
|
+
def _parse_min_disk(spec: str) -> int:
|
|
143
|
+
"""Parse --min-disk-gb GiB into a bounded positive int."""
|
|
144
|
+
try:
|
|
145
|
+
gb = int(spec)
|
|
146
|
+
except ValueError:
|
|
147
|
+
raise ValueError(
|
|
148
|
+
f"--min-disk-gb expects a whole number of GiB, got {spec!r}") from None
|
|
149
|
+
if gb < 1:
|
|
150
|
+
raise ValueError(
|
|
151
|
+
"--min-disk-gb must be >= 1 (omit the flag for the 256 GiB default)")
|
|
152
|
+
if gb > _MIN_DISK_GB_MAX:
|
|
153
|
+
raise ValueError(
|
|
154
|
+
f"--min-disk-gb {gb} exceeds the {_MIN_DISK_GB_MAX} GiB (16 TiB) "
|
|
155
|
+
"ceiling — no host can be provisioned that large")
|
|
156
|
+
return gb
|
|
157
|
+
|
|
158
|
+
|
|
159
|
+
def _reject_volume_upload_collision(volume_mounts, local) -> None:
|
|
160
|
+
"""Refuse a --volume mount that collides with an upload target.
|
|
161
|
+
|
|
162
|
+
run/job uploads land at /workspace/<name>. A volume mounted at that
|
|
163
|
+
same path would be silently shadowed by the upload, and the uploaded
|
|
164
|
+
files committed into the volume on close — volume pollution with no
|
|
165
|
+
error anywhere (BUG-1, 2026-08-20 report). Same for a mount below the
|
|
166
|
+
upload dir when the local tree has content at that subpath. A mount
|
|
167
|
+
below the upload dir over an EMPTY local subpath stays allowed: it
|
|
168
|
+
composes cleanly, and writes into it are ordinary volume writes.
|
|
169
|
+
"""
|
|
170
|
+
dest = f"/workspace/{local.name}"
|
|
171
|
+
for mount in volume_mounts:
|
|
172
|
+
name = mount.get("name", "")
|
|
173
|
+
mpath = mount.get("mount_path") or f"/workspace/{name}"
|
|
174
|
+
mpath = "/" + mpath.strip("/")
|
|
175
|
+
if mpath == dest:
|
|
176
|
+
print(
|
|
177
|
+
f"Error: volume_mount_target_not_empty — volume {name!r} "
|
|
178
|
+
f"mounts at {mpath}, exactly where the upload places "
|
|
179
|
+
f"{local.name!r}. The upload would shadow the volume and "
|
|
180
|
+
"be committed into it on close. Mount the volume at a "
|
|
181
|
+
f"different path (--volume {name}:/workspace/{name}) or "
|
|
182
|
+
"rename the upload.",
|
|
183
|
+
file=sys.stderr)
|
|
184
|
+
sys.exit(2)
|
|
185
|
+
if local.is_dir() and mpath.startswith(dest + "/"):
|
|
186
|
+
sub = local.joinpath(*mpath[len(dest) + 1:].split("/"))
|
|
187
|
+
if sub.is_file() or (sub.is_dir() and any(sub.iterdir())):
|
|
188
|
+
print(
|
|
189
|
+
f"Error: volume_mount_target_not_empty — volume "
|
|
190
|
+
f"{name!r} mounts at {mpath}, but the upload already "
|
|
191
|
+
f"has content at {sub}. Those files would shadow the "
|
|
192
|
+
"volume and be committed into it on close. Move them "
|
|
193
|
+
"or mount the volume elsewhere.",
|
|
194
|
+
file=sys.stderr)
|
|
195
|
+
sys.exit(2)
|
|
196
|
+
|
|
197
|
+
|
|
198
|
+
def _warn_sku_ignored_on_attach(client, sid: str, sku: str) -> None:
|
|
199
|
+
"""`--sku` with `--on` only takes effect when a COLD session resumes
|
|
200
|
+
(the resume re-provisions on the requested SKU). Against a live session
|
|
201
|
+
it is silently ignored — say so. Best-effort: never blocks the attach."""
|
|
202
|
+
if not sku:
|
|
203
|
+
return
|
|
204
|
+
try:
|
|
205
|
+
info = client.describe_session(sid)
|
|
206
|
+
except Exception: # noqa: BLE001 — a hint must not break the attach
|
|
207
|
+
return
|
|
208
|
+
if info.sku and info.sku != sku and info.status == "active":
|
|
209
|
+
print(
|
|
210
|
+
f"Warning: --sku {sku} ignored — session {sid} is already "
|
|
211
|
+
f"running on {info.sku}. --sku applies only when a cold session "
|
|
212
|
+
"resumes; create a new session to change hardware.",
|
|
213
|
+
file=sys.stderr)
|
|
214
|
+
|
|
215
|
+
|
|
216
|
+
def _extract_min_disk(args: list[str]) -> tuple[int, list[str]]:
|
|
217
|
+
"""Pull --min-disk-gb (alias --min-disk) N out of args; 0 if absent."""
|
|
218
|
+
rest: list[str] = []
|
|
219
|
+
val = 0
|
|
220
|
+
i = 0
|
|
221
|
+
while i < len(args):
|
|
222
|
+
if args[i] in ("--min-disk-gb", "--min-disk") and i + 1 < len(args):
|
|
223
|
+
val = _parse_min_disk(args[i + 1])
|
|
224
|
+
i += 2
|
|
225
|
+
else:
|
|
226
|
+
rest.append(args[i])
|
|
227
|
+
i += 1
|
|
228
|
+
return val, rest
|
|
229
|
+
|
|
230
|
+
|
|
231
|
+
def _extract_output_policy(args: list[str]) -> tuple[str, int | None, list[str]]:
|
|
232
|
+
output_range = "tail"
|
|
233
|
+
output_limit: int | None = 64 * 1024
|
|
234
|
+
rest: list[str] = []
|
|
235
|
+
i = 0
|
|
236
|
+
while i < len(args):
|
|
237
|
+
arg = args[i]
|
|
238
|
+
if arg in ("--output", "--stdout") and i + 1 < len(args):
|
|
239
|
+
output_range = args[i + 1].lower()
|
|
240
|
+
i += 2
|
|
241
|
+
elif arg in ("--output-bytes", "--stdout-bytes") and i + 1 < len(args):
|
|
242
|
+
output_limit = _parse_byte_count(args[i + 1])
|
|
243
|
+
i += 2
|
|
244
|
+
elif arg in ("--no-output-limit", "--stdout-all"):
|
|
245
|
+
output_range = "all"
|
|
246
|
+
output_limit = None
|
|
247
|
+
i += 1
|
|
248
|
+
else:
|
|
249
|
+
rest.append(arg)
|
|
250
|
+
i += 1
|
|
251
|
+
if output_range == "all":
|
|
252
|
+
output_limit = None
|
|
253
|
+
if output_range not in ("all", "head", "middle", "tail"):
|
|
254
|
+
raise ValueError("--output must be one of: all, head, middle, tail")
|
|
255
|
+
return output_range, output_limit, rest
|
|
256
|
+
|
|
257
|
+
|
|
258
|
+
def _extract_upload_options(args: list[str]) -> tuple[UploadOptions, list[str]]:
|
|
259
|
+
"""Pull `--include-hidden` / `--follow-symlinks` out of args.
|
|
260
|
+
|
|
261
|
+
Directory uploads skip hidden entries (.env, .git, .venv …) and symlinked
|
|
262
|
+
files by default — see fcloud.fileset. These flags are the explicit opt-in.
|
|
263
|
+
"""
|
|
264
|
+
opts = UploadOptions()
|
|
265
|
+
rest: list[str] = []
|
|
266
|
+
for a in args:
|
|
267
|
+
if a == "--include-hidden":
|
|
268
|
+
opts.include_hidden = True
|
|
269
|
+
elif a == "--follow-symlinks":
|
|
270
|
+
opts.follow_symlinks = True
|
|
271
|
+
else:
|
|
272
|
+
rest.append(a)
|
|
273
|
+
return opts, rest
|
|
274
|
+
|
|
275
|
+
|
|
276
|
+
def _extract_volumes(args: list[str]) -> tuple[list[dict[str, str]], list[str]]:
|
|
277
|
+
mounts: list[dict[str, str]] = []
|
|
278
|
+
rest: list[str] = []
|
|
279
|
+
i = 0
|
|
280
|
+
while i < len(args):
|
|
281
|
+
if args[i] == "--volume" and i + 1 < len(args):
|
|
282
|
+
mounts.append(_parse_volume_mount(args[i + 1]))
|
|
283
|
+
i += 2
|
|
284
|
+
else:
|
|
285
|
+
rest.append(args[i])
|
|
286
|
+
i += 1
|
|
287
|
+
return mounts, rest
|
|
288
|
+
|
|
289
|
+
|
|
290
|
+
def _parse_volume_mount(spec: str) -> dict[str, str]:
|
|
291
|
+
name, sep, mount = spec.partition(":")
|
|
292
|
+
name = name.strip()
|
|
293
|
+
if not name:
|
|
294
|
+
raise ValueError("--volume requires NAME[:MOUNT]")
|
|
295
|
+
out = {"name": name}
|
|
296
|
+
if sep:
|
|
297
|
+
out["mount_path"] = mount.strip()
|
|
298
|
+
return out
|
|
299
|
+
|
|
300
|
+
|
|
301
|
+
def _extract_secrets(args: list[str]) -> tuple[dict[str, str], list[str]]:
|
|
302
|
+
"""Pull repeatable `--secret KEY[=VALUE]` flags out of args.
|
|
303
|
+
|
|
304
|
+
Secrets are injected into the session's *runtime* env only — they never
|
|
305
|
+
touch fcloud.json (which is committed to git), so tokens like HF_TOKEN stay
|
|
306
|
+
out of version control. `--secret KEY=VALUE` sets the value inline;
|
|
307
|
+
`--secret KEY` inherits KEY from the caller's own environment, or —
|
|
308
|
+
failing that — from the nearest .env file. Naming the key is the opt-in:
|
|
309
|
+
a .env's non-fcloud entries never enter this process's environment.
|
|
310
|
+
"""
|
|
311
|
+
secrets: dict[str, str] = {}
|
|
312
|
+
rest: list[str] = []
|
|
313
|
+
i = 0
|
|
314
|
+
while i < len(args):
|
|
315
|
+
if args[i] == "--secret" and i + 1 < len(args):
|
|
316
|
+
key, sep, value = args[i + 1].partition("=")
|
|
317
|
+
key = key.strip()
|
|
318
|
+
if not key:
|
|
319
|
+
raise ValueError("--secret requires KEY[=VALUE]")
|
|
320
|
+
if sep:
|
|
321
|
+
secrets[key] = value
|
|
322
|
+
elif key in os.environ:
|
|
323
|
+
secrets[key] = os.environ[key]
|
|
324
|
+
elif (dotenv := dotenv_value(key)) is not None:
|
|
325
|
+
secrets[key] = dotenv
|
|
326
|
+
else:
|
|
327
|
+
raise ValueError(
|
|
328
|
+
f"--secret {key}: no value given and {key} is unset in the "
|
|
329
|
+
f"environment and the nearest .env (use --secret {key}=VALUE)")
|
|
330
|
+
i += 2
|
|
331
|
+
else:
|
|
332
|
+
rest.append(args[i])
|
|
333
|
+
i += 1
|
|
334
|
+
return secrets, rest
|
|
335
|
+
|
|
336
|
+
|
|
337
|
+
def _shape_exec_command(cmd_args: list[str]) -> list[str]:
|
|
338
|
+
"""Keep argv commands structured, but treat one CLI string as shell text."""
|
|
339
|
+
if len(cmd_args) == 1:
|
|
340
|
+
return ["bash", "-lc", cmd_args[0]]
|
|
341
|
+
return cmd_args
|
|
342
|
+
|
|
343
|
+
|
|
344
|
+
def _print_upload_summary(result: dict, *, verb: str = "Uploaded") -> None:
|
|
345
|
+
"""One line on stderr: what the host wrote and what the policy skipped.
|
|
346
|
+
|
|
347
|
+
Reports the count the HOST acknowledged, not the local file count — they
|
|
348
|
+
differ exactly when the pull went wrong, and printing the local count hid
|
|
349
|
+
that. The skip note names the flag that would include each class.
|
|
350
|
+
"""
|
|
351
|
+
written = result.get("written", 0)
|
|
352
|
+
total = result.get("bytes_total", 0)
|
|
353
|
+
line = f"{verb} {written} file(s), {total} bytes"
|
|
354
|
+
note = result.get("skip_note") or ""
|
|
355
|
+
if note:
|
|
356
|
+
line += f" — {note}"
|
|
357
|
+
print(line, file=sys.stderr)
|
|
358
|
+
_print_s3_route(result)
|
|
359
|
+
|
|
360
|
+
|
|
361
|
+
def _print_s3_route(result: dict) -> None:
|
|
362
|
+
"""Name the files that took the presigned-S3 route, if any.
|
|
363
|
+
|
|
364
|
+
Files above the inline threshold leave the WS and go client→S3→host,
|
|
365
|
+
which is a different path with different failure modes. Printing it
|
|
366
|
+
always (not only on failure) is what makes "the upload broke" traceable
|
|
367
|
+
to a file — one oversized results JSON in an otherwise tiny code upload
|
|
368
|
+
silently routes the batch through S3.
|
|
369
|
+
"""
|
|
370
|
+
large = result.get("s3_files") or []
|
|
371
|
+
if not large:
|
|
372
|
+
return
|
|
373
|
+
threshold = int(result.get("s3_threshold_bytes", 0))
|
|
374
|
+
limit_mib = threshold // (1 << 20) if threshold else 0
|
|
375
|
+
print(f" {len(large)} file(s) over {limit_mib} MiB took the S3 "
|
|
376
|
+
f"upload path (client → S3 → host):", file=sys.stderr)
|
|
377
|
+
for entry in large[:5]:
|
|
378
|
+
size_mb = int(entry.get("size", 0)) / 1e6
|
|
379
|
+
print(f" {entry.get('path', '')} ({size_mb:.1f} MB)",
|
|
380
|
+
file=sys.stderr)
|
|
381
|
+
if len(large) > 5:
|
|
382
|
+
print(f" … +{len(large) - 5} more", file=sys.stderr)
|