simulo 0.26.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- simulo/__init__.py +433 -0
- simulo/_client/__init__.py +6 -0
- simulo/_client/_entrypoint.py +313 -0
- simulo/_client/_mounts.py +25 -0
- simulo/_client/_runner.py +186 -0
- simulo/_client/_secure_downloads.py +1181 -0
- simulo/_client/app.py +1308 -0
- simulo/_client/asset.py +331 -0
- simulo/_client/asset_api.py +517 -0
- simulo/_client/asset_package.py +1103 -0
- simulo/_client/asset_pins.py +187 -0
- simulo/_client/builtin_aliases.py +107 -0
- simulo/_client/bundle.py +254 -0
- simulo/_client/cancel_api.py +104 -0
- simulo/_client/cli.py +9063 -0
- simulo/_client/config.py +186 -0
- simulo/_client/credentials.py +210 -0
- simulo/_client/discovery.py +214 -0
- simulo/_client/export_api.py +212 -0
- simulo/_client/export_bundle.py +296 -0
- simulo/_client/facades.py +581 -0
- simulo/_client/http.py +414 -0
- simulo/_client/identity_api.py +117 -0
- simulo/_client/install_samples.py +267 -0
- simulo/_client/jobs_api.py +224 -0
- simulo/_client/learning.py +393 -0
- simulo/_client/login.py +319 -0
- simulo/_client/mode.py +29 -0
- simulo/_client/outputs.py +116 -0
- simulo/_client/packaging.py +445 -0
- simulo/_client/preflight_api.py +186 -0
- simulo/_client/preflight_render.py +200 -0
- simulo/_client/registry.py +98 -0
- simulo/_client/runtime.py +185 -0
- simulo/_client/runtime_display.py +90 -0
- simulo/_client/seed_ref.py +76 -0
- simulo/_client/stub.py +41 -0
- simulo/_client/submit_api.py +1057 -0
- simulo/_client/templates/__init__.py +21 -0
- simulo/_client/templates/inference/app.py.tmpl +316 -0
- simulo/_client/templates/inference/simuloignore.tmpl +30 -0
- simulo/_client/templates/scenario/app.py.tmpl +93 -0
- simulo/_client/templates/scenario/simuloignore.tmpl +27 -0
- simulo/_client/templates/training/app.py.tmpl +235 -0
- simulo/_client/templates/training/simuloignore.tmpl +29 -0
- simulo/_client/view_fragment.py +21 -0
- simulo/_client/view_session_api.py +122 -0
- simulo/_client/volume.py +71 -0
- simulo/callbacks.py +274 -0
- simulo/py.typed +0 -0
- simulo-0.26.0.dist-info/METADATA +130 -0
- simulo-0.26.0.dist-info/RECORD +55 -0
- simulo-0.26.0.dist-info/WHEEL +5 -0
- simulo-0.26.0.dist-info/entry_points.txt +2 -0
- simulo-0.26.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,1057 @@
|
|
|
1
|
+
"""HTTP client for the cloud submit + recordings + models surface (stdlib only, torch-free).
|
|
2
|
+
|
|
3
|
+
Implements the client-facing half of ``simulo.interfaces.platform.submit``:
|
|
4
|
+
register a content-addressed package, upload its tar archive, create the job
|
|
5
|
+
that executes it, and list/download the MCAP recordings and trained-model
|
|
6
|
+
checkpoints a worker produced. Shares request plumbing (structured errors,
|
|
7
|
+
the https-when-token guard, the foreign-host bearer rule) with
|
|
8
|
+
``jobs_api.py`` via ``http.py``.
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
import hashlib
|
|
14
|
+
import urllib.parse
|
|
15
|
+
from dataclasses import asdict
|
|
16
|
+
from pathlib import Path
|
|
17
|
+
from typing import Any, Optional
|
|
18
|
+
from uuid import UUID
|
|
19
|
+
|
|
20
|
+
from simulo._client import http
|
|
21
|
+
from simulo._client._secure_downloads import (
|
|
22
|
+
DestinationDirectoryError,
|
|
23
|
+
DestinationObservation,
|
|
24
|
+
observe_destination,
|
|
25
|
+
open_destination_directory,
|
|
26
|
+
)
|
|
27
|
+
from simulo._client.preflight_render import sanitize_server_text
|
|
28
|
+
from simulo.interfaces.ids import parse_job_public_id
|
|
29
|
+
from simulo.interfaces.platform.asset_catalog import (
|
|
30
|
+
ASSETS_FIELD,
|
|
31
|
+
ASSETS_RESOLVE_ROUTE,
|
|
32
|
+
MAX_ASSETS_RESOLVE_BATCH,
|
|
33
|
+
AssetsResolveResponse,
|
|
34
|
+
ResolvedAssetPin,
|
|
35
|
+
ResolveMiss,
|
|
36
|
+
)
|
|
37
|
+
from simulo.interfaces.platform.outputs import (
|
|
38
|
+
JOB_OUTPUT_DOWNLOAD_ROUTE_TEMPLATE,
|
|
39
|
+
JOB_OUTPUTS_ROUTE_TEMPLATE,
|
|
40
|
+
OutputRecord,
|
|
41
|
+
)
|
|
42
|
+
from simulo.interfaces.platform.runs import JOB_SCOPE_MINE, JOBS_ROUTE
|
|
43
|
+
from simulo.interfaces.platform.submit import (
|
|
44
|
+
JOB_MODEL_DOWNLOAD_ROUTE_TEMPLATE,
|
|
45
|
+
JOB_MODEL_ROUTE_TEMPLATE,
|
|
46
|
+
JOB_MODELS_ROUTE_TEMPLATE,
|
|
47
|
+
JOB_RECORDING_DOWNLOAD_ROUTE_TEMPLATE,
|
|
48
|
+
JOB_RECORDINGS_ROUTE_TEMPLATE,
|
|
49
|
+
PACKAGE_ARCHIVE_ROUTE_TEMPLATE,
|
|
50
|
+
PACKAGES_ROUTE,
|
|
51
|
+
SEED_SOURCE_CONFLICT_CODE,
|
|
52
|
+
SEED_SOURCE_FIELDS,
|
|
53
|
+
)
|
|
54
|
+
|
|
55
|
+
_REQUEST_TIMEOUT_S = 10.0 # every outbound call has an explicit timeout (NFR)
|
|
56
|
+
_UPLOAD_TIMEOUT_S = 60.0 # package archives can be up to 64 MiB (MAX_PACKAGE_BYTES)
|
|
57
|
+
_DOWNLOAD_TIMEOUT_S = 60.0
|
|
58
|
+
|
|
59
|
+
_LIST_PAGE_LIMIT = 100
|
|
60
|
+
_LIST_MAX_PAGES = 1000
|
|
61
|
+
|
|
62
|
+
_UNAVAILABLE_HINT = "Check SIMULO_API_URL / SIMULO_ENV, and that you are logged in (`simulo login`)."
|
|
63
|
+
|
|
64
|
+
#: Width cap on any server-derived value echoed back in an error message here
|
|
65
|
+
#: (see :func:`_safe_server_text`). A real sha256
|
|
66
|
+
#: hex digest is 64 chars and a real download destination is a filename under
|
|
67
|
+
#: the user's own cwd, so this never truncates a legitimate value. Mirrors the
|
|
68
|
+
#: CLI's ``_MAX_OUTPUTS_LINE_CHARS`` posture (bound server text at the render
|
|
69
|
+
#: boundary) rather than importing it — ``cli`` imports THIS module, so the
|
|
70
|
+
#: dependency cannot run the other way.
|
|
71
|
+
_MAX_SERVER_TEXT_CHARS = 200
|
|
72
|
+
# Mirrors the control plane's bounded legacy-prefix query. A larger payload
|
|
73
|
+
# cannot be a response from the supported resolver contract.
|
|
74
|
+
_MAX_AMBIGUITY_CANDIDATES = 6
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def _output_record_from_payload(payload: dict[str, Any]) -> dict[str, Any]:
|
|
78
|
+
"""Validate and normalize one server output row through its contract.
|
|
79
|
+
|
|
80
|
+
``OutputRecord.from_payload`` is deliberately used only at this response
|
|
81
|
+
deserialization boundary: it tolerates fields added by a newer server,
|
|
82
|
+
while direct ``OutputRecord(...)`` construction remains strict. The
|
|
83
|
+
thin client's pre-existing mapping return shape is preserved for callers.
|
|
84
|
+
"""
|
|
85
|
+
record = OutputRecord.from_payload(payload)
|
|
86
|
+
normalized = asdict(record)
|
|
87
|
+
if normalized.get("public_id") is None:
|
|
88
|
+
normalized.pop("public_id", None)
|
|
89
|
+
return normalized
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
def _safe_server_text(value: Any) -> str:
|
|
93
|
+
"""Make one server-derived value safe to interpolate into an error message.
|
|
94
|
+
|
|
95
|
+
**The choke point EVERY server-derived component of the digest-mismatch
|
|
96
|
+
raises below passes through — the digest AND the destination path.** That
|
|
97
|
+
"every component" framing is the whole point and was learned the hard way
|
|
98
|
+
twice on this PR: sanitizing one field while a sibling on the SAME LINE
|
|
99
|
+
stays raw neutralizes nothing, because the attacker simply moves the
|
|
100
|
+
payload one field over. Here that sibling was ``dest_path``, which is
|
|
101
|
+
built from a portable canonical server name for directory-form downloads
|
|
102
|
+
— and that helper rejects raw or normalization-created path separators,
|
|
103
|
+
vets filesystem shape, and removes terminal/Unicode-format controls from
|
|
104
|
+
the basename. This error boundary
|
|
105
|
+
still sanitizes every component independently as defense in depth and for
|
|
106
|
+
user-supplied explicit destination paths. Measured on the PR-B11 final round through
|
|
107
|
+
``cli.main(["outputs", JOB, AID])``, with a wrong digest (which any
|
|
108
|
+
hostile server can force at will, needing no control of the digest
|
|
109
|
+
itself) and a name of ``report\\x1b[31mHACKED\\x1b[0m.bin``, the raw
|
|
110
|
+
interpolation emitted a live ESC to stderr inside the very message whose
|
|
111
|
+
purpose is to report a verification FAILURE.
|
|
112
|
+
|
|
113
|
+
Two hazards, handled in one place:
|
|
114
|
+
|
|
115
|
+
* Terminal control characters — the thin client's shared
|
|
116
|
+
``sanitize_server_text`` helper, the
|
|
117
|
+
package's single choke point for server text bound for a terminal (the
|
|
118
|
+
same one the CLI's ``_safe_output_detail_text`` builds on), plus the
|
|
119
|
+
lone-surrogate round-trip that stripping does not cover (a ``\\udcff``
|
|
120
|
+
in a name reaches ``print()`` as ``UnicodeEncodeError: surrogates not
|
|
121
|
+
allowed`` — a stack trace where an error message belongs).
|
|
122
|
+
``.lower().removeprefix("sha256:")`` on the digest is NOT a sanitizer;
|
|
123
|
+
measured, all of these survive it intact: ``\\x1b[31mHACKED\\x1b[0m``
|
|
124
|
+
(SGR — its final byte ``m`` is already lowercase),
|
|
125
|
+
``\\x1b]8;;https://evil.example\\x07trusted.simulo.ai\\x1b]8;;\\x07``
|
|
126
|
+
(an OSC 8 hyperlink rendering as a trusted domain pointing elsewhere),
|
|
127
|
+
``deadbeef\\rok — sha256 verified`` (a CR overwrite), and NUL.
|
|
128
|
+
Lowercasing happens to neuter some CSI sequences (``\\x1b[2K`` ->
|
|
129
|
+
``\\x1b[2k``, not a valid final byte) but nothing else.
|
|
130
|
+
* UNBOUNDED LENGTH — neither field is length-checked anywhere on the way
|
|
131
|
+
here. Measured: a 50,000-char server name yielded a 50,351-char error
|
|
132
|
+
message. Bounded at :data:`_MAX_SERVER_TEXT_CHARS`.
|
|
133
|
+
"""
|
|
134
|
+
text = sanitize_server_text(str(value))
|
|
135
|
+
text = text.encode("utf-8", "replace").decode("utf-8")
|
|
136
|
+
if len(text) > _MAX_SERVER_TEXT_CHARS:
|
|
137
|
+
text = text[: _MAX_SERVER_TEXT_CHARS - 1] + "…"
|
|
138
|
+
return text
|
|
139
|
+
|
|
140
|
+
|
|
141
|
+
def _safe_server_repr(value: Any) -> str:
|
|
142
|
+
"""``repr(value)`` for a server-derived value, bounded in width.
|
|
143
|
+
|
|
144
|
+
``repr`` escapes terminal control bytes and lone surrogates, leaving width
|
|
145
|
+
as the remaining hazard. The helper remains part of the module's pinned
|
|
146
|
+
defensive contract even though ordinary identifier errors no longer echo
|
|
147
|
+
machine identifiers.
|
|
148
|
+
"""
|
|
149
|
+
text = repr(value)
|
|
150
|
+
if len(text) > _MAX_SERVER_TEXT_CHARS:
|
|
151
|
+
text = text[: _MAX_SERVER_TEXT_CHARS - 1] + "…"
|
|
152
|
+
return text
|
|
153
|
+
|
|
154
|
+
|
|
155
|
+
#: The wire name the `--from <job-ref>` seed rides under (contract-pinned pair;
|
|
156
|
+
#: the CLI never sends the model-id sibling — the operator dropped
|
|
157
|
+
#: `--from-model` from the human surface, so a user never types a raw UUID).
|
|
158
|
+
_SEED_FROM_JOB_ID_FIELD, _SEED_FROM_MODEL_ID_FIELD = SEED_SOURCE_FIELDS
|
|
159
|
+
|
|
160
|
+
|
|
161
|
+
class SubmitApiError(http.HttpError):
|
|
162
|
+
"""Base error for the submit/recordings API."""
|
|
163
|
+
|
|
164
|
+
|
|
165
|
+
class _AmbiguousJobRefError(SubmitApiError):
|
|
166
|
+
"""Client-authored copy for the named legacy UUID-prefix exception."""
|
|
167
|
+
|
|
168
|
+
code = "job_ref_ambiguous"
|
|
169
|
+
|
|
170
|
+
def __init__(self, message: str, *, candidates: tuple[str, ...]) -> None:
|
|
171
|
+
super().__init__(message)
|
|
172
|
+
self.candidates = candidates
|
|
173
|
+
|
|
174
|
+
|
|
175
|
+
def _validated_job_ref_ambiguity_candidates(exc: http.HttpHTTPError) -> Optional[tuple[str, ...]]:
|
|
176
|
+
"""Canonical UUID candidates for the one named machine-ID exception.
|
|
177
|
+
|
|
178
|
+
The server owns the raw error envelope, so its ``code`` alone is not a
|
|
179
|
+
provenance boundary. Only a 409 with a non-empty, bounded list containing
|
|
180
|
+
2–6 distinct exact UUID strings is eligible for human display; malformed
|
|
181
|
+
or duplicate-bearing lists fail closed and ordinary UUID redaction remains
|
|
182
|
+
in force.
|
|
183
|
+
"""
|
|
184
|
+
if exc.status != 409 or exc.code != "job_ref_ambiguous":
|
|
185
|
+
return None
|
|
186
|
+
raw_candidates = exc.candidates
|
|
187
|
+
if not isinstance(raw_candidates, list) or not raw_candidates or len(raw_candidates) > _MAX_AMBIGUITY_CANDIDATES:
|
|
188
|
+
return None
|
|
189
|
+
candidates: list[str] = []
|
|
190
|
+
for raw_candidate in raw_candidates:
|
|
191
|
+
if not isinstance(raw_candidate, str):
|
|
192
|
+
return None
|
|
193
|
+
try:
|
|
194
|
+
canonical = str(UUID(raw_candidate))
|
|
195
|
+
except (ValueError, AttributeError):
|
|
196
|
+
return None
|
|
197
|
+
if raw_candidate != canonical:
|
|
198
|
+
return None
|
|
199
|
+
if canonical in candidates:
|
|
200
|
+
return None
|
|
201
|
+
candidates.append(canonical)
|
|
202
|
+
return tuple(candidates) if len(candidates) >= 2 else None
|
|
203
|
+
|
|
204
|
+
|
|
205
|
+
class AssetPinError(SubmitApiError):
|
|
206
|
+
"""A submit-time asset failure — an unpinned ref under ``--frozen``, a ref that
|
|
207
|
+
could not be resolved, or a warning promoted to an error under
|
|
208
|
+
``--strict-assets`` (see ``asset_pins.resolve_and_pin_assets``).
|
|
209
|
+
|
|
210
|
+
A ``SubmitApiError`` subclass so ``simulo run``'s existing handler renders it
|
|
211
|
+
as a clean one-liner (never a traceback) alongside every other submit error —
|
|
212
|
+
the failure happens in seconds, before the package is uploaded.
|
|
213
|
+
"""
|
|
214
|
+
|
|
215
|
+
|
|
216
|
+
class SeedNotHonoredError(SubmitApiError):
|
|
217
|
+
"""A seed was requested but the created job record carries no seed provenance.
|
|
218
|
+
|
|
219
|
+
The loud-degradation guard for ``--from`` (explicit-run-intent plan): an
|
|
220
|
+
older control plane that predates the seed fields ignores them as unknown
|
|
221
|
+
keys and creates a FRESH job — which would silently train the wrong thing
|
|
222
|
+
and report success, a wrong-result hazard, not a cosmetic one. Unlike
|
|
223
|
+
``simulo view``'s missing-route degradation (a warning-shaped fallback),
|
|
224
|
+
this is an ERROR: the job was already created without the seed by the
|
|
225
|
+
time the response reveals the problem, so the message names the created
|
|
226
|
+
Job ID when available (or gives a neutral lookup path) and explains how to
|
|
227
|
+
stop it. The client deliberately does NOT auto-cancel:
|
|
228
|
+
a 200 idempotent-replay response can attach to a pre-existing fresh job
|
|
229
|
+
the user started on purpose, and the wire response shape doesn't
|
|
230
|
+
distinguish 201-created from 200-replayed — cancelling would risk killing
|
|
231
|
+
a deliberate run, so the human makes that call.
|
|
232
|
+
"""
|
|
233
|
+
|
|
234
|
+
|
|
235
|
+
def _seed_error_from_http(exc: http.HttpHTTPError) -> Optional[SubmitApiError]:
|
|
236
|
+
"""Map a seed-validation error from ``POST /v1/jobs`` to actionable copy.
|
|
237
|
+
|
|
238
|
+
Returns ``None`` for codes this function doesn't recognize (the caller
|
|
239
|
+
re-raises the original error unchanged — "surface the server's error"
|
|
240
|
+
rather than inventing text for unknown failures). The recognized codes
|
|
241
|
+
are the explicit-run-intent contract set (control plane
|
|
242
|
+
``jobs/service.py::resolve_seed_source`` + ``jobs/job_ref.py``):
|
|
243
|
+
|
|
244
|
+
* ``job_ref_ambiguous`` (409) — the prefix matched several jobs; the
|
|
245
|
+
candidates are RENDERED (one per line), never silently picked from.
|
|
246
|
+
* ``job_not_found`` (404) — unknown/cross-org seed job: the same
|
|
247
|
+
friendly not-found copy ``simulo cancel``/``simulo view`` use, aimed
|
|
248
|
+
at the SEED reference rather than the submitted job.
|
|
249
|
+
* ``seed_artifact_invalid`` (422), ``seed_job_not_terminal`` /
|
|
250
|
+
``seed_model_unavailable`` / ``seed_model_not_checkpoint`` /
|
|
251
|
+
``seed_replay_conflict`` (409), ``seed_source_conflict`` (422) — the
|
|
252
|
+
server's message already says what is wrong; it is framed as CLI copy,
|
|
253
|
+
without echoing a caller-supplied legacy UUID.
|
|
254
|
+
"""
|
|
255
|
+
if exc.code == "job_ref_ambiguous":
|
|
256
|
+
candidates = _validated_job_ref_ambiguity_candidates(exc)
|
|
257
|
+
if candidates is None:
|
|
258
|
+
return SubmitApiError(
|
|
259
|
+
"The --from selection is ambiguous, but the server did not provide valid UUID candidates. "
|
|
260
|
+
"List jobs with `simulo jobs` and use a complete Job ID."
|
|
261
|
+
)
|
|
262
|
+
listing = "".join(f"\n {candidate}" for candidate in candidates)
|
|
263
|
+
return _AmbiguousJobRefError(
|
|
264
|
+
f"The --from selection is ambiguous — it matches more than one job:{listing}\n"
|
|
265
|
+
"Use a complete Job ID (see `simulo jobs`).",
|
|
266
|
+
candidates=candidates,
|
|
267
|
+
)
|
|
268
|
+
if exc.code == "job_not_found":
|
|
269
|
+
return SubmitApiError(
|
|
270
|
+
"The --from job doesn't exist, or you don't have access to it. List runs with `simulo jobs`."
|
|
271
|
+
)
|
|
272
|
+
if exc.code in (
|
|
273
|
+
"seed_artifact_invalid",
|
|
274
|
+
"seed_job_not_terminal",
|
|
275
|
+
"seed_model_unavailable",
|
|
276
|
+
"seed_model_not_checkpoint",
|
|
277
|
+
"seed_replay_conflict",
|
|
278
|
+
SEED_SOURCE_CONFLICT_CODE,
|
|
279
|
+
):
|
|
280
|
+
return SubmitApiError(f"The --from selection could not be used: {exc.message}")
|
|
281
|
+
return None
|
|
282
|
+
|
|
283
|
+
|
|
284
|
+
class SubmitApiClient:
|
|
285
|
+
"""Client for package submit, job create, and recordings — the cloud path."""
|
|
286
|
+
|
|
287
|
+
def __init__(self, base_url: str, *, token: Optional[str] = None) -> None:
|
|
288
|
+
if not base_url.startswith(("http://", "https://")):
|
|
289
|
+
raise SubmitApiError(f"API base URL must be an http(s) URL, got {base_url!r}.")
|
|
290
|
+
self._base_url = base_url.rstrip("/")
|
|
291
|
+
self._token = token
|
|
292
|
+
|
|
293
|
+
@property
|
|
294
|
+
def base_url(self) -> str:
|
|
295
|
+
return self._base_url
|
|
296
|
+
|
|
297
|
+
# ------------------------------------------------------------------
|
|
298
|
+
# Package submit + job create
|
|
299
|
+
# ------------------------------------------------------------------
|
|
300
|
+
|
|
301
|
+
def create_package(
|
|
302
|
+
self,
|
|
303
|
+
*,
|
|
304
|
+
package_id: str,
|
|
305
|
+
source_digest: str,
|
|
306
|
+
job_name: str,
|
|
307
|
+
archive_sha256: str,
|
|
308
|
+
archive_size: int,
|
|
309
|
+
) -> dict[str, Any]:
|
|
310
|
+
"""``POST /v1/packages`` — register a content-addressed package.
|
|
311
|
+
|
|
312
|
+
A 409 ``package_conflict`` is re-raised as actionable CLI copy:
|
|
313
|
+
the server already holds a package under this exact
|
|
314
|
+
``package_id`` whose archive bytes differ. ``package_id`` is a pure
|
|
315
|
+
function of (source, job, args), so for a user who changed nothing
|
|
316
|
+
this almost always means a DIFFERENT ``simulo`` client version
|
|
317
|
+
packaged the same app differently (the packaging format moved) —
|
|
318
|
+
not a mistake in their code, and the raw wire code gives them
|
|
319
|
+
nothing to act on. The message states the one real remedy available
|
|
320
|
+
today (any source or args change mints a fresh submission identity)
|
|
321
|
+
and what the platform will do about the underlying limitation. The
|
|
322
|
+
server-side conflict check itself is correct and stays — one
|
|
323
|
+
package id must never silently refer to two different payloads.
|
|
324
|
+
"""
|
|
325
|
+
try:
|
|
326
|
+
payload = self._post_json(
|
|
327
|
+
PACKAGES_ROUTE,
|
|
328
|
+
{
|
|
329
|
+
"package_id": package_id,
|
|
330
|
+
"source_digest": source_digest,
|
|
331
|
+
"job_name": job_name,
|
|
332
|
+
"archive_sha256": archive_sha256,
|
|
333
|
+
"archive_size": archive_size,
|
|
334
|
+
},
|
|
335
|
+
)
|
|
336
|
+
except http.HttpHTTPError as exc:
|
|
337
|
+
if exc.status == 409 and exc.code == "package_conflict":
|
|
338
|
+
raise SubmitApiError(
|
|
339
|
+
"This exact submission (same source, job, and args) is already registered "
|
|
340
|
+
"on the platform with different package bytes — usually because a different "
|
|
341
|
+
"simulo client version packaged it, not because of anything in your app. "
|
|
342
|
+
"Nothing was uploaded and no job was created. To resubmit now, change "
|
|
343
|
+
"anything about the app's source (even a comment) or its args so the "
|
|
344
|
+
"submission gets a fresh identity; a platform update that accepts unchanged "
|
|
345
|
+
"resubmissions automatically is planned."
|
|
346
|
+
) from exc
|
|
347
|
+
raise
|
|
348
|
+
if not isinstance(payload, dict):
|
|
349
|
+
raise SubmitApiError(f"Malformed create_package response: {payload!r}")
|
|
350
|
+
return payload
|
|
351
|
+
|
|
352
|
+
def upload_archive(self, package_id: str, archive_bytes: bytes) -> None:
|
|
353
|
+
"""``PUT /v1/packages/{id}/archive`` — upload the raw tar bytes.
|
|
354
|
+
|
|
355
|
+
Idempotent per the contract: re-uploading the same content-addressed
|
|
356
|
+
package id is a no-op.
|
|
357
|
+
"""
|
|
358
|
+
path = PACKAGE_ARCHIVE_ROUTE_TEMPLATE.format(package_id=http.quote_path_segment(package_id))
|
|
359
|
+
http.request_bytes(
|
|
360
|
+
"PUT",
|
|
361
|
+
self._base_url + path,
|
|
362
|
+
token=self._token,
|
|
363
|
+
api_base_url=self._base_url,
|
|
364
|
+
body=archive_bytes,
|
|
365
|
+
content_type="application/octet-stream",
|
|
366
|
+
timeout=_UPLOAD_TIMEOUT_S,
|
|
367
|
+
unavailable_hint=_UNAVAILABLE_HINT,
|
|
368
|
+
)
|
|
369
|
+
|
|
370
|
+
def resolve_assets(self, refs: list[str], runtime: str) -> AssetsResolveResponse:
|
|
371
|
+
"""``POST /v1/assets/resolve`` — resolve a batch of registry refs to exact
|
|
372
|
+
pins.
|
|
373
|
+
|
|
374
|
+
``refs`` are the refs AS AUTHORED (short, possibly unpinned); the response
|
|
375
|
+
is keyed by those same strings so the caller correlates every result back
|
|
376
|
+
to its own ref. ``runtime`` is REQUIRED — the control plane computes
|
|
377
|
+
``validated_on_runtime`` against it. Refuses a batch larger than
|
|
378
|
+
:data:`MAX_ASSETS_RESOLVE_BATCH` locally (a clear error rather than a
|
|
379
|
+
wire 422), and parses the two maps into the frozen wire dataclasses.
|
|
380
|
+
"""
|
|
381
|
+
if len(refs) > MAX_ASSETS_RESOLVE_BATCH:
|
|
382
|
+
raise AssetPinError(
|
|
383
|
+
f"too many assets to resolve in one submit ({len(refs)}; the limit is "
|
|
384
|
+
f"{MAX_ASSETS_RESOLVE_BATCH}). Reduce the number of mounted assets."
|
|
385
|
+
)
|
|
386
|
+
payload = self._post_json(ASSETS_RESOLVE_ROUTE, {"refs": refs, "runtime": runtime})
|
|
387
|
+
if (
|
|
388
|
+
not isinstance(payload, dict)
|
|
389
|
+
or not isinstance(payload.get("resolved"), dict)
|
|
390
|
+
or not isinstance(payload.get("missing"), dict)
|
|
391
|
+
):
|
|
392
|
+
raise SubmitApiError(f"Malformed assets/resolve response: {payload!r}")
|
|
393
|
+
try:
|
|
394
|
+
resolved = {
|
|
395
|
+
str(ref): ResolvedAssetPin(
|
|
396
|
+
canonical_ref=entry["canonical_ref"],
|
|
397
|
+
version=int(entry["version"]),
|
|
398
|
+
digest=entry["digest"],
|
|
399
|
+
kind=entry["kind"],
|
|
400
|
+
deprecated=bool(entry["deprecated"]),
|
|
401
|
+
validated_on_runtime=bool(entry["validated_on_runtime"]),
|
|
402
|
+
)
|
|
403
|
+
for ref, entry in payload["resolved"].items()
|
|
404
|
+
}
|
|
405
|
+
missing = {
|
|
406
|
+
str(ref): ResolveMiss(code=str(entry.get("code", "")), message=str(entry.get("message", "")))
|
|
407
|
+
for ref, entry in payload["missing"].items()
|
|
408
|
+
}
|
|
409
|
+
except (KeyError, TypeError, ValueError) as exc:
|
|
410
|
+
raise SubmitApiError(f"Malformed assets/resolve response: {payload!r} ({exc})") from exc
|
|
411
|
+
return AssetsResolveResponse(resolved=resolved, missing=missing)
|
|
412
|
+
|
|
413
|
+
def create_job(
|
|
414
|
+
self,
|
|
415
|
+
*,
|
|
416
|
+
package_id: str,
|
|
417
|
+
job_name: str,
|
|
418
|
+
args: dict[str, Any],
|
|
419
|
+
viewstream: bool = False,
|
|
420
|
+
seed_from_job_id: Optional[str] = None,
|
|
421
|
+
assets: Optional[list[dict[str, str]]] = None,
|
|
422
|
+
app_name: Optional[str] = None,
|
|
423
|
+
system: Optional[str] = None,
|
|
424
|
+
) -> dict[str, Any]:
|
|
425
|
+
"""``POST /v1/jobs`` — create the job that executes *package_id*.
|
|
426
|
+
|
|
427
|
+
``app_name`` (live-view wave) is the submitting ``App``'s
|
|
428
|
+
declared name — display context the platform's live viewer shows next
|
|
429
|
+
to ``job_name``. An OPTIONAL top-level sibling of ``args`` following
|
|
430
|
+
the exact same package-hash-invariance rule as ``viewstream`` /
|
|
431
|
+
``seed_from_job_id`` / ``assets`` below (``args``' canonical JSON
|
|
432
|
+
feeds the content-addressed ``package_id``; a display property must
|
|
433
|
+
never perturb package identity). Sent only when non-empty; a control
|
|
434
|
+
plane that predates the field ignores the extra key, and a job
|
|
435
|
+
without one simply renders by ``job_name`` alone — best-effort
|
|
436
|
+
metadata, never load-bearing.
|
|
437
|
+
|
|
438
|
+
``viewstream`` is an OPTIONAL top-level body field — a sibling of
|
|
439
|
+
``args``, never nested inside it (``args`` feeds the ``package_id``
|
|
440
|
+
sha256; a streaming flag must never change a package's content-
|
|
441
|
+
addressed identity). Sent only when ``True``: a control plane that
|
|
442
|
+
predates the live-visualization plan's ``JobSubmitRequest`` field
|
|
443
|
+
simply ignores an absent key, same as it ignores any unknown field.
|
|
444
|
+
|
|
445
|
+
``system`` (system-tier-catalog-1362, M50-persistence) is the
|
|
446
|
+
already-validated tier wire value (e.g. ``"tier1"``) from the job's
|
|
447
|
+
own manifest ``resources["system"]`` — an OPTIONAL top-level sibling
|
|
448
|
+
of ``args`` for the same package-hash-invariance reason as
|
|
449
|
+
``viewstream`` above. Sent only when truthy: a control plane that
|
|
450
|
+
predates this field ignores the extra key, and the run proceeds
|
|
451
|
+
exactly as it did before ``system=`` existed.
|
|
452
|
+
|
|
453
|
+
``seed_from_job_id`` (``simulo run --from``, explicit-run-intent
|
|
454
|
+
plan) follows the exact same sibling-of-``args`` rule for the exact
|
|
455
|
+
same package-hash reason — a seed is a property of THIS submission,
|
|
456
|
+
never of the package. The value is passed through VERBATIM
|
|
457
|
+
(``<job-ref>[:best|:latest]``, ref = complete Job ID, full UUID, or an
|
|
458
|
+
unambiguous UUID prefix): the control plane owns resolution, org-scoped, in one
|
|
459
|
+
round trip — the client never resolves a model id itself. Two
|
|
460
|
+
seed-specific behaviors on top of the plain call:
|
|
461
|
+
|
|
462
|
+
* seed-validation failures are re-raised with actionable CLI copy
|
|
463
|
+
(:func:`_seed_error_from_http`; unrecognized codes propagate
|
|
464
|
+
unchanged);
|
|
465
|
+
* a 2xx response whose record carries NO ``seed`` provenance means
|
|
466
|
+
an older control plane silently ignored the request — that job
|
|
467
|
+
would train FRESH while claiming success, so it raises
|
|
468
|
+
:class:`SeedNotHonoredError` instead of returning (see that
|
|
469
|
+
class's docstring for why it never auto-cancels).
|
|
470
|
+
|
|
471
|
+
``assets`` (submit-time pin resolution) is the resolved
|
|
472
|
+
``[{ref, digest}]`` list — a top-level sibling of ``args``, on the
|
|
473
|
+
:data:`~simulo.interfaces.platform.asset_catalog.ASSETS_FIELD` key,
|
|
474
|
+
following the EXACT same package-hash-invariance rule as ``viewstream`` /
|
|
475
|
+
``seed_from_job_id``: an asset pin is a property of THIS submission, never
|
|
476
|
+
of the package, so it must never enter ``args`` (whose canonical JSON
|
|
477
|
+
feeds ``package_id``). Sent only when non-empty; the control plane
|
|
478
|
+
re-verifies every pin in the submit transaction and writes ``job_assets``.
|
|
479
|
+
|
|
480
|
+
Replay conflicts: the control plane's submit is idempotent on the
|
|
481
|
+
package's in-flight job, and refuses (409) a replay whose intent
|
|
482
|
+
differs from that job's rather than silently answering with it.
|
|
483
|
+
Both replay-conflict codes are mapped to :class:`SubmitApiError`
|
|
484
|
+
here, the ``_seed_error_from_http`` convention — unmapped they
|
|
485
|
+
would render as a raw wire dump
|
|
486
|
+
(``error: viewstream_replay_conflict: ...``) rather than copy
|
|
487
|
+
framed around what this submit asked for:
|
|
488
|
+
|
|
489
|
+
* ``viewstream_replay_conflict`` — framed with whether THIS submit
|
|
490
|
+
asked for ``--viewstream``, since either direction can conflict;
|
|
491
|
+
* ``seed_replay_conflict`` with NO ``--from`` on this submit — the
|
|
492
|
+
fresh-resubmit-of-a-seeded-job direction, which the ``--from``-
|
|
493
|
+
gated :func:`_seed_error_from_http` branch below cannot reach
|
|
494
|
+
(when ``--from`` WAS passed, that branch keeps handling it).
|
|
495
|
+
There is no seed REF to frame with, but the flag's absence is a
|
|
496
|
+
user-visible fact, so the copy is framed on it (``without
|
|
497
|
+
--from``) — the same treatment as the viewstream direction.
|
|
498
|
+
|
|
499
|
+
Returns the :class:`~simulo.interfaces.platform.runs.JobRecord`
|
|
500
|
+
fields. The HTTP status is the created-vs-replayed signal — 201 on
|
|
501
|
+
first submit, 200 on an idempotent replay of an in-flight job — and
|
|
502
|
+
it is the ONLY signal: the body is exactly the JobRecord fields on
|
|
503
|
+
both paths (kept free of extras so ``JobRecord(**payload)``
|
|
504
|
+
reconstruction stays valid), and this client treats both statuses
|
|
505
|
+
alike.
|
|
506
|
+
"""
|
|
507
|
+
body: dict[str, Any] = {"package_id": package_id, "job_name": job_name, "args": args}
|
|
508
|
+
if app_name:
|
|
509
|
+
body["app_name"] = app_name
|
|
510
|
+
if viewstream:
|
|
511
|
+
body["viewstream"] = True
|
|
512
|
+
if system:
|
|
513
|
+
body["system"] = system
|
|
514
|
+
if seed_from_job_id is not None:
|
|
515
|
+
body[_SEED_FROM_JOB_ID_FIELD] = seed_from_job_id
|
|
516
|
+
if assets:
|
|
517
|
+
body[ASSETS_FIELD] = assets
|
|
518
|
+
try:
|
|
519
|
+
payload = self._post_json(JOBS_ROUTE, body)
|
|
520
|
+
except http.HttpHTTPError as exc:
|
|
521
|
+
# NOTE (#830/#838 fix round, measured): an unmapped HttpHTTPError
|
|
522
|
+
# never tracebacks out of `simulo run` — cli.py's
|
|
523
|
+
# `except (JobsApiError, SubmitApiError)` catches it, because
|
|
524
|
+
# JobsApiError is an ALIAS of HttpError (jobs_api.py). These
|
|
525
|
+
# mappings exist to upgrade the resulting `error: <wire-code>:
|
|
526
|
+
# <message>` dump to framed CLI copy, not to prevent a crash.
|
|
527
|
+
if exc.code == "viewstream_replay_conflict":
|
|
528
|
+
requested = "with --viewstream" if viewstream else "without --viewstream"
|
|
529
|
+
raise SubmitApiError(f"Cannot re-submit {requested}: {exc.message}") from exc
|
|
530
|
+
if seed_from_job_id is not None:
|
|
531
|
+
friendly = _seed_error_from_http(exc)
|
|
532
|
+
if friendly is not None:
|
|
533
|
+
raise friendly from exc
|
|
534
|
+
elif exc.code == "seed_replay_conflict":
|
|
535
|
+
# NOTE: this elif pairs with `if seed_from_job_id is not
|
|
536
|
+
# None:` above — NOT with the nested `if friendly is not
|
|
537
|
+
# None:` that ends that block. It is the no---from direction
|
|
538
|
+
# of the same server guard (a FRESH re-submit of a package
|
|
539
|
+
# whose in-flight job was seeded), reachable only when no
|
|
540
|
+
# seed was requested on THIS submit. _seed_error_from_http
|
|
541
|
+
# frames its copy around the user's --from value, which does
|
|
542
|
+
# not exist here — so frame on the flag's ABSENCE instead;
|
|
543
|
+
# the server's message names the in-flight job and remedy.
|
|
544
|
+
raise SubmitApiError(f"Cannot re-submit without --from: {exc.message}") from exc
|
|
545
|
+
raise
|
|
546
|
+
if not isinstance(payload, dict) or not payload.get("job_id"):
|
|
547
|
+
raise SubmitApiError("Malformed create_job response: no machine job identity was returned.")
|
|
548
|
+
if seed_from_job_id is not None:
|
|
549
|
+
seed = payload.get("seed")
|
|
550
|
+
if not isinstance(seed, dict) or not seed.get("model_id"):
|
|
551
|
+
try:
|
|
552
|
+
raw_public_id = payload.get("public_id")
|
|
553
|
+
if not isinstance(raw_public_id, str):
|
|
554
|
+
raise TypeError("public_id is not a string")
|
|
555
|
+
public_id = str(parse_job_public_id(raw_public_id))
|
|
556
|
+
except (TypeError, ValueError):
|
|
557
|
+
public_id = None
|
|
558
|
+
created_job = f"Job {public_id}" if public_id is not None else "The job"
|
|
559
|
+
cancel_hint = (
|
|
560
|
+
f"`simulo cancel {public_id}`" if public_id is not None else "`simulo jobs`, then `simulo cancel`"
|
|
561
|
+
)
|
|
562
|
+
raise SeedNotHonoredError(
|
|
563
|
+
"The --from selection was NOT honored: this control plane does not "
|
|
564
|
+
"support continuing a run from a checkpoint (it ignored the request), so the "
|
|
565
|
+
f"job it created would train FRESH — the wrong result, reported as success.\n"
|
|
566
|
+
f"{created_job} was created WITHOUT the checkpoint. Stop it with {cancel_hint} "
|
|
567
|
+
"if you did not intend a fresh run, and re-run without --from "
|
|
568
|
+
"(or against a control plane that supports it)."
|
|
569
|
+
)
|
|
570
|
+
return payload
|
|
571
|
+
|
|
572
|
+
# ------------------------------------------------------------------
|
|
573
|
+
# Recordings
|
|
574
|
+
# ------------------------------------------------------------------
|
|
575
|
+
|
|
576
|
+
def list_recordings(self, job_id: str, *, scope: str = JOB_SCOPE_MINE) -> list[dict[str, Any]]:
|
|
577
|
+
"""All :class:`RecordingRecord` entries for *job_id*, across every page."""
|
|
578
|
+
path = JOB_RECORDINGS_ROUTE_TEMPLATE.format(job_id=http.quote_path_segment(job_id))
|
|
579
|
+
items: list[dict[str, Any]] = []
|
|
580
|
+
page = 1
|
|
581
|
+
while page <= _LIST_MAX_PAGES:
|
|
582
|
+
payload = self._get(path, self._page_query(_LIST_PAGE_LIMIT, page, scope))
|
|
583
|
+
page_items = payload.get("items") if isinstance(payload, dict) else None
|
|
584
|
+
if not isinstance(page_items, list):
|
|
585
|
+
raise SubmitApiError("Malformed recordings list response (no 'items' array).")
|
|
586
|
+
items.extend(item for item in page_items if isinstance(item, dict))
|
|
587
|
+
if len(page_items) < _LIST_PAGE_LIMIT:
|
|
588
|
+
return items
|
|
589
|
+
page += 1
|
|
590
|
+
return items
|
|
591
|
+
|
|
592
|
+
def list_recordings_page(
|
|
593
|
+
self, job_id: str, *, limit: int, page: int = 1, scope: str = JOB_SCOPE_MINE
|
|
594
|
+
) -> tuple[list[dict[str, Any]], int]:
|
|
595
|
+
"""One page (up to *limit* rows) of *job_id*'s recordings, plus the
|
|
596
|
+
total recording count for the job.
|
|
597
|
+
|
|
598
|
+
The single-request counterpart to :meth:`list_recordings` (which
|
|
599
|
+
loops every page — still used for ``simulo recordings --all`` and
|
|
600
|
+
NAME resolution, both of which need the complete set). Backs the
|
|
601
|
+
CLI's default recent-N view / ``--limit N`` in the bare (no NAME, no
|
|
602
|
+
``--all``) branch only; *page* defaults to the first, and the CLI's
|
|
603
|
+
``_fetch_list_view`` walks it forward for a view wider than one
|
|
604
|
+
server page.
|
|
605
|
+
"""
|
|
606
|
+
path = JOB_RECORDINGS_ROUTE_TEMPLATE.format(job_id=http.quote_path_segment(job_id))
|
|
607
|
+
payload = self._get(path, self._page_query(limit, page, scope))
|
|
608
|
+
items = payload.get("items") if isinstance(payload, dict) else None
|
|
609
|
+
total = payload.get("total") if isinstance(payload, dict) else None
|
|
610
|
+
if not isinstance(items, list) or not isinstance(total, int):
|
|
611
|
+
raise SubmitApiError("Malformed recordings list response (no 'items'/'total').")
|
|
612
|
+
return [item for item in items if isinstance(item, dict)], total
|
|
613
|
+
|
|
614
|
+
def get_recording_download_url(
|
|
615
|
+
self, job_id: str, recording_id: str, *, scope: str = JOB_SCOPE_MINE
|
|
616
|
+
) -> dict[str, Any]:
|
|
617
|
+
"""``GET .../recordings/{id}/download`` — resolve a short-lived download link."""
|
|
618
|
+
path = JOB_RECORDING_DOWNLOAD_ROUTE_TEMPLATE.format(
|
|
619
|
+
job_id=http.quote_path_segment(job_id), recording_id=http.quote_path_segment(recording_id)
|
|
620
|
+
)
|
|
621
|
+
payload = self._get(path, {"scope": scope})
|
|
622
|
+
if not isinstance(payload, dict) or not payload.get("url"):
|
|
623
|
+
raise SubmitApiError("Malformed recording download-link response.")
|
|
624
|
+
return payload
|
|
625
|
+
|
|
626
|
+
def download_recording(
|
|
627
|
+
self,
|
|
628
|
+
job_id: str,
|
|
629
|
+
recording_id: str,
|
|
630
|
+
dest_path: Path,
|
|
631
|
+
*,
|
|
632
|
+
scope: str = JOB_SCOPE_MINE,
|
|
633
|
+
destination_observation: Optional[DestinationObservation] = None,
|
|
634
|
+
replace_existing: bool = True,
|
|
635
|
+
) -> Path:
|
|
636
|
+
"""Resolve and download one recording's MCAP bytes to *dest_path*.
|
|
637
|
+
|
|
638
|
+
The bearer is attached to the download URL only when its host equals
|
|
639
|
+
the API host (see ``http.may_attach_bearer`` — the foreign-host rule
|
|
640
|
+
for presigned S3 URLs).
|
|
641
|
+
|
|
642
|
+
Any failure after temp creation may leave a mode-0600 ``.part`` file
|
|
643
|
+
because no backend can prove a pathname still identifies this
|
|
644
|
+
invocation's temp entry before deleting it. Ordinary write failures
|
|
645
|
+
name the retained entry; interrupts and process exits propagate
|
|
646
|
+
unchanged. The client never automatically deletes a prior ``.part``
|
|
647
|
+
file.
|
|
648
|
+
|
|
649
|
+
SECURITY (destination-directory race hardening): the destination
|
|
650
|
+
directory is SNAPSHOTTED before any network I/O begins
|
|
651
|
+
(``observe_destination`` — read-only, so a failed fetch still
|
|
652
|
+
leaves zero filesystem footprint, exactly as before), and the
|
|
653
|
+
eventual atomic write (same-directory temp file + fsync + rename)
|
|
654
|
+
is bound to that snapshot (``open_destination_directory``), never
|
|
655
|
+
to a fresh, unqualified lookup of the path string taken after the
|
|
656
|
+
fetch. That binding — not mere adjacency of two statements — is
|
|
657
|
+
what closes the race across the fetch's entire duration, including
|
|
658
|
+
every ancestor a nested ``-o a/b/c`` would create. The observation
|
|
659
|
+
may hold an open file descriptor (POSIX) for the entire fetch, so
|
|
660
|
+
it is released in a ``finally`` on every path — a digest mismatch
|
|
661
|
+
or fetch failure included, not only the success path. See
|
|
662
|
+
``_secure_downloads`` for exactly what this closes and the
|
|
663
|
+
documented POSIX/Windows difference.
|
|
664
|
+
"""
|
|
665
|
+
destination = destination_observation or observe_destination(dest_path.parent)
|
|
666
|
+
if not destination.matches_directory(dest_path.parent):
|
|
667
|
+
destination.close()
|
|
668
|
+
raise DestinationDirectoryError(
|
|
669
|
+
f"Refusing to use an observation for {destination.expanded} to write {dest_path}."
|
|
670
|
+
)
|
|
671
|
+
try:
|
|
672
|
+
link = self.get_recording_download_url(job_id, recording_id, scope=scope)
|
|
673
|
+
download_url = str(link["url"])
|
|
674
|
+
data = http.request_bytes(
|
|
675
|
+
"GET",
|
|
676
|
+
download_url,
|
|
677
|
+
token=self._token,
|
|
678
|
+
api_base_url=self._base_url,
|
|
679
|
+
timeout=_DOWNLOAD_TIMEOUT_S,
|
|
680
|
+
unavailable_hint=_UNAVAILABLE_HINT,
|
|
681
|
+
)
|
|
682
|
+
destination_dir = open_destination_directory(destination)
|
|
683
|
+
try:
|
|
684
|
+
destination_dir.write_atomic(dest_path.name, data, replace_existing=replace_existing)
|
|
685
|
+
finally:
|
|
686
|
+
destination_dir.close()
|
|
687
|
+
finally:
|
|
688
|
+
destination.close()
|
|
689
|
+
return dest_path
|
|
690
|
+
|
|
691
|
+
# ------------------------------------------------------------------
|
|
692
|
+
# Models (trained checkpoints) — the recordings surface's analogue
|
|
693
|
+
# ------------------------------------------------------------------
|
|
694
|
+
|
|
695
|
+
def list_models(self, job_id: str, *, scope: str = JOB_SCOPE_MINE) -> list[dict[str, Any]]:
|
|
696
|
+
"""All :class:`ModelRecord` entries for *job_id*, across every page."""
|
|
697
|
+
path = JOB_MODELS_ROUTE_TEMPLATE.format(job_id=http.quote_path_segment(job_id))
|
|
698
|
+
items: list[dict[str, Any]] = []
|
|
699
|
+
page = 1
|
|
700
|
+
while page <= _LIST_MAX_PAGES:
|
|
701
|
+
payload = self._get(path, self._page_query(_LIST_PAGE_LIMIT, page, scope))
|
|
702
|
+
page_items = payload.get("items") if isinstance(payload, dict) else None
|
|
703
|
+
if not isinstance(page_items, list):
|
|
704
|
+
raise SubmitApiError("Malformed models list response (no 'items' array).")
|
|
705
|
+
items.extend(item for item in page_items if isinstance(item, dict))
|
|
706
|
+
if len(page_items) < _LIST_PAGE_LIMIT:
|
|
707
|
+
return items
|
|
708
|
+
page += 1
|
|
709
|
+
return items
|
|
710
|
+
|
|
711
|
+
def list_models_page(
|
|
712
|
+
self, job_id: str, *, limit: int, page: int = 1, scope: str = JOB_SCOPE_MINE
|
|
713
|
+
) -> tuple[list[dict[str, Any]], int]:
|
|
714
|
+
"""One page (up to *limit* rows) of *job_id*'s models, plus the total
|
|
715
|
+
model count for the job.
|
|
716
|
+
|
|
717
|
+
The single-request counterpart to :meth:`list_models` (which loops
|
|
718
|
+
every page — still used for ``simulo models --all`` and NAME/id
|
|
719
|
+
resolution, both of which need the complete set). Backs the CLI's
|
|
720
|
+
default recent-N view / ``--limit N`` in the bare (no NAME, no
|
|
721
|
+
``--all``) branch only; *page* defaults to the first, and the CLI's
|
|
722
|
+
``_fetch_list_view`` walks it forward for a view wider than one
|
|
723
|
+
server page.
|
|
724
|
+
"""
|
|
725
|
+
path = JOB_MODELS_ROUTE_TEMPLATE.format(job_id=http.quote_path_segment(job_id))
|
|
726
|
+
payload = self._get(path, self._page_query(limit, page, scope))
|
|
727
|
+
items = payload.get("items") if isinstance(payload, dict) else None
|
|
728
|
+
total = payload.get("total") if isinstance(payload, dict) else None
|
|
729
|
+
if not isinstance(items, list) or not isinstance(total, int):
|
|
730
|
+
raise SubmitApiError("Malformed models list response (no 'items'/'total').")
|
|
731
|
+
return [item for item in items if isinstance(item, dict)], total
|
|
732
|
+
|
|
733
|
+
def get_model(self, job_id: str, model_id: str, *, scope: str = JOB_SCOPE_MINE) -> dict[str, Any]:
|
|
734
|
+
"""``GET .../models/{id}`` — one model's metadata record."""
|
|
735
|
+
path = JOB_MODEL_ROUTE_TEMPLATE.format(
|
|
736
|
+
job_id=http.quote_path_segment(job_id), model_id=http.quote_path_segment(model_id)
|
|
737
|
+
)
|
|
738
|
+
payload = self._get(path, {"scope": scope})
|
|
739
|
+
if not isinstance(payload, dict) or not payload.get("model_id"):
|
|
740
|
+
raise SubmitApiError("Malformed model record response.")
|
|
741
|
+
return payload
|
|
742
|
+
|
|
743
|
+
def get_model_download_url(self, job_id: str, model_id: str, *, scope: str = JOB_SCOPE_MINE) -> dict[str, Any]:
|
|
744
|
+
"""``GET .../models/{id}/download`` — resolve a short-lived download link."""
|
|
745
|
+
path = JOB_MODEL_DOWNLOAD_ROUTE_TEMPLATE.format(
|
|
746
|
+
job_id=http.quote_path_segment(job_id), model_id=http.quote_path_segment(model_id)
|
|
747
|
+
)
|
|
748
|
+
payload = self._get(path, {"scope": scope})
|
|
749
|
+
if not isinstance(payload, dict) or not payload.get("url"):
|
|
750
|
+
raise SubmitApiError("Malformed model download-link response.")
|
|
751
|
+
return payload
|
|
752
|
+
|
|
753
|
+
def download_model(
|
|
754
|
+
self,
|
|
755
|
+
job_id: str,
|
|
756
|
+
model_id: str,
|
|
757
|
+
dest_path: Path,
|
|
758
|
+
*,
|
|
759
|
+
expected_sha256: Optional[str] = None,
|
|
760
|
+
scope: str = JOB_SCOPE_MINE,
|
|
761
|
+
destination_observation: Optional[DestinationObservation] = None,
|
|
762
|
+
replace_existing: bool = True,
|
|
763
|
+
) -> Path:
|
|
764
|
+
"""Resolve and download one model's checkpoint bytes to *dest_path*.
|
|
765
|
+
|
|
766
|
+
When *expected_sha256* is given (the record's ``digest_sha256`` —
|
|
767
|
+
always present for models), the downloaded bytes are verified
|
|
768
|
+
against it BEFORE anything is written to *dest_path*: a mismatch
|
|
769
|
+
raises :class:`SubmitApiError` and leaves no partial/corrupt file
|
|
770
|
+
behind. The bearer is attached to the download URL only when its
|
|
771
|
+
host equals the API host (``http.may_attach_bearer`` — the
|
|
772
|
+
foreign-host rule for presigned S3 URLs).
|
|
773
|
+
|
|
774
|
+
The write itself is ATOMIC (a deliberate security hardening): the
|
|
775
|
+
bytes land in a temp file created in *dest_path*'s OWN directory
|
|
776
|
+
(guaranteeing the same filesystem, so the final rename is atomic on
|
|
777
|
+
POSIX and Windows alike) and are only ``os.replace``'d into
|
|
778
|
+
*dest_path* after every byte is written and fsync'd. A killed
|
|
779
|
+
process, a full disk mid-write, or two concurrent ``simulo models``
|
|
780
|
+
invocations targeting the same path can therefore never leave a
|
|
781
|
+
truncated/torn file at *dest_path* — a reader always sees either the
|
|
782
|
+
prior complete file (if any) or the new complete one, never a
|
|
783
|
+
partial write in between.
|
|
784
|
+
|
|
785
|
+
Any failure after temp creation may leave a mode-0600 ``.part`` file
|
|
786
|
+
because no backend can prove a pathname still identifies this
|
|
787
|
+
invocation's temp entry before deleting it. Ordinary write failures
|
|
788
|
+
name the retained entry; interrupts and process exits propagate
|
|
789
|
+
unchanged. The client never automatically deletes a prior ``.part``
|
|
790
|
+
file.
|
|
791
|
+
|
|
792
|
+
SECURITY (destination-directory race hardening): the destination
|
|
793
|
+
directory is SNAPSHOTTED before any network I/O begins — before
|
|
794
|
+
even the download URL is resolved (``observe_destination`` —
|
|
795
|
+
read-only, so a digest mismatch or failed fetch still leaves zero
|
|
796
|
+
filesystem footprint, exactly as before, not even the destination
|
|
797
|
+
directory created) — and the eventual atomic write is bound to
|
|
798
|
+
that snapshot (``open_destination_directory``), never to a fresh,
|
|
799
|
+
unqualified lookup of the path string taken after the fetch. That
|
|
800
|
+
binding — not mere adjacency of two statements — is what closes
|
|
801
|
+
the race across the fetch's entire duration, including every
|
|
802
|
+
ancestor a nested ``-o a/b/c`` would create. The observation may
|
|
803
|
+
hold an open file descriptor (POSIX) for the entire fetch, so it
|
|
804
|
+
is released in a ``finally`` on every path — a digest mismatch or
|
|
805
|
+
fetch failure included, not only the success path. See
|
|
806
|
+
``_secure_downloads`` for exactly what this closes and the
|
|
807
|
+
documented POSIX/Windows difference.
|
|
808
|
+
|
|
809
|
+
``dest_path`` is the caller's own destination, but its BASENAME is
|
|
810
|
+
routinely server-derived. Directory-form CLI callers build it from
|
|
811
|
+
``_portable_directory_download_filename(record["name"])``; that
|
|
812
|
+
helper rejects raw or normalization-created path separators, chooses
|
|
813
|
+
a portable canonical basename, rejects NUL and Windows-unsafe forms,
|
|
814
|
+
and removes Unicode control/format characters. So *dest_path* remains
|
|
815
|
+
untrusted display text wherever
|
|
816
|
+
this method renders it, and the mismatch raise above routes it
|
|
817
|
+
through :func:`_safe_server_text` as defense in depth. Callers remain
|
|
818
|
+
responsible for the filesystem half before constructing it.
|
|
819
|
+
"""
|
|
820
|
+
destination = destination_observation or observe_destination(dest_path.parent)
|
|
821
|
+
if not destination.matches_directory(dest_path.parent):
|
|
822
|
+
destination.close()
|
|
823
|
+
raise DestinationDirectoryError(
|
|
824
|
+
f"Refusing to use an observation for {destination.expanded} to write {dest_path}."
|
|
825
|
+
)
|
|
826
|
+
try:
|
|
827
|
+
link = self.get_model_download_url(job_id, model_id, scope=scope)
|
|
828
|
+
download_url = str(link["url"])
|
|
829
|
+
data = http.request_bytes(
|
|
830
|
+
"GET",
|
|
831
|
+
download_url,
|
|
832
|
+
token=self._token,
|
|
833
|
+
api_base_url=self._base_url,
|
|
834
|
+
timeout=_DOWNLOAD_TIMEOUT_S,
|
|
835
|
+
unavailable_hint=_UNAVAILABLE_HINT,
|
|
836
|
+
)
|
|
837
|
+
if expected_sha256:
|
|
838
|
+
expected = expected_sha256.lower().removeprefix("sha256:")
|
|
839
|
+
actual = hashlib.sha256(data).hexdigest()
|
|
840
|
+
if actual != expected:
|
|
841
|
+
raise SubmitApiError(
|
|
842
|
+
"Model download digest mismatch for the selected Model: "
|
|
843
|
+
f"expected sha256:{_safe_server_text(expected)}, "
|
|
844
|
+
f"got sha256:{actual} ({len(data)} bytes). "
|
|
845
|
+
f"Nothing was written to {_safe_server_text(dest_path)}."
|
|
846
|
+
)
|
|
847
|
+
destination_dir = open_destination_directory(destination)
|
|
848
|
+
try:
|
|
849
|
+
destination_dir.write_atomic(dest_path.name, data, replace_existing=replace_existing)
|
|
850
|
+
finally:
|
|
851
|
+
destination_dir.close()
|
|
852
|
+
finally:
|
|
853
|
+
destination.close()
|
|
854
|
+
return dest_path
|
|
855
|
+
|
|
856
|
+
# ------------------------------------------------------------------
|
|
857
|
+
# Outputs (Plan B, #531, PR-B10) — the read-time UNION over recordings,
|
|
858
|
+
# models, and general job_artifacts. A projection, not a home: the same
|
|
859
|
+
# bytes stay downloadable from this route AND the output's original
|
|
860
|
+
# /recordings or /models surface (contract docstring on
|
|
861
|
+
# JOB_OUTPUT_DOWNLOAD_ROUTE_TEMPLATE) — this client method set exists
|
|
862
|
+
# so `simulo outputs` has one listing/download surface spanning all
|
|
863
|
+
# three, including the general kind (report/video/dataset/file/
|
|
864
|
+
# anomaly_capture) that has no other CLI download path.
|
|
865
|
+
# ------------------------------------------------------------------
|
|
866
|
+
|
|
867
|
+
def list_outputs(self, job_id: str, *, scope: str = JOB_SCOPE_MINE) -> list[dict[str, Any]]:
|
|
868
|
+
"""All :class:`OutputRecord` entries for *job_id*, across every page.
|
|
869
|
+
|
|
870
|
+
``scope`` (contract ``runs.JOB_SCOPE_MINE``/``JOB_SCOPE_ORG``) is sent
|
|
871
|
+
explicitly on every request. ``simulo export`` passes
|
|
872
|
+
``scope="org"``: the export job that owns a policy bundle is
|
|
873
|
+
SYSTEM-submitted (``submitted_by_user_id=None``), so the server's
|
|
874
|
+
``scope=mine`` default answers the non-enumerating 404 for it. The
|
|
875
|
+
same applies to :meth:`download_output`, which resolves access from
|
|
876
|
+
the same job.
|
|
877
|
+
"""
|
|
878
|
+
path = JOB_OUTPUTS_ROUTE_TEMPLATE.format(job_id=http.quote_path_segment(job_id))
|
|
879
|
+
items: list[dict[str, Any]] = []
|
|
880
|
+
page = 1
|
|
881
|
+
while page <= _LIST_MAX_PAGES:
|
|
882
|
+
payload = self._get(path, self._page_query(_LIST_PAGE_LIMIT, page, scope))
|
|
883
|
+
page_items = payload.get("items") if isinstance(payload, dict) else None
|
|
884
|
+
if not isinstance(page_items, list):
|
|
885
|
+
raise SubmitApiError("Malformed outputs list response (no 'items' array).")
|
|
886
|
+
try:
|
|
887
|
+
items.extend(_output_record_from_payload(item) for item in page_items)
|
|
888
|
+
except (TypeError, ValueError) as exc:
|
|
889
|
+
raise SubmitApiError(f"Malformed outputs list response: {_safe_server_text(exc)}") from exc
|
|
890
|
+
if len(page_items) < _LIST_PAGE_LIMIT:
|
|
891
|
+
return items
|
|
892
|
+
page += 1
|
|
893
|
+
return items
|
|
894
|
+
|
|
895
|
+
def list_outputs_page(
|
|
896
|
+
self, job_id: str, *, limit: int, page: int = 1, scope: str = JOB_SCOPE_MINE
|
|
897
|
+
) -> tuple[list[dict[str, Any]], int]:
|
|
898
|
+
"""One page (up to *limit* rows) of *job_id*'s outputs, plus the
|
|
899
|
+
total output count for the job.
|
|
900
|
+
|
|
901
|
+
The single-request counterpart to :meth:`list_outputs` (which loops
|
|
902
|
+
every page — used for ``simulo outputs --all`` and NAME/id
|
|
903
|
+
resolution, both of which need the complete set). Backs the CLI's
|
|
904
|
+
default recent-N view / ``--limit N`` in the bare (no NAME, no
|
|
905
|
+
``--all``) branch only; *page* defaults to the first, and the CLI's
|
|
906
|
+
``_fetch_list_view`` walks it forward for a view wider than one
|
|
907
|
+
server page.
|
|
908
|
+
"""
|
|
909
|
+
path = JOB_OUTPUTS_ROUTE_TEMPLATE.format(job_id=http.quote_path_segment(job_id))
|
|
910
|
+
payload = self._get(path, self._page_query(limit, page, scope))
|
|
911
|
+
items = payload.get("items") if isinstance(payload, dict) else None
|
|
912
|
+
total = payload.get("total") if isinstance(payload, dict) else None
|
|
913
|
+
if not isinstance(items, list) or not isinstance(total, int):
|
|
914
|
+
raise SubmitApiError("Malformed outputs list response (no 'items'/'total').")
|
|
915
|
+
try:
|
|
916
|
+
records = [_output_record_from_payload(item) for item in items]
|
|
917
|
+
except (TypeError, ValueError) as exc:
|
|
918
|
+
raise SubmitApiError(f"Malformed outputs list response: {_safe_server_text(exc)}") from exc
|
|
919
|
+
return records, total
|
|
920
|
+
|
|
921
|
+
def get_output_download_url(self, job_id: str, output_id: str, *, scope: str = JOB_SCOPE_MINE) -> dict[str, Any]:
|
|
922
|
+
"""``GET .../outputs/{id}/download`` — resolve a short-lived download link.
|
|
923
|
+
|
|
924
|
+
``scope``: see :meth:`list_outputs`. The download route resolves
|
|
925
|
+
access from the OWNING job exactly as the listing does, so a scope that
|
|
926
|
+
is right for the listing and absent here is a 404 halfway through a
|
|
927
|
+
download — they must move together.
|
|
928
|
+
"""
|
|
929
|
+
path = JOB_OUTPUT_DOWNLOAD_ROUTE_TEMPLATE.format(
|
|
930
|
+
job_id=http.quote_path_segment(job_id), output_id=http.quote_path_segment(output_id)
|
|
931
|
+
)
|
|
932
|
+
payload = self._get(path, {"scope": scope})
|
|
933
|
+
if not isinstance(payload, dict) or not payload.get("url"):
|
|
934
|
+
raise SubmitApiError("Malformed output download-link response.")
|
|
935
|
+
return payload
|
|
936
|
+
|
|
937
|
+
def download_output(
|
|
938
|
+
self,
|
|
939
|
+
job_id: str,
|
|
940
|
+
output_id: str,
|
|
941
|
+
dest_path: Path,
|
|
942
|
+
*,
|
|
943
|
+
expected_sha256: Optional[str] = None,
|
|
944
|
+
scope: str = JOB_SCOPE_MINE,
|
|
945
|
+
destination_observation: Optional[DestinationObservation] = None,
|
|
946
|
+
replace_existing: bool = True,
|
|
947
|
+
) -> Path:
|
|
948
|
+
"""Resolve and download one output's bytes to *dest_path*.
|
|
949
|
+
|
|
950
|
+
Mirrors :meth:`download_model`'s atomic-write pattern (a same-directory
|
|
951
|
+
temp file + ``os.fsync`` + ``os.replace``, so a killed process or a
|
|
952
|
+
full disk can never leave a truncated file at *dest_path*) rather than
|
|
953
|
+
the simpler ``download_recording`` write — a general output can be
|
|
954
|
+
as large as a video or dataset, and the same digest-verification path
|
|
955
|
+
applies when the caller has one.
|
|
956
|
+
|
|
957
|
+
Any failure after temp creation may leave a mode-0600 ``.part`` file
|
|
958
|
+
because no backend can prove a pathname still identifies this
|
|
959
|
+
invocation's temp entry before deleting it. Ordinary write failures
|
|
960
|
+
name the retained entry; interrupts and process exits propagate
|
|
961
|
+
unchanged. The client never automatically deletes a prior ``.part``
|
|
962
|
+
file.
|
|
963
|
+
|
|
964
|
+
*expected_sha256* is the record's ``digest_sha256``, verified BEFORE
|
|
965
|
+
anything is written when given. Pass ``None`` — not ``""`` — when the
|
|
966
|
+
record's digest is empty: :class:`~simulo.interfaces.platform.
|
|
967
|
+
outputs.OutputRecord` documents ``""`` as "no digest stored" for a
|
|
968
|
+
row whose home table records none (the recordings convention), which
|
|
969
|
+
means skip-verify, never a mismatch.
|
|
970
|
+
|
|
971
|
+
SECURITY (destination-directory race hardening): the destination
|
|
972
|
+
directory is SNAPSHOTTED before any network I/O begins — before
|
|
973
|
+
even the download URL is resolved (``observe_destination`` —
|
|
974
|
+
read-only, so a digest mismatch or failed fetch still leaves zero
|
|
975
|
+
filesystem footprint, exactly as before, not even the destination
|
|
976
|
+
directory created) — and the eventual atomic write is bound to
|
|
977
|
+
that snapshot (``open_destination_directory``), never to a fresh,
|
|
978
|
+
unqualified lookup of the path string taken after the fetch. That
|
|
979
|
+
binding — not mere adjacency of two statements — is what closes
|
|
980
|
+
the race across the fetch's entire duration, including every
|
|
981
|
+
ancestor a nested ``-o a/b/c`` would create. The observation may
|
|
982
|
+
hold an open file descriptor (POSIX) for the entire fetch, so it
|
|
983
|
+
is released in a ``finally`` on every path — a digest mismatch or
|
|
984
|
+
fetch failure included, not only the success path. See
|
|
985
|
+
``_secure_downloads`` for exactly what this closes and the
|
|
986
|
+
documented POSIX/Windows difference.
|
|
987
|
+
|
|
988
|
+
``scope``: see :meth:`list_outputs`.
|
|
989
|
+
"""
|
|
990
|
+
destination = destination_observation or observe_destination(dest_path.parent)
|
|
991
|
+
if not destination.matches_directory(dest_path.parent):
|
|
992
|
+
destination.close()
|
|
993
|
+
raise DestinationDirectoryError(
|
|
994
|
+
f"Refusing to use an observation for {destination.expanded} to write {dest_path}."
|
|
995
|
+
)
|
|
996
|
+
try:
|
|
997
|
+
link = self.get_output_download_url(job_id, output_id, scope=scope)
|
|
998
|
+
download_url = str(link["url"])
|
|
999
|
+
data = http.request_bytes(
|
|
1000
|
+
"GET",
|
|
1001
|
+
download_url,
|
|
1002
|
+
token=self._token,
|
|
1003
|
+
api_base_url=self._base_url,
|
|
1004
|
+
timeout=_DOWNLOAD_TIMEOUT_S,
|
|
1005
|
+
unavailable_hint=_UNAVAILABLE_HINT,
|
|
1006
|
+
)
|
|
1007
|
+
if expected_sha256:
|
|
1008
|
+
expected = expected_sha256.lower().removeprefix("sha256:")
|
|
1009
|
+
actual = hashlib.sha256(data).hexdigest()
|
|
1010
|
+
if actual != expected:
|
|
1011
|
+
raise SubmitApiError(
|
|
1012
|
+
"Output download digest mismatch for the selected Output: "
|
|
1013
|
+
f"expected sha256:{_safe_server_text(expected)}, "
|
|
1014
|
+
f"got sha256:{actual} ({len(data)} bytes). "
|
|
1015
|
+
f"Nothing was written to {_safe_server_text(dest_path)}."
|
|
1016
|
+
)
|
|
1017
|
+
destination_dir = open_destination_directory(destination)
|
|
1018
|
+
try:
|
|
1019
|
+
destination_dir.write_atomic(dest_path.name, data, replace_existing=replace_existing)
|
|
1020
|
+
finally:
|
|
1021
|
+
destination_dir.close()
|
|
1022
|
+
finally:
|
|
1023
|
+
destination.close()
|
|
1024
|
+
return dest_path
|
|
1025
|
+
|
|
1026
|
+
# ------------------------------------------------------------------
|
|
1027
|
+
# Internals
|
|
1028
|
+
# ------------------------------------------------------------------
|
|
1029
|
+
|
|
1030
|
+
@staticmethod
|
|
1031
|
+
def _page_query(limit: int, page: int, scope: str) -> dict[str, str]:
|
|
1032
|
+
"""Build one explicitly scoped pagination query."""
|
|
1033
|
+
return {"limit": str(limit), "page": str(page), "scope": scope}
|
|
1034
|
+
|
|
1035
|
+
def _get(self, path: str, query: Optional[dict[str, str]] = None) -> Any:
|
|
1036
|
+
url = self._base_url + path
|
|
1037
|
+
if query:
|
|
1038
|
+
url += "?" + urllib.parse.urlencode(query)
|
|
1039
|
+
return http.request_json(
|
|
1040
|
+
"GET",
|
|
1041
|
+
url,
|
|
1042
|
+
token=self._token,
|
|
1043
|
+
api_base_url=self._base_url,
|
|
1044
|
+
timeout=_REQUEST_TIMEOUT_S,
|
|
1045
|
+
unavailable_hint=_UNAVAILABLE_HINT,
|
|
1046
|
+
)
|
|
1047
|
+
|
|
1048
|
+
def _post_json(self, path: str, body: dict[str, Any]) -> Any:
|
|
1049
|
+
return http.request_json(
|
|
1050
|
+
"POST",
|
|
1051
|
+
self._base_url + path,
|
|
1052
|
+
token=self._token,
|
|
1053
|
+
api_base_url=self._base_url,
|
|
1054
|
+
json_body=body,
|
|
1055
|
+
timeout=_REQUEST_TIMEOUT_S,
|
|
1056
|
+
unavailable_hint=_UNAVAILABLE_HINT,
|
|
1057
|
+
)
|