mimiry 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- mimiry/__init__.py +54 -0
- mimiry/_auth.py +70 -0
- mimiry/_exceptions.py +48 -0
- mimiry/_session.py +262 -0
- mimiry/client.py +591 -0
- mimiry-0.1.0.dist-info/METADATA +118 -0
- mimiry-0.1.0.dist-info/RECORD +10 -0
- mimiry-0.1.0.dist-info/WHEEL +5 -0
- mimiry-0.1.0.dist-info/licenses/LICENSE +21 -0
- mimiry-0.1.0.dist-info/top_level.txt +1 -0
mimiry/__init__.py
ADDED
|
@@ -0,0 +1,54 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Mimiry Python SDK
|
|
3
|
+
=================
|
|
4
|
+
|
|
5
|
+
Simple, high-level client for the Mimiry compute platform.
|
|
6
|
+
|
|
7
|
+
Quick start::
|
|
8
|
+
|
|
9
|
+
from mimiry import MimiryClient
|
|
10
|
+
|
|
11
|
+
client = MimiryClient()
|
|
12
|
+
|
|
13
|
+
# Run a command and get the output
|
|
14
|
+
result = client.run("nvidia-smi")
|
|
15
|
+
result.print_logs()
|
|
16
|
+
|
|
17
|
+
# Run a Python script
|
|
18
|
+
result = client.run_script("train.py", verbose=True)
|
|
19
|
+
|
|
20
|
+
# Create a session for SSH access
|
|
21
|
+
session = client.create_session(auto_terminate=False)
|
|
22
|
+
session.wait_until_running()
|
|
23
|
+
print(session.ssh_command())
|
|
24
|
+
|
|
25
|
+
Authentication:
|
|
26
|
+
Run `mimiry auth login` once before using this SDK.
|
|
27
|
+
Tokens are refreshed automatically.
|
|
28
|
+
"""
|
|
29
|
+
|
|
30
|
+
from .client import (
|
|
31
|
+
CUDA_BASE,
|
|
32
|
+
PYTORCH_IMAGE,
|
|
33
|
+
PYTORCH_NGC,
|
|
34
|
+
MimiryClient,
|
|
35
|
+
RunResult,
|
|
36
|
+
)
|
|
37
|
+
from ._exceptions import AuthError, LogsError, MimiryError, QuotaError, SessionError
|
|
38
|
+
from ._session import Session
|
|
39
|
+
|
|
40
|
+
__all__ = [
|
|
41
|
+
"MimiryClient",
|
|
42
|
+
"RunResult",
|
|
43
|
+
"Session",
|
|
44
|
+
"CUDA_BASE",
|
|
45
|
+
"PYTORCH_IMAGE",
|
|
46
|
+
"PYTORCH_NGC",
|
|
47
|
+
"MimiryError",
|
|
48
|
+
"AuthError",
|
|
49
|
+
"QuotaError",
|
|
50
|
+
"SessionError",
|
|
51
|
+
"LogsError",
|
|
52
|
+
]
|
|
53
|
+
|
|
54
|
+
__version__ = "0.1.0"
|
mimiry/_auth.py
ADDED
|
@@ -0,0 +1,70 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Token management for the Mimiry SDK.
|
|
3
|
+
|
|
4
|
+
Platform note: JWTs issued by `mimiry auth token --refresh` expire in
|
|
5
|
+
approximately 5-8 minutes. Any API call after expiry returns 401, and
|
|
6
|
+
cleanup DELETE calls silently fail, leaving sessions running and billing
|
|
7
|
+
the account. This manager refreshes proactively every 4 minutes.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
import json
|
|
11
|
+
import subprocess
|
|
12
|
+
import time
|
|
13
|
+
|
|
14
|
+
from ._exceptions import AuthError
|
|
15
|
+
|
|
16
|
+
# Refresh well before the ~5-8 min expiry window
|
|
17
|
+
_REFRESH_INTERVAL = 240 # 4 minutes
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
class TokenManager:
|
|
21
|
+
"""
|
|
22
|
+
Retrieves and caches a Mimiry JWT, auto-refreshing before expiry.
|
|
23
|
+
|
|
24
|
+
Calls `mimiry auth token --refresh --json` under the hood.
|
|
25
|
+
Run `mimiry auth login` once before using the SDK.
|
|
26
|
+
"""
|
|
27
|
+
|
|
28
|
+
def __init__(self):
|
|
29
|
+
self._token: str | None = None
|
|
30
|
+
self._fetched_at: float = 0.0
|
|
31
|
+
|
|
32
|
+
def get(self, force: bool = False) -> str:
|
|
33
|
+
"""Return a valid token, refreshing from the CLI if needed."""
|
|
34
|
+
age = time.time() - self._fetched_at
|
|
35
|
+
if force or not self._token or age >= _REFRESH_INTERVAL:
|
|
36
|
+
self._token = self._fetch()
|
|
37
|
+
self._fetched_at = time.time()
|
|
38
|
+
return self._token
|
|
39
|
+
|
|
40
|
+
def _fetch(self) -> str:
|
|
41
|
+
result = subprocess.run(
|
|
42
|
+
["mimiry", "auth", "token", "--refresh", "--json"],
|
|
43
|
+
capture_output=True,
|
|
44
|
+
text=True,
|
|
45
|
+
timeout=30,
|
|
46
|
+
)
|
|
47
|
+
if result.returncode != 0:
|
|
48
|
+
raise AuthError(
|
|
49
|
+
"Failed to get auth token. Run: mimiry auth login\n"
|
|
50
|
+
f"stderr: {result.stderr.strip()}"
|
|
51
|
+
)
|
|
52
|
+
|
|
53
|
+
# Try JSON response first (CLI >= 1.1)
|
|
54
|
+
try:
|
|
55
|
+
data = json.loads(result.stdout)
|
|
56
|
+
token = data.get("access_token") or data.get("token")
|
|
57
|
+
if token:
|
|
58
|
+
return token
|
|
59
|
+
except json.JSONDecodeError:
|
|
60
|
+
pass
|
|
61
|
+
|
|
62
|
+
# Fallback: older CLI versions print a bare JWT line
|
|
63
|
+
for line in result.stdout.splitlines():
|
|
64
|
+
line = line.strip()
|
|
65
|
+
if line.startswith("eyJ"):
|
|
66
|
+
return line
|
|
67
|
+
|
|
68
|
+
raise AuthError(
|
|
69
|
+
f"Could not parse token from CLI output: {result.stdout[:300]!r}"
|
|
70
|
+
)
|
mimiry/_exceptions.py
ADDED
|
@@ -0,0 +1,48 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Mimiry SDK exception hierarchy.
|
|
3
|
+
"""
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
class MimiryError(Exception):
|
|
7
|
+
"""Base exception for all Mimiry SDK errors."""
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
class AuthError(MimiryError):
|
|
11
|
+
"""
|
|
12
|
+
Authentication failed or token could not be retrieved.
|
|
13
|
+
Fix: run `mimiry auth login` then retry.
|
|
14
|
+
"""
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
class QuotaError(MimiryError):
|
|
18
|
+
"""
|
|
19
|
+
Account quota exceeded or insufficient credits.
|
|
20
|
+
Check balance with client.balance() and active sessions with client.sessions().
|
|
21
|
+
"""
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
class SessionError(MimiryError):
|
|
25
|
+
"""
|
|
26
|
+
Session creation, provisioning, or execution failed.
|
|
27
|
+
The session ID is available on the exception when applicable.
|
|
28
|
+
"""
|
|
29
|
+
|
|
30
|
+
def __init__(self, message, session_id=None):
|
|
31
|
+
super().__init__(message)
|
|
32
|
+
self.session_id = session_id
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
class LogsError(MimiryError):
|
|
36
|
+
"""
|
|
37
|
+
Failed to retrieve session logs after all retries.
|
|
38
|
+
This typically means the VM setup took longer than expected.
|
|
39
|
+
"""
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
class _HttpError(Exception):
|
|
43
|
+
"""Internal: raised for non-2xx HTTP responses."""
|
|
44
|
+
|
|
45
|
+
def __init__(self, status, body):
|
|
46
|
+
self.status = status
|
|
47
|
+
self.body = body
|
|
48
|
+
super().__init__(f"HTTP {status}: {body}")
|
mimiry/_session.py
ADDED
|
@@ -0,0 +1,262 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Session resource — wraps a single Mimiry compute session.
|
|
3
|
+
|
|
4
|
+
Platform quirks handled here (transparent to callers):
|
|
5
|
+
- status='running' means VM SSH is up, NOT Docker/GPU ready (Bug 2)
|
|
6
|
+
- /logs returns 503 'vm_setup_in_progress' for ~60-90s after 'running'
|
|
7
|
+
- Token is auto-refreshed before every network call
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
import time
|
|
11
|
+
|
|
12
|
+
from ._exceptions import LogsError, SessionError, _HttpError
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
class Session:
|
|
16
|
+
"""
|
|
17
|
+
A Mimiry compute session.
|
|
18
|
+
|
|
19
|
+
Returned by MimiryClient.create_session() and MimiryClient.run().
|
|
20
|
+
Do not construct directly.
|
|
21
|
+
|
|
22
|
+
Attributes:
|
|
23
|
+
id : Session UUID.
|
|
24
|
+
status : Last-known status string (call refresh() to update).
|
|
25
|
+
host : SSH hostname (None until provisioning completes).
|
|
26
|
+
port : SSH port (default 22).
|
|
27
|
+
username : SSH username (default 'ubuntu').
|
|
28
|
+
"""
|
|
29
|
+
|
|
30
|
+
def __init__(self, session_id: str, client):
|
|
31
|
+
self.id = session_id
|
|
32
|
+
self._client = client
|
|
33
|
+
self.status = "pending"
|
|
34
|
+
self.host: str | None = None
|
|
35
|
+
self.port: int = 22
|
|
36
|
+
self.username: str = "ubuntu"
|
|
37
|
+
|
|
38
|
+
# ── State ──────────────────────────────────────────────────────────────────
|
|
39
|
+
|
|
40
|
+
def refresh(self) -> "Session":
|
|
41
|
+
"""Fetch the latest session state from the API."""
|
|
42
|
+
data = self._client._get(f"/sessions/{self.id}")
|
|
43
|
+
self.status = data.get("status", "unknown")
|
|
44
|
+
ssh = data.get("ssh") or {}
|
|
45
|
+
if ssh.get("host"):
|
|
46
|
+
self.host = ssh["host"]
|
|
47
|
+
self.port = ssh.get("port", 22)
|
|
48
|
+
self.username = ssh.get("username", "ubuntu")
|
|
49
|
+
return self
|
|
50
|
+
|
|
51
|
+
def wait_until_running(self, timeout: int = 300, poll_interval: int = 5) -> "Session":
|
|
52
|
+
"""
|
|
53
|
+
Block until status == 'running' (VM SSH daemon is up).
|
|
54
|
+
|
|
55
|
+
Typical time: 23-55 seconds. After this returns, wait another
|
|
56
|
+
60-90 seconds before calling logs() — Docker and the GPU driver
|
|
57
|
+
are still loading.
|
|
58
|
+
|
|
59
|
+
Raises:
|
|
60
|
+
SessionError: if the session fails or the timeout is exceeded.
|
|
61
|
+
"""
|
|
62
|
+
start = time.time()
|
|
63
|
+
last_status = ""
|
|
64
|
+
|
|
65
|
+
while True:
|
|
66
|
+
elapsed = time.time() - start
|
|
67
|
+
if elapsed >= timeout:
|
|
68
|
+
raise SessionError(
|
|
69
|
+
f"Timed out after {timeout}s waiting for 'running' "
|
|
70
|
+
f"(last status: {self.status})",
|
|
71
|
+
session_id=self.id,
|
|
72
|
+
)
|
|
73
|
+
|
|
74
|
+
self.refresh()
|
|
75
|
+
|
|
76
|
+
if self.status != last_status:
|
|
77
|
+
last_status = self.status
|
|
78
|
+
|
|
79
|
+
if self.status == "running":
|
|
80
|
+
return self
|
|
81
|
+
|
|
82
|
+
if self.status in ("completed", "done", "succeeded"):
|
|
83
|
+
# Fast session that completed before we polled
|
|
84
|
+
return self
|
|
85
|
+
|
|
86
|
+
if self.status in ("failed", "terminated", "cancelled"):
|
|
87
|
+
raise SessionError(
|
|
88
|
+
f"Session {self.id} reached '{self.status}' during provisioning",
|
|
89
|
+
session_id=self.id,
|
|
90
|
+
)
|
|
91
|
+
|
|
92
|
+
time.sleep(poll_interval)
|
|
93
|
+
|
|
94
|
+
# ── Logs ───────────────────────────────────────────────────────────────────
|
|
95
|
+
|
|
96
|
+
def logs(self, tail: int = 500, max_retries: int = 12, initial_delay: int = 15) -> str:
|
|
97
|
+
"""
|
|
98
|
+
Fetch container logs, retrying automatically on 503.
|
|
99
|
+
|
|
100
|
+
The /logs endpoint returns 503 'vm_setup_in_progress' for ~60-90s
|
|
101
|
+
after status=running while Docker and the GPU driver finish loading.
|
|
102
|
+
This is handled transparently with exponential backoff.
|
|
103
|
+
|
|
104
|
+
Returns:
|
|
105
|
+
Log text as a string. Empty string if the container produced
|
|
106
|
+
no output (normal for sessions without a command field set).
|
|
107
|
+
|
|
108
|
+
Raises:
|
|
109
|
+
LogsError: if all retry attempts are exhausted.
|
|
110
|
+
"""
|
|
111
|
+
delay = initial_delay
|
|
112
|
+
|
|
113
|
+
for attempt in range(1, max_retries + 1):
|
|
114
|
+
self._client._tokens.get() # proactive token refresh
|
|
115
|
+
|
|
116
|
+
try:
|
|
117
|
+
data = self._client._get(
|
|
118
|
+
f"/sessions/{self.id}/logs?tail={tail}×tamps=false"
|
|
119
|
+
)
|
|
120
|
+
return data.get("logs") or ""
|
|
121
|
+
|
|
122
|
+
except _HttpError as exc:
|
|
123
|
+
if exc.status == 503:
|
|
124
|
+
retry_after = (exc.body or {}).get("retry_after_seconds", delay)
|
|
125
|
+
time.sleep(retry_after)
|
|
126
|
+
delay = min(delay * 2, 60)
|
|
127
|
+
continue
|
|
128
|
+
if exc.status == 409:
|
|
129
|
+
# Session not in a loggable state — may have already exited cleanly
|
|
130
|
+
return ""
|
|
131
|
+
raise LogsError(
|
|
132
|
+
f"Unexpected HTTP {exc.status} fetching logs "
|
|
133
|
+
f"for session {self.id}: {exc.body}"
|
|
134
|
+
) from exc
|
|
135
|
+
|
|
136
|
+
raise LogsError(
|
|
137
|
+
f"Could not fetch logs for session {self.id} "
|
|
138
|
+
f"after {max_retries} attempts (VM setup still in progress)"
|
|
139
|
+
)
|
|
140
|
+
|
|
141
|
+
def poll_logs(
|
|
142
|
+
self,
|
|
143
|
+
tail: int = 1000,
|
|
144
|
+
poll_interval: int = 30,
|
|
145
|
+
timeout: int = 2400,
|
|
146
|
+
stop_pattern: str | None = None,
|
|
147
|
+
on_output=None,
|
|
148
|
+
) -> str:
|
|
149
|
+
"""
|
|
150
|
+
Poll logs repeatedly until the session ends or a stop_pattern matches.
|
|
151
|
+
|
|
152
|
+
Designed for long-running jobs (e.g. ML training) where you want to
|
|
153
|
+
see progress as it happens. Token is refreshed automatically.
|
|
154
|
+
|
|
155
|
+
Args:
|
|
156
|
+
tail: Number of log lines to fetch per poll.
|
|
157
|
+
poll_interval: Seconds between polls once logs are flowing.
|
|
158
|
+
timeout: Hard stop in seconds.
|
|
159
|
+
stop_pattern: Stop when this string appears in the logs.
|
|
160
|
+
on_output: Optional callable(new_text) — called with each
|
|
161
|
+
new chunk of log content.
|
|
162
|
+
|
|
163
|
+
Returns:
|
|
164
|
+
The final complete log string (all lines since session start).
|
|
165
|
+
"""
|
|
166
|
+
start = time.time()
|
|
167
|
+
last_log = ""
|
|
168
|
+
|
|
169
|
+
while True:
|
|
170
|
+
elapsed = time.time() - start
|
|
171
|
+
if elapsed >= timeout:
|
|
172
|
+
break
|
|
173
|
+
|
|
174
|
+
self._client._tokens.get()
|
|
175
|
+
|
|
176
|
+
try:
|
|
177
|
+
text = self.logs(tail=tail)
|
|
178
|
+
except LogsError:
|
|
179
|
+
break
|
|
180
|
+
|
|
181
|
+
if text and text != last_log:
|
|
182
|
+
new_content = text[len(last_log):]
|
|
183
|
+
last_log = text
|
|
184
|
+
if on_output and new_content.strip():
|
|
185
|
+
on_output(new_content)
|
|
186
|
+
|
|
187
|
+
if stop_pattern and last_log and stop_pattern in last_log:
|
|
188
|
+
break
|
|
189
|
+
|
|
190
|
+
self.refresh()
|
|
191
|
+
if self.status in ("terminated", "completed", "done", "failed"):
|
|
192
|
+
# One final log fetch to capture anything written right before exit
|
|
193
|
+
try:
|
|
194
|
+
final = self.logs(tail=tail)
|
|
195
|
+
if final and final != last_log:
|
|
196
|
+
new_content = final[len(last_log):]
|
|
197
|
+
last_log = final
|
|
198
|
+
if on_output and new_content.strip():
|
|
199
|
+
on_output(new_content)
|
|
200
|
+
except LogsError:
|
|
201
|
+
pass
|
|
202
|
+
break
|
|
203
|
+
|
|
204
|
+
time.sleep(poll_interval)
|
|
205
|
+
|
|
206
|
+
return last_log
|
|
207
|
+
|
|
208
|
+
# ── SSH ────────────────────────────────────────────────────────────────────
|
|
209
|
+
|
|
210
|
+
def ssh_command(self, key_path: str = "~/.ssh/mimiry_api") -> str:
|
|
211
|
+
"""
|
|
212
|
+
Return the SSH command string to connect to this session.
|
|
213
|
+
|
|
214
|
+
Note: 'running' status means SSH is up, but the GPU driver takes
|
|
215
|
+
another ~60-90s to load. Connecting too early will show
|
|
216
|
+
'NVIDIA-SMI has failed' — this is normal, just wait and retry.
|
|
217
|
+
|
|
218
|
+
Raises:
|
|
219
|
+
SessionError: if the session has no SSH host assigned.
|
|
220
|
+
"""
|
|
221
|
+
if not self.host:
|
|
222
|
+
self.refresh()
|
|
223
|
+
if not self.host:
|
|
224
|
+
raise SessionError(
|
|
225
|
+
f"Session {self.id} has no SSH host "
|
|
226
|
+
f"(status: {self.status}). "
|
|
227
|
+
"Did you set ssh_enabled=True when creating the session?",
|
|
228
|
+
session_id=self.id,
|
|
229
|
+
)
|
|
230
|
+
return (
|
|
231
|
+
f"ssh -i {key_path} "
|
|
232
|
+
f"-o StrictHostKeyChecking=no "
|
|
233
|
+
f"-p {self.port} "
|
|
234
|
+
f"{self.username}@{self.host}"
|
|
235
|
+
)
|
|
236
|
+
|
|
237
|
+
# ── Lifecycle ──────────────────────────────────────────────────────────────
|
|
238
|
+
|
|
239
|
+
def terminate(self) -> None:
|
|
240
|
+
"""
|
|
241
|
+
Terminate this session (DELETE /sessions/{id}).
|
|
242
|
+
|
|
243
|
+
Always call this when done — auto_terminate may not fire reliably
|
|
244
|
+
when background processes keep the host alive. Silently ignores
|
|
245
|
+
errors (e.g. session already terminated).
|
|
246
|
+
"""
|
|
247
|
+
try:
|
|
248
|
+
self._client._delete(f"/sessions/{self.id}")
|
|
249
|
+
except Exception:
|
|
250
|
+
pass
|
|
251
|
+
self.status = "terminated"
|
|
252
|
+
|
|
253
|
+
def __repr__(self) -> str:
|
|
254
|
+
return (
|
|
255
|
+
f"Session(id={self.id!r}, status={self.status!r}, host={self.host!r})"
|
|
256
|
+
)
|
|
257
|
+
|
|
258
|
+
def __enter__(self) -> "Session":
|
|
259
|
+
return self
|
|
260
|
+
|
|
261
|
+
def __exit__(self, *_) -> None:
|
|
262
|
+
self.terminate()
|
mimiry/client.py
ADDED
|
@@ -0,0 +1,591 @@
|
|
|
1
|
+
"""
|
|
2
|
+
MimiryClient — the main entry point for the Mimiry Python SDK.
|
|
3
|
+
|
|
4
|
+
Confirmed-working patterns (validated on Mimiry soft-launch, Feb 2026):
|
|
5
|
+
|
|
6
|
+
Payload:
|
|
7
|
+
- image.uri : use nvcr.io/nvidia/cuda:12.1.0-base-ubuntu22.04 for quick jobs
|
|
8
|
+
use pytorch/pytorch:2.3.1-cuda12.1-cudnn8-runtime for ML jobs
|
|
9
|
+
- command : runs inside user-container (DO NOT use startup_script — ignored)
|
|
10
|
+
- ssh_enabled : required for /logs endpoint to work
|
|
11
|
+
- auto_terminate: True recommended to avoid zombie sessions + billing
|
|
12
|
+
|
|
13
|
+
Timing:
|
|
14
|
+
- status='running' in 23-55s, but Docker/GPU take another 60-90s
|
|
15
|
+
- /logs returns 503 for ~60-90s after 'running' — handled automatically
|
|
16
|
+
- JWT tokens expire in ~5-8 min — refreshed automatically every 4 min
|
|
17
|
+
"""
|
|
18
|
+
|
|
19
|
+
import base64
|
|
20
|
+
import json
|
|
21
|
+
import time
|
|
22
|
+
import urllib.error
|
|
23
|
+
import urllib.request
|
|
24
|
+
from dataclasses import dataclass
|
|
25
|
+
from pathlib import Path
|
|
26
|
+
|
|
27
|
+
from ._auth import TokenManager
|
|
28
|
+
from ._exceptions import AuthError, MimiryError, QuotaError, SessionError, _HttpError
|
|
29
|
+
from ._session import Session
|
|
30
|
+
|
|
31
|
+
# ── Default images ─────────────────────────────────────────────────────────────
|
|
32
|
+
# Confirmed working, fastest startup. Contains CUDA 12.1 + drivers only.
|
|
33
|
+
CUDA_BASE = "nvcr.io/nvidia/cuda:12.1.0-base-ubuntu22.04"
|
|
34
|
+
|
|
35
|
+
# Best balance of size vs. capability for ML jobs.
|
|
36
|
+
# ~3.8 GB compressed, ~2-4 min pull. Contains torch + CUDA 12.1 + cuDNN 8.
|
|
37
|
+
PYTORCH_IMAGE = "pytorch/pytorch:2.3.1-cuda12.1-cudnn8-runtime"
|
|
38
|
+
|
|
39
|
+
# Full NGC PyTorch stack (~14 GB, 10-15 min pull). Only use if you need
|
|
40
|
+
# Jupyter, Apex, Triton, or other NGC extras. Increases job startup time
|
|
41
|
+
# from ~5 min to ~20 min.
|
|
42
|
+
PYTORCH_NGC = "nvcr.io/nvidia/pytorch:24.01-py3"
|
|
43
|
+
|
|
44
|
+
# Default API base
|
|
45
|
+
_API_BASE = "https://softlaunch.mimiry.com/api/compute/v1"
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
@dataclass
|
|
49
|
+
class RunResult:
|
|
50
|
+
"""
|
|
51
|
+
Result returned by MimiryClient.run() and MimiryClient.run_script().
|
|
52
|
+
|
|
53
|
+
Attributes:
|
|
54
|
+
logs: Full container output (stdout + stderr).
|
|
55
|
+
session_id: The session UUID.
|
|
56
|
+
success: True if the session completed without error status.
|
|
57
|
+
elapsed: Wall-clock seconds from session creation to log retrieval.
|
|
58
|
+
image: The container image that was used.
|
|
59
|
+
"""
|
|
60
|
+
|
|
61
|
+
logs: str
|
|
62
|
+
session_id: str
|
|
63
|
+
success: bool
|
|
64
|
+
elapsed: float
|
|
65
|
+
image: str
|
|
66
|
+
|
|
67
|
+
def print_logs(self) -> None:
|
|
68
|
+
"""Print logs to stdout."""
|
|
69
|
+
print(self.logs)
|
|
70
|
+
|
|
71
|
+
def __repr__(self) -> str:
|
|
72
|
+
lines = len(self.logs.splitlines()) if self.logs else 0
|
|
73
|
+
return (
|
|
74
|
+
f"RunResult(session_id={self.session_id!r}, "
|
|
75
|
+
f"success={self.success}, elapsed={self.elapsed:.1f}s, "
|
|
76
|
+
f"log_lines={lines})"
|
|
77
|
+
)
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
class MimiryClient:
|
|
81
|
+
"""
|
|
82
|
+
Client for the Mimiry compute platform.
|
|
83
|
+
|
|
84
|
+
Authenticates automatically via the `mimiry` CLI. Run `mimiry auth login`
|
|
85
|
+
once before using this client.
|
|
86
|
+
|
|
87
|
+
Args:
|
|
88
|
+
api_base: Override the API base URL (default: softlaunch endpoint).
|
|
89
|
+
ssh_key_path: Path to the SSH private key used for session access.
|
|
90
|
+
Default: ~/.ssh/mimiry_api
|
|
91
|
+
|
|
92
|
+
Example::
|
|
93
|
+
|
|
94
|
+
from mimiry import MimiryClient
|
|
95
|
+
|
|
96
|
+
client = MimiryClient()
|
|
97
|
+
result = client.run("nvidia-smi")
|
|
98
|
+
result.print_logs()
|
|
99
|
+
"""
|
|
100
|
+
|
|
101
|
+
def __init__(
|
|
102
|
+
self,
|
|
103
|
+
api_base: str = _API_BASE,
|
|
104
|
+
ssh_key_path: str = "~/.ssh/mimiry_api",
|
|
105
|
+
):
|
|
106
|
+
self._api_base = api_base.rstrip("/")
|
|
107
|
+
self._ssh_key_path = ssh_key_path
|
|
108
|
+
self._tokens = TokenManager()
|
|
109
|
+
|
|
110
|
+
# ── High-level API ─────────────────────────────────────────────────────────
|
|
111
|
+
|
|
112
|
+
def run(
|
|
113
|
+
self,
|
|
114
|
+
command: str,
|
|
115
|
+
*,
|
|
116
|
+
image: str = CUDA_BASE,
|
|
117
|
+
gpu: str = "T4",
|
|
118
|
+
gpu_count: int = 1,
|
|
119
|
+
name: str | None = None,
|
|
120
|
+
ssh_key_path: str | None = None,
|
|
121
|
+
provision_timeout: int = 300,
|
|
122
|
+
job_timeout: int = 600,
|
|
123
|
+
logs_tail: int = 500,
|
|
124
|
+
logs_retries: int = 12,
|
|
125
|
+
poll_interval: int = 30,
|
|
126
|
+
stop_pattern: str | None = None,
|
|
127
|
+
verbose: bool = False,
|
|
128
|
+
) -> RunResult:
|
|
129
|
+
"""
|
|
130
|
+
Submit a command, wait for it to complete, and return the logs.
|
|
131
|
+
|
|
132
|
+
This is the main method for running one-shot GPU workloads. It handles
|
|
133
|
+
the full lifecycle: auth, session creation, provisioning poll, log
|
|
134
|
+
retrieval with 503 retry, token refresh, and cleanup.
|
|
135
|
+
|
|
136
|
+
Args:
|
|
137
|
+
command: Shell command to run inside the container.
|
|
138
|
+
For multi-step commands use bash -c syntax:
|
|
139
|
+
'bash -c "pip install numpy && python3 job.py"'
|
|
140
|
+
image: Container image URI. Default is the small CUDA
|
|
141
|
+
base (fast start, no PyTorch). For ML jobs use
|
|
142
|
+
PYTORCH_IMAGE or pass your own.
|
|
143
|
+
gpu: GPU type. 'T4' is the only type on soft-launch.
|
|
144
|
+
gpu_count: Number of GPUs (default 1).
|
|
145
|
+
name: Session name shown in the dashboard. Auto-generated
|
|
146
|
+
from timestamp if not provided.
|
|
147
|
+
ssh_key_path: Path to SSH public key to embed in the session.
|
|
148
|
+
Defaults to client-level ssh_key_path + '.pub'.
|
|
149
|
+
provision_timeout: Seconds to wait for status='running' (default 300).
|
|
150
|
+
job_timeout: Seconds to wait for the job to complete (default 600).
|
|
151
|
+
For large image pulls (NGC PyTorch ~14 GB) use 2400+.
|
|
152
|
+
logs_tail: Lines of log output to retrieve per poll (default 500).
|
|
153
|
+
logs_retries: 503 retry attempts while VM setup is in progress.
|
|
154
|
+
Each retry waits ~15-60s, so 12 ≈ 3-12 min total.
|
|
155
|
+
poll_interval: Seconds between log polls for long-running jobs.
|
|
156
|
+
stop_pattern: Stop polling when this string appears in the logs.
|
|
157
|
+
Useful for jobs that don't auto-terminate.
|
|
158
|
+
verbose: Print progress and logs to stdout as they arrive.
|
|
159
|
+
|
|
160
|
+
Returns:
|
|
161
|
+
RunResult with .logs, .session_id, .success, .elapsed, .image
|
|
162
|
+
|
|
163
|
+
Raises:
|
|
164
|
+
AuthError: Not logged in.
|
|
165
|
+
QuotaError: Account quota exceeded.
|
|
166
|
+
SessionError: Session failed during provisioning.
|
|
167
|
+
|
|
168
|
+
Example — GPU check::
|
|
169
|
+
|
|
170
|
+
result = client.run("nvidia-smi")
|
|
171
|
+
result.print_logs()
|
|
172
|
+
|
|
173
|
+
Example — Install + run::
|
|
174
|
+
|
|
175
|
+
result = client.run(
|
|
176
|
+
'bash -c "pip install -q numpy && python3 -c \\"import numpy; print(numpy.__version__)\\""',
|
|
177
|
+
image=CUDA_BASE,
|
|
178
|
+
)
|
|
179
|
+
"""
|
|
180
|
+
t0 = time.time()
|
|
181
|
+
session_name = name or f"sdk-run-{int(t0)}"
|
|
182
|
+
|
|
183
|
+
# Resolve SSH public key
|
|
184
|
+
key_pub = self._resolve_pub_key(ssh_key_path)
|
|
185
|
+
|
|
186
|
+
if verbose:
|
|
187
|
+
print(f"[mimiry] Creating session '{session_name}' on {gpu}...")
|
|
188
|
+
print(f"[mimiry] Image: {image}")
|
|
189
|
+
|
|
190
|
+
session = self.create_session(
|
|
191
|
+
image=image,
|
|
192
|
+
gpu=gpu,
|
|
193
|
+
gpu_count=gpu_count,
|
|
194
|
+
name=session_name,
|
|
195
|
+
command=command,
|
|
196
|
+
ssh_key_pub=key_pub,
|
|
197
|
+
auto_terminate=True,
|
|
198
|
+
)
|
|
199
|
+
|
|
200
|
+
if verbose:
|
|
201
|
+
print(f"[mimiry] Session: {session.id}")
|
|
202
|
+
print("[mimiry] Waiting for VM to reach 'running'...")
|
|
203
|
+
|
|
204
|
+
try:
|
|
205
|
+
session.wait_until_running(timeout=provision_timeout)
|
|
206
|
+
|
|
207
|
+
if verbose:
|
|
208
|
+
elapsed = time.time() - t0
|
|
209
|
+
print(f"[mimiry] Running ({elapsed:.0f}s). Fetching logs...")
|
|
210
|
+
|
|
211
|
+
# Initial log fetch handles the 503 'vm_setup_in_progress' phase
|
|
212
|
+
logs = session.logs(tail=logs_tail, max_retries=logs_retries)
|
|
213
|
+
|
|
214
|
+
if verbose and logs:
|
|
215
|
+
print(logs, flush=True)
|
|
216
|
+
|
|
217
|
+
# For jobs that haven't terminated yet, keep polling
|
|
218
|
+
session.refresh()
|
|
219
|
+
if session.status not in ("terminated", "completed", "done", "failed"):
|
|
220
|
+
def _on_output(text):
|
|
221
|
+
if verbose:
|
|
222
|
+
print(text, end="", flush=True)
|
|
223
|
+
|
|
224
|
+
logs = session.poll_logs(
|
|
225
|
+
tail=logs_tail,
|
|
226
|
+
poll_interval=poll_interval,
|
|
227
|
+
timeout=job_timeout,
|
|
228
|
+
stop_pattern=stop_pattern,
|
|
229
|
+
on_output=_on_output,
|
|
230
|
+
)
|
|
231
|
+
|
|
232
|
+
success = session.status not in ("failed",)
|
|
233
|
+
elapsed = time.time() - t0
|
|
234
|
+
|
|
235
|
+
if verbose:
|
|
236
|
+
print(f"\n[mimiry] Done in {elapsed:.1f}s — session {session.id}")
|
|
237
|
+
|
|
238
|
+
return RunResult(
|
|
239
|
+
logs=logs,
|
|
240
|
+
session_id=session.id,
|
|
241
|
+
success=success,
|
|
242
|
+
elapsed=elapsed,
|
|
243
|
+
image=image,
|
|
244
|
+
)
|
|
245
|
+
|
|
246
|
+
finally:
|
|
247
|
+
session.terminate()
|
|
248
|
+
|
|
249
|
+
def run_script(
|
|
250
|
+
self,
|
|
251
|
+
script: str,
|
|
252
|
+
*,
|
|
253
|
+
image: str = PYTORCH_IMAGE,
|
|
254
|
+
gpu: str = "T4",
|
|
255
|
+
name: str | None = None,
|
|
256
|
+
job_timeout: int = 2400,
|
|
257
|
+
provision_timeout: int = 300,
|
|
258
|
+
stop_pattern: str | None = None,
|
|
259
|
+
verbose: bool = False,
|
|
260
|
+
**kwargs,
|
|
261
|
+
) -> RunResult:
|
|
262
|
+
"""
|
|
263
|
+
Run a Python script on a GPU session and return its output.
|
|
264
|
+
|
|
265
|
+
The script is base64-encoded and decoded inside the container, avoiding
|
|
266
|
+
all shell-quoting issues. This is the recommended pattern for any
|
|
267
|
+
non-trivial Python workload.
|
|
268
|
+
|
|
269
|
+
Confirmed working pattern (from run_ml_benchmark.sh, Feb 2026):
|
|
270
|
+
command = "bash -c 'echo <base64> | base64 -d | python3 -u'"
|
|
271
|
+
|
|
272
|
+
Args:
|
|
273
|
+
script: Path to a .py file (str or Path), or a Python source string.
|
|
274
|
+
image: Container image. Default is pytorch/pytorch:2.3.1 (~3.8 GB,
|
|
275
|
+
2-4 min pull). Use PYTORCH_NGC for the full NGC stack.
|
|
276
|
+
gpu: GPU type ('T4' on soft-launch).
|
|
277
|
+
name: Session name (auto-generated if not provided).
|
|
278
|
+
job_timeout: Max seconds to wait for job completion.
|
|
279
|
+
NGC PyTorch (~14 GB) needs 2400+ for image pull.
|
|
280
|
+
provision_timeout: Max seconds to wait for VM to reach 'running'.
|
|
281
|
+
stop_pattern: Stop polling when this string appears in logs.
|
|
282
|
+
verbose: Print progress and logs as they arrive.
|
|
283
|
+
**kwargs: Passed through to run() (e.g. logs_tail, poll_interval).
|
|
284
|
+
|
|
285
|
+
Returns:
|
|
286
|
+
RunResult with .logs, .session_id, .success, .elapsed, .image
|
|
287
|
+
|
|
288
|
+
Example — inline script::
|
|
289
|
+
|
|
290
|
+
result = client.run_script(
|
|
291
|
+
'''
|
|
292
|
+
import torch
|
|
293
|
+
print(f"CUDA: {torch.cuda.is_available()}")
|
|
294
|
+
print(f"GPU: {torch.cuda.get_device_name(0)}")
|
|
295
|
+
'''
|
|
296
|
+
)
|
|
297
|
+
result.print_logs()
|
|
298
|
+
|
|
299
|
+
Example — script file::
|
|
300
|
+
|
|
301
|
+
result = client.run_script("ml_benchmark.py", verbose=True)
|
|
302
|
+
"""
|
|
303
|
+
# Resolve script content
|
|
304
|
+
path = Path(script) if "\n" not in script and len(script) < 500 else None
|
|
305
|
+
if path and path.exists():
|
|
306
|
+
source = path.read_text()
|
|
307
|
+
script_label = path.name
|
|
308
|
+
else:
|
|
309
|
+
source = script.strip()
|
|
310
|
+
script_label = "inline_script.py"
|
|
311
|
+
|
|
312
|
+
b64 = base64.b64encode(source.encode()).decode()
|
|
313
|
+
command = f"bash -c 'echo {b64} | base64 -d | python3 -u'"
|
|
314
|
+
|
|
315
|
+
session_name = name or f"sdk-script-{int(time.time())}"
|
|
316
|
+
|
|
317
|
+
if verbose:
|
|
318
|
+
print(f"[mimiry] Script: {script_label} ({len(source)} bytes)")
|
|
319
|
+
print(f"[mimiry] Image: {image}")
|
|
320
|
+
if image == PYTORCH_IMAGE:
|
|
321
|
+
print("[mimiry] Note: ~3.8 GB image pull, expect 2-4 min before logs appear")
|
|
322
|
+
elif image == PYTORCH_NGC:
|
|
323
|
+
print("[mimiry] Note: ~14 GB NGC image pull, expect 10-15 min before logs appear")
|
|
324
|
+
|
|
325
|
+
return self.run(
|
|
326
|
+
command,
|
|
327
|
+
image=image,
|
|
328
|
+
gpu=gpu,
|
|
329
|
+
name=session_name,
|
|
330
|
+
job_timeout=job_timeout,
|
|
331
|
+
provision_timeout=provision_timeout,
|
|
332
|
+
stop_pattern=stop_pattern,
|
|
333
|
+
verbose=verbose,
|
|
334
|
+
**kwargs,
|
|
335
|
+
)
|
|
336
|
+
|
|
337
|
+
# ── Session management ─────────────────────────────────────────────────────
|
|
338
|
+
|
|
339
|
+
def create_session(
|
|
340
|
+
self,
|
|
341
|
+
*,
|
|
342
|
+
image: str = CUDA_BASE,
|
|
343
|
+
gpu: str = "T4",
|
|
344
|
+
gpu_count: int = 1,
|
|
345
|
+
name: str | None = None,
|
|
346
|
+
command: str | None = None,
|
|
347
|
+
ssh_key_pub: str | None = None,
|
|
348
|
+
ssh_key_path: str | None = None,
|
|
349
|
+
auto_terminate: bool = True,
|
|
350
|
+
) -> Session:
|
|
351
|
+
"""
|
|
352
|
+
Create a session and return a Session object for manual control.
|
|
353
|
+
|
|
354
|
+
Use this when you want to SSH in interactively, run multiple commands,
|
|
355
|
+
or manage the session lifecycle yourself.
|
|
356
|
+
|
|
357
|
+
Args:
|
|
358
|
+
image: Container image URI.
|
|
359
|
+
gpu: GPU type ('T4' on soft-launch).
|
|
360
|
+
gpu_count: Number of GPUs (default 1).
|
|
361
|
+
name: Session name.
|
|
362
|
+
command: Optional command to run on start. If None, the
|
|
363
|
+
container starts but runs nothing (useful for SSH).
|
|
364
|
+
ssh_key_pub: SSH public key contents (overrides ssh_key_path).
|
|
365
|
+
ssh_key_path: Path to SSH private key; .pub will be read automatically.
|
|
366
|
+
Defaults to client-level ssh_key_path.
|
|
367
|
+
auto_terminate: Terminate when command exits (default True).
|
|
368
|
+
Set to False for interactive/SSH sessions.
|
|
369
|
+
|
|
370
|
+
Returns:
|
|
371
|
+
Session object (NOT yet running — call wait_until_running()).
|
|
372
|
+
|
|
373
|
+
Example — SSH session::
|
|
374
|
+
|
|
375
|
+
with client.create_session(
|
|
376
|
+
auto_terminate=False,
|
|
377
|
+
command=None,
|
|
378
|
+
) as session:
|
|
379
|
+
session.wait_until_running()
|
|
380
|
+
print(session.ssh_command())
|
|
381
|
+
input("Press Enter when done...")
|
|
382
|
+
# session.terminate() called automatically by context manager
|
|
383
|
+
"""
|
|
384
|
+
session_name = name or f"sdk-session-{int(time.time())}"
|
|
385
|
+
|
|
386
|
+
# Resolve public key
|
|
387
|
+
pub_key = ssh_key_pub or self._resolve_pub_key(ssh_key_path)
|
|
388
|
+
|
|
389
|
+
payload: dict = {
|
|
390
|
+
"name": session_name,
|
|
391
|
+
"image": {"uri": image},
|
|
392
|
+
"gpu": {"types": [gpu], "count": gpu_count},
|
|
393
|
+
"ssh_enabled": True,
|
|
394
|
+
"auto_terminate": auto_terminate,
|
|
395
|
+
}
|
|
396
|
+
|
|
397
|
+
if pub_key:
|
|
398
|
+
payload["ssh_public_key"] = pub_key
|
|
399
|
+
|
|
400
|
+
if command:
|
|
401
|
+
payload["command"] = command
|
|
402
|
+
|
|
403
|
+
resp = self._post("/sessions", payload)
|
|
404
|
+
session_id = resp.get("id") or resp.get("session_id")
|
|
405
|
+
|
|
406
|
+
if not session_id:
|
|
407
|
+
raise SessionError(f"No session ID in API response: {resp}")
|
|
408
|
+
|
|
409
|
+
return Session(session_id, self)
|
|
410
|
+
|
|
411
|
+
def sessions(self, include_terminal: bool = False) -> list[Session]:
|
|
412
|
+
"""
|
|
413
|
+
List sessions on this account.
|
|
414
|
+
|
|
415
|
+
Args:
|
|
416
|
+
include_terminal: Include terminated/failed/completed sessions.
|
|
417
|
+
Default False (active sessions only).
|
|
418
|
+
|
|
419
|
+
Returns:
|
|
420
|
+
List of Session objects (status is populated, but host may be None
|
|
421
|
+
for sessions that have terminated).
|
|
422
|
+
"""
|
|
423
|
+
data = self._get("/sessions")
|
|
424
|
+
raw = data.get("sessions") or []
|
|
425
|
+
|
|
426
|
+
terminal = {"terminated", "failed", "completed", "done", "cancelled"}
|
|
427
|
+
result = []
|
|
428
|
+
|
|
429
|
+
for s in raw:
|
|
430
|
+
status = s.get("status", "")
|
|
431
|
+
if not include_terminal and status in terminal:
|
|
432
|
+
continue
|
|
433
|
+
|
|
434
|
+
session = Session(s["id"], self)
|
|
435
|
+
session.status = status
|
|
436
|
+
ssh = s.get("ssh") or {}
|
|
437
|
+
session.host = ssh.get("host")
|
|
438
|
+
session.port = ssh.get("port", 22)
|
|
439
|
+
session.username = ssh.get("username", "ubuntu")
|
|
440
|
+
result.append(session)
|
|
441
|
+
|
|
442
|
+
return result
|
|
443
|
+
|
|
444
|
+
def kill_all(self, confirm: bool = True) -> int:
|
|
445
|
+
"""
|
|
446
|
+
Terminate all active sessions.
|
|
447
|
+
|
|
448
|
+
Args:
|
|
449
|
+
confirm: If True (default), print a list and ask for confirmation.
|
|
450
|
+
Set to False for non-interactive scripts.
|
|
451
|
+
|
|
452
|
+
Returns:
|
|
453
|
+
Number of sessions terminated.
|
|
454
|
+
"""
|
|
455
|
+
active = self.sessions()
|
|
456
|
+
|
|
457
|
+
if not active:
|
|
458
|
+
print("No active sessions.")
|
|
459
|
+
return 0
|
|
460
|
+
|
|
461
|
+
print(f"Active sessions ({len(active)}):")
|
|
462
|
+
for s in active:
|
|
463
|
+
print(f" {s.id} [{s.status}] host={s.host or 'none'}")
|
|
464
|
+
|
|
465
|
+
if confirm:
|
|
466
|
+
answer = input(f"\nTerminate all {len(active)} session(s)? [y/N] ").strip()
|
|
467
|
+
if answer.lower() not in ("y", "yes"):
|
|
468
|
+
print("Aborted.")
|
|
469
|
+
return 0
|
|
470
|
+
|
|
471
|
+
count = 0
|
|
472
|
+
for s in active:
|
|
473
|
+
s.terminate()
|
|
474
|
+
count += 1
|
|
475
|
+
print(f" Terminated: {s.id}")
|
|
476
|
+
|
|
477
|
+
return count
|
|
478
|
+
|
|
479
|
+
# ── Account info ───────────────────────────────────────────────────────────
|
|
480
|
+
|
|
481
|
+
def balance(self) -> dict:
|
|
482
|
+
"""
|
|
483
|
+
Return account balance information.
|
|
484
|
+
|
|
485
|
+
Returns:
|
|
486
|
+
Dict with at least 'balance' (float, EUR).
|
|
487
|
+
|
|
488
|
+
Example::
|
|
489
|
+
|
|
490
|
+
info = client.balance()
|
|
491
|
+
print(f"Balance: €{info['balance']:.2f}")
|
|
492
|
+
"""
|
|
493
|
+
return self._get("/balance")
|
|
494
|
+
|
|
495
|
+
def quota(self) -> dict:
|
|
496
|
+
"""
|
|
497
|
+
Return current quota usage.
|
|
498
|
+
|
|
499
|
+
Returns:
|
|
500
|
+
Dict with 'current_jobs' and 'max_concurrent_jobs'.
|
|
501
|
+
|
|
502
|
+
Note: The 'active_sessions' field is always 0 due to a known platform
|
|
503
|
+
bug — use client.sessions() for accurate counts.
|
|
504
|
+
|
|
505
|
+
Example::
|
|
506
|
+
|
|
507
|
+
q = client.quota()
|
|
508
|
+
print(f"Jobs: {q['current_jobs']}/{q['max_concurrent_jobs']}")
|
|
509
|
+
"""
|
|
510
|
+
return self._get("/quota")
|
|
511
|
+
|
|
512
|
+
# ── HTTP helpers ───────────────────────────────────────────────────────────
|
|
513
|
+
|
|
514
|
+
def _get(self, path: str) -> dict:
|
|
515
|
+
return self._request("GET", path)
|
|
516
|
+
|
|
517
|
+
def _post(self, path: str, body: dict) -> dict:
|
|
518
|
+
return self._request("POST", path, json_body=body)
|
|
519
|
+
|
|
520
|
+
def _delete(self, path: str) -> dict:
|
|
521
|
+
return self._request("DELETE", path)
|
|
522
|
+
|
|
523
|
+
def _request(self, method: str, path: str, json_body: dict | None = None) -> dict:
|
|
524
|
+
"""Execute an authenticated API request, raising _HttpError on non-2xx."""
|
|
525
|
+
url = self._api_base + path
|
|
526
|
+
token = self._tokens.get()
|
|
527
|
+
|
|
528
|
+
headers = {
|
|
529
|
+
"Authorization": f"Bearer {token}",
|
|
530
|
+
"Content-Type": "application/json",
|
|
531
|
+
}
|
|
532
|
+
|
|
533
|
+
data: bytes | None = None
|
|
534
|
+
if json_body is not None:
|
|
535
|
+
data = json.dumps(json_body).encode()
|
|
536
|
+
|
|
537
|
+
req = urllib.request.Request(url, data=data, headers=headers, method=method)
|
|
538
|
+
|
|
539
|
+
try:
|
|
540
|
+
with urllib.request.urlopen(req, timeout=30) as resp:
|
|
541
|
+
body = resp.read()
|
|
542
|
+
return json.loads(body) if body else {}
|
|
543
|
+
|
|
544
|
+
except urllib.error.HTTPError as exc:
|
|
545
|
+
body_bytes = exc.read()
|
|
546
|
+
try:
|
|
547
|
+
body = json.loads(body_bytes)
|
|
548
|
+
except Exception:
|
|
549
|
+
body = {"raw": body_bytes.decode(errors="replace")}
|
|
550
|
+
|
|
551
|
+
if exc.code == 401:
|
|
552
|
+
# Token may have just expired — force refresh and retry once
|
|
553
|
+
self._tokens.get(force=True)
|
|
554
|
+
token = self._tokens._token
|
|
555
|
+
req.add_header("Authorization", f"Bearer {token}")
|
|
556
|
+
try:
|
|
557
|
+
with urllib.request.urlopen(req, timeout=30) as resp2:
|
|
558
|
+
body2 = resp2.read()
|
|
559
|
+
return json.loads(body2) if body2 else {}
|
|
560
|
+
except urllib.error.HTTPError as exc2:
|
|
561
|
+
body2 = exc2.read()
|
|
562
|
+
raise _HttpError(exc2.code, json.loads(body2) if body2 else {})
|
|
563
|
+
|
|
564
|
+
if exc.code == 402:
|
|
565
|
+
raise QuotaError(
|
|
566
|
+
"Insufficient credits (HTTP 402). Top up your account balance."
|
|
567
|
+
)
|
|
568
|
+
if exc.code == 429:
|
|
569
|
+
raise QuotaError(
|
|
570
|
+
f"Rate limited or quota exceeded (HTTP 429): {body}"
|
|
571
|
+
)
|
|
572
|
+
|
|
573
|
+
raise _HttpError(exc.code, body)
|
|
574
|
+
|
|
575
|
+
except urllib.error.URLError as exc:
|
|
576
|
+
raise MimiryError(f"Network error reaching {url}: {exc.reason}") from exc
|
|
577
|
+
|
|
578
|
+
# ── Internal helpers ───────────────────────────────────────────────────────
|
|
579
|
+
|
|
580
|
+
def _resolve_pub_key(self, ssh_key_path: str | None = None) -> str:
|
|
581
|
+
"""Read the SSH public key file. Returns empty string if not found."""
|
|
582
|
+
path_str = ssh_key_path or self._ssh_key_path
|
|
583
|
+
path = Path(path_str).expanduser()
|
|
584
|
+
pub = path.with_suffix(".pub") if path.suffix != ".pub" else path
|
|
585
|
+
if pub.exists():
|
|
586
|
+
return pub.read_text().strip()
|
|
587
|
+
return ""
|
|
588
|
+
|
|
589
|
+
|
|
590
|
+
# Re-export for convenience
|
|
591
|
+
from ._exceptions import AuthError, MimiryError, QuotaError, SessionError # noqa: F401
|
|
@@ -0,0 +1,118 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: mimiry
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Python SDK for the Mimiry GPU compute platform
|
|
5
|
+
Author-email: Mimiry <team@mimiry.com>
|
|
6
|
+
License: MIT
|
|
7
|
+
Keywords: gpu,cloud,compute,machine-learning,cuda,mimiry
|
|
8
|
+
Classifier: Development Status :: 3 - Alpha
|
|
9
|
+
Classifier: Intended Audience :: Developers
|
|
10
|
+
Classifier: Intended Audience :: Science/Research
|
|
11
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
12
|
+
Classifier: Programming Language :: Python :: 3
|
|
13
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
14
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
15
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
16
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
17
|
+
Classifier: Topic :: System :: Distributed Computing
|
|
18
|
+
Requires-Python: >=3.10
|
|
19
|
+
Description-Content-Type: text/markdown
|
|
20
|
+
License-File: LICENSE
|
|
21
|
+
Provides-Extra: dev
|
|
22
|
+
Requires-Dist: pytest; extra == "dev"
|
|
23
|
+
Requires-Dist: ruff; extra == "dev"
|
|
24
|
+
Dynamic: license-file
|
|
25
|
+
|
|
26
|
+
# Mimiry Python SDK
|
|
27
|
+
|
|
28
|
+
Run GPU jobs on [Mimiry](https://mimiry.com) from Python — no shell scripts, no manual API calls.
|
|
29
|
+
|
|
30
|
+
```bash
|
|
31
|
+
pip install mimiry
|
|
32
|
+
```
|
|
33
|
+
|
|
34
|
+
## Requirements
|
|
35
|
+
|
|
36
|
+
- Python 3.10 or newer
|
|
37
|
+
- [Mimiry CLI](https://mimiry.com) installed and authenticated (`mimiry auth login`)
|
|
38
|
+
|
|
39
|
+
## Quick start
|
|
40
|
+
|
|
41
|
+
**Run a command on a GPU and get the output:**
|
|
42
|
+
|
|
43
|
+
```python
|
|
44
|
+
from mimiry import MimiryClient
|
|
45
|
+
|
|
46
|
+
client = MimiryClient()
|
|
47
|
+
result = client.run("nvidia-smi", verbose=True)
|
|
48
|
+
result.print_logs()
|
|
49
|
+
```
|
|
50
|
+
|
|
51
|
+
**Run a Python script on a GPU:**
|
|
52
|
+
|
|
53
|
+
```python
|
|
54
|
+
from mimiry import MimiryClient, PYTORCH_IMAGE
|
|
55
|
+
|
|
56
|
+
client = MimiryClient()
|
|
57
|
+
result = client.run_script("train.py", image=PYTORCH_IMAGE, verbose=True)
|
|
58
|
+
result.print_logs()
|
|
59
|
+
```
|
|
60
|
+
|
|
61
|
+
**Or pass an inline script directly:**
|
|
62
|
+
|
|
63
|
+
```python
|
|
64
|
+
result = client.run_script(
|
|
65
|
+
"""
|
|
66
|
+
import torch
|
|
67
|
+
print(torch.cuda.get_device_name(0))
|
|
68
|
+
""",
|
|
69
|
+
image=PYTORCH_IMAGE,
|
|
70
|
+
)
|
|
71
|
+
```
|
|
72
|
+
|
|
73
|
+
**Check your balance and quota:**
|
|
74
|
+
|
|
75
|
+
```python
|
|
76
|
+
print(client.balance())
|
|
77
|
+
print(client.quota())
|
|
78
|
+
```
|
|
79
|
+
|
|
80
|
+
**Clean up all running sessions:**
|
|
81
|
+
|
|
82
|
+
```python
|
|
83
|
+
client.kill_all()
|
|
84
|
+
```
|
|
85
|
+
|
|
86
|
+
## How it works
|
|
87
|
+
|
|
88
|
+
Each `client.run()` call:
|
|
89
|
+
|
|
90
|
+
1. Creates a session via the Mimiry Sessions API
|
|
91
|
+
2. Polls until the VM reaches `running` status (~30–55 s)
|
|
92
|
+
3. Waits for Docker and the GPU driver to finish loading (~60–90 s)
|
|
93
|
+
4. Fetches the command output from the platform log endpoint
|
|
94
|
+
5. Terminates the session automatically
|
|
95
|
+
|
|
96
|
+
Sessions are always terminated after the job finishes. If a script crashes before cleanup, run `client.kill_all()`.
|
|
97
|
+
|
|
98
|
+
## Images
|
|
99
|
+
|
|
100
|
+
| Constant | Image | Size | Use when |
|
|
101
|
+
|---|---|---|---|
|
|
102
|
+
| `CUDA_BASE` | `nvcr.io/nvidia/cuda:12.1.0-base-ubuntu22.04` | ~0.2 GB | Quick checks, no PyTorch |
|
|
103
|
+
| `PYTORCH_IMAGE` | `pytorch/pytorch:2.3.1-cuda12.1-cudnn8-runtime` | ~3.8 GB | Most ML jobs |
|
|
104
|
+
| `PYTORCH_NGC` | `nvcr.io/nvidia/pytorch:24.01-py3` | ~14 GB | Full NGC stack |
|
|
105
|
+
|
|
106
|
+
## Authentication
|
|
107
|
+
|
|
108
|
+
The SDK reads credentials from the Mimiry CLI automatically. Run once before using the SDK:
|
|
109
|
+
|
|
110
|
+
```bash
|
|
111
|
+
mimiry auth login
|
|
112
|
+
```
|
|
113
|
+
|
|
114
|
+
Tokens expire in ~5–8 minutes but are refreshed automatically by the SDK.
|
|
115
|
+
|
|
116
|
+
## License
|
|
117
|
+
|
|
118
|
+
MIT
|
|
@@ -0,0 +1,10 @@
|
|
|
1
|
+
mimiry/__init__.py,sha256=u2f3LVwyc2ib644X93vlQV3skfu-wjlG9PU-vhJRO-A,1109
|
|
2
|
+
mimiry/_auth.py,sha256=rdj00JDfsTmkWws7XmooetYW_BOjj9kk0wvX-QNYlmA,2173
|
|
3
|
+
mimiry/_exceptions.py,sha256=m3lIsy5HpRN0ZtEWEkiLTipQVwiYoogC7LuhjBaVRwI,1155
|
|
4
|
+
mimiry/_session.py,sha256=3PrvA5wEF_rxfIxPxFDuGpcCQcBs3qZU8loFhLTTm-g,9554
|
|
5
|
+
mimiry/client.py,sha256=na0SYOrqAbXxgF44TsTSXiv6cAzIsgUNWQJ-l0mRcDA,21713
|
|
6
|
+
mimiry-0.1.0.dist-info/licenses/LICENSE,sha256=CPRMv5J9OLMZRk7wVbXsWsRRUI9cxI5_ZLHIlvztF3E,1063
|
|
7
|
+
mimiry-0.1.0.dist-info/METADATA,sha256=IUXeVWpTvWqIkTkNQAihBawLcFrxF2VEZ_7dT2qssRc,3006
|
|
8
|
+
mimiry-0.1.0.dist-info/WHEEL,sha256=YCfwYGOYMi5Jhw2fU4yNgwErybb2IX5PEwBKV4ZbdBo,91
|
|
9
|
+
mimiry-0.1.0.dist-info/top_level.txt,sha256=aUypA93QzhCqQ7eLjpQbmeHCzCnbYPIYVPAWZb8YEJk,7
|
|
10
|
+
mimiry-0.1.0.dist-info/RECORD,,
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Mimiry
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
mimiry
|