dockhand-cli 0.3.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
dockhand/manage.py ADDED
@@ -0,0 +1,459 @@
1
+ """Docker container lifecycle management (logs, stop, remove, stats)."""
2
+
3
+ import re
4
+ import time
5
+ from datetime import datetime, timezone
6
+
7
+ import typer
8
+ from rich.console import Console
9
+ from rich.table import Table
10
+ from rich.text import Text
11
+
12
+ from dockhand.client import get_client, get_client_for_host
13
+ from dockhand.client.base import Client
14
+ from dockhand.config import DockerConfig, cli_config
15
+ from dockhand.error import error_and_exit
16
+ from dockhand.history import get_history_entry, load_history, mark_job_time, mark_stopped, save_history
17
+ from dockhand.queue import ts_list, ts_max_slots, ts_start_times
18
+ from dockhand.transport import Transport, entry_handle, get_transport, transport_for_entry
19
+
20
+ _STATE_STYLES = {
21
+ "running": "bold green",
22
+ "queued": "yellow",
23
+ "finished": "dim",
24
+ "stopped": "yellow",
25
+ "failed": "bold red",
26
+ "skipped": "dim",
27
+ }
28
+
29
+
30
+ def _user_command(full_cmd: str, imagename: str) -> str:
31
+ """Strip docker run boilerplate, returning only the user command."""
32
+ if imagename in full_cmd:
33
+ return full_cmd.split(imagename, 1)[-1].strip()
34
+ return full_cmd
35
+
36
+
37
+ _JOBS_DISPLAY_LIMIT = 30
38
+
39
+ _TERMINAL_STATES = ("finished", "failed", "stopped")
40
+
41
+ # Sort rank for `dockhand jobs`: running, then queued, then everything terminal
42
+ # (finished/failed/stopped/skipped) grouped last.
43
+ _JOB_CATEGORY_ORDER = {"running": 0, "queued": 1}
44
+
45
+
46
+ def _format_duration(seconds: float) -> str:
47
+ """Compact elapsed-time string, e.g. ``45s``, ``5m30s``, ``2h15m``, ``1d04h``."""
48
+ seconds = max(0, int(seconds))
49
+ if seconds < 60:
50
+ return f"{seconds}s"
51
+ minutes, seconds = divmod(seconds, 60)
52
+ if minutes < 60:
53
+ return f"{minutes}m{seconds:02d}s" if seconds else f"{minutes}m"
54
+ hours, minutes = divmod(minutes, 60)
55
+ if hours < 24:
56
+ return f"{hours}h{minutes:02d}m" if minutes else f"{hours}h"
57
+ days, hours = divmod(hours, 24)
58
+ return f"{days}d{hours:02d}h" if hours else f"{days}d"
59
+
60
+
61
+ def _format_time(ts: float | None) -> str:
62
+ if ts is None:
63
+ return "-"
64
+ return datetime.fromtimestamp(ts).strftime("%Y-%m-%d %H:%M:%S")
65
+
66
+
67
+ _DOCKER_TIME_RE = re.compile(r"^(\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2}\.\d{6})\d*Z$")
68
+
69
+
70
+ def _docker_started_at(client: Client, container_name: str) -> float | None:
71
+ """The container's real start time via `docker inspect`, or None if it can't be
72
+ determined (container never existed under this name, already removed, etc.) — the
73
+ caller falls back to an approximate timestamp in that case."""
74
+ returncode, stdout = client.run(
75
+ f"docker inspect --format '{{{{.State.StartedAt}}}}' {container_name}",
76
+ cwd=cli_config.remote_path,
77
+ capture=True,
78
+ )
79
+ if returncode != 0:
80
+ return None
81
+ ts = stdout.strip()
82
+ match = _DOCKER_TIME_RE.match(ts)
83
+ if not match or ts.startswith("0001-01-01"): # Go zero value = never started
84
+ return None
85
+ dt = datetime.strptime(match.group(1), "%Y-%m-%dT%H:%M:%S.%f").replace(tzinfo=timezone.utc)
86
+ return dt.timestamp()
87
+
88
+
89
+ def _host_now(client: Client) -> float:
90
+ """Current time on the job host. started_at comes from the host's `docker inspect`, so
91
+ elapsed time must be measured against the host clock too — otherwise clock skew between
92
+ this machine and the host shows up as a wrong (or clamped-to-0s) running duration.
93
+ Falls back to the local clock if the host can't be asked."""
94
+ returncode, stdout = client.run("date +%s", cwd=cli_config.remote_path, capture=True)
95
+ try:
96
+ return float(stdout.strip()) if returncode == 0 else time.time()
97
+ except ValueError:
98
+ return time.time()
99
+
100
+
101
+ def _observe_job_time(
102
+ client: Client, transport: Transport, entry: dict, *, state: str, duration_seconds: float | None, now: float
103
+ ) -> bool:
104
+ """Fill in started_at/ended_at the first time a job is observed in that state.
105
+
106
+ started_at for a *running* job is fetched from the container's actual `docker
107
+ inspect` start time when possible (exact, independent of when this happens to run) —
108
+ falling back to "now" only if the container can't be inspected (e.g. a job whose
109
+ container predates deterministic naming). ended_at for a job that has already
110
+ finished is derived from ts's own authoritative elapsed-time field when available
111
+ (`started_at + duration_seconds`), rather than guessed from the observation time,
112
+ since the container may already be gone (`--rm`) by the time anyone checks.
113
+
114
+ Mutates ``entry`` in place. Returns whether anything changed.
115
+ """
116
+ changed = False
117
+ if state == "running" and "started_at" not in entry:
118
+ container = transport.container_name(entry)
119
+ started_at = _docker_started_at(client, container) if container else None
120
+ entry["started_at"] = started_at if started_at is not None else now
121
+ changed = True
122
+ elif state in _TERMINAL_STATES and "ended_at" not in entry:
123
+ started_at = entry.get("started_at")
124
+ if duration_seconds is not None and started_at is not None:
125
+ entry["ended_at"] = started_at + duration_seconds
126
+ else:
127
+ entry["ended_at"] = now
128
+ changed = True
129
+ return changed
130
+
131
+
132
+ def execute_stats(config: DockerConfig, all: bool = False):
133
+ """List live jobs for the active transport (queue or direct docker).
134
+
135
+ Defaults to the most recent 30 jobs, newest (highest ID) first. ``--all``
136
+ lifts the 30-job cap and also includes finished/failed/stopped jobs.
137
+
138
+ Started is the container's real `docker inspect` start time where available (see
139
+ ``_docker_started_at``); Duration for a finished task-spooler job comes straight from
140
+ ts's own elapsed-time field. Both are exact regardless of when this command happens to
141
+ run. The remaining fallback (stamping "now" the first time a job is observed in a
142
+ state) only kicks in for jobs whose container can't be inspected — e.g. one submitted
143
+ before container naming was added, or already cleaned up.
144
+ """
145
+ transport = get_transport()
146
+ with get_client() as client:
147
+ jobs = transport.list_jobs(client)
148
+
149
+ if not all:
150
+ jobs = [j for j in jobs if j["state"] in ("running", "queued", "finished")]
151
+
152
+ if not jobs:
153
+ typer.echo("No active jobs." if not all else "No jobs.")
154
+ return
155
+
156
+ history = load_history()
157
+ handle_to_local = {
158
+ str(entry_handle(e)): e["local_id"] for e in history if entry_handle(e) is not None and "local_id" in e
159
+ }
160
+ entry_by_local = {e["local_id"]: e for e in history if "local_id" in e}
161
+ stopped_locals = {e["local_id"] for e in history if e.get("stopped")}
162
+
163
+ def _effective_state(job: dict) -> str:
164
+ local_id = handle_to_local.get(str(job["handle"]))
165
+ state = job["state"]
166
+ if local_id in stopped_locals and state in ("finished", "failed"):
167
+ return "stopped"
168
+ return state
169
+
170
+ # Group by category (running, then queued, then finished/failed/stopped), newest
171
+ # job id first within each group — matches how an operator scans the list: what's
172
+ # active right now, what's next, then history.
173
+ jobs.sort(
174
+ key=lambda j: (
175
+ _JOB_CATEGORY_ORDER.get(_effective_state(j), 2),
176
+ -handle_to_local.get(str(j["handle"]), -1),
177
+ )
178
+ )
179
+ if not all:
180
+ jobs = jobs[:_JOBS_DISPLAY_LIMIT]
181
+
182
+ table = Table(show_header=True, header_style="bold", box=None, padding=(0, 2))
183
+ table.add_column("ID", justify="right", style="bold")
184
+ table.add_column("Status")
185
+ table.add_column("Started")
186
+ table.add_column("Ended")
187
+ table.add_column("Duration")
188
+ table.add_column("Command")
189
+
190
+ now = _host_now(client)
191
+ history_changed = False
192
+
193
+ for job in jobs:
194
+ state = job["state"]
195
+ local_id = handle_to_local.get(str(job["handle"]))
196
+ if local_id in stopped_locals and state in ("finished", "failed"):
197
+ state = "stopped"
198
+
199
+ entry = entry_by_local.get(local_id)
200
+ duration_seconds = job.get("duration_seconds")
201
+ if entry is not None:
202
+ job_transport = transport_for_entry(entry)
203
+ if _observe_job_time(
204
+ client, job_transport, entry, state=state, duration_seconds=duration_seconds, now=now
205
+ ):
206
+ history_changed = True
207
+
208
+ started_at = entry.get("started_at") if entry else None
209
+ ended_at = entry.get("ended_at") if entry else None
210
+ if duration_seconds is not None:
211
+ duration_str = _format_duration(duration_seconds)
212
+ elif started_at is not None:
213
+ duration_str = _format_duration((ended_at or now) - started_at)
214
+ else:
215
+ duration_str = "-"
216
+
217
+ style = _STATE_STYLES.get(state, "")
218
+ status_text = Text(state, style=style)
219
+ id_str = str(local_id) if local_id is not None else f"{transport.name}:{job['handle']}"
220
+ user_cmd = _user_command(job["command"], config.imagename)
221
+ table.add_row(
222
+ id_str, status_text, _format_time(started_at), _format_time(ended_at), duration_str, Text(user_cmd)
223
+ )
224
+
225
+ if history_changed:
226
+ save_history(history)
227
+
228
+ Console().print(table)
229
+
230
+
231
+ def execute_queue(config: DockerConfig):
232
+ """Show the host's whole task-spooler queue, including other projects' jobs.
233
+
234
+ Task spooler runs one shared queue per host, so jobs submitted by other projects
235
+ on the same host already sit in the same `tsp -l` output — this prints it directly
236
+ instead of going through this project's history, which only know about its own jobs.
237
+ """
238
+ with get_client() as client:
239
+ jobs = ts_list(client, cwd=cli_config.remote_path)
240
+ max_slots = ts_max_slots(client, cwd=cli_config.remote_path)
241
+ if jobs:
242
+ started = ts_start_times(client, [j["id"] for j in jobs if j["state"] != "queued"], cli_config.remote_path)
243
+ now = _host_now(client)
244
+
245
+ if not jobs:
246
+ typer.echo("Task spooler queue is empty.")
247
+ return
248
+
249
+ # Same grouping as `dockhand jobs`: running, then queued (in ts's queue order), then the
250
+ # rest newest-started first. ts IDs aren't chronological (they restart with the server).
251
+ # The sort is stable, so running/queued jobs keep ts's order.
252
+ jobs.sort(
253
+ key=lambda j: (
254
+ _JOB_CATEGORY_ORDER.get(j["state"], 2),
255
+ 0 if j["state"] in _JOB_CATEGORY_ORDER else -started.get(j["id"], 0),
256
+ )
257
+ )
258
+
259
+ table = Table(show_header=True, header_style="bold", box=None, padding=(0, 2))
260
+ table.add_column("ID", justify="right", style="bold")
261
+ table.add_column("Project")
262
+ table.add_column("Status")
263
+ table.add_column("Started")
264
+ table.add_column("Duration")
265
+ table.add_column("Command")
266
+ table.add_column("TS ID", justify="right", style="dim")
267
+
268
+ for job in jobs:
269
+ state = job["state"]
270
+ start = started.get(job["id"])
271
+ if job.get("duration_seconds") is not None:
272
+ duration = _format_duration(job["duration_seconds"])
273
+ elif state == "running" and start is not None:
274
+ duration = _format_duration(now - start)
275
+ else:
276
+ duration = "-"
277
+ # dockhand's own job number, as used by `logs`/`stop`/`urgent` in the submitting project.
278
+ # Jobs that predate container naming (or weren't submitted by dockhand) have none.
279
+ name_match = re.fullmatch(r"dockhand-(\d+)", job.get("container_name") or "")
280
+ project = (job.get("image") or "-").split(":", 1)[0]
281
+ table.add_row(
282
+ name_match.group(1) if name_match else "-",
283
+ project,
284
+ Text(state, style=_STATE_STYLES.get(state, "")),
285
+ _format_time(start),
286
+ duration,
287
+ Text(job["command"]),
288
+ str(job["id"]),
289
+ )
290
+
291
+ Console().print(table)
292
+
293
+ running = sum(1 for j in jobs if j["state"] == "running")
294
+ queued = sum(1 for j in jobs if j["state"] == "queued")
295
+ slots_str = f"{running} running / {max_slots} slots" if max_slots is not None else f"{running} running"
296
+ typer.echo(f"\n{slots_str}, {queued} queued.")
297
+
298
+
299
+ def _resolve_entry(job_id: int | None) -> tuple[int, dict]:
300
+ """Resolve a local job ID (or default to last) to (local_id, history_entry)."""
301
+ history = load_history()
302
+ if not history:
303
+ error_and_exit("No job history found. Provide a job ID.")
304
+ if job_id is None:
305
+ entry = history[-1]
306
+ return entry["local_id"], entry
307
+ entry = get_history_entry(job_id)
308
+ if entry is None:
309
+ error_and_exit(f"Job #{job_id} not found in history.")
310
+ return job_id, entry
311
+
312
+
313
+ def _duration_header(local_id: int, state: str, started_at: float | None, ended_at: float | None, now: float) -> str:
314
+ if state == "queued":
315
+ return f"Job #{local_id} — queued (not started yet)."
316
+ if state == "running":
317
+ duration = _format_duration(now - started_at) if started_at is not None else "unknown"
318
+ return f"Job #{local_id} — running for {duration}."
319
+ if started_at is not None and ended_at is not None:
320
+ return f"Job #{local_id} — {state} in {_format_duration(ended_at - started_at)}."
321
+ return f"Job #{local_id} — {state}."
322
+
323
+
324
+ def execute_logs(
325
+ config: DockerConfig,
326
+ *,
327
+ job_id: int | None,
328
+ n: int | None,
329
+ follow: bool,
330
+ ):
331
+ """Show logs from a job (via the tsp output file, or ``docker logs``)."""
332
+ local_id, entry = _resolve_entry(job_id)
333
+ host = entry.get("host", "localhost")
334
+ transport = transport_for_entry(entry)
335
+ with get_client_for_host(host) as client:
336
+ jobs = transport.list_jobs(client)
337
+ job = next((j for j in jobs if str(j["handle"]) == str(entry_handle(entry))), None)
338
+ if job is not None:
339
+ now = _host_now(client)
340
+ state = job["state"]
341
+ if entry.get("stopped") and state in ("finished", "failed"):
342
+ state = "stopped"
343
+ if _observe_job_time(
344
+ client, transport, entry, state=state, duration_seconds=job.get("duration_seconds"), now=now
345
+ ):
346
+ mark_job_time(local_id, started_at=entry.get("started_at"), ended_at=entry.get("ended_at"))
347
+ typer.echo(_duration_header(local_id, state, entry.get("started_at"), entry.get("ended_at"), now))
348
+ returncode = transport.logs(client, entry, n=n, follow=follow)
349
+ if returncode != 0:
350
+ error_and_exit(f"Could not read logs for job #{local_id}. The job may still be queued and not yet started.")
351
+
352
+
353
+ def execute_stop(config: DockerConfig, *, job_id: int | None = None):
354
+ """Stop a running job or cancel a queued one."""
355
+ local_id, entry = _resolve_entry(job_id)
356
+ host = entry.get("host", "localhost")
357
+ with get_client_for_host(host) as client:
358
+ if transport_for_entry(entry).stop(client, entry):
359
+ mark_stopped(local_id)
360
+ typer.echo(f"Stopped job #{local_id}.")
361
+ else:
362
+ error_and_exit(f"Failed to stop job #{local_id}.")
363
+
364
+
365
+ def execute_remove(
366
+ config: DockerConfig,
367
+ job_ids: list[int] | None = None,
368
+ from_history: bool = False,
369
+ ):
370
+ """Remove pending job(s) from the queue, or clean up direct-run containers."""
371
+ if not job_ids:
372
+ history = load_history()
373
+ if not history:
374
+ error_and_exit("No job history found. Provide a job ID.")
375
+ job_ids = [history[-1]["local_id"]]
376
+
377
+ removed = []
378
+ for local_id in job_ids:
379
+ entry = get_history_entry(local_id)
380
+ if entry is None:
381
+ typer.echo(f"Job #{local_id} not found in history.")
382
+ continue
383
+ host = entry.get("host", "localhost")
384
+ with get_client_for_host(host) as client:
385
+ if transport_for_entry(entry).remove(client, entry):
386
+ typer.echo(f"Removed job #{local_id}.")
387
+ removed.append(local_id)
388
+ else:
389
+ typer.echo(f"Failed to remove job #{local_id} (it may have already finished).")
390
+
391
+ if from_history and removed:
392
+ history = load_history()
393
+ ids_set = set(removed)
394
+ history = [e for e in history if e.get("local_id") not in ids_set]
395
+ save_history(history)
396
+ typer.echo(f"Removed {len(removed)} job(s) from history.")
397
+
398
+
399
+ def _baked_image_refs(history: list[dict]) -> dict[str, list]:
400
+ """Map each distinct baked image tag in history to the job IDs that used it.
401
+
402
+ Only images dockhand built for bake delivery qualify — their ``image_ref`` differs
403
+ from the base ``imagename``. Mount jobs (``image_ref`` == base name, or absent) are
404
+ never pruned since dockhand didn't create those tags.
405
+ """
406
+ refs: dict[str, list] = {}
407
+ for entry in history:
408
+ cfg = entry.get("config", {})
409
+ ref = cfg.get("image_ref")
410
+ if ref and ref != cfg.get("imagename"):
411
+ refs.setdefault(ref, []).append(entry.get("local_id"))
412
+ return refs
413
+
414
+
415
+ def execute_prune(config: DockerConfig, *, yes: bool = False, dry_run: bool = False):
416
+ """Remove baked images that no active (running/queued) job still references."""
417
+ history = load_history()
418
+ baked = _baked_image_refs(history)
419
+ if not baked:
420
+ typer.echo("No baked images to prune.")
421
+ return
422
+
423
+ # Keep images referenced by jobs that are still running or queued.
424
+ transport = get_transport()
425
+ with get_client() as client:
426
+ jobs = transport.list_jobs(client)
427
+ active_handles = {str(j["handle"]) for j in jobs if j["state"] in ("running", "queued")}
428
+ in_use = {
429
+ entry["config"]["image_ref"]
430
+ for entry in history
431
+ if str(entry_handle(entry)) in active_handles and entry.get("config", {}).get("image_ref")
432
+ }
433
+
434
+ candidates = [ref for ref in baked if ref not in in_use]
435
+ if not candidates:
436
+ typer.echo("Nothing to prune — all baked images are in use by active jobs.")
437
+ return
438
+
439
+ typer.echo("The following baked images will be removed:")
440
+ for ref in candidates:
441
+ job_ids = ", ".join(str(i) for i in baked[ref] if i is not None)
442
+ typer.echo(f" - {ref} (jobs: {job_ids})")
443
+
444
+ if dry_run:
445
+ return
446
+ if not yes and not typer.confirm("Remove these images?"):
447
+ typer.echo("Aborted.")
448
+ return
449
+
450
+ removed = 0
451
+ with get_client() as client:
452
+ for ref in candidates:
453
+ returncode, _ = client.run(f"docker image rm {ref}", cwd=cli_config.remote_path, capture=True)
454
+ if returncode == 0:
455
+ typer.echo(f"Removed {ref}")
456
+ removed += 1
457
+ else:
458
+ typer.echo(f"Could not remove {ref} (in use or already gone).")
459
+ typer.echo(f"Pruned {removed} image(s).")
dockhand/queue.py ADDED
@@ -0,0 +1,174 @@
1
+ """Task spooler (ts) queue integration."""
2
+
3
+ import re
4
+ from datetime import datetime
5
+
6
+ from dockhand.client.base import Client
7
+
8
+
9
+ def ts_submit(client: Client, docker_cmd: str, cwd: str, slots: int = 1) -> int:
10
+ """Submit a docker command to task spooler. Returns the ts job ID."""
11
+ slots_flag = f"-N {slots} " if slots > 1 else ""
12
+ returncode, stdout = client.run(f"tsp {slots_flag}{docker_cmd}", cwd=cwd, capture=True)
13
+ if returncode != 0:
14
+ from dockhand.error import error_and_exit
15
+
16
+ error_and_exit("Failed to submit job to task spooler. Is 'tsp' installed on the host?")
17
+ try:
18
+ return int(stdout.strip())
19
+ except ValueError:
20
+ from dockhand.error import error_and_exit
21
+
22
+ error_and_exit(f"Unexpected output from task spooler: {stdout.strip()!r}")
23
+
24
+
25
+ def ts_list(client: Client, cwd: str) -> list[dict]:
26
+ """List all task spooler jobs. Returns a list of parsed job dicts."""
27
+ returncode, stdout = client.run("tsp -l", cwd=cwd, capture=True)
28
+ if returncode != 0:
29
+ return []
30
+ return _parse_ts_list(stdout)
31
+
32
+
33
+ def ts_make_urgent(client: Client, job_id: int, cwd: str) -> bool:
34
+ """Move a queued job to the front of the queue. Returns True on success."""
35
+ returncode, _ = client.run(f"tsp -u {job_id}", cwd=cwd, capture=True)
36
+ return returncode == 0
37
+
38
+
39
+ def ts_remove(client: Client, job_id: int, cwd: str) -> bool:
40
+ """Remove a pending job from the queue. Returns True on success."""
41
+ returncode, _ = client.run(f"tsp -r {job_id}", cwd=cwd, capture=True)
42
+ return returncode == 0
43
+
44
+
45
+ def ts_kill(client: Client, job_id: int, cwd: str) -> bool:
46
+ """Send SIGTERM to a running job. Returns True on success."""
47
+ returncode, _ = client.run(f"tsp -k {job_id}", cwd=cwd, capture=True)
48
+ return returncode == 0
49
+
50
+
51
+ def ts_get_job(client: Client, job_id: int, cwd: str) -> dict | None:
52
+ """Look up a single job by ID from ts -l. Returns None if not found."""
53
+ jobs = ts_list(client, cwd=cwd)
54
+ return next((j for j in jobs if j["id"] == job_id), None)
55
+
56
+
57
+ def ts_max_slots(client: Client, cwd: str) -> int | None:
58
+ """Current max slot count configured for the host's task spooler (`tsp -S`)."""
59
+ returncode, stdout = client.run("tsp -S", cwd=cwd, capture=True)
60
+ if returncode != 0:
61
+ return None
62
+ try:
63
+ return int(stdout.strip())
64
+ except ValueError:
65
+ return None
66
+
67
+
68
+ def extract_docker_command(input_string):
69
+ # Regex breakdown:
70
+ # -v\s+\S+:\S+:\w+ -> Matches the -v flag and the path mapping
71
+ # .* -> Greedily matches as much as possible to find the LAST -v
72
+ # \s+\S+ -> Matches the image name
73
+ # \s+(.*) -> Captures everything after that image name
74
+ pattern = r".*-v\s+\S+:\S+:\w+\s+\S+\s+(.*)"
75
+ match = re.search(pattern, input_string)
76
+
77
+ if match:
78
+ return match.group(1).strip()
79
+ else:
80
+ return None
81
+
82
+
83
+ # Matches the "real/user/sys" Times(r/u/s) field ts prints for finished jobs, e.g.
84
+ # "25851.03/2.62/3.05". Only real (wall-clock) seconds are kept. Searched for rather than
85
+ # read off a fixed column: ts pads this field with spaces rather than "-/-/-" placeholders
86
+ # while a job is still running/queued, which shifts the column count under a positional split.
87
+ _TIMES_RE = re.compile(r"(\d+\.\d+)/\d+\.\d+/\d+\.\d+")
88
+
89
+ # The image is the token right after the last `-v host:container:perm` mount (same anchor
90
+ # as extract_docker_command); the container name is dockhand's `--name dockhand-<local_id>`.
91
+ _IMAGE_RE = re.compile(r".*-v\s+\S+:\S+:\w+\s+(\S+)")
92
+ _NAME_RE = re.compile(r"--name\s+(\S+)")
93
+
94
+
95
+ def _parse_ts_list(output: str) -> list[dict]:
96
+ """Parse ts -l output into a list of dicts.
97
+
98
+ Expected format (header line followed by job lines):
99
+ ID State Output E-Level Times(r/u/s) Command [run=N/M]
100
+ 0 finished /tmp/ts-out.XXX 0 0.1/0.0/0.0 docker run ...
101
+ 1 running /tmp/ts-out.YYY docker run ...
102
+ 2 queued (file) docker run ...
103
+ """
104
+ jobs = []
105
+ lines = output.strip().split("\n")
106
+ if len(lines) < 2:
107
+ return jobs
108
+
109
+ for line in lines[1:]: # skip header
110
+ line = line.strip()
111
+ if not line:
112
+ continue
113
+ # Split into at most 6 parts; the last part is the full command string
114
+ parts = re.split(r"\s+", line, maxsplit=5)
115
+ command = extract_docker_command(line)
116
+ if len(parts) < 2:
117
+ continue
118
+ try:
119
+ job_id = int(parts[0])
120
+ except ValueError:
121
+ continue
122
+ times_match = _TIMES_RE.search(line)
123
+ image_match = _IMAGE_RE.match(line)
124
+ name_match = _NAME_RE.search(line)
125
+ jobs.append(
126
+ {
127
+ "id": job_id,
128
+ "state": parts[1], # queued / running / finished / failed / skipped
129
+ "output_file": parts[2] if len(parts) > 2 else None,
130
+ "exit_code": parts[3] if len(parts) > 3 else None,
131
+ # Authoritative wall-clock duration from ts itself, once the job has
132
+ # finished — exact regardless of when this was observed.
133
+ "duration_seconds": float(times_match.group(1)) if times_match else None,
134
+ "command": command if command else "",
135
+ "image": image_match.group(1) if image_match else None,
136
+ "container_name": name_match.group(1) if name_match else None,
137
+ }
138
+ )
139
+
140
+ return jobs
141
+
142
+
143
+ def ts_start_times(client: Client, job_ids: list[int], cwd: str) -> dict[int, float]:
144
+ """Start time (epoch seconds) of each started job, from `tsp -i`, in one round trip.
145
+
146
+ `tsp -l` has no start time; `tsp -i` prints one per job in the host's local time
147
+ ("Start time: Sat Oct 10 07:18:53 2026"), so the host's UTC offset is fetched alongside
148
+ to convert it. Queued jobs have no start time and are simply absent from the result.
149
+ """
150
+ if not job_ids:
151
+ return {}
152
+ loop = "; ".join(f"echo @@{i}; tsp -i {i} | grep '^Start time:'" for i in job_ids)
153
+ returncode, stdout = client.run(f"date +%z; {loop}", cwd=cwd, capture=True)
154
+ lines = stdout.strip().split("\n")
155
+ if returncode != 0 or not lines:
156
+ return {}
157
+ try:
158
+ host_tz = datetime.strptime(lines[0].strip(), "%z").tzinfo
159
+ except ValueError:
160
+ return {}
161
+
162
+ starts = {}
163
+ current = None
164
+ for line in lines[1:]:
165
+ if line.startswith("@@"):
166
+ current = int(line[2:])
167
+ elif line.startswith("Start time:") and current is not None:
168
+ # ctime pads single-digit days ("Sep 7"); collapse whitespace before parsing.
169
+ text = " ".join(line.removeprefix("Start time:").split())
170
+ try:
171
+ starts[current] = datetime.strptime(text, "%a %b %d %H:%M:%S %Y").replace(tzinfo=host_tz).timestamp()
172
+ except ValueError:
173
+ pass
174
+ return starts
dockhand/resubmit.py ADDED
@@ -0,0 +1,49 @@
1
+ """Docker container resubmission."""
2
+ import dataclasses
3
+
4
+ import typer
5
+
6
+ from dockhand.config import DockerConfig, DockerResubmitConfig
7
+ from dockhand.error import error_and_exit
8
+
9
+
10
+ def execute_resubmit(docker_config: DockerConfig, resubmit_config: DockerResubmitConfig):
11
+ """Resubmit a previous docker run with optional overrides."""
12
+ from dockhand.history import get_history_entry, load_history
13
+ from dockhand.submit import execute_submit
14
+
15
+ history = load_history()
16
+
17
+ if not history:
18
+ error_and_exit("No docker history found. Submit a docker job first.")
19
+
20
+ local_id = int(resubmit_config.container_id) if resubmit_config.container_id else None
21
+ if local_id is not None:
22
+ entry = get_history_entry(local_id)
23
+ if entry is None:
24
+ error_and_exit(f"Job #{local_id} not found in history.")
25
+ else:
26
+ entry = history[-1]
27
+
28
+ original_config = entry["config"]
29
+
30
+ commands = resubmit_config.commands if resubmit_config.commands is not None else original_config.get("commands", [])
31
+ imagename = resubmit_config.imagename if resubmit_config.imagename is not None else original_config.get("imagename")
32
+ gpus = resubmit_config.gpus if resubmit_config.gpus is not None else original_config.get("gpus")
33
+
34
+ updated_config = dataclasses.replace(docker_config, imagename=imagename, gpus=gpus)
35
+
36
+ # Reproduce a baked job verbatim: rerun the exact image it originally built instead of
37
+ # re-resolving/rebuilding from the current code. Only when the original ran a distinct
38
+ # baked tag and the image name wasn't overridden.
39
+ original_image_ref = original_config.get("image_ref")
40
+ original_imagename = original_config.get("imagename")
41
+ pin_image = (
42
+ original_image_ref
43
+ if (resubmit_config.imagename is None and original_image_ref and original_image_ref != original_imagename)
44
+ else None
45
+ )
46
+ if pin_image is not None:
47
+ typer.echo(f"Reusing baked image {pin_image} from the original run.")
48
+
49
+ execute_submit(updated_config, commands, sync=False, imagename=imagename, gpus=gpus, image_ref=pin_image)