docker-stack 2.2.4__py3-none-any.whl → 2.3.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
docker_stack/cli.py CHANGED
@@ -36,12 +36,16 @@ from docker_stack.shell_auth import (
36
36
  run_managed_shell,
37
37
  shell_session_active,
38
38
  )
39
+ from docker_stack.conflict_prompt import ForceDeployPrompt
39
40
  from docker_stack.manager_api import (
40
- FEATURE_CLUSTER_CONTAINER_CLI,
41
+ DEPLOY_WAIT_ENV,
41
42
  FEATURE_STACK_DEPLOY,
42
43
  FEATURE_STACK_QUERY,
43
44
  ManagerApiClient,
45
+ describe_active_deployment,
46
+ describe_run,
44
47
  discover_manager_client,
48
+ format_deploy_error_body,
45
49
  )
46
50
  from docker_stack.registry import DockerRegistry
47
51
  from .envsubst import LineCheckResult, SubstitutionError, envsubst, envsubst_load_file
@@ -102,15 +106,23 @@ def _write_selected_node(value: Optional[str], display_name: str) -> None:
102
106
  Path(node_state).write_text(f"{display_name}\n", encoding="utf-8")
103
107
 
104
108
 
105
- def _manager_with_cluster_cli() -> ManagerApiClient:
106
- client = discover_manager_client()
107
- if client is None or not client.supports(FEATURE_CLUSTER_CONTAINER_CLI):
108
- raise RuntimeError("Docker-Manager does not support cluster container CLI; upgrade the manager and all node agents")
109
+ def _require_manager() -> ManagerApiClient:
110
+ """Return a client for the Docker-Manager behind this shell, or explain why not.
111
+
112
+ Only the current manager is supported, so there is no capability negotiation here -
113
+ either the target is a Docker-Manager or it is a plain Docker daemon.
114
+ """
115
+ client = discover_manager_client(strict=True)
116
+ if client is None or not client.is_manager_backend():
117
+ raise RuntimeError(
118
+ f"{client.manager_url if client else 'the current Docker endpoint'} is not a "
119
+ "Docker-Manager; node-local commands need one"
120
+ )
109
121
  return client
110
122
 
111
123
 
112
124
  def _select_node(value: str) -> str:
113
- client = _manager_with_cluster_cli()
125
+ client = _require_manager()
114
126
  requested = value.strip()
115
127
  if requested.lower() == "cluster":
116
128
  _write_selected_node(None, "cluster")
@@ -133,6 +145,9 @@ def _select_node(value: str) -> str:
133
145
  raise RuntimeError(f"node '{requested}' is not ready")
134
146
  node_id = str(node.get("id") or "").strip()
135
147
  hostname = str(node.get("hostname") or node_id).strip()
148
+ # Only this node has to be usable. The health of every other node - drained, down, or
149
+ # missing its agent - is none of this command's business.
150
+ client.check_node_agent(node_id or hostname)
136
151
  _write_selected_node(node_id, hostname)
137
152
  return hostname
138
153
 
@@ -339,6 +354,13 @@ def _print_cluster_containers(payload: Dict[str, object], *, no_trunc: bool = Fa
339
354
  columns.append("SIZE")
340
355
  columns.append("NODE")
341
356
  _print_table(columns, rows)
357
+ skipped = payload.get("skipped_nodes") if isinstance(payload.get("skipped_nodes"), list) else []
358
+ for node in skipped:
359
+ if isinstance(node, dict):
360
+ print(
361
+ f"docker ps: skipped node {node.get('node_name') or node.get('node_id')}: {node.get('reason')}",
362
+ file=sys.stderr,
363
+ )
342
364
  errors = payload.get("node_errors") if isinstance(payload.get("node_errors"), list) else []
343
365
  for error in errors:
344
366
  if isinstance(error, dict):
@@ -413,7 +435,7 @@ def _managed_docker(arguments: List[str]) -> int:
413
435
  return _exec_docker(values)
414
436
  try:
415
437
  client = discover_manager_client()
416
- if client is None or not client.supports(FEATURE_CLUSTER_CONTAINER_CLI):
438
+ if client is None or not client.is_manager_backend():
417
439
  return _exec_docker(values)
418
440
  filters = []
419
441
  all_containers = False
@@ -1483,6 +1505,7 @@ class DockerStack:
1483
1505
  options=options,
1484
1506
  )
1485
1507
  else:
1508
+ prompt = ForceDeployPrompt.create()
1486
1509
  try:
1487
1510
  payload = manager_client.deploy_stack_stream(
1488
1511
  stack=stack_name,
@@ -1490,6 +1513,7 @@ class DockerStack:
1490
1513
  compose=rendered_content,
1491
1514
  options=options,
1492
1515
  on_event=DockerStack._print_manager_deploy_event,
1516
+ wait_for_poll=prompt.wait if prompt else None,
1493
1517
  )
1494
1518
  except RuntimeError as exc:
1495
1519
  message = str(exc)
@@ -1554,6 +1578,51 @@ class DockerStack:
1554
1578
  message = data.get("message") or data.get("error") or ""
1555
1579
  if message:
1556
1580
  print(f"[manager] {service}: {status}: {message}", flush=True)
1581
+ elif event_name == "deployment":
1582
+ deployment_id = data.get("deployment_id")
1583
+ if deployment_id:
1584
+ mode = data.get("mode") or "deployment"
1585
+ print(f"[manager] {mode} {deployment_id}", flush=True)
1586
+ elif event_name == "waiting":
1587
+ DockerStack._print_manager_wait(data)
1588
+
1589
+ @staticmethod
1590
+ def _print_manager_wait(active: Dict[str, object]) -> None:
1591
+ """Narrate the wait for another run of this stack, and any attempt to force past it.
1592
+
1593
+ The manager will not queue an interactive deploy for us, so the CLI polls
1594
+ and reports each phase: waiting, forcing (asking the manager to abort the
1595
+ holder), and how that abort ended.
1596
+ """
1597
+ if not isinstance(active, dict):
1598
+ active = {}
1599
+ phase = str(active.get("phase") or "waiting")
1600
+ run = describe_run(active)
1601
+ if phase == "waiting":
1602
+ budget = active.get("wait_budget_secs")
1603
+ suffix = f" (waiting up to {budget}s, set {DEPLOY_WAIT_ENV} to change)" if isinstance(budget, int) else ""
1604
+ message = f"waiting: {describe_active_deployment(active)}{suffix}"
1605
+ elif phase == "forcing":
1606
+ message = f"forcing: asking the manager to abort {run}"
1607
+ elif phase == "aborted":
1608
+ message = f"aborted {run}; deploying now"
1609
+ elif phase == "aborted_other":
1610
+ message = f"aborted a different run than shown: {run}; deploying now"
1611
+ elif phase == "free":
1612
+ message = "the stack is free; deploying now"
1613
+ elif phase == "not_released":
1614
+ message = f"asked the manager to abort {run}, but it has not released the stack yet; waiting for it to finish"
1615
+ elif phase == "forbidden":
1616
+ message = f"you do not have permission to abort {run}; still waiting"
1617
+ elif phase == "still_held":
1618
+ message = f"the stack is still held by {run}; waiting"
1619
+ elif phase == "force_used":
1620
+ message = "force was already used once in this run; waiting for the stack to free up"
1621
+ elif phase == "interrupted":
1622
+ message = "interrupted while asking the manager to abort; that request may still complete"
1623
+ else:
1624
+ message = f"{phase}: {run}"
1625
+ print(f"[manager] {message}", file=sys.stderr, flush=True)
1557
1626
 
1558
1627
  @staticmethod
1559
1628
  def _rollback_via_manager(
@@ -1563,10 +1632,13 @@ class DockerStack:
1563
1632
  namespace: str,
1564
1633
  version: str,
1565
1634
  ) -> Optional[str]:
1635
+ prompt = ForceDeployPrompt.create()
1566
1636
  payload = manager_client.rollback_stack(
1567
1637
  stack=stack_name,
1568
1638
  namespace=namespace,
1569
1639
  version=version,
1640
+ on_wait=DockerStack._print_manager_wait,
1641
+ wait_for_poll=prompt.wait if prompt else None,
1570
1642
  )
1571
1643
  warnings = payload.get("warnings") or []
1572
1644
  for warning in warnings:
@@ -1851,7 +1923,7 @@ def _humanize_manager_error(message: str) -> str:
1851
1923
  return message
1852
1924
  if not isinstance(body, dict):
1853
1925
  return message
1854
- detail = str(body.get("message", "")).strip()
1926
+ detail = format_deploy_error_body(body) if body.get("code") else str(body.get("message", "")).strip()
1855
1927
  if not detail:
1856
1928
  return message
1857
1929
  incident = str(body.get("incident_id") or body.get("trace_id") or "").strip()
@@ -1879,6 +1951,17 @@ def _manager_deploy_suggestion(message: str) -> Optional[str]:
1879
1951
  "by this stack. Either allow that external network in the Docker-Manager deployment rule, attach the service to "
1880
1952
  "a stack-owned network, or relabel/recreate the existing network with the expected stack ownership before deploying."
1881
1953
  )
1954
+ if "is still running" in lowered and ("gave up after waiting" in lowered or "http 409" in lowered):
1955
+ return (
1956
+ "Suggestion: another run holds this stack's deploy lock. Rerun and let it wait, raise "
1957
+ f"{DEPLOY_WAIT_ENV} to queue longer, press Enter twice while waiting at a terminal to abort that run "
1958
+ "and force your deploy, or abort it from the Docker-Manager stack page."
1959
+ )
1960
+ if "failed services:" in lowered:
1961
+ return (
1962
+ "Suggestion: every service operation has finished; the failed ones are listed above with their incident ids. "
1963
+ "Fix the cause and redeploy, or use the rollback offered on the Docker-Manager stack page to restore the previous specs."
1964
+ )
1882
1965
  if "stack not found" in lowered:
1883
1966
  return (
1884
1967
  "Suggestion: the manager was reached but has no stack by that name in that namespace. "
@@ -1901,9 +1984,10 @@ def _manager_deploy_suggestion(message: str) -> Optional[str]:
1901
1984
  def _format_called_process_error(exc: subprocess.CalledProcessError) -> str:
1902
1985
  output = _error_output(exc)
1903
1986
  lines = [f"docker-stack: command failed with exit code {exc.returncode}: {_format_command(exc.cmd)}"]
1904
- if output:
1905
- lines.append(_humanize_manager_error(output))
1906
- suggestion = _manager_deploy_suggestion(output)
1987
+ humanized = _humanize_manager_error(output) if output else ""
1988
+ if humanized:
1989
+ lines.append(humanized)
1990
+ suggestion = _manager_deploy_suggestion(humanized or output)
1907
1991
  if suggestion:
1908
1992
  lines.extend(["", suggestion])
1909
1993
  return "\n".join(lines)
@@ -1911,8 +1995,9 @@ def _format_called_process_error(exc: subprocess.CalledProcessError) -> str:
1911
1995
 
1912
1996
  def _format_runtime_error(exc: RuntimeError) -> str:
1913
1997
  message = str(exc).strip() or exc.__class__.__name__
1914
- lines = [f"docker-stack: {_humanize_manager_error(message)}"]
1915
- suggestion = _manager_deploy_suggestion(message)
1998
+ humanized = _humanize_manager_error(message)
1999
+ lines = [f"docker-stack: {humanized}"]
2000
+ suggestion = _manager_deploy_suggestion(humanized)
1916
2001
  if suggestion:
1917
2002
  lines.extend(["", suggestion])
1918
2003
  return "\n".join(lines)
@@ -0,0 +1,134 @@
1
+ """Terminal prompt shown while a deploy waits for another run of the same stack.
2
+
3
+ Docker-Manager refuses to queue interactive applies: it answers ``409
4
+ deployment_in_progress`` and leaves the choice between waiting and aborting to
5
+ whoever is on the other end. On a CLI that is the person at the terminal, so
6
+ while the client polls for the stack to free up this prompt listens for two
7
+ Enter presses in quick succession and reports ``"force"``; the caller then asks
8
+ the manager to abort the running deployment. Ctrl+C is left alone and keeps
9
+ exiting the CLI.
10
+
11
+ Enter is line-oriented, so no terminal mode is changed and there is nothing to
12
+ restore on the way out. The prompt is only created when both stdin and stderr
13
+ are terminals and the process is in the foreground; anywhere else (CI, pipes,
14
+ the GitHub Action's heredoc) the caller simply sleeps between polls.
15
+ """
16
+
17
+ import io
18
+ import os
19
+ import select
20
+ import sys
21
+ import time
22
+ from typing import Any, Optional
23
+
24
+ DOUBLE_ENTER_SECS = 3.0
25
+ WINDOWS_POLL_SECS = 0.1
26
+ HINT = "[manager] press Enter twice to force your deploy (aborts that run; changes it already made to the daemon stay), Ctrl+C to quit"
27
+ SECOND_ENTER_HINT = f"[manager] press Enter again within {int(DOUBLE_ENTER_SECS)}s to force the deploy"
28
+
29
+ ENTER = "enter"
30
+ OTHER = "other"
31
+ TIMEOUT = "timeout"
32
+ EOF = "eof"
33
+
34
+
35
+ class _PosixEnterReader:
36
+ def __init__(self, stdin, fd: int):
37
+ self._stdin = stdin
38
+ self._fd = fd
39
+
40
+ def wait_for_enter(self, timeout: float) -> str:
41
+ try:
42
+ readable, _, _ = select.select([self._fd], [], [], max(0.0, timeout))
43
+ except (OSError, ValueError):
44
+ return EOF
45
+ if not readable:
46
+ return TIMEOUT
47
+ line = self._stdin.readline()
48
+ if line == "":
49
+ return EOF
50
+ return ENTER if line.strip() == "" else OTHER
51
+
52
+
53
+ class _WindowsEnterReader:
54
+ def __init__(self, msvcrt_module):
55
+ self._msvcrt = msvcrt_module
56
+
57
+ def wait_for_enter(self, timeout: float) -> str:
58
+ deadline = time.monotonic() + max(0.0, timeout)
59
+ while True:
60
+ if self._msvcrt.kbhit():
61
+ char = self._msvcrt.getwch()
62
+ return ENTER if char in ("\r", "\n") else OTHER
63
+ remaining = deadline - time.monotonic()
64
+ if remaining <= 0:
65
+ return TIMEOUT
66
+ time.sleep(min(WINDOWS_POLL_SECS, remaining))
67
+
68
+
69
+ class ForceDeployPrompt:
70
+ """Listens for a double Enter between polls; see the module docstring."""
71
+
72
+ def __init__(self, stderr, reader):
73
+ self._stderr = stderr
74
+ self._reader = reader
75
+ self._hinted = False
76
+ self._dead = False
77
+ self._first_enter_at: Optional[float] = None
78
+
79
+ @classmethod
80
+ def create(cls) -> Optional["ForceDeployPrompt"]:
81
+ """Build a prompt when a person can see and answer it, else ``None``."""
82
+ stdin, stderr = sys.stdin, sys.stderr
83
+ if stdin is None or stderr is None:
84
+ return None
85
+ try:
86
+ if not stdin.isatty() or not stderr.isatty():
87
+ return None
88
+ fd = stdin.fileno()
89
+ except (AttributeError, ValueError, OSError, io.UnsupportedOperation):
90
+ return None
91
+ if os.name == "nt":
92
+ try:
93
+ import msvcrt # type: ignore[import-not-found]
94
+ except ImportError:
95
+ return None
96
+ return cls(stderr, _WindowsEnterReader(msvcrt))
97
+ try:
98
+ # Reading a terminal from a background job stops the process (SIGTTIN).
99
+ if os.tcgetpgrp(fd) != os.getpgrp():
100
+ return None
101
+ except (AttributeError, OSError):
102
+ return None
103
+ return cls(stderr, _PosixEnterReader(stdin, fd))
104
+
105
+ def _say(self, message: str) -> None:
106
+ print(message, file=self._stderr, flush=True)
107
+
108
+ def wait(self, active: Any, seconds: float) -> Optional[str]:
109
+ """Wait up to ``seconds`` for the next poll; ``"force"`` if Enter was pressed twice."""
110
+ deadline = time.monotonic() + max(0.0, seconds)
111
+ if not self._hinted:
112
+ self._say(HINT)
113
+ self._hinted = True
114
+ while True:
115
+ remaining = deadline - time.monotonic()
116
+ if remaining <= 0:
117
+ return None
118
+ if self._dead:
119
+ time.sleep(remaining)
120
+ return None
121
+ result = self._reader.wait_for_enter(remaining)
122
+ if result == TIMEOUT:
123
+ return None
124
+ if result == EOF:
125
+ self._dead = True
126
+ continue
127
+ if result != ENTER:
128
+ continue
129
+ now = time.monotonic()
130
+ if self._first_enter_at is not None and now - self._first_enter_at <= DOUBLE_ENTER_SECS:
131
+ self._first_enter_at = None
132
+ return "force"
133
+ self._first_enter_at = now
134
+ self._say(SECOND_ENTER_HINT)
@@ -4,6 +4,8 @@ import shutil
4
4
  import socket
5
5
  import ssl
6
6
  import subprocess
7
+ import sys
8
+ import time
7
9
  import urllib.error
8
10
  import urllib.parse
9
11
  import urllib.request
@@ -215,6 +217,208 @@ def _manager_deploy_timeout_secs(default: int) -> int:
215
217
  )
216
218
 
217
219
 
220
+ # How long a deploy/rollback waits for another run of the same stack to finish
221
+ # before giving up. ``0`` fails immediately. Defaults to the deploy timeout.
222
+ DEPLOY_WAIT_ENV = "DOCKER_MANAGER_DEPLOY_WAIT_SECS"
223
+ DEPLOY_WAIT_POLL_SECS = 5
224
+ DEPLOY_WAIT_NOTICE_SECS = 30
225
+ # The manager's abort handler itself waits up to 30s for the holder to let go
226
+ # of the stack, so the client must give it comfortably more than that.
227
+ ABORT_REQUEST_TIMEOUT_SECS = 45
228
+
229
+
230
+ def _manager_deploy_wait_secs(default: int) -> int:
231
+ raw = os.getenv(DEPLOY_WAIT_ENV, "").strip()
232
+ if not raw:
233
+ return _manager_deploy_timeout_secs(default)
234
+ try:
235
+ return max(0, int(raw))
236
+ except ValueError:
237
+ return _manager_deploy_timeout_secs(default)
238
+
239
+
240
+ class ManagerDeploymentInProgressError(RuntimeError):
241
+ """The manager refused an apply because another run holds this stack's deploy guard.
242
+
243
+ Docker-Manager answers interactive apply paths (``deploy/stream``, version
244
+ rollback) with ``409 deployment_in_progress`` instead of queueing. The
245
+ ``active_deployment`` payload names the run: ``operation``, ``actor``,
246
+ ``deployment_id``, ``started_at_ms``.
247
+ """
248
+
249
+ def __init__(self, message: str, *, active_deployment: Optional[Dict[str, Any]] = None):
250
+ super().__init__(message)
251
+ self.active_deployment = active_deployment if isinstance(active_deployment, dict) else {}
252
+
253
+
254
+ class ManagerAbortForbiddenError(RuntimeError):
255
+ """The caller may deploy this stack but lacks the permission to abort another run of it."""
256
+
257
+
258
+ class ManagerNoActiveDeploymentError(RuntimeError):
259
+ """An abort found nothing running: the holder finished on its own."""
260
+
261
+
262
+ class ManagerDeployFailedError(RuntimeError):
263
+ """A manager deploy stream ended with an ``error`` event.
264
+
265
+ ``services`` carries the per-service outcome when the manager reports a
266
+ partial failure (``code`` = ``deployment_partial_failure``); every parallel
267
+ service operation has completed by then, so this is the full picture.
268
+ """
269
+
270
+ def __init__(
271
+ self,
272
+ message: str,
273
+ *,
274
+ code: Optional[str] = None,
275
+ deployment_id: Optional[str] = None,
276
+ rollback_available: bool = False,
277
+ services: Optional[list] = None,
278
+ ):
279
+ super().__init__(message)
280
+ self.code = code
281
+ self.deployment_id = deployment_id
282
+ self.rollback_available = rollback_available
283
+ self.services = services or []
284
+
285
+
286
+ def describe_run(active: Any) -> str:
287
+ """Name a deploy run: who started what, when, and its deployment id if known."""
288
+ if not isinstance(active, dict):
289
+ return "another deployment of this stack"
290
+ operation = str(active.get("operation") or "deployment").strip() or "deployment"
291
+ actor = str(active.get("actor") or "").strip() or "another caller"
292
+ parts = [f"a {operation} started by {actor}"]
293
+ started_at_ms = active.get("started_at_ms")
294
+ if isinstance(started_at_ms, (int, float)) and started_at_ms > 0:
295
+ elapsed = max(0, int(time.time() - started_at_ms / 1000))
296
+ parts.append(f"{elapsed}s ago")
297
+ deployment_id = str(active.get("deployment_id") or "").strip()
298
+ if deployment_id:
299
+ parts.append(f"(deployment {deployment_id})")
300
+ return " ".join(parts)
301
+
302
+
303
+ def describe_active_deployment(active: Any) -> str:
304
+ """One line naming the run that holds a stack: who, what, since when."""
305
+ return describe_run(active) + " is still running"
306
+
307
+
308
+ def same_deploy_run(shown: Any, aborted: Any) -> bool:
309
+ """Whether the run the manager aborted is the one the user was looking at.
310
+
311
+ ``deployment_id`` is authoritative when both sides have one; it is ``null``
312
+ for a run that has not persisted its row yet and for non-stream deploys, so
313
+ fall back to who started what and when.
314
+ """
315
+ if not isinstance(shown, dict) or not isinstance(aborted, dict):
316
+ return True
317
+ shown_id = str(shown.get("deployment_id") or "").strip()
318
+ aborted_id = str(aborted.get("deployment_id") or "").strip()
319
+ if shown_id and aborted_id:
320
+ return shown_id == aborted_id
321
+ keys = ("operation", "actor", "started_at_ms")
322
+ return all(shown.get(key) == aborted.get(key) for key in keys)
323
+
324
+
325
+ def format_failed_services(services: Any) -> list:
326
+ """Render the manager's per-service results as one line per failed service."""
327
+ lines = []
328
+ if not isinstance(services, list):
329
+ return lines
330
+ for item in services:
331
+ if not isinstance(item, dict):
332
+ continue
333
+ if str(item.get("status") or "").strip() not in {"failed", "error"}:
334
+ continue
335
+ service = str(item.get("service") or item.get("logical_name") or "service").strip()
336
+ action = str(item.get("action") or "").strip()
337
+ label = f"{service} ({action})" if action else service
338
+ error = str(item.get("error") or "").strip() or "failed without an error message"
339
+ incident = str(item.get("incident_id") or "").strip()
340
+ suffix = f" (incident {incident})" if incident else ""
341
+ lines.append(f" - {label}: {error}{suffix}")
342
+ return lines
343
+
344
+
345
+ def format_deploy_error_body(body: Dict[str, Any]) -> str:
346
+ """Expand a manager deploy error body into prose, keeping every reported cause."""
347
+ message = str(body.get("message") or "").strip()
348
+ code = str(body.get("code") or "").strip()
349
+ if code == "deployment_in_progress":
350
+ return describe_active_deployment(body.get("active_deployment")) if body.get("active_deployment") else message
351
+ lines = [message] if message else []
352
+ failed = format_failed_services(body.get("services"))
353
+ if failed:
354
+ lines.append("failed services:")
355
+ lines.extend(failed)
356
+ deployment_id = str(body.get("deployment_id") or "").strip()
357
+ if body.get("rollback_available") and deployment_id:
358
+ lines.append(f"rollback is available from the Docker-Manager stack page (deployment {deployment_id})")
359
+ elif deployment_id:
360
+ lines.append(f"deployment {deployment_id}")
361
+ return "\n".join(lines)
362
+
363
+
364
+ def _wait_notifier(on_event):
365
+ """Surface deploy-guard waits through the same event callback as the stream."""
366
+ if on_event is None:
367
+ return None
368
+ return lambda active: on_event({"event": "waiting", "data": active})
369
+
370
+
371
+ ABORT_DEPLOYMENT_PATH = "/api/stacks/deploy/abort"
372
+
373
+
374
+ def _decode_http_error_body(exc: urllib.error.HTTPError) -> str:
375
+ try:
376
+ return exc.read().decode("utf-8", errors="replace").strip()
377
+ except Exception:
378
+ return ""
379
+
380
+
381
+ def _manager_http_error(method: str, path: str, exc: urllib.error.HTTPError) -> RuntimeError:
382
+ """Map an HTTP failure to the most specific error the caller can act on."""
383
+ payload = _decode_http_error_body(exc)
384
+ body: Any = None
385
+ if payload:
386
+ try:
387
+ body = json.loads(payload)
388
+ except json.JSONDecodeError:
389
+ body = None
390
+ if exc.code == 409 and isinstance(body, dict) and body.get("code") == "deployment_in_progress":
391
+ active = body.get("active_deployment")
392
+ return ManagerDeploymentInProgressError(
393
+ f"Manager request failed ({method} {path}): HTTP 409: {describe_active_deployment(active)}",
394
+ active_deployment=active if isinstance(active, dict) else {},
395
+ )
396
+ if exc.code == 404 and isinstance(body, dict) and body.get("code") == "no_active_deployment":
397
+ return ManagerNoActiveDeploymentError(f"Manager request failed ({method} {path}): HTTP 404: no deployment is running")
398
+ if exc.code == 403 and path == ABORT_DEPLOYMENT_PATH:
399
+ detail = str(body.get("message") or "").strip() if isinstance(body, dict) else ""
400
+ return ManagerAbortForbiddenError(
401
+ f"Manager request failed ({method} {path}): HTTP 403: {detail or 'stack deploy permission required'}"
402
+ )
403
+ suffix = f": {payload}" if payload else ""
404
+ return RuntimeError(f"Manager request failed ({method} {path}): HTTP {exc.code}{suffix}")
405
+
406
+
407
+ def _stream_error(data: Any, default: str) -> ManagerDeployFailedError:
408
+ if not isinstance(data, dict):
409
+ message = str(data).strip() if data is not None else ""
410
+ return ManagerDeployFailedError(message or default)
411
+ message = format_deploy_error_body(data) or default
412
+ services = data.get("services")
413
+ return ManagerDeployFailedError(
414
+ message,
415
+ code=str(data.get("code") or "").strip() or None,
416
+ deployment_id=str(data.get("deployment_id") or "").strip() or None,
417
+ rollback_available=bool(data.get("rollback_available")),
418
+ services=services if isinstance(services, list) else [],
419
+ )
420
+
421
+
218
422
  def _manager_target_from_env() -> Optional[str]:
219
423
  manager_url = os.getenv("DOCKER_MANAGER_URL", "").strip()
220
424
  if manager_url:
@@ -293,14 +497,7 @@ class ManagerApiClient:
293
497
  with urllib.request.urlopen(request, timeout=timeout, context=context) as response:
294
498
  return json.loads(response.read().decode("utf-8"))
295
499
  except urllib.error.HTTPError as exc:
296
- body = ""
297
- try:
298
- payload = exc.read().decode("utf-8", errors="replace").strip()
299
- if payload:
300
- body = f": {payload}"
301
- except Exception:
302
- body = ""
303
- raise RuntimeError(f"Manager request failed ({method} {path}): HTTP {exc.code}{body}") from exc
500
+ raise _manager_http_error(method, path, exc) from exc
304
501
  except urllib.error.URLError as exc:
305
502
  raise RuntimeError(f"Manager request failed ({method} {path}): {exc.reason}") from exc
306
503
  except (TimeoutError, socket.timeout) as exc:
@@ -361,14 +558,7 @@ class ManagerApiClient:
361
558
  event_data = {"raw": raw_data}
362
559
  yield {"event": event_name, "data": event_data}
363
560
  except urllib.error.HTTPError as exc:
364
- body = ""
365
- try:
366
- payload = exc.read().decode("utf-8", errors="replace").strip()
367
- if payload:
368
- body = f": {payload}"
369
- except Exception:
370
- body = ""
371
- raise RuntimeError(f"Manager request failed ({method} {path}): HTTP {exc.code}{body}") from exc
561
+ raise _manager_http_error(method, path, exc) from exc
372
562
  except urllib.error.URLError as exc:
373
563
  raise RuntimeError(f"Manager request failed ({method} {path}): {exc.reason}") from exc
374
564
  except (TimeoutError, socket.timeout) as exc:
@@ -416,6 +606,144 @@ class ManagerApiClient:
416
606
  endpoint_id = self._resolve_endpoint_id()
417
607
  return f"/api/endpoints/{endpoint_id}{normalized}"
418
608
 
609
+ def is_manager_backend(self) -> bool:
610
+ """True when the target is a Docker-Manager stack API rather than a raw daemon."""
611
+ return self._detect_manager_backend()
612
+
613
+ def abort_active_deployment(self, *, stack: str, namespace: str) -> Optional[Dict[str, Any]]:
614
+ """Ask the manager to abort whatever holds this stack's deploy guard.
615
+
616
+ Returns the manager's ``{"aborted": <run>, "released": bool}`` payload, or
617
+ ``None`` when nothing was running any more. The manager waits up to 30s
618
+ for the holder to release before answering, hence the long timeout.
619
+ """
620
+ try:
621
+ payload = self._request_json(
622
+ ABORT_DEPLOYMENT_PATH,
623
+ method="POST",
624
+ payload={"namespace": namespace, "stack": stack},
625
+ timeout_secs=max(ABORT_REQUEST_TIMEOUT_SECS, int(self.timeout_secs or 0)),
626
+ )
627
+ except ManagerNoActiveDeploymentError:
628
+ return None
629
+ if not isinstance(payload, dict):
630
+ raise RuntimeError("Docker-Manager abort response is invalid")
631
+ return payload
632
+
633
+ def _force_takeover(self, *, stack: str, namespace: str, shown: Dict[str, Any], notify) -> str:
634
+ """Abort the run holding the stack so ours can go. Returns ``"retry"`` or ``"wait"``.
635
+
636
+ A daemon request the aborted run already made still completes; abort stops
637
+ its orchestration, and our deploy then applies over whatever state that left.
638
+ """
639
+ notify("forcing", shown)
640
+ try:
641
+ outcome = self.abort_active_deployment(stack=stack, namespace=namespace)
642
+ except ManagerAbortForbiddenError:
643
+ notify("forbidden", shown)
644
+ return "wait"
645
+ except ManagerDeploymentInProgressError as exc:
646
+ notify("still_held", exc.active_deployment or shown)
647
+ return "wait"
648
+ except KeyboardInterrupt:
649
+ notify("interrupted", shown)
650
+ raise
651
+ if outcome is None:
652
+ notify("free", shown)
653
+ return "retry"
654
+ aborted = outcome.get("aborted")
655
+ aborted = aborted if isinstance(aborted, dict) else shown
656
+ if not outcome.get("released"):
657
+ notify("not_released", aborted)
658
+ return "wait"
659
+ notify("aborted" if same_deploy_run(shown, aborted) else "aborted_other", aborted)
660
+ return "retry"
661
+
662
+ def _run_when_stack_is_free(
663
+ self,
664
+ operation,
665
+ *,
666
+ stack: Optional[str] = None,
667
+ namespace: Optional[str] = None,
668
+ on_wait=None,
669
+ wait_for_poll=None,
670
+ ):
671
+ """Run ``operation``, waiting out another deploy of the same stack.
672
+
673
+ The manager's interactive apply paths answer ``409 deployment_in_progress``
674
+ rather than queueing, because a person in the UI can choose to wait or
675
+ abort. Nobody is there to choose on a CLI or CI run, so queue here: retry
676
+ until the guard frees or ``DOCKER_MANAGER_DEPLOY_WAIT_SECS`` runs out,
677
+ and tell the caller what is being waited on. A budget of ``0`` fails
678
+ on the first collision.
679
+
680
+ ``wait_for_poll(active, seconds)`` replaces the sleep between polls for a
681
+ caller that can listen to a person; returning ``"force"`` aborts the
682
+ running deployment (once per run) so this one can take the stack.
683
+ ``on_wait`` receives the active run plus a ``phase`` for every notice.
684
+ """
685
+ budget = _manager_deploy_wait_secs(self.timeout_secs)
686
+ started = time.monotonic()
687
+ last_notice: Optional[float] = None
688
+ force_used = False
689
+ force_used_notified = False
690
+ skip_budget_check = False
691
+
692
+ def notify(phase: str, active: Any) -> None:
693
+ if on_wait:
694
+ base = active if isinstance(active, dict) else {}
695
+ on_wait({**base, "phase": phase})
696
+
697
+ while True:
698
+ try:
699
+ return operation()
700
+ except ManagerDeploymentInProgressError as exc:
701
+ waited = int(time.monotonic() - started)
702
+ if skip_budget_check:
703
+ skip_budget_check = False
704
+ elif waited + DEPLOY_WAIT_POLL_SECS > budget:
705
+ if budget > 0:
706
+ raise ManagerDeploymentInProgressError(
707
+ f"{exc}; gave up after waiting {waited}s "
708
+ f"(raise {DEPLOY_WAIT_ENV} to wait longer, or abort it from the Docker-Manager stack page)",
709
+ active_deployment=exc.active_deployment,
710
+ ) from exc
711
+ raise
712
+ now = time.monotonic()
713
+ if on_wait and (last_notice is None or now - last_notice >= DEPLOY_WAIT_NOTICE_SECS):
714
+ on_wait(
715
+ {**exc.active_deployment, "phase": "waiting", "waited_secs": waited, "wait_budget_secs": budget}
716
+ )
717
+ last_notice = now
718
+ if wait_for_poll is None:
719
+ time.sleep(DEPLOY_WAIT_POLL_SECS)
720
+ continue
721
+ decision = wait_for_poll(exc.active_deployment, DEPLOY_WAIT_POLL_SECS)
722
+ if decision != "force":
723
+ continue
724
+ if force_used:
725
+ if not force_used_notified:
726
+ notify("force_used", exc.active_deployment)
727
+ force_used_notified = True
728
+ continue
729
+ force_used = True
730
+ if stack is None or namespace is None:
731
+ raise RuntimeError("cannot force a deploy without knowing its stack and namespace")
732
+ if self._force_takeover(stack=stack, namespace=namespace, shown=exc.active_deployment, notify=notify) == "retry":
733
+ skip_budget_check = True
734
+
735
+ def check_node_agent(self, selector: str) -> Dict[str, Any]:
736
+ """Report whether one node can serve node-local Docker commands.
737
+
738
+ Scoped to ``selector`` on purpose: an unrelated node that is drained, down, or
739
+ missing its agent must never make a usable node look unusable.
740
+ """
741
+ node = urllib.parse.quote(selector, safe="")
742
+ payload = self._request_json(f"/api/docker-stack/nodes/{node}/agent")
743
+ if not isinstance(payload, dict):
744
+ raise RuntimeError("Docker-Manager node agent response is invalid")
745
+ return payload
746
+
419
747
  def supports(self, feature_name: str) -> bool:
420
748
  features = self.detect_features()
421
749
  if FEATURE_MESUDIP_DOCKER_ENTERPRISE in features and feature_name in {
@@ -633,6 +961,7 @@ class ManagerApiClient:
633
961
  images: Dict[str, str],
634
962
  options: Optional[Dict[str, Any]] = None,
635
963
  on_event=None,
964
+ wait_for_poll=None,
636
965
  ) -> Dict[str, Any]:
637
966
  payload: Dict[str, Any] = {
638
967
  "stack": stack,
@@ -643,23 +972,18 @@ class ManagerApiClient:
643
972
  if prepared_options:
644
973
  payload["options"] = prepared_options
645
974
 
646
- done: Optional[Dict[str, Any]] = None
647
- for event in self._request_sse_events(
648
- "/api/stacks/deploy/images",
649
- method="POST",
650
- payload=payload,
651
- timeout_secs=_manager_deploy_timeout_secs(self.timeout_secs),
652
- ):
653
- if on_event:
654
- on_event(event)
655
- event_name = event.get("event")
656
- data = event.get("data")
657
- if event_name == "error":
658
- message = data.get("message") if isinstance(data, dict) else str(data)
659
- raise RuntimeError(message or "manager image deploy stream failed")
660
- if event_name == "done" and isinstance(data, dict):
661
- done = data
662
- return done or {"warnings": [], "stdout": "", "stderr": ""}
975
+ return self._run_when_stack_is_free(
976
+ lambda: self._consume_deploy_stream(
977
+ "/api/stacks/deploy/images",
978
+ payload,
979
+ on_event=on_event,
980
+ default_error="manager image deploy stream failed",
981
+ ),
982
+ stack=stack,
983
+ namespace=namespace,
984
+ on_wait=_wait_notifier(on_event),
985
+ wait_for_poll=wait_for_poll,
986
+ )
663
987
 
664
988
  def deploy_stack_stream(
665
989
  self,
@@ -669,15 +993,37 @@ class ManagerApiClient:
669
993
  compose: str,
670
994
  options: Optional[Dict[str, Any]] = None,
671
995
  on_event=None,
996
+ wait_for_poll=None,
672
997
  ) -> Dict[str, Any]:
673
998
  payload: Dict[str, Any] = {"stack": stack, "namespace": namespace, "compose": compose}
674
999
  prepared_options = _public_deploy_options(options, compose)
675
1000
  if prepared_options:
676
1001
  payload["options"] = prepared_options
677
1002
 
1003
+ return self._run_when_stack_is_free(
1004
+ lambda: self._consume_deploy_stream(
1005
+ "/api/stacks/deploy/stream",
1006
+ payload,
1007
+ on_event=on_event,
1008
+ default_error="manager deploy stream failed",
1009
+ ),
1010
+ stack=stack,
1011
+ namespace=namespace,
1012
+ on_wait=_wait_notifier(on_event),
1013
+ wait_for_poll=wait_for_poll,
1014
+ )
1015
+
1016
+ def _consume_deploy_stream(
1017
+ self,
1018
+ path: str,
1019
+ payload: Dict[str, Any],
1020
+ *,
1021
+ on_event,
1022
+ default_error: str,
1023
+ ) -> Dict[str, Any]:
678
1024
  done: Optional[Dict[str, Any]] = None
679
1025
  for event in self._request_sse_events(
680
- "/api/stacks/deploy/stream",
1026
+ path,
681
1027
  method="POST",
682
1028
  payload=payload,
683
1029
  timeout_secs=_manager_deploy_timeout_secs(self.timeout_secs),
@@ -687,19 +1033,32 @@ class ManagerApiClient:
687
1033
  event_name = event.get("event")
688
1034
  data = event.get("data")
689
1035
  if event_name == "error":
690
- message = data.get("message") if isinstance(data, dict) else str(data)
691
- raise RuntimeError(message or "manager deploy stream failed")
1036
+ raise _stream_error(data, default_error)
692
1037
  if event_name == "done" and isinstance(data, dict):
693
1038
  done = data
694
1039
  return done or {"warnings": [], "stdout": "", "stderr": ""}
695
1040
 
696
- def rollback_stack(self, *, stack: str, namespace: str, version: str) -> Dict[str, Any]:
1041
+ def rollback_stack(
1042
+ self,
1043
+ *,
1044
+ stack: str,
1045
+ namespace: str,
1046
+ version: str,
1047
+ on_wait=None,
1048
+ wait_for_poll=None,
1049
+ ) -> Dict[str, Any]:
697
1050
  quoted_stack = urllib.parse.quote(stack, safe="")
698
1051
  if self._detect_manager_backend():
699
- return self._request_json(
700
- f"/api/stacks/{quoted_stack}/rollback",
701
- method="POST",
702
- payload={"namespace": namespace, "version": version},
1052
+ return self._run_when_stack_is_free(
1053
+ lambda: self._request_json(
1054
+ f"/api/stacks/{quoted_stack}/rollback",
1055
+ method="POST",
1056
+ payload={"namespace": namespace, "version": version},
1057
+ ),
1058
+ stack=stack,
1059
+ namespace=namespace,
1060
+ on_wait=on_wait,
1061
+ wait_for_poll=wait_for_poll,
703
1062
  )
704
1063
  return self._request_json(
705
1064
  self._endpoint_path(f"/inventory/stacks/{quoted_stack}/rollback"),
@@ -721,7 +1080,15 @@ class ManagerApiClient:
721
1080
  )
722
1081
 
723
1082
 
724
- def discover_manager_client(timeout_secs: int = 5) -> Optional[ManagerApiClient]:
1083
+ def discover_manager_client(
1084
+ timeout_secs: int = 5, *, strict: bool = False
1085
+ ) -> Optional[ManagerApiClient]:
1086
+ """Build a client for the manager behind DOCKER_MANAGER_URL or the Docker context.
1087
+
1088
+ ``strict`` raises the concrete reason instead of returning ``None``. Callers that
1089
+ fall back to the plain Docker CLI want the quiet form; callers that are about to
1090
+ report a failure to the user must not discard why discovery failed.
1091
+ """
725
1092
  try:
726
1093
  target = _manager_target_from_env()
727
1094
  if target:
@@ -729,11 +1096,26 @@ def discover_manager_client(timeout_secs: int = 5) -> Optional[ManagerApiClient]
729
1096
  elif shutil.which("docker"):
730
1097
  _, context_target = current_docker_context_target()
731
1098
  if not context_target or not context_target.startswith(("tcp://", "http://", "https://")):
1099
+ if strict:
1100
+ raise RuntimeError(
1101
+ "no Docker-Manager endpoint found: set DOCKER_MANAGER_URL, or select a "
1102
+ f"docker context with a tcp:// endpoint (current target: {context_target or 'none'})"
1103
+ )
732
1104
  return None
733
1105
  config = resolve_login_config(manager_target=context_target)
734
1106
  else:
1107
+ if strict:
1108
+ raise RuntimeError(
1109
+ "no Docker-Manager endpoint found: DOCKER_MANAGER_URL is unset and the docker CLI is not installed"
1110
+ )
735
1111
  return None
736
- except Exception:
1112
+ except RuntimeError:
1113
+ if strict:
1114
+ raise
1115
+ return None
1116
+ except Exception as exc:
1117
+ if strict:
1118
+ raise RuntimeError(f"failed resolving the Docker-Manager endpoint: {exc}") from exc
737
1119
  return None
738
1120
 
739
1121
  return ManagerApiClient(
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: docker-stack
3
- Version: 2.2.4
3
+ Version: 2.3.0
4
4
  Summary: CLI for deploying and managing Docker stacks.
5
5
  Home-page: https://github.com/mesudip/docker-stack
6
6
  Author: Sudip Bhattarai
@@ -213,6 +213,18 @@ are sent unchanged to the Docker CLI.
213
213
  - **Docker Stack Versioning and Config Backup for Rollback:**
214
214
  The utility automatically versions your Docker configs and secrets, allowing for easy tracking of changes and seamless rollbacks to previous states. This provides a safety net for your deployments, ensuring you can always revert to a stable configuration.
215
215
 
216
+ - **Waiting for another deploy of the same stack (Docker-Manager):**
217
+ Docker-Manager runs one apply per stack at a time. When `docker-stack deploy` or `docker-stack checkout` finds another run in progress (a UI deploy, a CI image bump, another operator), it waits and prints who started it:
218
+
219
+ ```
220
+ [manager] waiting: a deploy started by alice 42s ago is still running (waiting up to 300s, set DOCKER_MANAGER_DEPLOY_WAIT_SECS to change)
221
+ [manager] press Enter twice to force your deploy (aborts that run; changes it already made to the daemon stay), Ctrl+C to quit
222
+ ```
223
+
224
+ - It retries every 5 seconds until the stack is free or `DOCKER_MANAGER_DEPLOY_WAIT_SECS` runs out (default: the deploy timeout; `0` fails immediately). CI runs wait the same way, without the prompt.
225
+ - At a terminal, pressing Enter twice within 3 seconds asks the manager to abort the running deployment and deploys as soon as the stack is released. This needs stack deploy permission and is offered once per run. The aborted run stops orchestrating, but daemon requests it already made still complete; your deploy then applies over that state.
226
+ - Ctrl+C exits without touching the other run.
227
+
216
228
  ## Why Use It?
217
229
 
218
230
  Vanilla Docker Stack deployments can sometimes lack the flexibility needed for dynamic environments or robust secret management. This utility bridges those gaps by:
@@ -1,20 +1,21 @@
1
1
  docker_stack/__init__.py,sha256=qgHdW8ZDBlyk6FzIf8Mld_cJD1SjOfqo2dELg4pK668,880
2
- docker_stack/cli.py,sha256=_IWypxyKZ4reUaB4qb7WpRt24WtUVWLlmrr3ALY8riM,96017
2
+ docker_stack/cli.py,sha256=rEIjauHe9Cz0k-2j6guaeThQR2aaVrFeEW16qG1y8QI,100468
3
3
  docker_stack/command_runner.py,sha256=mNaUAVKtrJbji_5ARVPLKhc92avqllNKrVQj1A6S0dA,1252
4
4
  docker_stack/compose.py,sha256=_fAVesjyW5ecid46sYYhcu02yF1vqknDSCFYflJPDOE,534
5
+ docker_stack/conflict_prompt.py,sha256=1pmsyJZyZbzKYAtLJ8kYNVSC0LJb6fsu8Qz2zWvyqBE,4911
5
6
  docker_stack/docker_objects.py,sha256=U6szytv5gKrmBYKqThSeMMXdiJ2u4STIQLkeqQ-Fn4Q,12103
6
7
  docker_stack/envsubst.py,sha256=gdEUpsjWJeNbbahySk56vUdQh1nns85o3LbU6SUBo3I,8337
7
8
  docker_stack/envsubst_merge.py,sha256=TwOu8BducG0rWnSreJGwesNi-VghCetljcf-aWceE8M,4971
8
9
  docker_stack/helpers.py,sha256=nJHaIS1vtFyPYclXVl3-Qd-1h55jnSy1LwdATDrF4hk,5953
9
10
  docker_stack/login.py,sha256=jnrmrDh79Brlzj89YdZ1sGpg-ex16yb3ZN47eFpcW5g,41754
10
- docker_stack/manager_api.py,sha256=xb8RwY4LEn1usg9qhG9AfiCvBJ2tpQawVMLRwF3vYHY,28361
11
+ docker_stack/manager_api.py,sha256=9AnVzsbYDllo1NeQM5vq7zTlQfPXXEYX24vXeuQCj74,44403
11
12
  docker_stack/markers.py,sha256=gqStnr0tNWZ1kzqvREc2jVQVQrevheShgJtL5QzSx4g,2388
12
13
  docker_stack/merge_conf.py,sha256=Pmsabcgf3SDyaWvf2OLsBPodzEqsvOIjj9xzn_qWXH0,1917
13
14
  docker_stack/registry.py,sha256=sWC1J9JDIrcYyhei1POYCZHJ0xUCKC2pveisSHMgYsQ,9036
14
15
  docker_stack/shell_auth.py,sha256=UUu9y4wMFKfc_j86yGcasvS_TLZ2uV4kv0T4XKTMQBQ,33424
15
16
  docker_stack/url_parser.py,sha256=Sk8GQE0nEiwCkijp1OltP2AfgtKEmM2hybcH_rBhfiI,6824
16
- docker_stack-2.2.4.dist-info/METADATA,sha256=__Y1aknQuctP1Nuf2zhuinUnlyYDu-qGY54rUma8alM,13821
17
- docker_stack-2.2.4.dist-info/WHEEL,sha256=SmOxYU7pzNKBqASvQJ7DjX3XGUF92lrGhMb3R6_iiqI,91
18
- docker_stack-2.2.4.dist-info/entry_points.txt,sha256=mpe2RwIguARsosXIUBQEN2pKU53URqZCh2S4ATwDFL4,55
19
- docker_stack-2.2.4.dist-info/top_level.txt,sha256=zT6TPL54cLrt9LO_MNkhEpGGOmsoe2HV6Na5Ohy3_2c,13
20
- docker_stack-2.2.4.dist-info/RECORD,,
17
+ docker_stack-2.3.0.dist-info/METADATA,sha256=AOgiK8qI7kxr18yUxvOqEDrbOhpLMR_vRCi9zhuKCcU,15013
18
+ docker_stack-2.3.0.dist-info/WHEEL,sha256=SmOxYU7pzNKBqASvQJ7DjX3XGUF92lrGhMb3R6_iiqI,91
19
+ docker_stack-2.3.0.dist-info/entry_points.txt,sha256=mpe2RwIguARsosXIUBQEN2pKU53URqZCh2S4ATwDFL4,55
20
+ docker_stack-2.3.0.dist-info/top_level.txt,sha256=zT6TPL54cLrt9LO_MNkhEpGGOmsoe2HV6Na5Ohy3_2c,13
21
+ docker_stack-2.3.0.dist-info/RECORD,,