stacktrace-cli 0.5.1__py3-none-any.whl → 0.5.2__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -2,7 +2,7 @@
2
2
 
3
3
  import logging as _logging
4
4
 
5
- __version__ = "0.5.1"
5
+ __version__ = "0.5.2"
6
6
 
7
7
  # Only the daemon attaches a handler (ADR-0048). Without this one, a WARNING
8
8
  # logged in any other process would reach Python's last-resort handler and
stacktrace_cli/cli.py CHANGED
@@ -40,6 +40,7 @@ from .sessions.access import collect_sessions
40
40
  from .sessions.render import render_json, render_text
41
41
  from .telemetry import emit_error
42
42
  from .telemetry.cli import telemetry as telemetry_cmd
43
+ from .webhook.cli import webhook as webhook_cmd
43
44
 
44
45
  PASSTHROUGH: Final = ("scan", "bom", "policy")
45
46
 
@@ -586,3 +587,4 @@ main.add_command(monitor)
586
587
  main.add_command(daemon_cmd, "daemon")
587
588
  main.add_command(findings)
588
589
  main.add_command(telemetry_cmd, "telemetry")
590
+ main.add_command(webhook_cmd, "webhook")
@@ -0,0 +1,39 @@
1
+ """One entry point for a Claude Code cloud environment's setup script.
2
+
3
+ A cloud environment's setup script runs once, as root, before the filesystem is
4
+ snapshotted and reused by every later session. What a session needs that is not
5
+ the CLI itself — the per-session boot hook and the asset identity — has to be
6
+ laid down there so it lands inside the snapshot.
7
+
8
+ Rather than paste that into the setup-script box, a setup script installs the
9
+ CLI and runs this:
10
+
11
+ UV_TOOL_DIR=/usr/local/share/uv-tools UV_TOOL_BIN_DIR=/usr/local/bin \
12
+ uv tool install stacktrace-cli
13
+ stacktrace-cloud-setup
14
+
15
+ Both variables matter: setup runs as root, and `uv tool install`'s root
16
+ defaults put the tool's environment under `/root/.local/share/uv/tools` and the
17
+ linked executable under `/root/.local/bin` — both inside `/root`, which a later
18
+ non-root session user cannot traverse. `UV_TOOL_BIN_DIR` alone moves the
19
+ symlink but not the environment it points into, so `command -v stacktrace` would
20
+ find the binary while running it still fails; both must move.
21
+
22
+ The work itself is `setup.sh`, shipped beside this module so it travels in the
23
+ wheel. This shim only locates it and hands off, so the logic stays one auditable
24
+ bash file rather than being reimplemented in Python.
25
+ """
26
+
27
+ from __future__ import annotations
28
+
29
+ import os
30
+ import sys
31
+ from pathlib import Path
32
+
33
+
34
+ def main() -> None:
35
+ script = Path(__file__).with_name("setup.sh")
36
+ # execvp so the process *becomes* bash: its exit status is the setup
37
+ # script's, which is what the environment reads to decide the session may
38
+ # start, and there is no Python frame left to swallow a signal.
39
+ os.execvp("bash", ["bash", str(script), *sys.argv[1:]])
@@ -0,0 +1,247 @@
1
+ #!/usr/bin/env bash
2
+ # Lay down everything a Claude Code cloud session needs, at environment setup.
3
+ #
4
+ # Run once by the environment's setup script, as root, before the filesystem is
5
+ # snapshotted. Two things, both of which must be inside that snapshot because
6
+ # the setup script does not run again for later sessions:
7
+ #
8
+ # 1. /usr/local/bin/stacktrace-boot.sh — the per-session hook that configures
9
+ # the remote and starts the daemon. Registered as a SessionStart hook in
10
+ # every home so it fires whichever user the session runs as.
11
+ # 2. The asset identity, in every home's ~/.config/stacktrace/asset-id, so
12
+ # `stacktrace remote sync` resolves it at the persisted rung (ADR-0065)
13
+ # instead of minting a fresh asset per session.
14
+ #
15
+ # The telemetry install id is deliberately NOT seeded here. Seeding it would
16
+ # make `read_install_id()` non-None before any event runs, so the one-time
17
+ # `installed` event (which `emit()` synthesizes only when the id is absent)
18
+ # would never fire. Left unseeded, the daemon's first in-session event mints the
19
+ # id and fires `installed` once per VM, correctly attributed `remote=true`
20
+ # because it runs inside the session. Counting one install per VM rather than
21
+ # per environment is ADR-0043's documented cloud tradeoff; per-environment is
22
+ # not cleanly achievable for an ephemeral VM (a setup-time emit would carry the
23
+ # wrong `remote` value and could not flush from a one-shot process).
24
+ #
25
+ # Every home rather than $HOME: setup runs as root and the user a session runs
26
+ # as is not documented, so writing all of them removes the guess. The asset id
27
+ # is unconditional — each environment instance gets its own, and a leftover from
28
+ # a previous snapshot must not win.
29
+ #
30
+ # Scope (ADR-0070): this seeds the snapshot and starts the daemon per session.
31
+ # It does not own daemon lifetime across a supervisor — that is ADR-0066, and a
32
+ # cloud VM is one session per VM, so the daemon dying with the VM is correct.
33
+ # The environment this targets is a fresh, single-tenant snapshot: it does not
34
+ # defend against adversarial or pre-existing symlinks and special files in a
35
+ # home, because that home does not exist until this run creates it.
36
+ #
37
+ # Options (all optional):
38
+ # --asset-id VALUE use this asset id instead of minting one; also taken
39
+ # from STACKTRACE_ASSET_EXTERNAL_ID in the environment
40
+ # --root DIR treat DIR as / (for testing)
41
+ #
42
+ # The remote token and API URL are read by the boot hook at session time from
43
+ # STACKTRACE_REMOTE_TOKEN and STACKTRACE_REMOTE_API_URL, not here: they belong
44
+ # to the environment's variables, not to the snapshot this seeds.
45
+ set -uo pipefail
46
+
47
+ ASSET_ID="${STACKTRACE_ASSET_EXTERNAL_ID:-}"
48
+ ROOT=""
49
+
50
+ while [ $# -gt 0 ]; do
51
+ case "$1" in
52
+ # `$# -ge 2` before the shift: without it, `--root` with no operand leaves
53
+ # the positional parameters unchanged under `set -u` with no `set -e`, and
54
+ # the loop spins forever — a malformed invocation would hang setup.
55
+ --asset-id) [ $# -ge 2 ] || { echo "✗ --asset-id needs a value" >&2; exit 2; }; ASSET_ID="$2"; shift 2 ;;
56
+ --root) [ $# -ge 2 ] || { echo "✗ --root needs a value" >&2; exit 2; }; ROOT="$2"; shift 2 ;;
57
+ -h|--help) sed -n '2,44p' "$0" | sed 's/^# \{0,1\}//'; exit 0 ;;
58
+ *) echo "✗ unknown argument: $1" >&2; exit 2 ;;
59
+ esac
60
+ done
61
+
62
+ LOG="${ROOT}/var/log/stacktrace-cloud-setup.log"
63
+ mkdir -p "$(dirname "$LOG")" 2>/dev/null || true
64
+ exec > >(tee -a "$LOG") 2>&1
65
+ echo "=== stacktrace cloud setup $(date -u +%FT%TZ) ==="
66
+
67
+ # ---- identities ----------------------------------------------------------
68
+ _uuid() {
69
+ if [ -r /proc/sys/kernel/random/uuid ]; then
70
+ cat /proc/sys/kernel/random/uuid
71
+ else
72
+ python3 -c 'import uuid; print(uuid.uuid4())'
73
+ fi
74
+ }
75
+
76
+ [ -n "$ASSET_ID" ] || ASSET_ID="cloud-$(_uuid)"
77
+ # remote/identity.py:_is_identifier rejects whitespace, non-printable
78
+ # characters, and anything over 255 characters. A value this accepts but the
79
+ # runtime rejects is worse than no file: every session reads it, rejects it,
80
+ # and mints a fresh asset — the per-session proliferation this exists to stop.
81
+ case "$ASSET_ID" in
82
+ *[[:space:]]*|"") echo "✗ asset id must be non-empty with no whitespace" >&2; exit 2 ;;
83
+ esac
84
+ if [ "${#ASSET_ID}" -gt 255 ]; then
85
+ echo "✗ asset id must be 255 characters or fewer" >&2
86
+ exit 2
87
+ fi
88
+ if ! python3 -c 'import sys; sys.exit(0 if sys.argv[1].isprintable() else 1)' "$ASSET_ID"; then
89
+ echo "✗ asset id must contain no non-printable characters" >&2
90
+ exit 2
91
+ fi
92
+ # ---- the per-session boot hook -------------------------------------------
93
+ # Failing hard here: if the hook cannot be written, every session's SessionStart
94
+ # command points at a file that does not exist and the daemon never starts —
95
+ # a silent total failure worse than a setup that stops and says so.
96
+ BOOT="${ROOT}/usr/local/bin/stacktrace-boot.sh"
97
+ mkdir -p "$(dirname "$BOOT")" || { echo "✗ cannot create $(dirname "$BOOT")" >&2; exit 1; }
98
+ cat > "$BOOT" <<'BOOTSCRIPT' || { echo "✗ cannot write $BOOT" >&2; exit 1; }
99
+ #!/usr/bin/env bash
100
+ # Configure the remote and start the daemon for this cloud session. No-op
101
+ # anywhere else. The asset id is already on disk from setup, so this does not
102
+ # touch it: it only configures the token and starts the daemon, which resolves
103
+ # the persisted asset id and mints the telemetry id on its first event.
104
+ #
105
+ # `daemon run` is started detached, not through a supervisor: a cloud VM runs
106
+ # one session, so the daemon's lifetime is the VM's (ADR-0070). Where a host
107
+ # runs several sessions, ADR-0066's `daemon start` replaces this one line.
108
+ set -u
109
+ [ "${CLAUDE_CODE_REMOTE:-}" = "true" ] || exit 0
110
+ command -v stacktrace >/dev/null 2>&1 || exit 0
111
+ [ -n "${STACKTRACE_REMOTE_TOKEN:-}" ] || exit 0
112
+
113
+ STATE="$HOME/.local/state/stacktrace"
114
+ mkdir -p "$STATE" && chmod 700 "$STATE"
115
+
116
+ stacktrace remote configure \
117
+ --token "$STACKTRACE_REMOTE_TOKEN" \
118
+ --api-url "${STACKTRACE_REMOTE_API_URL:-https://api.stacktrace.ai}" >/dev/null 2>&1 || true
119
+
120
+ # Streams redirected because own_std_streams only re-points fd 1/2 on log
121
+ # rollover; without this the daemon inherits the hook's pipe to Claude Code.
122
+ if ! stacktrace daemon status 2>/dev/null | grep -q '^running: yes'; then
123
+ setsid nohup stacktrace daemon run </dev/null >>"$STATE/daemon.log" 2>&1 &
124
+ fi
125
+ exit 0
126
+ BOOTSCRIPT
127
+ chmod 755 "$BOOT" || { echo "✗ cannot chmod $BOOT" >&2; exit 1; }
128
+ echo "✓ $BOOT"
129
+
130
+ # ---- seed every home -----------------------------------------------------
131
+ # Every existing home is seeded because the future session user is unknown, so a
132
+ # home that exists but cannot be seeded is a failure, not a skip: the session
133
+ # might run as exactly that user and find no daemon. `failed` records any such
134
+ # home and fails the whole run at the end, covering the mixed-success case where
135
+ # one home seeds and another does not. A candidate that is not a directory (the
136
+ # `/home/*` glob with no match, a stray file) is a genuine skip, not a home.
137
+ seeded=0
138
+ failed=0
139
+ for home in "${ROOT}"/root "${ROOT}"/home/*; do
140
+ [ -d "$home" ] || continue
141
+
142
+ cfg="$home/.config/stacktrace"
143
+ if ! install -d -m 700 "$cfg" 2>/dev/null; then
144
+ echo "✗ cannot create $cfg — not seeding $home" >&2; failed=1; continue
145
+ fi
146
+
147
+ if ! printf '%s\n' "$ASSET_ID" > "$cfg/asset-id" 2>/dev/null; then
148
+ echo "✗ cannot write $cfg/asset-id — not seeding $home" >&2; failed=1; continue
149
+ fi
150
+ chmod 600 "$cfg/asset-id" 2>/dev/null || true
151
+
152
+ # Register the SessionStart hook, merging so any existing settings survive.
153
+ # The hook is the point of seeding a home — a home whose registration fails
154
+ # (an unwritable path, or a settings.json that is valid JSON of an unexpected
155
+ # shape) is a failure, not a skip, or setup would exit 0 having left that
156
+ # session with no daemon.
157
+ if ! python3 - "$home" <<'PY'
158
+ import json, pathlib, sys
159
+ home = pathlib.Path(sys.argv[1])
160
+ p = home / ".claude" / "settings.json"
161
+ command = "/usr/local/bin/stacktrace-boot.sh"
162
+ try:
163
+ p.parent.mkdir(parents=True, exist_ok=True)
164
+ if p.exists():
165
+ # Present but unreadable or not valid JSON: do not clobber a real file.
166
+ s = json.loads(p.read_text())
167
+ else:
168
+ s = {}
169
+ except Exception:
170
+ sys.exit(1)
171
+ # A settings.json Claude Code accepts is a JSON object with object-shaped
172
+ # `hooks` and a list-shaped `SessionStart`; anything else is a shape this must
173
+ # not silently rewrite, so it fails the home rather than guessing.
174
+ if not isinstance(s, dict):
175
+ sys.exit(1)
176
+ hooks = s.setdefault("hooks", {})
177
+ if not isinstance(hooks, dict):
178
+ sys.exit(1)
179
+ starts = hooks.setdefault("SessionStart", [])
180
+ if not isinstance(starts, list):
181
+ sys.exit(1)
182
+ # Dedup only against an entry that actually fires on the events we need. An
183
+ # existing entry carrying this command under a narrower matcher (say `compact`)
184
+ # does not start the daemon at session start, so it must not suppress the
185
+ # `startup|resume` registration. A matcher covers us when it names both events,
186
+ # or when it is absent/empty — Claude Code treats that as matching every event.
187
+ required = {"startup", "resume"}
188
+
189
+
190
+ def covers(entry):
191
+ matcher = entry.get("matcher")
192
+ if matcher is None or matcher == "":
193
+ return True
194
+ return required.issubset({t.strip() for t in str(matcher).split("|")})
195
+
196
+
197
+ already = any(
198
+ isinstance(e, dict)
199
+ and covers(e)
200
+ and any(isinstance(h, dict) and h.get("command") == command for h in e.get("hooks", []))
201
+ for e in starts
202
+ )
203
+ if not already:
204
+ starts.append({"matcher": "startup|resume",
205
+ "hooks": [{"type": "command", "command": command}]})
206
+ try:
207
+ p.write_text(json.dumps(s, indent=2) + "\n")
208
+ except Exception:
209
+ sys.exit(1)
210
+ PY
211
+ then
212
+ echo "✗ could not register the hook in $home — not seeding it" >&2
213
+ failed=1
214
+ continue
215
+ fi
216
+
217
+ # Root created these directories, so they are root-owned; give them to the
218
+ # home's owner or the session user cannot write its own config. The parent
219
+ # `.config` that `install -d` created is chowned as well as the leaf, since a
220
+ # root-owned .config blocks the user from creating any other config beneath it.
221
+ for d in "$home/.config" "$cfg" "$home/.claude"; do
222
+ [ -e "$d" ] && chown --reference="$home" "$d" 2>/dev/null || true
223
+ done
224
+ chown --reference="$home" "$cfg/asset-id" \
225
+ "$home/.claude/settings.json" 2>/dev/null || true
226
+
227
+ echo "✓ seeded $home"
228
+ seeded=$((seeded + 1))
229
+ done
230
+
231
+ if [ "$seeded" -eq 0 ]; then
232
+ echo "✗ no home directory was writable — every session will mint its own identity" >&2
233
+ exit 1
234
+ fi
235
+
236
+ if [ "$failed" -ne 0 ]; then
237
+ echo "✗ one or more home directories could not be seeded — a session running as" \
238
+ "that user would have no daemon, so failing the whole setup rather than" \
239
+ "accepting a snapshot that works for some users and not others" >&2
240
+ exit 1
241
+ fi
242
+
243
+ command -v stacktrace >/dev/null 2>&1 || echo "! stacktrace is not on PATH yet; the boot hook waits for it" >&2
244
+
245
+ echo "✓ asset id: $ASSET_ID"
246
+ echo "✓ homes seeded: $seeded"
247
+ exit 0
@@ -50,9 +50,9 @@ class _PrivateRotatingFileHandler(RotatingFileHandler):
50
50
 
51
51
  def doRollover(self) -> None:
52
52
  super().doRollover()
53
- # `ensure_daemon` pointed this process's stdout and stderr at the file
54
- # just renamed away. Follow it, or a crash traceback lands in the
55
- # backup and is deleted with it at the next rollover.
53
+ # The detached launcher points this process's stdout and stderr at
54
+ # the file just renamed away. Follow it, or a crash traceback lands in
55
+ # the backup and is deleted with it at the next rollover.
56
56
  if self._own_std_streams and self.stream is not None:
57
57
  sys.stdout.flush()
58
58
  sys.stderr.flush()
@@ -6,7 +6,11 @@ import json
6
6
  import os
7
7
  import sqlite3
8
8
  import sys
9
+ import time
10
+ from collections.abc import Callable
9
11
  from datetime import UTC, datetime
12
+ from pathlib import Path
13
+ from typing import TypeVar
10
14
 
11
15
  import click
12
16
 
@@ -15,13 +19,61 @@ from .paths import RuntimePaths
15
19
  from .presentation import render_finding
16
20
  from .store import FindingStore, SessionKey
17
21
 
22
+ _T = TypeVar("_T")
23
+
18
24
 
19
25
  @click.group()
20
26
  def daemon() -> None:
21
27
  """Manage automatic session detection."""
22
28
 
23
29
 
24
- @daemon.command("run", hidden=True)
30
+ #: Bounds how long `status`/`stop` retry a control request against a daemon
31
+ #: whose lock is held but that has not yet answered. The accept loop starts
32
+ #: before composition startup, but the lock is taken earlier than that
33
+ #: (version lookup, opening the store, binding the socket). A request in
34
+ #: that gap would otherwise see its control timeout expire with nobody yet
35
+ #: reading it. Comfortably larger than the gap so an ordinary start never
36
+ #: surfaces as unresponsive.
37
+ _STARTUP_GRACE_SECONDS = 10.0
38
+
39
+ #: How long `stop` waits for the singleton lock once shutdown has been
40
+ #: accepted. Its own budget, not shared with `_STARTUP_GRACE_SECONDS`:
41
+ #: retrying during startup must not eat the time the in-progress tick needs.
42
+ #: Accepting shutdown only sets the stop event. The tick — a scan, then
43
+ #: Fleet drain and BOM upload — runs to completion before the loop notices,
44
+ #: and the lock stays held through that work, the telemetry flush, and
45
+ #: `compositions.close()`. Fleet's HTTP client allows 30s per attempt and
46
+ #: three attempts (~95s for one call), and one tick can make several. A 10s
47
+ #: budget expired during an ordinary pass, so `stop` reported failure and
48
+ #: dropped the transition lock while the process was still shutting down;
49
+ #: the next `start` then saw `ALREADY_RUNNING`. 600s covers several retried
50
+ #: cloud calls plus the scan and the shutdown tail, and still gives up when
51
+ #: the process is stuck.
52
+ _STOP_TIMEOUT_SECONDS = 600.0
53
+
54
+
55
+ def _retry_while_starting(
56
+ lock_path: Path, call: Callable[[], _T], errors: tuple[type[Exception], ...]
57
+ ) -> _T:
58
+ """Retry a lifecycle control request while the daemon lock is held.
59
+
60
+ Treats an early failure as evidence of the startup gap described above,
61
+ not as proof the daemon is actually unresponsive -- as long as the lock
62
+ is still held, that gap closes on its own within a few seconds.
63
+ """
64
+ from .lifecycle import daemon_lock_held
65
+
66
+ deadline = time.monotonic() + _STARTUP_GRACE_SECONDS
67
+ while True:
68
+ try:
69
+ return call()
70
+ except errors:
71
+ if not daemon_lock_held(lock_path) or time.monotonic() >= deadline:
72
+ raise
73
+ time.sleep(0.1)
74
+
75
+
76
+ @daemon.command("run")
25
77
  # No `envvar=` on any of these. Click resolves an environment variable *into*
26
78
  # the parameter, so a value that came from the environment would arrive here
27
79
  # indistinguishable from one typed on the command line — and `daemon status`
@@ -53,6 +105,12 @@ def daemon() -> None:
53
105
  " that fallback: a finding still uploads immediately when the scan commits it."
54
106
  " Defaults to STACKTRACE_FLEET_INTERVAL, then 5m.",
55
107
  )
108
+ @click.option(
109
+ "--webhook-interval",
110
+ default=None,
111
+ help="How often to retry configured webhook delivery. 0 disables only the fallback;"
112
+ " new findings still wake delivery. Defaults to STACKTRACE_WEBHOOK_INTERVAL, then 1m.",
113
+ )
56
114
  @click.option(
57
115
  "--bom-interval",
58
116
  default=None,
@@ -64,8 +122,10 @@ def run_command(
64
122
  scan_since: str | None,
65
123
  agent_kinds: tuple[str, ...],
66
124
  fleet_interval: str | None,
125
+ webhook_interval: str | None,
67
126
  bom_interval: str | None,
68
127
  ) -> None:
128
+ """Run the daemon in the foreground for a host supervisor."""
69
129
  _, _, run_daemon = _runtime()
70
130
  paths = RuntimePaths.from_environment()
71
131
  try:
@@ -74,14 +134,21 @@ def run_command(
74
134
  scan_since=scan_since,
75
135
  agent_kinds=agent_kinds,
76
136
  fleet_interval=fleet_interval,
137
+ webhook_interval=webhook_interval,
77
138
  bom_interval=bom_interval,
78
139
  )
79
140
  _reject_unobservable_kinds(config)
80
141
  except ValueError as error:
81
142
  raise click.ClickException(str(error)) from error
82
- # The real daemon process: its stdout and stderr are the log file
83
- # (`ensure_daemon`), so they follow it across a rollover (ADR-0048).
84
- run_daemon(paths, config, own_std_streams=True)
143
+ from .lifecycle import DETACHED_ENV
144
+
145
+ status = run_daemon(
146
+ paths,
147
+ config,
148
+ own_std_streams=os.environ.get(DETACHED_ENV) == "1",
149
+ )
150
+ if status:
151
+ raise click.exceptions.Exit(status)
85
152
 
86
153
 
87
154
  def _reject_unobservable_kinds(config: DaemonConfig) -> None:
@@ -112,18 +179,30 @@ def status() -> None:
112
179
  grows means retention is not pruning; and a configuration without its source
113
180
  makes fleet-wide drift undebuggable (`docs/specs/daemon-scanning.md`).
114
181
  """
115
- daemon_is_running, _, _ = _runtime()
116
- from .runtime import last_error, read_run_state
182
+ from stacktrace_cli.build import version_label
117
183
 
118
- paths = RuntimePaths.from_environment()
119
- running = daemon_is_running(paths.socket)
120
- run_state = read_run_state(paths)
184
+ from .client import ProtocolError, request_status
185
+ from .lifecycle import daemon_lock_held
186
+ from .runtime import last_error
121
187
 
122
- click.echo(f"running: {'yes' if running else 'no'}")
123
- if isinstance(run_state, dict):
124
- click.echo(f"pid: {run_state.get('pid')}")
125
- click.echo(f"version: {run_state.get('version')}")
126
- click.echo(f"started: {run_state.get('started_at')}")
188
+ paths = RuntimePaths.from_environment()
189
+ live = None
190
+ try:
191
+ live = _retry_while_starting(
192
+ paths.lock,
193
+ lambda: request_status(paths.socket),
194
+ (ConnectionError, OSError, ProtocolError),
195
+ )
196
+ except (ConnectionError, OSError, ProtocolError):
197
+ pass
198
+ if live is None and daemon_lock_held(paths.lock):
199
+ click.echo("running: unresponsive")
200
+ else:
201
+ click.echo(f"running: {'yes' if live is not None else 'no'}")
202
+ if live is not None:
203
+ click.echo(f"version: {live.build}")
204
+ click.echo(f"installed: {version_label()}")
205
+ click.echo(f"started: {live.started_at}")
127
206
 
128
207
  try:
129
208
  # One `_store()` call apiece, all against the same file: if the first
@@ -145,9 +224,9 @@ def status() -> None:
145
224
  click.echo(f"last error: {last_error(paths.log) or 'none reported'}")
146
225
 
147
226
  try:
148
- config = _effective_config(run_state)
227
+ config = _effective_config(live.config if live is not None else None)
149
228
  except ValueError as exc:
150
- # No readable run-state means `resolve()`'s own environment fallback
229
+ # No live status means `resolve()`'s own environment fallback
151
230
  # ran, and its validation is exactly what could be *why* the daemon
152
231
  # never started -- the one diagnostic surface for that case must not
153
232
  # itself crash on it. `from_document` above already never raises for
@@ -159,7 +238,7 @@ def status() -> None:
159
238
  click.echo(f" {name}: {entry['value']} ({entry['source']})")
160
239
 
161
240
 
162
- def _effective_config(run_state: dict[str, object] | None) -> DaemonConfig:
241
+ def _effective_config(document: dict[str, object] | None) -> DaemonConfig:
163
242
  """What the daemon resolved, when it is there to say so.
164
243
 
165
244
  Falling back to this process's own environment is a worse answer and is
@@ -167,8 +246,8 @@ def _effective_config(run_state: dict[str, object] | None) -> DaemonConfig:
167
246
  printing nothing, because the common case for an operator running `status`
168
247
  is a daemon that is not running at all.
169
248
  """
170
- if isinstance(run_state, dict) and isinstance(run_state.get("config"), dict):
171
- return DaemonConfig.from_document(run_state["config"]) # type: ignore[arg-type]
249
+ if document is not None:
250
+ return DaemonConfig.from_document(document)
172
251
  return DaemonConfig.resolve()
173
252
 
174
253
 
@@ -238,6 +317,58 @@ def _database_line(paths: RuntimePaths) -> str:
238
317
  return f"{store.database_size()} bytes ({paths.state})"
239
318
 
240
319
 
320
+ @daemon.command()
321
+ def start() -> None:
322
+ """Start one detached daemon when no host service manager is available."""
323
+ _runtime()
324
+ from .lifecycle import StartOutcome, start_daemon, transition_lock
325
+
326
+ paths = RuntimePaths.from_environment()
327
+ try:
328
+ with transition_lock(paths.transition_lock):
329
+ outcome = start_daemon(paths)
330
+ except (OSError, RuntimeError) as error:
331
+ raise click.ClickException(f"could not start daemon: {error}") from error
332
+ if outcome is StartOutcome.ALREADY_RUNNING:
333
+ click.echo("Stacktrace daemon is already running.")
334
+ else:
335
+ click.echo("Stacktrace daemon started.")
336
+
337
+
338
+ @daemon.command()
339
+ def stop() -> None:
340
+ """Ask the running daemon to shut down cleanly."""
341
+ from .client import ProtocolError, request_shutdown
342
+ from .lifecycle import daemon_lock_held, transition_lock
343
+
344
+ paths = RuntimePaths.from_environment()
345
+ with transition_lock(paths.transition_lock):
346
+ if not daemon_lock_held(paths.lock):
347
+ click.echo("Stacktrace daemon is not running.")
348
+ return
349
+ try:
350
+ _retry_while_starting(
351
+ paths.lock,
352
+ lambda: request_shutdown(paths.socket),
353
+ (ConnectionError, OSError, ProtocolError),
354
+ )
355
+ except (ConnectionError, OSError, ProtocolError) as error:
356
+ if not daemon_lock_held(paths.lock):
357
+ click.echo("Stacktrace daemon stopped.")
358
+ return
359
+ raise click.ClickException(
360
+ "daemon is running but unresponsive; restart it through its host "
361
+ "supervisor or use OS process tools"
362
+ ) from error
363
+ deadline = time.monotonic() + _STOP_TIMEOUT_SECONDS
364
+ while time.monotonic() < deadline:
365
+ if not daemon_lock_held(paths.lock):
366
+ click.echo("Stacktrace daemon stopped.")
367
+ return
368
+ time.sleep(0.1)
369
+ raise click.ClickException(f"daemon did not stop; see {paths.log}")
370
+
371
+
241
372
  @daemon.command()
242
373
  @click.option(
243
374
  "--agent-kind",