stacktrace-cli 0.5.1__py3-none-any.whl → 0.5.2__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- stacktrace_cli/__init__.py +1 -1
- stacktrace_cli/cli.py +2 -0
- stacktrace_cli/cloud/__init__.py +39 -0
- stacktrace_cli/cloud/setup.sh +247 -0
- stacktrace_cli/daemon/activity_log.py +3 -3
- stacktrace_cli/daemon/cli.py +150 -19
- stacktrace_cli/daemon/client.py +134 -26
- stacktrace_cli/daemon/config.py +14 -5
- stacktrace_cli/daemon/lifecycle.py +172 -0
- stacktrace_cli/daemon/paths.py +3 -6
- stacktrace_cli/daemon/protocol.py +20 -1
- stacktrace_cli/daemon/runtime.py +119 -98
- stacktrace_cli/daemon/server.py +78 -0
- stacktrace_cli/daemon/service.py +32 -17
- stacktrace_cli/daemon/store.py +36 -3
- stacktrace_cli/daemon/webhook.py +508 -0
- stacktrace_cli/detector/render.py +8 -4
- stacktrace_cli/monitor/site/app.js +1 -1
- stacktrace_cli/monitor/site/index.html +3 -4
- stacktrace_cli/telemetry/events.py +3 -4
- stacktrace_cli/webhook/__init__.py +1 -0
- stacktrace_cli/webhook/cli.py +169 -0
- stacktrace_cli/webhook/config.py +193 -0
- {stacktrace_cli-0.5.1.dist-info → stacktrace_cli-0.5.2.dist-info}/METADATA +30 -2
- {stacktrace_cli-0.5.1.dist-info → stacktrace_cli-0.5.2.dist-info}/RECORD +27 -21
- {stacktrace_cli-0.5.1.dist-info → stacktrace_cli-0.5.2.dist-info}/entry_points.txt +1 -0
- stacktrace_cli/daemon/version.py +0 -49
- {stacktrace_cli-0.5.1.dist-info → stacktrace_cli-0.5.2.dist-info}/WHEEL +0 -0
stacktrace_cli/__init__.py
CHANGED
stacktrace_cli/cli.py
CHANGED
|
@@ -40,6 +40,7 @@ from .sessions.access import collect_sessions
|
|
|
40
40
|
from .sessions.render import render_json, render_text
|
|
41
41
|
from .telemetry import emit_error
|
|
42
42
|
from .telemetry.cli import telemetry as telemetry_cmd
|
|
43
|
+
from .webhook.cli import webhook as webhook_cmd
|
|
43
44
|
|
|
44
45
|
PASSTHROUGH: Final = ("scan", "bom", "policy")
|
|
45
46
|
|
|
@@ -586,3 +587,4 @@ main.add_command(monitor)
|
|
|
586
587
|
main.add_command(daemon_cmd, "daemon")
|
|
587
588
|
main.add_command(findings)
|
|
588
589
|
main.add_command(telemetry_cmd, "telemetry")
|
|
590
|
+
main.add_command(webhook_cmd, "webhook")
|
|
@@ -0,0 +1,39 @@
|
|
|
1
|
+
"""One entry point for a Claude Code cloud environment's setup script.
|
|
2
|
+
|
|
3
|
+
A cloud environment's setup script runs once, as root, before the filesystem is
|
|
4
|
+
snapshotted and reused by every later session. What a session needs that is not
|
|
5
|
+
the CLI itself — the per-session boot hook and the asset identity — has to be
|
|
6
|
+
laid down there so it lands inside the snapshot.
|
|
7
|
+
|
|
8
|
+
Rather than paste that into the setup-script box, a setup script installs the
|
|
9
|
+
CLI and runs this:
|
|
10
|
+
|
|
11
|
+
UV_TOOL_DIR=/usr/local/share/uv-tools UV_TOOL_BIN_DIR=/usr/local/bin \
|
|
12
|
+
uv tool install stacktrace-cli
|
|
13
|
+
stacktrace-cloud-setup
|
|
14
|
+
|
|
15
|
+
Both variables matter: setup runs as root, and `uv tool install`'s root
|
|
16
|
+
defaults put the tool's environment under `/root/.local/share/uv/tools` and the
|
|
17
|
+
linked executable under `/root/.local/bin` — both inside `/root`, which a later
|
|
18
|
+
non-root session user cannot traverse. `UV_TOOL_BIN_DIR` alone moves the
|
|
19
|
+
symlink but not the environment it points into, so `command -v stacktrace` would
|
|
20
|
+
find the binary while running it still fails; both must move.
|
|
21
|
+
|
|
22
|
+
The work itself is `setup.sh`, shipped beside this module so it travels in the
|
|
23
|
+
wheel. This shim only locates it and hands off, so the logic stays one auditable
|
|
24
|
+
bash file rather than being reimplemented in Python.
|
|
25
|
+
"""
|
|
26
|
+
|
|
27
|
+
from __future__ import annotations
|
|
28
|
+
|
|
29
|
+
import os
|
|
30
|
+
import sys
|
|
31
|
+
from pathlib import Path
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def main() -> None:
|
|
35
|
+
script = Path(__file__).with_name("setup.sh")
|
|
36
|
+
# execvp so the process *becomes* bash: its exit status is the setup
|
|
37
|
+
# script's, which is what the environment reads to decide the session may
|
|
38
|
+
# start, and there is no Python frame left to swallow a signal.
|
|
39
|
+
os.execvp("bash", ["bash", str(script), *sys.argv[1:]])
|
|
@@ -0,0 +1,247 @@
|
|
|
1
|
+
#!/usr/bin/env bash
|
|
2
|
+
# Lay down everything a Claude Code cloud session needs, at environment setup.
|
|
3
|
+
#
|
|
4
|
+
# Run once by the environment's setup script, as root, before the filesystem is
|
|
5
|
+
# snapshotted. Two things, both of which must be inside that snapshot because
|
|
6
|
+
# the setup script does not run again for later sessions:
|
|
7
|
+
#
|
|
8
|
+
# 1. /usr/local/bin/stacktrace-boot.sh — the per-session hook that configures
|
|
9
|
+
# the remote and starts the daemon. Registered as a SessionStart hook in
|
|
10
|
+
# every home so it fires whichever user the session runs as.
|
|
11
|
+
# 2. The asset identity, in every home's ~/.config/stacktrace/asset-id, so
|
|
12
|
+
# `stacktrace remote sync` resolves it at the persisted rung (ADR-0065)
|
|
13
|
+
# instead of minting a fresh asset per session.
|
|
14
|
+
#
|
|
15
|
+
# The telemetry install id is deliberately NOT seeded here. Seeding it would
|
|
16
|
+
# make `read_install_id()` non-None before any event runs, so the one-time
|
|
17
|
+
# `installed` event (which `emit()` synthesizes only when the id is absent)
|
|
18
|
+
# would never fire. Left unseeded, the daemon's first in-session event mints the
|
|
19
|
+
# id and fires `installed` once per VM, correctly attributed `remote=true`
|
|
20
|
+
# because it runs inside the session. Counting one install per VM rather than
|
|
21
|
+
# per environment is ADR-0043's documented cloud tradeoff; per-environment is
|
|
22
|
+
# not cleanly achievable for an ephemeral VM (a setup-time emit would carry the
|
|
23
|
+
# wrong `remote` value and could not flush from a one-shot process).
|
|
24
|
+
#
|
|
25
|
+
# Every home rather than $HOME: setup runs as root and the user a session runs
|
|
26
|
+
# as is not documented, so writing all of them removes the guess. The asset id
|
|
27
|
+
# is unconditional — each environment instance gets its own, and a leftover from
|
|
28
|
+
# a previous snapshot must not win.
|
|
29
|
+
#
|
|
30
|
+
# Scope (ADR-0070): this seeds the snapshot and starts the daemon per session.
|
|
31
|
+
# It does not own daemon lifetime across a supervisor — that is ADR-0066, and a
|
|
32
|
+
# cloud VM is one session per VM, so the daemon dying with the VM is correct.
|
|
33
|
+
# The environment this targets is a fresh, single-tenant snapshot: it does not
|
|
34
|
+
# defend against adversarial or pre-existing symlinks and special files in a
|
|
35
|
+
# home, because that home does not exist until this run creates it.
|
|
36
|
+
#
|
|
37
|
+
# Options (all optional):
|
|
38
|
+
# --asset-id VALUE use this asset id instead of minting one; also taken
|
|
39
|
+
# from STACKTRACE_ASSET_EXTERNAL_ID in the environment
|
|
40
|
+
# --root DIR treat DIR as / (for testing)
|
|
41
|
+
#
|
|
42
|
+
# The remote token and API URL are read by the boot hook at session time from
|
|
43
|
+
# STACKTRACE_REMOTE_TOKEN and STACKTRACE_REMOTE_API_URL, not here: they belong
|
|
44
|
+
# to the environment's variables, not to the snapshot this seeds.
|
|
45
|
+
set -uo pipefail
|
|
46
|
+
|
|
47
|
+
ASSET_ID="${STACKTRACE_ASSET_EXTERNAL_ID:-}"
|
|
48
|
+
ROOT=""
|
|
49
|
+
|
|
50
|
+
while [ $# -gt 0 ]; do
|
|
51
|
+
case "$1" in
|
|
52
|
+
# `$# -ge 2` before the shift: without it, `--root` with no operand leaves
|
|
53
|
+
# the positional parameters unchanged under `set -u` with no `set -e`, and
|
|
54
|
+
# the loop spins forever — a malformed invocation would hang setup.
|
|
55
|
+
--asset-id) [ $# -ge 2 ] || { echo "✗ --asset-id needs a value" >&2; exit 2; }; ASSET_ID="$2"; shift 2 ;;
|
|
56
|
+
--root) [ $# -ge 2 ] || { echo "✗ --root needs a value" >&2; exit 2; }; ROOT="$2"; shift 2 ;;
|
|
57
|
+
-h|--help) sed -n '2,44p' "$0" | sed 's/^# \{0,1\}//'; exit 0 ;;
|
|
58
|
+
*) echo "✗ unknown argument: $1" >&2; exit 2 ;;
|
|
59
|
+
esac
|
|
60
|
+
done
|
|
61
|
+
|
|
62
|
+
LOG="${ROOT}/var/log/stacktrace-cloud-setup.log"
|
|
63
|
+
mkdir -p "$(dirname "$LOG")" 2>/dev/null || true
|
|
64
|
+
exec > >(tee -a "$LOG") 2>&1
|
|
65
|
+
echo "=== stacktrace cloud setup $(date -u +%FT%TZ) ==="
|
|
66
|
+
|
|
67
|
+
# ---- identities ----------------------------------------------------------
|
|
68
|
+
_uuid() {
|
|
69
|
+
if [ -r /proc/sys/kernel/random/uuid ]; then
|
|
70
|
+
cat /proc/sys/kernel/random/uuid
|
|
71
|
+
else
|
|
72
|
+
python3 -c 'import uuid; print(uuid.uuid4())'
|
|
73
|
+
fi
|
|
74
|
+
}
|
|
75
|
+
|
|
76
|
+
[ -n "$ASSET_ID" ] || ASSET_ID="cloud-$(_uuid)"
|
|
77
|
+
# remote/identity.py:_is_identifier rejects whitespace, non-printable
|
|
78
|
+
# characters, and anything over 255 characters. A value this accepts but the
|
|
79
|
+
# runtime rejects is worse than no file: every session reads it, rejects it,
|
|
80
|
+
# and mints a fresh asset — the per-session proliferation this exists to stop.
|
|
81
|
+
case "$ASSET_ID" in
|
|
82
|
+
*[[:space:]]*|"") echo "✗ asset id must be non-empty with no whitespace" >&2; exit 2 ;;
|
|
83
|
+
esac
|
|
84
|
+
if [ "${#ASSET_ID}" -gt 255 ]; then
|
|
85
|
+
echo "✗ asset id must be 255 characters or fewer" >&2
|
|
86
|
+
exit 2
|
|
87
|
+
fi
|
|
88
|
+
if ! python3 -c 'import sys; sys.exit(0 if sys.argv[1].isprintable() else 1)' "$ASSET_ID"; then
|
|
89
|
+
echo "✗ asset id must contain no non-printable characters" >&2
|
|
90
|
+
exit 2
|
|
91
|
+
fi
|
|
92
|
+
# ---- the per-session boot hook -------------------------------------------
|
|
93
|
+
# Failing hard here: if the hook cannot be written, every session's SessionStart
|
|
94
|
+
# command points at a file that does not exist and the daemon never starts —
|
|
95
|
+
# a silent total failure worse than a setup that stops and says so.
|
|
96
|
+
BOOT="${ROOT}/usr/local/bin/stacktrace-boot.sh"
|
|
97
|
+
mkdir -p "$(dirname "$BOOT")" || { echo "✗ cannot create $(dirname "$BOOT")" >&2; exit 1; }
|
|
98
|
+
cat > "$BOOT" <<'BOOTSCRIPT' || { echo "✗ cannot write $BOOT" >&2; exit 1; }
|
|
99
|
+
#!/usr/bin/env bash
|
|
100
|
+
# Configure the remote and start the daemon for this cloud session. No-op
|
|
101
|
+
# anywhere else. The asset id is already on disk from setup, so this does not
|
|
102
|
+
# touch it: it only configures the token and starts the daemon, which resolves
|
|
103
|
+
# the persisted asset id and mints the telemetry id on its first event.
|
|
104
|
+
#
|
|
105
|
+
# `daemon run` is started detached, not through a supervisor: a cloud VM runs
|
|
106
|
+
# one session, so the daemon's lifetime is the VM's (ADR-0070). Where a host
|
|
107
|
+
# runs several sessions, ADR-0066's `daemon start` replaces this one line.
|
|
108
|
+
set -u
|
|
109
|
+
[ "${CLAUDE_CODE_REMOTE:-}" = "true" ] || exit 0
|
|
110
|
+
command -v stacktrace >/dev/null 2>&1 || exit 0
|
|
111
|
+
[ -n "${STACKTRACE_REMOTE_TOKEN:-}" ] || exit 0
|
|
112
|
+
|
|
113
|
+
STATE="$HOME/.local/state/stacktrace"
|
|
114
|
+
mkdir -p "$STATE" && chmod 700 "$STATE"
|
|
115
|
+
|
|
116
|
+
stacktrace remote configure \
|
|
117
|
+
--token "$STACKTRACE_REMOTE_TOKEN" \
|
|
118
|
+
--api-url "${STACKTRACE_REMOTE_API_URL:-https://api.stacktrace.ai}" >/dev/null 2>&1 || true
|
|
119
|
+
|
|
120
|
+
# Streams redirected because own_std_streams only re-points fd 1/2 on log
|
|
121
|
+
# rollover; without this the daemon inherits the hook's pipe to Claude Code.
|
|
122
|
+
if ! stacktrace daemon status 2>/dev/null | grep -q '^running: yes'; then
|
|
123
|
+
setsid nohup stacktrace daemon run </dev/null >>"$STATE/daemon.log" 2>&1 &
|
|
124
|
+
fi
|
|
125
|
+
exit 0
|
|
126
|
+
BOOTSCRIPT
|
|
127
|
+
chmod 755 "$BOOT" || { echo "✗ cannot chmod $BOOT" >&2; exit 1; }
|
|
128
|
+
echo "✓ $BOOT"
|
|
129
|
+
|
|
130
|
+
# ---- seed every home -----------------------------------------------------
|
|
131
|
+
# Every existing home is seeded because the future session user is unknown, so a
|
|
132
|
+
# home that exists but cannot be seeded is a failure, not a skip: the session
|
|
133
|
+
# might run as exactly that user and find no daemon. `failed` records any such
|
|
134
|
+
# home and fails the whole run at the end, covering the mixed-success case where
|
|
135
|
+
# one home seeds and another does not. A candidate that is not a directory (the
|
|
136
|
+
# `/home/*` glob with no match, a stray file) is a genuine skip, not a home.
|
|
137
|
+
seeded=0
|
|
138
|
+
failed=0
|
|
139
|
+
for home in "${ROOT}"/root "${ROOT}"/home/*; do
|
|
140
|
+
[ -d "$home" ] || continue
|
|
141
|
+
|
|
142
|
+
cfg="$home/.config/stacktrace"
|
|
143
|
+
if ! install -d -m 700 "$cfg" 2>/dev/null; then
|
|
144
|
+
echo "✗ cannot create $cfg — not seeding $home" >&2; failed=1; continue
|
|
145
|
+
fi
|
|
146
|
+
|
|
147
|
+
if ! printf '%s\n' "$ASSET_ID" > "$cfg/asset-id" 2>/dev/null; then
|
|
148
|
+
echo "✗ cannot write $cfg/asset-id — not seeding $home" >&2; failed=1; continue
|
|
149
|
+
fi
|
|
150
|
+
chmod 600 "$cfg/asset-id" 2>/dev/null || true
|
|
151
|
+
|
|
152
|
+
# Register the SessionStart hook, merging so any existing settings survive.
|
|
153
|
+
# The hook is the point of seeding a home — a home whose registration fails
|
|
154
|
+
# (an unwritable path, or a settings.json that is valid JSON of an unexpected
|
|
155
|
+
# shape) is a failure, not a skip, or setup would exit 0 having left that
|
|
156
|
+
# session with no daemon.
|
|
157
|
+
if ! python3 - "$home" <<'PY'
|
|
158
|
+
import json, pathlib, sys
|
|
159
|
+
home = pathlib.Path(sys.argv[1])
|
|
160
|
+
p = home / ".claude" / "settings.json"
|
|
161
|
+
command = "/usr/local/bin/stacktrace-boot.sh"
|
|
162
|
+
try:
|
|
163
|
+
p.parent.mkdir(parents=True, exist_ok=True)
|
|
164
|
+
if p.exists():
|
|
165
|
+
# Present but unreadable or not valid JSON: do not clobber a real file.
|
|
166
|
+
s = json.loads(p.read_text())
|
|
167
|
+
else:
|
|
168
|
+
s = {}
|
|
169
|
+
except Exception:
|
|
170
|
+
sys.exit(1)
|
|
171
|
+
# A settings.json Claude Code accepts is a JSON object with object-shaped
|
|
172
|
+
# `hooks` and a list-shaped `SessionStart`; anything else is a shape this must
|
|
173
|
+
# not silently rewrite, so it fails the home rather than guessing.
|
|
174
|
+
if not isinstance(s, dict):
|
|
175
|
+
sys.exit(1)
|
|
176
|
+
hooks = s.setdefault("hooks", {})
|
|
177
|
+
if not isinstance(hooks, dict):
|
|
178
|
+
sys.exit(1)
|
|
179
|
+
starts = hooks.setdefault("SessionStart", [])
|
|
180
|
+
if not isinstance(starts, list):
|
|
181
|
+
sys.exit(1)
|
|
182
|
+
# Dedup only against an entry that actually fires on the events we need. An
|
|
183
|
+
# existing entry carrying this command under a narrower matcher (say `compact`)
|
|
184
|
+
# does not start the daemon at session start, so it must not suppress the
|
|
185
|
+
# `startup|resume` registration. A matcher covers us when it names both events,
|
|
186
|
+
# or when it is absent/empty — Claude Code treats that as matching every event.
|
|
187
|
+
required = {"startup", "resume"}
|
|
188
|
+
|
|
189
|
+
|
|
190
|
+
def covers(entry):
|
|
191
|
+
matcher = entry.get("matcher")
|
|
192
|
+
if matcher is None or matcher == "":
|
|
193
|
+
return True
|
|
194
|
+
return required.issubset({t.strip() for t in str(matcher).split("|")})
|
|
195
|
+
|
|
196
|
+
|
|
197
|
+
already = any(
|
|
198
|
+
isinstance(e, dict)
|
|
199
|
+
and covers(e)
|
|
200
|
+
and any(isinstance(h, dict) and h.get("command") == command for h in e.get("hooks", []))
|
|
201
|
+
for e in starts
|
|
202
|
+
)
|
|
203
|
+
if not already:
|
|
204
|
+
starts.append({"matcher": "startup|resume",
|
|
205
|
+
"hooks": [{"type": "command", "command": command}]})
|
|
206
|
+
try:
|
|
207
|
+
p.write_text(json.dumps(s, indent=2) + "\n")
|
|
208
|
+
except Exception:
|
|
209
|
+
sys.exit(1)
|
|
210
|
+
PY
|
|
211
|
+
then
|
|
212
|
+
echo "✗ could not register the hook in $home — not seeding it" >&2
|
|
213
|
+
failed=1
|
|
214
|
+
continue
|
|
215
|
+
fi
|
|
216
|
+
|
|
217
|
+
# Root created these directories, so they are root-owned; give them to the
|
|
218
|
+
# home's owner or the session user cannot write its own config. The parent
|
|
219
|
+
# `.config` that `install -d` created is chowned as well as the leaf, since a
|
|
220
|
+
# root-owned .config blocks the user from creating any other config beneath it.
|
|
221
|
+
for d in "$home/.config" "$cfg" "$home/.claude"; do
|
|
222
|
+
[ -e "$d" ] && chown --reference="$home" "$d" 2>/dev/null || true
|
|
223
|
+
done
|
|
224
|
+
chown --reference="$home" "$cfg/asset-id" \
|
|
225
|
+
"$home/.claude/settings.json" 2>/dev/null || true
|
|
226
|
+
|
|
227
|
+
echo "✓ seeded $home"
|
|
228
|
+
seeded=$((seeded + 1))
|
|
229
|
+
done
|
|
230
|
+
|
|
231
|
+
if [ "$seeded" -eq 0 ]; then
|
|
232
|
+
echo "✗ no home directory was writable — every session will mint its own identity" >&2
|
|
233
|
+
exit 1
|
|
234
|
+
fi
|
|
235
|
+
|
|
236
|
+
if [ "$failed" -ne 0 ]; then
|
|
237
|
+
echo "✗ one or more home directories could not be seeded — a session running as" \
|
|
238
|
+
"that user would have no daemon, so failing the whole setup rather than" \
|
|
239
|
+
"accepting a snapshot that works for some users and not others" >&2
|
|
240
|
+
exit 1
|
|
241
|
+
fi
|
|
242
|
+
|
|
243
|
+
command -v stacktrace >/dev/null 2>&1 || echo "! stacktrace is not on PATH yet; the boot hook waits for it" >&2
|
|
244
|
+
|
|
245
|
+
echo "✓ asset id: $ASSET_ID"
|
|
246
|
+
echo "✓ homes seeded: $seeded"
|
|
247
|
+
exit 0
|
|
@@ -50,9 +50,9 @@ class _PrivateRotatingFileHandler(RotatingFileHandler):
|
|
|
50
50
|
|
|
51
51
|
def doRollover(self) -> None:
|
|
52
52
|
super().doRollover()
|
|
53
|
-
#
|
|
54
|
-
# just renamed away. Follow it, or a crash traceback lands in
|
|
55
|
-
# backup and is deleted with it at the next rollover.
|
|
53
|
+
# The detached launcher points this process's stdout and stderr at
|
|
54
|
+
# the file just renamed away. Follow it, or a crash traceback lands in
|
|
55
|
+
# the backup and is deleted with it at the next rollover.
|
|
56
56
|
if self._own_std_streams and self.stream is not None:
|
|
57
57
|
sys.stdout.flush()
|
|
58
58
|
sys.stderr.flush()
|
stacktrace_cli/daemon/cli.py
CHANGED
|
@@ -6,7 +6,11 @@ import json
|
|
|
6
6
|
import os
|
|
7
7
|
import sqlite3
|
|
8
8
|
import sys
|
|
9
|
+
import time
|
|
10
|
+
from collections.abc import Callable
|
|
9
11
|
from datetime import UTC, datetime
|
|
12
|
+
from pathlib import Path
|
|
13
|
+
from typing import TypeVar
|
|
10
14
|
|
|
11
15
|
import click
|
|
12
16
|
|
|
@@ -15,13 +19,61 @@ from .paths import RuntimePaths
|
|
|
15
19
|
from .presentation import render_finding
|
|
16
20
|
from .store import FindingStore, SessionKey
|
|
17
21
|
|
|
22
|
+
_T = TypeVar("_T")
|
|
23
|
+
|
|
18
24
|
|
|
19
25
|
@click.group()
|
|
20
26
|
def daemon() -> None:
|
|
21
27
|
"""Manage automatic session detection."""
|
|
22
28
|
|
|
23
29
|
|
|
24
|
-
|
|
30
|
+
#: Bounds how long `status`/`stop` retry a control request against a daemon
|
|
31
|
+
#: whose lock is held but that has not yet answered. The accept loop starts
|
|
32
|
+
#: before composition startup, but the lock is taken earlier than that
|
|
33
|
+
#: (version lookup, opening the store, binding the socket). A request in
|
|
34
|
+
#: that gap would otherwise see its control timeout expire with nobody yet
|
|
35
|
+
#: reading it. Comfortably larger than the gap so an ordinary start never
|
|
36
|
+
#: surfaces as unresponsive.
|
|
37
|
+
_STARTUP_GRACE_SECONDS = 10.0
|
|
38
|
+
|
|
39
|
+
#: How long `stop` waits for the singleton lock once shutdown has been
|
|
40
|
+
#: accepted. Its own budget, not shared with `_STARTUP_GRACE_SECONDS`:
|
|
41
|
+
#: retrying during startup must not eat the time the in-progress tick needs.
|
|
42
|
+
#: Accepting shutdown only sets the stop event. The tick — a scan, then
|
|
43
|
+
#: Fleet drain and BOM upload — runs to completion before the loop notices,
|
|
44
|
+
#: and the lock stays held through that work, the telemetry flush, and
|
|
45
|
+
#: `compositions.close()`. Fleet's HTTP client allows 30s per attempt and
|
|
46
|
+
#: three attempts (~95s for one call), and one tick can make several. A 10s
|
|
47
|
+
#: budget expired during an ordinary pass, so `stop` reported failure and
|
|
48
|
+
#: dropped the transition lock while the process was still shutting down;
|
|
49
|
+
#: the next `start` then saw `ALREADY_RUNNING`. 600s covers several retried
|
|
50
|
+
#: cloud calls plus the scan and the shutdown tail, and still gives up when
|
|
51
|
+
#: the process is stuck.
|
|
52
|
+
_STOP_TIMEOUT_SECONDS = 600.0
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def _retry_while_starting(
|
|
56
|
+
lock_path: Path, call: Callable[[], _T], errors: tuple[type[Exception], ...]
|
|
57
|
+
) -> _T:
|
|
58
|
+
"""Retry a lifecycle control request while the daemon lock is held.
|
|
59
|
+
|
|
60
|
+
Treats an early failure as evidence of the startup gap described above,
|
|
61
|
+
not as proof the daemon is actually unresponsive -- as long as the lock
|
|
62
|
+
is still held, that gap closes on its own within a few seconds.
|
|
63
|
+
"""
|
|
64
|
+
from .lifecycle import daemon_lock_held
|
|
65
|
+
|
|
66
|
+
deadline = time.monotonic() + _STARTUP_GRACE_SECONDS
|
|
67
|
+
while True:
|
|
68
|
+
try:
|
|
69
|
+
return call()
|
|
70
|
+
except errors:
|
|
71
|
+
if not daemon_lock_held(lock_path) or time.monotonic() >= deadline:
|
|
72
|
+
raise
|
|
73
|
+
time.sleep(0.1)
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
@daemon.command("run")
|
|
25
77
|
# No `envvar=` on any of these. Click resolves an environment variable *into*
|
|
26
78
|
# the parameter, so a value that came from the environment would arrive here
|
|
27
79
|
# indistinguishable from one typed on the command line — and `daemon status`
|
|
@@ -53,6 +105,12 @@ def daemon() -> None:
|
|
|
53
105
|
" that fallback: a finding still uploads immediately when the scan commits it."
|
|
54
106
|
" Defaults to STACKTRACE_FLEET_INTERVAL, then 5m.",
|
|
55
107
|
)
|
|
108
|
+
@click.option(
|
|
109
|
+
"--webhook-interval",
|
|
110
|
+
default=None,
|
|
111
|
+
help="How often to retry configured webhook delivery. 0 disables only the fallback;"
|
|
112
|
+
" new findings still wake delivery. Defaults to STACKTRACE_WEBHOOK_INTERVAL, then 1m.",
|
|
113
|
+
)
|
|
56
114
|
@click.option(
|
|
57
115
|
"--bom-interval",
|
|
58
116
|
default=None,
|
|
@@ -64,8 +122,10 @@ def run_command(
|
|
|
64
122
|
scan_since: str | None,
|
|
65
123
|
agent_kinds: tuple[str, ...],
|
|
66
124
|
fleet_interval: str | None,
|
|
125
|
+
webhook_interval: str | None,
|
|
67
126
|
bom_interval: str | None,
|
|
68
127
|
) -> None:
|
|
128
|
+
"""Run the daemon in the foreground for a host supervisor."""
|
|
69
129
|
_, _, run_daemon = _runtime()
|
|
70
130
|
paths = RuntimePaths.from_environment()
|
|
71
131
|
try:
|
|
@@ -74,14 +134,21 @@ def run_command(
|
|
|
74
134
|
scan_since=scan_since,
|
|
75
135
|
agent_kinds=agent_kinds,
|
|
76
136
|
fleet_interval=fleet_interval,
|
|
137
|
+
webhook_interval=webhook_interval,
|
|
77
138
|
bom_interval=bom_interval,
|
|
78
139
|
)
|
|
79
140
|
_reject_unobservable_kinds(config)
|
|
80
141
|
except ValueError as error:
|
|
81
142
|
raise click.ClickException(str(error)) from error
|
|
82
|
-
|
|
83
|
-
|
|
84
|
-
run_daemon(
|
|
143
|
+
from .lifecycle import DETACHED_ENV
|
|
144
|
+
|
|
145
|
+
status = run_daemon(
|
|
146
|
+
paths,
|
|
147
|
+
config,
|
|
148
|
+
own_std_streams=os.environ.get(DETACHED_ENV) == "1",
|
|
149
|
+
)
|
|
150
|
+
if status:
|
|
151
|
+
raise click.exceptions.Exit(status)
|
|
85
152
|
|
|
86
153
|
|
|
87
154
|
def _reject_unobservable_kinds(config: DaemonConfig) -> None:
|
|
@@ -112,18 +179,30 @@ def status() -> None:
|
|
|
112
179
|
grows means retention is not pruning; and a configuration without its source
|
|
113
180
|
makes fleet-wide drift undebuggable (`docs/specs/daemon-scanning.md`).
|
|
114
181
|
"""
|
|
115
|
-
|
|
116
|
-
from .runtime import last_error, read_run_state
|
|
182
|
+
from stacktrace_cli.build import version_label
|
|
117
183
|
|
|
118
|
-
|
|
119
|
-
|
|
120
|
-
|
|
184
|
+
from .client import ProtocolError, request_status
|
|
185
|
+
from .lifecycle import daemon_lock_held
|
|
186
|
+
from .runtime import last_error
|
|
121
187
|
|
|
122
|
-
|
|
123
|
-
|
|
124
|
-
|
|
125
|
-
|
|
126
|
-
|
|
188
|
+
paths = RuntimePaths.from_environment()
|
|
189
|
+
live = None
|
|
190
|
+
try:
|
|
191
|
+
live = _retry_while_starting(
|
|
192
|
+
paths.lock,
|
|
193
|
+
lambda: request_status(paths.socket),
|
|
194
|
+
(ConnectionError, OSError, ProtocolError),
|
|
195
|
+
)
|
|
196
|
+
except (ConnectionError, OSError, ProtocolError):
|
|
197
|
+
pass
|
|
198
|
+
if live is None and daemon_lock_held(paths.lock):
|
|
199
|
+
click.echo("running: unresponsive")
|
|
200
|
+
else:
|
|
201
|
+
click.echo(f"running: {'yes' if live is not None else 'no'}")
|
|
202
|
+
if live is not None:
|
|
203
|
+
click.echo(f"version: {live.build}")
|
|
204
|
+
click.echo(f"installed: {version_label()}")
|
|
205
|
+
click.echo(f"started: {live.started_at}")
|
|
127
206
|
|
|
128
207
|
try:
|
|
129
208
|
# One `_store()` call apiece, all against the same file: if the first
|
|
@@ -145,9 +224,9 @@ def status() -> None:
|
|
|
145
224
|
click.echo(f"last error: {last_error(paths.log) or 'none reported'}")
|
|
146
225
|
|
|
147
226
|
try:
|
|
148
|
-
config = _effective_config(
|
|
227
|
+
config = _effective_config(live.config if live is not None else None)
|
|
149
228
|
except ValueError as exc:
|
|
150
|
-
# No
|
|
229
|
+
# No live status means `resolve()`'s own environment fallback
|
|
151
230
|
# ran, and its validation is exactly what could be *why* the daemon
|
|
152
231
|
# never started -- the one diagnostic surface for that case must not
|
|
153
232
|
# itself crash on it. `from_document` above already never raises for
|
|
@@ -159,7 +238,7 @@ def status() -> None:
|
|
|
159
238
|
click.echo(f" {name}: {entry['value']} ({entry['source']})")
|
|
160
239
|
|
|
161
240
|
|
|
162
|
-
def _effective_config(
|
|
241
|
+
def _effective_config(document: dict[str, object] | None) -> DaemonConfig:
|
|
163
242
|
"""What the daemon resolved, when it is there to say so.
|
|
164
243
|
|
|
165
244
|
Falling back to this process's own environment is a worse answer and is
|
|
@@ -167,8 +246,8 @@ def _effective_config(run_state: dict[str, object] | None) -> DaemonConfig:
|
|
|
167
246
|
printing nothing, because the common case for an operator running `status`
|
|
168
247
|
is a daemon that is not running at all.
|
|
169
248
|
"""
|
|
170
|
-
if
|
|
171
|
-
return DaemonConfig.from_document(
|
|
249
|
+
if document is not None:
|
|
250
|
+
return DaemonConfig.from_document(document)
|
|
172
251
|
return DaemonConfig.resolve()
|
|
173
252
|
|
|
174
253
|
|
|
@@ -238,6 +317,58 @@ def _database_line(paths: RuntimePaths) -> str:
|
|
|
238
317
|
return f"{store.database_size()} bytes ({paths.state})"
|
|
239
318
|
|
|
240
319
|
|
|
320
|
+
@daemon.command()
|
|
321
|
+
def start() -> None:
|
|
322
|
+
"""Start one detached daemon when no host service manager is available."""
|
|
323
|
+
_runtime()
|
|
324
|
+
from .lifecycle import StartOutcome, start_daemon, transition_lock
|
|
325
|
+
|
|
326
|
+
paths = RuntimePaths.from_environment()
|
|
327
|
+
try:
|
|
328
|
+
with transition_lock(paths.transition_lock):
|
|
329
|
+
outcome = start_daemon(paths)
|
|
330
|
+
except (OSError, RuntimeError) as error:
|
|
331
|
+
raise click.ClickException(f"could not start daemon: {error}") from error
|
|
332
|
+
if outcome is StartOutcome.ALREADY_RUNNING:
|
|
333
|
+
click.echo("Stacktrace daemon is already running.")
|
|
334
|
+
else:
|
|
335
|
+
click.echo("Stacktrace daemon started.")
|
|
336
|
+
|
|
337
|
+
|
|
338
|
+
@daemon.command()
|
|
339
|
+
def stop() -> None:
|
|
340
|
+
"""Ask the running daemon to shut down cleanly."""
|
|
341
|
+
from .client import ProtocolError, request_shutdown
|
|
342
|
+
from .lifecycle import daemon_lock_held, transition_lock
|
|
343
|
+
|
|
344
|
+
paths = RuntimePaths.from_environment()
|
|
345
|
+
with transition_lock(paths.transition_lock):
|
|
346
|
+
if not daemon_lock_held(paths.lock):
|
|
347
|
+
click.echo("Stacktrace daemon is not running.")
|
|
348
|
+
return
|
|
349
|
+
try:
|
|
350
|
+
_retry_while_starting(
|
|
351
|
+
paths.lock,
|
|
352
|
+
lambda: request_shutdown(paths.socket),
|
|
353
|
+
(ConnectionError, OSError, ProtocolError),
|
|
354
|
+
)
|
|
355
|
+
except (ConnectionError, OSError, ProtocolError) as error:
|
|
356
|
+
if not daemon_lock_held(paths.lock):
|
|
357
|
+
click.echo("Stacktrace daemon stopped.")
|
|
358
|
+
return
|
|
359
|
+
raise click.ClickException(
|
|
360
|
+
"daemon is running but unresponsive; restart it through its host "
|
|
361
|
+
"supervisor or use OS process tools"
|
|
362
|
+
) from error
|
|
363
|
+
deadline = time.monotonic() + _STOP_TIMEOUT_SECONDS
|
|
364
|
+
while time.monotonic() < deadline:
|
|
365
|
+
if not daemon_lock_held(paths.lock):
|
|
366
|
+
click.echo("Stacktrace daemon stopped.")
|
|
367
|
+
return
|
|
368
|
+
time.sleep(0.1)
|
|
369
|
+
raise click.ClickException(f"daemon did not stop; see {paths.log}")
|
|
370
|
+
|
|
371
|
+
|
|
241
372
|
@daemon.command()
|
|
242
373
|
@click.option(
|
|
243
374
|
"--agent-kind",
|