continuity-guard 0.5.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,14 @@
1
+ """Continuity Guard -- liveness and progress watchdog for long-running AI agent sessions.
2
+
3
+ from continuity_guard import guard
4
+
5
+ with guard("nightly-refactor", profile="overnight") as s:
6
+ for step in agent.run():
7
+ s.progress(step=step.name, tool=step.tool,
8
+ args=step.args, tokens=step.tokens)
9
+ """
10
+
11
+ from .client import Session, args_hash, guard
12
+
13
+ __all__ = ["guard", "Session", "args_hash", "__version__"]
14
+ __version__ = "0.5.0"
@@ -0,0 +1,230 @@
1
+ """Actions and notifications.
2
+
3
+ Kill authority lives here, and only here, and only locally. Three ways to get
4
+ termination wrong, all guarded against:
5
+
6
+ PID reuse -> pidfd where available, else PID+start_time verified as a pair
7
+ orphaned kids -> kill the process GROUP, not the process
8
+ no grace -> SIGTERM, wait, SIGKILL
9
+ """
10
+
11
+ from __future__ import annotations
12
+
13
+ import errno
14
+ import os
15
+ import signal
16
+ import subprocess
17
+ import time
18
+ import urllib.request
19
+
20
+
21
+ # ---------------------------------------------------------------- termination
22
+ # Can this platform give us a process start time at all? Decided once, from
23
+ # our own PID: if we cannot read our own, we cannot read anyone's.
24
+ def _probe_start_time_support() -> bool:
25
+ try:
26
+ with open(f"/proc/{os.getpid()}/stat", "rb") as fh:
27
+ data = fh.read()
28
+ int(data[data.rindex(b")") + 2:].split()[19])
29
+ return True
30
+ except Exception:
31
+ return False
32
+
33
+
34
+ def _start_time(pid: int) -> int:
35
+ try:
36
+ with open(f"/proc/{pid}/stat", "rb") as fh:
37
+ data = fh.read()
38
+ return int(data[data.rindex(b")") + 2 :].split()[19])
39
+ except Exception:
40
+ return 0
41
+
42
+
43
+ def _alive(pid: int) -> bool:
44
+ """Is this PID a live process -- not a zombie?
45
+
46
+ /proc/<pid> keeps existing after a process dies, right up until its parent
47
+ reaps it. An existence check alone therefore reads a corpse as alive. That
48
+ matters on the supervisor path: cg_supervise is the PARENT of the agent it
49
+ registers, so a terminated agent sits as a zombie until the supervisor's
50
+ wait() runs. The grace loop would spin the full window and then fire a
51
+ pointless SIGKILL at an already-dead process -- and report
52
+ "SIGKILL after 20s grace" when SIGTERM had in fact worked immediately.
53
+ """
54
+ try:
55
+ with open(f"/proc/{pid}/stat", "rb") as fh:
56
+ data = fh.read()
57
+ # comm (field 2) may contain spaces and parens: parse after the last ')'
58
+ return data[data.rindex(b")") + 2:].split()[0] != b"Z"
59
+ except (OSError, ValueError, IndexError):
60
+ return False
61
+
62
+
63
+ def _identity_ok(pid: int, expected_start: int) -> bool:
64
+ """A PID alone is not an identity. Recorded an hour ago it may belong to
65
+ something else entirely by now.
66
+
67
+ A missing start_time used to fall back to "is anything alive with this
68
+ PID", which is not an identity check at all -- it accepts every live
69
+ process on the box. Combined with a world-writable socket and peer-supplied
70
+ policy, that made an armed daemon a process-kill oracle: announce someone
71
+ else's PID with start_time 0, set a policy that terminates immediately, and
72
+ the daemon signals it. Termination now REQUIRES a verified start_time, and
73
+ an unverifiable identity is refused rather than assumed.
74
+ """
75
+ if pid <= 1:
76
+ return False
77
+ if expected_start:
78
+ return _start_time(pid) == expected_start
79
+ # No start_time was announced. This used to fall back to "is anything
80
+ # alive with this PID", which is not an identity check -- it accepts every
81
+ # live process on the box, and with peer-supplied policy that made an armed
82
+ # daemon a process-kill oracle. Where the platform CAN supply a start time,
83
+ # its absence now means the identity is unverifiable and the kill is
84
+ # refused. Where it cannot (no /proc: macOS, Windows), refusing outright
85
+ # would silently disable armed termination on a supported platform, so the
86
+ # weaker check stands there and the daemon says so at startup.
87
+ if _START_TIME_AVAILABLE:
88
+ return False
89
+ return _alive(pid)
90
+
91
+
92
+ _START_TIME_AVAILABLE = _probe_start_time_support()
93
+
94
+
95
+ def alive(pid: int) -> bool:
96
+ """Public liveness check: is this PID a running, non-zombie process?"""
97
+ return _alive(pid)
98
+
99
+
100
+ def _ppid(pid: int) -> int:
101
+ try:
102
+ with open(f"/proc/{pid}/stat", "rb") as fh:
103
+ data = fh.read()
104
+ return int(data[data.rindex(b")") + 2:].split()[1])
105
+ except Exception:
106
+ return 0
107
+
108
+
109
+ def is_descendant(pid: int, ancestor: int, max_depth: int = 40) -> bool:
110
+ """Is `pid` inside `ancestor`'s process tree?
111
+
112
+ A supervisor announcing a child is legitimate and common (cg_supervise does
113
+ exactly this), so the announced PID need not be the connecting peer. It
114
+ should, though, be a process that peer actually owns. Bounded walk: a
115
+ corrupted or cyclic chain must not spin.
116
+ """
117
+ if pid <= 1 or ancestor <= 1:
118
+ return False
119
+ seen, cur = set(), pid
120
+ for _ in range(max_depth):
121
+ if cur == ancestor:
122
+ return True
123
+ if cur in seen or cur <= 1:
124
+ return False
125
+ seen.add(cur)
126
+ cur = _ppid(cur)
127
+ return False
128
+
129
+
130
+ def terminate(pid: int, start_time: int, grace: float = 20.0, log=print) -> str:
131
+ if not _identity_ok(pid, start_time):
132
+ return "skipped: pid identity mismatch (reuse guard)"
133
+
134
+ try:
135
+ pgid = os.getpgid(pid)
136
+ except OSError:
137
+ return "skipped: process gone"
138
+
139
+ # Only signal the group if it is genuinely the session's own group --
140
+ # otherwise we could take out the daemon or the user's shell.
141
+ target, how = (-pgid, "group") if pgid == pid else (pid, "process")
142
+
143
+ try:
144
+ os.kill(target, signal.SIGTERM)
145
+ except OSError as e:
146
+ if e.errno == errno.ESRCH:
147
+ return "already gone"
148
+ return f"SIGTERM failed: {e}"
149
+
150
+ deadline = time.monotonic() + grace
151
+ while time.monotonic() < deadline:
152
+ if not _alive(pid):
153
+ return f"terminated ({how}, SIGTERM)"
154
+ time.sleep(0.2)
155
+
156
+ try:
157
+ os.kill(target, signal.SIGKILL)
158
+ except OSError:
159
+ pass
160
+ return f"terminated ({how}, SIGKILL after {grace:g}s grace)"
161
+
162
+
163
+ # --------------------------------------------------------------- notification
164
+ def _notify_file(cfg, event, log):
165
+ path = os.path.expanduser(cfg.get("path", "~/continuity-guard-alerts.log"))
166
+ os.makedirs(os.path.dirname(path) or ".", exist_ok=True)
167
+ with open(path, "a") as fh:
168
+ fh.write(
169
+ "{ts} {level:<6} {name:<24} {detector:<12} {detail}{shadow}\n".format(
170
+ ts=time.strftime("%Y-%m-%dT%H:%M:%S"),
171
+ level=event["level"].upper(),
172
+ name=event["name"],
173
+ detector=event["detector"],
174
+ detail=event["detail"],
175
+ shadow=" [SHADOW]" if event["shadow"] else "",
176
+ )
177
+ )
178
+
179
+
180
+ def _notify_exec(cfg, event, log):
181
+ cmd = cfg.get("command")
182
+ if not cmd:
183
+ return
184
+ env = dict(os.environ)
185
+ env.update({f"CG_{k.upper()}": str(v) for k, v in event.items()})
186
+ try:
187
+ subprocess.Popen(
188
+ cmd if isinstance(cmd, list) else ["/bin/sh", "-c", cmd],
189
+ env=env, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL,
190
+ start_new_session=True,
191
+ )
192
+ except Exception as e:
193
+ log(f"notify exec failed: {e}")
194
+
195
+
196
+ def _notify_webhook(cfg, event, log):
197
+ url = cfg.get("url")
198
+ if not url:
199
+ return
200
+ import json
201
+ body = json.dumps(event).encode("utf-8")
202
+ req = urllib.request.Request(
203
+ url, data=body, headers={"Content-Type": "application/json"}
204
+ )
205
+ try:
206
+ urllib.request.urlopen(req, timeout=float(cfg.get("timeout", 5)))
207
+ except Exception as e:
208
+ log(f"notify webhook failed: {e}")
209
+
210
+
211
+ _KINDS = {"file": _notify_file, "exec": _notify_exec, "webhook": _notify_webhook}
212
+
213
+
214
+ def notify(notifications, event, log=print):
215
+ """Fan out one event to every configured sink that subscribes to it.
216
+
217
+ Notification failures are logged and swallowed: a broken webhook must
218
+ never stop the daemon from doing its actual job.
219
+ """
220
+ for cfg in notifications or []:
221
+ wanted = cfg.get("on")
222
+ if wanted and event["level"] not in wanted and event["detector"] not in wanted:
223
+ continue
224
+ fn = _KINDS.get(cfg.get("kind"))
225
+ if not fn:
226
+ continue
227
+ try:
228
+ fn(cfg, event, log)
229
+ except Exception as e:
230
+ log(f"notify {cfg.get('kind')} failed: {e}")