continuity-guard 0.5.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- continuity_guard/__init__.py +14 -0
- continuity_guard/actions.py +230 -0
- continuity_guard/cli.py +565 -0
- continuity_guard/client.py +311 -0
- continuity_guard/config.example.toml +227 -0
- continuity_guard/config.py +243 -0
- continuity_guard/daemon.py +1060 -0
- continuity_guard/detectors.py +500 -0
- continuity_guard/handoff.py +195 -0
- continuity_guard/integrations.py +259 -0
- continuity_guard/protocol.py +212 -0
- continuity_guard/proxy.py +414 -0
- continuity_guard/quota.py +572 -0
- continuity_guard/state.py +200 -0
- continuity_guard/supervise.py +250 -0
- continuity_guard/transport.py +131 -0
- continuity_guard-0.5.0.dist-info/METADATA +768 -0
- continuity_guard-0.5.0.dist-info/RECORD +22 -0
- continuity_guard-0.5.0.dist-info/WHEEL +5 -0
- continuity_guard-0.5.0.dist-info/entry_points.txt +5 -0
- continuity_guard-0.5.0.dist-info/licenses/LICENSE +21 -0
- continuity_guard-0.5.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
"""Continuity Guard -- liveness and progress watchdog for long-running AI agent sessions.
|
|
2
|
+
|
|
3
|
+
from continuity_guard import guard
|
|
4
|
+
|
|
5
|
+
with guard("nightly-refactor", profile="overnight") as s:
|
|
6
|
+
for step in agent.run():
|
|
7
|
+
s.progress(step=step.name, tool=step.tool,
|
|
8
|
+
args=step.args, tokens=step.tokens)
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from .client import Session, args_hash, guard
|
|
12
|
+
|
|
13
|
+
__all__ = ["guard", "Session", "args_hash", "__version__"]
|
|
14
|
+
__version__ = "0.5.0"
|
|
@@ -0,0 +1,230 @@
|
|
|
1
|
+
"""Actions and notifications.
|
|
2
|
+
|
|
3
|
+
Kill authority lives here, and only here, and only locally. Three ways to get
|
|
4
|
+
termination wrong, all guarded against:
|
|
5
|
+
|
|
6
|
+
PID reuse -> pidfd where available, else PID+start_time verified as a pair
|
|
7
|
+
orphaned kids -> kill the process GROUP, not the process
|
|
8
|
+
no grace -> SIGTERM, wait, SIGKILL
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
import errno
|
|
14
|
+
import os
|
|
15
|
+
import signal
|
|
16
|
+
import subprocess
|
|
17
|
+
import time
|
|
18
|
+
import urllib.request
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
# ---------------------------------------------------------------- termination
|
|
22
|
+
# Can this platform give us a process start time at all? Decided once, from
|
|
23
|
+
# our own PID: if we cannot read our own, we cannot read anyone's.
|
|
24
|
+
def _probe_start_time_support() -> bool:
|
|
25
|
+
try:
|
|
26
|
+
with open(f"/proc/{os.getpid()}/stat", "rb") as fh:
|
|
27
|
+
data = fh.read()
|
|
28
|
+
int(data[data.rindex(b")") + 2:].split()[19])
|
|
29
|
+
return True
|
|
30
|
+
except Exception:
|
|
31
|
+
return False
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def _start_time(pid: int) -> int:
|
|
35
|
+
try:
|
|
36
|
+
with open(f"/proc/{pid}/stat", "rb") as fh:
|
|
37
|
+
data = fh.read()
|
|
38
|
+
return int(data[data.rindex(b")") + 2 :].split()[19])
|
|
39
|
+
except Exception:
|
|
40
|
+
return 0
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def _alive(pid: int) -> bool:
|
|
44
|
+
"""Is this PID a live process -- not a zombie?
|
|
45
|
+
|
|
46
|
+
/proc/<pid> keeps existing after a process dies, right up until its parent
|
|
47
|
+
reaps it. An existence check alone therefore reads a corpse as alive. That
|
|
48
|
+
matters on the supervisor path: cg_supervise is the PARENT of the agent it
|
|
49
|
+
registers, so a terminated agent sits as a zombie until the supervisor's
|
|
50
|
+
wait() runs. The grace loop would spin the full window and then fire a
|
|
51
|
+
pointless SIGKILL at an already-dead process -- and report
|
|
52
|
+
"SIGKILL after 20s grace" when SIGTERM had in fact worked immediately.
|
|
53
|
+
"""
|
|
54
|
+
try:
|
|
55
|
+
with open(f"/proc/{pid}/stat", "rb") as fh:
|
|
56
|
+
data = fh.read()
|
|
57
|
+
# comm (field 2) may contain spaces and parens: parse after the last ')'
|
|
58
|
+
return data[data.rindex(b")") + 2:].split()[0] != b"Z"
|
|
59
|
+
except (OSError, ValueError, IndexError):
|
|
60
|
+
return False
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
def _identity_ok(pid: int, expected_start: int) -> bool:
|
|
64
|
+
"""A PID alone is not an identity. Recorded an hour ago it may belong to
|
|
65
|
+
something else entirely by now.
|
|
66
|
+
|
|
67
|
+
A missing start_time used to fall back to "is anything alive with this
|
|
68
|
+
PID", which is not an identity check at all -- it accepts every live
|
|
69
|
+
process on the box. Combined with a world-writable socket and peer-supplied
|
|
70
|
+
policy, that made an armed daemon a process-kill oracle: announce someone
|
|
71
|
+
else's PID with start_time 0, set a policy that terminates immediately, and
|
|
72
|
+
the daemon signals it. Termination now REQUIRES a verified start_time, and
|
|
73
|
+
an unverifiable identity is refused rather than assumed.
|
|
74
|
+
"""
|
|
75
|
+
if pid <= 1:
|
|
76
|
+
return False
|
|
77
|
+
if expected_start:
|
|
78
|
+
return _start_time(pid) == expected_start
|
|
79
|
+
# No start_time was announced. This used to fall back to "is anything
|
|
80
|
+
# alive with this PID", which is not an identity check -- it accepts every
|
|
81
|
+
# live process on the box, and with peer-supplied policy that made an armed
|
|
82
|
+
# daemon a process-kill oracle. Where the platform CAN supply a start time,
|
|
83
|
+
# its absence now means the identity is unverifiable and the kill is
|
|
84
|
+
# refused. Where it cannot (no /proc: macOS, Windows), refusing outright
|
|
85
|
+
# would silently disable armed termination on a supported platform, so the
|
|
86
|
+
# weaker check stands there and the daemon says so at startup.
|
|
87
|
+
if _START_TIME_AVAILABLE:
|
|
88
|
+
return False
|
|
89
|
+
return _alive(pid)
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
_START_TIME_AVAILABLE = _probe_start_time_support()
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
def alive(pid: int) -> bool:
|
|
96
|
+
"""Public liveness check: is this PID a running, non-zombie process?"""
|
|
97
|
+
return _alive(pid)
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
def _ppid(pid: int) -> int:
|
|
101
|
+
try:
|
|
102
|
+
with open(f"/proc/{pid}/stat", "rb") as fh:
|
|
103
|
+
data = fh.read()
|
|
104
|
+
return int(data[data.rindex(b")") + 2:].split()[1])
|
|
105
|
+
except Exception:
|
|
106
|
+
return 0
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
def is_descendant(pid: int, ancestor: int, max_depth: int = 40) -> bool:
|
|
110
|
+
"""Is `pid` inside `ancestor`'s process tree?
|
|
111
|
+
|
|
112
|
+
A supervisor announcing a child is legitimate and common (cg_supervise does
|
|
113
|
+
exactly this), so the announced PID need not be the connecting peer. It
|
|
114
|
+
should, though, be a process that peer actually owns. Bounded walk: a
|
|
115
|
+
corrupted or cyclic chain must not spin.
|
|
116
|
+
"""
|
|
117
|
+
if pid <= 1 or ancestor <= 1:
|
|
118
|
+
return False
|
|
119
|
+
seen, cur = set(), pid
|
|
120
|
+
for _ in range(max_depth):
|
|
121
|
+
if cur == ancestor:
|
|
122
|
+
return True
|
|
123
|
+
if cur in seen or cur <= 1:
|
|
124
|
+
return False
|
|
125
|
+
seen.add(cur)
|
|
126
|
+
cur = _ppid(cur)
|
|
127
|
+
return False
|
|
128
|
+
|
|
129
|
+
|
|
130
|
+
def terminate(pid: int, start_time: int, grace: float = 20.0, log=print) -> str:
|
|
131
|
+
if not _identity_ok(pid, start_time):
|
|
132
|
+
return "skipped: pid identity mismatch (reuse guard)"
|
|
133
|
+
|
|
134
|
+
try:
|
|
135
|
+
pgid = os.getpgid(pid)
|
|
136
|
+
except OSError:
|
|
137
|
+
return "skipped: process gone"
|
|
138
|
+
|
|
139
|
+
# Only signal the group if it is genuinely the session's own group --
|
|
140
|
+
# otherwise we could take out the daemon or the user's shell.
|
|
141
|
+
target, how = (-pgid, "group") if pgid == pid else (pid, "process")
|
|
142
|
+
|
|
143
|
+
try:
|
|
144
|
+
os.kill(target, signal.SIGTERM)
|
|
145
|
+
except OSError as e:
|
|
146
|
+
if e.errno == errno.ESRCH:
|
|
147
|
+
return "already gone"
|
|
148
|
+
return f"SIGTERM failed: {e}"
|
|
149
|
+
|
|
150
|
+
deadline = time.monotonic() + grace
|
|
151
|
+
while time.monotonic() < deadline:
|
|
152
|
+
if not _alive(pid):
|
|
153
|
+
return f"terminated ({how}, SIGTERM)"
|
|
154
|
+
time.sleep(0.2)
|
|
155
|
+
|
|
156
|
+
try:
|
|
157
|
+
os.kill(target, signal.SIGKILL)
|
|
158
|
+
except OSError:
|
|
159
|
+
pass
|
|
160
|
+
return f"terminated ({how}, SIGKILL after {grace:g}s grace)"
|
|
161
|
+
|
|
162
|
+
|
|
163
|
+
# --------------------------------------------------------------- notification
|
|
164
|
+
def _notify_file(cfg, event, log):
|
|
165
|
+
path = os.path.expanduser(cfg.get("path", "~/continuity-guard-alerts.log"))
|
|
166
|
+
os.makedirs(os.path.dirname(path) or ".", exist_ok=True)
|
|
167
|
+
with open(path, "a") as fh:
|
|
168
|
+
fh.write(
|
|
169
|
+
"{ts} {level:<6} {name:<24} {detector:<12} {detail}{shadow}\n".format(
|
|
170
|
+
ts=time.strftime("%Y-%m-%dT%H:%M:%S"),
|
|
171
|
+
level=event["level"].upper(),
|
|
172
|
+
name=event["name"],
|
|
173
|
+
detector=event["detector"],
|
|
174
|
+
detail=event["detail"],
|
|
175
|
+
shadow=" [SHADOW]" if event["shadow"] else "",
|
|
176
|
+
)
|
|
177
|
+
)
|
|
178
|
+
|
|
179
|
+
|
|
180
|
+
def _notify_exec(cfg, event, log):
|
|
181
|
+
cmd = cfg.get("command")
|
|
182
|
+
if not cmd:
|
|
183
|
+
return
|
|
184
|
+
env = dict(os.environ)
|
|
185
|
+
env.update({f"CG_{k.upper()}": str(v) for k, v in event.items()})
|
|
186
|
+
try:
|
|
187
|
+
subprocess.Popen(
|
|
188
|
+
cmd if isinstance(cmd, list) else ["/bin/sh", "-c", cmd],
|
|
189
|
+
env=env, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL,
|
|
190
|
+
start_new_session=True,
|
|
191
|
+
)
|
|
192
|
+
except Exception as e:
|
|
193
|
+
log(f"notify exec failed: {e}")
|
|
194
|
+
|
|
195
|
+
|
|
196
|
+
def _notify_webhook(cfg, event, log):
|
|
197
|
+
url = cfg.get("url")
|
|
198
|
+
if not url:
|
|
199
|
+
return
|
|
200
|
+
import json
|
|
201
|
+
body = json.dumps(event).encode("utf-8")
|
|
202
|
+
req = urllib.request.Request(
|
|
203
|
+
url, data=body, headers={"Content-Type": "application/json"}
|
|
204
|
+
)
|
|
205
|
+
try:
|
|
206
|
+
urllib.request.urlopen(req, timeout=float(cfg.get("timeout", 5)))
|
|
207
|
+
except Exception as e:
|
|
208
|
+
log(f"notify webhook failed: {e}")
|
|
209
|
+
|
|
210
|
+
|
|
211
|
+
_KINDS = {"file": _notify_file, "exec": _notify_exec, "webhook": _notify_webhook}
|
|
212
|
+
|
|
213
|
+
|
|
214
|
+
def notify(notifications, event, log=print):
|
|
215
|
+
"""Fan out one event to every configured sink that subscribes to it.
|
|
216
|
+
|
|
217
|
+
Notification failures are logged and swallowed: a broken webhook must
|
|
218
|
+
never stop the daemon from doing its actual job.
|
|
219
|
+
"""
|
|
220
|
+
for cfg in notifications or []:
|
|
221
|
+
wanted = cfg.get("on")
|
|
222
|
+
if wanted and event["level"] not in wanted and event["detector"] not in wanted:
|
|
223
|
+
continue
|
|
224
|
+
fn = _KINDS.get(cfg.get("kind"))
|
|
225
|
+
if not fn:
|
|
226
|
+
continue
|
|
227
|
+
try:
|
|
228
|
+
fn(cfg, event, log)
|
|
229
|
+
except Exception as e:
|
|
230
|
+
log(f"notify {cfg.get('kind')} failed: {e}")
|