@m13v/s4l 1.7.1 → 1.7.2-rc.10
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/mcp/dist/index.js +21 -1
- package/mcp/dist/version.json +2 -2
- package/mcp/manifest.json +1 -1
- package/mcp/menubar/s4l_card.py +92 -1
- package/mcp/menubar/s4l_menubar.py +218 -20
- package/mcp/menubar/s4l_state.py +72 -0
- package/mcp/package.json +1 -1
- package/package.json +1 -1
- package/scripts/claude_job.py +33 -5
- package/scripts/merge_review_queue.py +51 -21
- package/scripts/producer_deathwatch.py +210 -0
- package/scripts/run_claude.sh +33 -0
- package/scripts/salvage_orphaned_prep_results.py +69 -6
- package/scripts/scheduled_task_selfheal.py +26 -0
- package/skill/run-twitter-cycle.sh +44 -152
package/scripts/claude_job.py
CHANGED
|
@@ -60,7 +60,6 @@ except Exception: # pragma: no cover - cosmetic only
|
|
|
60
60
|
|
|
61
61
|
# script_tag -> queue type. ONLY pure text->JSON claude calls belong here.
|
|
62
62
|
TAG_TO_TYPE = {
|
|
63
|
-
"run-twitter-cycle-queries": "twitter-query",
|
|
64
63
|
"run-twitter-cycle-prep": "twitter-prep",
|
|
65
64
|
"feedback-digest": "feedback-digest",
|
|
66
65
|
# Topic-invention lane (queue-native since 2026-07-06; invent_topics.py
|
|
@@ -75,11 +74,10 @@ TAG_TO_TYPE = {
|
|
|
75
74
|
}
|
|
76
75
|
|
|
77
76
|
# queue type -> (activity state, label) the menu bar shows while the job is in
|
|
78
|
-
# flight. Phase-
|
|
79
|
-
#
|
|
80
|
-
#
|
|
77
|
+
# flight. Phase-2b prep is the reply drafting. Both the launchd provider (which
|
|
78
|
+
# blocks for minutes) and the scheduled-task worker (which does the LLM turn)
|
|
79
|
+
# narrate from this one map.
|
|
81
80
|
TYPE_TO_ACTIVITY = {
|
|
82
|
-
"twitter-query": ("scanning", "search"),
|
|
83
81
|
"twitter-prep": ("drafting", "draft"),
|
|
84
82
|
"feedback-digest": ("learning", "feedback"),
|
|
85
83
|
"invent-topic": ("learning", "new topic"),
|
|
@@ -335,6 +333,32 @@ def heartbeat_path() -> str:
|
|
|
335
333
|
return os.path.join(queue_root(), "worker-heartbeat.json")
|
|
336
334
|
|
|
337
335
|
|
|
336
|
+
def _arm_deathwatch(job_id: str, qtype: str, batch: str) -> None:
|
|
337
|
+
"""Best-effort dead-man's-switch (2026-07-08): arm scripts/producer_deathwatch.py
|
|
338
|
+
to flag an UNEXPECTED death (SIGKILL/OOM/hard crash) of THIS process while
|
|
339
|
+
it's blocked in the cmd_provider() poll loop below — the exact gap that
|
|
340
|
+
made orphaned salvage results ("worker drafted, no card") unexplainable.
|
|
341
|
+
Every normal return path in cmd_provider() calls _disarm_deathwatch()
|
|
342
|
+
first, so a clean exit never produces a report. Shared with
|
|
343
|
+
run_claude.sh's direct-exec path (every non-queue platform), which calls
|
|
344
|
+
producer_deathwatch.py's `arm`/`disarm` CLI directly instead of through
|
|
345
|
+
this Python wrapper — see that file for the single implementation both
|
|
346
|
+
callers share."""
|
|
347
|
+
try:
|
|
348
|
+
import producer_deathwatch as pdw
|
|
349
|
+
pdw.arm(os.getpid(), job_id, qtype, batch, call_path="queue")
|
|
350
|
+
except Exception:
|
|
351
|
+
pass
|
|
352
|
+
|
|
353
|
+
|
|
354
|
+
def _disarm_deathwatch(job_id: str) -> None:
|
|
355
|
+
try:
|
|
356
|
+
import producer_deathwatch as pdw
|
|
357
|
+
pdw.disarm(job_id)
|
|
358
|
+
except Exception:
|
|
359
|
+
pass
|
|
360
|
+
|
|
361
|
+
|
|
338
362
|
def _stamp_heartbeat(event: str, qtype: str | None = None) -> None:
|
|
339
363
|
"""Best-effort: never let a heartbeat write failure break the queue."""
|
|
340
364
|
try:
|
|
@@ -690,6 +714,7 @@ def cmd_provider(ns) -> int:
|
|
|
690
714
|
running_path = os.path.join(running_dir(), fname)
|
|
691
715
|
_atomic_write(pending_path, job)
|
|
692
716
|
_plog(f"enqueued {qtype} job {job_id} batch={batch}; waiting for a scheduled task (timeout {ns.timeout}s)")
|
|
717
|
+
_arm_deathwatch(job_id, qtype, batch)
|
|
693
718
|
# Narrate the (multi-minute) block to the menu bar. The launchd draft lane has
|
|
694
719
|
# no other activity writer, so without this the box looks idle while it works.
|
|
695
720
|
# Cleared by run-draft-and-publish.sh's exit trap at cycle end (and by the
|
|
@@ -732,6 +757,7 @@ def cmd_provider(ns) -> int:
|
|
|
732
757
|
os.remove(res_path)
|
|
733
758
|
if res.get("status") == "error":
|
|
734
759
|
_plog(f"job {job_id} returned error: {res.get('error', 'unknown')}")
|
|
760
|
+
_disarm_deathwatch(job_id)
|
|
735
761
|
return 1
|
|
736
762
|
obj = res.get("result")
|
|
737
763
|
# Emit a claude `--output-format json` shaped envelope so the
|
|
@@ -759,6 +785,7 @@ def cmd_provider(ns) -> int:
|
|
|
759
785
|
except Exception:
|
|
760
786
|
_ncand = "?"
|
|
761
787
|
_plog(f"consumed result for job {job_id} batch={batch} ({qtype}); {_ncand} candidates -> producer assembles the plan")
|
|
788
|
+
_disarm_deathwatch(job_id)
|
|
762
789
|
return 0
|
|
763
790
|
time.sleep(POLL_INTERVAL_S)
|
|
764
791
|
|
|
@@ -779,6 +806,7 @@ def cmd_provider(ns) -> int:
|
|
|
779
806
|
# flicker the ⚠ off). Cleared only when a draft actually drains.
|
|
780
807
|
_bump_drain_timeout()
|
|
781
808
|
_plog(f"timed out after {ns.timeout}s waiting for job {job_id} batch={batch} ({qtype}); removed the job")
|
|
809
|
+
_disarm_deathwatch(job_id)
|
|
782
810
|
return 79 # mirror run_claude.sh's "blocked, skip cleanly" exit code
|
|
783
811
|
|
|
784
812
|
|
|
@@ -129,16 +129,31 @@ STATS_KEYS = (
|
|
|
129
129
|
)
|
|
130
130
|
|
|
131
131
|
|
|
132
|
-
def
|
|
133
|
-
"""
|
|
134
|
-
|
|
135
|
-
|
|
136
|
-
|
|
137
|
-
|
|
138
|
-
|
|
139
|
-
|
|
140
|
-
|
|
141
|
-
|
|
132
|
+
def _sync_with_backend(cands: list) -> tuple[int, int]:
|
|
133
|
+
"""One bulk /api/v1/twitter-candidates lookup for every still-open candidate
|
|
134
|
+
(not posted, not terminal), used for two things:
|
|
135
|
+
|
|
136
|
+
- stamp the discovery-time `stats` sidecar the card renders (candidates
|
|
137
|
+
that already have one are left alone), same as the old
|
|
138
|
+
_enrich_with_stats this replaces.
|
|
139
|
+
- notice when the backend has ALREADY retired a candidate this plan
|
|
140
|
+
still thinks is 'pending' (most commonly the Phase 0 freshness gate
|
|
141
|
+
flipping status='expired' after FRESHNESS_HOURS — see
|
|
142
|
+
skill/run-twitter-cycle.sh) and mark it terminal here too, same as a
|
|
143
|
+
human "discard all pending" would. Without this, a card can sit in
|
|
144
|
+
the review queue as an approvable draft long after the backend has
|
|
145
|
+
moved on; approving it later silently no-ops (post_drafts returns
|
|
146
|
+
posted:0, no browser ever launches, no post-*.log — see the
|
|
147
|
+
2026-07-09 "approved 3 cards, nothing posted" investigation).
|
|
148
|
+
|
|
149
|
+
Runs on EVERY merge (every cycle), not just once per candidate, so status
|
|
150
|
+
drift after the initial stamp is still caught while a card is still
|
|
151
|
+
pending. Best-effort: any API failure leaves every candidate untouched
|
|
152
|
+
(fail open, same as before). Returns (stamped_count, pruned_count)."""
|
|
153
|
+
pending = [c for c in cands if not c.get("posted") and not c.get("terminal") and _thread_url(c)]
|
|
154
|
+
if not pending:
|
|
155
|
+
return 0, 0
|
|
156
|
+
urls = sorted({_thread_url(c) for c in pending})[:500]
|
|
142
157
|
try:
|
|
143
158
|
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
|
|
144
159
|
from http_api import api_get
|
|
@@ -149,17 +164,24 @@ def _enrich_with_stats(cands: list) -> int:
|
|
|
149
164
|
)
|
|
150
165
|
rows = (resp.get("data") or {}).get("candidates") or []
|
|
151
166
|
except BaseException as e: # http_api raises SystemExit on terminal failure
|
|
152
|
-
print(f"[merge_review_queue]
|
|
153
|
-
return 0
|
|
167
|
+
print(f"[merge_review_queue] backend sync skipped: {e}", file=sys.stderr)
|
|
168
|
+
return 0, 0
|
|
154
169
|
by_url = {str(r.get("tweet_url")): r for r in rows if r.get("tweet_url")}
|
|
155
170
|
stamped = 0
|
|
156
|
-
|
|
171
|
+
pruned = 0
|
|
172
|
+
for c in pending:
|
|
157
173
|
row = by_url.get(_thread_url(c))
|
|
158
174
|
if not row:
|
|
159
175
|
continue
|
|
160
|
-
|
|
161
|
-
|
|
162
|
-
|
|
176
|
+
if not c.get("stats"):
|
|
177
|
+
c["stats"] = {k: row.get(k) for k in STATS_KEYS}
|
|
178
|
+
stamped += 1
|
|
179
|
+
status = row.get("status")
|
|
180
|
+
if status and status != "pending":
|
|
181
|
+
c["terminal"] = True
|
|
182
|
+
c["discard_reason"] = f"backend_status_{status}"
|
|
183
|
+
pruned += 1
|
|
184
|
+
return stamped, pruned
|
|
163
185
|
|
|
164
186
|
|
|
165
187
|
def main() -> int:
|
|
@@ -255,9 +277,15 @@ def main() -> int:
|
|
|
255
277
|
merged.append(c)
|
|
256
278
|
added += 1
|
|
257
279
|
|
|
258
|
-
stamped =
|
|
280
|
+
stamped, pruned = _sync_with_backend(merged)
|
|
259
281
|
if stamped:
|
|
260
282
|
print(f"[merge_review_queue] stamped stats on {stamped} candidate(s)", file=sys.stderr)
|
|
283
|
+
if pruned:
|
|
284
|
+
print(
|
|
285
|
+
f"[merge_review_queue] pruned {pruned} candidate(s) already retired by the "
|
|
286
|
+
"backend (expired/etc.) before they were reviewed",
|
|
287
|
+
file=sys.stderr,
|
|
288
|
+
)
|
|
261
289
|
|
|
262
290
|
plan_obj = {"candidates": merged}
|
|
263
291
|
if plan_created_at:
|
|
@@ -274,15 +302,17 @@ def main() -> int:
|
|
|
274
302
|
_atomic_write(dst, plan_obj)
|
|
275
303
|
ensure_store_symlink()
|
|
276
304
|
|
|
277
|
-
# Refresh the review-request marker the menu bar polls (count = pending,
|
|
278
|
-
|
|
305
|
+
# Refresh the review-request marker the menu bar polls (count = pending,
|
|
306
|
+
# not posted, not terminal -- a just-pruned expired card must not still
|
|
307
|
+
# inflate the badge).
|
|
308
|
+
pending_count = len([c for c in merged if not c.get("posted") and not c.get("terminal")])
|
|
279
309
|
project = ns.project or batch.get("project") or (new_cands[0].get("matched_project") if new_cands else None)
|
|
280
310
|
_atomic_write(
|
|
281
311
|
review_request_path(),
|
|
282
312
|
{
|
|
283
313
|
"batch_id": REVIEW_QUEUE_ID,
|
|
284
314
|
"project": project,
|
|
285
|
-
"count":
|
|
315
|
+
"count": pending_count,
|
|
286
316
|
"plan_path": dst,
|
|
287
317
|
"created_at": time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime()),
|
|
288
318
|
},
|
|
@@ -306,7 +336,7 @@ def main() -> int:
|
|
|
306
336
|
|
|
307
337
|
print(
|
|
308
338
|
f"[merge_review_queue] merged {added} new draft(s) into {REVIEW_QUEUE_ID} "
|
|
309
|
-
f"({
|
|
339
|
+
f"({pending_count} pending total) from {os.path.basename(src)}",
|
|
310
340
|
file=sys.stderr,
|
|
311
341
|
)
|
|
312
342
|
# Clean up the consumed batch plan so /tmp doesn't fill with orphans.
|
|
@@ -0,0 +1,210 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""producer_deathwatch.py — dead-man's-switch for every Claude-calling
|
|
3
|
+
producer process in the pipeline.
|
|
4
|
+
|
|
5
|
+
Watches a single PID: either scripts/claude_job.py's blocking provider wait
|
|
6
|
+
(cmd_provider, the queue path used by twitter-prep/feedback-digest/etc.), or
|
|
7
|
+
scripts/run_claude.sh's direct `claude -p` exec (every other platform:
|
|
8
|
+
reddit, linkedin, github, moltbook, instagram, dm-outreach-*, ...). If that
|
|
9
|
+
PID disappears while its arm marker still exists, something killed it
|
|
10
|
+
(SIGKILL / OOM / hard crash) rather than a normal return — a clean return
|
|
11
|
+
always disarms first. This is the exact gap that made orphaned salvage
|
|
12
|
+
results ("worker drafted, no card") unexplainable: the dying process can
|
|
13
|
+
never log its own death, and salvage only sees the aftermath up to
|
|
14
|
+
--max-age-hours later.
|
|
15
|
+
|
|
16
|
+
On an unexpected death this:
|
|
17
|
+
1. Snapshots memory pressure + related processes.
|
|
18
|
+
2. Appends a structured JSON line to producer-deathwatch.jsonl (local,
|
|
19
|
+
box-only, for offline/box-local debugging).
|
|
20
|
+
3. POSTs the same event to /api/v1/producer-death-events so it's queryable
|
|
21
|
+
across every install from Postgres, not just grep-able on one box (see
|
|
22
|
+
migrations/2026-07-09-producer-death-events.sql).
|
|
23
|
+
4. Emits a one-line summary via claude_job._plog() into provider.log,
|
|
24
|
+
which scripts/relay_provider_log.py already ships to Cloud Logging.
|
|
25
|
+
|
|
26
|
+
Three subcommands, so both Python (claude_job.py) and bash (run_claude.sh)
|
|
27
|
+
callers share one implementation:
|
|
28
|
+
arm — write the marker + spawn `watch` detached (start_new_session=True,
|
|
29
|
+
so it survives being in the same process group as the watched
|
|
30
|
+
pid if that group gets signaled). Called right before a caller
|
|
31
|
+
starts blocking on the watched pid.
|
|
32
|
+
disarm — remove the marker. Called on every normal return path AND from
|
|
33
|
+
a signal-trap cleanup (e.g. run_claude.sh's _sa_cleanup, which
|
|
34
|
+
itself SIGKILLs the watched process group as part of ordinary
|
|
35
|
+
TERM/INT/HUP handling — that must disarm too, or a normal
|
|
36
|
+
watchdog-triggered shutdown would misreport as an unexpected
|
|
37
|
+
death).
|
|
38
|
+
watch — the actual poll loop (internal; `arm` spawns this, nothing else
|
|
39
|
+
should call it directly).
|
|
40
|
+
|
|
41
|
+
Best-effort throughout: this is diagnostics only, never allowed to affect
|
|
42
|
+
the real job either way.
|
|
43
|
+
"""
|
|
44
|
+
from __future__ import annotations
|
|
45
|
+
|
|
46
|
+
import argparse
|
|
47
|
+
import json
|
|
48
|
+
import os
|
|
49
|
+
import subprocess
|
|
50
|
+
import sys
|
|
51
|
+
import time
|
|
52
|
+
|
|
53
|
+
HERE = os.path.dirname(os.path.abspath(__file__))
|
|
54
|
+
sys.path.insert(0, HERE)
|
|
55
|
+
from claude_job import queue_root, _plog # noqa: E402
|
|
56
|
+
|
|
57
|
+
POLL_S = 5.0
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def arm_path(job_id: str) -> str:
|
|
61
|
+
return os.path.join(queue_root(), f"deathwatch-armed-{job_id}.marker")
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def arm(watch_pid: int, job_id: str, qtype: str, batch: str, call_path: str) -> None:
|
|
65
|
+
"""Write the marker and spawn a detached `watch` subprocess. Best-effort:
|
|
66
|
+
any failure here must never block the real caller."""
|
|
67
|
+
try:
|
|
68
|
+
marker = arm_path(job_id)
|
|
69
|
+
os.makedirs(queue_root(), exist_ok=True)
|
|
70
|
+
with open(marker, "w") as f:
|
|
71
|
+
f.write(str(watch_pid))
|
|
72
|
+
subprocess.Popen(
|
|
73
|
+
[sys.executable, os.path.abspath(__file__), "watch",
|
|
74
|
+
"--watch-pid", str(watch_pid), "--job-id", job_id,
|
|
75
|
+
"--qtype", qtype, "--batch", batch, "--call-path", call_path],
|
|
76
|
+
start_new_session=True,
|
|
77
|
+
stdin=subprocess.DEVNULL, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL,
|
|
78
|
+
)
|
|
79
|
+
except Exception:
|
|
80
|
+
pass
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
def disarm(job_id: str) -> None:
|
|
84
|
+
try:
|
|
85
|
+
os.remove(arm_path(job_id))
|
|
86
|
+
except Exception:
|
|
87
|
+
pass
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
def _snapshot() -> tuple[str, str]:
|
|
91
|
+
try:
|
|
92
|
+
vm = subprocess.run(["vm_stat"], capture_output=True, text=True, timeout=5).stdout.strip()
|
|
93
|
+
except Exception as e:
|
|
94
|
+
vm = f"(vm_stat failed: {e})"
|
|
95
|
+
try:
|
|
96
|
+
procs = subprocess.run(
|
|
97
|
+
["/bin/ps", "-axo", "pid=,ppid=,%mem=,%cpu=,command="],
|
|
98
|
+
capture_output=True, text=True, timeout=5,
|
|
99
|
+
).stdout
|
|
100
|
+
related = "\n".join(
|
|
101
|
+
ln for ln in procs.splitlines()
|
|
102
|
+
if "claude" in ln or "run-twitter-cycle" in ln or "run_claude" in ln
|
|
103
|
+
) or "(none matching claude/run-twitter-cycle/run_claude)"
|
|
104
|
+
except Exception as e:
|
|
105
|
+
related = f"(ps failed: {e})"
|
|
106
|
+
# Cap length: this rides into a Postgres text column and a JSON line;
|
|
107
|
+
# keep it bounded so a busy box's ps dump can't balloon either.
|
|
108
|
+
return vm[:4000], related[:4000]
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
def _report_to_db(event: dict) -> None:
|
|
112
|
+
"""Best-effort POST to /api/v1/producer-death-events. Catches
|
|
113
|
+
BaseException, not just Exception: http_api._request raises SystemExit
|
|
114
|
+
on a terminal 4xx/5xx, which must never be allowed to break the
|
|
115
|
+
diagnostic path (mirrors autopilot_stall_watch.py's same guard)."""
|
|
116
|
+
try:
|
|
117
|
+
import http_api # noqa: E402 (sibling module, HERE already on sys.path)
|
|
118
|
+
http_api.api_post("/api/v1/producer-death-events", {
|
|
119
|
+
"watch_pid": event["watch_pid"],
|
|
120
|
+
"job_id": event["job_id"],
|
|
121
|
+
"batch_id": event["batch"] if event["batch"] != "-" else None,
|
|
122
|
+
"qtype": event["qtype"],
|
|
123
|
+
"call_path": event["call_path"],
|
|
124
|
+
"vm_stat_summary": event["vm_stat"],
|
|
125
|
+
"related_processes": event["related_processes"],
|
|
126
|
+
})
|
|
127
|
+
except BaseException:
|
|
128
|
+
pass
|
|
129
|
+
|
|
130
|
+
|
|
131
|
+
def watch(watch_pid: int, job_id: str, qtype: str, batch: str, call_path: str) -> int:
|
|
132
|
+
marker = arm_path(job_id)
|
|
133
|
+
while True:
|
|
134
|
+
if not os.path.exists(marker):
|
|
135
|
+
return 0 # disarmed: the caller returned cleanly, nothing to report
|
|
136
|
+
try:
|
|
137
|
+
os.kill(watch_pid, 0)
|
|
138
|
+
except ProcessLookupError:
|
|
139
|
+
break # pid gone but still armed -> unexpected death
|
|
140
|
+
except PermissionError:
|
|
141
|
+
pass # exists, just not signalable from here; keep watching
|
|
142
|
+
except Exception:
|
|
143
|
+
pass
|
|
144
|
+
time.sleep(POLL_S)
|
|
145
|
+
|
|
146
|
+
ts = time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime())
|
|
147
|
+
vm, related = _snapshot()
|
|
148
|
+
event = {
|
|
149
|
+
"ts": ts,
|
|
150
|
+
"event": "unexpected_death",
|
|
151
|
+
"watch_pid": watch_pid,
|
|
152
|
+
"job_id": job_id,
|
|
153
|
+
"qtype": qtype,
|
|
154
|
+
"batch": batch,
|
|
155
|
+
"call_path": call_path,
|
|
156
|
+
"vm_stat": vm,
|
|
157
|
+
"related_processes": related,
|
|
158
|
+
}
|
|
159
|
+
try:
|
|
160
|
+
os.makedirs(queue_root(), exist_ok=True)
|
|
161
|
+
with open(os.path.join(queue_root(), "producer-deathwatch.jsonl"), "a") as f:
|
|
162
|
+
f.write(json.dumps(event) + "\n")
|
|
163
|
+
except Exception:
|
|
164
|
+
pass
|
|
165
|
+
_report_to_db(event)
|
|
166
|
+
try:
|
|
167
|
+
_plog(f"[deathwatch] UNEXPECTED DEATH watch_pid={watch_pid} job={job_id} "
|
|
168
|
+
f"type={qtype} batch={batch} path={call_path} -> see producer-deathwatch.jsonl")
|
|
169
|
+
except Exception:
|
|
170
|
+
pass
|
|
171
|
+
try:
|
|
172
|
+
os.remove(marker)
|
|
173
|
+
except Exception:
|
|
174
|
+
pass
|
|
175
|
+
return 0
|
|
176
|
+
|
|
177
|
+
|
|
178
|
+
def main() -> int:
|
|
179
|
+
ap = argparse.ArgumentParser()
|
|
180
|
+
sub = ap.add_subparsers(dest="cmd", required=True)
|
|
181
|
+
|
|
182
|
+
pa = sub.add_parser("arm")
|
|
183
|
+
pa.add_argument("--watch-pid", type=int, required=True)
|
|
184
|
+
pa.add_argument("--job-id", required=True)
|
|
185
|
+
pa.add_argument("--qtype", default="?")
|
|
186
|
+
pa.add_argument("--batch", default="-")
|
|
187
|
+
pa.add_argument("--call-path", default="queue", choices=["queue", "direct"])
|
|
188
|
+
|
|
189
|
+
pd = sub.add_parser("disarm")
|
|
190
|
+
pd.add_argument("--job-id", required=True)
|
|
191
|
+
|
|
192
|
+
pw = sub.add_parser("watch")
|
|
193
|
+
pw.add_argument("--watch-pid", type=int, required=True)
|
|
194
|
+
pw.add_argument("--job-id", required=True)
|
|
195
|
+
pw.add_argument("--qtype", default="?")
|
|
196
|
+
pw.add_argument("--batch", default="-")
|
|
197
|
+
pw.add_argument("--call-path", default="queue", choices=["queue", "direct"])
|
|
198
|
+
|
|
199
|
+
ns = ap.parse_args()
|
|
200
|
+
if ns.cmd == "arm":
|
|
201
|
+
arm(ns.watch_pid, ns.job_id, ns.qtype, ns.batch, ns.call_path)
|
|
202
|
+
return 0
|
|
203
|
+
if ns.cmd == "disarm":
|
|
204
|
+
disarm(ns.job_id)
|
|
205
|
+
return 0
|
|
206
|
+
return watch(ns.watch_pid, ns.job_id, ns.qtype, ns.batch, ns.call_path)
|
|
207
|
+
|
|
208
|
+
|
|
209
|
+
if __name__ == "__main__":
|
|
210
|
+
sys.exit(main())
|
package/scripts/run_claude.sh
CHANGED
|
@@ -200,6 +200,16 @@ _sa_cleanup() {
|
|
|
200
200
|
rm -f "$SIDE_LOG"
|
|
201
201
|
rm -f "$ACTIVE_FILE"
|
|
202
202
|
|
|
203
|
+
# Disarm the deathwatch (see the arm call in the retry loop below). This
|
|
204
|
+
# path SIGKILLs $CLAUDE_PG's process group a few lines down as ordinary
|
|
205
|
+
# TERM/INT/HUP handling (e.g. the watchdog killing a genuinely hung run),
|
|
206
|
+
# so it must disarm too — otherwise a normal forced shutdown would
|
|
207
|
+
# misreport as an unexpected death once that kill lands.
|
|
208
|
+
if [ -n "${_SA_DW_JOB:-}" ]; then
|
|
209
|
+
python3 "$REPO_DIR/scripts/producer_deathwatch.py" disarm --job-id "$_SA_DW_JOB" \
|
|
210
|
+
>/dev/null 2>&1 || true
|
|
211
|
+
fi
|
|
212
|
+
|
|
203
213
|
# Sweep orphan claude descendants. Process groups survive the parent's
|
|
204
214
|
# death (kids reparented to launchd keep their PGID), so killing
|
|
205
215
|
# `kill -- -PGID` reaches every grandchild, including ones reparented
|
|
@@ -308,8 +318,31 @@ EOF
|
|
|
308
318
|
{ claude --session-id "$SESSION_ID" ${MODEL_ARGS[@]+"${MODEL_ARGS[@]}"} "$@" | tee -a "$SIDE_LOG"; exit "${PIPESTATUS[0]}"; } &
|
|
309
319
|
CLAUDE_PG=$!
|
|
310
320
|
set +m
|
|
321
|
+
# Dead-man's-switch (2026-07-09): every non-queue-routed tag (reddit,
|
|
322
|
+
# linkedin, github, moltbook, instagram, dm-outreach-*, ...) blocks
|
|
323
|
+
# here exactly like claude_job.py's queue provider does, with the
|
|
324
|
+
# same silent-death risk (SIGKILL/OOM/hard crash while waiting).
|
|
325
|
+
# Watches THIS SCRIPT's own pid ($$), not $CLAUDE_PG: if only the
|
|
326
|
+
# claude child dies, `wait` unblocks normally and this script keeps
|
|
327
|
+
# running (RC-checked, logged, retried) — no observability gap.
|
|
328
|
+
# The gap is when the WHOLE TREE (this script included) is killed
|
|
329
|
+
# together, which is what actually happened in the salvage-orphan
|
|
330
|
+
# cases this was built for. Arm per-attempt (job id includes
|
|
331
|
+
# $CLAUDE_PG so a retry never collides with a still-unwinding prior
|
|
332
|
+
# attempt's watcher); disarm right after `wait` returns AND from
|
|
333
|
+
# _sa_cleanup's trap (that path SIGKILLs $CLAUDE_PG's group as
|
|
334
|
+
# ordinary TERM/INT/HUP handling, which must disarm too or a normal
|
|
335
|
+
# watchdog-triggered shutdown would misreport as an unexpected
|
|
336
|
+
# death). See scripts/producer_deathwatch.py.
|
|
337
|
+
_SA_DW_JOB="${SESSION_ID}-${CLAUDE_PG}"
|
|
338
|
+
python3 "$REPO_DIR/scripts/producer_deathwatch.py" arm \
|
|
339
|
+
--watch-pid "$$" --job-id "$_SA_DW_JOB" --qtype "$SCRIPT_TAG" \
|
|
340
|
+
--batch "${BATCH_ID:-${SA_CYCLE_ID:--}}" --call-path direct \
|
|
341
|
+
>/dev/null 2>&1 || true
|
|
311
342
|
wait "$CLAUDE_PG"
|
|
312
343
|
RC=$?
|
|
344
|
+
python3 "$REPO_DIR/scripts/producer_deathwatch.py" disarm --job-id "$_SA_DW_JOB" \
|
|
345
|
+
>/dev/null 2>&1 || true
|
|
313
346
|
if [ "$RC" -ne 127 ]; then
|
|
314
347
|
break
|
|
315
348
|
fi
|
|
@@ -18,10 +18,14 @@ Safe by construction:
|
|
|
18
18
|
- Best-effort: any single failure is logged and skipped; never raises.
|
|
19
19
|
|
|
20
20
|
Degradation vs a normal cycle: salvaged candidates skip the cycle's post-provider
|
|
21
|
-
top-N selection (so MORE cards, which is fine)
|
|
22
|
-
arm stamp that run-twitter-cycle.sh's plan writer adds after the provider returns
|
|
23
|
-
|
|
24
|
-
|
|
21
|
+
top-N selection (so MORE cards, which is fine), lack the tail-link / experiments
|
|
22
|
+
arm stamp that run-twitter-cycle.sh's plan writer adds after the provider returns,
|
|
23
|
+
and (two-draft schema only) lack assigned_style/assigned_mode, which live in the
|
|
24
|
+
cycle's shell variables, not the model output. The reply text itself IS complete
|
|
25
|
+
end-to-end (_mirror_two_draft_fields backfills reply_text/drafts from
|
|
26
|
+
draft_a_text/draft_b_text for the post-2026-07-07/08 two-draft schema, since the
|
|
27
|
+
model output alone has no reply_text field to check for completeness). A salvaged
|
|
28
|
+
card is strictly better than a lost draft.
|
|
25
29
|
|
|
26
30
|
Usage:
|
|
27
31
|
python3 scripts/salvage_orphaned_prep_results.py # automated (safe age gate)
|
|
@@ -59,15 +63,73 @@ except Exception: # standalone fallbacks
|
|
|
59
63
|
pass
|
|
60
64
|
|
|
61
65
|
|
|
66
|
+
def _mirror_two_draft_fields(candidates):
|
|
67
|
+
"""Backfill reply_text/drafts for two-draft-schema candidates (2026-07-07/08
|
|
68
|
+
redesign) that reach salvage. The normal cycle path (run-twitter-cycle.sh)
|
|
69
|
+
mirrors draft_a_text onto reply_text/engagement_style/drafts right after the
|
|
70
|
+
model returns, but salvage bypasses that shell-side step entirely, so an
|
|
71
|
+
orphaned post-redesign result reached review cards with NEITHER field set.
|
|
72
|
+
The menubar card (s4l_card.py) reads d.get("drafts") first, then falls back
|
|
73
|
+
to d.get("reply_text") or "", so those cards rendered the thread with a
|
|
74
|
+
completely empty editable reply box despite draft_a_text/draft_b_text
|
|
75
|
+
holding real, already-drafted content (root-caused 2026-07-09 via
|
|
76
|
+
candidate 374925 and 4 siblings, all missing 'experiments' too, confirming
|
|
77
|
+
they came through this salvage path rather than a normal cycle write).
|
|
78
|
+
|
|
79
|
+
assigned_style/assigned_mode are deliberately left OUT (not set to None,
|
|
80
|
+
just absent): the picker's per-cycle style assignment lives only in
|
|
81
|
+
run-twitter-cycle.sh's shell variables, not in the model's JSON output, so
|
|
82
|
+
it can't be recovered here. twitter_post_plan.py already has a documented
|
|
83
|
+
fallback for that ("assigned_mode key absent" -> use the plan-level
|
|
84
|
+
assignment, itself None for a salvaged plan), so leaving the keys out is
|
|
85
|
+
the safe, already-supported degradation, same class as the existing
|
|
86
|
+
no-experiments-stamp degradation.
|
|
87
|
+
"""
|
|
88
|
+
for c in candidates:
|
|
89
|
+
if not isinstance(c, dict) or "draft_a_text" not in c or "reply_text" in c:
|
|
90
|
+
continue
|
|
91
|
+
c["reply_text"] = c.get("draft_a_text") or ""
|
|
92
|
+
c["engagement_style"] = c.get("draft_a_style") or ""
|
|
93
|
+
c["new_style"] = c.get("draft_a_new_style")
|
|
94
|
+
if c.get("draft_a_text_en"):
|
|
95
|
+
c["reply_text_en"] = c["draft_a_text_en"]
|
|
96
|
+
draft_b_text = c.get("draft_b_text")
|
|
97
|
+
if not c.get("is_reused_draft") and draft_b_text:
|
|
98
|
+
c["drafts"] = [
|
|
99
|
+
{
|
|
100
|
+
"variant": "a", "text": c.get("draft_a_text") or "",
|
|
101
|
+
"style": c.get("draft_a_style") or "",
|
|
102
|
+
"text_en": c.get("draft_a_text_en"),
|
|
103
|
+
},
|
|
104
|
+
{
|
|
105
|
+
"variant": "b", "text": draft_b_text,
|
|
106
|
+
"style": c.get("draft_b_style") or "",
|
|
107
|
+
"text_en": c.get("draft_b_text_en"),
|
|
108
|
+
},
|
|
109
|
+
]
|
|
110
|
+
|
|
111
|
+
|
|
62
112
|
def _is_prep_result(obj):
|
|
63
|
-
"""True iff obj looks like a twitter-prep result (candidates
|
|
113
|
+
"""True iff obj looks like a twitter-prep result (drafted candidates).
|
|
114
|
+
|
|
115
|
+
"reply_text" was the single-draft field before the 2026-07-07/08 two-draft
|
|
116
|
+
redesign (draft_a_text/draft_b_text per candidate, no single recommended
|
|
117
|
+
reply). Checking only "reply_text" made every post-redesign orphaned
|
|
118
|
+
result silently misclassified as non-prep and marked .skipped instead of
|
|
119
|
+
recovered — the exact "worker drafted but no card" bug this script exists
|
|
120
|
+
to prevent. Accept either field so both old and current schema results
|
|
121
|
+
are recognized.
|
|
122
|
+
"""
|
|
64
123
|
if not isinstance(obj, dict):
|
|
65
124
|
return False
|
|
66
125
|
cands = obj.get("candidates")
|
|
67
126
|
if not isinstance(cands, list) or not cands:
|
|
68
127
|
return False
|
|
69
128
|
c0 = cands[0]
|
|
70
|
-
|
|
129
|
+
if not isinstance(c0, dict):
|
|
130
|
+
return False
|
|
131
|
+
has_text = "reply_text" in c0 or "draft_a_text" in c0
|
|
132
|
+
return has_text and ("candidate_url" in c0 or "candidate_id" in c0)
|
|
71
133
|
|
|
72
134
|
|
|
73
135
|
def main():
|
|
@@ -130,6 +192,7 @@ def main():
|
|
|
130
192
|
pass
|
|
131
193
|
continue
|
|
132
194
|
|
|
195
|
+
_mirror_two_draft_fields(obj["candidates"])
|
|
133
196
|
n = len(obj["candidates"])
|
|
134
197
|
age_min = (now - st.st_mtime) / 60.0
|
|
135
198
|
_plog(f"[salvage] ORPHAN prep result job {job_id}: producer never consumed it "
|
|
@@ -266,6 +266,32 @@ def heal() -> dict:
|
|
|
266
266
|
return summary
|
|
267
267
|
|
|
268
268
|
|
|
269
|
+
def can_create_for_active_account() -> bool:
|
|
270
|
+
"""Read-only: would fix 5 (see heal()) actually be able to create a fresh
|
|
271
|
+
registration right now? True only if the active account (resolved the same
|
|
272
|
+
way heal() does, via schedule_state's config.json lookup) has at least one
|
|
273
|
+
EXISTING session directory to write into — fix 5 never fabricates one.
|
|
274
|
+
Used by callers (the menu bar) to decide whether to offer an automatic
|
|
275
|
+
"restart to finish setup" action or fall back to the manual re-arm prompt,
|
|
276
|
+
BEFORE committing to a restart that would turn out to fix nothing."""
|
|
277
|
+
try:
|
|
278
|
+
for cfg in schedule_state._config_json_paths():
|
|
279
|
+
root = os.path.dirname(cfg)
|
|
280
|
+
uuid = schedule_state._active_account_uuid(cfg)
|
|
281
|
+
if not uuid:
|
|
282
|
+
continue
|
|
283
|
+
account_dir = os.path.join(root, "claude-code-sessions", uuid)
|
|
284
|
+
session_dirs = [
|
|
285
|
+
p for p in glob.glob(os.path.join(account_dir, "*"))
|
|
286
|
+
if os.path.isdir(p)
|
|
287
|
+
]
|
|
288
|
+
if session_dirs:
|
|
289
|
+
return True
|
|
290
|
+
except Exception:
|
|
291
|
+
pass
|
|
292
|
+
return False
|
|
293
|
+
|
|
294
|
+
|
|
269
295
|
def main() -> int:
|
|
270
296
|
out = heal()
|
|
271
297
|
print(json.dumps(out))
|