@m13v/s4l 1.7.4-rc.4 → 1.7.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +134 -101
- package/bin/cli.js +23 -0
- package/mcp/dist/index.js +24 -5
- package/mcp/dist/runtime.js +110 -0
- package/mcp/dist/version.js +163 -54
- package/mcp/dist/version.json +2 -2
- package/mcp/manifest.json +1 -1
- package/mcp/menubar/s4l_card.py +55 -19
- package/mcp/menubar/s4l_menubar.py +179 -56
- package/mcp/menubar/s4l_state.py +41 -1
- package/mcp/package.json +1 -1
- package/package.json +4 -2
- package/scripts/active_experiments.py +30 -6
- package/scripts/autopilot_stall_watch.py +185 -63
- package/scripts/cdp_ready_check.py +94 -0
- package/scripts/engage_twitter_helper.py +83 -17
- package/scripts/engagement_styles.py +221 -26
- package/scripts/enrich_reply_parents.py +68 -22
- package/scripts/feedback_digest.py +24 -1
- package/scripts/generate_daily_human_style.py +31 -7
- package/scripts/invent_styles.py +103 -26
- package/scripts/linkedin_cadence.py +234 -0
- package/scripts/linkedin_killswitch.py +43 -0
- package/scripts/memory_snapshot.py +57 -0
- package/scripts/patches/browser-harness-loopback-cdp-proxy.patch +118 -0
- package/scripts/recent_self_posts.py +6 -0
- package/scripts/reddit_browser.py +57 -3
- package/scripts/s4l_box_update.sh +20 -1
- package/scripts/s4l_mode.py +3 -4
- package/scripts/scan_x_profile.py +25 -7
- package/scripts/schedule_state.py +23 -0
- package/scripts/scheduled_task_selfheal.py +213 -16
- package/scripts/setup_twitter_auth.py +6 -1
- package/scripts/snapshot.py +179 -61
- package/scripts/stall_guard.py +70 -0
- package/scripts/test_scheduled_task_selfheal.py +121 -0
- package/scripts/top_performers.py +34 -0
- package/scripts/twitter_browser.py +146 -8
- package/scripts/twitter_post_plan.py +10 -1
- package/scripts/twitter_scan.py +26 -0
- package/scripts/watchdog_hung_runs.py +117 -3
- package/skill/amplitude-24h-signups.sh +1 -1
- package/skill/audit-dm-staleness.sh +1 -1
- package/skill/audit-linkedin.sh +1 -1
- package/skill/audit-reddit-resurrect.sh +1 -1
- package/skill/audit.sh +5 -5
- package/skill/check-web-chats.sh +1 -1
- package/skill/dm-outreach-linkedin.sh +4 -4
- package/skill/dm-outreach-reddit.sh +1 -1
- package/skill/dm-outreach-twitter.sh +1 -1
- package/skill/engage-dm-replies.sh +16 -3
- package/skill/engage-linkedin.sh +2 -2
- package/skill/engage-moltbook.sh +1 -1
- package/skill/engage-twitter.sh +4 -4
- package/skill/github-engage.sh +1 -1
- package/skill/lib/browser-launch.sh +54 -0
- package/skill/lib/linkedin-backend.sh +22 -15
- package/skill/lib/reddit-backend.sh +17 -11
- package/skill/lib/twitter-backend.sh +137 -16
- package/skill/link-edit-github.sh +1 -1
- package/skill/link-edit-moltbook.sh +1 -1
- package/skill/link-edit-reddit.sh +1 -1
- package/skill/linkedin-cadence.sh +20 -0
- package/skill/linkedin-presence.sh +1 -1
- package/skill/lock.sh +30 -6
- package/skill/precompute-stats.sh +1 -1
- package/skill/refresh-instagram-tokens.sh +1 -1
- package/skill/refresh-twitter-following.sh +1 -1
- package/skill/run-generate-daily-style.sh +1 -1
- package/skill/run-instagram-daily.sh +1 -1
- package/skill/run-instagram-render.sh +1 -1
- package/skill/run-linkedin-launchd.sh +1 -1
- package/skill/run-linkedin-unipile.sh +1 -1
- package/skill/run-linkedin.sh +1 -1
- package/skill/run-reddit-search.sh +2 -2
- package/skill/run-reddit-threads.sh +1 -1
- package/skill/run-twitter-cycle.sh +190 -55
- package/skill/scan-instagram-replies.sh +1 -1
- package/skill/sentry-digest.sh +1 -1
- package/skill/stats-instagram.sh +1 -1
- package/skill/stats-linkedin.sh +2 -2
- package/skill/stats.sh +1 -1
- package/skill/sweep-link-clicks.sh +1 -1
- package/scripts/_lock_preempt_test.py +0 -60
- package/scripts/saps_activity.py +0 -18
- package/scripts/saps_mode.py +0 -18
|
@@ -14,7 +14,8 @@ autopilot" item). This watcher is the part the user can't see: a fleet-side aler
|
|
|
14
14
|
so a sustained stall pages us even when nobody is looking at the menu bar.
|
|
15
15
|
|
|
16
16
|
Design mirrors the stall signal in mcp/menubar/s4l_menubar.py (_autopilot_stalled)
|
|
17
|
-
and mcp/src/index.ts (autopilotStalled)
|
|
17
|
+
and mcp/src/index.ts (autopilotStalled); the thresholds are shared via
|
|
18
|
+
schedule_state.py (index.ts mirrors by hand):
|
|
18
19
|
stalled = the autopilot is configured (a complete worker set's SKILL.md
|
|
19
20
|
files present — see WORKER_TASK_SETS)
|
|
20
21
|
AND a draft job has sat unclaimed in pending/ past STALL_SECONDS.
|
|
@@ -46,19 +47,34 @@ import identity # noqa: E402 (lives next to this file in scripts/)
|
|
|
46
47
|
|
|
47
48
|
s4l_env.mirror()
|
|
48
49
|
|
|
49
|
-
#
|
|
50
|
-
STALL_SECONDS =
|
|
51
|
-
#
|
|
52
|
-
|
|
53
|
-
# or crashed. Must be generous enough to clear the longest real drafting turn so a
|
|
54
|
-
# healthy run never trips it. Keep in sync with AUTOPILOT_RUNNING_STALL_SECONDS
|
|
55
|
-
# (menubar). See _oldest_running_age.
|
|
56
|
-
RUNNING_STALL_SECONDS = 900
|
|
50
|
+
# Stall thresholds live in schedule_state.py (HEALTHY_DRAIN_MAX_SECONDS and the
|
|
51
|
+
# names derived from it) — retune there, not here. STALL_SECONDS = pending job
|
|
52
|
+
# unclaimed; RUNNING_STALL_SECONDS = claimed but wedged (see _oldest_running_age).
|
|
53
|
+
from schedule_state import STALL_SECONDS, RUNNING_STALL_SECONDS # noqa: E402
|
|
57
54
|
# Require the stall to persist this many consecutive checks before paging, so a
|
|
58
55
|
# transient slow claim (e.g. right after a Claude restart) doesn't false-alarm.
|
|
59
56
|
# At StartInterval 120 that is ~6 min of continuous stall.
|
|
60
57
|
ALERT_AFTER = 3
|
|
61
58
|
|
|
59
|
+
# --- Host-sleep awareness (2026-07-13, Sentry S4L-4B) ------------------------
|
|
60
|
+
# launchd does NOT fire StartInterval jobs while the host is suspended, so the
|
|
61
|
+
# wall-clock gap between consecutive ticks of THIS watchdog is a reliable sleep
|
|
62
|
+
# detector: a gap of 3+ intervals means the box was asleep (or powered off),
|
|
63
|
+
# not stalled. S4L-4B (Nhat's MacBook Air, 2026-07-11): laptop slept mid-cycle,
|
|
64
|
+
# one producer enqueue timeout latched, no cycle ran to clear it, and the page
|
|
65
|
+
# fired with pending=0/running=0 and a bogus "account change?" cause while the
|
|
66
|
+
# telemetry showed 9 samples in a 2h window (40-min gaps). Laptops that sleep
|
|
67
|
+
# between cycles would re-trigger that page forever.
|
|
68
|
+
TICK_INTERVAL_SECONDS = 120 # keep in sync with launchd StartInterval
|
|
69
|
+
SLEEP_GAP_SECONDS = TICK_INTERVAL_SECONDS * 3 # missed >=3 ticks -> host slept
|
|
70
|
+
# For this long after a detected sleep gap, pre-existing latch/age signals are
|
|
71
|
+
# treated as sleep-tainted: they must be corroborated by actual queued work
|
|
72
|
+
# (pending/running > 0) or the outcome-level batches_stuck backstop to page.
|
|
73
|
+
SLEEP_GAP_RECENT_SECONDS = 1800
|
|
74
|
+
# A worker-task transcript modified this recently proves the routines are
|
|
75
|
+
# firing, which rules out the "orphaned routines / account change" cause.
|
|
76
|
+
WORKER_RECENT_SECONDS = 1800
|
|
77
|
+
|
|
62
78
|
# A box counts as configured when ANY complete worker set has its SKILL.md on
|
|
63
79
|
# disk: the universal type-blind worker, its short-lived staging predecessor, or
|
|
64
80
|
# the legacy per-type pair. Keep in sync with WORKER_TASK_SETS in
|
|
@@ -142,6 +158,46 @@ def _recent_rate_limit(window: int = 1200) -> bool:
|
|
|
142
158
|
return False
|
|
143
159
|
|
|
144
160
|
|
|
161
|
+
def _worker_ran_recently(window: int = WORKER_RECENT_SECONDS) -> bool:
|
|
162
|
+
"""True if any s4l-worker scheduled-task transcript was written in the last
|
|
163
|
+
`window` seconds — direct on-disk proof the worker routines are firing, so
|
|
164
|
+
a stall can NOT be the orphaned-routines / account-change shape. Same
|
|
165
|
+
transcript bucket _recent_rate_limit reads."""
|
|
166
|
+
try:
|
|
167
|
+
now = time.time()
|
|
168
|
+
for f in glob.glob(os.path.expanduser("~/.claude/projects/*s4l-worker*/*.jsonl")):
|
|
169
|
+
try:
|
|
170
|
+
if (now - os.path.getmtime(f)) <= window:
|
|
171
|
+
return True
|
|
172
|
+
except OSError:
|
|
173
|
+
continue
|
|
174
|
+
except Exception:
|
|
175
|
+
pass
|
|
176
|
+
return False
|
|
177
|
+
|
|
178
|
+
|
|
179
|
+
def _api_reachable(timeout: float = 3.0) -> bool:
|
|
180
|
+
"""True if the S4L API host answers HTTP at all. ANY HTTP status counts as
|
|
181
|
+
reachable (a 4xx/5xx still proves the network path is up); only a socket /
|
|
182
|
+
URL error means offline. Used to hold the Sentry page while the box has no
|
|
183
|
+
network — an offline box is a different, self-healing condition, and the
|
|
184
|
+
page would be misleading (plus the event can't ship anyway). The stall
|
|
185
|
+
episode keeps counting, so if it's still stalled when connectivity returns,
|
|
186
|
+
the page fires then."""
|
|
187
|
+
import urllib.error
|
|
188
|
+
import urllib.request
|
|
189
|
+
|
|
190
|
+
base = os.environ.get("AUTOPOSTER_API_BASE", "https://s4l.ai").rstrip("/")
|
|
191
|
+
try:
|
|
192
|
+
req = urllib.request.Request(base, method="HEAD")
|
|
193
|
+
urllib.request.urlopen(req, timeout=timeout)
|
|
194
|
+
return True
|
|
195
|
+
except urllib.error.HTTPError:
|
|
196
|
+
return True # server answered; network is up
|
|
197
|
+
except Exception:
|
|
198
|
+
return False
|
|
199
|
+
|
|
200
|
+
|
|
145
201
|
BATCH_PROGRESSION_MIN_BATCHES = 5
|
|
146
202
|
BATCH_PROGRESSED_PHASES = {"phase2b-gen", "phase2b-post"}
|
|
147
203
|
|
|
@@ -348,6 +404,30 @@ def _report_queue_health_sample(
|
|
|
348
404
|
|
|
349
405
|
|
|
350
406
|
def main() -> int:
|
|
407
|
+
now = time.time()
|
|
408
|
+
st = _read_state()
|
|
409
|
+
|
|
410
|
+
# Sleep detection: launchd skips ticks while the host is suspended, so a
|
|
411
|
+
# wall-clock gap between this tick and the previous one of >= 3 intervals
|
|
412
|
+
# means the box slept (see the S4L-4B block by the constants above).
|
|
413
|
+
last_tick_at = st.get("last_tick_at")
|
|
414
|
+
last_sleep_gap_at = st.get("last_sleep_gap_at")
|
|
415
|
+
if last_tick_at is not None:
|
|
416
|
+
tick_gap = now - float(last_tick_at)
|
|
417
|
+
if tick_gap > SLEEP_GAP_SECONDS:
|
|
418
|
+
last_sleep_gap_at = now
|
|
419
|
+
sys.stderr.write(
|
|
420
|
+
f"[stall-watch] tick gap {int(tick_gap)}s (> {SLEEP_GAP_SECONDS}s) — "
|
|
421
|
+
"host was asleep/off; treating stall signals as sleep-tainted for "
|
|
422
|
+
f"{SLEEP_GAP_RECENT_SECONDS}s\n"
|
|
423
|
+
)
|
|
424
|
+
sleep_gap_recent = bool(
|
|
425
|
+
last_sleep_gap_at is not None
|
|
426
|
+
and (now - float(last_sleep_gap_at)) < SLEEP_GAP_RECENT_SECONDS
|
|
427
|
+
)
|
|
428
|
+
|
|
429
|
+
pending = _pending_count()
|
|
430
|
+
running = _running_count()
|
|
351
431
|
age = _oldest_pending_age()
|
|
352
432
|
run_age = _oldest_running_age()
|
|
353
433
|
timeouts = _consecutive_timeouts()
|
|
@@ -362,8 +442,16 @@ def main() -> int:
|
|
|
362
442
|
# never claimed), (3) running-age (job claimed then wedged mid-run) — (3) is
|
|
363
443
|
# the only one of the first three that catches a worker dying after it picked
|
|
364
444
|
# up the job — (4) batches_stuck, the outcome-level backstop above.
|
|
445
|
+
# Latch sanity gate (S4L-4B): consecutive_timeouts is durable and only
|
|
446
|
+
# clears on a successful drain, so ONE timed-out enqueue keeps a box
|
|
447
|
+
# looking stalled for as long as no cycle runs — even with a provably idle
|
|
448
|
+
# queue (pending=0, running=0). A single timeout only counts when there is
|
|
449
|
+
# actual queued work to corroborate it; an idle-queue latch needs >= 2.
|
|
450
|
+
# NOTE: deliberately stricter than the menubar/_index.ts stall HINT
|
|
451
|
+
# (timeouts >= 1) — a UI hint may over-trigger, a fleet page must not.
|
|
452
|
+
timeouts_signal = timeouts >= 2 or (timeouts >= 1 and (pending > 0 or running > 0))
|
|
365
453
|
stalled = configured and (
|
|
366
|
-
|
|
454
|
+
timeouts_signal
|
|
367
455
|
or (age is not None and age > STALL_SECONDS)
|
|
368
456
|
or (run_age is not None and run_age > RUNNING_STALL_SECONDS)
|
|
369
457
|
or batches_stuck
|
|
@@ -373,13 +461,17 @@ def main() -> int:
|
|
|
373
461
|
# episode resets and a LATER real stall (orphaned routines) still alerts.
|
|
374
462
|
if stalled and _recent_rate_limit():
|
|
375
463
|
stalled = False
|
|
464
|
+
# Sleep suppression: right after a wake, the latch (and any pre-nap ages)
|
|
465
|
+
# predate the sleep, not a worker failure. Require corroboration by real
|
|
466
|
+
# queued work or the outcome-level backstop before calling it a stall.
|
|
467
|
+
if stalled and sleep_gap_recent and not (pending > 0 or running > 0 or batches_stuck):
|
|
468
|
+
stalled = False
|
|
376
469
|
|
|
377
470
|
# Record this tick regardless of outcome — see _report_queue_health_sample.
|
|
378
471
|
_report_queue_health_sample(
|
|
379
|
-
|
|
472
|
+
pending, running, timeouts, age, run_age, stalled, batches_stuck
|
|
380
473
|
)
|
|
381
474
|
|
|
382
|
-
st = _read_state()
|
|
383
475
|
consecutive = int(st.get("consecutive", 0))
|
|
384
476
|
alerted = bool(st.get("alerted", False))
|
|
385
477
|
# first_seen_at: first check this episode looked stalled at all (predates
|
|
@@ -390,6 +482,10 @@ def main() -> int:
|
|
|
390
482
|
first_seen_at = st.get("first_seen_at")
|
|
391
483
|
alerted_at = st.get("alerted_at")
|
|
392
484
|
|
|
485
|
+
# Tick bookkeeping persisted on EVERY exit path — the sleep detector needs
|
|
486
|
+
# an unbroken last_tick_at chain even (especially) while healthy.
|
|
487
|
+
tick_state = {"last_tick_at": now, "last_sleep_gap_at": last_sleep_gap_at}
|
|
488
|
+
|
|
393
489
|
if not stalled:
|
|
394
490
|
# Recovered (or never stalled) -> reset the episode so the next stall pages.
|
|
395
491
|
if consecutive or alerted:
|
|
@@ -397,7 +493,7 @@ def main() -> int:
|
|
|
397
493
|
total_duration = time.time() - float(first_seen_at)
|
|
398
494
|
paged_duration = (time.time() - float(alerted_at)) if alerted_at else None
|
|
399
495
|
_report_recovery(total_duration, paged_duration)
|
|
400
|
-
|
|
496
|
+
_write_state({**tick_state, "consecutive": 0, "alerted": False})
|
|
401
497
|
return 0
|
|
402
498
|
|
|
403
499
|
consecutive += 1
|
|
@@ -411,60 +507,86 @@ def main() -> int:
|
|
|
411
507
|
# the fallback is the classic orphaned-routine case.
|
|
412
508
|
wedged_inflight = run_age is not None and run_age > RUNNING_STALL_SECONDS
|
|
413
509
|
if consecutive >= ALERT_AFTER and not alerted:
|
|
414
|
-
|
|
415
|
-
|
|
416
|
-
|
|
417
|
-
|
|
418
|
-
|
|
419
|
-
"a worker claimed a draft job and then died mid-run (claude -p child "
|
|
420
|
-
"never came up / crashed)"
|
|
421
|
-
)
|
|
422
|
-
stall_shape = "inflight_wedged"
|
|
423
|
-
elif batches_stuck:
|
|
424
|
-
cause = (
|
|
425
|
-
f"last {BATCH_PROGRESSION_MIN_BATCHES} twitter_batches all failed to "
|
|
426
|
-
"reach phase2b-gen — drafting is not actually happening even if the "
|
|
427
|
-
"queue-level counters look ambiguous"
|
|
428
|
-
)
|
|
429
|
-
stall_shape = "batches_not_progressing"
|
|
430
|
-
else:
|
|
431
|
-
cause = "scheduled-task routines likely orphaned — Claude Desktop account change?"
|
|
432
|
-
stall_shape = "not_draining"
|
|
433
|
-
# Our own staging/QA/dev boxes (identity.is_internal_install) get set
|
|
434
|
-
# up and rebuilt with nobody actively feeding the queue, which looks
|
|
435
|
-
# identical to a real stall on the signals above. Downgrade those to
|
|
436
|
-
# warning instead of error so they don't page as a customer incident
|
|
437
|
-
# (the digest only scans error/fatal) while still leaving a Sentry
|
|
438
|
-
# record if we ever need to look one up by hand.
|
|
439
|
-
is_internal = identity.is_internal_install()
|
|
440
|
-
sentry.capture_message(
|
|
441
|
-
"social-autoposter autopilot stalled: draft jobs are not being "
|
|
442
|
-
f"drained ({cause}). producer consecutive timeouts={timeouts}, "
|
|
443
|
-
f"oldest pending job age={age_str}, oldest in-flight (running) job "
|
|
444
|
-
f"age={run_age_str}, sustained {consecutive} checks.",
|
|
445
|
-
level=("warning" if is_internal else "error"),
|
|
446
|
-
tags={
|
|
447
|
-
"component": "autopilot",
|
|
448
|
-
"issue": "stall",
|
|
449
|
-
"stall_shape": stall_shape,
|
|
450
|
-
"consecutive_timeouts": str(timeouts),
|
|
451
|
-
"oldest_pending_age_s": str(int(age)) if age is not None else "",
|
|
452
|
-
"oldest_running_age_s": str(int(run_age)) if run_age is not None else "",
|
|
453
|
-
"batches_stuck": str(batches_stuck),
|
|
454
|
-
"internal_install": str(is_internal),
|
|
455
|
-
},
|
|
456
|
-
)
|
|
457
|
-
sentry.flush()
|
|
458
|
-
except Exception:
|
|
459
|
-
# No Sentry (helper/SDK missing) -> at least leave a local breadcrumb.
|
|
510
|
+
if not _api_reachable():
|
|
511
|
+
# No network: an offline box is a different, self-healing condition,
|
|
512
|
+
# the page would misattribute it to the worker (and the event can't
|
|
513
|
+
# ship anyway). Keep the episode counting so a stall that survives
|
|
514
|
+
# the outage still pages the tick connectivity returns.
|
|
460
515
|
sys.stderr.write(
|
|
461
|
-
f"[stall-watch]
|
|
462
|
-
|
|
516
|
+
f"[stall-watch] stall persisted {consecutive} checks but the API is "
|
|
517
|
+
"unreachable (box offline?); deferring the page until network returns\n"
|
|
463
518
|
)
|
|
464
|
-
|
|
465
|
-
|
|
519
|
+
else:
|
|
520
|
+
try:
|
|
521
|
+
sentry = _sentry()
|
|
522
|
+
sentry.init()
|
|
523
|
+
if wedged_inflight:
|
|
524
|
+
cause = (
|
|
525
|
+
"a worker claimed a draft job and then died mid-run (claude -p child "
|
|
526
|
+
"never came up / crashed)"
|
|
527
|
+
)
|
|
528
|
+
stall_shape = "inflight_wedged"
|
|
529
|
+
elif batches_stuck:
|
|
530
|
+
cause = (
|
|
531
|
+
f"last {BATCH_PROGRESSION_MIN_BATCHES} twitter_batches all failed to "
|
|
532
|
+
"reach phase2b-gen — drafting is not actually happening even if the "
|
|
533
|
+
"queue-level counters look ambiguous"
|
|
534
|
+
)
|
|
535
|
+
stall_shape = "batches_not_progressing"
|
|
536
|
+
elif _worker_ran_recently():
|
|
537
|
+
# On-disk transcripts prove the worker routines ARE firing, so
|
|
538
|
+
# this cannot be the orphaned-routines shape (S4L-4B paged with
|
|
539
|
+
# exactly that bogus cause while the registry sample showed the
|
|
540
|
+
# task had run seconds earlier).
|
|
541
|
+
cause = (
|
|
542
|
+
"producer timeout latch, but worker-task transcripts show recent "
|
|
543
|
+
"runs — routines are firing; suspect a transient enqueue timeout"
|
|
544
|
+
+ (" or a host sleep gap" if sleep_gap_recent else "")
|
|
545
|
+
+ ", NOT orphaned routines"
|
|
546
|
+
)
|
|
547
|
+
stall_shape = "latch_worker_alive"
|
|
548
|
+
else:
|
|
549
|
+
cause = "scheduled-task routines likely orphaned — Claude Desktop account change?"
|
|
550
|
+
stall_shape = "not_draining"
|
|
551
|
+
# Our own staging/QA/dev boxes (identity.is_internal_install) get set
|
|
552
|
+
# up and rebuilt with nobody actively feeding the queue, which looks
|
|
553
|
+
# identical to a real stall on the signals above. Downgrade those to
|
|
554
|
+
# warning instead of error so they don't page as a customer incident
|
|
555
|
+
# (the digest only scans error/fatal) while still leaving a Sentry
|
|
556
|
+
# record if we ever need to look one up by hand.
|
|
557
|
+
is_internal = identity.is_internal_install()
|
|
558
|
+
sentry.capture_message(
|
|
559
|
+
"social-autoposter autopilot stalled: draft jobs are not being "
|
|
560
|
+
f"drained ({cause}). producer consecutive timeouts={timeouts}, "
|
|
561
|
+
f"oldest pending job age={age_str}, oldest in-flight (running) job "
|
|
562
|
+
f"age={run_age_str}, sustained {consecutive} checks.",
|
|
563
|
+
level=("warning" if is_internal else "error"),
|
|
564
|
+
tags={
|
|
565
|
+
"component": "autopilot",
|
|
566
|
+
"issue": "stall",
|
|
567
|
+
"stall_shape": stall_shape,
|
|
568
|
+
"consecutive_timeouts": str(timeouts),
|
|
569
|
+
"oldest_pending_age_s": str(int(age)) if age is not None else "",
|
|
570
|
+
"oldest_running_age_s": str(int(run_age)) if run_age is not None else "",
|
|
571
|
+
"batches_stuck": str(batches_stuck),
|
|
572
|
+
"internal_install": str(is_internal),
|
|
573
|
+
"pending": str(pending),
|
|
574
|
+
"running": str(running),
|
|
575
|
+
"recent_sleep_gap": str(sleep_gap_recent),
|
|
576
|
+
},
|
|
577
|
+
)
|
|
578
|
+
sentry.flush()
|
|
579
|
+
except Exception:
|
|
580
|
+
# No Sentry (helper/SDK missing) -> at least leave a local breadcrumb.
|
|
581
|
+
sys.stderr.write(
|
|
582
|
+
f"[stall-watch] autopilot stalled (timeouts={timeouts}, "
|
|
583
|
+
f"pending_age={age_str}, running_age={run_age_str}) but Sentry report failed\n"
|
|
584
|
+
)
|
|
585
|
+
alerted = True
|
|
586
|
+
alerted_at = time.time()
|
|
466
587
|
|
|
467
588
|
_write_state({
|
|
589
|
+
**tick_state,
|
|
468
590
|
"consecutive": consecutive,
|
|
469
591
|
"alerted": alerted,
|
|
470
592
|
"first_seen_at": first_seen_at,
|
|
@@ -0,0 +1,94 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""Real CDP readiness probe for the harness Chrome.
|
|
3
|
+
|
|
4
|
+
Exit 0 when a full Playwright connect_over_cdp handshake completes against the
|
|
5
|
+
given CDP URL, 1 when it does not. /json/version alone is a LIVENESS check: a
|
|
6
|
+
wedged Chrome (process alive, HTTP answering, websocket upgrade completing,
|
|
7
|
+
but the browser loop never servicing the CDP session) passes it, and every
|
|
8
|
+
downstream attach then eats Playwright's 180s default timeout while holding
|
|
9
|
+
the browser lock (S4L-4H, Karol 2026-07-11; identical wedge locally 2026-07-09,
|
|
10
|
+
twice, same Chrome instance both times).
|
|
11
|
+
|
|
12
|
+
Usage: cdp_ready_check.py [CDP_URL] [TIMEOUT_MS]
|
|
13
|
+
|
|
14
|
+
Prints a one-line JSON verdict to stdout so the caller can persist it
|
|
15
|
+
(twitter-backend.sh writes it to skill/logs/cdp-health.json, which
|
|
16
|
+
memory_snapshot.py carries onto the per-minute heartbeat sample).
|
|
17
|
+
|
|
18
|
+
Falls back to an HTTP-only probe when playwright is not importable under the
|
|
19
|
+
invoking interpreter, so a bare python3 caller degrades to the legacy
|
|
20
|
+
liveness behavior instead of hard-failing.
|
|
21
|
+
"""
|
|
22
|
+
import json
|
|
23
|
+
import sys
|
|
24
|
+
import time
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def main() -> int:
|
|
28
|
+
url = (sys.argv[1] if len(sys.argv) > 1 else "http://127.0.0.1:9555").rstrip("/")
|
|
29
|
+
timeout_ms = int(sys.argv[2]) if len(sys.argv) > 2 else 8000
|
|
30
|
+
t0 = time.time()
|
|
31
|
+
try:
|
|
32
|
+
from playwright.sync_api import sync_playwright
|
|
33
|
+
except Exception:
|
|
34
|
+
import urllib.request
|
|
35
|
+
try:
|
|
36
|
+
# ProxyHandler({}): loopback CDP must never route through a proxy.
|
|
37
|
+
# macOS system proxy settings leak into urllib's default opener, and
|
|
38
|
+
# a box-wide forwarder 403s 127.0.0.1 probes (2026-07-13 root cause
|
|
39
|
+
# of the "wedged Chrome" misdiagnosis).
|
|
40
|
+
opener = urllib.request.build_opener(urllib.request.ProxyHandler({}))
|
|
41
|
+
opener.open(f"{url}/json/version", timeout=3)
|
|
42
|
+
print(json.dumps({"ready": True, "mode": "http-only"}))
|
|
43
|
+
return 0
|
|
44
|
+
except Exception as e:
|
|
45
|
+
print(json.dumps({
|
|
46
|
+
"ready": False, "mode": "http-only", "error": str(e)[:120],
|
|
47
|
+
}))
|
|
48
|
+
return 1
|
|
49
|
+
try:
|
|
50
|
+
with sync_playwright() as p:
|
|
51
|
+
browser = p.chromium.connect_over_cdp(url, timeout=timeout_ms)
|
|
52
|
+
n_contexts = len(browser.contexts)
|
|
53
|
+
# Renderer-liveness sweep (2026-07-14). A tab whose RENDERER
|
|
54
|
+
# crashed ("Aw, Snap", error code 5) keeps its normal title/url in
|
|
55
|
+
# every CDP listing and the browser-level handshake stays green,
|
|
56
|
+
# so it sat visibly dead for 20+ minutes with nothing entitled to
|
|
57
|
+
# touch it — below the wedge detector (browser-level) and the
|
|
58
|
+
# stall guard (scan-progress-level). Probe each page with a
|
|
59
|
+
# trivial evaluate; on failure reload the tab IN PLACE, which
|
|
60
|
+
# spawns a fresh renderer. No kill, no new window, no focus
|
|
61
|
+
# change. Best-effort: revival must never fail the readiness
|
|
62
|
+
# verdict the wedge gate depends on.
|
|
63
|
+
revived = 0
|
|
64
|
+
for ctx in browser.contexts:
|
|
65
|
+
for page in ctx.pages:
|
|
66
|
+
try:
|
|
67
|
+
page.set_default_timeout(4000)
|
|
68
|
+
page.evaluate("1")
|
|
69
|
+
except Exception:
|
|
70
|
+
try:
|
|
71
|
+
page.reload(timeout=15000, wait_until="commit")
|
|
72
|
+
revived += 1
|
|
73
|
+
except Exception:
|
|
74
|
+
pass
|
|
75
|
+
browser.close()
|
|
76
|
+
out = {
|
|
77
|
+
"ready": True, "mode": "cdp", "contexts": n_contexts,
|
|
78
|
+
"elapsed_s": round(time.time() - t0, 2),
|
|
79
|
+
}
|
|
80
|
+
if revived:
|
|
81
|
+
out["revived"] = revived
|
|
82
|
+
print(json.dumps(out))
|
|
83
|
+
return 0
|
|
84
|
+
except Exception as e:
|
|
85
|
+
print(json.dumps({
|
|
86
|
+
"ready": False, "mode": "cdp",
|
|
87
|
+
"elapsed_s": round(time.time() - t0, 2),
|
|
88
|
+
"error": str(e)[:200].replace("\n", " "),
|
|
89
|
+
}))
|
|
90
|
+
return 1
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
if __name__ == "__main__":
|
|
94
|
+
raise SystemExit(main())
|
|
@@ -144,6 +144,59 @@ def _render_media_block(media) -> str:
|
|
|
144
144
|
)
|
|
145
145
|
|
|
146
146
|
|
|
147
|
+
def _build_chain_block(row) -> str:
|
|
148
|
+
"""Conversation chain reconstructed from replies.parent_reply_id linkage.
|
|
149
|
+
|
|
150
|
+
Walks ancestors bottom-up via GET /api/v1/replies/:id and renders the
|
|
151
|
+
chain root-first, each hop showing the inbound comment and (when we
|
|
152
|
+
responded) our reply. Empty string when the row has no parent linkage:
|
|
153
|
+
the root post itself already rides PENDING_DATA via the posts JOIN
|
|
154
|
+
(our_content / thread_title), so a chain block would add nothing.
|
|
155
|
+
|
|
156
|
+
Like counterparty_history_block, the block is self-titled and lands
|
|
157
|
+
inline in PENDING_DATA — no shell-side prompt change needed.
|
|
158
|
+
"""
|
|
159
|
+
parent_id = row.get("parent_reply_id")
|
|
160
|
+
if not parent_id:
|
|
161
|
+
return ""
|
|
162
|
+
hops = []
|
|
163
|
+
seen = set()
|
|
164
|
+
cur = parent_id
|
|
165
|
+
for _ in range(10):
|
|
166
|
+
if not cur or cur in seen:
|
|
167
|
+
break
|
|
168
|
+
seen.add(cur)
|
|
169
|
+
try:
|
|
170
|
+
resp = api_get(f"/api/v1/replies/{cur}")
|
|
171
|
+
except Exception:
|
|
172
|
+
break
|
|
173
|
+
r = (resp.get("data") or {}).get("reply") or {}
|
|
174
|
+
if not r:
|
|
175
|
+
break
|
|
176
|
+
hops.append(r)
|
|
177
|
+
cur = r.get("parent_reply_id")
|
|
178
|
+
if not hops:
|
|
179
|
+
return ""
|
|
180
|
+
|
|
181
|
+
def _one_line(text):
|
|
182
|
+
return " ".join((text or "").split())
|
|
183
|
+
|
|
184
|
+
lines = [
|
|
185
|
+
"## Conversation chain (reconstructed from our DB; root first — "
|
|
186
|
+
"the row you are drafting for replies to the LAST message)"
|
|
187
|
+
]
|
|
188
|
+
for r in reversed(hops):
|
|
189
|
+
lines.append(f"@{r.get('their_author') or '?'}: {_one_line(r.get('their_content'))}")
|
|
190
|
+
ours = _one_line(r.get("our_reply_content"))
|
|
191
|
+
if ours:
|
|
192
|
+
lines.append(f" our reply: {ours}")
|
|
193
|
+
lines.append(
|
|
194
|
+
f"@{row.get('their_author') or '?'}: {_one_line(row.get('their_content'))}"
|
|
195
|
+
" <- you are replying to this"
|
|
196
|
+
)
|
|
197
|
+
return "\n".join(lines)
|
|
198
|
+
|
|
199
|
+
|
|
147
200
|
def cmd_pending_data(batch_size: int) -> int:
|
|
148
201
|
try:
|
|
149
202
|
from account_resolver import resolve as _resolve_account # noqa: WPS433
|
|
@@ -173,38 +226,50 @@ def cmd_pending_data(batch_size: int) -> int:
|
|
|
173
226
|
# top slot then and get enriched.
|
|
174
227
|
ENRICH_TOP_N = 60
|
|
175
228
|
history_blocks = [""] * len(rows)
|
|
229
|
+
chain_blocks = [""] * len(rows)
|
|
176
230
|
try:
|
|
177
231
|
from concurrent.futures import ThreadPoolExecutor
|
|
178
232
|
from counterparty_history import get_counterparty_history_block
|
|
179
233
|
|
|
180
234
|
def _enrich(r):
|
|
181
235
|
author = r.get("their_author")
|
|
182
|
-
|
|
183
|
-
|
|
236
|
+
history = ""
|
|
237
|
+
if author:
|
|
238
|
+
try:
|
|
239
|
+
_disengage, history = get_counterparty_history_block(
|
|
240
|
+
platform="x",
|
|
241
|
+
author=author,
|
|
242
|
+
current_post_id=r.get("post_id"),
|
|
243
|
+
current_reply_id=r.get("id"),
|
|
244
|
+
)
|
|
245
|
+
history = history or ""
|
|
246
|
+
except Exception as e:
|
|
247
|
+
print(
|
|
248
|
+
f"[engage_twitter_helper] counterparty_history failed "
|
|
249
|
+
f"for @{author}: {e}",
|
|
250
|
+
file=sys.stderr,
|
|
251
|
+
)
|
|
184
252
|
try:
|
|
185
|
-
|
|
186
|
-
platform="x",
|
|
187
|
-
author=author,
|
|
188
|
-
current_post_id=r.get("post_id"),
|
|
189
|
-
current_reply_id=r.get("id"),
|
|
190
|
-
)
|
|
191
|
-
return block or ""
|
|
253
|
+
chain = _build_chain_block(r)
|
|
192
254
|
except Exception as e:
|
|
193
255
|
print(
|
|
194
|
-
f"[engage_twitter_helper]
|
|
195
|
-
f"for
|
|
256
|
+
f"[engage_twitter_helper] chain block failed "
|
|
257
|
+
f"for reply {r.get('id')}: {e}",
|
|
196
258
|
file=sys.stderr,
|
|
197
259
|
)
|
|
198
|
-
|
|
260
|
+
chain = ""
|
|
261
|
+
return (history, chain)
|
|
199
262
|
|
|
200
263
|
top_rows = rows[:ENRICH_TOP_N]
|
|
201
264
|
with ThreadPoolExecutor(max_workers=8) as ex:
|
|
202
|
-
for idx,
|
|
203
|
-
history_blocks[idx] =
|
|
265
|
+
for idx, (history, chain) in enumerate(ex.map(_enrich, top_rows)):
|
|
266
|
+
history_blocks[idx] = history
|
|
267
|
+
chain_blocks[idx] = chain
|
|
204
268
|
non_empty = sum(1 for b in history_blocks if b)
|
|
269
|
+
chains_non_empty = sum(1 for b in chain_blocks if b)
|
|
205
270
|
print(
|
|
206
|
-
f"[engage_twitter_helper]
|
|
207
|
-
f"
|
|
271
|
+
f"[engage_twitter_helper] enriched {len(top_rows)}/{len(rows)} rows "
|
|
272
|
+
f"(history={non_empty}, chain={chains_non_empty} non-empty)",
|
|
208
273
|
file=sys.stderr,
|
|
209
274
|
)
|
|
210
275
|
except Exception as e:
|
|
@@ -215,7 +280,7 @@ def cmd_pending_data(batch_size: int) -> int:
|
|
|
215
280
|
)
|
|
216
281
|
|
|
217
282
|
out = []
|
|
218
|
-
for r, history_block in zip(rows, history_blocks):
|
|
283
|
+
for r, history_block, chain_block in zip(rows, history_blocks, chain_blocks):
|
|
219
284
|
out.append({
|
|
220
285
|
"id": r.get("id"),
|
|
221
286
|
"platform": r.get("platform"),
|
|
@@ -231,6 +296,7 @@ def cmd_pending_data(batch_size: int) -> int:
|
|
|
231
296
|
"is_our_original_post": int(r.get("is_our_original_post") or 0),
|
|
232
297
|
"project_name": r.get("project_name"),
|
|
233
298
|
"counterparty_history_block": history_block,
|
|
299
|
+
"conversation_chain_block": chain_block,
|
|
234
300
|
"their_media_block": _render_media_block(r.get("their_media")),
|
|
235
301
|
})
|
|
236
302
|
# json_agg(...) returns null when the array is empty; engage-twitter.sh's
|