@m13v/s4l 1.7.4-rc.23 → 1.7.4-rc.25

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,4 +1,4 @@
1
1
  {
2
- "version": "1.7.4-rc.23",
3
- "installedAt": "2026-07-12T19:50:52.791Z"
2
+ "version": "1.7.4-rc.25",
3
+ "installedAt": "2026-07-13T15:58:56.002Z"
4
4
  }
package/mcp/manifest.json CHANGED
@@ -2,7 +2,7 @@
2
2
  "dxt_version": "0.1",
3
3
  "name": "social-autoposter",
4
4
  "display_name": "S4L",
5
- "version": "1.7.4-rc.23",
5
+ "version": "1.7.4-rc.25",
6
6
  "description": "Draft, review, approve, and autopilot X/Twitter posts.",
7
7
  "long_description": "## **⚠️ The disclaimer above is generic Claude boilerplate.** Anthropic shows the same warning on every plugin regardless of what it does; any plugin has the same level of access as any app you download from the internet.\n\nS4L is an open source product developed by Mediar.ai Incorporated, a VC-backed San Francisco-based startup.\n\nTo get started:\n\n1\\. Copy this prompt: **Set me up on S4L plugin end to end**\n\n2\\. Quit with CMD+Q, reopen Claude, paste into a new chat.\n\nWhat happens next:\n\n* About every 5 minutes S4L scans X for posts that match your topics and drafts replies in your voice.\n* Drafts show up as review cards, usually the first within a few minutes. Nothing is posted automatically; you approve each one.\n* Posting autopilot stays off until you explicitly turn it on.",
8
8
  "author": {
package/mcp/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@m13v/s4l-mcp",
3
- "version": "1.7.4-rc.23",
3
+ "version": "1.7.4-rc.25",
4
4
  "private": true,
5
5
  "description": "Desktop MCP client for social-autoposter (X/Twitter rail): manual draft/review/approve loop, autopilot control, and stats. Thin wrapper over the existing pipeline scripts.",
6
6
  "license": "MIT",
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@m13v/s4l",
3
- "version": "1.7.4-rc.23",
3
+ "version": "1.7.4-rc.25",
4
4
  "description": "Automated social posting pipeline for Reddit, X/Twitter, LinkedIn, and Moltbook. Install as a Claude Code agent skill.",
5
5
  "bin": {
6
6
  "social-autoposter": "bin/cli.js",
@@ -59,6 +59,25 @@ RUNNING_STALL_SECONDS = 1200
59
59
  # At StartInterval 120 that is ~6 min of continuous stall.
60
60
  ALERT_AFTER = 3
61
61
 
62
+ # --- Host-sleep awareness (2026-07-13, Sentry S4L-4B) ------------------------
63
+ # launchd does NOT fire StartInterval jobs while the host is suspended, so the
64
+ # wall-clock gap between consecutive ticks of THIS watchdog is a reliable sleep
65
+ # detector: a gap of 3+ intervals means the box was asleep (or powered off),
66
+ # not stalled. S4L-4B (Nhat's MacBook Air, 2026-07-11): laptop slept mid-cycle,
67
+ # one producer enqueue timeout latched, no cycle ran to clear it, and the page
68
+ # fired with pending=0/running=0 and a bogus "account change?" cause while the
69
+ # telemetry showed 9 samples in a 2h window (40-min gaps). Laptops that sleep
70
+ # between cycles would re-trigger that page forever.
71
+ TICK_INTERVAL_SECONDS = 120 # keep in sync with launchd StartInterval
72
+ SLEEP_GAP_SECONDS = TICK_INTERVAL_SECONDS * 3 # missed >=3 ticks -> host slept
73
+ # For this long after a detected sleep gap, pre-existing latch/age signals are
74
+ # treated as sleep-tainted: they must be corroborated by actual queued work
75
+ # (pending/running > 0) or the outcome-level batches_stuck backstop to page.
76
+ SLEEP_GAP_RECENT_SECONDS = 1800
77
+ # A worker-task transcript modified this recently proves the routines are
78
+ # firing, which rules out the "orphaned routines / account change" cause.
79
+ WORKER_RECENT_SECONDS = 1800
80
+
62
81
  # A box counts as configured when ANY complete worker set has its SKILL.md on
63
82
  # disk: the universal type-blind worker, its short-lived staging predecessor, or
64
83
  # the legacy per-type pair. Keep in sync with WORKER_TASK_SETS in
@@ -142,6 +161,46 @@ def _recent_rate_limit(window: int = 1200) -> bool:
142
161
  return False
143
162
 
144
163
 
164
+ def _worker_ran_recently(window: int = WORKER_RECENT_SECONDS) -> bool:
165
+ """True if any s4l-worker scheduled-task transcript was written in the last
166
+ `window` seconds — direct on-disk proof the worker routines are firing, so
167
+ a stall can NOT be the orphaned-routines / account-change shape. Same
168
+ transcript bucket _recent_rate_limit reads."""
169
+ try:
170
+ now = time.time()
171
+ for f in glob.glob(os.path.expanduser("~/.claude/projects/*s4l-worker*/*.jsonl")):
172
+ try:
173
+ if (now - os.path.getmtime(f)) <= window:
174
+ return True
175
+ except OSError:
176
+ continue
177
+ except Exception:
178
+ pass
179
+ return False
180
+
181
+
182
+ def _api_reachable(timeout: float = 3.0) -> bool:
183
+ """True if the S4L API host answers HTTP at all. ANY HTTP status counts as
184
+ reachable (a 4xx/5xx still proves the network path is up); only a socket /
185
+ URL error means offline. Used to hold the Sentry page while the box has no
186
+ network — an offline box is a different, self-healing condition, and the
187
+ page would be misleading (plus the event can't ship anyway). The stall
188
+ episode keeps counting, so if it's still stalled when connectivity returns,
189
+ the page fires then."""
190
+ import urllib.error
191
+ import urllib.request
192
+
193
+ base = os.environ.get("AUTOPOSTER_API_BASE", "https://s4l.ai").rstrip("/")
194
+ try:
195
+ req = urllib.request.Request(base, method="HEAD")
196
+ urllib.request.urlopen(req, timeout=timeout)
197
+ return True
198
+ except urllib.error.HTTPError:
199
+ return True # server answered; network is up
200
+ except Exception:
201
+ return False
202
+
203
+
145
204
  BATCH_PROGRESSION_MIN_BATCHES = 5
146
205
  BATCH_PROGRESSED_PHASES = {"phase2b-gen", "phase2b-post"}
147
206
 
@@ -348,6 +407,30 @@ def _report_queue_health_sample(
348
407
 
349
408
 
350
409
  def main() -> int:
410
+ now = time.time()
411
+ st = _read_state()
412
+
413
+ # Sleep detection: launchd skips ticks while the host is suspended, so a
414
+ # wall-clock gap between this tick and the previous one of >= 3 intervals
415
+ # means the box slept (see the S4L-4B block by the constants above).
416
+ last_tick_at = st.get("last_tick_at")
417
+ last_sleep_gap_at = st.get("last_sleep_gap_at")
418
+ if last_tick_at is not None:
419
+ tick_gap = now - float(last_tick_at)
420
+ if tick_gap > SLEEP_GAP_SECONDS:
421
+ last_sleep_gap_at = now
422
+ sys.stderr.write(
423
+ f"[stall-watch] tick gap {int(tick_gap)}s (> {SLEEP_GAP_SECONDS}s) — "
424
+ "host was asleep/off; treating stall signals as sleep-tainted for "
425
+ f"{SLEEP_GAP_RECENT_SECONDS}s\n"
426
+ )
427
+ sleep_gap_recent = bool(
428
+ last_sleep_gap_at is not None
429
+ and (now - float(last_sleep_gap_at)) < SLEEP_GAP_RECENT_SECONDS
430
+ )
431
+
432
+ pending = _pending_count()
433
+ running = _running_count()
351
434
  age = _oldest_pending_age()
352
435
  run_age = _oldest_running_age()
353
436
  timeouts = _consecutive_timeouts()
@@ -362,8 +445,16 @@ def main() -> int:
362
445
  # never claimed), (3) running-age (job claimed then wedged mid-run) — (3) is
363
446
  # the only one of the first three that catches a worker dying after it picked
364
447
  # up the job — (4) batches_stuck, the outcome-level backstop above.
448
+ # Latch sanity gate (S4L-4B): consecutive_timeouts is durable and only
449
+ # clears on a successful drain, so ONE timed-out enqueue keeps a box
450
+ # looking stalled for as long as no cycle runs — even with a provably idle
451
+ # queue (pending=0, running=0). A single timeout only counts when there is
452
+ # actual queued work to corroborate it; an idle-queue latch needs >= 2.
453
+ # NOTE: deliberately stricter than the menubar/_index.ts stall HINT
454
+ # (timeouts >= 1) — a UI hint may over-trigger, a fleet page must not.
455
+ timeouts_signal = timeouts >= 2 or (timeouts >= 1 and (pending > 0 or running > 0))
365
456
  stalled = configured and (
366
- timeouts >= 1
457
+ timeouts_signal
367
458
  or (age is not None and age > STALL_SECONDS)
368
459
  or (run_age is not None and run_age > RUNNING_STALL_SECONDS)
369
460
  or batches_stuck
@@ -373,13 +464,17 @@ def main() -> int:
373
464
  # episode resets and a LATER real stall (orphaned routines) still alerts.
374
465
  if stalled and _recent_rate_limit():
375
466
  stalled = False
467
+ # Sleep suppression: right after a wake, the latch (and any pre-nap ages)
468
+ # predate the sleep, not a worker failure. Require corroboration by real
469
+ # queued work or the outcome-level backstop before calling it a stall.
470
+ if stalled and sleep_gap_recent and not (pending > 0 or running > 0 or batches_stuck):
471
+ stalled = False
376
472
 
377
473
  # Record this tick regardless of outcome — see _report_queue_health_sample.
378
474
  _report_queue_health_sample(
379
- _pending_count(), _running_count(), timeouts, age, run_age, stalled, batches_stuck
475
+ pending, running, timeouts, age, run_age, stalled, batches_stuck
380
476
  )
381
477
 
382
- st = _read_state()
383
478
  consecutive = int(st.get("consecutive", 0))
384
479
  alerted = bool(st.get("alerted", False))
385
480
  # first_seen_at: first check this episode looked stalled at all (predates
@@ -390,6 +485,10 @@ def main() -> int:
390
485
  first_seen_at = st.get("first_seen_at")
391
486
  alerted_at = st.get("alerted_at")
392
487
 
488
+ # Tick bookkeeping persisted on EVERY exit path — the sleep detector needs
489
+ # an unbroken last_tick_at chain even (especially) while healthy.
490
+ tick_state = {"last_tick_at": now, "last_sleep_gap_at": last_sleep_gap_at}
491
+
393
492
  if not stalled:
394
493
  # Recovered (or never stalled) -> reset the episode so the next stall pages.
395
494
  if consecutive or alerted:
@@ -397,7 +496,7 @@ def main() -> int:
397
496
  total_duration = time.time() - float(first_seen_at)
398
497
  paged_duration = (time.time() - float(alerted_at)) if alerted_at else None
399
498
  _report_recovery(total_duration, paged_duration)
400
- _write_state({"consecutive": 0, "alerted": False})
499
+ _write_state({**tick_state, "consecutive": 0, "alerted": False})
401
500
  return 0
402
501
 
403
502
  consecutive += 1
@@ -411,60 +510,86 @@ def main() -> int:
411
510
  # the fallback is the classic orphaned-routine case.
412
511
  wedged_inflight = run_age is not None and run_age > RUNNING_STALL_SECONDS
413
512
  if consecutive >= ALERT_AFTER and not alerted:
414
- try:
415
- sentry = _sentry()
416
- sentry.init()
417
- if wedged_inflight:
418
- cause = (
419
- "a worker claimed a draft job and then died mid-run (claude -p child "
420
- "never came up / crashed)"
421
- )
422
- stall_shape = "inflight_wedged"
423
- elif batches_stuck:
424
- cause = (
425
- f"last {BATCH_PROGRESSION_MIN_BATCHES} twitter_batches all failed to "
426
- "reach phase2b-gen — drafting is not actually happening even if the "
427
- "queue-level counters look ambiguous"
428
- )
429
- stall_shape = "batches_not_progressing"
430
- else:
431
- cause = "scheduled-task routines likely orphaned — Claude Desktop account change?"
432
- stall_shape = "not_draining"
433
- # Our own staging/QA/dev boxes (identity.is_internal_install) get set
434
- # up and rebuilt with nobody actively feeding the queue, which looks
435
- # identical to a real stall on the signals above. Downgrade those to
436
- # warning instead of error so they don't page as a customer incident
437
- # (the digest only scans error/fatal) while still leaving a Sentry
438
- # record if we ever need to look one up by hand.
439
- is_internal = identity.is_internal_install()
440
- sentry.capture_message(
441
- "social-autoposter autopilot stalled: draft jobs are not being "
442
- f"drained ({cause}). producer consecutive timeouts={timeouts}, "
443
- f"oldest pending job age={age_str}, oldest in-flight (running) job "
444
- f"age={run_age_str}, sustained {consecutive} checks.",
445
- level=("warning" if is_internal else "error"),
446
- tags={
447
- "component": "autopilot",
448
- "issue": "stall",
449
- "stall_shape": stall_shape,
450
- "consecutive_timeouts": str(timeouts),
451
- "oldest_pending_age_s": str(int(age)) if age is not None else "",
452
- "oldest_running_age_s": str(int(run_age)) if run_age is not None else "",
453
- "batches_stuck": str(batches_stuck),
454
- "internal_install": str(is_internal),
455
- },
456
- )
457
- sentry.flush()
458
- except Exception:
459
- # No Sentry (helper/SDK missing) -> at least leave a local breadcrumb.
513
+ if not _api_reachable():
514
+ # No network: an offline box is a different, self-healing condition,
515
+ # the page would misattribute it to the worker (and the event can't
516
+ # ship anyway). Keep the episode counting so a stall that survives
517
+ # the outage still pages the tick connectivity returns.
460
518
  sys.stderr.write(
461
- f"[stall-watch] autopilot stalled (timeouts={timeouts}, "
462
- f"pending_age={age_str}, running_age={run_age_str}) but Sentry report failed\n"
519
+ f"[stall-watch] stall persisted {consecutive} checks but the API is "
520
+ "unreachable (box offline?); deferring the page until network returns\n"
463
521
  )
464
- alerted = True
465
- alerted_at = time.time()
522
+ else:
523
+ try:
524
+ sentry = _sentry()
525
+ sentry.init()
526
+ if wedged_inflight:
527
+ cause = (
528
+ "a worker claimed a draft job and then died mid-run (claude -p child "
529
+ "never came up / crashed)"
530
+ )
531
+ stall_shape = "inflight_wedged"
532
+ elif batches_stuck:
533
+ cause = (
534
+ f"last {BATCH_PROGRESSION_MIN_BATCHES} twitter_batches all failed to "
535
+ "reach phase2b-gen — drafting is not actually happening even if the "
536
+ "queue-level counters look ambiguous"
537
+ )
538
+ stall_shape = "batches_not_progressing"
539
+ elif _worker_ran_recently():
540
+ # On-disk transcripts prove the worker routines ARE firing, so
541
+ # this cannot be the orphaned-routines shape (S4L-4B paged with
542
+ # exactly that bogus cause while the registry sample showed the
543
+ # task had run seconds earlier).
544
+ cause = (
545
+ "producer timeout latch, but worker-task transcripts show recent "
546
+ "runs — routines are firing; suspect a transient enqueue timeout"
547
+ + (" or a host sleep gap" if sleep_gap_recent else "")
548
+ + ", NOT orphaned routines"
549
+ )
550
+ stall_shape = "latch_worker_alive"
551
+ else:
552
+ cause = "scheduled-task routines likely orphaned — Claude Desktop account change?"
553
+ stall_shape = "not_draining"
554
+ # Our own staging/QA/dev boxes (identity.is_internal_install) get set
555
+ # up and rebuilt with nobody actively feeding the queue, which looks
556
+ # identical to a real stall on the signals above. Downgrade those to
557
+ # warning instead of error so they don't page as a customer incident
558
+ # (the digest only scans error/fatal) while still leaving a Sentry
559
+ # record if we ever need to look one up by hand.
560
+ is_internal = identity.is_internal_install()
561
+ sentry.capture_message(
562
+ "social-autoposter autopilot stalled: draft jobs are not being "
563
+ f"drained ({cause}). producer consecutive timeouts={timeouts}, "
564
+ f"oldest pending job age={age_str}, oldest in-flight (running) job "
565
+ f"age={run_age_str}, sustained {consecutive} checks.",
566
+ level=("warning" if is_internal else "error"),
567
+ tags={
568
+ "component": "autopilot",
569
+ "issue": "stall",
570
+ "stall_shape": stall_shape,
571
+ "consecutive_timeouts": str(timeouts),
572
+ "oldest_pending_age_s": str(int(age)) if age is not None else "",
573
+ "oldest_running_age_s": str(int(run_age)) if run_age is not None else "",
574
+ "batches_stuck": str(batches_stuck),
575
+ "internal_install": str(is_internal),
576
+ "pending": str(pending),
577
+ "running": str(running),
578
+ "recent_sleep_gap": str(sleep_gap_recent),
579
+ },
580
+ )
581
+ sentry.flush()
582
+ except Exception:
583
+ # No Sentry (helper/SDK missing) -> at least leave a local breadcrumb.
584
+ sys.stderr.write(
585
+ f"[stall-watch] autopilot stalled (timeouts={timeouts}, "
586
+ f"pending_age={age_str}, running_age={run_age_str}) but Sentry report failed\n"
587
+ )
588
+ alerted = True
589
+ alerted_at = time.time()
466
590
 
467
591
  _write_state({
592
+ **tick_state,
468
593
  "consecutive": consecutive,
469
594
  "alerted": alerted,
470
595
  "first_seen_at": first_seen_at,
@@ -33,7 +33,12 @@ def main() -> int:
33
33
  except Exception:
34
34
  import urllib.request
35
35
  try:
36
- urllib.request.urlopen(f"{url}/json/version", timeout=3)
36
+ # ProxyHandler({}): loopback CDP must never route through a proxy.
37
+ # macOS system proxy settings leak into urllib's default opener, and
38
+ # a box-wide forwarder 403s 127.0.0.1 probes (2026-07-13 root cause
39
+ # of the "wedged Chrome" misdiagnosis).
40
+ opener = urllib.request.build_opener(urllib.request.ProxyHandler({}))
41
+ opener.open(f"{url}/json/version", timeout=3)
37
42
  print(json.dumps({"ready": True, "mode": "http-only"}))
38
43
  return 0
39
44
  except Exception as e:
package/skill/lock.sh CHANGED
@@ -132,7 +132,31 @@ if [ -z "${_SA_LOCK_DIRS+x}" ]; then
132
132
  rm -f "$t"
133
133
  done
134
134
  }
135
- trap _sa_release_locks EXIT INT TERM HUP
135
+ trap _sa_release_locks EXIT
136
+ # 2026-07-12: INT/TERM/HUP release locks AND EXIT, instead of sharing the
137
+ # plain EXIT handler. Scenario this fixes (observed 2026-07-12 12:02:09,
138
+ # engage-dm-replies.sh pid 72465): a stray signal fired the old combined
139
+ # trap, which released BOTH held locks and then RESUMED the script (bash
140
+ # continues after a trapped signal when the handler doesn't exit). The run
141
+ # kept driving the shared twitter Chrome without its twitter-browser lock,
142
+ # the twitter cycle legitimately acquired the freed lock and navigated the
143
+ # same tab mid-send, and all 3 send-dm attempts died with
144
+ # thread_url_redirected ("shared-tab drift", human_dm_replies id 100045).
145
+ # A signaled run must never continue lock-protected work: log WHICH signal
146
+ # hit (the old single-line event gave no clue), clean up once, exit 128+N.
147
+ # Scripts that install their own combined traps after sourcing this file
148
+ # (e.g. run-twitter-cycle.sh's _sa_combined_exit) still override these.
149
+ _sa_exit_on_signal() {
150
+ local _sig="$1" _code="$2"
151
+ _sa_lock_event signal_exit "-" "signal=$_sig"
152
+ echo "[lock] caught SIG${_sig} pid=$$; releasing locks and exiting $_code at $(date +%H:%M:%S)" >&2
153
+ _sa_release_locks
154
+ trap - EXIT
155
+ exit "$_code"
156
+ }
157
+ trap '_sa_exit_on_signal INT 130' INT
158
+ trap '_sa_exit_on_signal TERM 143' TERM
159
+ trap '_sa_exit_on_signal HUP 129' HUP
136
160
  fi
137
161
 
138
162
  acquire_lock() {