@m13v/s4l 1.7.2 → 1.7.4-rc.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -29,7 +29,16 @@ Usage:
29
29
 
30
30
  Output (stdout, last line is JSON):
31
31
  {"ok": true, "handle": "...", "profile": {...}, "posts": [...],
32
- "comments": [...], "counts": {...}, "grounding_instructions": "..."}
32
+ "comments": [...], "top_posts": [...], "top_replies": [...],
33
+ "counts": {...}, "grounding_instructions": "..."}
34
+
35
+ top_posts / top_replies are the same items ranked by real engagement
36
+ (likes*3 + retweets*5 + replies*2), each stamped with rank + engagement_score.
37
+ On /with_replies each reply also carries `parent` = {author, text, url} (the
38
+ tweet it replied to, best-effort DOM-adjacent pairing), and the top posts get
39
+ their thread continuation expanded (`thread`: [tweet texts]) by visiting the
40
+ permalink. scripts/voice_exemplars.py turns these into voice.examples +
41
+ persona_corpus.txt exemplars.
33
42
  """
34
43
 
35
44
  from __future__ import annotations
@@ -182,7 +191,11 @@ def scrape_profile(send) -> dict:
182
191
  # --------------------------------------------------------------------------- #
183
192
  _TIMELINE_JS_TMPL = r"""(function(){
184
193
  var ME=%s; // lowercase handle without @
194
+ var WITH_PARENTS=%s; // true only on /with_replies: pair each reply with the
195
+ // other-author article directly above it (same
196
+ // conversation cell renders parent then our reply)
185
197
  var out=[];
198
+ var lastOther=null;
186
199
  var arts=document.querySelectorAll('article');
187
200
  for(var i=0;i<arts.length;i++){
188
201
  var art=arts[i];
@@ -194,10 +207,8 @@ _TIMELINE_JS_TMPL = r"""(function(){
194
207
  var mm=hh.match(/^\/([A-Za-z0-9_]{1,15})$/);
195
208
  if(mm){authorHandle=mm[1].toLowerCase();break;}
196
209
  }
197
- if(authorHandle && authorHandle!==ME) continue; // skip others' posts (reposts/quotes/threads)
198
210
  var tEl=art.querySelector('[data-testid="tweetText"]');
199
211
  var text=tEl?(tEl.innerText||'').trim():'';
200
- if(!text) continue;
201
212
  // permalink + id
202
213
  var url='';var id='';
203
214
  var statusLinks=art.querySelectorAll('a[href*="/status/"]');
@@ -206,6 +217,16 @@ _TIMELINE_JS_TMPL = r"""(function(){
206
217
  var sm=sh.match(/\/status\/(\d+)/);
207
218
  if(sm){id=sm[1];url='https://x.com'+sh.split('?')[0];break;}
208
219
  }
220
+ if(authorHandle && authorHandle!==ME){
221
+ // Someone else's article. On /with_replies it is the parent of our next
222
+ // reply in the same conversation cell; remember it. On the posts tab it
223
+ // is a repost/quote we simply skip (WITH_PARENTS=false there).
224
+ if(WITH_PARENTS && text){
225
+ lastOther={author:'@'+authorHandle,text:text.slice(0,500),url:url};
226
+ }
227
+ continue;
228
+ }
229
+ if(!text) continue;
209
230
  // reply? presence of a "Replying to" header in the cell
210
231
  var isReply=false, replyTo='';
211
232
  var spans=art.querySelectorAll('span,div');
@@ -226,15 +247,21 @@ _TIMELINE_JS_TMPL = r"""(function(){
226
247
  var m=al.match(/([\d,]+)/);
227
248
  return m?parseInt(m[1].replace(/,/g,''),10):0;
228
249
  }
229
- out.push({text:text,url:url,id:id,is_reply:isReply,reply_to:replyTo,
230
- likes:metric('like'),replies:metric('reply'),retweets:metric('retweet')});
250
+ var item={text:text,url:url,id:id,is_reply:isReply,reply_to:replyTo,
251
+ likes:metric('like'),replies:metric('reply'),retweets:metric('retweet')};
252
+ if(WITH_PARENTS){
253
+ item.parent=lastOther; // may be null (their own standalone post)
254
+ lastOther=null; // consume: never pair one parent with two replies
255
+ }
256
+ out.push(item);
231
257
  }
232
258
  return JSON.stringify(out);
233
259
  })()"""
234
260
 
235
261
 
236
262
  def scrape_timeline(send, me: str, want: int, max_scrolls: int = 30,
237
- exclude_ids: "set | None" = None) -> list:
263
+ exclude_ids: "set | None" = None,
264
+ capture_parents: bool = False) -> list:
238
265
  """Scroll the current timeline, collecting up to `want` of the user's OWN
239
266
  authored articles (in DOM order = newest first). `exclude_ids` drops items
240
267
  already captured elsewhere — that's how the comments pass (/with_replies)
@@ -250,7 +277,8 @@ def scrape_timeline(send, me: str, want: int, max_scrolls: int = 30,
250
277
  scrolling to the bottom each step to force the next lazy-load batch."""
251
278
  seen: dict[str, dict] = {}
252
279
  exclude_ids = exclude_ids or set()
253
- expr = _TIMELINE_JS_TMPL % json.dumps(me.lower())
280
+ expr = _TIMELINE_JS_TMPL % (json.dumps(me.lower()),
281
+ "true" if capture_parents else "false")
254
282
  STALL_LIMIT = 4
255
283
  stall = 0
256
284
  for n in range(max_scrolls):
@@ -283,6 +311,69 @@ def scrape_timeline(send, me: str, want: int, max_scrolls: int = 30,
283
311
  return items[:want]
284
312
 
285
313
 
314
+ # --------------------------------------------------------------------------- #
315
+ # Engagement ranking + thread expansion for exemplar extraction.
316
+ # --------------------------------------------------------------------------- #
317
+ def _engagement_score(item: dict) -> int:
318
+ """Same weighting everywhere exemplars are ranked (voice_exemplars.py
319
+ re-derives it if absent): a retweet is a stronger signal than a like,
320
+ a like stronger than a reply-back."""
321
+ return (int(item.get("likes") or 0) * 3
322
+ + int(item.get("retweets") or 0) * 5
323
+ + int(item.get("replies") or 0) * 2)
324
+
325
+
326
+ def rank_top(items: list, n: int = 5) -> list:
327
+ """Top n items by engagement, stamped with rank + engagement_score.
328
+ Zero-engagement items still rank (small accounts often have nothing else);
329
+ the score on each entry lets downstream decide what to keep."""
330
+ ranked = sorted(items, key=_engagement_score, reverse=True)[:n]
331
+ out = []
332
+ for i, it in enumerate(ranked, 1):
333
+ e = dict(it)
334
+ e["rank"] = i
335
+ e["engagement_score"] = _engagement_score(it)
336
+ out.append(e)
337
+ return out
338
+
339
+
340
+ _THREAD_JS_TMPL = r"""(function(){
341
+ var ME=%s;
342
+ var out=[];var started=false;
343
+ var arts=document.querySelectorAll('article');
344
+ for(var i=0;i<arts.length;i++){
345
+ var art=arts[i];
346
+ var authorHandle='';
347
+ var links=art.querySelectorAll('a[href^="/"]');
348
+ for(var j=0;j<links.length;j++){
349
+ var hh=links[j].getAttribute('href')||'';
350
+ var mm=hh.match(/^\/([A-Za-z0-9_]{1,15})$/);
351
+ if(mm){authorHandle=mm[1].toLowerCase();break;}
352
+ }
353
+ var tEl=art.querySelector('[data-testid="tweetText"]');
354
+ var text=tEl?(tEl.innerText||'').trim():'';
355
+ if(authorHandle===ME && text){out.push(text);started=true;}
356
+ else if(started){break;} // first non-ME article after the run = end of thread
357
+ }
358
+ return JSON.stringify(out);
359
+ })()"""
360
+
361
+
362
+ def expand_thread(send, me: str, url: str) -> list:
363
+ """Visit a post's permalink and return the consecutive run of the user's own
364
+ tweets starting at the focal one ([focal, continuation, ...]). Length 1 =
365
+ not a thread. Bonus: the permalink shows full text, so this also untruncates
366
+ posts the timeline cut with 'Show more'. Best-effort; [] on failure."""
367
+ if not url or not _navigate(send, url, settle=3.5, expect="/status/"):
368
+ return []
369
+ raw = _eval(send, _THREAD_JS_TMPL % json.dumps(me.lower())) or "[]"
370
+ try:
371
+ parts = json.loads(raw)
372
+ except Exception:
373
+ parts = []
374
+ return parts if isinstance(parts, list) else []
375
+
376
+
286
377
  GROUNDING_INSTRUCTIONS = (
287
378
  "You now have this user's real X profile (bio, original posts, and their own "
288
379
  "replies). Use it as GROUND TRUTH to draft their autoposter config fields in "
@@ -294,9 +385,15 @@ GROUNDING_INSTRUCTIONS = (
294
385
  "capitalization habits, emoji/punctuation usage, and recurring phrases or tics. "
295
386
  "Write the `voice` field so a reply drafted with it would be indistinguishable "
296
387
  "from something they'd actually type.\n"
297
- "3. GOLDEN-RULE EXAMPLES: pick 2-4 of their strongest real replies/posts "
298
- "verbatim and keep them as exemplars (these become few-shot anchors). Choose "
299
- "ones that show the target reply behavior: helpful, specific, in-voice.\n"
388
+ "3. GOLDEN-RULE EXAMPLES: `top_replies` and `top_posts` are already ranked by "
389
+ "REAL engagement (each entry carries likes/retweets/replies, engagement_score, "
390
+ "the parent tweet it replied to, and thread continuations). From them keep up "
391
+ "to 5 replies verbatim as exemplars, skipping throwaway one-liners that show "
392
+ "no voice; on small accounts engagement is sparse, so judge voice quality too. "
393
+ "STORE them in the project's `voice.examples` (every drafter on every platform "
394
+ "mirrors that field), or run "
395
+ "`voice_exemplars.py apply --scan <scan.json> --project <name>` to write "
396
+ "voice.examples + the persona_corpus.txt exemplar section deterministically.\n"
300
397
  "4. PHRASE BANK: list the kinds of phrases / openers / sign-offs they reuse, "
301
398
  "and any words/claims they clearly AVOID (for `content_guardrails`).\n"
302
399
  "5. ICP: infer who they engage with (who they reply to, what communities) to "
@@ -314,6 +411,10 @@ def main() -> int:
314
411
  ap.add_argument("--handle", default=None, help="@handle to scan (default: live logged-in handle)")
315
412
  ap.add_argument("--posts", type=int, default=20, help="max original posts to collect")
316
413
  ap.add_argument("--comments", type=int, default=50, help="max replies/comments to collect")
414
+ ap.add_argument("--top", type=int, default=5, help="how many top posts/replies to rank")
415
+ ap.add_argument("--expand-threads", type=int, default=3,
416
+ help="visit this many top posts' permalinks to capture thread "
417
+ "continuations + full text (0 = off)")
317
418
  args = ap.parse_args()
318
419
 
319
420
  if create_connection is None:
@@ -354,10 +455,23 @@ def main() -> int:
354
455
  on_replies = _navigate(send, f"https://x.com/{handle}/with_replies",
355
456
  settle=4.0, expect=f"/{handle}/with_replies")
356
457
  comments = (
357
- scrape_timeline(send, handle, args.comments, exclude_ids=post_ids)
458
+ scrape_timeline(send, handle, args.comments, exclude_ids=post_ids,
459
+ capture_parents=True)
358
460
  if on_replies else []
359
461
  )
360
462
 
463
+ # 4. Rank both surfaces by real engagement, then expand the top posts'
464
+ # permalinks to capture thread continuations (and untruncated text).
465
+ top_posts = rank_top(posts, args.top)
466
+ top_replies = rank_top(comments, args.top)
467
+ for tp in top_posts[:max(args.expand_threads, 0)]:
468
+ parts = expand_thread(send, handle, tp.get("url") or "")
469
+ if parts:
470
+ if len(parts[0]) > len(tp.get("text") or ""):
471
+ tp["text"] = parts[0] # permalink text is never truncated
472
+ if len(parts) > 1:
473
+ tp["thread"] = parts
474
+
361
475
  result = {
362
476
  "ok": True,
363
477
  "state": "scanned",
@@ -365,9 +479,27 @@ def main() -> int:
365
479
  "profile": profile,
366
480
  "posts": posts,
367
481
  "comments": comments,
482
+ "top_posts": top_posts,
483
+ "top_replies": top_replies,
368
484
  "counts": {"posts": len(posts), "comments": len(comments)},
369
485
  "grounding_instructions": GROUNDING_INSTRUCTIONS,
370
486
  }
487
+
488
+ # Persist the corpus beside config.json so the later project-save step
489
+ # (setup writes the project, then voice_exemplars.py auto-applies the
490
+ # exemplars) works without a re-scan. The scan can run BEFORE config.json
491
+ # exists on a fresh onboarding, hence mkdir. Best-effort: the scan is
492
+ # still fully usable from stdout if the write fails.
493
+ try:
494
+ sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
495
+ import s4l_mode
496
+ sidecar = s4l_mode.config_path().parent / "last_profile_scan.json"
497
+ sidecar.parent.mkdir(parents=True, exist_ok=True)
498
+ sidecar.write_text(json.dumps(result, ensure_ascii=False))
499
+ result["scan_file"] = str(sidecar)
500
+ except Exception:
501
+ result["scan_file"] = None
502
+
371
503
  print(json.dumps(result, ensure_ascii=False))
372
504
  return 0
373
505
  except Exception as e:
@@ -0,0 +1,92 @@
1
+ import re, os, subprocess, json, hashlib
2
+ from collections import Counter, defaultdict
3
+ from itertools import combinations
4
+
5
+ DATA = os.path.expanduser("~/social-autoposter/mixer/remotion/src/mixer/data.ts")
6
+ PUB = os.path.expanduser("~/social-autoposter/mixer/remotion/public/mixer")
7
+ src = open(DATA).read()
8
+
9
+ # Extract every clipsV2 block and legacy clips arrays referencing tlh-*.mp4
10
+ # Find all "mixer/tlh-...mp4" occurrences grouped per variant via clipsV2:[ ... ] blocks
11
+ sets = []
12
+ for m in re.finditer(r'clipsV2:\s*\[(.*?)\]', src, re.S):
13
+ clips = re.findall(r'mixer/(tlh-[0-9a-z\-]+\.mp4)', m.group(1))
14
+ if clips:
15
+ sets.append(clips)
16
+ # legacy v1: clips arrays inside TLH? lesson-1 uses clips[] with tlh-1..5
17
+ for m in re.finditer(r'clips:\s*\[(.*?)\]', src, re.S):
18
+ clips = re.findall(r'"(tlh-[0-9a-z\-]+\.mp4)"', m.group(1))
19
+ if clips:
20
+ sets.append(clips)
21
+
22
+ def family(fn):
23
+ # tlh-<firstint>-... -> first integer group
24
+ mm = re.match(r'tlh-(\d+)', fn)
25
+ return mm.group(1) if mm else fn
26
+
27
+ usage = Counter()
28
+ cooc = Counter()
29
+ for s in sets:
30
+ for c in set(s):
31
+ usage[c]+=1
32
+ for a,b in combinations(sorted(set(s)),2):
33
+ cooc[(a,b)]+=1
34
+
35
+ # available encoded tlh clips on disk
36
+ avail = sorted([f for f in os.listdir(PUB) if re.match(r'tlh-[0-9a-z\-]+\.mp4$', f)])
37
+
38
+ def probe(fn):
39
+ p = os.path.join(PUB, fn)
40
+ out = subprocess.run(["ffprobe","-v","error","-select_streams","v:0",
41
+ "-show_entries","stream=width,height:format=duration","-of","json",p],
42
+ capture_output=True,text=True).stdout
43
+ j = json.loads(out)
44
+ st = j["stream"][0] if "stream" in j else j["streams"][0]
45
+ dur = float(j["format"]["duration"])
46
+ return st["width"], st["height"], dur
47
+
48
+ def blackframes(fn):
49
+ p = os.path.join(PUB, fn)
50
+ r = subprocess.run(["ffmpeg","-v","error","-i",p,"-vf","blackdetect=d=0.1:pic_th=0.98",
51
+ "-an","-f","null","-"],capture_output=True,text=True)
52
+ return "black_start" in r.stderr
53
+
54
+ def sha(fn):
55
+ return hashlib.sha256(open(os.path.join(PUB,fn),'rb').read()).hexdigest()
56
+
57
+ # Filter avail to valid 1080x1920, dur>=1.75 (holds a 2.0s slot acceptably or exact), non-black
58
+ os.environ["PATH"]="/opt/homebrew/Cellar/ffmpeg/8.1.1/bin:"+os.environ["PATH"]
59
+ valid=[]
60
+ info={}
61
+ for f in avail:
62
+ try:
63
+ w,h,d = probe(f)
64
+ except Exception as e:
65
+ continue
66
+ if (w,h)!=(1080,1920): continue
67
+ info[f]=(w,h,d)
68
+ valid.append(f)
69
+
70
+ print("total prior sets:", len(sets))
71
+ print("valid 1080x1920 clips on disk:", len(valid))
72
+
73
+ # Build candidate 4-sets: distinct families, zero pairwise co-occurrence, low total usage.
74
+ # Rank by (max pairwise cooc, total usage). Verify durations >=1.75 and non-black + distinct sha lazily on the winner.
75
+ from itertools import combinations as comb
76
+ # group valid by family, prefer low-usage clips
77
+ valid_sorted = sorted(valid, key=lambda f:(usage[f], f))
78
+ best=None
79
+ # to keep it tractable, restrict candidate pool to the 40 lowest-usage valid clips across distinct families
80
+ pool = valid_sorted[:60]
81
+ results=[]
82
+ for quad in comb(pool,4):
83
+ fams=[family(f) for f in quad]
84
+ if len(set(fams))!=4: continue
85
+ pairs=list(comb(sorted(quad),2))
86
+ maxco=max(cooc[p] for p in pairs)
87
+ tot=sum(usage[f] for f in quad)
88
+ results.append((maxco,tot,quad))
89
+ results.sort(key=lambda x:(x[0],x[1]))
90
+ print("\ntop 15 candidate quads (maxcooc, totalusage, clips):")
91
+ for maxco,tot,quad in results[:15]:
92
+ print(maxco,tot,list(quad),"durs",[round(info[f][2],3) for f in quad])
@@ -388,15 +388,17 @@ def format_report(summary, top, bottom, project=None, platform=None,
388
388
  # Projects with zero total_clicks across many posts are the canaries
389
389
  # for "this product/voice combination isn't landing" (the 'General'
390
390
  # bucket in the 7d audit on 2026-05-12: 56 posts, 0 clicks).
391
- lines.append("### Posts per Project per Platform")
392
- for row in summary:
393
- lines.append(
394
- f" {row[0]:<20} {row[1]:<12} {row[2]:>5} posts "
395
- f"avg_clicks={row[5]} avg_cm={row[4]} avg_up={row[3]} "
396
- f"best_clicks={row[8]} best_cm={row[7]} best_up={row[6]} "
397
- f"total_clicks={row[9]}"
398
- )
399
- lines.append("")
391
+ # Empty summary (--no-project-sections) skips the section entirely.
392
+ if summary:
393
+ lines.append("### Posts per Project per Platform")
394
+ for row in summary:
395
+ lines.append(
396
+ f" {row[0]:<20} {row[1]:<12} {row[2]:>5} posts "
397
+ f"avg_clicks={row[5]} avg_cm={row[4]} avg_up={row[3]} "
398
+ f"best_clicks={row[8]} best_cm={row[7]} best_up={row[6]} "
399
+ f"total_clicks={row[9]}"
400
+ )
401
+ lines.append("")
400
402
 
401
403
  # Per-project top performers (when no project filter)
402
404
  if top_by_group:
@@ -425,6 +427,16 @@ def format_report(summary, top, bottom, project=None, platform=None,
425
427
  for p in fallback_top:
426
428
  lines.append(format_post(p, suffix_strip_list=suffix_strip_list))
427
429
  lines.append("")
430
+ elif project:
431
+ # Project filter given, nothing qualified, and no fallback rows
432
+ # either (or the caller suppressed them). Say so explicitly so the
433
+ # on-demand --brief caller sees a definitive answer, not a bare
434
+ # header it might re-query.
435
+ lines.append(
436
+ f"### No {project} posts meeting {threshold_label} in the recency window. "
437
+ "Draft from the thread and the project's config voice; there is no winner to ground on."
438
+ )
439
+ lines.append("")
428
440
 
429
441
  # Bottom posts with failure annotations
430
442
  if bottom:
@@ -540,6 +552,21 @@ def main():
540
552
  "the few-shot exemplar section shows only the matching "
541
553
  "high-scoring posts instead of every style. Summary, "
542
554
  "fallback_top, and top_by_group are not affected."))
555
+ parser.add_argument("--no-project-sections", action="store_true",
556
+ help=("Omit the per-project summary table and the Top "
557
+ "Posts by Project section from the report. Added "
558
+ "2026-07-10 for the cycle orchestrators: the full "
559
+ "multi-project winner corpus is no longer bulk-"
560
+ "injected into every draft prompt (it homogenized "
561
+ "drafts). Instead the drafting session queries "
562
+ "`--project <name> --top 3` on demand AFTER it "
563
+ "has routed a candidate to a project."))
564
+ parser.add_argument("--brief", action="store_true",
565
+ help=("Render ONLY the top-posts list for the given "
566
+ "--project (no style table, no exemplars, no "
567
+ "summary, no bottom posts). This is the lean "
568
+ "on-demand shape the drafting session calls "
569
+ "after routing a candidate to a project."))
543
570
  parser.add_argument("--json", action="store_true", help="Output as JSON")
544
571
  args = parser.parse_args()
545
572
 
@@ -557,6 +584,25 @@ def main():
557
584
  if row and len(row) > 12 and row[12] in wanted
558
585
  ]
559
586
 
587
+ if args.no_project_sections:
588
+ summary = []
589
+ top_by_group = None
590
+ # Also drop the flat top-posts list: with top_by_group gone,
591
+ # format_report's `elif top:` branch would otherwise render a
592
+ # platform-wide "Top N Posts" block, which is the same shared
593
+ # winner corpus under a different header.
594
+ top = []
595
+ fallback_top = None
596
+
597
+ if args.brief:
598
+ # Lean per-project view: keep `top` (and its fallback message),
599
+ # strip everything else.
600
+ style_perf = []
601
+ top_by_style = []
602
+ summary = []
603
+ top_by_group = None
604
+ bottom = []
605
+
560
606
  if args.json:
561
607
  output = {
562
608
  "summary": [list(row) for row in summary],
@@ -43,12 +43,59 @@ import json
43
43
  import os
44
44
  import random
45
45
  import re
46
+ import signal
46
47
  import subprocess
47
48
  import sys
48
49
  import time
49
50
  from datetime import datetime, timezone
50
51
  from pathlib import Path
51
52
 
53
+
54
+ def _neuter_stream(stream) -> None:
55
+ """Point a dead pipe's fd at /dev/null so later writes (including the
56
+ interpreter-shutdown flush) can't raise BrokenPipeError again."""
57
+ try:
58
+ devnull = os.open(os.devnull, os.O_WRONLY)
59
+ os.dup2(devnull, stream.fileno())
60
+ os.close(devnull)
61
+ except Exception:
62
+ pass
63
+
64
+
65
+ _builtin_print = print
66
+
67
+
68
+ def print(*args, **kwargs): # noqa: A001 -- deliberate builtins.print shadow
69
+ # When the parent (cycle shell tree or MCP server) is killed mid-batch, our
70
+ # stdout/stderr pipes close and the next print raises BrokenPipeError. The
71
+ # old behavior unwound through excepthook (Sentry S4L-8), killing the batch
72
+ # BETWEEN posting a reply and recording it — which is how live tweets ended
73
+ # up unlogged and candidates re-feedable. Output to a dead parent is
74
+ # worthless; the bookkeeping (log_post, update_candidate, audit JSONL) is
75
+ # not. Swallow the error, neuter the fds, keep going.
76
+ try:
77
+ _builtin_print(*args, **kwargs)
78
+ except BrokenPipeError:
79
+ _neuter_stream(sys.stdout)
80
+ _neuter_stream(sys.stderr)
81
+
82
+
83
+ # Graceful SIGTERM: the MCP post_drafts timeout (and anything else that asks
84
+ # nicely before SIGKILL) sends SIGTERM. Dying instantly reopens the same
85
+ # posted-but-unrecorded window as the broken pipe, so instead flag the loop to
86
+ # stop at the next candidate boundary: current candidate finishes its
87
+ # bookkeeping, the summary and audit line still get written, exit stays 0.
88
+ _terminate_requested = False
89
+
90
+
91
+ def _on_sigterm(signum, frame):
92
+ global _terminate_requested
93
+ _terminate_requested = True
94
+ print("[post] SIGTERM received; stopping after current candidate", flush=True)
95
+
96
+
97
+ signal.signal(signal.SIGTERM, _on_sigterm)
98
+
52
99
  # This pipeline ONLY posts (never scans), so mark every twitter_browser.py reply
53
100
  # subprocess it spawns as the high-priority "post" lock role. run_subprocess
54
101
  # inherits this process env, so the child twitter_browser.py reads S4L_LOCK_ROLE
@@ -1150,6 +1197,10 @@ def main() -> int:
1150
1197
 
1151
1198
  try:
1152
1199
  for _idx, c in enumerate(candidates, start=1):
1200
+ if _terminate_requested:
1201
+ print(f"[post] stopping early on SIGTERM: {_idx - 1}/{_total} "
1202
+ f"candidates processed, posted={posted}", flush=True)
1203
+ break
1153
1204
  # Live per-post status for the S4L menu bar. LEAD with `posted` (the
1154
1205
  # REAL count of replies that actually landed), not `_idx` (the loop
1155
1206
  # position). _idx races through already-posted / deleted cards as instant
@@ -1224,6 +1275,22 @@ def main() -> int:
1224
1275
  "reason": reason or "",
1225
1276
  "our_url": c.get("our_url") or "",
1226
1277
  })
1278
+ # Per-candidate durable record: the run-level audit at the bottom of
1279
+ # main() is lost if this process is SIGKILLed mid-batch (browser-lock
1280
+ # hijack), leaving no local trace of what already posted. Separate
1281
+ # file from post-results.jsonl on purpose: that one is run-level and
1282
+ # the menu bar/dashboard parse it; don't mix schemas.
1283
+ try:
1284
+ _prog_path = os.path.join(
1285
+ REPO_DIR, "skill", "logs", "post-candidates.jsonl")
1286
+ with open(_prog_path, "a", encoding="utf-8") as _pf:
1287
+ _pf.write(json.dumps({
1288
+ "at": datetime.now(timezone.utc).isoformat(),
1289
+ "plan": plan_path.name,
1290
+ **candidate_results[-1],
1291
+ }) + "\n")
1292
+ except Exception:
1293
+ pass
1227
1294
  finally:
1228
1295
  _clear_activity()
1229
1296
  # Release the batch hold so the next scan/post can take the browser