@m13v/s4l 1.7.4-rc.2 → 1.7.4-rc.21

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -15,12 +15,26 @@ This job runs OUTSIDE any drafting context, on the operator Mac only:
15
15
  style), so invention must NOT fan out per install the way topic
16
16
  invention does. One central daily run is the correct scope; that is
17
17
  the deliberate difference from invent_topics.py's per-install kicker.
18
- - The prompt carries the existing style universe and explicitly forbids
19
- the dominant structural family; the ask is a style from an
20
- UNREPRESENTED family, not a riff on the champion.
18
+ - VERBALIZED SAMPLING (2026-07-10, arXiv 2510.01171): a single "invent
19
+ one style" ask reliably returns the modal answer (typicality bias from
20
+ preference training), which is why 653 of ~950 registered styles
21
+ cluster at 53-98 target_chars and one rhetorical family. Instead the
22
+ prompt asks for N_CANDIDATES candidate styles WITH a verbalized
23
+ typicality probability each, and selection takes the LOWEST-probability
24
+ candidate that survives dedup: we sample the tail of the distribution,
25
+ not the mode. This is the principled fix; the family blocklist below is
26
+ only a backstop.
27
+ - The existing registry is framed as OCCUPIED NICHES to stay out of
28
+ (quality-diversity framing, arXiv 2310.13032), NOT as reference
29
+ inspiration; showing top performers as exemplars is what anchored the
30
+ retired inline path to the champion family.
31
+ - target_chars DIVERSITY: the prompt shows live length-band occupancy
32
+ and requires candidates to spread across bands, biased to the
33
+ least-occupied ones, so invention stops minting 80-char styles.
21
34
  - Post-hoc semantic dedup (token-Jaccard on description+example, plus a
22
- reframe-family heuristic) rejects near-clones; a rejection re-prompts
23
- with a grown avoid-list, at most DUPE_RETRIES times.
35
+ reframe-family heuristic) rejects near-clones; when every candidate is
36
+ rejected we re-prompt with a grown avoid-list, at most DUPE_RETRIES
37
+ times.
24
38
  - Accepted styles are registered via engagement_styles.register_style
25
39
  (kind='model_invented'), same as the retired inline path, so pickers
26
40
  see them on their next tick with zero other wiring.
@@ -52,6 +66,12 @@ SCRIPT_TAG = "invent-styles"
52
66
  CALL_TIMEOUT_SEC = 420
53
67
  DUPE_RETRIES = 3
54
68
  SIMILARITY_THRESHOLD = 0.5 # Jaccard on description+example tokens
69
+ N_CANDIDATES = 5 # verbalized-sampling fan-out per call
70
+
71
+ # Length bands for target_chars diversity. Occupancy is computed live from
72
+ # the registry and shown in the prompt; candidates must spread across bands
73
+ # and bias toward the least-occupied ones.
74
+ LENGTH_BANDS = [(30, 60), (60, 100), (100, 150), (150, 200), (200, 250)]
55
75
 
56
76
  # Heuristic markers of the saturated agree-then-relocate/reframe family.
57
77
  # A proposal whose description/example leans on these is a clone of the
@@ -98,6 +118,23 @@ def _find_near_dupe(proposal, universe):
98
118
  return None
99
119
 
100
120
 
121
+ def band_occupancy(universe):
122
+ """[(lo, hi, count), ...] for LENGTH_BANDS over the registry's
123
+ target_chars. The live histogram the prompt shows so candidates can
124
+ claim underrepresented bands."""
125
+ counts = [0] * len(LENGTH_BANDS)
126
+ for e in universe.values():
127
+ try:
128
+ tc = int(e.get("target_chars") or 0)
129
+ except (TypeError, ValueError):
130
+ continue
131
+ for i, (lo, hi) in enumerate(LENGTH_BANDS):
132
+ if lo <= tc < hi:
133
+ counts[i] += 1
134
+ break
135
+ return [(lo, hi, counts[i]) for i, (lo, hi) in enumerate(LENGTH_BANDS)]
136
+
137
+
101
138
  def build_prompt(universe, avoid):
102
139
  names = sorted(universe.keys())
103
140
  # Detail only a bounded sample (prompt-size guard): the seeds plus the
@@ -117,25 +154,42 @@ def build_prompt(universe, avoid):
117
154
  "\nAlready proposed and REJECTED this run (do not resubmit or "
118
155
  "paraphrase): " + ", ".join(avoid) + "\n"
119
156
  )
157
+ bands = band_occupancy(universe)
158
+ band_lines = "\n".join(
159
+ f"- {lo}-{hi} chars: {count} existing styles"
160
+ f"{' <- UNDERREPRESENTED, prefer this band' if count == min(c for _, _, c in bands) else ''}"
161
+ for lo, hi, count in bands
162
+ )
120
163
  return f"""You maintain the engagement-style registry for a social reply system. A style is a named rhetorical TEMPLATE (description + one example reply) that drafters follow when writing short replies on X/Twitter and Reddit.
121
164
 
122
- PROBLEM: the registry is saturated with ONE structural family: agree-then-relocate (concede the surface point, move the spotlight to the hidden/harder/unmeasured part, often "X is easy, Y is the real work" or "X was never the point, Y is"). Do NOT invent another member of that family, however disguised.
165
+ The registry below is OCCUPIED TERRITORY, not inspiration. Your job is to find empty niches: structural families and length bands that no existing style covers. Do not riff on what is already there.
166
+
167
+ The registry is saturated with ONE structural family: agree-then-relocate (concede the surface point, move the spotlight to the hidden/harder/unmeasured part, often "X is easy, Y is the real work" or "X was never the point, Y is"). Do NOT propose any member of that family, however disguised.
123
168
 
124
169
  EXISTING STYLE NAMES ({len(names)} total):
125
170
  {", ".join(names)}
126
171
 
127
- REPRESENTATIVE DETAILS (sample):
172
+ REPRESENTATIVE DETAILS (occupied niches, sample):
128
173
  {chr(10).join(detailed)}
174
+
175
+ LENGTH-BAND OCCUPANCY (live registry histogram of target_chars):
176
+ {band_lines}
129
177
  {avoid_block}
130
- TASK: invent exactly ONE genuinely new style from a structural family that is missing or rare above. Families worth mining (pick ONE, or another you identify): direct answer with zero framing; first-person confession of a specific failure; pure curious question with no thesis; dry understatement one-liner; enthusiastic cosign with one concrete addition; flat disagreement stated plainly without conceding anything first; a tiny numbered checklist; a vivid analogy that does NOT end in a lesson; deadpan humor riffing on the thread's wording.
178
+ TASK (verbalized sampling): generate {N_CANDIDATES} CANDIDATE styles. For each, verbalize a `probability`: how likely this exact style would be as a language model's single first answer to "invent a new reply style" (0.0-1.0, honest, they need not sum to 1). Deliberately include tail candidates: at least 3 of the {N_CANDIDATES} must have probability under 0.10, meaning genuinely atypical moves a model would almost never produce first. We will programmatically select from the LOW-probability tail, so the obvious candidates are effectively discards; put your creativity into the tail.
179
+
180
+ Diversity requirements across the {N_CANDIDATES} candidates:
181
+ - Each from a DIFFERENT structural family. Families worth mining (or others you identify): direct answer with zero framing; first-person confession of a specific failure; pure curious question with no thesis; dry understatement one-liner; enthusiastic cosign with one concrete addition; flat disagreement stated plainly without conceding anything first; a tiny numbered checklist; a vivid analogy that does NOT end in a lesson; deadpan humor riffing on the thread's wording.
182
+ - Each in a DIFFERENT length band from the histogram above, biased toward the least-occupied bands. `target_chars` must be the actual length of your example, and the example must genuinely inhabit its band (a 200-char style is narrative, not a padded one-liner).
131
183
 
132
- Rules:
133
- - The example must read like a real human reply (lowercase ok), 40-220 chars, NO links, NO product names.
184
+ Per-candidate rules:
185
+ - `description` must state the style's DEFINING MOVE (the one thing every draft in this style must contain) and its OPENING (how the first words enter: e.g. lowercase noun, a number, the question itself). A future model reading only the description must know exactly what shape to write.
186
+ - `note` states when to use / when not to, and how a product mention would enter this style if ever (one clause; the link/CTA layer is downstream, so no URLs or link mechanics).
187
+ - The example must read like a real human reply (lowercase ok), NO links, NO product names.
134
188
  - The style must be usable across many products and threads, not thread-specific.
135
- - Your answer will be dedup-checked by token similarity against every existing style's description+example; if you cannot find a genuinely different move, return the saturation envelope instead of forcing a paraphrase.
189
+ - Every candidate is dedup-checked by token similarity against every existing style's description+example; if you cannot field {N_CANDIDATES} genuinely different moves, return the saturation envelope instead of forcing paraphrases.
136
190
 
137
191
  Answer with ONLY one JSON object, no prose, in one of these two shapes:
138
- {{"name": "snake_case_name", "description": "...", "example": "...", "why_existing_didnt_fit": "...", "target_chars": <int, the length the example is>}}
192
+ {{"candidates": [{{"name": "snake_case_name", "description": "...", "example": "...", "note": "...", "why_existing_didnt_fit": "...", "target_chars": <int>, "probability": <float>}}, ...]}}
139
193
  {{"saturated": true, "reason": "..."}}"""
140
194
 
141
195
 
@@ -202,21 +256,43 @@ def main():
202
256
  print(f"[invent_styles] slot={slot} model reports saturation: "
203
257
  f"{obj.get('reason', '')[:200]}", file=sys.stderr)
204
258
  break
205
- name = str(obj.get("name") or "").strip()
206
- if not re.fullmatch(r"[a-z0-9_]{3,60}", name):
207
- print(f"[invent_styles] slot={slot} bad name {name!r}; retrying",
208
- file=sys.stderr)
209
- avoid.append(name or "(unnamed)")
210
- continue
211
- dupe = _find_near_dupe({**obj, "name": name}, universe)
212
- if dupe:
213
- reason, existing = dupe
214
- print(f"[invent_styles] slot={slot} rejected {name!r}: {reason} "
215
- f"vs {existing}", file=sys.stderr)
216
- avoid.append(name)
259
+ candidates = obj.get("candidates")
260
+ if not isinstance(candidates, list) or not candidates:
261
+ print(f"[invent_styles] slot={slot} attempt={attempt} no "
262
+ f"candidates array in output", file=sys.stderr)
217
263
  continue
218
- accepted = {**obj, "name": name}
219
- break
264
+ # Verbalized-sampling selection: walk the tail first (ascending
265
+ # verbalized probability = least typical candidate first) and
266
+ # take the first that survives dedup. The modal candidates at
267
+ # the top of the distribution only get a chance if every tail
268
+ # candidate is a clone.
269
+ def _prob(c):
270
+ try:
271
+ return float(c.get("probability"))
272
+ except (TypeError, ValueError):
273
+ return 1.0 # unparseable prob sorts as maximally typical
274
+ for cand in sorted(candidates, key=_prob):
275
+ name = str(cand.get("name") or "").strip()
276
+ if not re.fullmatch(r"[a-z0-9_]{3,60}", name):
277
+ print(f"[invent_styles] slot={slot} bad name {name!r}; "
278
+ f"skipping candidate", file=sys.stderr)
279
+ avoid.append(name or "(unnamed)")
280
+ continue
281
+ dupe = _find_near_dupe({**cand, "name": name}, universe)
282
+ if dupe:
283
+ reason, existing = dupe
284
+ print(f"[invent_styles] slot={slot} rejected {name!r} "
285
+ f"(p={_prob(cand):.2f}): {reason} vs {existing}",
286
+ file=sys.stderr)
287
+ avoid.append(name)
288
+ continue
289
+ accepted = {**cand, "name": name}
290
+ print(f"[invent_styles] slot={slot} selected tail candidate "
291
+ f"{name!r} (p={_prob(cand):.2f}) from "
292
+ f"{len(candidates)} candidates", file=sys.stderr)
293
+ break
294
+ if accepted:
295
+ break
220
296
  if not accepted:
221
297
  continue
222
298
  if args.dry_run:
@@ -227,6 +303,7 @@ def main():
227
303
  {
228
304
  "description": accepted.get("description", ""),
229
305
  "example": accepted.get("example", ""),
306
+ "note": accepted.get("note", ""),
230
307
  "why_existing_didnt_fit": accepted.get("why_existing_didnt_fit", ""),
231
308
  "target_chars": accepted.get("target_chars"),
232
309
  },
@@ -0,0 +1,234 @@
1
+ #!/usr/bin/env python3
2
+ """LinkedIn posting cadence: 4 active days, then a mandatory 2-day break.
3
+
4
+ Per user instruction (2026-07-11): LinkedIn activity should not run more than
5
+ four days at a time before taking a two-day break. "Active day" is counted
6
+ only when real posting/engagement activity actually happened that day (an
7
+ outage day does not count toward the four), and the break pauses ALL LinkedIn
8
+ traffic, including the passive presence-check job, not just posting.
9
+
10
+ Every LinkedIn entrypoint (skill/run-linkedin.sh, engage-linkedin.sh,
11
+ engage-dm-replies.sh, dm-outreach-linkedin.sh, audit-linkedin.sh,
12
+ linkedin-presence.sh) already gates on the existence of ONE file:
13
+ ~/.claude/social-autoposter/linkedin.killswitch
14
+ That file is scripts/linkedin_killswitch.py's antibot killswitch. This module
15
+ reuses the SAME file for a scheduled break (signal="scheduled_break") so every
16
+ entrypoint pauses automatically, with zero edits to the locked entrypoint
17
+ scripts. It never overwrites a REAL antibot signal, and linkedin_killswitch.py
18
+ is patched (recover-check) to never try to auto-recover a scheduled_break.
19
+
20
+ State lives at ~/.claude/social-autoposter/linkedin_cadence.json:
21
+ {
22
+ "phase": "active" | "break",
23
+ "phase_started": "2026-07-11T20:00:00Z",
24
+ "active_days": ["2026-07-13", "2026-07-14", ...] # UTC dates with
25
+ # confirmed posts,
26
+ # only meaningful
27
+ # while phase=active
28
+ }
29
+
30
+ CLI:
31
+ python3 scripts/linkedin_cadence.py enforce # one tick; called every 15m
32
+ python3 scripts/linkedin_cadence.py status # print state (json)
33
+ """
34
+
35
+ import json
36
+ import os
37
+ import sys
38
+ from datetime import datetime, timedelta, timezone
39
+
40
+ sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
41
+
42
+ import http_api # noqa: E402
43
+ import linkedin_killswitch as ks # noqa: E402
44
+
45
+ STATE_DIR = os.path.expanduser(
46
+ os.environ.get("LINKEDIN_KILLSWITCH_DIR", "~/.claude/social-autoposter")
47
+ )
48
+ STATE_FILE = os.path.expanduser(
49
+ os.environ.get("LINKEDIN_CADENCE_FILE", os.path.join(STATE_DIR, "linkedin_cadence.json"))
50
+ )
51
+
52
+ ACTIVE_DAYS_TARGET = int(os.environ.get("LINKEDIN_CADENCE_ACTIVE_DAYS", "4"))
53
+ BREAK_DAYS = int(os.environ.get("LINKEDIN_CADENCE_BREAK_DAYS", "2"))
54
+
55
+ SCHEDULED_BREAK_SIGNAL = "scheduled_break"
56
+
57
+
58
+ def _now():
59
+ return datetime.now(timezone.utc)
60
+
61
+
62
+ def _now_iso():
63
+ return _now().strftime("%Y-%m-%dT%H:%M:%SZ")
64
+
65
+
66
+ def _today_str():
67
+ return _now().date().isoformat()
68
+
69
+
70
+ def _parse_ts(ts):
71
+ return datetime.strptime(ts, "%Y-%m-%dT%H:%M:%SZ").replace(tzinfo=timezone.utc)
72
+
73
+
74
+ def _ensure_dir():
75
+ os.makedirs(STATE_DIR, exist_ok=True)
76
+
77
+
78
+ def load_state():
79
+ try:
80
+ with open(STATE_FILE, "r") as f:
81
+ return json.load(f)
82
+ except Exception:
83
+ return None
84
+
85
+
86
+ def save_state(state):
87
+ _ensure_dir()
88
+ tmp = STATE_FILE + ".tmp"
89
+ with open(tmp, "w") as f:
90
+ json.dump(state, f, indent=2)
91
+ f.write("\n")
92
+ os.replace(tmp, STATE_FILE)
93
+
94
+
95
+ def _default_state_starting_break():
96
+ # First-ever run: user asked to start immediately with a two-day break.
97
+ return {"phase": "break", "phase_started": _now_iso(), "active_days": []}
98
+
99
+
100
+ def _today_post_count():
101
+ """LinkedIn posts made today (UTC), via the same API the dashboard uses.
102
+
103
+ Best-effort: returns None on any failure so callers can skip counting
104
+ this tick rather than wrongly recording 0 activity."""
105
+ try:
106
+ resp = http_api.api_get(
107
+ "/api/v1/dashboard/posts-per-day", {"days": 1, "platform": "linkedin"}
108
+ )
109
+ rows = (resp or {}).get("data", {}).get("rows", [])
110
+ today = _today_str()
111
+ for row in rows:
112
+ if row.get("day") == today:
113
+ return int(row.get("posts_made") or 0)
114
+ return 0
115
+ except Exception as exc:
116
+ print(f"[linkedin_cadence] WARN: posts-per-day query failed: {exc}", file=sys.stderr)
117
+ return None
118
+
119
+
120
+ def _pause_marker_set():
121
+ """Set the shared killswitch file for a scheduled break, unless a REAL
122
+ antibot signal is already in charge (never stomp a genuine block)."""
123
+ payload = ks.read()
124
+ if payload is None:
125
+ ks.engage(
126
+ signal=SCHEDULED_BREAK_SIGNAL,
127
+ detail="cadence: scheduled 2-day pause after 4 active days",
128
+ send_email=False,
129
+ )
130
+ print("[linkedin_cadence] pause marker set (scheduled_break)", file=sys.stderr)
131
+ elif payload.get("signal") == SCHEDULED_BREAK_SIGNAL:
132
+ pass # already set by us; idempotent, no trail spam
133
+ else:
134
+ print(
135
+ f"[linkedin_cadence] real killswitch already active (signal="
136
+ f"{payload.get('signal')!r}); deferring to it, not overwriting",
137
+ file=sys.stderr,
138
+ )
139
+
140
+
141
+ def _pause_marker_clear_if_ours():
142
+ payload = ks.read()
143
+ if payload is not None and payload.get("signal") == SCHEDULED_BREAK_SIGNAL:
144
+ ks.clear()
145
+ print("[linkedin_cadence] pause marker cleared (break ended)", file=sys.stderr)
146
+
147
+
148
+ def enforce():
149
+ state = load_state() or _default_state_starting_break()
150
+ if load_state() is None:
151
+ save_state(state)
152
+ print(
153
+ f"[linkedin_cadence] no prior state; starting BREAK phase now "
154
+ f"({BREAK_DAYS}d)",
155
+ file=sys.stderr,
156
+ )
157
+
158
+ now = _now()
159
+ phase = state["phase"]
160
+
161
+ if phase == "break":
162
+ started = _parse_ts(state["phase_started"])
163
+ elapsed = now - started
164
+ if elapsed >= timedelta(days=BREAK_DAYS):
165
+ state = {"phase": "active", "phase_started": _now_iso(), "active_days": []}
166
+ save_state(state)
167
+ _pause_marker_clear_if_ours()
168
+ print(
169
+ f"[linkedin_cadence] break ended after {elapsed}; switching to ACTIVE",
170
+ file=sys.stderr,
171
+ )
172
+ phase = "active"
173
+ else:
174
+ remaining = timedelta(days=BREAK_DAYS) - elapsed
175
+ _pause_marker_set()
176
+ print(
177
+ f"[linkedin_cadence] BREAK phase: {remaining} remaining",
178
+ file=sys.stderr,
179
+ )
180
+ return
181
+
182
+ # phase == "active"
183
+ real_block = ks.read()
184
+ if real_block is not None and real_block.get("signal") != SCHEDULED_BREAK_SIGNAL:
185
+ print(
186
+ f"[linkedin_cadence] account down for a real reason (signal="
187
+ f"{real_block.get('signal')!r}); not counting today, not pausing",
188
+ file=sys.stderr,
189
+ )
190
+ return
191
+
192
+ count = _today_post_count()
193
+ today = _today_str()
194
+ if count is not None and count > 0 and today not in state["active_days"]:
195
+ state["active_days"].append(today)
196
+ save_state(state)
197
+ print(
198
+ f"[linkedin_cadence] activity confirmed today ({count} posts); "
199
+ f"active_days={len(state['active_days'])}/{ACTIVE_DAYS_TARGET} "
200
+ f"{state['active_days']}",
201
+ file=sys.stderr,
202
+ )
203
+
204
+ if len(state["active_days"]) >= ACTIVE_DAYS_TARGET:
205
+ state = {"phase": "break", "phase_started": _now_iso(), "active_days": []}
206
+ save_state(state)
207
+ _pause_marker_set()
208
+ print(
209
+ f"[linkedin_cadence] {ACTIVE_DAYS_TARGET} active days reached; "
210
+ f"switching to BREAK for {BREAK_DAYS}d",
211
+ file=sys.stderr,
212
+ )
213
+ else:
214
+ print(
215
+ f"[linkedin_cadence] ACTIVE phase: "
216
+ f"{len(state['active_days'])}/{ACTIVE_DAYS_TARGET} active days so far",
217
+ file=sys.stderr,
218
+ )
219
+
220
+
221
+ def main():
222
+ if len(sys.argv) < 2 or sys.argv[1] not in ("enforce", "status"):
223
+ print("usage: linkedin_cadence.py [enforce|status]", file=sys.stderr)
224
+ sys.exit(2)
225
+ if sys.argv[1] == "status":
226
+ state = load_state()
227
+ print(json.dumps(state if state is not None else {"phase": None}, indent=2))
228
+ sys.exit(0)
229
+ enforce()
230
+ sys.exit(0)
231
+
232
+
233
+ if __name__ == "__main__":
234
+ main()
@@ -145,6 +145,10 @@ VALID_SIGNALS = {
145
145
  "session_invalid_marker",
146
146
  "captcha_detected",
147
147
  "manual",
148
+ # Deliberate pause from scripts/linkedin_cadence.py (4-active-days /
149
+ # 2-day-break schedule). Not an incident: recover-check must never try to
150
+ # auto-recover it, and engage() should never email on it.
151
+ "scheduled_break",
148
152
  }
149
153
 
150
154
 
@@ -894,6 +898,14 @@ def _cmd_recover_check(args):
894
898
  if not is_active():
895
899
  print("recover-check: killswitch not active, nothing to recover", file=sys.stderr)
896
900
  sys.exit(1)
901
+ payload = read() or {}
902
+ if payload.get("signal") == "scheduled_break":
903
+ print(
904
+ "recover-check: scheduled_break active (cadence pause, not a real "
905
+ "incident); deferring to scripts/linkedin_cadence.py, not probing",
906
+ file=sys.stderr,
907
+ )
908
+ sys.exit(1)
897
909
  if is_terminal():
898
910
  print(
899
911
  "recover-check: TERMINAL (auto-recovery gave up); "
@@ -95,6 +95,12 @@ def build_block(platform, limit):
95
95
  "order_by": "posted_at",
96
96
  "order_dir": "desc",
97
97
  "limit": str(limit),
98
+ # Scope to THIS install's own posts (server filters on the
99
+ # authenticated X-Installation identity). Without this the
100
+ # default read returns the whole fleet's posts and the block
101
+ # would claim another account's replies as "yours"
102
+ # (found 2026-07-10: 8 of 20 rows were another install's).
103
+ "own_install": "true",
98
104
  },
99
105
  )
100
106
  rows = ((resp or {}).get("data") or {}).get("posts") or []
@@ -311,6 +311,50 @@ def scrape_timeline(send, me: str, want: int, max_scrolls: int = 30,
311
311
  return items[:want]
312
312
 
313
313
 
314
+ def _own_posted_ids() -> set:
315
+ """Status ids of everything S4L itself has posted from this install, via
316
+ /api/v1/posts (install-scoped by the X-Installation header). Used to keep
317
+ the bot's own output OUT of the author-voice exemplar pool: on accounts
318
+ where S4L has been active, a recency scan is dominated by S4L drafts, and
319
+ feeding those back as 'the author's voice' is a feedback loop. Best-effort:
320
+ any failure (offline, fresh install, no API) returns an empty set and the
321
+ scan proceeds unfiltered."""
322
+ try:
323
+ sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
324
+ from http_api import api_get # noqa: PLC0415
325
+ import re
326
+ ids: set = set()
327
+ # The route caps limit at 500, so walk the FULL posting history with a
328
+ # forward posted_at cursor (order_dir=asc + since>=). On prolific
329
+ # accounts a single newest-500 page misses older S4L posts that the
330
+ # profile scan still reaches (bit the operator account: 12.8k posted
331
+ # statuses, 2 of 5 picked exemplars were S4L's own). The since filter
332
+ # is inclusive, so overlap rows dedupe via the set; a page whose max
333
+ # posted_at equals the cursor would loop and breaks instead.
334
+ cursor = None
335
+ for _ in range(80): # hard cap 40k statuses
336
+ q = {"platform": "twitter", "has_our_url": "true", "limit": 500,
337
+ "order_by": "posted_at", "order_dir": "asc"}
338
+ if cursor:
339
+ q["since"] = cursor
340
+ resp = api_get("/api/v1/posts", q)
341
+ rows = ((resp or {}).get("data") or {}).get("posts") or []
342
+ for r in rows:
343
+ m = re.search(r"/status/(\d+)", str(r.get("our_url") or ""))
344
+ if m:
345
+ ids.add(m.group(1))
346
+ if len(rows) < 500:
347
+ break
348
+ page_max = max((r.get("posted_at") or "" for r in rows), default="")
349
+ if not page_max or page_max == cursor:
350
+ break
351
+ cursor = page_max
352
+ return ids
353
+ except Exception as e:
354
+ print(f"[scan_x_profile] own-posts exclusion unavailable: {e}", file=sys.stderr)
355
+ return set()
356
+
357
+
314
358
  # --------------------------------------------------------------------------- #
315
359
  # Engagement ranking + thread expansion for exemplar extraction.
316
360
  # --------------------------------------------------------------------------- #
@@ -409,8 +453,8 @@ GROUNDING_INSTRUCTIONS = (
409
453
  def main() -> int:
410
454
  ap = argparse.ArgumentParser()
411
455
  ap.add_argument("--handle", default=None, help="@handle to scan (default: live logged-in handle)")
412
- ap.add_argument("--posts", type=int, default=20, help="max original posts to collect")
413
- ap.add_argument("--comments", type=int, default=50, help="max replies/comments to collect")
456
+ ap.add_argument("--posts", type=int, default=60, help="max original posts to collect")
457
+ ap.add_argument("--comments", type=int, default=150, help="max replies/comments to collect")
414
458
  ap.add_argument("--top", type=int, default=5, help="how many top posts/replies to rank")
415
459
  ap.add_argument("--expand-threads", type=int, default=3,
416
460
  help="visit this many top posts' permalinks to capture thread "
@@ -446,21 +490,37 @@ def main() -> int:
446
490
  expect=f"/{handle}")
447
491
  profile = scrape_profile(send) if on_profile else {}
448
492
 
449
- # 2. Original posts (current page = posts tab).
450
- posts = scrape_timeline(send, handle, args.posts) if on_profile else []
493
+ # 2. Everything S4L itself posted from this install gets excluded from
494
+ # BOTH surfaces (posts and replies): the exemplars must be the
495
+ # human's writing, not the bot's own output echoed back.
496
+ s4l_ids = _own_posted_ids()
497
+ if s4l_ids:
498
+ print(f"[scan_x_profile] excluding {len(s4l_ids)} s4l-posted statuses "
499
+ "from the exemplar pool", file=sys.stderr)
500
+
501
+ # 3. Original posts (current page = posts tab). max_scrolls tracks the
502
+ # requested depth: the scan is programmatic, so scrolling deeper
503
+ # costs only time, and end-of-feed stall detection stops it early
504
+ # on small accounts.
505
+ posts = (scrape_timeline(send, handle, args.posts,
506
+ max_scrolls=max(30, args.posts),
507
+ exclude_ids=s4l_ids)
508
+ if on_profile else [])
451
509
  post_ids = {p.get("id") for p in posts if p.get("id")}
452
510
 
453
- # 3. Replies / comments = the user's own articles on /with_replies that
511
+ # 4. Replies / comments = the user's own articles on /with_replies that
454
512
  # are NOT among the original posts (set subtraction, not DOM text).
455
513
  on_replies = _navigate(send, f"https://x.com/{handle}/with_replies",
456
514
  settle=4.0, expect=f"/{handle}/with_replies")
457
515
  comments = (
458
- scrape_timeline(send, handle, args.comments, exclude_ids=post_ids,
516
+ scrape_timeline(send, handle, args.comments,
517
+ max_scrolls=max(30, args.comments),
518
+ exclude_ids=post_ids | s4l_ids,
459
519
  capture_parents=True)
460
520
  if on_replies else []
461
521
  )
462
522
 
463
- # 4. Rank both surfaces by real engagement, then expand the top posts'
523
+ # 5. Rank both surfaces by real engagement, then expand the top posts'
464
524
  # permalinks to capture thread continuations (and untruncated text).
465
525
  top_posts = rank_top(posts, args.top)
466
526
  top_replies = rank_top(comments, args.top)
@@ -481,7 +541,8 @@ def main() -> int:
481
541
  "comments": comments,
482
542
  "top_posts": top_posts,
483
543
  "top_replies": top_replies,
484
- "counts": {"posts": len(posts), "comments": len(comments)},
544
+ "counts": {"posts": len(posts), "comments": len(comments),
545
+ "s4l_posted_excluded": len(s4l_ids)},
485
546
  "grounding_instructions": GROUNDING_INSTRUCTIONS,
486
547
  }
487
548
 
@@ -70,7 +70,7 @@ PERSONA_REQUIRED_FIELDS = ["name", "description", "voice", "search_topics"]
70
70
  CURRENT_WORKER_TASK_IDS = ("s4l-worker", "saps-worker")
71
71
  LEGACY_WORKER_TASK_IDS = ("saps-phase1-query", "saps-phase2b-draft")
72
72
  UPDATER_LABEL = "com.m13v.social-autoposter-update"
73
- AUTOPILOT_STALL_MS = 180_000
73
+ AUTOPILOT_STALL_MS = 1_200_000
74
74
 
75
75
  # Milestones overlaid with LIVE state for display (the rest keep their ledger
76
76
  # value). Mirrors the overlay in buildSnapshot().
@@ -567,9 +567,43 @@ def main():
567
567
  "summary, no bottom posts). This is the lean "
568
568
  "on-demand shape the drafting session calls "
569
569
  "after routing a candidate to a project."))
570
+ parser.add_argument("--invoked-by", default=None,
571
+ help=("Caller tag recorded in the on-demand invocation "
572
+ "ledger (the drafting prompt passes the cycle's "
573
+ "batch_id). Tracking only; no behavior change."))
570
574
  parser.add_argument("--json", action="store_true", help="Output as JSON")
571
575
  args = parser.parse_args()
572
576
 
577
+ # On-demand invocation ledger (2026-07-10). Any --project call is the
578
+ # on-demand per-project winners lookup the draft prompt offers; record it
579
+ # at the TOOL level (model self-reports are unreliable) so we can measure
580
+ # whether drafting sessions actually use the query. One JSON line per
581
+ # call in $S4L_STATE_DIR/top-performers-invocations.jsonl; the cycle
582
+ # counts lines for its batch_id after prep and logs a
583
+ # [project_top_performers] marker. Best-effort: never block the report.
584
+ if args.project:
585
+ try:
586
+ import datetime
587
+ state_dir = os.environ.get(
588
+ "S4L_STATE_DIR",
589
+ os.path.expanduser("~/.social-autoposter-mcp"))
590
+ os.makedirs(state_dir, exist_ok=True)
591
+ with open(os.path.join(state_dir,
592
+ "top-performers-invocations.jsonl"),
593
+ "a") as fh:
594
+ fh.write(json.dumps({
595
+ "ts": datetime.datetime.now(
596
+ datetime.timezone.utc).isoformat(),
597
+ "project": args.project,
598
+ "platform": args.platform,
599
+ "brief": bool(args.brief),
600
+ "top": args.top,
601
+ "invoked_by": args.invoked_by,
602
+ }) + "\n")
603
+ except Exception as exc:
604
+ print(f"[top_performers] invocation ledger write failed: {exc!r}",
605
+ file=sys.stderr)
606
+
573
607
  (summary, style_perf, top, bottom, fallback_top,
574
608
  top_by_group, top_by_style) = _fetch_report_via_api(
575
609
  platform=args.platform, project=args.project, top=args.top, bottom=args.bottom,