@m13v/s4l 1.7.4-rc.2 → 1.7.4-rc.21

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -40,16 +40,39 @@ ENV_PREFIX = "S4L_EXP_"
40
40
  # Keep draft_prompt entries in sync with bin/server.js
41
41
  # DRAFT_PROMPT_VARIANT_DEFS and the arm strings in run-twitter-cycle.sh.
42
42
  DESCRIPTIONS = {
43
+ "draft_b_source": {
44
+ "human_derived": (
45
+ "Draft B explore slot: style distilled from real top-performing "
46
+ "human replies (daily synthesizer), least-used first"
47
+ ),
48
+ "model_invented": (
49
+ "Draft B explore slot: freshly invented style from the "
50
+ "standalone invention job (post 2026-07-10), least-used first"
51
+ ),
52
+ "scored_fallback": (
53
+ "Draft B explore pool was empty; fell back to a second scored "
54
+ "pick from the proven-style pool"
55
+ ),
56
+ },
43
57
  "draft_prompt": {
58
+ "treatment_v3": (
59
+ "style-as-form: the assigned style is the binding FORM (defining "
60
+ "move + per-style length + self-check), learned preferences apply "
61
+ "inside it; keeps the v2 skeleton ban"
62
+ ),
63
+ "control_v3": (
64
+ "plain draft directive, uniform length clamp, no structure ban"
65
+ ),
66
+ # v2 arms (skeleton-ban) retired 2026-07-10; v1 arms
67
+ # (decoupled-product-pivot) retired 2026-07-06; kept so any
68
+ # straggler card from an old plan still explains itself.
44
69
  "treatment_v2": (
45
- "skeleton ban: forbids the concede-then-reverse "
70
+ "v2, retired: skeleton ban, forbids the concede-then-reverse "
46
71
  '"easy X / hard Y" structure and forces varied entry points'
47
72
  ),
48
73
  "control_v2": (
49
- "current draft directive (style + product pivot), no structure ban"
74
+ "v2, retired: draft directive (style + product pivot), no structure ban"
50
75
  ),
51
- # v1 arms (decoupled-product-pivot) retired 2026-07-06; kept so any
52
- # straggler card from an old plan still explains itself.
53
76
  "treatment": "v1, retired: product pivot decoupled from the reply",
54
77
  "control": "v1, retired: original draft directive",
55
78
  },
@@ -99,8 +122,9 @@ def collect(env=None):
99
122
  if var.startswith(ENV_PREFIX) and (v or "").strip():
100
123
  out[var[len(ENV_PREFIX):].lower()] = v.strip()
101
124
  # 2026-07-06: the personal_brand persona directive is now ARM-AWARE in
102
- # run-twitter-cycle.sh (treatment_v2 adds the concede-then-reverse skeleton ban,
103
- # control_v2 does not), so the assigned draft_prompt arm DOES touch persona
125
+ # run-twitter-cycle.sh (treatment_v3 adds the skeleton ban + two-layer
126
+ # style/preferences contract, control_v3 does not), so the assigned
127
+ # draft_prompt arm DOES touch persona
104
128
  # drafts. Keep it stamped so the arm surfaces on persona cards and the per-arm
105
129
  # readout covers both lanes. (Previously dropped here because the persona
106
130
  # directive overrode both arms wholesale; that is no longer the case.)
@@ -47,13 +47,13 @@ import identity # noqa: E402 (lives next to this file in scripts/)
47
47
  s4l_env.mirror()
48
48
 
49
49
  # Keep in sync with AUTOPILOT_STALL_SECONDS (menubar) / AUTOPILOT_STALL_MS (index.ts).
50
- STALL_SECONDS = 180
50
+ STALL_SECONDS = 1200
51
51
  # A job CLAIMED but never finished (sits in running/ this long) means a worker
52
52
  # picked it up and then died mid-run — the claude -p drafting child never came up
53
53
  # or crashed. Must be generous enough to clear the longest real drafting turn so a
54
54
  # healthy run never trips it. Keep in sync with AUTOPILOT_RUNNING_STALL_SECONDS
55
55
  # (menubar). See _oldest_running_age.
56
- RUNNING_STALL_SECONDS = 900
56
+ RUNNING_STALL_SECONDS = 1200
57
57
  # Require the stall to persist this many consecutive checks before paging, so a
58
58
  # transient slow claim (e.g. right after a Claude restart) doesn't false-alarm.
59
59
  # At StartInterval 120 that is ~6 min of continuous stall.
@@ -144,6 +144,59 @@ def _render_media_block(media) -> str:
144
144
  )
145
145
 
146
146
 
147
+ def _build_chain_block(row) -> str:
148
+ """Conversation chain reconstructed from replies.parent_reply_id linkage.
149
+
150
+ Walks ancestors bottom-up via GET /api/v1/replies/:id and renders the
151
+ chain root-first, each hop showing the inbound comment and (when we
152
+ responded) our reply. Empty string when the row has no parent linkage:
153
+ the root post itself already rides PENDING_DATA via the posts JOIN
154
+ (our_content / thread_title), so a chain block would add nothing.
155
+
156
+ Like counterparty_history_block, the block is self-titled and lands
157
+ inline in PENDING_DATA — no shell-side prompt change needed.
158
+ """
159
+ parent_id = row.get("parent_reply_id")
160
+ if not parent_id:
161
+ return ""
162
+ hops = []
163
+ seen = set()
164
+ cur = parent_id
165
+ for _ in range(10):
166
+ if not cur or cur in seen:
167
+ break
168
+ seen.add(cur)
169
+ try:
170
+ resp = api_get(f"/api/v1/replies/{cur}")
171
+ except Exception:
172
+ break
173
+ r = (resp.get("data") or {}).get("reply") or {}
174
+ if not r:
175
+ break
176
+ hops.append(r)
177
+ cur = r.get("parent_reply_id")
178
+ if not hops:
179
+ return ""
180
+
181
+ def _one_line(text):
182
+ return " ".join((text or "").split())
183
+
184
+ lines = [
185
+ "## Conversation chain (reconstructed from our DB; root first — "
186
+ "the row you are drafting for replies to the LAST message)"
187
+ ]
188
+ for r in reversed(hops):
189
+ lines.append(f"@{r.get('their_author') or '?'}: {_one_line(r.get('their_content'))}")
190
+ ours = _one_line(r.get("our_reply_content"))
191
+ if ours:
192
+ lines.append(f" our reply: {ours}")
193
+ lines.append(
194
+ f"@{row.get('their_author') or '?'}: {_one_line(row.get('their_content'))}"
195
+ " <- you are replying to this"
196
+ )
197
+ return "\n".join(lines)
198
+
199
+
147
200
  def cmd_pending_data(batch_size: int) -> int:
148
201
  try:
149
202
  from account_resolver import resolve as _resolve_account # noqa: WPS433
@@ -173,38 +226,50 @@ def cmd_pending_data(batch_size: int) -> int:
173
226
  # top slot then and get enriched.
174
227
  ENRICH_TOP_N = 60
175
228
  history_blocks = [""] * len(rows)
229
+ chain_blocks = [""] * len(rows)
176
230
  try:
177
231
  from concurrent.futures import ThreadPoolExecutor
178
232
  from counterparty_history import get_counterparty_history_block
179
233
 
180
234
  def _enrich(r):
181
235
  author = r.get("their_author")
182
- if not author:
183
- return ""
236
+ history = ""
237
+ if author:
238
+ try:
239
+ _disengage, history = get_counterparty_history_block(
240
+ platform="x",
241
+ author=author,
242
+ current_post_id=r.get("post_id"),
243
+ current_reply_id=r.get("id"),
244
+ )
245
+ history = history or ""
246
+ except Exception as e:
247
+ print(
248
+ f"[engage_twitter_helper] counterparty_history failed "
249
+ f"for @{author}: {e}",
250
+ file=sys.stderr,
251
+ )
184
252
  try:
185
- _disengage, block = get_counterparty_history_block(
186
- platform="x",
187
- author=author,
188
- current_post_id=r.get("post_id"),
189
- current_reply_id=r.get("id"),
190
- )
191
- return block or ""
253
+ chain = _build_chain_block(r)
192
254
  except Exception as e:
193
255
  print(
194
- f"[engage_twitter_helper] counterparty_history failed "
195
- f"for @{author}: {e}",
256
+ f"[engage_twitter_helper] chain block failed "
257
+ f"for reply {r.get('id')}: {e}",
196
258
  file=sys.stderr,
197
259
  )
198
- return ""
260
+ chain = ""
261
+ return (history, chain)
199
262
 
200
263
  top_rows = rows[:ENRICH_TOP_N]
201
264
  with ThreadPoolExecutor(max_workers=8) as ex:
202
- for idx, block in enumerate(ex.map(_enrich, top_rows)):
203
- history_blocks[idx] = block
265
+ for idx, (history, chain) in enumerate(ex.map(_enrich, top_rows)):
266
+ history_blocks[idx] = history
267
+ chain_blocks[idx] = chain
204
268
  non_empty = sum(1 for b in history_blocks if b)
269
+ chains_non_empty = sum(1 for b in chain_blocks if b)
205
270
  print(
206
- f"[engage_twitter_helper] counterparty_history enriched "
207
- f"{len(top_rows)}/{len(rows)} rows ({non_empty} with non-empty block)",
271
+ f"[engage_twitter_helper] enriched {len(top_rows)}/{len(rows)} rows "
272
+ f"(history={non_empty}, chain={chains_non_empty} non-empty)",
208
273
  file=sys.stderr,
209
274
  )
210
275
  except Exception as e:
@@ -215,7 +280,7 @@ def cmd_pending_data(batch_size: int) -> int:
215
280
  )
216
281
 
217
282
  out = []
218
- for r, history_block in zip(rows, history_blocks):
283
+ for r, history_block, chain_block in zip(rows, history_blocks, chain_blocks):
219
284
  out.append({
220
285
  "id": r.get("id"),
221
286
  "platform": r.get("platform"),
@@ -231,6 +296,7 @@ def cmd_pending_data(batch_size: int) -> int:
231
296
  "is_our_original_post": int(r.get("is_our_original_post") or 0),
232
297
  "project_name": r.get("project_name"),
233
298
  "counterparty_history_block": history_block,
299
+ "conversation_chain_block": chain_block,
234
300
  "their_media_block": _render_media_block(r.get("their_media")),
235
301
  })
236
302
  # json_agg(...) returns null when the array is empty; engage-twitter.sh's
@@ -36,7 +36,7 @@ import json
36
36
  import os
37
37
  import random
38
38
  import sys as _sys_mod
39
- from datetime import datetime, timezone
39
+ from datetime import datetime, timedelta, timezone
40
40
 
41
41
  # ── Style taxonomy ──────────────────────────────────────────────────
42
42
 
@@ -1341,6 +1341,106 @@ def pick_style_for_post(platform, context="posting",
1341
1341
  }
1342
1342
 
1343
1343
 
1344
+ # Days back a model_invented style still counts as "recent" for the Draft-B
1345
+ # exploration pool. The registry holds ~900 pre-2026-07-10 inventions that
1346
+ # are mostly clones of the agree-then-relocate skeleton (see the
1347
+ # invent_styles.py docstring); the exploration slot exists to trial the
1348
+ # standalone job's structurally-diverse output, not to resurrect those.
1349
+ # human_derived rows are NOT date-gated: they arrive one per platform per
1350
+ # day and least-used ordering naturally favors the fresh ones.
1351
+ EXPLORATION_INVENTED_MAX_AGE_DAYS = 14
1352
+ # Hard floor: never trial inventions from before the inline INVENT_RATE was
1353
+ # zeroed (2026-07-10). Anything older is the clone flood, regardless of how
1354
+ # recent the rolling window makes it look.
1355
+ EXPLORATION_INVENTED_EPOCH = datetime(2026, 7, 10, 21, 0,
1356
+ tzinfo=timezone.utc)
1357
+
1358
+
1359
+ def pick_exploration_style(platform, context="posting", exclude=None,
1360
+ rng=None):
1361
+ """Draft-B exploration picker (2026-07-11). NEVER invents.
1362
+
1363
+ Returns an assignment dict in the exact pick_style_for_post() shape
1364
+ (so s4l_render_style_block and every downstream consumer work
1365
+ unchanged) with an extra "source" key in {"human_derived",
1366
+ "model_invented"}, or None when the pool is empty / anything fails
1367
+ (caller falls back to the scored picker).
1368
+
1369
+ Pool: registry rows with kind='human_derived' (any age) plus
1370
+ kind='model_invented' rows registered in the last
1371
+ EXPLORATION_INVENTED_MAX_AGE_DAYS days. Selection is LEAST-USED first
1372
+ (30-day post count on this platform via compute_target_distribution),
1373
+ uniform among the up-to-5 least-used, so every new style gets trial
1374
+ exposure instead of waiting behind the score-weighted sampler's
1375
+ winner-take-most weights. This is the distribution channel for the
1376
+ standalone invent_styles.py job; the review card's pick plus the
1377
+ posted draft's engagement write the style's first real score, and
1378
+ winners graduate into the Draft-A pool through the normal sampler.
1379
+ """
1380
+ rnd = rng or random
1381
+ try:
1382
+ never = set(PLATFORM_POLICY.get(platform, {}).get("never", []))
1383
+ skip = set(exclude or ()) | never
1384
+ registry = _fetch_registry_styles()
1385
+ cutoff = max(
1386
+ datetime.now(timezone.utc)
1387
+ - timedelta(days=EXPLORATION_INVENTED_MAX_AGE_DAYS),
1388
+ EXPLORATION_INVENTED_EPOCH,
1389
+ )
1390
+ pool = {}
1391
+ for name, entry in registry.items():
1392
+ if name in skip:
1393
+ continue
1394
+ if (entry.get("status") or "active") != "active":
1395
+ continue
1396
+ kind = entry.get("kind")
1397
+ if kind == "human_derived":
1398
+ pool[name] = entry
1399
+ elif kind == "model_invented":
1400
+ invented_at = _parse_iso_utc(entry.get("invented_at"))
1401
+ if invented_at is not None and invented_at >= cutoff:
1402
+ pool[name] = entry
1403
+ if not pool:
1404
+ return None
1405
+ usage = {r["style"]: int(r.get("n") or 0)
1406
+ for r in compute_target_distribution(platform,
1407
+ context=context)}
1408
+ names = list(pool.keys())
1409
+ rnd.shuffle(names) # random tie order before the stable sort
1410
+ names.sort(key=lambda s: usage.get(s, 0))
1411
+ chosen = rnd.choice(names[:5])
1412
+ entry = pool[chosen]
1413
+ return {
1414
+ "mode": "use",
1415
+ "style": chosen,
1416
+ "description": entry.get("description"),
1417
+ "example": entry.get("example"),
1418
+ "note": entry.get("note"),
1419
+ "target_chars": entry.get("target_chars") or DEFAULT_TARGET_CHARS,
1420
+ "source": entry.get("kind"),
1421
+ "usage_n_30d": usage.get(chosen, 0),
1422
+ "reference_styles": [],
1423
+ "distribution_snapshot": [],
1424
+ "picked_at": datetime.now(timezone.utc).isoformat(
1425
+ timespec="seconds"),
1426
+ }
1427
+ except Exception:
1428
+ return None
1429
+
1430
+
1431
+ def _parse_iso_utc(value):
1432
+ """Parse an ISO timestamp into aware-UTC; None on any failure."""
1433
+ if not value:
1434
+ return None
1435
+ try:
1436
+ dt = datetime.fromisoformat(str(value).replace("Z", "+00:00"))
1437
+ if dt.tzinfo is None:
1438
+ dt = dt.replace(tzinfo=timezone.utc)
1439
+ return dt.astimezone(timezone.utc)
1440
+ except (ValueError, TypeError):
1441
+ return None
1442
+
1443
+
1344
1444
  def get_assigned_style_prompt(platform, assignment, context="posting"):
1345
1445
  """Compact prompt block built from a pick_style_for_post() assignment.
1346
1446
 
@@ -1364,31 +1464,126 @@ def get_assigned_style_prompt(platform, assignment, context="posting"):
1364
1464
  lines = []
1365
1465
 
1366
1466
  if assignment["mode"] == "use":
1367
- lines.append(f"## Your assigned engagement style: **{assignment['style']}**")
1368
- lines.append("")
1369
- lines.append(
1370
- f"This style was selected by the picker (weighted by live "
1371
- f"click-driven performance across {platform}). Use it. Do not "
1372
- f"swap it for a different listed style."
1373
- )
1374
- lines.append("")
1375
- lines.append(f"Platform tone: {policy.get('note', '')}")
1376
- lines.append("")
1377
- lines.append(f"**{assignment['style']}**: {assignment.get('description', '')}")
1378
- if assignment.get("example"):
1379
- lines.append(f' Example: "{assignment["example"]}"')
1380
- if assignment.get("note"):
1381
- lines.append(f" Note: {assignment['note']}")
1382
- # LENGTH A/B CONCLUDED 2026-06-04: control won, so the prompt always
1383
- # uses the legacy generic length guidance. The treatment's per-style
1384
- # target prompt remains preserved only in the shipped experiment card.
1385
- lines.append("")
1386
- lines.append(
1387
- "**LENGTH: keep it tight.** One or two sentences, well under the "
1388
- "250-character Twitter limit. A short, sharp reply almost always "
1389
- "beats a paragraph. This applies to the comment text only; any "
1390
- "link/CTA the system appends afterward is separate."
1391
- )
1467
+ # Draft-prompt A/B v3 (style-as-form, 2026-07-10): the treatment_v3
1468
+ # arm renders the style as the BINDING FORM of the draft (defining
1469
+ # move + per-style length for EVERY assignment + end-of-block
1470
+ # self-check + two-layer learned_preferences contract). The arm is
1471
+ # read from the env HERE because this block is rendered inside the
1472
+ # cycle process where run-twitter-cycle.sh assigns and exports
1473
+ # S4L_DRAFT_PROMPT_VARIANT (stamp-at-source: that same process also
1474
+ # stamps the arm onto every plan candidate via active_experiments).
1475
+ # Downstream/post-time consumers never re-read this env. Unset or
1476
+ # control_v3 (and every non-twitter caller, which never has the env)
1477
+ # renders the legacy block below unchanged.
1478
+ _dp_arm = (os.environ.get("S4L_DRAFT_PROMPT_VARIANT") or "").strip()
1479
+ if _dp_arm == "treatment_v3":
1480
+ _tc = assignment.get("target_chars") or DEFAULT_TARGET_CHARS
1481
+ lines.append(
1482
+ f"## Your assigned engagement style: **{assignment['style']}** "
1483
+ "(this is the FORM of the draft, not a flavor hint)"
1484
+ )
1485
+ lines.append("")
1486
+ lines.append(
1487
+ f"This style was selected by the picker (weighted by live "
1488
+ f"click-driven performance across {platform}). Use it. Do not "
1489
+ f"swap it for a different listed style."
1490
+ )
1491
+ lines.append("")
1492
+ lines.append(f"Platform tone: {policy.get('note', '')}")
1493
+ lines.append("")
1494
+ lines.append(f"**{assignment['style']}**: {assignment.get('description', '')}")
1495
+ if assignment.get("example"):
1496
+ lines.append(f' Example: "{assignment["example"]}"')
1497
+ if assignment.get("note"):
1498
+ lines.append(f" Note: {assignment['note']}")
1499
+ lines.append("")
1500
+ lines.append(
1501
+ "Commit to the form BEFORE writing. From the description and "
1502
+ "example above, identify the style's DEFINING MOVE (the one "
1503
+ "thing a draft in this style must contain: a specific number, "
1504
+ "a question, a flat disagreement, a confession, whatever the "
1505
+ "description names) and build the reply around that move. The "
1506
+ "test: with the topic removed, a reader should be able to "
1507
+ "identify this style from the draft's shape alone. If your "
1508
+ "draft would read the same under any other style name, it "
1509
+ "does not conform; rewrite it."
1510
+ )
1511
+ lines.append("")
1512
+ lines.append(
1513
+ f"**LENGTH: aim for about {int(_tc)} characters** "
1514
+ "(this style's own winning length; within about 30% either "
1515
+ "way is fine, never above 250). Let the target set the form: "
1516
+ "a very short target means one clipped line, a long one can "
1517
+ "breathe. Do NOT default to the usual two-sentence shape. "
1518
+ "This applies to the comment text only; any link/CTA the "
1519
+ "system appends afterward is separate."
1520
+ )
1521
+ lines.append("")
1522
+ lines.append(
1523
+ "Learned user preferences (the learned_preferences block in "
1524
+ "the project context) apply INSIDE this form: they control "
1525
+ "voice, wording, and what to avoid; they do not replace the "
1526
+ "style's structure or length. If a preference seems to "
1527
+ "conflict with the style's defining move, keep the move and "
1528
+ "satisfy the preference within it."
1529
+ )
1530
+ lines.append("")
1531
+ lines.append(
1532
+ "SELF-CHECK before returning: (1) does the draft contain the "
1533
+ "style's defining move? (2) is the length within about 30% "
1534
+ "of the target? If either fails, rewrite once."
1535
+ )
1536
+ else:
1537
+ lines.append(f"## Your assigned engagement style: **{assignment['style']}**")
1538
+ lines.append("")
1539
+ lines.append(
1540
+ f"This style was selected by the picker (weighted by live "
1541
+ f"click-driven performance across {platform}). Use it. Do not "
1542
+ f"swap it for a different listed style."
1543
+ )
1544
+ lines.append("")
1545
+ lines.append(f"Platform tone: {policy.get('note', '')}")
1546
+ lines.append("")
1547
+ lines.append(f"**{assignment['style']}**: {assignment.get('description', '')}")
1548
+ if assignment.get("example"):
1549
+ lines.append(f' Example: "{assignment["example"]}"')
1550
+ if assignment.get("note"):
1551
+ lines.append(f" Note: {assignment['note']}")
1552
+ # LENGTH A/B CONCLUDED 2026-06-04: control won, so the prompt always
1553
+ # uses the legacy generic length guidance. The treatment's per-style
1554
+ # target prompt remains preserved only in the shipped experiment card.
1555
+ #
1556
+ # EXCEPTION (2026-07-11, Draft-B explore slot): exploration
1557
+ # assignments (pick_exploration_style, marked by `source`) honor the
1558
+ # style's own target_chars so the A/B pair diverges on the length
1559
+ # axis too. The uniform clamp flattened every draft to the same
1560
+ # 2-sentence shape (user: "the older drafts looked all very similar
1561
+ # in terms of the length"). The 06-04 conclusion still governs the
1562
+ # scored path, which never carries `source`.
1563
+ #
1564
+ # (The draft-prompt v3 treatment arm above supersedes this clamp
1565
+ # for its cycles; this legacy branch IS the v3 control arm.)
1566
+ _explore_tc = assignment.get("target_chars")
1567
+ if (assignment.get("source") in ("human_derived", "model_invented")
1568
+ and _explore_tc):
1569
+ lines.append("")
1570
+ lines.append(
1571
+ f"**LENGTH: aim for about {int(_explore_tc)} characters** "
1572
+ "(this style's own winning length; within about 30% either "
1573
+ "way is fine, never above 250). Let the target set the form: "
1574
+ "a very short target means one clipped line, a long one can "
1575
+ "breathe. Do NOT default to the usual two-sentence shape. "
1576
+ "This applies to the comment text only; any link/CTA the "
1577
+ "system appends afterward is separate."
1578
+ )
1579
+ else:
1580
+ lines.append("")
1581
+ lines.append(
1582
+ "**LENGTH: keep it tight.** One or two sentences, well under the "
1583
+ "250-character Twitter limit. A short, sharp reply almost always "
1584
+ "beats a paragraph. This applies to the comment text only; any "
1585
+ "link/CTA the system appends afterward is separate."
1586
+ )
1392
1587
  lines.append("")
1393
1588
  lines.append(
1394
1589
  'In your output JSON, set "engagement_style" to exactly '