@m13v/s4l 1.7.4-rc.2 → 1.7.4-rc.21

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,312 @@
1
+ #!/usr/bin/env python3
2
+ """Fill parent-thread linkage on X replies discovered via the notifications lane.
3
+
4
+ The X notifications feed does not expose the parent tweet id, so
5
+ scan_twitter_mentions_browser.py inserts `replies` rows with mention_id only:
6
+ no post_id, no parent_reply_id, and often no project_name. This script
7
+ resolves the parent chain AFTER the fact, deterministically, with no browser
8
+ and no model: fxtwitter's public JSON (already used by fetch_twitter_t1.py)
9
+ returns `replying_to_status` for any tweet.
10
+
11
+ Per row with (post_id IS NULL AND parent_reply_id IS NULL):
12
+ 1. fxtwitter GET on their_comment_id -> parent tweet id + handle.
13
+ 2. Walk up the ancestor chain (bounded hops) until the root.
14
+ 3. First ancestor that is one of OUR posts (/api/v1/posts/lookup, wide
15
+ window) -> PATCH replies.post_id (+ project_name when the row has none).
16
+ 4. Immediate parent that is another tracked reply
17
+ (/api/v1/replies?their_comment_id=) -> PATCH parent_reply_id + depth.
18
+ 5. Root author -> PATCH thread_author_handle.
19
+
20
+ Terminal misses (tweet deleted, protected, not-a-reply with nothing to link)
21
+ are remembered in a local state file so recurring runs don't refetch forever.
22
+
23
+ Usage:
24
+ python3 scripts/enrich_reply_parents.py --limit 20 # recurring lane
25
+ python3 scripts/enrich_reply_parents.py --backfill # all missing rows
26
+ python3 scripts/enrich_reply_parents.py --ids 546832 542579 # specific rows
27
+ python3 scripts/enrich_reply_parents.py --limit 5 --dry-run
28
+
29
+ All DB I/O goes through the s4l.ai HTTP API (http_api), never direct SQL.
30
+ """
31
+ import argparse
32
+ import json
33
+ import os
34
+ import sys
35
+ import time
36
+ import urllib.error
37
+ import urllib.request
38
+
39
+ sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
40
+ from http_api import api_get, api_patch # noqa: E402
41
+
42
+ STATE_PATH = os.environ.get(
43
+ "S4L_ENRICH_PARENTS_STATE",
44
+ os.path.expanduser("~/.social-autoposter-enrich-parents.json"),
45
+ )
46
+ MAX_HOPS = 6
47
+ LOOKUP_DAYS = 3650
48
+
49
+
50
+ def load_state():
51
+ try:
52
+ with open(STATE_PATH) as f:
53
+ return json.load(f)
54
+ except Exception:
55
+ return {}
56
+
57
+
58
+ def save_state(state):
59
+ tmp = STATE_PATH + ".tmp"
60
+ with open(tmp, "w") as f:
61
+ json.dump(state, f)
62
+ os.replace(tmp, STATE_PATH)
63
+
64
+
65
+ def fetch_fxtwitter(handle, tweet_id):
66
+ """Returns (status, tweet_dict). status: 'ok' | 'gone' | 'transient'."""
67
+ url = f"https://api.fxtwitter.com/{handle or 'i'}/status/{tweet_id}"
68
+ req = urllib.request.Request(url, headers={"User-Agent": "social-autoposter/1.0"})
69
+ try:
70
+ with urllib.request.urlopen(req, timeout=15) as resp:
71
+ data = json.loads(resp.read())
72
+ except urllib.error.HTTPError as e:
73
+ # 401 = protected account, 404 = deleted. Both terminal.
74
+ if e.code in (401, 404):
75
+ return "gone", None
76
+ return "transient", None
77
+ except Exception:
78
+ return "transient", None
79
+ code = data.get("code")
80
+ if code == 200 and data.get("tweet"):
81
+ return "ok", data["tweet"]
82
+ if code in (401, 404):
83
+ return "gone", None
84
+ return "transient", None
85
+
86
+
87
+ def walk_ancestors(handle, tweet_id, sleep_s):
88
+ """Ancestor chain bottom-up: [(id, handle), ...] parent first, root last.
89
+
90
+ Returns (focal, chain, terminal): focal is the fetched tweet object for
91
+ tweet_id itself (None when it is gone/unfetchable — needed for quote
92
+ linkage), terminal is 'root' when the walk reached a non-reply tweet,
93
+ 'gone'/'transient' when a hop was deleted/protected/errored (the chain up
94
+ to that point is still usable, but root attribution is not), or 'hop_cap'.
95
+ """
96
+ chain = []
97
+ focal = None
98
+ cur_handle, cur_id = handle, tweet_id
99
+ for _ in range(MAX_HOPS):
100
+ status, tweet = fetch_fxtwitter(cur_handle, cur_id)
101
+ time.sleep(sleep_s)
102
+ if status != "ok":
103
+ # 'gone' is terminal (deleted/protected); 'transient' must NOT be
104
+ # remembered, the next run retries it.
105
+ return focal, chain, status
106
+ if focal is None:
107
+ focal = tweet
108
+ parent_id = tweet.get("replying_to_status")
109
+ parent_handle = tweet.get("replying_to") or ""
110
+ if not parent_id:
111
+ return focal, chain, "root"
112
+ chain.append((str(parent_id), parent_handle))
113
+ cur_handle, cur_id = parent_handle, parent_id
114
+ return focal, chain, "hop_cap"
115
+
116
+
117
+ def lookup_our_post(tweet_id):
118
+ resp = api_get(
119
+ "/api/v1/posts/lookup",
120
+ query={"platform": "twitter", "post_id": str(tweet_id), "days": str(LOOKUP_DAYS)},
121
+ )
122
+ return (resp.get("data") or {}).get("post") or None
123
+
124
+
125
+ def lookup_tracked_reply(tweet_id):
126
+ resp = api_get(
127
+ "/api/v1/replies",
128
+ query={"platform": "x", "their_comment_id": str(tweet_id), "limit": "1"},
129
+ )
130
+ rows = (resp.get("data") or {}).get("replies") or []
131
+ return rows[0] if rows else None
132
+
133
+
134
+ def lookup_our_posted_reply(tweet_id):
135
+ """Reply row where WE authored the tweet (our_reply_id / our_reply_url)."""
136
+ resp = api_get(
137
+ "/api/v1/replies",
138
+ query={"platform": "x", "our_reply_status_id": str(tweet_id), "limit": "1"},
139
+ )
140
+ rows = (resp.get("data") or {}).get("replies") or []
141
+ return rows[0] if rows else None
142
+
143
+
144
+ def fetch_work(limit, our_account=None, ids=None, before_id=None):
145
+ if ids:
146
+ out = []
147
+ for rid in ids:
148
+ resp = api_get(f"/api/v1/replies/{rid}")
149
+ row = (resp.get("data") or {}).get("reply")
150
+ if row:
151
+ out.append(row)
152
+ return out
153
+ query = {
154
+ "platform": "x",
155
+ "missing_parent": "1",
156
+ "order_by": "id",
157
+ "limit": str(limit),
158
+ }
159
+ if our_account:
160
+ query["our_account"] = our_account
161
+ if before_id:
162
+ query["before_id"] = str(before_id)
163
+ resp = api_get("/api/v1/replies", query=query)
164
+ return (resp.get("data") or {}).get("replies") or []
165
+
166
+
167
+ def enrich_row(row, state, sleep_s, dry_run):
168
+ rid = row["id"]
169
+ tid = str(row.get("their_comment_id") or "")
170
+ handle = (row.get("their_author") or "").lstrip("@")
171
+ if not tid:
172
+ return "no_tweet_id"
173
+ if state.get(tid):
174
+ return "state_skip"
175
+
176
+ focal, chain, terminal = walk_ancestors(handle, tid, sleep_s)
177
+ if focal is None:
178
+ if terminal == "transient":
179
+ return "transient" # retry next run, no state write
180
+ state[tid] = "gone" # deleted/protected focal tweet
181
+ return "gone"
182
+
183
+ patch = {}
184
+ matched_post = None
185
+ for anc_id, _anc_handle in chain:
186
+ post = lookup_our_post(anc_id)
187
+ if post:
188
+ matched_post = post
189
+ break
190
+ if matched_post:
191
+ patch["post_id"] = matched_post["id"]
192
+ if not row.get("project_name") and matched_post.get("project_name"):
193
+ patch["project_name"] = matched_post["project_name"]
194
+
195
+ if chain:
196
+ parent_id, _parent_handle = chain[0]
197
+ # Immediate parent: another inbound reply we track, or a reply WE
198
+ # posted (the dominant case: a fan replying to our engagement reply).
199
+ tracked = lookup_tracked_reply(parent_id)
200
+ if tracked and tracked["id"] != rid:
201
+ patch["parent_reply_id"] = tracked["id"]
202
+ patch["depth"] = (tracked.get("depth") or 1) + 1
203
+ else:
204
+ ours = lookup_our_posted_reply(parent_id)
205
+ if ours and ours["id"] != rid:
206
+ patch["parent_reply_id"] = ours["id"]
207
+ patch["depth"] = (ours.get("depth") or 1) + 1
208
+ if "post_id" not in patch and ours.get("post_id"):
209
+ patch["post_id"] = ours["post_id"]
210
+ if not row.get("project_name") and not patch.get("project_name") \
211
+ and ours.get("project_name"):
212
+ patch["project_name"] = ours["project_name"]
213
+
214
+ # Quote linkage: a quote-tweet of our post (or of a reply we posted) is
215
+ # engagement on our content even when replying_to_status is empty.
216
+ quote_id = str(((focal.get("quote") or {}).get("id")) or "")
217
+ if quote_id and "post_id" not in patch:
218
+ qpost = lookup_our_post(quote_id)
219
+ if qpost:
220
+ patch["post_id"] = qpost["id"]
221
+ if not row.get("project_name") and not patch.get("project_name") \
222
+ and qpost.get("project_name"):
223
+ patch["project_name"] = qpost["project_name"]
224
+ elif "parent_reply_id" not in patch:
225
+ qours = lookup_our_posted_reply(quote_id)
226
+ if qours and qours["id"] != rid:
227
+ patch["parent_reply_id"] = qours["id"]
228
+ patch["depth"] = (qours.get("depth") or 1) + 1
229
+ if qours.get("post_id"):
230
+ patch["post_id"] = qours["post_id"]
231
+
232
+ if terminal == "root":
233
+ root_handle = (chain[-1][1] if chain else (focal.get("author") or {}).get("screen_name") or "")
234
+ root_handle = (root_handle or "").lstrip("@")
235
+ if chain and root_handle and not row.get("thread_author_handle"):
236
+ patch["thread_author_handle"] = root_handle
237
+
238
+ if not patch:
239
+ if terminal == "transient":
240
+ return "transient" # incomplete walk; retry next run
241
+ # Chain walked to root / cut by a deleted ancestor / not a reply at
242
+ # all, and nothing of ours anywhere in it: remember permanently.
243
+ state[tid] = "no_link" if chain else "not_a_reply"
244
+ return state[tid]
245
+
246
+ if dry_run:
247
+ print(f" [DRY] reply {rid}: {json.dumps(patch)}")
248
+ return "would_patch"
249
+
250
+ resp = api_patch(f"/api/v1/replies/{rid}", patch)
251
+ if resp.get("error"):
252
+ print(f" ERROR patching reply {rid}: {resp['error']}", file=sys.stderr)
253
+ return "patch_error"
254
+ state[tid] = "linked"
255
+ kinds = "+".join(k for k in ("post_id", "parent_reply_id", "thread_author_handle") if k in patch)
256
+ print(f" linked reply {rid}: {kinds} {json.dumps(patch)}")
257
+ return "linked"
258
+
259
+
260
+ def main():
261
+ ap = argparse.ArgumentParser(description="Backfill parent-thread linkage on X replies")
262
+ ap.add_argument("--limit", type=int, default=20, help="rows per run (recurring lane)")
263
+ ap.add_argument("--backfill", action="store_true", help="keep paging until no work is left")
264
+ ap.add_argument("--ids", nargs="*", type=int, help="enrich specific reply ids")
265
+ ap.add_argument("--our-account", default=None, help="scope to one posting handle")
266
+ ap.add_argument("--all-accounts", action="store_true", help="do not scope by handle")
267
+ ap.add_argument("--sleep", type=float, default=0.5, help="seconds between fxtwitter calls")
268
+ ap.add_argument("--dry-run", action="store_true")
269
+ args = ap.parse_args()
270
+
271
+ our_account = args.our_account
272
+ if not our_account and not args.all_accounts and not args.ids:
273
+ try:
274
+ from account_resolver import resolve as _resolve_account
275
+ our_account = _resolve_account("twitter")
276
+ except Exception:
277
+ our_account = None
278
+ if not our_account:
279
+ print("No twitter account resolvable; pass --our-account or --all-accounts", file=sys.stderr)
280
+ sys.exit(1)
281
+
282
+ state = load_state()
283
+ totals = {}
284
+ seen_ids = set()
285
+ before_id = None
286
+ while True:
287
+ rows = fetch_work(
288
+ args.limit if not args.backfill else 200, our_account, args.ids, before_id
289
+ )
290
+ rows = [r for r in rows if r["id"] not in seen_ids]
291
+ if not rows:
292
+ break
293
+ for row in rows:
294
+ seen_ids.add(row["id"])
295
+ # Terminal misses (deleted tweet, foreign thread) never leave the
296
+ # missing_parent queue; the id cursor pages past them.
297
+ before_id = row["id"] if before_id is None else min(before_id, row["id"])
298
+ outcome = enrich_row(row, state, args.sleep, args.dry_run)
299
+ totals[outcome] = totals.get(outcome, 0) + 1
300
+ if len(seen_ids) % 25 == 0 and not args.dry_run:
301
+ save_state(state)
302
+ if not args.dry_run:
303
+ save_state(state)
304
+ if args.ids or not args.backfill:
305
+ break
306
+
307
+ print(f"[enrich_reply_parents] processed={len(seen_ids)} " +
308
+ " ".join(f"{k}={v}" for k, v in sorted(totals.items())))
309
+
310
+
311
+ if __name__ == "__main__":
312
+ main()
@@ -82,6 +82,13 @@ MAX_EVENTS_PER_RUN = 200
82
82
  # degraded (a bad search topic, a draft-quality regression), not that any
83
83
  # single draft was wrong in a way the digest model could articulate.
84
84
  BULK_NO_REASON_THRESHOLD = int(os.environ.get("S4L_BULK_NO_REASON_THRESHOLD", "3"))
85
+ # Two-draft cards ship raw per-box pointer-dwell ms in draft_choice; this is
86
+ # the read-vs-skim floor for the draft the reviewer did NOT pick. At or above
87
+ # it, keeping the preselected Draft A counts as an informed keep (they read B
88
+ # and stayed); below it the approval says nothing about B. Threshold lives
89
+ # HERE, not in the menubar client, so it can be tuned without a client
90
+ # release.
91
+ DRAFT_READ_MS = int(os.environ.get("S4L_DRAFT_READ_MS", "1000"))
85
92
 
86
93
  DISALLOWED_TOOLS = (
87
94
  "ScheduleWakeup,CronCreate,CronDelete,CronList,EnterPlanMode,EnterWorktree,"
@@ -106,10 +113,54 @@ def load_config():
106
113
  return {"projects": []}
107
114
 
108
115
 
116
+ def _draft_choice(e: dict) -> dict | None:
117
+ """Parsed draft_choice payload (two-draft cards only). The API returns
118
+ jsonb as a dict; a locally-buffered event may still carry it as a JSON
119
+ string. None when absent, unparseable, or missing the unchosen draft
120
+ (nothing pairwise to say without the loser)."""
121
+ dc = e.get("draft_choice")
122
+ if isinstance(dc, str):
123
+ try:
124
+ dc = json.loads(dc)
125
+ except Exception:
126
+ return None
127
+ if not isinstance(dc, dict) or not (dc.get("unchosen_text") or "").strip():
128
+ return None
129
+ return dc
130
+
131
+
109
132
  def _event_line(e: dict) -> str:
110
133
  """One compact evidence line per event for the prompt."""
111
134
  parts = [f"[{e.get('decision')}{'+loved' if e.get('loved') else ''}]"]
112
135
  note = (e.get("reject_note") or "").strip()
136
+ # Two-draft pairwise flags (approvals only: on a reject BOTH drafts died,
137
+ # so which box the caret sat in carries no preference). Weighting ladder,
138
+ # explained to the model in build_prompt: an active switch to Draft B is
139
+ # strong (they necessarily read both); keeping the preselected A counts
140
+ # only when hover dwell shows they actually read B; a fast approve with B
141
+ # unread is flagged as exactly that so no preference gets fabricated.
142
+ dc = _draft_choice(e) if e.get("decision") == "approved" else None
143
+ show_unchosen = False
144
+ if dc:
145
+ if not dc.get("auto_selected"):
146
+ parts.append("picked_draft_b_over_default_a")
147
+ show_unchosen = True
148
+ elif dc.get("visited_other"):
149
+ # Clicked into B, then came back and approved A: an explicit
150
+ # head-to-head choice of the default, same strength as a switch
151
+ # (2026-07-10 user rule: only a zero-interaction approve is
152
+ # no-signal).
153
+ parts.append("chose_default_a_after_trying_b")
154
+ show_unchosen = True
155
+ else:
156
+ other_ms = dc.get("hover_b_ms") or 0
157
+ if other_ms >= DRAFT_READ_MS:
158
+ parts.append(
159
+ f"kept_default_a_after_reading_b={round(other_ms / 1000, 1)}s"
160
+ )
161
+ show_unchosen = True
162
+ else:
163
+ parts.append("second_draft_not_read")
113
164
  if e.get("reject_category"):
114
165
  parts.append(f"category={e['reject_category']}")
115
166
  elif e.get("decision") == "rejected" and not note:
@@ -147,6 +198,13 @@ def _event_line(e: dict) -> str:
147
198
  line += f"\n user REWROTE it to: {draft[:300]}"
148
199
  elif draft:
149
200
  line += f"\n our draft was: {draft[:200]}"
201
+ if dc and show_unchosen:
202
+ line += f"\n the draft they did NOT pick was: {(dc.get('unchosen_text') or '')[:300]}"
203
+ if dc.get("style") or dc.get("unchosen_style"):
204
+ line += (
205
+ f"\n styles: picked={dc.get('style') or '?'}"
206
+ f" not_picked={dc.get('unchosen_style') or '?'}"
207
+ )
150
208
  url = (e.get("thread_url") or "").strip()
151
209
  if url:
152
210
  line += f"\n thread: {url}"
@@ -205,6 +263,8 @@ NEW REVIEW EVENTS since the last digest ({len(rejected)} rejected, {len(no_reaso
205
263
 
206
264
  Categories: wrong_author = the thread's author/audience was a bad fit; off_topic = the thread itself was a bad fit; bad_draft = thread was fine but the written reply was off; other = see the note. "no_reason_given" means the user rejected without picking a category or typing a note: the rejection itself is real, but WHY is your inference from the author/thread/draft context alone, so treat it as weak evidence. It can corroborate a pattern that reasoned events already show, but a no_reason_given reject never justifies a new entry or an author block on its own, and 2+ of them agreeing still only justify an entry when the shared pattern in their context is unmistakable. "edited_before_approving" with an ORIGINAL/REWROTE pair means the user hand-corrected our draft before posting: the rewrite is a direct statement of the voice they want. Diff the pair; when 2+ edits show the same correction (a phrase type removed, a structure replaced, tone shifted, length cut), distill that recurring pattern into draft_style_notes. Ignore edit content that is lead-specific or cosmetic (typo fixes, one-off facts); learn only what generalizes. "user_checked=profile_click" means the user opened the author's profile before deciding (a strong author-quality signal even without a note). "[approved+loved]" means the user picked the heart in the approve row ("this was a really good one"; approve_level_N in interactions carries the strength, 2 = best of the best): strong positive evidence for audience_prefer and thread selection, worth roughly two plain approvals.
207
265
 
266
+ Two-draft cards show a "did NOT pick" pair. "picked_draft_b_over_default_a" means the card offered two drafts with A preselected and the user deliberately clicked into B and approved it: a direct head-to-head preference for the picked draft over the shown alternative, evidence on par with a hand rewrite. "chose_default_a_after_trying_b" means they clicked into B (trying it as the selection) and then came back and approved A: equally explicit, the same head-to-head strength as a switch, just with the default as the winner. "kept_default_a_after_reading_b=Xs" means they kept the preselected A but spent Xs with the pointer over B first: an informed keep, weaker than either explicit choice (reading B does not prove they weighed it; treat like no_reason_given, corroborating a pattern that stronger events already show rather than founding one). "second_draft_not_read" means they approved the default without touching or reading the alternative: NO pairwise signal, never infer anything against the unread draft. When 2+ pairwise events agree, diff the picked texts against the not-picked ones and distill WHAT recurs (tone, structure, length, opener type, directness) into draft_style_notes; the "styles:" line names each side's engagement style, useful when the same style keeps winning or losing.
267
+
208
268
  You can also block SPECIFIC authors via the plan's block_authors list. A block is a permanent hard exclusion of that one handle from all future thread selection, so it is YOUR judgment call, never automatic. Block when the evidence is strong: a wrong_author reject IS a direct human statement about that author (especially with profile_click), and the author context (author_followers, their post, found_via_topic) or the user's note confirms the account itself was the problem rather than the topic. Do NOT block when the reject looks topic-driven (off_topic/bad_draft on a reasonable account) or when you are unsure; the generalizable TYPE entry in audience_avoid is the softer tool for that.
209
269
 
210
270
  Propose changes to the block. RULES, in priority order:
@@ -333,6 +393,22 @@ def _is_actionable(e: dict) -> bool:
333
393
  # as actionable as a reject (and it feeds edit_examples).
334
394
  if e.get("edited"):
335
395
  return True
396
+ # Any pairwise draft signal triggers the digest (2026-07-10 user rule,
397
+ # widened same evening): an explicit choice (switched to B, or tried B
398
+ # and came back to A) is style evidence on par with an edit, and even a
399
+ # hover-read of the other draft (>= DRAFT_READ_MS) is a small but real
400
+ # signal that must flow through the pipeline rather than bank forever
401
+ # behind the plain-approvals gate. The PROMPT still weights the tiers
402
+ # (explicit strong, hover-read weak/corroborating); this gate only
403
+ # decides when a Claude turn is worth burning. A zero-interaction
404
+ # approve (second_draft_not_read) remains non-actionable: no signal.
405
+ dc = _draft_choice(e)
406
+ if dc and (
407
+ not dc.get("auto_selected")
408
+ or dc.get("visited_other")
409
+ or (dc.get("hover_b_ms") or 0) >= DRAFT_READ_MS
410
+ ):
411
+ return True
336
412
  return bool((e.get("reject_note") or "").strip())
337
413
 
338
414
 
@@ -193,11 +193,23 @@ def build_prompt(platform, replies, reserved_names):
193
193
  lines.append("")
194
194
  lines.append(
195
195
  "These replies all WON the thread (top of the conversation by likes). "
196
- "Find the shared pattern that makes them work — the rhetorical move, "
196
+ "Find the shared pattern that makes them work: the rhetorical move, "
197
197
  "the structural shape, the relationship to the OP. Most winners "
198
198
  "share ONE pattern; that pattern is your new engagement style."
199
199
  )
200
200
  lines.append("")
201
+ lines.append(
202
+ "PRESERVE THE STRUCTURAL FINGERPRINT. Do not smooth the pattern "
203
+ "into a generic one-liner ('add a thoughtful counterpoint'); keep "
204
+ "what the winning replies actually DO on the page: how the first "
205
+ "words enter (lowercase noun, a number, the question itself, a "
206
+ "quoted phrase), how many sentences and of what shape (one clipped "
207
+ "line, two short + one long, a fragment), where the punch sits, "
208
+ "and what the reply refuses to do (no greeting, no hedge, no "
209
+ "summary). The char counts shown per reply are part of the "
210
+ "fingerprint; notice where the winners cluster."
211
+ )
212
+ lines.append("")
201
213
  lines.append(
202
214
  "Ignore: replies that win because of follower count, fame, or "
203
215
  "non-repeatable luck. Focus on the structural move that we (a "
@@ -209,14 +221,15 @@ def build_prompt(platform, replies, reserved_names):
209
221
  )
210
222
  lines.append("")
211
223
  for i, r in enumerate(replies, 1):
224
+ content = r.get("reply_content") or ""
212
225
  lines.append(
213
226
  f"### #{i} (likes={r['likes']}, replies={r['replies_count']}, "
214
- f"rt={r['retweets']})"
227
+ f"rt={r['retweets']}, chars={len(content.strip())})"
215
228
  )
216
229
  lines.append(f"Thread: {r['thread_url']}")
217
230
  handle = r.get("reply_author_handle") or "(unknown)"
218
231
  lines.append(f"Reply by @{handle}:")
219
- lines.append(f"> {r['reply_content']}")
232
+ lines.append(f"> {content}")
220
233
  lines.append("")
221
234
 
222
235
  lines.append("## Schema (match exactly)")
@@ -230,7 +243,12 @@ def build_prompt(platform, replies, reserved_names):
230
243
  lines.append("```")
231
244
  lines.append("{")
232
245
  lines.append(' "name": "<snake_case_name>",')
233
- lines.append(' "description": "<one to three sentences describing the style>",')
246
+ lines.append(
247
+ ' "description": "<one to three sentences carrying the structural '
248
+ "fingerprint: the DEFINING MOVE (the one thing every draft in this "
249
+ "style must contain), the OPENING (how the first words enter), and "
250
+ 'the sentence shape>",'
251
+ )
234
252
  lines.append(' "example": "<one short OP + reply pair demonstrating the style>",')
235
253
  lines.append(' "best_in": {')
236
254
  lines.append(f' "{platform}": ["<short context label>", ...],')
@@ -255,8 +273,11 @@ def build_prompt(platform, replies, reserved_names):
255
273
  "MOVE (e.g. `mirror_and_extend`, `flip_to_alt`, not `good_reply`)."
256
274
  )
257
275
  lines.append(
258
- "3. The description should make the style copyable: a future model "
259
- "reading just that one sentence should know what to write."
276
+ "3. The description must make the style copyable AS A FORM: a future "
277
+ "model reading only the description must know the defining move, how "
278
+ "to open, and what sentence shape to write. 'Add a sharp "
279
+ "counterpoint' fails this test; 'one clipped sentence, no greeting, "
280
+ "opens with the concrete number the OP left out' passes."
260
281
  )
261
282
  lines.append(
262
283
  "4. The example should be a realistic OP + reply pair, not lifted "
@@ -270,7 +291,10 @@ def build_prompt(platform, replies, reserved_names):
270
291
  lines.append(
271
292
  "6. NEVER propose a style about including a product, a URL, or a "
272
293
  "mechanism. Our link-tail layer handles that downstream. The style "
273
- "is about the text BEFORE the link."
294
+ "is about the text BEFORE the link. The note MAY state in one "
295
+ "clause how a plain product mention would enter this style if ever "
296
+ "(e.g. 'product name fits as the concrete example slot; never in "
297
+ "the opening'), but no URLs or link mechanics."
274
298
  )
275
299
  return "\n".join(lines)
276
300