@m13v/s4l 1.7.7-rc.5 → 1.7.7-rc.6

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/mcp/dist/index.js CHANGED
@@ -131,10 +131,13 @@ const STALL_WATCH_INTERVAL_SECS = 120;
131
131
  // S4L_OVERLAY_WATCH=0.
132
132
  const OVERLAY_WATCH_LABEL = "com.m13v.social-overlay-watch";
133
133
  const OVERLAY_WATCH_PLIST = path.join(os.homedir(), "Library", "LaunchAgents", `${OVERLAY_WATCH_LABEL}.plist`);
134
- // Daily self-updater. Enabled alongside autopilot so a hands-free (headless)
135
- // install keeps itself current — the interactive `runtime` tool (action:'update')
136
- // only helps when
137
- // a human-facing agent session is open, which an autopilot box never has.
134
+ // Legacy daily self-updater launchd job. BANNED (user decision 2026-07-31):
135
+ // it ran s4l_box_update.sh unattended, which kills and relaunches Claude
136
+ // Desktop with no human in the loop. Updates are user-initiated only (menubar
137
+ // "Update now & restart Claude Desktop", the panel button, `runtime`
138
+ // action:'update', or an operator running s4l_box_update.sh explicitly). The
139
+ // label/plist constants survive so ensureUpdaterRemoved() can clean the job
140
+ // off every box that got it from v1.7.1..v1.7.7-rc.4.
138
141
  const UPDATER_LABEL = "com.m13v.social-autoposter-update";
139
142
  const UPDATER_PLIST = path.join(os.homedir(), "Library", "LaunchAgents", `${UPDATER_LABEL}.plist`);
140
143
  // A sane PATH for launchd jobs (launchd starts with a bare PATH). Include the
@@ -4874,68 +4877,21 @@ async function ensureOverlayWatchInstalled() {
4874
4877
  return { ok: false, detail: e?.message || String(e) };
4875
4878
  }
4876
4879
  }
4877
- // Install/refresh the daily self-updater launchd job. This used to be bundled
4878
- // into the now-deleted `autopilot` MCP tool (removed 2026-06-19, 88bd1cb9):
4879
- // calling `autopilot enable` installed this job as a side effect, so removing
4880
- // the tool silently orphaned it — UPDATER_LABEL/UPDATER_PLIST and the
4881
- // auto_update_on status check survived (buildSnapshot still reports them), but
4882
- // nothing has installed the plist since, on ANY box provisioned after that
4883
- // commit (auto_update_on reads false forever). Restored here as its own
4884
- // deterministic boot-time job, same pattern as its five siblings above.
4885
- //
4886
- // Points at scripts/s4l_box_update.sh, NOT skill/social-autoposter-update.sh:
4887
- // the latter is the npm-lane updater (npm view + npx update) and is a silent
4888
- // no-op on a .mcpb box, which has no npm/npx on PATH (see version.ts). The
4889
- // .mcpb-lane equivalent downloads the .mcpb directly from the channel-resolved
4890
- // GitHub release and unpacks it over the extension dir — see that script's own
4891
- // header for the channel/no-downgrade/retry guards. Default mode there
4892
- // downloads + unpacks + restarts Claude Desktop with NO human in the loop,
4893
- // matching the original bundled updater's intent ("keeps a headless install
4894
- // current"); RunAtLoad so a box that boots already-behind checks promptly.
4895
- async function ensureUpdaterInstalled() {
4880
+ // Remove the legacy daily self-updater launchd job (see UPDATER_LABEL's
4881
+ // comment). Boot-time, idempotent, best-effort: unload the job if launchd has
4882
+ // it and delete the plist so the daily silent Claude Desktop restart can never
4883
+ // fire again. Do NOT reintroduce an install path for this job — any future
4884
+ // auto-update mechanism must be user-visible and user-confirmed.
4885
+ async function ensureUpdaterRemoved() {
4896
4886
  try {
4897
4887
  if (process.platform !== "darwin")
4898
- return { ok: false, detail: "not macOS" };
4899
- if ((process.env.S4L_AUTO_UPDATE) === "0")
4900
- return { ok: false, detail: "disabled (S4L_AUTO_UPDATE=0)" };
4901
- const logDir = path.join(repoDir(), "skill", "logs");
4902
- try {
4903
- fs.mkdirSync(logDir, { recursive: true });
4904
- }
4905
- catch {
4906
- /* best-effort */
4907
- }
4908
- const xml = plistXml({
4909
- label: UPDATER_LABEL,
4910
- programArgs: ["/bin/bash", path.join(repoDir(), "scripts", "s4l_box_update.sh")],
4911
- intervalSecs: 86_400,
4912
- runAtLoad: true,
4913
- stdoutLog: path.join(logDir, "launchd-self-update-stdout.log"),
4914
- stderrLog: path.join(logDir, "launchd-self-update-stderr.log"),
4915
- });
4888
+ return { ok: true, detail: "not macOS" };
4916
4889
  const uid = process.getuid ? process.getuid() : 0;
4917
- let cur = null;
4918
- try {
4919
- cur = fs.readFileSync(UPDATER_PLIST, "utf-8");
4920
- }
4921
- catch {
4922
- cur = null;
4923
- }
4924
- let detail;
4925
- if (cur === xml) {
4926
- const res = await loadPlist(UPDATER_LABEL, UPDATER_PLIST, uid);
4927
- detail = `current (load rc=${res.code})`;
4928
- }
4929
- else {
4930
- if (cur !== null) {
4931
- await unloadPlist(UPDATER_LABEL, UPDATER_PLIST, uid);
4932
- }
4933
- fs.mkdirSync(path.dirname(UPDATER_PLIST), { recursive: true });
4934
- fs.writeFileSync(UPDATER_PLIST, xml, "utf-8");
4935
- const res = await loadPlist(UPDATER_LABEL, UPDATER_PLIST, uid);
4936
- detail = cur === null ? "installed + loaded" : `rewritten + reloaded (rc=${res.code})`;
4937
- }
4938
- return { ok: true, detail };
4890
+ const existed = fs.existsSync(UPDATER_PLIST);
4891
+ await unloadPlist(UPDATER_LABEL, UPDATER_PLIST, uid);
4892
+ if (existed)
4893
+ fs.rmSync(UPDATER_PLIST, { force: true });
4894
+ return { ok: true, detail: existed ? "removed" : "absent" };
4939
4895
  }
4940
4896
  catch (e) {
4941
4897
  return { ok: false, detail: e?.message || String(e) };
@@ -6134,13 +6090,12 @@ async function main() {
6134
6090
  void ensureOverlayWatchInstalled()
6135
6091
  .then((r) => console.error(`[overlay-watch] launchd supervisor: ${r.ok ? "ok" : "skip"} (${r.detail})`))
6136
6092
  .catch((e) => console.error("[overlay-watch] supervisor install failed:", e?.message || e));
6137
- // Daily self-updater: restored 2026-07-08 after 88bd1cb9 ("Remove autopilot
6138
- // tool") silently dropped its only install path (it used to be bundled into
6139
- // the deleted `autopilot enable` action). Best-effort; never blocks boot.
6140
- // Disable with S4L_AUTO_UPDATE=0.
6141
- void ensureUpdaterInstalled()
6142
- .then((r) => console.error(`[self-update] launchd updater: ${r.ok ? "ok" : "skip"} (${r.detail})`))
6143
- .catch((e) => console.error("[self-update] updater install failed:", e?.message || e));
6093
+ // Silent daily self-updates are banned (user decision 2026-07-31): the job
6094
+ // restarted Claude Desktop unattended. Clean it off every box that got it
6095
+ // from v1.7.1..v1.7.7-rc.4. Best-effort; never blocks boot.
6096
+ void ensureUpdaterRemoved()
6097
+ .then((r) => console.error(`[self-update] legacy launchd updater: ${r.ok ? "ok" : "skip"} (${r.detail})`))
6098
+ .catch((e) => console.error("[self-update] updater removal failed:", e?.message || e));
6144
6099
  // Heal installs onboarded before short_links_live defaulted to false: such a
6145
6100
  // project wraps short links against the customer's own domain, which has no
6146
6101
  // /r/[code] resolver, so every minted link 404s. Re-point them at the s4l.ai
@@ -1,4 +1,4 @@
1
1
  {
2
- "version": "1.7.7-rc.5",
3
- "installedAt": "2026-07-31T22:56:07.561Z"
2
+ "version": "1.7.7-rc.6",
3
+ "installedAt": "2026-07-31T23:52:05.555Z"
4
4
  }
package/mcp/manifest.json CHANGED
@@ -2,7 +2,7 @@
2
2
  "dxt_version": "0.1",
3
3
  "name": "social-autoposter",
4
4
  "display_name": "S4L",
5
- "version": "1.7.7-rc.5",
5
+ "version": "1.7.7-rc.6",
6
6
  "description": "Draft, review, approve, and autopilot X/Twitter posts.",
7
7
  "long_description": "## **⚠️ The disclaimer above is generic Claude boilerplate.** Anthropic shows the same warning on every plugin regardless of what it does; any plugin has the same level of access as any app you download from the internet.\n\nS4L is an open source product developed by Mediar.ai Incorporated, a VC-backed San Francisco-based startup.\n\nTo get started:\n\n1\\. Copy this prompt: **Set me up on S4L plugin end to end**\n\n2\\. Quit with CMD+Q, reopen Claude, paste into a new chat.\n\nWhat happens next:\n\n* About every 5 minutes S4L scans X for posts that match your topics and drafts replies in your voice.\n* Drafts show up as review cards, usually the first within a few minutes. Nothing is posted automatically; you approve each one.\n* Posting autopilot stays off until you explicitly turn it on.",
8
8
  "author": {
package/mcp/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@m13v/s4l-mcp",
3
- "version": "1.7.7-rc.5",
3
+ "version": "1.7.7-rc.6",
4
4
  "private": true,
5
5
  "description": "Desktop MCP client for social-autoposter (X/Twitter rail): manual draft/review/approve loop, autopilot control, and stats. Thin wrapper over the existing pipeline scripts.",
6
6
  "license": "MIT",
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@m13v/s4l",
3
- "version": "1.7.7-rc.5",
3
+ "version": "1.7.7-rc.6",
4
4
  "description": "Automated social posting pipeline for Reddit, X/Twitter, LinkedIn, and Moltbook. Install as a Claude Code agent skill.",
5
5
  "bin": {
6
6
  "social-autoposter": "bin/cli.js",
@@ -76,6 +76,12 @@ TAG_TO_TYPE = {
76
76
  # Python and inlined; no Bash tools), mirroring the twitter Phase 2b
77
77
  # tool-free conversion, so it rides the same universal worker.
78
78
  "post-reddit-draft": "reddit-draft",
79
+ # Context mining (2026-07-31): scripts/context_mining.py inlines compacted
80
+ # transcripts + the corpus into one pure text->JSON turn. Execution
81
+ # guidance lives at the top of the prompt itself (not only in
82
+ # TYPE_TO_WORKER_NOTES) because an installed-package worker claims with
83
+ # ITS copy of this map, which may predate this type.
84
+ "context-mining": "context-mining",
79
85
  }
80
86
 
81
87
  # queue type -> (activity state, label) the menu bar shows while the job is in
@@ -92,6 +98,7 @@ TYPE_TO_ACTIVITY = {
92
98
  # must stay concise (user rule 2026-07-14); the platform shows on the
93
99
  # review card itself, not in the tray.
94
100
  "reddit-draft": ("drafting", "draft"),
101
+ "context-mining": ("learning", "mining context"),
95
102
  }
96
103
 
97
104
  # queue type -> execution notes PREPENDED to the prompt sidecar at claim time.
@@ -140,6 +147,13 @@ TYPE_TO_WORKER_NOTES = {
140
147
  "absent from posts (add a rejects entry only for the structural "
141
148
  "false-positive cases the prompt describes)."
142
149
  ),
150
+ "context-mining": (
151
+ "WORKER EXECUTION NOTES (queue metadata; follow while executing the "
152
+ "prompt below): single-turn, tool-free job. Everything you need is "
153
+ "inlined in the prompt (transcripts, corpus, prior decisions). Never "
154
+ "use tools. Read, then submit ONE result object matching the schema: "
155
+ "{\"proposals\": [...]}. An empty proposals list is a valid answer."
156
+ ),
143
157
  }
144
158
 
145
159
 
@@ -0,0 +1,638 @@
1
+ #!/usr/bin/env python3
2
+ """
3
+ context_mining.py -- PROTOTYPE: mine the user's own Claude conversation
4
+ transcripts for insights worth adding to a personal context corpus, using the
5
+ existing claude_job.py file queue as the LLM lane (design agreed 2026-07-31).
6
+
7
+ Flow (test-run shape; toggle/watermark/cards come later):
8
+ gather -> deterministically collect recent human-interactive sessions,
9
+ compact each FULL transcript (user prose kept, tool plumbing
10
+ dropped), inline the current corpus (numbered lines) plus the
11
+ already-considered ledger, enqueue ONE pure text->JSON job via
12
+ run_claude.sh (tag: context-mining), block for the result, and
13
+ write proposals to pending.json.
14
+ review -> print pending proposals.
15
+ approve -> apply a proposal to corpus.md (append or replace a line) and
16
+ record it in the considered ledger.
17
+ skip -> record the proposal as skipped so it is never re-pitched.
18
+
19
+ State lives under <state_dir>/context-mining/ where state_dir is
20
+ $S4L_STATE_DIR or ~/.social-autoposter-mcp (same convention as the queue):
21
+ corpus.md the corpus; every non-blank line is one numbered memory
22
+ pending.json proposals awaiting a human decision
23
+ considered.jsonl append-only ledger of every approve/skip decision
24
+
25
+ Sources walked (same two stores as extract_user_messages_today.py):
26
+ ~/.claude/projects/*/*.jsonl Claude Code (CLI + desktop)
27
+ ~/Library/Application Support/Claude/
28
+ local-agent-mode-sessions/**/.claude/projects/**/*.jsonl Cowork
29
+
30
+ Eligibility: a session must contain >= MIN_HUMAN_MSGS user messages classified
31
+ HUMAN (typed prose, not command echoes / task notifications / reminders), and
32
+ its cwd must not be an S4L worker sandbox. S4L's own automation transcripts are
33
+ never mined -- that would be a feedback loop.
34
+
35
+ Usage:
36
+ python3 scripts/context_mining.py gather --days 3
37
+ python3 scripts/context_mining.py gather --days 3 --dry-run # build, no enqueue
38
+ python3 scripts/context_mining.py review
39
+ python3 scripts/context_mining.py approve cm-1a2b3c4d [id...]
40
+ python3 scripts/context_mining.py skip cm-1a2b3c4d [id...]
41
+ """
42
+
43
+ from __future__ import annotations
44
+
45
+ import argparse
46
+ import hashlib
47
+ import json
48
+ import os
49
+ import re
50
+ import subprocess
51
+ import sys
52
+ import time
53
+ from datetime import datetime, timedelta, timezone
54
+ from pathlib import Path
55
+
56
+ HOME = Path.home()
57
+ REPO_DIR = Path(__file__).resolve().parent.parent
58
+
59
+ CLAUDE_CODE_PROJECTS_ROOT = HOME / ".claude" / "projects"
60
+ COWORK_SESSIONS_ROOT = (
61
+ HOME / "Library" / "Application Support" / "Claude" / "local-agent-mode-sessions"
62
+ )
63
+
64
+
65
+ def state_dir() -> Path:
66
+ base = os.environ.get("S4L_STATE_DIR") or str(HOME / ".social-autoposter-mcp")
67
+ d = Path(base) / "context-mining"
68
+ d.mkdir(parents=True, exist_ok=True)
69
+ return d
70
+
71
+
72
+ def corpus_path() -> Path:
73
+ return state_dir() / "corpus.md"
74
+
75
+
76
+ def pending_path() -> Path:
77
+ return state_dir() / "pending.json"
78
+
79
+
80
+ def ledger_path() -> Path:
81
+ return state_dir() / "considered.jsonl"
82
+
83
+
84
+ # ---------------------------------------------------------------- gathering
85
+
86
+ MIN_HUMAN_MSGS = 2
87
+ PER_USER_MSG_CAP = 2500
88
+ PER_ASSISTANT_MSG_CAP = 700
89
+ PER_SESSION_CAP = 14_000
90
+ DEFAULT_TOTAL_BUDGET = 80_000
91
+
92
+ # cwd substrings that mark a session as S4L's own automation, never mined
93
+ EXCLUDED_CWD_MARKERS = (".s4l-worker", ".saps-worker")
94
+
95
+
96
+ def _classify(text: str) -> str:
97
+ """HUMAN vs harness-injected user-role messages (same taxonomy as
98
+ extract_user_messages_today.py)."""
99
+ stripped = text.lstrip()
100
+ if stripped.startswith("<task-notification>"):
101
+ return "TASK_NOTIF"
102
+ if stripped.startswith("<command-name>") or stripped.startswith("<command-message>"):
103
+ return "COMMAND"
104
+ if stripped.startswith("<local-command-stdout>") or stripped.startswith(
105
+ "<local-command-stderr>"
106
+ ):
107
+ return "CMD_STDOUT"
108
+ if "<<autonomous-loop" in stripped or stripped.startswith("<loop-"):
109
+ return "SCHED_WAKE"
110
+ if stripped.startswith("<user-prompt-submit-hook>"):
111
+ return "HOOK"
112
+ if stripped.startswith("<system-reminder>"):
113
+ without = re.sub(
114
+ r"<system-reminder>.*?</system-reminder>", "", stripped, flags=re.DOTALL
115
+ ).strip()
116
+ if not without:
117
+ return "SYS_REMIND"
118
+ return "HUMAN"
119
+
120
+
121
+ def _strip_reminders(text: str) -> str:
122
+ return re.sub(r"<system-reminder>.*?</system-reminder>", "", text, flags=re.DOTALL).strip()
123
+
124
+
125
+ def _text_blocks(content) -> str | None:
126
+ """Textual payload of a message; None for tool_result plumbing."""
127
+ if isinstance(content, str):
128
+ return content
129
+ if isinstance(content, list):
130
+ texts = []
131
+ for block in content:
132
+ if not isinstance(block, dict):
133
+ continue
134
+ btype = block.get("type")
135
+ if btype == "tool_result":
136
+ return None
137
+ if btype == "text" and isinstance(block.get("text"), str):
138
+ texts.append(block["text"])
139
+ if texts:
140
+ return "\n".join(texts)
141
+ return None
142
+
143
+
144
+ def _iter_transcript_files():
145
+ if CLAUDE_CODE_PROJECTS_ROOT.is_dir():
146
+ for p in CLAUDE_CODE_PROJECTS_ROOT.glob("*/*.jsonl"):
147
+ yield "claude_code", p
148
+ if COWORK_SESSIONS_ROOT.is_dir():
149
+ for p in COWORK_SESSIONS_ROOT.glob("**/.claude/projects/**/*.jsonl"):
150
+ yield "cowork", p
151
+
152
+
153
+ def _compact_session(path: Path) -> dict | None:
154
+ """Read one full transcript and compact it to the conversation arc:
155
+ HUMAN user messages (near-full) + assistant prose (trimmed). Returns None
156
+ if the session is not human-interactive or is S4L automation."""
157
+ cwd = None
158
+ first_ts = last_ts = None
159
+ human_count = 0
160
+ turns: list[str] = []
161
+ try:
162
+ with path.open("r", encoding="utf-8", errors="replace") as f:
163
+ for line in f:
164
+ try:
165
+ d = json.loads(line)
166
+ except Exception:
167
+ continue
168
+ ts = d.get("timestamp")
169
+ if ts:
170
+ first_ts = first_ts or ts
171
+ last_ts = ts
172
+ if cwd is None and d.get("cwd"):
173
+ cwd = d["cwd"]
174
+ if d.get("isSidechain"):
175
+ continue
176
+ msg = d.get("message") or {}
177
+ role = msg.get("role")
178
+ if d.get("type") == "user" and role == "user":
179
+ text = _text_blocks(msg.get("content"))
180
+ if text is None:
181
+ continue
182
+ text = text.strip()
183
+ if not text or _classify(text) != "HUMAN":
184
+ continue
185
+ text = _strip_reminders(text)
186
+ if not text:
187
+ continue
188
+ human_count += 1
189
+ turns.append("USER: " + text[:PER_USER_MSG_CAP])
190
+ elif d.get("type") == "assistant" and role == "assistant":
191
+ text = _text_blocks(msg.get("content"))
192
+ if not text:
193
+ continue
194
+ text = text.strip()
195
+ if len(text) < 40: # skip one-liner tool narration
196
+ continue
197
+ turns.append("ASSISTANT: " + text[:PER_ASSISTANT_MSG_CAP])
198
+ except OSError:
199
+ return None
200
+
201
+ if human_count < MIN_HUMAN_MSGS:
202
+ return None
203
+ if cwd and any(m in cwd for m in EXCLUDED_CWD_MARKERS):
204
+ return None
205
+
206
+ body = "\n\n".join(turns)
207
+ if len(body) > PER_SESSION_CAP:
208
+ # keep the head and the tail; the middle is elided
209
+ half = PER_SESSION_CAP // 2
210
+ body = body[:half] + "\n\n[... middle of session elided ...]\n\n" + body[-half:]
211
+ return {
212
+ "session_id": path.stem,
213
+ "path": str(path),
214
+ "cwd": cwd or "?",
215
+ "first_ts": first_ts or "?",
216
+ "last_ts": last_ts or "?",
217
+ "human_msgs": human_count,
218
+ "body": body,
219
+ }
220
+
221
+
222
+ def gather_sessions(days: int) -> list[dict]:
223
+ """ALL eligible human sessions within the window, most recent first."""
224
+ cutoff = time.time() - days * 86400
225
+ candidates = []
226
+ for _source, p in _iter_transcript_files():
227
+ try:
228
+ mtime = p.stat().st_mtime
229
+ except OSError:
230
+ continue
231
+ if mtime < cutoff:
232
+ continue
233
+ candidates.append((mtime, p))
234
+ candidates.sort(reverse=True)
235
+
236
+ kept: list[dict] = []
237
+ seen_ids: set[str] = set()
238
+ for _mtime, p in candidates:
239
+ if p.stem in seen_ids:
240
+ continue
241
+ sess = _compact_session(p)
242
+ if sess is None:
243
+ continue
244
+ seen_ids.add(p.stem)
245
+ kept.append(sess)
246
+ return kept
247
+
248
+
249
+ def chunk_by_budget(sessions: list[dict], budget: int) -> list[list[dict]]:
250
+ """Split into batches whose combined body size fits the budget. A batch
251
+ always takes at least one session, so oversized sessions still ship."""
252
+ batches: list[list[dict]] = []
253
+ cur: list[dict] = []
254
+ used = 0
255
+ for s in sessions:
256
+ cost = len(s["body"])
257
+ if cur and used + cost > budget:
258
+ batches.append(cur)
259
+ cur, used = [], 0
260
+ cur.append(s)
261
+ used += cost
262
+ if cur:
263
+ batches.append(cur)
264
+ return batches
265
+
266
+
267
+ # ------------------------------------------------------------------- corpus
268
+
269
+
270
+ def read_corpus_lines() -> list[str]:
271
+ if not corpus_path().exists():
272
+ return []
273
+ return [ln.rstrip("\n") for ln in corpus_path().read_text().splitlines() if ln.strip()]
274
+
275
+
276
+ def write_corpus_lines(lines: list[str]) -> None:
277
+ corpus_path().write_text("\n".join(lines) + ("\n" if lines else ""))
278
+
279
+
280
+ def read_ledger() -> list[dict]:
281
+ if not ledger_path().exists():
282
+ return []
283
+ out = []
284
+ for ln in ledger_path().read_text().splitlines():
285
+ try:
286
+ out.append(json.loads(ln))
287
+ except Exception:
288
+ continue
289
+ return out
290
+
291
+
292
+ def append_ledger(rec: dict) -> None:
293
+ with ledger_path().open("a") as f:
294
+ f.write(json.dumps(rec, ensure_ascii=False) + "\n")
295
+
296
+
297
+ # ------------------------------------------------------------------- prompt
298
+
299
+ SCHEMA = {
300
+ "type": "object",
301
+ "required": ["proposals"],
302
+ "properties": {
303
+ "proposals": {
304
+ "type": "array",
305
+ "items": {
306
+ "type": "object",
307
+ "required": ["action", "text", "why", "source_session", "source_date"],
308
+ "properties": {
309
+ "action": {"type": "string", "enum": ["add", "revise"]},
310
+ "revises_line": {"type": ["integer", "null"]},
311
+ "text": {"type": "string"},
312
+ "why": {"type": "string"},
313
+ "quote": {"type": "string"},
314
+ "source_session": {"type": "string"},
315
+ "source_date": {"type": "string"},
316
+ },
317
+ },
318
+ }
319
+ },
320
+ }
321
+
322
+
323
+ def build_prompt(sessions: list[dict], pending_props: list[dict] | None = None) -> str:
324
+ corpus_lines = read_corpus_lines()
325
+ corpus_block = (
326
+ "\n".join(f"[{i + 1}] {ln}" for i, ln in enumerate(corpus_lines))
327
+ if corpus_lines
328
+ else "(the corpus is currently empty)"
329
+ )
330
+ considered = [
331
+ {"status": r.get("status"), "text": r.get("text", "")} for r in read_ledger()[-60:]
332
+ ] + [{"status": "pending", "text": p.get("text", "")} for p in (pending_props or [])]
333
+ considered_block = (
334
+ "\n".join(f"- ({r['status']}) {r['text'][:200]}" for r in considered)
335
+ if considered
336
+ else "(nothing has been considered yet)"
337
+ )
338
+ transcript_parts = []
339
+ for s in sessions:
340
+ transcript_parts.append(
341
+ f"### Session {s['session_id'][:8]} | {s['first_ts'][:10]} | cwd: {s['cwd']}\n\n{s['body']}"
342
+ )
343
+ transcripts_block = "\n\n---\n\n".join(transcript_parts)
344
+
345
+ return f"""EXECUTION NOTES: this is a single-turn, tool-free job. Read everything below, then submit ONE result object matching the provided JSON schema. Do not use any tools. Do not fetch anything.
346
+
347
+ # Role
348
+
349
+ You are the daily context miner for a personal "context corpus": a numbered list of durable, one-line memories distilled from the user's own Claude conversations. Each line is an insight worth keeping and potentially worth sharing with the world: a specific experience, a hard-won conclusion, a non-obvious observation, or a concrete data point.
350
+
351
+ The user is Matthew, a solo founder building S4L (a social-media autoposting agent), Fazm, Mediar, and related products, and doing everything else through Claude sessions: fundraising, debugging, ops, legal, marketing.
352
+
353
+ # What qualifies
354
+
355
+ - GENERAL: the line states a transferable lesson a stranger in tech could apply to
356
+ their own work, without knowing this codebase or these products. The specific
357
+ incident is EVIDENCE (it goes in the quote and the why), not the line itself.
358
+ Test: would a founder who has never heard of S4L or Fazm reshare this?
359
+ - Durable: still true and useful months from now, not transient state ("the deploy is broken today").
360
+ - Non-obvious: something a smart peer would not already assume. Generic best practices do NOT qualify.
361
+ - First-hand: grounded in what actually happened in these sessions (an experiment, an outage, a metric, a decision and its reasoning, a surprising vendor/platform behavior).
362
+ - SYNTHESIZED: if several incidents (even across sessions) point at one underlying
363
+ lesson, propose ONE line for the lesson, never one line per incident.
364
+
365
+ # What never qualifies
366
+
367
+ - Vendor-specific micro-gotchas (a field length limit, one API's quirky error) unless
368
+ the line is elevated to the general pattern the gotcha exemplifies.
369
+ - Anything that only matters inside this codebase or product internals.
370
+ - Credentials, API keys, tokens, passwords, account numbers, or anything that looks like one, even partially. If a great insight touches one, redact the secret.
371
+ - Names, handles, or identifying details of customers, users, or other private
372
+ individuals. Describe them generically ("a stalled user", "an early customer").
373
+ - Private details about clients, other people's finances, legal disputes, or immigration status.
374
+ - Injected harness noise: usage-limit banners, system reminders, scheduler chatter. These appear inside transcripts and are not the user's words.
375
+ - Anything already covered by an existing corpus line (see below), unless you are proposing a REVISION of that line.
376
+
377
+ # Current corpus (numbered)
378
+
379
+ {corpus_block}
380
+
381
+ # Already considered (do NOT re-propose anything equivalent, including skipped items)
382
+
383
+ {considered_block}
384
+
385
+ # Transcripts to mine ({len(sessions)} sessions, most recent first)
386
+
387
+ {transcripts_block}
388
+
389
+ # Your task
390
+
391
+ Propose 0-3 corpus changes, and only what clears EVERY bar above; most days the
392
+ right answer is 0 or 1. An empty proposals list is a valid, common answer. For each proposal:
393
+ - action: "add" for a new line, or "revise" when a new learning supersedes an existing corpus line (set revises_line to that line's number).
394
+ - text: the corpus line itself. One sentence, self-contained, max ~300 chars, written in the user's plainspoken voice. No hashtags, no em dashes.
395
+ - why: one sentence on why this is worth keeping (what makes it non-obvious or shareable).
396
+ - quote: a short supporting excerpt from the transcript (redact any secrets).
397
+ - source_session: the 8-char session id from the transcript header.
398
+ - source_date: the session date (YYYY-MM-DD).
399
+
400
+ Submit the result object now."""
401
+
402
+
403
+ # -------------------------------------------------------------------- queue
404
+
405
+
406
+ def _run_one_batch(sessions: list[dict], pending_props: list[dict], ns) -> list[dict] | None:
407
+ """Enqueue one mining job for this batch; return stamped proposals, or None
408
+ on failure (caller stops the loop and keeps what it has)."""
409
+ prompt = build_prompt(sessions, pending_props)
410
+ schema_file = state_dir() / "schema.json"
411
+ schema_file.write_text(json.dumps(SCHEMA))
412
+
413
+ env = dict(os.environ)
414
+ env["S4L_REPO_DIR"] = str(REPO_DIR) # route through THIS repo's claude_job.py
415
+ cmd = [
416
+ "bash",
417
+ str(REPO_DIR / "scripts" / "run_claude.sh"),
418
+ "context-mining",
419
+ "--json-schema",
420
+ str(schema_file),
421
+ "-p",
422
+ ]
423
+ try:
424
+ proc = subprocess.run(
425
+ cmd,
426
+ input=prompt,
427
+ capture_output=True,
428
+ text=True,
429
+ env=env,
430
+ timeout=ns.timeout + 120,
431
+ )
432
+ except subprocess.TimeoutExpired:
433
+ print("[gather] hard timeout waiting for the queue result", file=sys.stderr)
434
+ return None
435
+
436
+ if proc.returncode == 79:
437
+ print(
438
+ "[gather] queue timed out (rc=79): no worker claimed the job. Is Claude "
439
+ "Desktop open and the s4l-worker task firing?",
440
+ file=sys.stderr,
441
+ )
442
+ return None
443
+ if proc.returncode != 0:
444
+ print(f"[gather] provider failed rc={proc.returncode}", file=sys.stderr)
445
+ sys.stderr.write((proc.stderr or "")[-2000:] + "\n")
446
+ return None
447
+
448
+ try:
449
+ envelope = json.loads(proc.stdout[proc.stdout.index("{"):])
450
+ obj = envelope.get("structured_output")
451
+ if obj is None:
452
+ obj = json.loads(envelope.get("result") or "{}")
453
+ except Exception as e:
454
+ print(f"[gather] could not parse result envelope: {e}", file=sys.stderr)
455
+ sys.stderr.write((proc.stdout or "")[-2000:] + "\n")
456
+ return None
457
+
458
+ stamped = []
459
+ for p in obj.get("proposals") or []:
460
+ pid = "cm-" + hashlib.sha1(
461
+ (p.get("text", "") + p.get("source_session", "")).encode()
462
+ ).hexdigest()[:8]
463
+ p["id"] = pid
464
+ p["mined_at"] = datetime.now(timezone.utc).isoformat(timespec="seconds")
465
+ stamped.append(p)
466
+ return stamped
467
+
468
+
469
+ def run_gather(ns) -> int:
470
+ sessions = gather_sessions(ns.days)
471
+ if not sessions:
472
+ print(json.dumps({"ok": False, "reason": "no_eligible_sessions", "days": ns.days}))
473
+ return 1
474
+ batches = chunk_by_budget(sessions, ns.budget)
475
+ total_chars = sum(len(s["body"]) for s in sessions)
476
+ print(
477
+ f"[gather] {len(sessions)} sessions in window ({ns.days}d), "
478
+ f"{total_chars:,} chars -> {len(batches)} batches (budget {ns.budget:,})",
479
+ file=sys.stderr,
480
+ )
481
+
482
+ if ns.dry_run:
483
+ out = state_dir() / "last_prompt.txt"
484
+ out.write_text(build_prompt(batches[0], []))
485
+ for i, b in enumerate(batches, 1):
486
+ ids = ", ".join(s["session_id"][:8] for s in b)
487
+ print(f"[gather] batch {i}/{len(batches)}: {len(b)} sessions ({ids})", file=sys.stderr)
488
+ print(f"[gather] dry-run: batch-1 prompt written to {out}, nothing enqueued", file=sys.stderr)
489
+ return 0
490
+
491
+ # carry over any still-pending proposals so batches dedup against them too
492
+ all_props: list[dict] = _load_pending()
493
+ known_ids = {p["id"] for p in all_props}
494
+ failed = 0
495
+ for i, batch in enumerate(batches, 1):
496
+ ids = ", ".join(s["session_id"][:8] for s in batch)
497
+ print(
498
+ f"[gather] batch {i}/{len(batches)}: {len(batch)} sessions "
499
+ f"({sum(len(s['body']) for s in batch):,} chars) [{ids}]",
500
+ file=sys.stderr,
501
+ )
502
+ stamped = _run_one_batch(batch, all_props, ns)
503
+ if stamped is None:
504
+ failed += 1
505
+ print(f"[gather] batch {i} failed; stopping loop, keeping prior results", file=sys.stderr)
506
+ break
507
+ fresh = [p for p in stamped if p["id"] not in known_ids]
508
+ known_ids.update(p["id"] for p in fresh)
509
+ all_props.extend(fresh)
510
+ pending_path().write_text(
511
+ json.dumps({"proposals": all_props}, indent=2, ensure_ascii=False)
512
+ )
513
+ print(f"[gather] batch {i} done: +{len(fresh)} proposals ({len(all_props)} total)", file=sys.stderr)
514
+
515
+ print(
516
+ f"[gather] run complete: {len(batches) - failed}/{len(batches)} batches, "
517
+ f"{len(all_props)} pending proposals -> {pending_path()}",
518
+ file=sys.stderr,
519
+ )
520
+ cmd_review(None)
521
+ return 0 if not failed else 1
522
+
523
+
524
+ # ------------------------------------------------------------------- review
525
+
526
+
527
+ def _load_pending() -> list[dict]:
528
+ if not pending_path().exists():
529
+ return []
530
+ try:
531
+ return json.loads(pending_path().read_text()).get("proposals", [])
532
+ except Exception:
533
+ return []
534
+
535
+
536
+ def cmd_review(_ns) -> int:
537
+ props = _load_pending()
538
+ if not props:
539
+ print("no pending proposals")
540
+ return 0
541
+ corpus_lines = read_corpus_lines()
542
+ for p in props:
543
+ print("=" * 72)
544
+ head = p["id"]
545
+ if p.get("action") == "revise" and p.get("revises_line"):
546
+ n = p["revises_line"]
547
+ old = corpus_lines[n - 1] if 0 < n <= len(corpus_lines) else "(missing line)"
548
+ head += f" REVISES [{n}]: {old[:120]}"
549
+ else:
550
+ head += " ADD"
551
+ print(head)
552
+ print(f" text : {p.get('text', '')}")
553
+ print(f" why : {p.get('why', '')}")
554
+ if p.get("quote"):
555
+ print(f" quote: {p['quote'][:220]}")
556
+ print(f" src : {p.get('source_session', '?')} @ {p.get('source_date', '?')}")
557
+ print("=" * 72)
558
+ print(f"{len(props)} pending. approve/skip with: context_mining.py approve <id...>")
559
+ return 0
560
+
561
+
562
+ def _decide(ids: list[str], status: str) -> int:
563
+ props = _load_pending()
564
+ if not props:
565
+ print("no pending proposals")
566
+ return 1
567
+ if ids == ["all"]:
568
+ ids = [p["id"] for p in props]
569
+ by_id = {p["id"]: p for p in props}
570
+ corpus_lines = read_corpus_lines()
571
+ done = []
572
+ for pid in ids:
573
+ p = by_id.get(pid)
574
+ if not p:
575
+ print(f"unknown id: {pid}")
576
+ continue
577
+ if status == "approved":
578
+ if p.get("action") == "revise" and p.get("revises_line"):
579
+ n = p["revises_line"]
580
+ if 0 < n <= len(corpus_lines):
581
+ corpus_lines[n - 1] = p["text"]
582
+ else:
583
+ corpus_lines.append(p["text"])
584
+ else:
585
+ corpus_lines.append(p["text"])
586
+ append_ledger(
587
+ {
588
+ "id": pid,
589
+ "status": status,
590
+ "text": p.get("text", ""),
591
+ "source_session": p.get("source_session"),
592
+ "decided_at": datetime.now(timezone.utc).isoformat(timespec="seconds"),
593
+ }
594
+ )
595
+ done.append(pid)
596
+ if status == "approved":
597
+ write_corpus_lines(corpus_lines)
598
+ remaining = [p for p in props if p["id"] not in set(done)]
599
+ pending_path().write_text(
600
+ json.dumps({"proposals": remaining}, indent=2, ensure_ascii=False)
601
+ )
602
+ print(f"{status}: {', '.join(done) if done else 'nothing'}; {len(remaining)} still pending")
603
+ if status == "approved":
604
+ print(f"corpus now has {len(corpus_lines)} lines: {corpus_path()}")
605
+ return 0
606
+
607
+
608
+ # ---------------------------------------------------------------------- CLI
609
+
610
+
611
+ def main() -> int:
612
+ ap = argparse.ArgumentParser(description="mine Claude transcripts into a context corpus")
613
+ sub = ap.add_subparsers(dest="cmd", required=True)
614
+
615
+ g = sub.add_parser("gather", help="collect sessions, enqueue one mining job, save proposals")
616
+ g.add_argument("--days", type=int, default=3)
617
+ g.add_argument("--budget", type=int, default=DEFAULT_TOTAL_BUDGET)
618
+ g.add_argument("--timeout", type=int, default=1800)
619
+ g.add_argument("--dry-run", action="store_true", help="build the prompt, don't enqueue")
620
+ g.set_defaults(func=run_gather)
621
+
622
+ r = sub.add_parser("review", help="print pending proposals")
623
+ r.set_defaults(func=cmd_review)
624
+
625
+ a = sub.add_parser("approve", help="apply proposal(s) to the corpus")
626
+ a.add_argument("ids", nargs="+")
627
+ a.set_defaults(func=lambda ns: _decide(ns.ids, "approved"))
628
+
629
+ s = sub.add_parser("skip", help="reject proposal(s), never re-pitched")
630
+ s.add_argument("ids", nargs="+")
631
+ s.set_defaults(func=lambda ns: _decide(ns.ids, "skipped"))
632
+
633
+ ns = ap.parse_args()
634
+ return ns.func(ns)
635
+
636
+
637
+ if __name__ == "__main__":
638
+ sys.exit(main())