@m13v/s4l 1.7.7-rc.5 → 1.7.7-rc.7

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/mcp/dist/index.js CHANGED
@@ -131,10 +131,13 @@ const STALL_WATCH_INTERVAL_SECS = 120;
131
131
  // S4L_OVERLAY_WATCH=0.
132
132
  const OVERLAY_WATCH_LABEL = "com.m13v.social-overlay-watch";
133
133
  const OVERLAY_WATCH_PLIST = path.join(os.homedir(), "Library", "LaunchAgents", `${OVERLAY_WATCH_LABEL}.plist`);
134
- // Daily self-updater. Enabled alongside autopilot so a hands-free (headless)
135
- // install keeps itself current — the interactive `runtime` tool (action:'update')
136
- // only helps when
137
- // a human-facing agent session is open, which an autopilot box never has.
134
+ // Legacy daily self-updater launchd job. BANNED (user decision 2026-07-31):
135
+ // it ran s4l_box_update.sh unattended, which kills and relaunches Claude
136
+ // Desktop with no human in the loop. Updates are user-initiated only (menubar
137
+ // "Update now & restart Claude Desktop", the panel button, `runtime`
138
+ // action:'update', or an operator running s4l_box_update.sh explicitly). The
139
+ // label/plist constants survive so ensureUpdaterRemoved() can clean the job
140
+ // off every box that got it from v1.7.1..v1.7.7-rc.4.
138
141
  const UPDATER_LABEL = "com.m13v.social-autoposter-update";
139
142
  const UPDATER_PLIST = path.join(os.homedir(), "Library", "LaunchAgents", `${UPDATER_LABEL}.plist`);
140
143
  // A sane PATH for launchd jobs (launchd starts with a bare PATH). Include the
@@ -4874,68 +4877,21 @@ async function ensureOverlayWatchInstalled() {
4874
4877
  return { ok: false, detail: e?.message || String(e) };
4875
4878
  }
4876
4879
  }
4877
- // Install/refresh the daily self-updater launchd job. This used to be bundled
4878
- // into the now-deleted `autopilot` MCP tool (removed 2026-06-19, 88bd1cb9):
4879
- // calling `autopilot enable` installed this job as a side effect, so removing
4880
- // the tool silently orphaned it — UPDATER_LABEL/UPDATER_PLIST and the
4881
- // auto_update_on status check survived (buildSnapshot still reports them), but
4882
- // nothing has installed the plist since, on ANY box provisioned after that
4883
- // commit (auto_update_on reads false forever). Restored here as its own
4884
- // deterministic boot-time job, same pattern as its five siblings above.
4885
- //
4886
- // Points at scripts/s4l_box_update.sh, NOT skill/social-autoposter-update.sh:
4887
- // the latter is the npm-lane updater (npm view + npx update) and is a silent
4888
- // no-op on a .mcpb box, which has no npm/npx on PATH (see version.ts). The
4889
- // .mcpb-lane equivalent downloads the .mcpb directly from the channel-resolved
4890
- // GitHub release and unpacks it over the extension dir — see that script's own
4891
- // header for the channel/no-downgrade/retry guards. Default mode there
4892
- // downloads + unpacks + restarts Claude Desktop with NO human in the loop,
4893
- // matching the original bundled updater's intent ("keeps a headless install
4894
- // current"); RunAtLoad so a box that boots already-behind checks promptly.
4895
- async function ensureUpdaterInstalled() {
4880
+ // Remove the legacy daily self-updater launchd job (see UPDATER_LABEL's
4881
+ // comment). Boot-time, idempotent, best-effort: unload the job if launchd has
4882
+ // it and delete the plist so the daily silent Claude Desktop restart can never
4883
+ // fire again. Do NOT reintroduce an install path for this job — any future
4884
+ // auto-update mechanism must be user-visible and user-confirmed.
4885
+ async function ensureUpdaterRemoved() {
4896
4886
  try {
4897
4887
  if (process.platform !== "darwin")
4898
- return { ok: false, detail: "not macOS" };
4899
- if ((process.env.S4L_AUTO_UPDATE) === "0")
4900
- return { ok: false, detail: "disabled (S4L_AUTO_UPDATE=0)" };
4901
- const logDir = path.join(repoDir(), "skill", "logs");
4902
- try {
4903
- fs.mkdirSync(logDir, { recursive: true });
4904
- }
4905
- catch {
4906
- /* best-effort */
4907
- }
4908
- const xml = plistXml({
4909
- label: UPDATER_LABEL,
4910
- programArgs: ["/bin/bash", path.join(repoDir(), "scripts", "s4l_box_update.sh")],
4911
- intervalSecs: 86_400,
4912
- runAtLoad: true,
4913
- stdoutLog: path.join(logDir, "launchd-self-update-stdout.log"),
4914
- stderrLog: path.join(logDir, "launchd-self-update-stderr.log"),
4915
- });
4888
+ return { ok: true, detail: "not macOS" };
4916
4889
  const uid = process.getuid ? process.getuid() : 0;
4917
- let cur = null;
4918
- try {
4919
- cur = fs.readFileSync(UPDATER_PLIST, "utf-8");
4920
- }
4921
- catch {
4922
- cur = null;
4923
- }
4924
- let detail;
4925
- if (cur === xml) {
4926
- const res = await loadPlist(UPDATER_LABEL, UPDATER_PLIST, uid);
4927
- detail = `current (load rc=${res.code})`;
4928
- }
4929
- else {
4930
- if (cur !== null) {
4931
- await unloadPlist(UPDATER_LABEL, UPDATER_PLIST, uid);
4932
- }
4933
- fs.mkdirSync(path.dirname(UPDATER_PLIST), { recursive: true });
4934
- fs.writeFileSync(UPDATER_PLIST, xml, "utf-8");
4935
- const res = await loadPlist(UPDATER_LABEL, UPDATER_PLIST, uid);
4936
- detail = cur === null ? "installed + loaded" : `rewritten + reloaded (rc=${res.code})`;
4937
- }
4938
- return { ok: true, detail };
4890
+ const existed = fs.existsSync(UPDATER_PLIST);
4891
+ await unloadPlist(UPDATER_LABEL, UPDATER_PLIST, uid);
4892
+ if (existed)
4893
+ fs.rmSync(UPDATER_PLIST, { force: true });
4894
+ return { ok: true, detail: existed ? "removed" : "absent" };
4939
4895
  }
4940
4896
  catch (e) {
4941
4897
  return { ok: false, detail: e?.message || String(e) };
@@ -6134,13 +6090,12 @@ async function main() {
6134
6090
  void ensureOverlayWatchInstalled()
6135
6091
  .then((r) => console.error(`[overlay-watch] launchd supervisor: ${r.ok ? "ok" : "skip"} (${r.detail})`))
6136
6092
  .catch((e) => console.error("[overlay-watch] supervisor install failed:", e?.message || e));
6137
- // Daily self-updater: restored 2026-07-08 after 88bd1cb9 ("Remove autopilot
6138
- // tool") silently dropped its only install path (it used to be bundled into
6139
- // the deleted `autopilot enable` action). Best-effort; never blocks boot.
6140
- // Disable with S4L_AUTO_UPDATE=0.
6141
- void ensureUpdaterInstalled()
6142
- .then((r) => console.error(`[self-update] launchd updater: ${r.ok ? "ok" : "skip"} (${r.detail})`))
6143
- .catch((e) => console.error("[self-update] updater install failed:", e?.message || e));
6093
+ // Silent daily self-updates are banned (user decision 2026-07-31): the job
6094
+ // restarted Claude Desktop unattended. Clean it off every box that got it
6095
+ // from v1.7.1..v1.7.7-rc.4. Best-effort; never blocks boot.
6096
+ void ensureUpdaterRemoved()
6097
+ .then((r) => console.error(`[self-update] legacy launchd updater: ${r.ok ? "ok" : "skip"} (${r.detail})`))
6098
+ .catch((e) => console.error("[self-update] updater removal failed:", e?.message || e));
6144
6099
  // Heal installs onboarded before short_links_live defaulted to false: such a
6145
6100
  // project wraps short links against the customer's own domain, which has no
6146
6101
  // /r/[code] resolver, so every minted link 404s. Re-point them at the s4l.ai
package/mcp/dist/repo.js CHANGED
@@ -211,7 +211,9 @@ export function readPlan(batchId) {
211
211
  }
212
212
  }
213
213
  export function writePlan(batchId, plan) {
214
- fs.writeFileSync(planPath(batchId), JSON.stringify(plan, null, 2), "utf-8");
214
+ // Compact (no indent): the review store reached 65 MB with pretty-printing
215
+ // (2026-08-02 lag incident) and every reader pays the parse.
216
+ fs.writeFileSync(planPath(batchId), JSON.stringify(plan), "utf-8");
215
217
  }
216
218
  // Find the newest plan file when no batch id is supplied.
217
219
  export function latestBatchId() {
@@ -1,4 +1,4 @@
1
1
  {
2
- "version": "1.7.7-rc.5",
3
- "installedAt": "2026-07-31T22:56:07.561Z"
2
+ "version": "1.7.7-rc.7",
3
+ "installedAt": "2026-08-02T22:02:14.223Z"
4
4
  }
package/mcp/manifest.json CHANGED
@@ -2,7 +2,7 @@
2
2
  "dxt_version": "0.1",
3
3
  "name": "social-autoposter",
4
4
  "display_name": "S4L",
5
- "version": "1.7.7-rc.5",
5
+ "version": "1.7.7-rc.7",
6
6
  "description": "Draft, review, approve, and autopilot X/Twitter posts.",
7
7
  "long_description": "## **⚠️ The disclaimer above is generic Claude boilerplate.** Anthropic shows the same warning on every plugin regardless of what it does; any plugin has the same level of access as any app you download from the internet.\n\nS4L is an open source product developed by Mediar.ai Incorporated, a VC-backed San Francisco-based startup.\n\nTo get started:\n\n1\\. Copy this prompt: **Set me up on S4L plugin end to end**\n\n2\\. Quit with CMD+Q, reopen Claude, paste into a new chat.\n\nWhat happens next:\n\n* About every 5 minutes S4L scans X for posts that match your topics and drafts replies in your voice.\n* Drafts show up as review cards, usually the first within a few minutes. Nothing is posted automatically; you approve each one.\n* Posting autopilot stays off until you explicitly turn it on.",
8
8
  "author": {
@@ -27,6 +27,20 @@ import sys
27
27
  import tempfile
28
28
  import threading
29
29
  import time
30
+ import warnings
31
+
32
+ # PyObjC emits an ObjCPointerWarning for every CGColor() handed to a CALayer
33
+ # (s4l_card's tile styling). The message embeds the pointer address, so the
34
+ # warnings module's once-per-location dedup never matches and a single card
35
+ # render sprayed ~100 lines/sec into menubar.err.log (13k+ lines by the
36
+ # 2026-08-02 lag incident). The pointers are handled correctly; silence the
37
+ # category before any card module loads.
38
+ try:
39
+ import objc
40
+
41
+ warnings.filterwarnings("ignore", category=objc.ObjCPointerWarning)
42
+ except Exception:
43
+ pass
30
44
 
31
45
  # --- stderr timestamp wrapper -------------------------------------------------
32
46
  # menubar.err.log (this process's stderr, redirected by the launchd plist) has
@@ -715,8 +729,28 @@ class S4LMenuBar(rumps.App):
715
729
  self._tick_stats_at = 0.0
716
730
  self._reloc_timer = rumps.Timer(self._maybe_relocate_tasks, 90)
717
731
  self._reloc_timer.start()
732
+ # Store compaction: archive heavy fields off settled candidates so the
733
+ # review-queue file (and every 1s/5s poll that parses it) stays small.
734
+ # Boot pass runs off the main thread — the first pass after an
735
+ # un-compacted stretch can chew through tens of MB.
736
+ self._compacted_at = 0.0
737
+ self._compact_store_async()
718
738
  self._tick(None)
719
739
 
740
+ def _compact_store_async(self):
741
+ self._compacted_at = time.time()
742
+
743
+ def run():
744
+ try:
745
+ n = st.compact_store()
746
+ if n:
747
+ sys.stderr.write(f"[s4l-menubar] store compaction archived {n} candidate(s)\n")
748
+ sys.stderr.flush()
749
+ except Exception:
750
+ pass
751
+
752
+ threading.Thread(target=run, daemon=True, name="s4l-compact-store").start()
753
+
720
754
  # ---- side effects -----------------------------------------------------
721
755
  def _open_claude(self, _=None):
722
756
  subprocess.run(["open", "-a", CLAUDE_APP], capture_output=True,
@@ -3009,6 +3043,10 @@ class S4LMenuBar(rumps.App):
3009
3043
  sys.stderr.flush()
3010
3044
  except Exception:
3011
3045
  pass
3046
+ # Hourly store compaction keeps the review-queue file small as posts
3047
+ # settle (boot already ran one pass; see _compact_store_async).
3048
+ if now_rc - self._compacted_at >= 3600:
3049
+ self._compact_store_async()
3012
3050
 
3013
3051
  # ---- draft review pop-ups ---------------------------------------------
3014
3052
  def _posting_activity_label_locked(self):
@@ -721,7 +721,9 @@ def _store_update(mutate):
721
721
  data = {"candidates": []}
722
722
  rv = mutate(data)
723
723
  tmp = f"{sp}.tmp.{os.getpid()}"
724
- Path(tmp).write_text(json.dumps(data, indent=2))
724
+ # Compact separators: this file reached 65 MB with indent=2
725
+ # (2026-08-02 lag incident) and every reader pays the parse.
726
+ Path(tmp).write_text(json.dumps(data, separators=(",", ":")))
725
727
  os.replace(tmp, sp)
726
728
  return rv
727
729
  finally:
@@ -779,6 +781,97 @@ def candidate_state(c):
779
781
  return "awaiting_review"
780
782
 
781
783
 
784
+ # Draft-time payloads that are only needed while a candidate can still POST
785
+ # (awaiting_review: card rendering; approved/post_failed: the drain path
786
+ # rebuilds a mini-plan from reddit_decision/reddit_plan_meta). Once a candidate
787
+ # is posted or terminal these blobs are dead weight — and they dominated the
788
+ # 65 MB store that caused the 2026-08-02 UI lag (single reddit candidates
789
+ # carried ~180 KB). Compaction moves them to an append-only sidecar so the
790
+ # append-forever ledger keeps every byte while the hot file stays small.
791
+ HEAVY_ARCHIVE_FIELDS = ("reddit_plan_meta", "reddit_decision", "thread_selftext")
792
+
793
+ # Audit-only blobs inside reddit_plan_meta.style_assignment, stamped by
794
+ # engagement_styles.pick_style_for_post "for audit" and copied onto EVERY
795
+ # reddit candidate by merge_review_queue. distribution_snapshot alone was
796
+ # ~107 KB per candidate (7.6 MB across one pending backlog). Nothing on the
797
+ # card-render or posting path reads them (post_reddit.py --phase post uses
798
+ # only .style/.mode; index.ts forwards the dict opaquely), so they are safe
799
+ # to archive off candidates in ANY state, pending included.
800
+ STYLE_AUDIT_FIELDS = ("distribution_snapshot", "reference_styles")
801
+
802
+ ARCHIVE_STORE = "review-queue-archive.jsonl"
803
+
804
+
805
+ def compact_store():
806
+ """Archive heavy payloads out of the hot store into review-queue-archive
807
+ .jsonl, then rewrite the store compactly. Two passes over the candidates:
808
+ HEAVY_ARCHIVE_FIELDS come off SETTLED (posted/terminal) candidates only —
809
+ approved/post_failed rows still need reddit_decision/reddit_plan_meta for
810
+ a (re-)approval drain — while STYLE_AUDIT_FIELDS come off every candidate.
811
+ Runs under the same store lock as every other python writer. The archive
812
+ lines land (flush+fsync) BEFORE any field is stripped, so no data is ever
813
+ lost — a crash in between only risks a duplicate archive line, never a
814
+ missing one. Returns the number of candidates compacted (0 when there was
815
+ nothing to do), None on failure."""
816
+
817
+ ap = str(Path(state_dir()) / ARCHIVE_STORE)
818
+
819
+ def mutate(data):
820
+ records = [] # (candidate, archived-fields dict, strip callback)
821
+ for c in data.get("candidates") or []:
822
+ fields = {}
823
+ settled = candidate_state(c) in ("posted", "terminal")
824
+ if settled:
825
+ fields.update(
826
+ {k: c[k] for k in HEAVY_ARCHIVE_FIELDS if c.get(k)}
827
+ )
828
+ sa = (c.get("reddit_plan_meta") or {}).get("style_assignment")
829
+ audit = (
830
+ {}
831
+ if (settled and "reddit_plan_meta" in fields) or not isinstance(sa, dict)
832
+ else {k: sa[k] for k in STYLE_AUDIT_FIELDS if sa.get(k)}
833
+ )
834
+ if audit:
835
+ fields["style_assignment_audit"] = audit
836
+
837
+ def strip(c=c, settled=settled, sa=sa, audit=audit):
838
+ if settled:
839
+ for k in HEAVY_ARCHIVE_FIELDS:
840
+ c.pop(k, None)
841
+ for k in audit:
842
+ sa.pop(k, None)
843
+
844
+ if fields:
845
+ records.append((c, fields, strip))
846
+ if not records:
847
+ return 0
848
+ with open(ap, "a") as f:
849
+ for c, fields, _ in records:
850
+ f.write(
851
+ json.dumps(
852
+ {
853
+ "candidate_id": c.get("candidate_id"),
854
+ "our_url": c.get("our_url"),
855
+ "thread_url": c.get("thread_url"),
856
+ "archived_at": time_iso(),
857
+ "fields": fields,
858
+ },
859
+ separators=(",", ":"),
860
+ )
861
+ + "\n"
862
+ )
863
+ f.flush()
864
+ os.fsync(f.fileno())
865
+ for c, fields, strip in records:
866
+ strip()
867
+ c["archived_fields"] = sorted(
868
+ set(c.get("archived_fields") or []) | set(fields)
869
+ )
870
+ return len(records)
871
+
872
+ return _store_update(mutate)
873
+
874
+
782
875
  def store_stamp_decision(batch, decision):
783
876
  """Write a card decision INTO the store the instant the user clicks. This is
784
877
  the durable record (the old approved-queue.json ledger is no longer
@@ -1001,13 +1094,22 @@ def store_reconcile_decisions(batch, decisions):
1001
1094
  return fixed
1002
1095
 
1003
1096
 
1097
+ # (path, mtime_ns, size) -> posted count. review_queue_posted_count() is on the
1098
+ # menubar's 1-second activity poll; without this cache that poll re-parsed the
1099
+ # whole store every tick, which saturated the AppKit main thread once the file
1100
+ # grew (2026-08-02: 65 MB, 98% CPU, multi-second card lag).
1101
+ _posted_count_cache = {"key": None, "count": None}
1102
+
1103
+
1004
1104
  def review_queue_posted_count():
1005
1105
  """Posts that have LANDED in the review-queue plan — the durable, cross-process
1006
1106
  truth. Independent of the menu bar's in-memory burst queue (which dies on a
1007
1107
  restart) and of WHICH process is posting (the menu bar worker, the autopilot,
1008
1108
  or a host agent draining via approve_drafts). Returns the posted count, or None
1009
1109
  when the plan can't be read. Drives the menu-bar posting indicator so progress
1010
- stays visible regardless of how the drain is driven."""
1110
+ stays visible regardless of how the drain is driven. Cached on the plan
1111
+ file's (mtime, size): the caller polls every second, the file changes only
1112
+ when a writer lands."""
1011
1113
  plan_path = None
1012
1114
  req = read_review_request()
1013
1115
  if req:
@@ -1016,11 +1118,22 @@ def review_queue_posted_count():
1016
1118
  plan_path = store_path()
1017
1119
  if not Path(plan_path).exists():
1018
1120
  plan_path = "/tmp/twitter_cycle_plan_review-queue.json"
1121
+ try:
1122
+ stt = os.stat(plan_path)
1123
+ cache_key = (os.path.realpath(plan_path), stt.st_mtime_ns, stt.st_size)
1124
+ except OSError:
1125
+ cache_key = None
1126
+ if cache_key is not None and _posted_count_cache["key"] == cache_key:
1127
+ return _posted_count_cache["count"]
1019
1128
  plan = read_plan(plan_path)
1020
1129
  cands = (plan or {}).get("candidates")
1021
- if not cands:
1022
- return None
1023
- return sum(1 for c in cands if candidate_state(c) == "posted")
1130
+ count = None if not cands else sum(
1131
+ 1 for c in cands if candidate_state(c) == "posted"
1132
+ )
1133
+ if cache_key is not None:
1134
+ _posted_count_cache["key"] = cache_key
1135
+ _posted_count_cache["count"] = count
1136
+ return count
1024
1137
 
1025
1138
 
1026
1139
  def review_drafts(plan, batch="review-queue"):
package/mcp/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@m13v/s4l-mcp",
3
- "version": "1.7.7-rc.5",
3
+ "version": "1.7.7-rc.7",
4
4
  "private": true,
5
5
  "description": "Desktop MCP client for social-autoposter (X/Twitter rail): manual draft/review/approve loop, autopilot control, and stats. Thin wrapper over the existing pipeline scripts.",
6
6
  "license": "MIT",
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@m13v/s4l",
3
- "version": "1.7.7-rc.5",
3
+ "version": "1.7.7-rc.7",
4
4
  "description": "Automated social posting pipeline for Reddit, X/Twitter, LinkedIn, and Moltbook. Install as a Claude Code agent skill.",
5
5
  "bin": {
6
6
  "social-autoposter": "bin/cli.js",
@@ -76,6 +76,12 @@ TAG_TO_TYPE = {
76
76
  # Python and inlined; no Bash tools), mirroring the twitter Phase 2b
77
77
  # tool-free conversion, so it rides the same universal worker.
78
78
  "post-reddit-draft": "reddit-draft",
79
+ # Context mining (2026-07-31): scripts/context_mining.py inlines compacted
80
+ # transcripts + the corpus into one pure text->JSON turn. Execution
81
+ # guidance lives at the top of the prompt itself (not only in
82
+ # TYPE_TO_WORKER_NOTES) because an installed-package worker claims with
83
+ # ITS copy of this map, which may predate this type.
84
+ "context-mining": "context-mining",
79
85
  }
80
86
 
81
87
  # queue type -> (activity state, label) the menu bar shows while the job is in
@@ -92,6 +98,7 @@ TYPE_TO_ACTIVITY = {
92
98
  # must stay concise (user rule 2026-07-14); the platform shows on the
93
99
  # review card itself, not in the tray.
94
100
  "reddit-draft": ("drafting", "draft"),
101
+ "context-mining": ("learning", "mining context"),
95
102
  }
96
103
 
97
104
  # queue type -> execution notes PREPENDED to the prompt sidecar at claim time.
@@ -140,6 +147,13 @@ TYPE_TO_WORKER_NOTES = {
140
147
  "absent from posts (add a rejects entry only for the structural "
141
148
  "false-positive cases the prompt describes)."
142
149
  ),
150
+ "context-mining": (
151
+ "WORKER EXECUTION NOTES (queue metadata; follow while executing the "
152
+ "prompt below): single-turn, tool-free job. Everything you need is "
153
+ "inlined in the prompt (transcripts, corpus, prior decisions). Never "
154
+ "use tools. Read, then submit ONE result object matching the schema: "
155
+ "{\"proposals\": [...]}. An empty proposals list is a valid answer."
156
+ ),
143
157
  }
144
158
 
145
159
 
@@ -0,0 +1,664 @@
1
+ #!/usr/bin/env python3
2
+ """
3
+ context_mining.py -- PROTOTYPE: mine the user's own Claude conversation
4
+ transcripts for insights worth adding to a personal context corpus, using the
5
+ existing claude_job.py file queue as the LLM lane (design agreed 2026-07-31).
6
+
7
+ Flow (test-run shape; toggle/watermark/cards come later):
8
+ gather -> deterministically collect recent human-interactive sessions,
9
+ compact each FULL transcript (user prose kept, tool plumbing
10
+ dropped), inline the current corpus (numbered lines) plus the
11
+ already-considered ledger, enqueue ONE pure text->JSON job via
12
+ run_claude.sh (tag: context-mining), block for the result, and
13
+ write proposals to pending.json.
14
+ review -> print pending proposals.
15
+ approve -> apply a proposal to corpus.md (append or replace a line) and
16
+ record it in the considered ledger.
17
+ skip -> record the proposal as skipped so it is never re-pitched.
18
+
19
+ State lives under <state_dir>/context-mining/ where state_dir is
20
+ $S4L_STATE_DIR or ~/.social-autoposter-mcp (same convention as the queue):
21
+ corpus.md the corpus; every non-blank line is one numbered memory
22
+ pending.json proposals awaiting a human decision
23
+ considered.jsonl append-only ledger of every approve/skip decision
24
+
25
+ Sources walked (same two stores as extract_user_messages_today.py):
26
+ ~/.claude/projects/*/*.jsonl Claude Code (CLI + desktop)
27
+ ~/Library/Application Support/Claude/
28
+ local-agent-mode-sessions/**/.claude/projects/**/*.jsonl Cowork
29
+
30
+ Eligibility: a session must contain >= MIN_HUMAN_MSGS user messages classified
31
+ HUMAN (typed prose, not command echoes / task notifications / reminders), and
32
+ its cwd must not be an S4L worker sandbox. S4L's own automation transcripts are
33
+ never mined -- that would be a feedback loop.
34
+
35
+ Usage:
36
+ python3 scripts/context_mining.py gather --days 3
37
+ python3 scripts/context_mining.py gather --days 3 --dry-run # build, no enqueue
38
+ python3 scripts/context_mining.py review
39
+ python3 scripts/context_mining.py approve cm-1a2b3c4d [id...]
40
+ python3 scripts/context_mining.py skip cm-1a2b3c4d [id...]
41
+ """
42
+
43
+ from __future__ import annotations
44
+
45
+ import argparse
46
+ import hashlib
47
+ import json
48
+ import os
49
+ import re
50
+ import subprocess
51
+ import sys
52
+ import time
53
+ from datetime import datetime, timedelta, timezone
54
+ from pathlib import Path
55
+
56
+ HOME = Path.home()
57
+ REPO_DIR = Path(__file__).resolve().parent.parent
58
+
59
+ CLAUDE_CODE_PROJECTS_ROOT = HOME / ".claude" / "projects"
60
+ COWORK_SESSIONS_ROOT = (
61
+ HOME / "Library" / "Application Support" / "Claude" / "local-agent-mode-sessions"
62
+ )
63
+
64
+
65
+ def state_dir() -> Path:
66
+ base = os.environ.get("S4L_STATE_DIR") or str(HOME / ".social-autoposter-mcp")
67
+ d = Path(base) / "context-mining"
68
+ d.mkdir(parents=True, exist_ok=True)
69
+ return d
70
+
71
+
72
+ def corpus_path() -> Path:
73
+ return state_dir() / "corpus.md"
74
+
75
+
76
+ def pending_path() -> Path:
77
+ return state_dir() / "pending.json"
78
+
79
+
80
+ def ledger_path() -> Path:
81
+ return state_dir() / "considered.jsonl"
82
+
83
+
84
+ # ---------------------------------------------------------------- gathering
85
+
86
+ MIN_HUMAN_MSGS = 2
87
+ PER_USER_MSG_CAP = 2500
88
+ PER_ASSISTANT_MSG_CAP = 700
89
+ PER_SESSION_CAP = 14_000
90
+ DEFAULT_TOTAL_BUDGET = 80_000
91
+
92
+ # cwd substrings that mark a session as S4L's own automation, never mined
93
+ EXCLUDED_CWD_MARKERS = (".s4l-worker", ".saps-worker")
94
+
95
+
96
+ def _classify(text: str) -> str:
97
+ """HUMAN vs harness-injected user-role messages (same taxonomy as
98
+ extract_user_messages_today.py)."""
99
+ stripped = text.lstrip()
100
+ if stripped.startswith("<task-notification>"):
101
+ return "TASK_NOTIF"
102
+ if stripped.startswith("<command-name>") or stripped.startswith("<command-message>"):
103
+ return "COMMAND"
104
+ if stripped.startswith("<local-command-stdout>") or stripped.startswith(
105
+ "<local-command-stderr>"
106
+ ):
107
+ return "CMD_STDOUT"
108
+ if "<<autonomous-loop" in stripped or stripped.startswith("<loop-"):
109
+ return "SCHED_WAKE"
110
+ if stripped.startswith("<user-prompt-submit-hook>"):
111
+ return "HOOK"
112
+ if stripped.startswith("<system-reminder>"):
113
+ without = re.sub(
114
+ r"<system-reminder>.*?</system-reminder>", "", stripped, flags=re.DOTALL
115
+ ).strip()
116
+ if not without:
117
+ return "SYS_REMIND"
118
+ return "HUMAN"
119
+
120
+
121
+ def _strip_reminders(text: str) -> str:
122
+ return re.sub(r"<system-reminder>.*?</system-reminder>", "", text, flags=re.DOTALL).strip()
123
+
124
+
125
+ def _text_blocks(content) -> str | None:
126
+ """Textual payload of a message; None for tool_result plumbing."""
127
+ if isinstance(content, str):
128
+ return content
129
+ if isinstance(content, list):
130
+ texts = []
131
+ for block in content:
132
+ if not isinstance(block, dict):
133
+ continue
134
+ btype = block.get("type")
135
+ if btype == "tool_result":
136
+ return None
137
+ if btype == "text" and isinstance(block.get("text"), str):
138
+ texts.append(block["text"])
139
+ if texts:
140
+ return "\n".join(texts)
141
+ return None
142
+
143
+
144
+ def _iter_transcript_files():
145
+ if CLAUDE_CODE_PROJECTS_ROOT.is_dir():
146
+ for p in CLAUDE_CODE_PROJECTS_ROOT.glob("*/*.jsonl"):
147
+ yield "claude_code", p
148
+ if COWORK_SESSIONS_ROOT.is_dir():
149
+ for p in COWORK_SESSIONS_ROOT.glob("**/.claude/projects/**/*.jsonl"):
150
+ yield "cowork", p
151
+
152
+
153
+ def _compact_session(path: Path) -> dict | None:
154
+ """Read one full transcript and compact it to the conversation arc:
155
+ HUMAN user messages (near-full) + assistant prose (trimmed). Returns None
156
+ if the session is not human-interactive or is S4L automation."""
157
+ cwd = None
158
+ first_ts = last_ts = None
159
+ human_count = 0
160
+ turns: list[str] = []
161
+ try:
162
+ with path.open("r", encoding="utf-8", errors="replace") as f:
163
+ for line in f:
164
+ try:
165
+ d = json.loads(line)
166
+ except Exception:
167
+ continue
168
+ ts = d.get("timestamp")
169
+ if ts:
170
+ first_ts = first_ts or ts
171
+ last_ts = ts
172
+ if cwd is None and d.get("cwd"):
173
+ cwd = d["cwd"]
174
+ if d.get("isSidechain"):
175
+ continue
176
+ msg = d.get("message") or {}
177
+ role = msg.get("role")
178
+ if d.get("type") == "user" and role == "user":
179
+ text = _text_blocks(msg.get("content"))
180
+ if text is None:
181
+ continue
182
+ text = text.strip()
183
+ if not text or _classify(text) != "HUMAN":
184
+ continue
185
+ text = _strip_reminders(text)
186
+ if not text:
187
+ continue
188
+ human_count += 1
189
+ turns.append("USER: " + text[:PER_USER_MSG_CAP])
190
+ elif d.get("type") == "assistant" and role == "assistant":
191
+ text = _text_blocks(msg.get("content"))
192
+ if not text:
193
+ continue
194
+ text = text.strip()
195
+ if len(text) < 40: # skip one-liner tool narration
196
+ continue
197
+ turns.append("ASSISTANT: " + text[:PER_ASSISTANT_MSG_CAP])
198
+ except OSError:
199
+ return None
200
+
201
+ if human_count < MIN_HUMAN_MSGS:
202
+ return None
203
+ if cwd and any(m in cwd for m in EXCLUDED_CWD_MARKERS):
204
+ return None
205
+
206
+ body = "\n\n".join(turns)
207
+ if len(body) > PER_SESSION_CAP:
208
+ # keep the head and the tail; the middle is elided
209
+ half = PER_SESSION_CAP // 2
210
+ body = body[:half] + "\n\n[... middle of session elided ...]\n\n" + body[-half:]
211
+ return {
212
+ "session_id": path.stem,
213
+ "path": str(path),
214
+ "cwd": cwd or "?",
215
+ "first_ts": first_ts or "?",
216
+ "last_ts": last_ts or "?",
217
+ "human_msgs": human_count,
218
+ "body": body,
219
+ }
220
+
221
+
222
+ def gather_sessions(days: int) -> list[dict]:
223
+ """ALL eligible human sessions within the window, most recent first."""
224
+ cutoff = time.time() - days * 86400
225
+ candidates = []
226
+ for _source, p in _iter_transcript_files():
227
+ try:
228
+ mtime = p.stat().st_mtime
229
+ except OSError:
230
+ continue
231
+ if mtime < cutoff:
232
+ continue
233
+ candidates.append((mtime, p))
234
+ candidates.sort(reverse=True)
235
+
236
+ kept: list[dict] = []
237
+ seen_ids: set[str] = set()
238
+ for _mtime, p in candidates:
239
+ if p.stem in seen_ids:
240
+ continue
241
+ sess = _compact_session(p)
242
+ if sess is None:
243
+ continue
244
+ seen_ids.add(p.stem)
245
+ kept.append(sess)
246
+ return kept
247
+
248
+
249
+ def chunk_by_budget(sessions: list[dict], budget: int) -> list[list[dict]]:
250
+ """Split into batches whose combined body size fits the budget. A batch
251
+ always takes at least one session, so oversized sessions still ship."""
252
+ batches: list[list[dict]] = []
253
+ cur: list[dict] = []
254
+ used = 0
255
+ for s in sessions:
256
+ cost = len(s["body"])
257
+ if cur and used + cost > budget:
258
+ batches.append(cur)
259
+ cur, used = [], 0
260
+ cur.append(s)
261
+ used += cost
262
+ if cur:
263
+ batches.append(cur)
264
+ return batches
265
+
266
+
267
+ # ------------------------------------------------------------------- corpus
268
+
269
+
270
+ def read_corpus_lines() -> list[str]:
271
+ if not corpus_path().exists():
272
+ return []
273
+ return [ln.rstrip("\n") for ln in corpus_path().read_text().splitlines() if ln.strip()]
274
+
275
+
276
+ def write_corpus_lines(lines: list[str]) -> None:
277
+ corpus_path().write_text("\n".join(lines) + ("\n" if lines else ""))
278
+
279
+
280
+ def read_ledger() -> list[dict]:
281
+ if not ledger_path().exists():
282
+ return []
283
+ out = []
284
+ for ln in ledger_path().read_text().splitlines():
285
+ try:
286
+ out.append(json.loads(ln))
287
+ except Exception:
288
+ continue
289
+ return out
290
+
291
+
292
+ def append_ledger(rec: dict) -> None:
293
+ with ledger_path().open("a") as f:
294
+ f.write(json.dumps(rec, ensure_ascii=False) + "\n")
295
+
296
+
297
+ # ------------------------------------------------------------------- prompt
298
+
299
+ SCHEMA = {
300
+ "type": "object",
301
+ "required": ["proposals"],
302
+ "properties": {
303
+ "proposals": {
304
+ "type": "array",
305
+ "items": {
306
+ "type": "object",
307
+ "required": [
308
+ "action",
309
+ "title",
310
+ "story",
311
+ "conclusion",
312
+ "source_session",
313
+ "source_date",
314
+ ],
315
+ "properties": {
316
+ "action": {"type": "string", "enum": ["add", "revise"]},
317
+ "revises_line": {"type": ["integer", "null"]},
318
+ "title": {"type": "string"},
319
+ "story": {"type": "string"},
320
+ "conclusion": {"type": "string"},
321
+ "source_session": {"type": "string"},
322
+ "source_date": {"type": "string"},
323
+ },
324
+ },
325
+ }
326
+ },
327
+ }
328
+
329
+
330
+ def corpus_entry(p: dict) -> str:
331
+ """Render one approved proposal as a single corpus line (numbering is
332
+ line-based, so the entry must not contain newlines)."""
333
+ story = " ".join((p.get("story") or "").split())
334
+ concl = " ".join((p.get("conclusion") or "").split())
335
+ title = " ".join((p.get("title") or "").split())
336
+ return f"{title}: {story} Conclusion: {concl}"
337
+
338
+
339
+ def build_prompt(sessions: list[dict], pending_props: list[dict] | None = None) -> str:
340
+ corpus_lines = read_corpus_lines()
341
+ corpus_block = (
342
+ "\n".join(f"[{i + 1}] {ln}" for i, ln in enumerate(corpus_lines))
343
+ if corpus_lines
344
+ else "(the corpus is currently empty)"
345
+ )
346
+ def _gist(r: dict) -> str:
347
+ # works for both the old one-line shape (text) and the story shape
348
+ return r.get("text") or f"{r.get('title', '')}: {r.get('conclusion', '')}"
349
+
350
+ considered = [
351
+ {"status": r.get("status"), "gist": _gist(r)} for r in read_ledger()[-60:]
352
+ ] + [{"status": "pending", "gist": _gist(p)} for p in (pending_props or [])]
353
+ considered_block = (
354
+ "\n".join(f"- ({r['status']}) {r['gist'][:200]}" for r in considered)
355
+ if considered
356
+ else "(nothing has been considered yet)"
357
+ )
358
+ transcript_parts = []
359
+ for s in sessions:
360
+ transcript_parts.append(
361
+ f"### Session {s['session_id'][:8]} | {s['first_ts'][:10]} | cwd: {s['cwd']}\n\n{s['body']}"
362
+ )
363
+ transcripts_block = "\n\n---\n\n".join(transcript_parts)
364
+
365
+ return f"""EXECUTION NOTES: this is a single-turn, tool-free job. Read everything below, then submit ONE result object matching the provided JSON schema. Do not use any tools. Do not fetch anything.
366
+
367
+ # Role
368
+
369
+ You are the daily context miner for a personal "context corpus": a numbered list of durable memories distilled from the user's own Claude conversations. Each entry is a short STORY plus a CONCLUSION: what actually happened (with its credible specifics) and the transferable lesson it taught. Entries are worth keeping and potentially worth sharing with the world: a specific experience, a hard-won conclusion, a non-obvious observation, or a concrete data point.
370
+
371
+ The user is Matthew, a solo founder building S4L (a social-media autoposting agent), Fazm, Mediar, and related products, and doing everything else through Claude sessions: fundraising, debugging, ops, legal, marketing.
372
+
373
+ # What qualifies
374
+
375
+ - GENERAL: the line states a transferable lesson a stranger in tech could apply to
376
+ their own work, without knowing this codebase or these products. The specific
377
+ incident is EVIDENCE (it goes in the quote and the why), not the line itself.
378
+ Test: would a founder who has never heard of S4L or Fazm reshare this?
379
+ - Durable: still true and useful months from now, not transient state ("the deploy is broken today").
380
+ - Non-obvious: something a smart peer would not already assume. Generic best practices do NOT qualify.
381
+ - First-hand: grounded in what actually happened in these sessions (an experiment, an outage, a metric, a decision and its reasoning, a surprising vendor/platform behavior).
382
+ - SYNTHESIZED: if several incidents (even across sessions) point at one underlying
383
+ lesson, propose ONE line for the lesson, never one line per incident.
384
+
385
+ # What never qualifies
386
+
387
+ - Vendor-specific micro-gotchas (a field length limit, one API's quirky error) unless
388
+ the line is elevated to the general pattern the gotcha exemplifies.
389
+ - Anything that only matters inside this codebase or product internals.
390
+ - Credentials, API keys, tokens, passwords, account numbers, or anything that looks like one, even partially. If a great insight touches one, redact the secret.
391
+ - Names, handles, or identifying details of customers, users, or other private
392
+ individuals. Describe them generically ("a stalled user", "an early customer").
393
+ - Private details about clients, other people's finances, legal disputes, or immigration status.
394
+ - Injected harness noise: usage-limit banners, system reminders, scheduler chatter. These appear inside transcripts and are not the user's words.
395
+ - Anything already covered by an existing corpus line (see below), unless you are proposing a REVISION of that line.
396
+
397
+ # Current corpus (numbered)
398
+
399
+ {corpus_block}
400
+
401
+ # Already considered (do NOT re-propose anything equivalent, including skipped items)
402
+
403
+ {considered_block}
404
+
405
+ # Transcripts to mine ({len(sessions)} sessions, most recent first)
406
+
407
+ {transcripts_block}
408
+
409
+ # Your task
410
+
411
+ Propose 0-3 corpus changes, and only what clears EVERY bar above; most days the
412
+ right answer is 0 or 1. An empty proposals list is a valid, common answer. Each
413
+ proposal is a small STORY with a CONCLUSION, not a bare aphorism:
414
+ - action: "add" for a new entry, or "revise" when a new learning supersedes an existing corpus entry (set revises_line to that entry's number).
415
+ - title: a short, specific handle for the insight (max ~70 chars, not clickbait).
416
+ - story: 2-5 sentences telling what actually happened, in the user's plainspoken first-person voice. Keep the concrete specifics that make it credible and vivid (the numbers, the timeline, the failure mode, what was tried), but generalize or omit anything only an insider would care about. Anonymize people, redact secrets. No hashtags, no em dashes.
417
+ - conclusion: 1-2 sentences stating the transferable lesson a stranger could apply to their own work. This is the part that must stand on its own.
418
+ - source_session: the 8-char session id from the transcript header.
419
+ - source_date: the session date (YYYY-MM-DD).
420
+
421
+ Submit the result object now."""
422
+
423
+
424
+ # -------------------------------------------------------------------- queue
425
+
426
+
427
+ def _run_one_batch(sessions: list[dict], pending_props: list[dict], ns) -> list[dict] | None:
428
+ """Enqueue one mining job for this batch; return stamped proposals, or None
429
+ on failure (caller stops the loop and keeps what it has)."""
430
+ prompt = build_prompt(sessions, pending_props)
431
+ schema_file = state_dir() / "schema.json"
432
+ schema_file.write_text(json.dumps(SCHEMA))
433
+
434
+ env = dict(os.environ)
435
+ env["S4L_REPO_DIR"] = str(REPO_DIR) # route through THIS repo's claude_job.py
436
+ cmd = [
437
+ "bash",
438
+ str(REPO_DIR / "scripts" / "run_claude.sh"),
439
+ "context-mining",
440
+ "--json-schema",
441
+ str(schema_file),
442
+ "-p",
443
+ ]
444
+ try:
445
+ proc = subprocess.run(
446
+ cmd,
447
+ input=prompt,
448
+ capture_output=True,
449
+ text=True,
450
+ env=env,
451
+ timeout=ns.timeout + 120,
452
+ )
453
+ except subprocess.TimeoutExpired:
454
+ print("[gather] hard timeout waiting for the queue result", file=sys.stderr)
455
+ return None
456
+
457
+ if proc.returncode == 79:
458
+ print(
459
+ "[gather] queue timed out (rc=79): no worker claimed the job. Is Claude "
460
+ "Desktop open and the s4l-worker task firing?",
461
+ file=sys.stderr,
462
+ )
463
+ return None
464
+ if proc.returncode != 0:
465
+ print(f"[gather] provider failed rc={proc.returncode}", file=sys.stderr)
466
+ sys.stderr.write((proc.stderr or "")[-2000:] + "\n")
467
+ return None
468
+
469
+ try:
470
+ envelope = json.loads(proc.stdout[proc.stdout.index("{"):])
471
+ obj = envelope.get("structured_output")
472
+ if obj is None:
473
+ obj = json.loads(envelope.get("result") or "{}")
474
+ except Exception as e:
475
+ print(f"[gather] could not parse result envelope: {e}", file=sys.stderr)
476
+ sys.stderr.write((proc.stdout or "")[-2000:] + "\n")
477
+ return None
478
+
479
+ stamped = []
480
+ for p in obj.get("proposals") or []:
481
+ key = (p.get("title") or p.get("text") or "") + p.get("source_session", "")
482
+ pid = "cm-" + hashlib.sha1(key.encode()).hexdigest()[:8]
483
+ p["id"] = pid
484
+ p["mined_at"] = datetime.now(timezone.utc).isoformat(timespec="seconds")
485
+ stamped.append(p)
486
+ return stamped
487
+
488
+
489
+ def run_gather(ns) -> int:
490
+ sessions = gather_sessions(ns.days)
491
+ if not sessions:
492
+ print(json.dumps({"ok": False, "reason": "no_eligible_sessions", "days": ns.days}))
493
+ return 1
494
+ batches = chunk_by_budget(sessions, ns.budget)
495
+ total_chars = sum(len(s["body"]) for s in sessions)
496
+ print(
497
+ f"[gather] {len(sessions)} sessions in window ({ns.days}d), "
498
+ f"{total_chars:,} chars -> {len(batches)} batches (budget {ns.budget:,})",
499
+ file=sys.stderr,
500
+ )
501
+
502
+ if ns.dry_run:
503
+ out = state_dir() / "last_prompt.txt"
504
+ out.write_text(build_prompt(batches[0], []))
505
+ for i, b in enumerate(batches, 1):
506
+ ids = ", ".join(s["session_id"][:8] for s in b)
507
+ print(f"[gather] batch {i}/{len(batches)}: {len(b)} sessions ({ids})", file=sys.stderr)
508
+ print(f"[gather] dry-run: batch-1 prompt written to {out}, nothing enqueued", file=sys.stderr)
509
+ return 0
510
+
511
+ # carry over any still-pending proposals so batches dedup against them too
512
+ all_props: list[dict] = _load_pending()
513
+ known_ids = {p["id"] for p in all_props}
514
+ failed = 0
515
+ for i, batch in enumerate(batches, 1):
516
+ ids = ", ".join(s["session_id"][:8] for s in batch)
517
+ print(
518
+ f"[gather] batch {i}/{len(batches)}: {len(batch)} sessions "
519
+ f"({sum(len(s['body']) for s in batch):,} chars) [{ids}]",
520
+ file=sys.stderr,
521
+ )
522
+ stamped = _run_one_batch(batch, all_props, ns)
523
+ if stamped is None:
524
+ failed += 1
525
+ print(f"[gather] batch {i} failed; stopping loop, keeping prior results", file=sys.stderr)
526
+ break
527
+ fresh = [p for p in stamped if p["id"] not in known_ids]
528
+ known_ids.update(p["id"] for p in fresh)
529
+ all_props.extend(fresh)
530
+ pending_path().write_text(
531
+ json.dumps({"proposals": all_props}, indent=2, ensure_ascii=False)
532
+ )
533
+ print(f"[gather] batch {i} done: +{len(fresh)} proposals ({len(all_props)} total)", file=sys.stderr)
534
+
535
+ print(
536
+ f"[gather] run complete: {len(batches) - failed}/{len(batches)} batches, "
537
+ f"{len(all_props)} pending proposals -> {pending_path()}",
538
+ file=sys.stderr,
539
+ )
540
+ cmd_review(None)
541
+ return 0 if not failed else 1
542
+
543
+
544
+ # ------------------------------------------------------------------- review
545
+
546
+
547
+ def _load_pending() -> list[dict]:
548
+ if not pending_path().exists():
549
+ return []
550
+ try:
551
+ return json.loads(pending_path().read_text()).get("proposals", [])
552
+ except Exception:
553
+ return []
554
+
555
+
556
+ def cmd_review(_ns) -> int:
557
+ props = _load_pending()
558
+ if not props:
559
+ print("no pending proposals")
560
+ return 0
561
+ corpus_lines = read_corpus_lines()
562
+ for p in props:
563
+ print("=" * 72)
564
+ head = p["id"]
565
+ if p.get("action") == "revise" and p.get("revises_line"):
566
+ n = p["revises_line"]
567
+ old = corpus_lines[n - 1] if 0 < n <= len(corpus_lines) else "(missing line)"
568
+ head += f" REVISES [{n}]: {old[:120]}"
569
+ else:
570
+ head += " ADD"
571
+ print(head)
572
+ if p.get("title"):
573
+ print(f" title: {p['title']}")
574
+ print(f" story: {p.get('story', '')}")
575
+ print(f" concl: {p.get('conclusion', '')}")
576
+ else: # legacy one-line shape
577
+ print(f" text : {p.get('text', '')}")
578
+ print(f" why : {p.get('why', '')}")
579
+ print(f" src : {p.get('source_session', '?')} @ {p.get('source_date', '?')}")
580
+ print("=" * 72)
581
+ print(f"{len(props)} pending. approve/skip with: context_mining.py approve <id...>")
582
+ return 0
583
+
584
+
585
+ def _decide(ids: list[str], status: str) -> int:
586
+ props = _load_pending()
587
+ if not props:
588
+ print("no pending proposals")
589
+ return 1
590
+ if ids == ["all"]:
591
+ ids = [p["id"] for p in props]
592
+ by_id = {p["id"]: p for p in props}
593
+ corpus_lines = read_corpus_lines()
594
+ done = []
595
+ for pid in ids:
596
+ p = by_id.get(pid)
597
+ if not p:
598
+ print(f"unknown id: {pid}")
599
+ continue
600
+ if status == "approved":
601
+ entry = corpus_entry(p) if p.get("title") else p.get("text", "")
602
+ if p.get("action") == "revise" and p.get("revises_line"):
603
+ n = p["revises_line"]
604
+ if 0 < n <= len(corpus_lines):
605
+ corpus_lines[n - 1] = entry
606
+ else:
607
+ corpus_lines.append(entry)
608
+ else:
609
+ corpus_lines.append(entry)
610
+ append_ledger(
611
+ {
612
+ "id": pid,
613
+ "status": status,
614
+ "title": p.get("title", ""),
615
+ "conclusion": p.get("conclusion", ""),
616
+ "text": p.get("text", ""),
617
+ "source_session": p.get("source_session"),
618
+ "decided_at": datetime.now(timezone.utc).isoformat(timespec="seconds"),
619
+ }
620
+ )
621
+ done.append(pid)
622
+ if status == "approved":
623
+ write_corpus_lines(corpus_lines)
624
+ remaining = [p for p in props if p["id"] not in set(done)]
625
+ pending_path().write_text(
626
+ json.dumps({"proposals": remaining}, indent=2, ensure_ascii=False)
627
+ )
628
+ print(f"{status}: {', '.join(done) if done else 'nothing'}; {len(remaining)} still pending")
629
+ if status == "approved":
630
+ print(f"corpus now has {len(corpus_lines)} lines: {corpus_path()}")
631
+ return 0
632
+
633
+
634
+ # ---------------------------------------------------------------------- CLI
635
+
636
+
637
+ def main() -> int:
638
+ ap = argparse.ArgumentParser(description="mine Claude transcripts into a context corpus")
639
+ sub = ap.add_subparsers(dest="cmd", required=True)
640
+
641
+ g = sub.add_parser("gather", help="collect sessions, enqueue one mining job, save proposals")
642
+ g.add_argument("--days", type=int, default=3)
643
+ g.add_argument("--budget", type=int, default=DEFAULT_TOTAL_BUDGET)
644
+ g.add_argument("--timeout", type=int, default=1800)
645
+ g.add_argument("--dry-run", action="store_true", help="build the prompt, don't enqueue")
646
+ g.set_defaults(func=run_gather)
647
+
648
+ r = sub.add_parser("review", help="print pending proposals")
649
+ r.set_defaults(func=cmd_review)
650
+
651
+ a = sub.add_parser("approve", help="apply proposal(s) to the corpus")
652
+ a.add_argument("ids", nargs="+")
653
+ a.set_defaults(func=lambda ns: _decide(ns.ids, "approved"))
654
+
655
+ s = sub.add_parser("skip", help="reject proposal(s), never re-pitched")
656
+ s.add_argument("ids", nargs="+")
657
+ s.set_defaults(func=lambda ns: _decide(ns.ids, "skipped"))
658
+
659
+ ns = ap.parse_args()
660
+ return ns.func(ns)
661
+
662
+
663
+ if __name__ == "__main__":
664
+ sys.exit(main())
@@ -101,7 +101,9 @@ def _atomic_write(path: str, obj) -> None:
101
101
  os.makedirs(os.path.dirname(path), exist_ok=True)
102
102
  tmp = f"{path}.tmp.{os.getpid()}"
103
103
  with open(tmp, "w") as f:
104
- json.dump(obj, f, indent=2)
104
+ # Compact separators: the review store reached 65 MB with indent=2
105
+ # (2026-08-02 lag incident) and every reader pays the parse.
106
+ json.dump(obj, f, separators=(",", ":"))
105
107
  os.replace(tmp, path)
106
108
 
107
109