@m13v/s4l 1.7.7-rc.6 → 1.7.7-rc.8
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/mcp/dist/repo.js +3 -1
- package/mcp/dist/version.json +2 -2
- package/mcp/manifest.json +1 -1
- package/mcp/menubar/s4l_card.py +9 -1
- package/mcp/menubar/s4l_card_canvas.py +29 -26
- package/mcp/menubar/s4l_menubar.py +38 -0
- package/mcp/menubar/s4l_state.py +118 -5
- package/mcp/package.json +1 -1
- package/package.json +1 -1
- package/scripts/browser_lifecycle.py +30 -0
- package/scripts/context_mining.py +49 -23
- package/scripts/harness_target_monitor.py +105 -0
- package/scripts/merge_review_queue.py +3 -1
- package/scripts/reddit_browser.py +9 -3
- package/scripts/reddit_browser_fetch.py +5 -1
- package/scripts/reddit_tools.py +50 -98
- package/scripts/stats.py +26 -53
- package/scripts/twitter_browser.py +8 -2
package/mcp/dist/repo.js
CHANGED
|
@@ -211,7 +211,9 @@ export function readPlan(batchId) {
|
|
|
211
211
|
}
|
|
212
212
|
}
|
|
213
213
|
export function writePlan(batchId, plan) {
|
|
214
|
-
|
|
214
|
+
// Compact (no indent): the review store reached 65 MB with pretty-printing
|
|
215
|
+
// (2026-08-02 lag incident) and every reader pays the parse.
|
|
216
|
+
fs.writeFileSync(planPath(batchId), JSON.stringify(plan), "utf-8");
|
|
215
217
|
}
|
|
216
218
|
// Find the newest plan file when no batch id is supplied.
|
|
217
219
|
export function latestBatchId() {
|
package/mcp/dist/version.json
CHANGED
package/mcp/manifest.json
CHANGED
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
"dxt_version": "0.1",
|
|
3
3
|
"name": "social-autoposter",
|
|
4
4
|
"display_name": "S4L",
|
|
5
|
-
"version": "1.7.7-rc.
|
|
5
|
+
"version": "1.7.7-rc.8",
|
|
6
6
|
"description": "Draft, review, approve, and autopilot X/Twitter posts.",
|
|
7
7
|
"long_description": "## **⚠️ The disclaimer above is generic Claude boilerplate.** Anthropic shows the same warning on every plugin regardless of what it does; any plugin has the same level of access as any app you download from the internet.\n\nS4L is an open source product developed by Mediar.ai Incorporated, a VC-backed San Francisco-based startup.\n\nTo get started:\n\n1\\. Copy this prompt: **Set me up on S4L plugin end to end**\n\n2\\. Quit with CMD+Q, reopen Claude, paste into a new chat.\n\nWhat happens next:\n\n* About every 5 minutes S4L scans X for posts that match your topics and drafts replies in your voice.\n* Drafts show up as review cards, usually the first within a few minutes. Nothing is posted automatically; you approve each one.\n* Posting autopilot stays off until you explicitly turn it on.",
|
|
8
8
|
"author": {
|
package/mcp/menubar/s4l_card.py
CHANGED
|
@@ -1964,7 +1964,12 @@ class _ReviewController(NSObject):
|
|
|
1964
1964
|
"""NSTimer target (2026-07-15): re-renders the header's age/expiry
|
|
1965
1965
|
label every second so its countdown visibly counts down without
|
|
1966
1966
|
needing hover. Not a python_method -- NSTimer invokes this through
|
|
1967
|
-
the ObjC runtime.
|
|
1967
|
+
the ObjC runtime. No-op when the rendered text is unchanged: the
|
|
1968
|
+
label usually shows a coarse "3h"-style value that only changes
|
|
1969
|
+
every few minutes, and unconditionally re-styling it dirtied one
|
|
1970
|
+
layer per tile per second -- with a 130-tile canvas that was a
|
|
1971
|
+
constant CoreAnimation commit churn keeping the app at ~15% CPU
|
|
1972
|
+
while idle (2026-08-02 lag incident)."""
|
|
1968
1973
|
if self._age_expiry_label is None:
|
|
1969
1974
|
return
|
|
1970
1975
|
try:
|
|
@@ -1975,6 +1980,9 @@ class _ReviewController(NSObject):
|
|
|
1975
1980
|
)
|
|
1976
1981
|
if not text:
|
|
1977
1982
|
return
|
|
1983
|
+
if (text, urgent) == getattr(self, "_age_expiry_last", None):
|
|
1984
|
+
return
|
|
1985
|
+
self._age_expiry_last = (text, urgent)
|
|
1978
1986
|
self._age_expiry_label.setStringValue_(text)
|
|
1979
1987
|
self._age_expiry_label.setFont_(_font(11, urgent))
|
|
1980
1988
|
self._age_expiry_label.setTextColor_(
|
|
@@ -461,14 +461,13 @@ class _CanvasController(NSObject):
|
|
|
461
461
|
|
|
462
462
|
@objc.python_method
|
|
463
463
|
def _reflow_from(self, start_idx):
|
|
464
|
-
"""
|
|
465
|
-
|
|
466
|
-
|
|
467
|
-
|
|
468
|
-
|
|
469
|
-
|
|
470
|
-
|
|
471
|
-
the reviewer hadn't yet acted on."""
|
|
464
|
+
"""Build every slot's CONTENT from start_idx onward from the current
|
|
465
|
+
self._order. Only two callers: the initial grid build (start 0) and
|
|
466
|
+
extend_drafts (start = old length, so only the NEW empty slots get
|
|
467
|
+
content). Decisions no longer come through here -- rebuilding ~130
|
|
468
|
+
full tile views per click was the dominant cost of an approval on a
|
|
469
|
+
big backlog (2026-08-02 lag incident, second act); _remove_and_reflow
|
|
470
|
+
now MOVES the surviving slot views instead."""
|
|
472
471
|
for i in range(start_idx, len(self._slots)):
|
|
473
472
|
slot = self._slots[i]
|
|
474
473
|
for sv in list(slot["view"].subviews()):
|
|
@@ -476,8 +475,8 @@ class _CanvasController(NSObject):
|
|
|
476
475
|
d = self._order[i]
|
|
477
476
|
tile = _ReviewController.alloc().initWithDrafts_onDecision_onComplete_focus_hostView_hostWindow_(
|
|
478
477
|
[d],
|
|
479
|
-
self._tile_decision_cb(
|
|
480
|
-
self._tile_complete_cb(
|
|
478
|
+
self._tile_decision_cb(slot),
|
|
479
|
+
self._tile_complete_cb(slot),
|
|
481
480
|
True,
|
|
482
481
|
slot["view"],
|
|
483
482
|
self._panel,
|
|
@@ -487,19 +486,23 @@ class _CanvasController(NSObject):
|
|
|
487
486
|
|
|
488
487
|
@objc.python_method
|
|
489
488
|
def _remove_and_reflow(self, n):
|
|
490
|
-
"""Pop draft `n` out of self._order, drop
|
|
491
|
-
|
|
492
|
-
|
|
493
|
-
|
|
489
|
+
"""Pop draft `n` out of self._order, drop ITS slot (view and all),
|
|
490
|
+
and shift every later slot's VIEW up into the vacated grid position
|
|
491
|
+
("snake" reflow, 2026-07-16 user direction). Positions move, content
|
|
492
|
+
doesn't: each surviving tile keeps its live view -- O(N) setFrame
|
|
493
|
+
calls instead of O(N) full tile rebuilds (2026-08-02 lag fix), and
|
|
494
|
+
an in-progress edit on a later card now survives earlier decisions
|
|
495
|
+
instead of being clobbered by the rebuild."""
|
|
494
496
|
try:
|
|
495
497
|
idx = next(i for i, d in enumerate(self._order) if d.get("n") == n)
|
|
496
498
|
except StopIteration:
|
|
497
499
|
return
|
|
498
500
|
self._order.pop(idx)
|
|
499
|
-
if self._slots:
|
|
500
|
-
|
|
501
|
-
|
|
502
|
-
self.
|
|
501
|
+
if idx < len(self._slots):
|
|
502
|
+
gone = self._slots.pop(idx)
|
|
503
|
+
gone["view"].removeFromSuperview()
|
|
504
|
+
for i in range(idx, len(self._slots)):
|
|
505
|
+
self._slots[i]["view"].setFrame_(self._slot_frame(i))
|
|
503
506
|
self._resize_doc()
|
|
504
507
|
self._refresh_header()
|
|
505
508
|
# Last card decided -> nothing left to review, so close the canvas
|
|
@@ -535,7 +538,7 @@ class _CanvasController(NSObject):
|
|
|
535
538
|
_log(f"canvas discard-all handler failed: {e}")
|
|
536
539
|
|
|
537
540
|
@objc.python_method
|
|
538
|
-
def _tile_decision_cb(self,
|
|
541
|
+
def _tile_decision_cb(self, slot):
|
|
539
542
|
def _cb(decision):
|
|
540
543
|
self._decisions.append(decision)
|
|
541
544
|
self._last_decision_at = time.time()
|
|
@@ -549,16 +552,16 @@ class _CanvasController(NSObject):
|
|
|
549
552
|
return _cb
|
|
550
553
|
|
|
551
554
|
@objc.python_method
|
|
552
|
-
def _tile_complete_cb(self,
|
|
555
|
+
def _tile_complete_cb(self, slot):
|
|
553
556
|
def _cb(_tile_decisions):
|
|
554
557
|
# The tile's own single-draft stack finished -- remove it from
|
|
555
558
|
# the ranking and let everything after it shift up ("snake"
|
|
556
|
-
# reflow; see _remove_and_reflow).
|
|
557
|
-
#
|
|
558
|
-
#
|
|
559
|
-
#
|
|
560
|
-
|
|
561
|
-
n = slot["n"]
|
|
559
|
+
# reflow; see _remove_and_reflow). Closing over the SLOT DICT
|
|
560
|
+
# (not a positional index) keeps the lookup correct no matter
|
|
561
|
+
# how many earlier slots have been popped since this closure
|
|
562
|
+
# was created: a slot keeps its `n` for life now that reflow
|
|
563
|
+
# moves views instead of rebuilding content.
|
|
564
|
+
n = slot["n"]
|
|
562
565
|
if n is not None:
|
|
563
566
|
self._remove_and_reflow(n)
|
|
564
567
|
|
|
@@ -27,6 +27,20 @@ import sys
|
|
|
27
27
|
import tempfile
|
|
28
28
|
import threading
|
|
29
29
|
import time
|
|
30
|
+
import warnings
|
|
31
|
+
|
|
32
|
+
# PyObjC emits an ObjCPointerWarning for every CGColor() handed to a CALayer
|
|
33
|
+
# (s4l_card's tile styling). The message embeds the pointer address, so the
|
|
34
|
+
# warnings module's once-per-location dedup never matches and a single card
|
|
35
|
+
# render sprayed ~100 lines/sec into menubar.err.log (13k+ lines by the
|
|
36
|
+
# 2026-08-02 lag incident). The pointers are handled correctly; silence the
|
|
37
|
+
# category before any card module loads.
|
|
38
|
+
try:
|
|
39
|
+
import objc
|
|
40
|
+
|
|
41
|
+
warnings.filterwarnings("ignore", category=objc.ObjCPointerWarning)
|
|
42
|
+
except Exception:
|
|
43
|
+
pass
|
|
30
44
|
|
|
31
45
|
# --- stderr timestamp wrapper -------------------------------------------------
|
|
32
46
|
# menubar.err.log (this process's stderr, redirected by the launchd plist) has
|
|
@@ -715,8 +729,28 @@ class S4LMenuBar(rumps.App):
|
|
|
715
729
|
self._tick_stats_at = 0.0
|
|
716
730
|
self._reloc_timer = rumps.Timer(self._maybe_relocate_tasks, 90)
|
|
717
731
|
self._reloc_timer.start()
|
|
732
|
+
# Store compaction: archive heavy fields off settled candidates so the
|
|
733
|
+
# review-queue file (and every 1s/5s poll that parses it) stays small.
|
|
734
|
+
# Boot pass runs off the main thread — the first pass after an
|
|
735
|
+
# un-compacted stretch can chew through tens of MB.
|
|
736
|
+
self._compacted_at = 0.0
|
|
737
|
+
self._compact_store_async()
|
|
718
738
|
self._tick(None)
|
|
719
739
|
|
|
740
|
+
def _compact_store_async(self):
|
|
741
|
+
self._compacted_at = time.time()
|
|
742
|
+
|
|
743
|
+
def run():
|
|
744
|
+
try:
|
|
745
|
+
n = st.compact_store()
|
|
746
|
+
if n:
|
|
747
|
+
sys.stderr.write(f"[s4l-menubar] store compaction archived {n} candidate(s)\n")
|
|
748
|
+
sys.stderr.flush()
|
|
749
|
+
except Exception:
|
|
750
|
+
pass
|
|
751
|
+
|
|
752
|
+
threading.Thread(target=run, daemon=True, name="s4l-compact-store").start()
|
|
753
|
+
|
|
720
754
|
# ---- side effects -----------------------------------------------------
|
|
721
755
|
def _open_claude(self, _=None):
|
|
722
756
|
subprocess.run(["open", "-a", CLAUDE_APP], capture_output=True,
|
|
@@ -3009,6 +3043,10 @@ class S4LMenuBar(rumps.App):
|
|
|
3009
3043
|
sys.stderr.flush()
|
|
3010
3044
|
except Exception:
|
|
3011
3045
|
pass
|
|
3046
|
+
# Hourly store compaction keeps the review-queue file small as posts
|
|
3047
|
+
# settle (boot already ran one pass; see _compact_store_async).
|
|
3048
|
+
if now_rc - self._compacted_at >= 3600:
|
|
3049
|
+
self._compact_store_async()
|
|
3012
3050
|
|
|
3013
3051
|
# ---- draft review pop-ups ---------------------------------------------
|
|
3014
3052
|
def _posting_activity_label_locked(self):
|
package/mcp/menubar/s4l_state.py
CHANGED
|
@@ -721,7 +721,9 @@ def _store_update(mutate):
|
|
|
721
721
|
data = {"candidates": []}
|
|
722
722
|
rv = mutate(data)
|
|
723
723
|
tmp = f"{sp}.tmp.{os.getpid()}"
|
|
724
|
-
|
|
724
|
+
# Compact separators: this file reached 65 MB with indent=2
|
|
725
|
+
# (2026-08-02 lag incident) and every reader pays the parse.
|
|
726
|
+
Path(tmp).write_text(json.dumps(data, separators=(",", ":")))
|
|
725
727
|
os.replace(tmp, sp)
|
|
726
728
|
return rv
|
|
727
729
|
finally:
|
|
@@ -779,6 +781,97 @@ def candidate_state(c):
|
|
|
779
781
|
return "awaiting_review"
|
|
780
782
|
|
|
781
783
|
|
|
784
|
+
# Draft-time payloads that are only needed while a candidate can still POST
|
|
785
|
+
# (awaiting_review: card rendering; approved/post_failed: the drain path
|
|
786
|
+
# rebuilds a mini-plan from reddit_decision/reddit_plan_meta). Once a candidate
|
|
787
|
+
# is posted or terminal these blobs are dead weight — and they dominated the
|
|
788
|
+
# 65 MB store that caused the 2026-08-02 UI lag (single reddit candidates
|
|
789
|
+
# carried ~180 KB). Compaction moves them to an append-only sidecar so the
|
|
790
|
+
# append-forever ledger keeps every byte while the hot file stays small.
|
|
791
|
+
HEAVY_ARCHIVE_FIELDS = ("reddit_plan_meta", "reddit_decision", "thread_selftext")
|
|
792
|
+
|
|
793
|
+
# Audit-only blobs inside reddit_plan_meta.style_assignment, stamped by
|
|
794
|
+
# engagement_styles.pick_style_for_post "for audit" and copied onto EVERY
|
|
795
|
+
# reddit candidate by merge_review_queue. distribution_snapshot alone was
|
|
796
|
+
# ~107 KB per candidate (7.6 MB across one pending backlog). Nothing on the
|
|
797
|
+
# card-render or posting path reads them (post_reddit.py --phase post uses
|
|
798
|
+
# only .style/.mode; index.ts forwards the dict opaquely), so they are safe
|
|
799
|
+
# to archive off candidates in ANY state, pending included.
|
|
800
|
+
STYLE_AUDIT_FIELDS = ("distribution_snapshot", "reference_styles")
|
|
801
|
+
|
|
802
|
+
ARCHIVE_STORE = "review-queue-archive.jsonl"
|
|
803
|
+
|
|
804
|
+
|
|
805
|
+
def compact_store():
|
|
806
|
+
"""Archive heavy payloads out of the hot store into review-queue-archive
|
|
807
|
+
.jsonl, then rewrite the store compactly. Two passes over the candidates:
|
|
808
|
+
HEAVY_ARCHIVE_FIELDS come off SETTLED (posted/terminal) candidates only —
|
|
809
|
+
approved/post_failed rows still need reddit_decision/reddit_plan_meta for
|
|
810
|
+
a (re-)approval drain — while STYLE_AUDIT_FIELDS come off every candidate.
|
|
811
|
+
Runs under the same store lock as every other python writer. The archive
|
|
812
|
+
lines land (flush+fsync) BEFORE any field is stripped, so no data is ever
|
|
813
|
+
lost — a crash in between only risks a duplicate archive line, never a
|
|
814
|
+
missing one. Returns the number of candidates compacted (0 when there was
|
|
815
|
+
nothing to do), None on failure."""
|
|
816
|
+
|
|
817
|
+
ap = str(Path(state_dir()) / ARCHIVE_STORE)
|
|
818
|
+
|
|
819
|
+
def mutate(data):
|
|
820
|
+
records = [] # (candidate, archived-fields dict, strip callback)
|
|
821
|
+
for c in data.get("candidates") or []:
|
|
822
|
+
fields = {}
|
|
823
|
+
settled = candidate_state(c) in ("posted", "terminal")
|
|
824
|
+
if settled:
|
|
825
|
+
fields.update(
|
|
826
|
+
{k: c[k] for k in HEAVY_ARCHIVE_FIELDS if c.get(k)}
|
|
827
|
+
)
|
|
828
|
+
sa = (c.get("reddit_plan_meta") or {}).get("style_assignment")
|
|
829
|
+
audit = (
|
|
830
|
+
{}
|
|
831
|
+
if (settled and "reddit_plan_meta" in fields) or not isinstance(sa, dict)
|
|
832
|
+
else {k: sa[k] for k in STYLE_AUDIT_FIELDS if sa.get(k)}
|
|
833
|
+
)
|
|
834
|
+
if audit:
|
|
835
|
+
fields["style_assignment_audit"] = audit
|
|
836
|
+
|
|
837
|
+
def strip(c=c, settled=settled, sa=sa, audit=audit):
|
|
838
|
+
if settled:
|
|
839
|
+
for k in HEAVY_ARCHIVE_FIELDS:
|
|
840
|
+
c.pop(k, None)
|
|
841
|
+
for k in audit:
|
|
842
|
+
sa.pop(k, None)
|
|
843
|
+
|
|
844
|
+
if fields:
|
|
845
|
+
records.append((c, fields, strip))
|
|
846
|
+
if not records:
|
|
847
|
+
return 0
|
|
848
|
+
with open(ap, "a") as f:
|
|
849
|
+
for c, fields, _ in records:
|
|
850
|
+
f.write(
|
|
851
|
+
json.dumps(
|
|
852
|
+
{
|
|
853
|
+
"candidate_id": c.get("candidate_id"),
|
|
854
|
+
"our_url": c.get("our_url"),
|
|
855
|
+
"thread_url": c.get("thread_url"),
|
|
856
|
+
"archived_at": time_iso(),
|
|
857
|
+
"fields": fields,
|
|
858
|
+
},
|
|
859
|
+
separators=(",", ":"),
|
|
860
|
+
)
|
|
861
|
+
+ "\n"
|
|
862
|
+
)
|
|
863
|
+
f.flush()
|
|
864
|
+
os.fsync(f.fileno())
|
|
865
|
+
for c, fields, strip in records:
|
|
866
|
+
strip()
|
|
867
|
+
c["archived_fields"] = sorted(
|
|
868
|
+
set(c.get("archived_fields") or []) | set(fields)
|
|
869
|
+
)
|
|
870
|
+
return len(records)
|
|
871
|
+
|
|
872
|
+
return _store_update(mutate)
|
|
873
|
+
|
|
874
|
+
|
|
782
875
|
def store_stamp_decision(batch, decision):
|
|
783
876
|
"""Write a card decision INTO the store the instant the user clicks. This is
|
|
784
877
|
the durable record (the old approved-queue.json ledger is no longer
|
|
@@ -1001,13 +1094,22 @@ def store_reconcile_decisions(batch, decisions):
|
|
|
1001
1094
|
return fixed
|
|
1002
1095
|
|
|
1003
1096
|
|
|
1097
|
+
# (path, mtime_ns, size) -> posted count. review_queue_posted_count() is on the
|
|
1098
|
+
# menubar's 1-second activity poll; without this cache that poll re-parsed the
|
|
1099
|
+
# whole store every tick, which saturated the AppKit main thread once the file
|
|
1100
|
+
# grew (2026-08-02: 65 MB, 98% CPU, multi-second card lag).
|
|
1101
|
+
_posted_count_cache = {"key": None, "count": None}
|
|
1102
|
+
|
|
1103
|
+
|
|
1004
1104
|
def review_queue_posted_count():
|
|
1005
1105
|
"""Posts that have LANDED in the review-queue plan — the durable, cross-process
|
|
1006
1106
|
truth. Independent of the menu bar's in-memory burst queue (which dies on a
|
|
1007
1107
|
restart) and of WHICH process is posting (the menu bar worker, the autopilot,
|
|
1008
1108
|
or a host agent draining via approve_drafts). Returns the posted count, or None
|
|
1009
1109
|
when the plan can't be read. Drives the menu-bar posting indicator so progress
|
|
1010
|
-
stays visible regardless of how the drain is driven.
|
|
1110
|
+
stays visible regardless of how the drain is driven. Cached on the plan
|
|
1111
|
+
file's (mtime, size): the caller polls every second, the file changes only
|
|
1112
|
+
when a writer lands."""
|
|
1011
1113
|
plan_path = None
|
|
1012
1114
|
req = read_review_request()
|
|
1013
1115
|
if req:
|
|
@@ -1016,11 +1118,22 @@ def review_queue_posted_count():
|
|
|
1016
1118
|
plan_path = store_path()
|
|
1017
1119
|
if not Path(plan_path).exists():
|
|
1018
1120
|
plan_path = "/tmp/twitter_cycle_plan_review-queue.json"
|
|
1121
|
+
try:
|
|
1122
|
+
stt = os.stat(plan_path)
|
|
1123
|
+
cache_key = (os.path.realpath(plan_path), stt.st_mtime_ns, stt.st_size)
|
|
1124
|
+
except OSError:
|
|
1125
|
+
cache_key = None
|
|
1126
|
+
if cache_key is not None and _posted_count_cache["key"] == cache_key:
|
|
1127
|
+
return _posted_count_cache["count"]
|
|
1019
1128
|
plan = read_plan(plan_path)
|
|
1020
1129
|
cands = (plan or {}).get("candidates")
|
|
1021
|
-
if not cands
|
|
1022
|
-
|
|
1023
|
-
|
|
1130
|
+
count = None if not cands else sum(
|
|
1131
|
+
1 for c in cands if candidate_state(c) == "posted"
|
|
1132
|
+
)
|
|
1133
|
+
if cache_key is not None:
|
|
1134
|
+
_posted_count_cache["key"] = cache_key
|
|
1135
|
+
_posted_count_cache["count"] = count
|
|
1136
|
+
return count
|
|
1024
1137
|
|
|
1025
1138
|
|
|
1026
1139
|
def review_drafts(plan, batch="review-queue"):
|
package/mcp/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@m13v/s4l-mcp",
|
|
3
|
-
"version": "1.7.7-rc.
|
|
3
|
+
"version": "1.7.7-rc.8",
|
|
4
4
|
"private": true,
|
|
5
5
|
"description": "Desktop MCP client for social-autoposter (X/Twitter rail): manual draft/review/approve loop, autopilot control, and stats. Thin wrapper over the existing pipeline scripts.",
|
|
6
6
|
"license": "MIT",
|
package/package.json
CHANGED
|
@@ -77,6 +77,36 @@ def park_tabs(cdp_base: str, host_markers, park_url: str, label: str) -> None:
|
|
|
77
77
|
pass
|
|
78
78
|
|
|
79
79
|
|
|
80
|
+
def background_new_page(browser, context, url: str = "about:blank", timeout_ms: int = 10_000):
|
|
81
|
+
"""Create a Playwright page WITHOUT raising the Chrome window.
|
|
82
|
+
|
|
83
|
+
Playwright's context.new_page() maps to a FOREGROUND Target.createTarget,
|
|
84
|
+
which activates Chrome on macOS and steals app focus (the July 2026
|
|
85
|
+
focus-steal class; same call the bh helpers and the harness daemon were
|
|
86
|
+
already fixed to avoid). Playwright's public API has no background
|
|
87
|
+
option, so this creates the target via a browser-level CDP session with
|
|
88
|
+
background:true and returns the Page that `context` adopts for it.
|
|
89
|
+
|
|
90
|
+
ONE shared implementation for every attach path that previously called
|
|
91
|
+
context.new_page() on the harness (reddit_browser, twitter_browser,
|
|
92
|
+
reddit_browser_fetch). Falls back to context.new_page() if the CDP path
|
|
93
|
+
fails: a rare focus blip beats a dead pipeline.
|
|
94
|
+
"""
|
|
95
|
+
try:
|
|
96
|
+
cdp = browser.new_browser_cdp_session()
|
|
97
|
+
try:
|
|
98
|
+
with context.expect_page(timeout=timeout_ms) as pg_info:
|
|
99
|
+
cdp.send("Target.createTarget", {"url": url, "background": True})
|
|
100
|
+
return pg_info.value
|
|
101
|
+
finally:
|
|
102
|
+
try:
|
|
103
|
+
cdp.detach()
|
|
104
|
+
except Exception:
|
|
105
|
+
pass
|
|
106
|
+
except Exception:
|
|
107
|
+
return context.new_page()
|
|
108
|
+
|
|
109
|
+
|
|
80
110
|
def register_park_on_exit(cdp_base: str, host_markers, park_url: str, label: str) -> None:
|
|
81
111
|
"""Arm park_tabs to run at process exit, once per (endpoint, park_url).
|
|
82
112
|
Call from a platform lib's get_browser_and_page so only processes that
|
|
@@ -304,13 +304,20 @@ SCHEMA = {
|
|
|
304
304
|
"type": "array",
|
|
305
305
|
"items": {
|
|
306
306
|
"type": "object",
|
|
307
|
-
"required": [
|
|
307
|
+
"required": [
|
|
308
|
+
"action",
|
|
309
|
+
"title",
|
|
310
|
+
"story",
|
|
311
|
+
"conclusion",
|
|
312
|
+
"source_session",
|
|
313
|
+
"source_date",
|
|
314
|
+
],
|
|
308
315
|
"properties": {
|
|
309
316
|
"action": {"type": "string", "enum": ["add", "revise"]},
|
|
310
317
|
"revises_line": {"type": ["integer", "null"]},
|
|
311
|
-
"
|
|
312
|
-
"
|
|
313
|
-
"
|
|
318
|
+
"title": {"type": "string"},
|
|
319
|
+
"story": {"type": "string"},
|
|
320
|
+
"conclusion": {"type": "string"},
|
|
314
321
|
"source_session": {"type": "string"},
|
|
315
322
|
"source_date": {"type": "string"},
|
|
316
323
|
},
|
|
@@ -320,6 +327,15 @@ SCHEMA = {
|
|
|
320
327
|
}
|
|
321
328
|
|
|
322
329
|
|
|
330
|
+
def corpus_entry(p: dict) -> str:
|
|
331
|
+
"""Render one approved proposal as a single corpus line (numbering is
|
|
332
|
+
line-based, so the entry must not contain newlines)."""
|
|
333
|
+
story = " ".join((p.get("story") or "").split())
|
|
334
|
+
concl = " ".join((p.get("conclusion") or "").split())
|
|
335
|
+
title = " ".join((p.get("title") or "").split())
|
|
336
|
+
return f"{title}: {story} Conclusion: {concl}"
|
|
337
|
+
|
|
338
|
+
|
|
323
339
|
def build_prompt(sessions: list[dict], pending_props: list[dict] | None = None) -> str:
|
|
324
340
|
corpus_lines = read_corpus_lines()
|
|
325
341
|
corpus_block = (
|
|
@@ -327,11 +343,15 @@ def build_prompt(sessions: list[dict], pending_props: list[dict] | None = None)
|
|
|
327
343
|
if corpus_lines
|
|
328
344
|
else "(the corpus is currently empty)"
|
|
329
345
|
)
|
|
346
|
+
def _gist(r: dict) -> str:
|
|
347
|
+
# works for both the old one-line shape (text) and the story shape
|
|
348
|
+
return r.get("text") or f"{r.get('title', '')}: {r.get('conclusion', '')}"
|
|
349
|
+
|
|
330
350
|
considered = [
|
|
331
|
-
{"status": r.get("status"), "
|
|
332
|
-
] + [{"status": "pending", "
|
|
351
|
+
{"status": r.get("status"), "gist": _gist(r)} for r in read_ledger()[-60:]
|
|
352
|
+
] + [{"status": "pending", "gist": _gist(p)} for p in (pending_props or [])]
|
|
333
353
|
considered_block = (
|
|
334
|
-
"\n".join(f"- ({r['status']}) {r['
|
|
354
|
+
"\n".join(f"- ({r['status']}) {r['gist'][:200]}" for r in considered)
|
|
335
355
|
if considered
|
|
336
356
|
else "(nothing has been considered yet)"
|
|
337
357
|
)
|
|
@@ -346,7 +366,7 @@ def build_prompt(sessions: list[dict], pending_props: list[dict] | None = None)
|
|
|
346
366
|
|
|
347
367
|
# Role
|
|
348
368
|
|
|
349
|
-
You are the daily context miner for a personal "context corpus": a numbered list of durable
|
|
369
|
+
You are the daily context miner for a personal "context corpus": a numbered list of durable memories distilled from the user's own Claude conversations. Each entry is a short STORY plus a CONCLUSION: what actually happened (with its credible specifics) and the transferable lesson it taught. Entries are worth keeping and potentially worth sharing with the world: a specific experience, a hard-won conclusion, a non-obvious observation, or a concrete data point.
|
|
350
370
|
|
|
351
371
|
The user is Matthew, a solo founder building S4L (a social-media autoposting agent), Fazm, Mediar, and related products, and doing everything else through Claude sessions: fundraising, debugging, ops, legal, marketing.
|
|
352
372
|
|
|
@@ -389,11 +409,12 @@ The user is Matthew, a solo founder building S4L (a social-media autoposting age
|
|
|
389
409
|
# Your task
|
|
390
410
|
|
|
391
411
|
Propose 0-3 corpus changes, and only what clears EVERY bar above; most days the
|
|
392
|
-
right answer is 0 or 1. An empty proposals list is a valid, common answer.
|
|
393
|
-
|
|
394
|
-
-
|
|
395
|
-
-
|
|
396
|
-
-
|
|
412
|
+
right answer is 0 or 1. An empty proposals list is a valid, common answer. Each
|
|
413
|
+
proposal is a small STORY with a CONCLUSION, not a bare aphorism:
|
|
414
|
+
- action: "add" for a new entry, or "revise" when a new learning supersedes an existing corpus entry (set revises_line to that entry's number).
|
|
415
|
+
- title: a short, specific handle for the insight (max ~70 chars, not clickbait).
|
|
416
|
+
- story: 2-5 sentences telling what actually happened, in the user's plainspoken first-person voice. Keep the concrete specifics that make it credible and vivid (the numbers, the timeline, the failure mode, what was tried), but generalize or omit anything only an insider would care about. Anonymize people, redact secrets. No hashtags, no em dashes.
|
|
417
|
+
- conclusion: 1-2 sentences stating the transferable lesson a stranger could apply to their own work. This is the part that must stand on its own.
|
|
397
418
|
- source_session: the 8-char session id from the transcript header.
|
|
398
419
|
- source_date: the session date (YYYY-MM-DD).
|
|
399
420
|
|
|
@@ -457,9 +478,8 @@ def _run_one_batch(sessions: list[dict], pending_props: list[dict], ns) -> list[
|
|
|
457
478
|
|
|
458
479
|
stamped = []
|
|
459
480
|
for p in obj.get("proposals") or []:
|
|
460
|
-
|
|
461
|
-
|
|
462
|
-
).hexdigest()[:8]
|
|
481
|
+
key = (p.get("title") or p.get("text") or "") + p.get("source_session", "")
|
|
482
|
+
pid = "cm-" + hashlib.sha1(key.encode()).hexdigest()[:8]
|
|
463
483
|
p["id"] = pid
|
|
464
484
|
p["mined_at"] = datetime.now(timezone.utc).isoformat(timespec="seconds")
|
|
465
485
|
stamped.append(p)
|
|
@@ -549,10 +569,13 @@ def cmd_review(_ns) -> int:
|
|
|
549
569
|
else:
|
|
550
570
|
head += " ADD"
|
|
551
571
|
print(head)
|
|
552
|
-
|
|
553
|
-
|
|
554
|
-
|
|
555
|
-
print(f"
|
|
572
|
+
if p.get("title"):
|
|
573
|
+
print(f" title: {p['title']}")
|
|
574
|
+
print(f" story: {p.get('story', '')}")
|
|
575
|
+
print(f" concl: {p.get('conclusion', '')}")
|
|
576
|
+
else: # legacy one-line shape
|
|
577
|
+
print(f" text : {p.get('text', '')}")
|
|
578
|
+
print(f" why : {p.get('why', '')}")
|
|
556
579
|
print(f" src : {p.get('source_session', '?')} @ {p.get('source_date', '?')}")
|
|
557
580
|
print("=" * 72)
|
|
558
581
|
print(f"{len(props)} pending. approve/skip with: context_mining.py approve <id...>")
|
|
@@ -575,18 +598,21 @@ def _decide(ids: list[str], status: str) -> int:
|
|
|
575
598
|
print(f"unknown id: {pid}")
|
|
576
599
|
continue
|
|
577
600
|
if status == "approved":
|
|
601
|
+
entry = corpus_entry(p) if p.get("title") else p.get("text", "")
|
|
578
602
|
if p.get("action") == "revise" and p.get("revises_line"):
|
|
579
603
|
n = p["revises_line"]
|
|
580
604
|
if 0 < n <= len(corpus_lines):
|
|
581
|
-
corpus_lines[n - 1] =
|
|
605
|
+
corpus_lines[n - 1] = entry
|
|
582
606
|
else:
|
|
583
|
-
corpus_lines.append(
|
|
607
|
+
corpus_lines.append(entry)
|
|
584
608
|
else:
|
|
585
|
-
corpus_lines.append(
|
|
609
|
+
corpus_lines.append(entry)
|
|
586
610
|
append_ledger(
|
|
587
611
|
{
|
|
588
612
|
"id": pid,
|
|
589
613
|
"status": status,
|
|
614
|
+
"title": p.get("title", ""),
|
|
615
|
+
"conclusion": p.get("conclusion", ""),
|
|
590
616
|
"text": p.get("text", ""),
|
|
591
617
|
"source_session": p.get("source_session"),
|
|
592
618
|
"decided_at": datetime.now(timezone.utc).isoformat(timespec="seconds"),
|
|
@@ -0,0 +1,105 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""Attribution monitor for harness-Chrome focus steals (2026-08-03).
|
|
3
|
+
|
|
4
|
+
Connects to a harness Chrome's browser-level CDP websocket, enables target
|
|
5
|
+
discovery, and appends one timestamped line per Target lifecycle event
|
|
6
|
+
(created / destroyed / info-changed => navigations) to a log file. Correlate
|
|
7
|
+
these against the `[browser-foreground]` activation lines in
|
|
8
|
+
~/.social-autoposter-mcp/menubar/menubar.err.log to attribute WHICH tab
|
|
9
|
+
operation coincided with an app activation, something none of the existing
|
|
10
|
+
logs capture (the daemon log has no timestamps; python new_page sites have
|
|
11
|
+
no logging at all; bh [bh_tab_event] covers only the bh lanes).
|
|
12
|
+
|
|
13
|
+
Read-only: never creates, closes, or navigates anything.
|
|
14
|
+
|
|
15
|
+
Usage:
|
|
16
|
+
harness_target_monitor.py [--port 9557] [--log PATH]
|
|
17
|
+
|
|
18
|
+
Runs forever; reconnects with backoff when Chrome restarts. Intended to run
|
|
19
|
+
under nohup during a diagnosis window. It is NOT part of the pipeline.
|
|
20
|
+
"""
|
|
21
|
+
import argparse
|
|
22
|
+
import json
|
|
23
|
+
import os
|
|
24
|
+
import sys
|
|
25
|
+
import time
|
|
26
|
+
from datetime import datetime, timezone
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def ts() -> str:
|
|
30
|
+
return datetime.now(timezone.utc).strftime("%Y-%m-%dT%H:%M:%S.%f")[:-3] + "Z"
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def log_line(path: str, msg: str) -> None:
|
|
34
|
+
with open(path, "a") as f:
|
|
35
|
+
f.write(f"[{ts()}] {msg}\n")
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def monitor_once(port: int, log_path: str) -> None:
|
|
39
|
+
import urllib.request
|
|
40
|
+
import websocket
|
|
41
|
+
|
|
42
|
+
opener = urllib.request.build_opener(urllib.request.ProxyHandler({}))
|
|
43
|
+
info = json.loads(opener.open(f"http://127.0.0.1:{port}/json/version", timeout=3).read())
|
|
44
|
+
ws = websocket.create_connection(
|
|
45
|
+
info["webSocketDebuggerUrl"], timeout=5, suppress_origin=True
|
|
46
|
+
)
|
|
47
|
+
try:
|
|
48
|
+
ws.send(json.dumps({"id": 1, "method": "Target.setDiscoverTargets",
|
|
49
|
+
"params": {"discover": True}}))
|
|
50
|
+
log_line(log_path, f"monitor attached port={port} chrome={info.get('Browser','?')}")
|
|
51
|
+
ws.settimeout(60)
|
|
52
|
+
last_url = {}
|
|
53
|
+
while True:
|
|
54
|
+
try:
|
|
55
|
+
msg = json.loads(ws.recv())
|
|
56
|
+
except Exception as e:
|
|
57
|
+
if "timed out" in str(e).lower():
|
|
58
|
+
# Idle is fine; poke the connection so a dead Chrome errors out.
|
|
59
|
+
ws.send(json.dumps({"id": 2, "method": "Browser.getVersion"}))
|
|
60
|
+
continue
|
|
61
|
+
raise
|
|
62
|
+
method = msg.get("method", "")
|
|
63
|
+
p = msg.get("params", {})
|
|
64
|
+
t = p.get("targetInfo", {})
|
|
65
|
+
if t.get("type") not in ("page", ""):
|
|
66
|
+
continue
|
|
67
|
+
tid = t.get("targetId") or p.get("targetId", "?")
|
|
68
|
+
url = t.get("url", "")
|
|
69
|
+
if method == "Target.targetCreated":
|
|
70
|
+
log_line(log_path, f"CREATED {tid} url={url[:100]}")
|
|
71
|
+
elif method == "Target.targetDestroyed":
|
|
72
|
+
log_line(log_path, f"DESTROYED {tid} (last_url={last_url.get(tid, '?')[:100]})")
|
|
73
|
+
elif method == "Target.targetInfoChanged":
|
|
74
|
+
if url and url != last_url.get(tid):
|
|
75
|
+
log_line(log_path, f"NAVIGATED {tid} url={url[:100]}")
|
|
76
|
+
if tid != "?" and url:
|
|
77
|
+
last_url[tid] = url
|
|
78
|
+
finally:
|
|
79
|
+
try:
|
|
80
|
+
ws.close()
|
|
81
|
+
except Exception:
|
|
82
|
+
pass
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
def main() -> int:
|
|
86
|
+
ap = argparse.ArgumentParser()
|
|
87
|
+
ap.add_argument("--port", type=int, default=9557)
|
|
88
|
+
ap.add_argument("--log", default=os.path.expanduser(
|
|
89
|
+
"~/social-autoposter/skill/logs/harness-target-events-9557.log"))
|
|
90
|
+
args = ap.parse_args()
|
|
91
|
+
while True:
|
|
92
|
+
try:
|
|
93
|
+
monitor_once(args.port, args.log)
|
|
94
|
+
except KeyboardInterrupt:
|
|
95
|
+
return 0
|
|
96
|
+
except Exception as e:
|
|
97
|
+
try:
|
|
98
|
+
log_line(args.log, f"monitor disconnected ({type(e).__name__}: {str(e)[:120]}); retry in 15s")
|
|
99
|
+
except OSError:
|
|
100
|
+
pass
|
|
101
|
+
time.sleep(15)
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
if __name__ == "__main__":
|
|
105
|
+
sys.exit(main())
|
|
@@ -101,7 +101,9 @@ def _atomic_write(path: str, obj) -> None:
|
|
|
101
101
|
os.makedirs(os.path.dirname(path), exist_ok=True)
|
|
102
102
|
tmp = f"{path}.tmp.{os.getpid()}"
|
|
103
103
|
with open(tmp, "w") as f:
|
|
104
|
-
|
|
104
|
+
# Compact separators: the review store reached 65 MB with indent=2
|
|
105
|
+
# (2026-08-02 lag incident) and every reader pays the parse.
|
|
106
|
+
json.dump(obj, f, separators=(",", ":"))
|
|
105
107
|
os.replace(tmp, path)
|
|
106
108
|
|
|
107
109
|
|
|
@@ -441,7 +441,11 @@ def _get_browser_and_page_raw(playwright):
|
|
|
441
441
|
return cdp_browser, pg, True
|
|
442
442
|
if chosen.pages:
|
|
443
443
|
return cdp_browser, chosen.pages[0], True
|
|
444
|
-
|
|
444
|
+
# Zero pages: create in the BACKGROUND (2026-08-03). A plain
|
|
445
|
+
# new_page() here is a foreground Target.createTarget, which
|
|
446
|
+
# activates Chrome and steals macOS app focus.
|
|
447
|
+
from browser_lifecycle import background_new_page
|
|
448
|
+
page = background_new_page(cdp_browser, chosen)
|
|
445
449
|
return cdp_browser, page, True
|
|
446
450
|
# No usable context: do NOT close the CDP browser (would kill the
|
|
447
451
|
# harness Chrome); just disconnect by falling through.
|
|
@@ -464,13 +468,15 @@ def _get_browser_and_page_raw(playwright):
|
|
|
464
468
|
for c in cookies
|
|
465
469
|
)
|
|
466
470
|
if has_session:
|
|
467
|
-
# Reuse an existing tab (no focus-steal); only
|
|
471
|
+
# Reuse an existing tab (no focus-steal); only create if none,
|
|
472
|
+
# and then in the BACKGROUND (2026-08-03, see above).
|
|
468
473
|
for pg in ctx.pages:
|
|
469
474
|
if "reddit.com" in (pg.url or "") and "login" not in (pg.url or ""):
|
|
470
475
|
return cdp_browser, pg, True
|
|
471
476
|
if ctx.pages:
|
|
472
477
|
return cdp_browser, ctx.pages[0], True
|
|
473
|
-
|
|
478
|
+
from browser_lifecycle import background_new_page
|
|
479
|
+
page = background_new_page(cdp_browser, ctx)
|
|
474
480
|
return cdp_browser, page, True
|
|
475
481
|
try:
|
|
476
482
|
cdp_browser.close()
|
|
@@ -165,7 +165,11 @@ def browser_get_json(url, cdp_url=None, timeout_ms=25000):
|
|
|
165
165
|
if page is None and ctx.pages:
|
|
166
166
|
page = ctx.pages[0]
|
|
167
167
|
if page is None:
|
|
168
|
-
|
|
168
|
+
# Zero pages: create in the BACKGROUND (2026-08-03). A plain
|
|
169
|
+
# new_page() is a foreground Target.createTarget, which
|
|
170
|
+
# activates Chrome and steals macOS app focus.
|
|
171
|
+
from browser_lifecycle import background_new_page
|
|
172
|
+
page = background_new_page(browser, ctx)
|
|
169
173
|
# Load the matching host root so the subsequent fetch() is same-origin
|
|
170
174
|
# (no CORS between www/old) and carries the logged-in session.
|
|
171
175
|
try:
|
package/scripts/reddit_tools.py
CHANGED
|
@@ -74,71 +74,61 @@ def _wait_if_needed():
|
|
|
74
74
|
def _fetch_via_browser(url):
|
|
75
75
|
"""Fetch a Reddit URL through the reddit-harness logged-in Chrome.
|
|
76
76
|
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
|
|
77
|
+
Browser-ONLY transport (2026-08-03, user decision: no urllib fallback).
|
|
78
|
+
Reddit's TLS-fingerprint wall (2026-05-28) 403s urllib/curl on *.json
|
|
79
|
+
unconditionally, so the old fallback could never succeed: it burned ~2s
|
|
80
|
+
per browser hiccup and then lost the query anyway. Transient failures
|
|
81
|
+
(Reddit 503s, harness contention) get ONE in-transport retry instead.
|
|
82
|
+
The REDDIT_FETCH_BACKEND=urllib debug knob is gone for the same reason:
|
|
83
|
+
forcing a transport that is guaranteed to 403 debugs nothing.
|
|
84
|
+
|
|
85
|
+
Returns the raw response body (str) on HTTP 200. Raises on final failure,
|
|
86
|
+
urllib.error.HTTPError for HTTP statuses and urllib.error.URLError for
|
|
87
|
+
transport-level failures, matching the exception shapes callers already
|
|
88
|
+
handle from the urllib era. 429 keeps the old inline-wait contract
|
|
89
|
+
(absorb a short wait, else RateLimitedError).
|
|
86
90
|
"""
|
|
87
|
-
|
|
88
|
-
|
|
89
|
-
|
|
90
|
-
|
|
91
|
-
except Exception as e:
|
|
92
|
-
sys.stderr.write(f"[reddit_tools] browser fetch unavailable ({e}); urllib fallback\n")
|
|
93
|
-
return None
|
|
94
|
-
try:
|
|
91
|
+
from reddit_browser_fetch import browser_get_json
|
|
92
|
+
|
|
93
|
+
last_status = 0
|
|
94
|
+
for attempt in (1, 2):
|
|
95
95
|
body, status = browser_get_json(url)
|
|
96
96
|
if status == 200 and body:
|
|
97
97
|
return body
|
|
98
|
-
|
|
99
|
-
|
|
100
|
-
|
|
101
|
-
|
|
98
|
+
last_status = status
|
|
99
|
+
if status == 429:
|
|
100
|
+
# Browser transport exposes no X-Ratelimit-Reset header; assume
|
|
101
|
+
# Reddit's standard 60s window.
|
|
102
|
+
_write_ratelimit(0, 60)
|
|
103
|
+
if 60 > MAX_INLINE_WAIT_SECONDS:
|
|
104
|
+
raise RateLimitedError(60)
|
|
105
|
+
if attempt == 1:
|
|
106
|
+
print("Rate limited. Waiting 62s...", file=sys.stderr)
|
|
107
|
+
time.sleep(62)
|
|
108
|
+
continue
|
|
109
|
+
sys.stderr.write(
|
|
110
|
+
f"[reddit_tools] browser fetch status={status} for {url[:80]} (attempt {attempt}/2)\n"
|
|
111
|
+
)
|
|
112
|
+
if attempt == 1:
|
|
113
|
+
time.sleep(2.5)
|
|
114
|
+
if last_status == 429:
|
|
115
|
+
raise RateLimitedError(60)
|
|
116
|
+
if last_status:
|
|
117
|
+
raise urllib.error.HTTPError(
|
|
118
|
+
url, last_status,
|
|
119
|
+
f"reddit browser fetch failed (status={last_status})", None, None,
|
|
120
|
+
)
|
|
121
|
+
raise urllib.error.URLError(f"reddit browser transport failed for {url[:80]}")
|
|
102
122
|
|
|
103
123
|
|
|
104
124
|
def _do_request(url):
|
|
105
|
-
"""Make a Reddit API request
|
|
125
|
+
"""Make a Reddit API request via the harness browser (sole transport).
|
|
106
126
|
|
|
107
|
-
|
|
108
|
-
|
|
109
|
-
immediately if the reset would require a long wait, else absorbs short waits.
|
|
127
|
+
429 handling (inline wait / RateLimitedError) lives inside
|
|
128
|
+
_fetch_via_browser; HTTP and transport failures raise there too.
|
|
110
129
|
"""
|
|
111
130
|
_wait_if_needed()
|
|
112
|
-
|
|
113
|
-
# if the harness is down or returns a non-200.
|
|
114
|
-
_body = _fetch_via_browser(url)
|
|
115
|
-
if _body is not None:
|
|
116
|
-
try:
|
|
117
|
-
return json.loads(_body)
|
|
118
|
-
except Exception:
|
|
119
|
-
sys.stderr.write(f"[reddit_tools] browser body not JSON for {url[:80]}; urllib fallback\n")
|
|
120
|
-
req = urllib.request.Request(url, headers={"User-Agent": USER_AGENT})
|
|
121
|
-
try:
|
|
122
|
-
resp = urllib.request.urlopen(req, timeout=20)
|
|
123
|
-
remaining = float(resp.headers.get("X-Ratelimit-Remaining", 100))
|
|
124
|
-
reset = float(resp.headers.get("X-Ratelimit-Reset", 0))
|
|
125
|
-
_write_ratelimit(remaining, reset)
|
|
126
|
-
return json.loads(resp.read())
|
|
127
|
-
except urllib.error.HTTPError as e:
|
|
128
|
-
if e.code == 429:
|
|
129
|
-
reset = float(e.headers.get("X-Ratelimit-Reset", 60))
|
|
130
|
-
_write_ratelimit(0, reset)
|
|
131
|
-
if reset > MAX_INLINE_WAIT_SECONDS:
|
|
132
|
-
raise RateLimitedError(reset)
|
|
133
|
-
print(f"Rate limited. Waiting {int(reset)+2}s...", file=sys.stderr)
|
|
134
|
-
time.sleep(int(reset) + 2)
|
|
135
|
-
# Retry once
|
|
136
|
-
resp = urllib.request.urlopen(req, timeout=20)
|
|
137
|
-
remaining = float(resp.headers.get("X-Ratelimit-Remaining", 100))
|
|
138
|
-
reset2 = float(resp.headers.get("X-Ratelimit-Reset", 0))
|
|
139
|
-
_write_ratelimit(remaining, reset2)
|
|
140
|
-
return json.loads(resp.read())
|
|
141
|
-
raise
|
|
131
|
+
return json.loads(_fetch_via_browser(url))
|
|
142
132
|
|
|
143
133
|
|
|
144
134
|
def batch_fetch_info(thing_ids, user_agent=USER_AGENT):
|
|
@@ -158,41 +148,9 @@ def batch_fetch_info(thing_ids, user_agent=USER_AGENT):
|
|
|
158
148
|
ids_str = ",".join(chunk)
|
|
159
149
|
url = f"https://old.reddit.com/api/info.json?id={ids_str}"
|
|
160
150
|
_wait_if_needed()
|
|
161
|
-
# Browser-
|
|
162
|
-
|
|
163
|
-
|
|
164
|
-
try:
|
|
165
|
-
data = json.loads(_body)
|
|
166
|
-
for child in data.get("data", {}).get("children", []):
|
|
167
|
-
cd = child.get("data", {})
|
|
168
|
-
name = cd.get("name")
|
|
169
|
-
if name:
|
|
170
|
-
results[name] = cd
|
|
171
|
-
continue
|
|
172
|
-
except Exception:
|
|
173
|
-
sys.stderr.write("[reddit_tools] browser info.json not JSON; urllib fallback\n")
|
|
174
|
-
req = urllib.request.Request(url, headers={"User-Agent": user_agent})
|
|
175
|
-
try:
|
|
176
|
-
resp = urllib.request.urlopen(req, timeout=30)
|
|
177
|
-
remaining = float(resp.headers.get("X-Ratelimit-Remaining", 100))
|
|
178
|
-
reset = float(resp.headers.get("X-Ratelimit-Reset", 0))
|
|
179
|
-
_write_ratelimit(remaining, reset)
|
|
180
|
-
data = json.loads(resp.read())
|
|
181
|
-
except urllib.error.HTTPError as e:
|
|
182
|
-
if e.code == 429:
|
|
183
|
-
reset = float(e.headers.get("X-Ratelimit-Reset", 60))
|
|
184
|
-
_write_ratelimit(0, reset)
|
|
185
|
-
if reset > MAX_INLINE_WAIT_SECONDS:
|
|
186
|
-
raise RateLimitedError(reset)
|
|
187
|
-
print(f"Rate limited. Waiting {int(reset)+2}s...", file=sys.stderr)
|
|
188
|
-
time.sleep(int(reset) + 2)
|
|
189
|
-
resp = urllib.request.urlopen(req, timeout=30)
|
|
190
|
-
remaining = float(resp.headers.get("X-Ratelimit-Remaining", 100))
|
|
191
|
-
reset2 = float(resp.headers.get("X-Ratelimit-Reset", 0))
|
|
192
|
-
_write_ratelimit(remaining, reset2)
|
|
193
|
-
data = json.loads(resp.read())
|
|
194
|
-
else:
|
|
195
|
-
raise
|
|
151
|
+
# Browser-only transport (see _fetch_via_browser; raises on failure,
|
|
152
|
+
# including the RateLimitedError this loop's callers already handle).
|
|
153
|
+
data = json.loads(_fetch_via_browser(url))
|
|
196
154
|
|
|
197
155
|
for child in data.get("data", {}).get("children", []):
|
|
198
156
|
d = child.get("data", {})
|
|
@@ -620,15 +578,9 @@ def _html_postable_check(thread_url):
|
|
|
620
578
|
try:
|
|
621
579
|
url = thread_url.replace("www.reddit.com", "old.reddit.com").rstrip("/") + "/"
|
|
622
580
|
_wait_if_needed()
|
|
623
|
-
# Browser-
|
|
581
|
+
# Browser-only transport (see _fetch_via_browser). Raises on failure;
|
|
582
|
+
# the outer except maps that to None ("network error") as before.
|
|
624
583
|
html = _fetch_via_browser(url)
|
|
625
|
-
if html is None:
|
|
626
|
-
req = urllib.request.Request(url, headers={"User-Agent": USER_AGENT})
|
|
627
|
-
resp = urllib.request.urlopen(req, timeout=15)
|
|
628
|
-
remaining = float(resp.headers.get("X-Ratelimit-Remaining", 100))
|
|
629
|
-
reset = float(resp.headers.get("X-Ratelimit-Reset", 0))
|
|
630
|
-
_write_ratelimit(remaining, reset)
|
|
631
|
-
html = resp.read().decode("utf-8", errors="ignore")
|
|
632
584
|
# Scope the lock check to the post header only. r/Entrepreneur (and
|
|
633
585
|
# similar subs) sticky an AutoMod comment that is itself locked,
|
|
634
586
|
# rendering `<span class="locked-tagline">locked comment</span>`
|
package/scripts/stats.py
CHANGED
|
@@ -447,64 +447,37 @@ def fetch_reddit_json(url, user_agent, max_retries=2, timeout=15):
|
|
|
447
447
|
(success AND error) into _reddit_rate_state so the caller can pace.
|
|
448
448
|
On 429, honors Retry-After (capped to 120s) and retries.
|
|
449
449
|
"""
|
|
450
|
-
# 2026-
|
|
451
|
-
#
|
|
452
|
-
#
|
|
453
|
-
#
|
|
454
|
-
#
|
|
455
|
-
#
|
|
456
|
-
#
|
|
457
|
-
|
|
458
|
-
|
|
459
|
-
|
|
460
|
-
|
|
461
|
-
if code == 200 and body:
|
|
462
|
-
try:
|
|
463
|
-
return ("ok", json.loads(body))
|
|
464
|
-
except Exception:
|
|
465
|
-
pass # non-JSON body -> fall through to urllib
|
|
466
|
-
elif code == 404:
|
|
467
|
-
return ("not_found", None)
|
|
468
|
-
except Exception:
|
|
469
|
-
pass
|
|
470
|
-
req = urllib.request.Request(url, headers={"User-Agent": user_agent})
|
|
450
|
+
# Browser-ONLY transport (2026-08-03, user decision: no urllib fallback).
|
|
451
|
+
# Reddit's TLS-fingerprint wall (2026-05-28) 403s urllib/curl on *.json
|
|
452
|
+
# unconditionally, so the old urllib fallback below could never succeed;
|
|
453
|
+
# it burned retries on guaranteed 403s and returned 'error' anyway.
|
|
454
|
+
# Transient browser failures (Reddit 503s, harness contention) get the
|
|
455
|
+
# same retry budget the urllib path had. The browser transport exposes
|
|
456
|
+
# no rate-limit headers, so 429 waits assume Reddit's standard 60s
|
|
457
|
+
# window; `user_agent`/`timeout` stay in the signature for callers but
|
|
458
|
+
# are unused (the harness browser supplies its own identity).
|
|
459
|
+
from reddit_browser_fetch import browser_get_json
|
|
460
|
+
|
|
471
461
|
for attempt in range(max_retries + 1):
|
|
472
462
|
try:
|
|
473
|
-
|
|
474
|
-
_update_reddit_rate_state(resp.headers)
|
|
475
|
-
body = resp.read()
|
|
476
|
-
if not body:
|
|
477
|
-
return ("empty", None)
|
|
478
|
-
try:
|
|
479
|
-
return ("ok", json.loads(body))
|
|
480
|
-
except Exception:
|
|
481
|
-
return ("empty", None)
|
|
482
|
-
except urllib.error.HTTPError as e:
|
|
483
|
-
_update_reddit_rate_state(e.headers)
|
|
484
|
-
if e.code == 404:
|
|
485
|
-
return ("not_found", None)
|
|
486
|
-
if e.code == 429:
|
|
487
|
-
retry_after = None
|
|
488
|
-
if e.headers:
|
|
489
|
-
ra = e.headers.get("Retry-After")
|
|
490
|
-
if ra:
|
|
491
|
-
try:
|
|
492
|
-
retry_after = int(ra)
|
|
493
|
-
except (TypeError, ValueError):
|
|
494
|
-
retry_after = None
|
|
495
|
-
if retry_after is None:
|
|
496
|
-
retry_after = int(_reddit_rate_state.get("reset_in") or 60)
|
|
497
|
-
retry_after = max(1, min(retry_after, 120))
|
|
498
|
-
if attempt < max_retries:
|
|
499
|
-
time.sleep(retry_after)
|
|
500
|
-
continue
|
|
501
|
-
return ("rate_limited", None)
|
|
502
|
-
return ("error", None)
|
|
463
|
+
body, code = browser_get_json(url)
|
|
503
464
|
except Exception:
|
|
465
|
+
body, code = None, 0
|
|
466
|
+
if code == 200 and body:
|
|
467
|
+
try:
|
|
468
|
+
return ("ok", json.loads(body))
|
|
469
|
+
except Exception:
|
|
470
|
+
return ("empty", None)
|
|
471
|
+
if code == 404:
|
|
472
|
+
return ("not_found", None)
|
|
473
|
+
if code == 429:
|
|
504
474
|
if attempt < max_retries:
|
|
505
|
-
time.sleep(
|
|
475
|
+
time.sleep(60)
|
|
506
476
|
continue
|
|
507
|
-
return ("
|
|
477
|
+
return ("rate_limited", None)
|
|
478
|
+
if attempt < max_retries:
|
|
479
|
+
time.sleep(5 * (attempt + 1))
|
|
480
|
+
continue
|
|
508
481
|
return ("error", None)
|
|
509
482
|
|
|
510
483
|
|
|
@@ -515,7 +515,11 @@ def _get_browser_and_page_raw(playwright):
|
|
|
515
515
|
# Otherwise reuse the first page (caller will navigate it).
|
|
516
516
|
if context.pages:
|
|
517
517
|
return browser, context.pages[0], True
|
|
518
|
-
|
|
518
|
+
# Zero pages: create in the BACKGROUND (2026-08-03). A plain
|
|
519
|
+
# new_page() is a foreground Target.createTarget, which
|
|
520
|
+
# activates Chrome and steals macOS app focus.
|
|
521
|
+
from browser_lifecycle import background_new_page
|
|
522
|
+
return browser, background_new_page(browser, context), True
|
|
519
523
|
# No contexts present (unusual on a fresh harness Chrome) — create one.
|
|
520
524
|
context = browser.new_context()
|
|
521
525
|
return browser, context.new_page(), True
|
|
@@ -545,7 +549,9 @@ def _get_browser_and_page_raw(playwright):
|
|
|
545
549
|
return browser, pg, True
|
|
546
550
|
if context.pages:
|
|
547
551
|
return browser, context.pages[0], True
|
|
548
|
-
|
|
552
|
+
# Zero pages: background create (2026-08-03, see above).
|
|
553
|
+
from browser_lifecycle import background_new_page
|
|
554
|
+
return browser, background_new_page(browser, context), True
|
|
549
555
|
except Exception as e:
|
|
550
556
|
_release_browser_lock()
|
|
551
557
|
print(json.dumps({
|