@m13v/s4l 1.7.7-rc.7 → 1.7.7-rc.8
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/mcp/dist/version.json +2 -2
- package/mcp/manifest.json +1 -1
- package/mcp/menubar/s4l_card.py +9 -1
- package/mcp/menubar/s4l_card_canvas.py +29 -26
- package/mcp/package.json +1 -1
- package/package.json +1 -1
- package/scripts/browser_lifecycle.py +30 -0
- package/scripts/harness_target_monitor.py +105 -0
- package/scripts/reddit_browser.py +9 -3
- package/scripts/reddit_browser_fetch.py +5 -1
- package/scripts/reddit_tools.py +50 -98
- package/scripts/stats.py +26 -53
- package/scripts/twitter_browser.py +8 -2
package/mcp/dist/version.json
CHANGED
package/mcp/manifest.json
CHANGED
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
"dxt_version": "0.1",
|
|
3
3
|
"name": "social-autoposter",
|
|
4
4
|
"display_name": "S4L",
|
|
5
|
-
"version": "1.7.7-rc.
|
|
5
|
+
"version": "1.7.7-rc.8",
|
|
6
6
|
"description": "Draft, review, approve, and autopilot X/Twitter posts.",
|
|
7
7
|
"long_description": "## **⚠️ The disclaimer above is generic Claude boilerplate.** Anthropic shows the same warning on every plugin regardless of what it does; any plugin has the same level of access as any app you download from the internet.\n\nS4L is an open source product developed by Mediar.ai Incorporated, a VC-backed San Francisco-based startup.\n\nTo get started:\n\n1\\. Copy this prompt: **Set me up on S4L plugin end to end**\n\n2\\. Quit with CMD+Q, reopen Claude, paste into a new chat.\n\nWhat happens next:\n\n* About every 5 minutes S4L scans X for posts that match your topics and drafts replies in your voice.\n* Drafts show up as review cards, usually the first within a few minutes. Nothing is posted automatically; you approve each one.\n* Posting autopilot stays off until you explicitly turn it on.",
|
|
8
8
|
"author": {
|
package/mcp/menubar/s4l_card.py
CHANGED
|
@@ -1964,7 +1964,12 @@ class _ReviewController(NSObject):
|
|
|
1964
1964
|
"""NSTimer target (2026-07-15): re-renders the header's age/expiry
|
|
1965
1965
|
label every second so its countdown visibly counts down without
|
|
1966
1966
|
needing hover. Not a python_method -- NSTimer invokes this through
|
|
1967
|
-
the ObjC runtime.
|
|
1967
|
+
the ObjC runtime. No-op when the rendered text is unchanged: the
|
|
1968
|
+
label usually shows a coarse "3h"-style value that only changes
|
|
1969
|
+
every few minutes, and unconditionally re-styling it dirtied one
|
|
1970
|
+
layer per tile per second -- with a 130-tile canvas that was a
|
|
1971
|
+
constant CoreAnimation commit churn keeping the app at ~15% CPU
|
|
1972
|
+
while idle (2026-08-02 lag incident)."""
|
|
1968
1973
|
if self._age_expiry_label is None:
|
|
1969
1974
|
return
|
|
1970
1975
|
try:
|
|
@@ -1975,6 +1980,9 @@ class _ReviewController(NSObject):
|
|
|
1975
1980
|
)
|
|
1976
1981
|
if not text:
|
|
1977
1982
|
return
|
|
1983
|
+
if (text, urgent) == getattr(self, "_age_expiry_last", None):
|
|
1984
|
+
return
|
|
1985
|
+
self._age_expiry_last = (text, urgent)
|
|
1978
1986
|
self._age_expiry_label.setStringValue_(text)
|
|
1979
1987
|
self._age_expiry_label.setFont_(_font(11, urgent))
|
|
1980
1988
|
self._age_expiry_label.setTextColor_(
|
|
@@ -461,14 +461,13 @@ class _CanvasController(NSObject):
|
|
|
461
461
|
|
|
462
462
|
@objc.python_method
|
|
463
463
|
def _reflow_from(self, start_idx):
|
|
464
|
-
"""
|
|
465
|
-
|
|
466
|
-
|
|
467
|
-
|
|
468
|
-
|
|
469
|
-
|
|
470
|
-
|
|
471
|
-
the reviewer hadn't yet acted on."""
|
|
464
|
+
"""Build every slot's CONTENT from start_idx onward from the current
|
|
465
|
+
self._order. Only two callers: the initial grid build (start 0) and
|
|
466
|
+
extend_drafts (start = old length, so only the NEW empty slots get
|
|
467
|
+
content). Decisions no longer come through here -- rebuilding ~130
|
|
468
|
+
full tile views per click was the dominant cost of an approval on a
|
|
469
|
+
big backlog (2026-08-02 lag incident, second act); _remove_and_reflow
|
|
470
|
+
now MOVES the surviving slot views instead."""
|
|
472
471
|
for i in range(start_idx, len(self._slots)):
|
|
473
472
|
slot = self._slots[i]
|
|
474
473
|
for sv in list(slot["view"].subviews()):
|
|
@@ -476,8 +475,8 @@ class _CanvasController(NSObject):
|
|
|
476
475
|
d = self._order[i]
|
|
477
476
|
tile = _ReviewController.alloc().initWithDrafts_onDecision_onComplete_focus_hostView_hostWindow_(
|
|
478
477
|
[d],
|
|
479
|
-
self._tile_decision_cb(
|
|
480
|
-
self._tile_complete_cb(
|
|
478
|
+
self._tile_decision_cb(slot),
|
|
479
|
+
self._tile_complete_cb(slot),
|
|
481
480
|
True,
|
|
482
481
|
slot["view"],
|
|
483
482
|
self._panel,
|
|
@@ -487,19 +486,23 @@ class _CanvasController(NSObject):
|
|
|
487
486
|
|
|
488
487
|
@objc.python_method
|
|
489
488
|
def _remove_and_reflow(self, n):
|
|
490
|
-
"""Pop draft `n` out of self._order, drop
|
|
491
|
-
|
|
492
|
-
|
|
493
|
-
|
|
489
|
+
"""Pop draft `n` out of self._order, drop ITS slot (view and all),
|
|
490
|
+
and shift every later slot's VIEW up into the vacated grid position
|
|
491
|
+
("snake" reflow, 2026-07-16 user direction). Positions move, content
|
|
492
|
+
doesn't: each surviving tile keeps its live view -- O(N) setFrame
|
|
493
|
+
calls instead of O(N) full tile rebuilds (2026-08-02 lag fix), and
|
|
494
|
+
an in-progress edit on a later card now survives earlier decisions
|
|
495
|
+
instead of being clobbered by the rebuild."""
|
|
494
496
|
try:
|
|
495
497
|
idx = next(i for i, d in enumerate(self._order) if d.get("n") == n)
|
|
496
498
|
except StopIteration:
|
|
497
499
|
return
|
|
498
500
|
self._order.pop(idx)
|
|
499
|
-
if self._slots:
|
|
500
|
-
|
|
501
|
-
|
|
502
|
-
self.
|
|
501
|
+
if idx < len(self._slots):
|
|
502
|
+
gone = self._slots.pop(idx)
|
|
503
|
+
gone["view"].removeFromSuperview()
|
|
504
|
+
for i in range(idx, len(self._slots)):
|
|
505
|
+
self._slots[i]["view"].setFrame_(self._slot_frame(i))
|
|
503
506
|
self._resize_doc()
|
|
504
507
|
self._refresh_header()
|
|
505
508
|
# Last card decided -> nothing left to review, so close the canvas
|
|
@@ -535,7 +538,7 @@ class _CanvasController(NSObject):
|
|
|
535
538
|
_log(f"canvas discard-all handler failed: {e}")
|
|
536
539
|
|
|
537
540
|
@objc.python_method
|
|
538
|
-
def _tile_decision_cb(self,
|
|
541
|
+
def _tile_decision_cb(self, slot):
|
|
539
542
|
def _cb(decision):
|
|
540
543
|
self._decisions.append(decision)
|
|
541
544
|
self._last_decision_at = time.time()
|
|
@@ -549,16 +552,16 @@ class _CanvasController(NSObject):
|
|
|
549
552
|
return _cb
|
|
550
553
|
|
|
551
554
|
@objc.python_method
|
|
552
|
-
def _tile_complete_cb(self,
|
|
555
|
+
def _tile_complete_cb(self, slot):
|
|
553
556
|
def _cb(_tile_decisions):
|
|
554
557
|
# The tile's own single-draft stack finished -- remove it from
|
|
555
558
|
# the ranking and let everything after it shift up ("snake"
|
|
556
|
-
# reflow; see _remove_and_reflow).
|
|
557
|
-
#
|
|
558
|
-
#
|
|
559
|
-
#
|
|
560
|
-
|
|
561
|
-
n = slot["n"]
|
|
559
|
+
# reflow; see _remove_and_reflow). Closing over the SLOT DICT
|
|
560
|
+
# (not a positional index) keeps the lookup correct no matter
|
|
561
|
+
# how many earlier slots have been popped since this closure
|
|
562
|
+
# was created: a slot keeps its `n` for life now that reflow
|
|
563
|
+
# moves views instead of rebuilding content.
|
|
564
|
+
n = slot["n"]
|
|
562
565
|
if n is not None:
|
|
563
566
|
self._remove_and_reflow(n)
|
|
564
567
|
|
package/mcp/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@m13v/s4l-mcp",
|
|
3
|
-
"version": "1.7.7-rc.
|
|
3
|
+
"version": "1.7.7-rc.8",
|
|
4
4
|
"private": true,
|
|
5
5
|
"description": "Desktop MCP client for social-autoposter (X/Twitter rail): manual draft/review/approve loop, autopilot control, and stats. Thin wrapper over the existing pipeline scripts.",
|
|
6
6
|
"license": "MIT",
|
package/package.json
CHANGED
|
@@ -77,6 +77,36 @@ def park_tabs(cdp_base: str, host_markers, park_url: str, label: str) -> None:
|
|
|
77
77
|
pass
|
|
78
78
|
|
|
79
79
|
|
|
80
|
+
def background_new_page(browser, context, url: str = "about:blank", timeout_ms: int = 10_000):
|
|
81
|
+
"""Create a Playwright page WITHOUT raising the Chrome window.
|
|
82
|
+
|
|
83
|
+
Playwright's context.new_page() maps to a FOREGROUND Target.createTarget,
|
|
84
|
+
which activates Chrome on macOS and steals app focus (the July 2026
|
|
85
|
+
focus-steal class; same call the bh helpers and the harness daemon were
|
|
86
|
+
already fixed to avoid). Playwright's public API has no background
|
|
87
|
+
option, so this creates the target via a browser-level CDP session with
|
|
88
|
+
background:true and returns the Page that `context` adopts for it.
|
|
89
|
+
|
|
90
|
+
ONE shared implementation for every attach path that previously called
|
|
91
|
+
context.new_page() on the harness (reddit_browser, twitter_browser,
|
|
92
|
+
reddit_browser_fetch). Falls back to context.new_page() if the CDP path
|
|
93
|
+
fails: a rare focus blip beats a dead pipeline.
|
|
94
|
+
"""
|
|
95
|
+
try:
|
|
96
|
+
cdp = browser.new_browser_cdp_session()
|
|
97
|
+
try:
|
|
98
|
+
with context.expect_page(timeout=timeout_ms) as pg_info:
|
|
99
|
+
cdp.send("Target.createTarget", {"url": url, "background": True})
|
|
100
|
+
return pg_info.value
|
|
101
|
+
finally:
|
|
102
|
+
try:
|
|
103
|
+
cdp.detach()
|
|
104
|
+
except Exception:
|
|
105
|
+
pass
|
|
106
|
+
except Exception:
|
|
107
|
+
return context.new_page()
|
|
108
|
+
|
|
109
|
+
|
|
80
110
|
def register_park_on_exit(cdp_base: str, host_markers, park_url: str, label: str) -> None:
|
|
81
111
|
"""Arm park_tabs to run at process exit, once per (endpoint, park_url).
|
|
82
112
|
Call from a platform lib's get_browser_and_page so only processes that
|
|
@@ -0,0 +1,105 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""Attribution monitor for harness-Chrome focus steals (2026-08-03).
|
|
3
|
+
|
|
4
|
+
Connects to a harness Chrome's browser-level CDP websocket, enables target
|
|
5
|
+
discovery, and appends one timestamped line per Target lifecycle event
|
|
6
|
+
(created / destroyed / info-changed => navigations) to a log file. Correlate
|
|
7
|
+
these against the `[browser-foreground]` activation lines in
|
|
8
|
+
~/.social-autoposter-mcp/menubar/menubar.err.log to attribute WHICH tab
|
|
9
|
+
operation coincided with an app activation, something none of the existing
|
|
10
|
+
logs capture (the daemon log has no timestamps; python new_page sites have
|
|
11
|
+
no logging at all; bh [bh_tab_event] covers only the bh lanes).
|
|
12
|
+
|
|
13
|
+
Read-only: never creates, closes, or navigates anything.
|
|
14
|
+
|
|
15
|
+
Usage:
|
|
16
|
+
harness_target_monitor.py [--port 9557] [--log PATH]
|
|
17
|
+
|
|
18
|
+
Runs forever; reconnects with backoff when Chrome restarts. Intended to run
|
|
19
|
+
under nohup during a diagnosis window. It is NOT part of the pipeline.
|
|
20
|
+
"""
|
|
21
|
+
import argparse
|
|
22
|
+
import json
|
|
23
|
+
import os
|
|
24
|
+
import sys
|
|
25
|
+
import time
|
|
26
|
+
from datetime import datetime, timezone
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def ts() -> str:
|
|
30
|
+
return datetime.now(timezone.utc).strftime("%Y-%m-%dT%H:%M:%S.%f")[:-3] + "Z"
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def log_line(path: str, msg: str) -> None:
|
|
34
|
+
with open(path, "a") as f:
|
|
35
|
+
f.write(f"[{ts()}] {msg}\n")
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def monitor_once(port: int, log_path: str) -> None:
|
|
39
|
+
import urllib.request
|
|
40
|
+
import websocket
|
|
41
|
+
|
|
42
|
+
opener = urllib.request.build_opener(urllib.request.ProxyHandler({}))
|
|
43
|
+
info = json.loads(opener.open(f"http://127.0.0.1:{port}/json/version", timeout=3).read())
|
|
44
|
+
ws = websocket.create_connection(
|
|
45
|
+
info["webSocketDebuggerUrl"], timeout=5, suppress_origin=True
|
|
46
|
+
)
|
|
47
|
+
try:
|
|
48
|
+
ws.send(json.dumps({"id": 1, "method": "Target.setDiscoverTargets",
|
|
49
|
+
"params": {"discover": True}}))
|
|
50
|
+
log_line(log_path, f"monitor attached port={port} chrome={info.get('Browser','?')}")
|
|
51
|
+
ws.settimeout(60)
|
|
52
|
+
last_url = {}
|
|
53
|
+
while True:
|
|
54
|
+
try:
|
|
55
|
+
msg = json.loads(ws.recv())
|
|
56
|
+
except Exception as e:
|
|
57
|
+
if "timed out" in str(e).lower():
|
|
58
|
+
# Idle is fine; poke the connection so a dead Chrome errors out.
|
|
59
|
+
ws.send(json.dumps({"id": 2, "method": "Browser.getVersion"}))
|
|
60
|
+
continue
|
|
61
|
+
raise
|
|
62
|
+
method = msg.get("method", "")
|
|
63
|
+
p = msg.get("params", {})
|
|
64
|
+
t = p.get("targetInfo", {})
|
|
65
|
+
if t.get("type") not in ("page", ""):
|
|
66
|
+
continue
|
|
67
|
+
tid = t.get("targetId") or p.get("targetId", "?")
|
|
68
|
+
url = t.get("url", "")
|
|
69
|
+
if method == "Target.targetCreated":
|
|
70
|
+
log_line(log_path, f"CREATED {tid} url={url[:100]}")
|
|
71
|
+
elif method == "Target.targetDestroyed":
|
|
72
|
+
log_line(log_path, f"DESTROYED {tid} (last_url={last_url.get(tid, '?')[:100]})")
|
|
73
|
+
elif method == "Target.targetInfoChanged":
|
|
74
|
+
if url and url != last_url.get(tid):
|
|
75
|
+
log_line(log_path, f"NAVIGATED {tid} url={url[:100]}")
|
|
76
|
+
if tid != "?" and url:
|
|
77
|
+
last_url[tid] = url
|
|
78
|
+
finally:
|
|
79
|
+
try:
|
|
80
|
+
ws.close()
|
|
81
|
+
except Exception:
|
|
82
|
+
pass
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
def main() -> int:
|
|
86
|
+
ap = argparse.ArgumentParser()
|
|
87
|
+
ap.add_argument("--port", type=int, default=9557)
|
|
88
|
+
ap.add_argument("--log", default=os.path.expanduser(
|
|
89
|
+
"~/social-autoposter/skill/logs/harness-target-events-9557.log"))
|
|
90
|
+
args = ap.parse_args()
|
|
91
|
+
while True:
|
|
92
|
+
try:
|
|
93
|
+
monitor_once(args.port, args.log)
|
|
94
|
+
except KeyboardInterrupt:
|
|
95
|
+
return 0
|
|
96
|
+
except Exception as e:
|
|
97
|
+
try:
|
|
98
|
+
log_line(args.log, f"monitor disconnected ({type(e).__name__}: {str(e)[:120]}); retry in 15s")
|
|
99
|
+
except OSError:
|
|
100
|
+
pass
|
|
101
|
+
time.sleep(15)
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
if __name__ == "__main__":
|
|
105
|
+
sys.exit(main())
|
|
@@ -441,7 +441,11 @@ def _get_browser_and_page_raw(playwright):
|
|
|
441
441
|
return cdp_browser, pg, True
|
|
442
442
|
if chosen.pages:
|
|
443
443
|
return cdp_browser, chosen.pages[0], True
|
|
444
|
-
|
|
444
|
+
# Zero pages: create in the BACKGROUND (2026-08-03). A plain
|
|
445
|
+
# new_page() here is a foreground Target.createTarget, which
|
|
446
|
+
# activates Chrome and steals macOS app focus.
|
|
447
|
+
from browser_lifecycle import background_new_page
|
|
448
|
+
page = background_new_page(cdp_browser, chosen)
|
|
445
449
|
return cdp_browser, page, True
|
|
446
450
|
# No usable context: do NOT close the CDP browser (would kill the
|
|
447
451
|
# harness Chrome); just disconnect by falling through.
|
|
@@ -464,13 +468,15 @@ def _get_browser_and_page_raw(playwright):
|
|
|
464
468
|
for c in cookies
|
|
465
469
|
)
|
|
466
470
|
if has_session:
|
|
467
|
-
# Reuse an existing tab (no focus-steal); only
|
|
471
|
+
# Reuse an existing tab (no focus-steal); only create if none,
|
|
472
|
+
# and then in the BACKGROUND (2026-08-03, see above).
|
|
468
473
|
for pg in ctx.pages:
|
|
469
474
|
if "reddit.com" in (pg.url or "") and "login" not in (pg.url or ""):
|
|
470
475
|
return cdp_browser, pg, True
|
|
471
476
|
if ctx.pages:
|
|
472
477
|
return cdp_browser, ctx.pages[0], True
|
|
473
|
-
|
|
478
|
+
from browser_lifecycle import background_new_page
|
|
479
|
+
page = background_new_page(cdp_browser, ctx)
|
|
474
480
|
return cdp_browser, page, True
|
|
475
481
|
try:
|
|
476
482
|
cdp_browser.close()
|
|
@@ -165,7 +165,11 @@ def browser_get_json(url, cdp_url=None, timeout_ms=25000):
|
|
|
165
165
|
if page is None and ctx.pages:
|
|
166
166
|
page = ctx.pages[0]
|
|
167
167
|
if page is None:
|
|
168
|
-
|
|
168
|
+
# Zero pages: create in the BACKGROUND (2026-08-03). A plain
|
|
169
|
+
# new_page() is a foreground Target.createTarget, which
|
|
170
|
+
# activates Chrome and steals macOS app focus.
|
|
171
|
+
from browser_lifecycle import background_new_page
|
|
172
|
+
page = background_new_page(browser, ctx)
|
|
169
173
|
# Load the matching host root so the subsequent fetch() is same-origin
|
|
170
174
|
# (no CORS between www/old) and carries the logged-in session.
|
|
171
175
|
try:
|
package/scripts/reddit_tools.py
CHANGED
|
@@ -74,71 +74,61 @@ def _wait_if_needed():
|
|
|
74
74
|
def _fetch_via_browser(url):
|
|
75
75
|
"""Fetch a Reddit URL through the reddit-harness logged-in Chrome.
|
|
76
76
|
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
|
|
77
|
+
Browser-ONLY transport (2026-08-03, user decision: no urllib fallback).
|
|
78
|
+
Reddit's TLS-fingerprint wall (2026-05-28) 403s urllib/curl on *.json
|
|
79
|
+
unconditionally, so the old fallback could never succeed: it burned ~2s
|
|
80
|
+
per browser hiccup and then lost the query anyway. Transient failures
|
|
81
|
+
(Reddit 503s, harness contention) get ONE in-transport retry instead.
|
|
82
|
+
The REDDIT_FETCH_BACKEND=urllib debug knob is gone for the same reason:
|
|
83
|
+
forcing a transport that is guaranteed to 403 debugs nothing.
|
|
84
|
+
|
|
85
|
+
Returns the raw response body (str) on HTTP 200. Raises on final failure,
|
|
86
|
+
urllib.error.HTTPError for HTTP statuses and urllib.error.URLError for
|
|
87
|
+
transport-level failures, matching the exception shapes callers already
|
|
88
|
+
handle from the urllib era. 429 keeps the old inline-wait contract
|
|
89
|
+
(absorb a short wait, else RateLimitedError).
|
|
86
90
|
"""
|
|
87
|
-
|
|
88
|
-
|
|
89
|
-
|
|
90
|
-
|
|
91
|
-
except Exception as e:
|
|
92
|
-
sys.stderr.write(f"[reddit_tools] browser fetch unavailable ({e}); urllib fallback\n")
|
|
93
|
-
return None
|
|
94
|
-
try:
|
|
91
|
+
from reddit_browser_fetch import browser_get_json
|
|
92
|
+
|
|
93
|
+
last_status = 0
|
|
94
|
+
for attempt in (1, 2):
|
|
95
95
|
body, status = browser_get_json(url)
|
|
96
96
|
if status == 200 and body:
|
|
97
97
|
return body
|
|
98
|
-
|
|
99
|
-
|
|
100
|
-
|
|
101
|
-
|
|
98
|
+
last_status = status
|
|
99
|
+
if status == 429:
|
|
100
|
+
# Browser transport exposes no X-Ratelimit-Reset header; assume
|
|
101
|
+
# Reddit's standard 60s window.
|
|
102
|
+
_write_ratelimit(0, 60)
|
|
103
|
+
if 60 > MAX_INLINE_WAIT_SECONDS:
|
|
104
|
+
raise RateLimitedError(60)
|
|
105
|
+
if attempt == 1:
|
|
106
|
+
print("Rate limited. Waiting 62s...", file=sys.stderr)
|
|
107
|
+
time.sleep(62)
|
|
108
|
+
continue
|
|
109
|
+
sys.stderr.write(
|
|
110
|
+
f"[reddit_tools] browser fetch status={status} for {url[:80]} (attempt {attempt}/2)\n"
|
|
111
|
+
)
|
|
112
|
+
if attempt == 1:
|
|
113
|
+
time.sleep(2.5)
|
|
114
|
+
if last_status == 429:
|
|
115
|
+
raise RateLimitedError(60)
|
|
116
|
+
if last_status:
|
|
117
|
+
raise urllib.error.HTTPError(
|
|
118
|
+
url, last_status,
|
|
119
|
+
f"reddit browser fetch failed (status={last_status})", None, None,
|
|
120
|
+
)
|
|
121
|
+
raise urllib.error.URLError(f"reddit browser transport failed for {url[:80]}")
|
|
102
122
|
|
|
103
123
|
|
|
104
124
|
def _do_request(url):
|
|
105
|
-
"""Make a Reddit API request
|
|
125
|
+
"""Make a Reddit API request via the harness browser (sole transport).
|
|
106
126
|
|
|
107
|
-
|
|
108
|
-
|
|
109
|
-
immediately if the reset would require a long wait, else absorbs short waits.
|
|
127
|
+
429 handling (inline wait / RateLimitedError) lives inside
|
|
128
|
+
_fetch_via_browser; HTTP and transport failures raise there too.
|
|
110
129
|
"""
|
|
111
130
|
_wait_if_needed()
|
|
112
|
-
|
|
113
|
-
# if the harness is down or returns a non-200.
|
|
114
|
-
_body = _fetch_via_browser(url)
|
|
115
|
-
if _body is not None:
|
|
116
|
-
try:
|
|
117
|
-
return json.loads(_body)
|
|
118
|
-
except Exception:
|
|
119
|
-
sys.stderr.write(f"[reddit_tools] browser body not JSON for {url[:80]}; urllib fallback\n")
|
|
120
|
-
req = urllib.request.Request(url, headers={"User-Agent": USER_AGENT})
|
|
121
|
-
try:
|
|
122
|
-
resp = urllib.request.urlopen(req, timeout=20)
|
|
123
|
-
remaining = float(resp.headers.get("X-Ratelimit-Remaining", 100))
|
|
124
|
-
reset = float(resp.headers.get("X-Ratelimit-Reset", 0))
|
|
125
|
-
_write_ratelimit(remaining, reset)
|
|
126
|
-
return json.loads(resp.read())
|
|
127
|
-
except urllib.error.HTTPError as e:
|
|
128
|
-
if e.code == 429:
|
|
129
|
-
reset = float(e.headers.get("X-Ratelimit-Reset", 60))
|
|
130
|
-
_write_ratelimit(0, reset)
|
|
131
|
-
if reset > MAX_INLINE_WAIT_SECONDS:
|
|
132
|
-
raise RateLimitedError(reset)
|
|
133
|
-
print(f"Rate limited. Waiting {int(reset)+2}s...", file=sys.stderr)
|
|
134
|
-
time.sleep(int(reset) + 2)
|
|
135
|
-
# Retry once
|
|
136
|
-
resp = urllib.request.urlopen(req, timeout=20)
|
|
137
|
-
remaining = float(resp.headers.get("X-Ratelimit-Remaining", 100))
|
|
138
|
-
reset2 = float(resp.headers.get("X-Ratelimit-Reset", 0))
|
|
139
|
-
_write_ratelimit(remaining, reset2)
|
|
140
|
-
return json.loads(resp.read())
|
|
141
|
-
raise
|
|
131
|
+
return json.loads(_fetch_via_browser(url))
|
|
142
132
|
|
|
143
133
|
|
|
144
134
|
def batch_fetch_info(thing_ids, user_agent=USER_AGENT):
|
|
@@ -158,41 +148,9 @@ def batch_fetch_info(thing_ids, user_agent=USER_AGENT):
|
|
|
158
148
|
ids_str = ",".join(chunk)
|
|
159
149
|
url = f"https://old.reddit.com/api/info.json?id={ids_str}"
|
|
160
150
|
_wait_if_needed()
|
|
161
|
-
# Browser-
|
|
162
|
-
|
|
163
|
-
|
|
164
|
-
try:
|
|
165
|
-
data = json.loads(_body)
|
|
166
|
-
for child in data.get("data", {}).get("children", []):
|
|
167
|
-
cd = child.get("data", {})
|
|
168
|
-
name = cd.get("name")
|
|
169
|
-
if name:
|
|
170
|
-
results[name] = cd
|
|
171
|
-
continue
|
|
172
|
-
except Exception:
|
|
173
|
-
sys.stderr.write("[reddit_tools] browser info.json not JSON; urllib fallback\n")
|
|
174
|
-
req = urllib.request.Request(url, headers={"User-Agent": user_agent})
|
|
175
|
-
try:
|
|
176
|
-
resp = urllib.request.urlopen(req, timeout=30)
|
|
177
|
-
remaining = float(resp.headers.get("X-Ratelimit-Remaining", 100))
|
|
178
|
-
reset = float(resp.headers.get("X-Ratelimit-Reset", 0))
|
|
179
|
-
_write_ratelimit(remaining, reset)
|
|
180
|
-
data = json.loads(resp.read())
|
|
181
|
-
except urllib.error.HTTPError as e:
|
|
182
|
-
if e.code == 429:
|
|
183
|
-
reset = float(e.headers.get("X-Ratelimit-Reset", 60))
|
|
184
|
-
_write_ratelimit(0, reset)
|
|
185
|
-
if reset > MAX_INLINE_WAIT_SECONDS:
|
|
186
|
-
raise RateLimitedError(reset)
|
|
187
|
-
print(f"Rate limited. Waiting {int(reset)+2}s...", file=sys.stderr)
|
|
188
|
-
time.sleep(int(reset) + 2)
|
|
189
|
-
resp = urllib.request.urlopen(req, timeout=30)
|
|
190
|
-
remaining = float(resp.headers.get("X-Ratelimit-Remaining", 100))
|
|
191
|
-
reset2 = float(resp.headers.get("X-Ratelimit-Reset", 0))
|
|
192
|
-
_write_ratelimit(remaining, reset2)
|
|
193
|
-
data = json.loads(resp.read())
|
|
194
|
-
else:
|
|
195
|
-
raise
|
|
151
|
+
# Browser-only transport (see _fetch_via_browser; raises on failure,
|
|
152
|
+
# including the RateLimitedError this loop's callers already handle).
|
|
153
|
+
data = json.loads(_fetch_via_browser(url))
|
|
196
154
|
|
|
197
155
|
for child in data.get("data", {}).get("children", []):
|
|
198
156
|
d = child.get("data", {})
|
|
@@ -620,15 +578,9 @@ def _html_postable_check(thread_url):
|
|
|
620
578
|
try:
|
|
621
579
|
url = thread_url.replace("www.reddit.com", "old.reddit.com").rstrip("/") + "/"
|
|
622
580
|
_wait_if_needed()
|
|
623
|
-
# Browser-
|
|
581
|
+
# Browser-only transport (see _fetch_via_browser). Raises on failure;
|
|
582
|
+
# the outer except maps that to None ("network error") as before.
|
|
624
583
|
html = _fetch_via_browser(url)
|
|
625
|
-
if html is None:
|
|
626
|
-
req = urllib.request.Request(url, headers={"User-Agent": USER_AGENT})
|
|
627
|
-
resp = urllib.request.urlopen(req, timeout=15)
|
|
628
|
-
remaining = float(resp.headers.get("X-Ratelimit-Remaining", 100))
|
|
629
|
-
reset = float(resp.headers.get("X-Ratelimit-Reset", 0))
|
|
630
|
-
_write_ratelimit(remaining, reset)
|
|
631
|
-
html = resp.read().decode("utf-8", errors="ignore")
|
|
632
584
|
# Scope the lock check to the post header only. r/Entrepreneur (and
|
|
633
585
|
# similar subs) sticky an AutoMod comment that is itself locked,
|
|
634
586
|
# rendering `<span class="locked-tagline">locked comment</span>`
|
package/scripts/stats.py
CHANGED
|
@@ -447,64 +447,37 @@ def fetch_reddit_json(url, user_agent, max_retries=2, timeout=15):
|
|
|
447
447
|
(success AND error) into _reddit_rate_state so the caller can pace.
|
|
448
448
|
On 429, honors Retry-After (capped to 120s) and retries.
|
|
449
449
|
"""
|
|
450
|
-
# 2026-
|
|
451
|
-
#
|
|
452
|
-
#
|
|
453
|
-
#
|
|
454
|
-
#
|
|
455
|
-
#
|
|
456
|
-
#
|
|
457
|
-
|
|
458
|
-
|
|
459
|
-
|
|
460
|
-
|
|
461
|
-
if code == 200 and body:
|
|
462
|
-
try:
|
|
463
|
-
return ("ok", json.loads(body))
|
|
464
|
-
except Exception:
|
|
465
|
-
pass # non-JSON body -> fall through to urllib
|
|
466
|
-
elif code == 404:
|
|
467
|
-
return ("not_found", None)
|
|
468
|
-
except Exception:
|
|
469
|
-
pass
|
|
470
|
-
req = urllib.request.Request(url, headers={"User-Agent": user_agent})
|
|
450
|
+
# Browser-ONLY transport (2026-08-03, user decision: no urllib fallback).
|
|
451
|
+
# Reddit's TLS-fingerprint wall (2026-05-28) 403s urllib/curl on *.json
|
|
452
|
+
# unconditionally, so the old urllib fallback below could never succeed;
|
|
453
|
+
# it burned retries on guaranteed 403s and returned 'error' anyway.
|
|
454
|
+
# Transient browser failures (Reddit 503s, harness contention) get the
|
|
455
|
+
# same retry budget the urllib path had. The browser transport exposes
|
|
456
|
+
# no rate-limit headers, so 429 waits assume Reddit's standard 60s
|
|
457
|
+
# window; `user_agent`/`timeout` stay in the signature for callers but
|
|
458
|
+
# are unused (the harness browser supplies its own identity).
|
|
459
|
+
from reddit_browser_fetch import browser_get_json
|
|
460
|
+
|
|
471
461
|
for attempt in range(max_retries + 1):
|
|
472
462
|
try:
|
|
473
|
-
|
|
474
|
-
_update_reddit_rate_state(resp.headers)
|
|
475
|
-
body = resp.read()
|
|
476
|
-
if not body:
|
|
477
|
-
return ("empty", None)
|
|
478
|
-
try:
|
|
479
|
-
return ("ok", json.loads(body))
|
|
480
|
-
except Exception:
|
|
481
|
-
return ("empty", None)
|
|
482
|
-
except urllib.error.HTTPError as e:
|
|
483
|
-
_update_reddit_rate_state(e.headers)
|
|
484
|
-
if e.code == 404:
|
|
485
|
-
return ("not_found", None)
|
|
486
|
-
if e.code == 429:
|
|
487
|
-
retry_after = None
|
|
488
|
-
if e.headers:
|
|
489
|
-
ra = e.headers.get("Retry-After")
|
|
490
|
-
if ra:
|
|
491
|
-
try:
|
|
492
|
-
retry_after = int(ra)
|
|
493
|
-
except (TypeError, ValueError):
|
|
494
|
-
retry_after = None
|
|
495
|
-
if retry_after is None:
|
|
496
|
-
retry_after = int(_reddit_rate_state.get("reset_in") or 60)
|
|
497
|
-
retry_after = max(1, min(retry_after, 120))
|
|
498
|
-
if attempt < max_retries:
|
|
499
|
-
time.sleep(retry_after)
|
|
500
|
-
continue
|
|
501
|
-
return ("rate_limited", None)
|
|
502
|
-
return ("error", None)
|
|
463
|
+
body, code = browser_get_json(url)
|
|
503
464
|
except Exception:
|
|
465
|
+
body, code = None, 0
|
|
466
|
+
if code == 200 and body:
|
|
467
|
+
try:
|
|
468
|
+
return ("ok", json.loads(body))
|
|
469
|
+
except Exception:
|
|
470
|
+
return ("empty", None)
|
|
471
|
+
if code == 404:
|
|
472
|
+
return ("not_found", None)
|
|
473
|
+
if code == 429:
|
|
504
474
|
if attempt < max_retries:
|
|
505
|
-
time.sleep(
|
|
475
|
+
time.sleep(60)
|
|
506
476
|
continue
|
|
507
|
-
return ("
|
|
477
|
+
return ("rate_limited", None)
|
|
478
|
+
if attempt < max_retries:
|
|
479
|
+
time.sleep(5 * (attempt + 1))
|
|
480
|
+
continue
|
|
508
481
|
return ("error", None)
|
|
509
482
|
|
|
510
483
|
|
|
@@ -515,7 +515,11 @@ def _get_browser_and_page_raw(playwright):
|
|
|
515
515
|
# Otherwise reuse the first page (caller will navigate it).
|
|
516
516
|
if context.pages:
|
|
517
517
|
return browser, context.pages[0], True
|
|
518
|
-
|
|
518
|
+
# Zero pages: create in the BACKGROUND (2026-08-03). A plain
|
|
519
|
+
# new_page() is a foreground Target.createTarget, which
|
|
520
|
+
# activates Chrome and steals macOS app focus.
|
|
521
|
+
from browser_lifecycle import background_new_page
|
|
522
|
+
return browser, background_new_page(browser, context), True
|
|
519
523
|
# No contexts present (unusual on a fresh harness Chrome) — create one.
|
|
520
524
|
context = browser.new_context()
|
|
521
525
|
return browser, context.new_page(), True
|
|
@@ -545,7 +549,9 @@ def _get_browser_and_page_raw(playwright):
|
|
|
545
549
|
return browser, pg, True
|
|
546
550
|
if context.pages:
|
|
547
551
|
return browser, context.pages[0], True
|
|
548
|
-
|
|
552
|
+
# Zero pages: background create (2026-08-03, see above).
|
|
553
|
+
from browser_lifecycle import background_new_page
|
|
554
|
+
return browser, background_new_page(browser, context), True
|
|
549
555
|
except Exception as e:
|
|
550
556
|
_release_browser_lock()
|
|
551
557
|
print(json.dumps({
|