substack-saved-mcp 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- substack_saved_mcp/__init__.py +3 -0
- substack_saved_mcp/cli.py +413 -0
- substack_saved_mcp/config.py +55 -0
- substack_saved_mcp/content_utils.py +151 -0
- substack_saved_mcp/database.py +550 -0
- substack_saved_mcp/mcp_server.py +307 -0
- substack_saved_mcp/models.py +92 -0
- substack_saved_mcp/substack_client.py +718 -0
- substack_saved_mcp/sync.py +267 -0
- substack_saved_mcp/url_utils.py +55 -0
- substack_saved_mcp-0.1.0.dist-info/METADATA +218 -0
- substack_saved_mcp-0.1.0.dist-info/RECORD +15 -0
- substack_saved_mcp-0.1.0.dist-info/WHEEL +4 -0
- substack_saved_mcp-0.1.0.dist-info/entry_points.txt +2 -0
- substack_saved_mcp-0.1.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,718 @@
|
|
|
1
|
+
"""Playwright client for Substack authentication, saved post extractions, and write operations."""
|
|
2
|
+
|
|
3
|
+
import concurrent.futures
|
|
4
|
+
import logging
|
|
5
|
+
import time
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
from typing import Any
|
|
8
|
+
from urllib.parse import urlparse
|
|
9
|
+
|
|
10
|
+
from substack_saved_mcp.config import get_browser_dir, get_storage_state_path
|
|
11
|
+
from substack_saved_mcp.models import SavedPost
|
|
12
|
+
from substack_saved_mcp.url_utils import canonicalize_url
|
|
13
|
+
|
|
14
|
+
logger = logging.getLogger(__name__)
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def _run_playwright_sync(func, *args, **kwargs):
|
|
18
|
+
"""Execute a function using Playwright Sync API safely, dispatching to a worker thread if an asyncio loop is active."""
|
|
19
|
+
try:
|
|
20
|
+
import asyncio
|
|
21
|
+
|
|
22
|
+
loop = asyncio.get_running_loop()
|
|
23
|
+
except RuntimeError:
|
|
24
|
+
loop = None
|
|
25
|
+
|
|
26
|
+
if loop and loop.is_running():
|
|
27
|
+
with concurrent.futures.ThreadPoolExecutor(max_workers=1) as executor:
|
|
28
|
+
future = executor.submit(func, *args, **kwargs)
|
|
29
|
+
return future.result()
|
|
30
|
+
return func(*args, **kwargs)
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
class SubstackClientError(Exception):
|
|
34
|
+
"""Base exception for Substack client operations."""
|
|
35
|
+
|
|
36
|
+
pass
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
class AuthRequiredError(SubstackClientError):
|
|
40
|
+
"""Raised when Substack session is expired, invalid, or unauthenticated."""
|
|
41
|
+
|
|
42
|
+
pass
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def perform_interactive_login(browser_dir: Path | None = None) -> Path:
|
|
46
|
+
"""Launch a visible browser window for the user to log in to Substack.
|
|
47
|
+
|
|
48
|
+
Saves storage state (cookies, local storage) to storage_state.json once complete.
|
|
49
|
+
"""
|
|
50
|
+
return _run_playwright_sync(
|
|
51
|
+
_perform_interactive_login_impl, browser_dir=browser_dir
|
|
52
|
+
)
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def _perform_interactive_login_impl(browser_dir: Path | None = None) -> Path:
|
|
56
|
+
try:
|
|
57
|
+
from playwright.sync_api import sync_playwright
|
|
58
|
+
except ImportError:
|
|
59
|
+
raise SubstackClientError(
|
|
60
|
+
"Playwright is not installed. Please run 'pip install playwright && playwright install'"
|
|
61
|
+
) from None
|
|
62
|
+
|
|
63
|
+
from substack_saved_mcp.config import ensure_app_dirs
|
|
64
|
+
|
|
65
|
+
ensure_app_dirs()
|
|
66
|
+
target_dir = browser_dir or get_browser_dir()
|
|
67
|
+
state_file = target_dir / "storage_state.json"
|
|
68
|
+
|
|
69
|
+
print("Opening Substack sign-in window...")
|
|
70
|
+
print("Please log in to your Substack account in the opened browser window.")
|
|
71
|
+
print(
|
|
72
|
+
"Once logged in and viewing your feed or saved posts, press ENTER in this terminal to save session.\n"
|
|
73
|
+
)
|
|
74
|
+
|
|
75
|
+
with sync_playwright() as p:
|
|
76
|
+
# Launch visible browser
|
|
77
|
+
browser = p.chromium.launch(headless=False)
|
|
78
|
+
context_kwargs = {}
|
|
79
|
+
if state_file.exists():
|
|
80
|
+
context_kwargs["storage_state"] = str(state_file)
|
|
81
|
+
|
|
82
|
+
context = browser.new_context(**context_kwargs)
|
|
83
|
+
page = context.new_page()
|
|
84
|
+
page.goto("https://substack.com/sign-in")
|
|
85
|
+
|
|
86
|
+
input(
|
|
87
|
+
"--> Press ENTER here AFTER you have successfully completed sign-in in the browser window: "
|
|
88
|
+
)
|
|
89
|
+
|
|
90
|
+
# Verify navigation to saved page or authenticated state
|
|
91
|
+
try:
|
|
92
|
+
if page.is_closed():
|
|
93
|
+
pages = [p for p in context.pages if not p.is_closed()]
|
|
94
|
+
if pages:
|
|
95
|
+
page = pages[0]
|
|
96
|
+
else:
|
|
97
|
+
page = context.new_page()
|
|
98
|
+
page.goto(
|
|
99
|
+
"https://substack.com/saved",
|
|
100
|
+
wait_until="domcontentloaded",
|
|
101
|
+
timeout=10000,
|
|
102
|
+
)
|
|
103
|
+
except Exception as err:
|
|
104
|
+
logger.warning(f"Notice during login verification: {err}")
|
|
105
|
+
|
|
106
|
+
try:
|
|
107
|
+
context.storage_state(path=str(state_file))
|
|
108
|
+
except Exception as err:
|
|
109
|
+
logger.warning(f"Notice saving storage state: {err}")
|
|
110
|
+
|
|
111
|
+
try:
|
|
112
|
+
browser.close()
|
|
113
|
+
except Exception:
|
|
114
|
+
pass
|
|
115
|
+
|
|
116
|
+
# Restrict permissions on session storage state file
|
|
117
|
+
import os
|
|
118
|
+
|
|
119
|
+
if os.name == "posix" and state_file.exists():
|
|
120
|
+
try:
|
|
121
|
+
state_file.chmod(0o600)
|
|
122
|
+
except Exception:
|
|
123
|
+
pass
|
|
124
|
+
|
|
125
|
+
print(f"--> Authentication state saved successfully to {state_file}")
|
|
126
|
+
return state_file
|
|
127
|
+
|
|
128
|
+
|
|
129
|
+
class SubstackSavedPostsClient:
|
|
130
|
+
"""Client for fetching and managing saved Substack posts via storage_state.json."""
|
|
131
|
+
|
|
132
|
+
def __init__(self, storage_state_path: Path | None = None):
|
|
133
|
+
self.state_path = storage_state_path or get_storage_state_path()
|
|
134
|
+
self._dom_cache: list[dict[str, Any]] | None = None
|
|
135
|
+
self._api_cache: list[dict[str, Any]] | None = None
|
|
136
|
+
self._api_failed: bool = False
|
|
137
|
+
|
|
138
|
+
def reset_cache(self) -> None:
|
|
139
|
+
"""Reset cached posts extraction (reader API and DOM fallback)."""
|
|
140
|
+
self._dom_cache = None
|
|
141
|
+
self._api_cache = None
|
|
142
|
+
self._api_failed = False
|
|
143
|
+
|
|
144
|
+
def _ensure_authenticated(self) -> None:
|
|
145
|
+
"""Check if storage state exists."""
|
|
146
|
+
if not self.state_path.exists():
|
|
147
|
+
raise AuthRequiredError(
|
|
148
|
+
f"No saved Substack session found at {self.state_path}. "
|
|
149
|
+
"Please run 'substack-saved-mcp login' first to authenticate."
|
|
150
|
+
)
|
|
151
|
+
|
|
152
|
+
def fetch_saved_posts_page(
|
|
153
|
+
self, limit: int = 50, offset: int = 0
|
|
154
|
+
) -> list[dict[str, Any]]:
|
|
155
|
+
"""Fetch a page of saved posts from Substack using Playwright request context."""
|
|
156
|
+
return _run_playwright_sync(
|
|
157
|
+
self._fetch_saved_posts_page_impl, limit=limit, offset=offset
|
|
158
|
+
)
|
|
159
|
+
|
|
160
|
+
def _fetch_saved_posts_page_impl(
|
|
161
|
+
self, limit: int = 50, offset: int = 0
|
|
162
|
+
) -> list[dict[str, Any]]:
|
|
163
|
+
self._ensure_authenticated()
|
|
164
|
+
|
|
165
|
+
try:
|
|
166
|
+
from playwright.sync_api import sync_playwright
|
|
167
|
+
except ImportError:
|
|
168
|
+
raise SubstackClientError("Playwright is not installed.") from None
|
|
169
|
+
|
|
170
|
+
with sync_playwright() as p:
|
|
171
|
+
# Prefer the reader inbox API: it returns the real bookmark timestamp
|
|
172
|
+
# (saved_at) and an ISO publication date (post_date) per post. The full
|
|
173
|
+
# saved list is fetched once via cursor pagination and cached, then sliced
|
|
174
|
+
# by offset so the caller keeps a simple offset/limit interface.
|
|
175
|
+
if self._api_cache is None and not self._api_failed:
|
|
176
|
+
api_context = p.request.new_context(storage_state=str(self.state_path))
|
|
177
|
+
self._api_cache = self._fetch_all_saved_via_reader_api(api_context)
|
|
178
|
+
if self._api_cache is None:
|
|
179
|
+
self._api_failed = True
|
|
180
|
+
logger.warning(
|
|
181
|
+
"Reader inbox API unavailable; falling back to DOM extraction."
|
|
182
|
+
)
|
|
183
|
+
|
|
184
|
+
if self._api_cache is not None:
|
|
185
|
+
return self._api_cache[offset : offset + limit]
|
|
186
|
+
|
|
187
|
+
# Fallback: headless DOM extraction on https://substack.com/saved
|
|
188
|
+
return self._fetch_via_dom(
|
|
189
|
+
offset=offset, limit=limit, playwright_instance=p
|
|
190
|
+
)
|
|
191
|
+
|
|
192
|
+
@staticmethod
|
|
193
|
+
def _retry_after_seconds(res: Any, attempt: int, cap: float = 30.0) -> float:
|
|
194
|
+
"""Compute how long to wait before retrying a throttled/failed reader-API request.
|
|
195
|
+
|
|
196
|
+
Honors the ``Retry-After`` response header when it's an integer number of
|
|
197
|
+
seconds (the form Substack/most APIs send), clamped to ``cap`` so a hostile
|
|
198
|
+
or absurd value can't hang the sync. Falls back to capped exponential
|
|
199
|
+
backoff (0.5s, 1s, 2s, ...) when the header is absent or unparseable.
|
|
200
|
+
"""
|
|
201
|
+
headers = getattr(res, "headers", None) or {}
|
|
202
|
+
raw = headers.get("retry-after")
|
|
203
|
+
if raw is not None:
|
|
204
|
+
try:
|
|
205
|
+
return min(float(int(str(raw).strip())), cap)
|
|
206
|
+
except (ValueError, TypeError):
|
|
207
|
+
pass # Not an integer (possibly an HTTP-date); use backoff instead.
|
|
208
|
+
return min(0.5 * (2**attempt), cap)
|
|
209
|
+
|
|
210
|
+
def _reader_api_get(
|
|
211
|
+
self, api_context: Any, url: str, max_retries: int = 3, sleep_func=time.sleep
|
|
212
|
+
) -> Any:
|
|
213
|
+
"""GET a reader-API URL, retrying on 429/5xx with Retry-After-aware backoff.
|
|
214
|
+
|
|
215
|
+
Only transient statuses (429 Too Many Requests and 5xx server errors) are
|
|
216
|
+
retried. Auth failures (401/403) and other 4xx are returned unretried so the
|
|
217
|
+
caller's existing handling (raise ``AuthRequiredError`` / fall back) applies.
|
|
218
|
+
"""
|
|
219
|
+
res = api_context.get(url)
|
|
220
|
+
attempts = 0
|
|
221
|
+
while (res.status == 429 or 500 <= res.status < 600) and attempts < max_retries:
|
|
222
|
+
delay = self._retry_after_seconds(res, attempts)
|
|
223
|
+
logger.warning(
|
|
224
|
+
f"Reader inbox API returned {res.status}; backing off {delay}s "
|
|
225
|
+
f"before retry {attempts + 1}/{max_retries}."
|
|
226
|
+
)
|
|
227
|
+
sleep_func(delay)
|
|
228
|
+
attempts += 1
|
|
229
|
+
res = api_context.get(url)
|
|
230
|
+
return res
|
|
231
|
+
|
|
232
|
+
def _fetch_all_saved_via_reader_api(
|
|
233
|
+
self,
|
|
234
|
+
api_context: Any,
|
|
235
|
+
page_size: int = 20,
|
|
236
|
+
max_posts: int = 2000,
|
|
237
|
+
max_retries: int = 3,
|
|
238
|
+
sleep_func=time.sleep,
|
|
239
|
+
) -> list[dict[str, Any]] | None:
|
|
240
|
+
"""Fetch the full saved list from the reader inbox API via cursor pagination.
|
|
241
|
+
|
|
242
|
+
Returns a list of enriched post dicts (each carrying its real ``saved_at``,
|
|
243
|
+
ISO ``post_date``, and an attached ``publication`` object), or ``None`` if the
|
|
244
|
+
endpoint is unavailable so the caller can fall back to DOM extraction. Raises
|
|
245
|
+
``AuthRequiredError`` when the session is expired. Transient 429/5xx responses
|
|
246
|
+
are retried with ``Retry-After``-aware backoff (see ``_reader_api_get``); a
|
|
247
|
+
429 that survives all retries is treated as "unavailable" (partial list or
|
|
248
|
+
DOM fallback) rather than as silent success.
|
|
249
|
+
"""
|
|
250
|
+
from urllib.parse import quote
|
|
251
|
+
|
|
252
|
+
all_posts: list[dict[str, Any]] = []
|
|
253
|
+
seen_urls = set()
|
|
254
|
+
# "after=X" returns posts saved before X (newest first); a far-future sentinel
|
|
255
|
+
# yields the first (most recently saved) page. Substack always sends this param.
|
|
256
|
+
cursor: str = "2999-01-01T00:00:00.000Z"
|
|
257
|
+
|
|
258
|
+
while len(all_posts) < max_posts:
|
|
259
|
+
url = f"https://substack.com/api/v1/reader/posts?inboxType=saved&limit={page_size}&after={quote(cursor)}"
|
|
260
|
+
|
|
261
|
+
res = self._reader_api_get(
|
|
262
|
+
api_context, url, max_retries=max_retries, sleep_func=sleep_func
|
|
263
|
+
)
|
|
264
|
+
if res.status in (401, 403) or "sign-in" in res.url:
|
|
265
|
+
raise AuthRequiredError(
|
|
266
|
+
"Substack session has expired or is invalid. Please run 'substack-saved-mcp login'."
|
|
267
|
+
)
|
|
268
|
+
if not res.ok:
|
|
269
|
+
# Unavailable on the first page → signal fallback; mid-stream → keep what we have.
|
|
270
|
+
return all_posts if all_posts else None
|
|
271
|
+
|
|
272
|
+
try:
|
|
273
|
+
data = res.json()
|
|
274
|
+
except Exception as e:
|
|
275
|
+
logger.warning(f"JSON parsing error from reader inbox API: {e}.")
|
|
276
|
+
return all_posts if all_posts else None
|
|
277
|
+
|
|
278
|
+
posts = data.get("posts") or []
|
|
279
|
+
if not posts:
|
|
280
|
+
break
|
|
281
|
+
|
|
282
|
+
pubs_by_id = {
|
|
283
|
+
pub.get("id"): pub for pub in (data.get("publications") or [])
|
|
284
|
+
}
|
|
285
|
+
before_len = len(all_posts)
|
|
286
|
+
page_min_saved: str | None = None
|
|
287
|
+
|
|
288
|
+
for post in posts:
|
|
289
|
+
saved_at = post.get("saved_at")
|
|
290
|
+
if saved_at and (page_min_saved is None or saved_at < page_min_saved):
|
|
291
|
+
page_min_saved = saved_at
|
|
292
|
+
|
|
293
|
+
# Attach publication object and author so parse_remote_post can read them.
|
|
294
|
+
pub = pubs_by_id.get(post.get("publication_id"))
|
|
295
|
+
if pub:
|
|
296
|
+
post["publication"] = pub
|
|
297
|
+
bylines = post.get("publishedBylines") or []
|
|
298
|
+
if bylines and isinstance(bylines[0], dict) and bylines[0].get("name"):
|
|
299
|
+
post.setdefault("author_name", bylines[0]["name"])
|
|
300
|
+
|
|
301
|
+
clean = canonicalize_url(post.get("canonical_url") or "")
|
|
302
|
+
if clean and clean not in seen_urls:
|
|
303
|
+
seen_urls.add(clean)
|
|
304
|
+
all_posts.append(post)
|
|
305
|
+
|
|
306
|
+
# Stop when the server says there is no more, we made no progress, or the
|
|
307
|
+
# cursor did not advance (guards against an inclusive-boundary infinite loop).
|
|
308
|
+
if (
|
|
309
|
+
not data.get("more")
|
|
310
|
+
or not page_min_saved
|
|
311
|
+
or len(all_posts) == before_len
|
|
312
|
+
):
|
|
313
|
+
break
|
|
314
|
+
if page_min_saved == cursor:
|
|
315
|
+
break
|
|
316
|
+
cursor = page_min_saved
|
|
317
|
+
|
|
318
|
+
return all_posts
|
|
319
|
+
|
|
320
|
+
def _fetch_via_dom(
|
|
321
|
+
self, offset: int = 0, limit: int = 20, playwright_instance: Any = None
|
|
322
|
+
) -> list[dict[str, Any]]:
|
|
323
|
+
"""Fallback method: Render https://substack.com/saved in headless browser and extract post cards."""
|
|
324
|
+
if playwright_instance is not None:
|
|
325
|
+
return self._fetch_via_dom_impl(
|
|
326
|
+
offset=offset, limit=limit, playwright_instance=playwright_instance
|
|
327
|
+
)
|
|
328
|
+
return _run_playwright_sync(
|
|
329
|
+
self._fetch_via_dom_impl,
|
|
330
|
+
offset=offset,
|
|
331
|
+
limit=limit,
|
|
332
|
+
playwright_instance=playwright_instance,
|
|
333
|
+
)
|
|
334
|
+
|
|
335
|
+
def _fetch_via_dom_impl(
|
|
336
|
+
self, offset: int = 0, limit: int = 20, playwright_instance: Any = None
|
|
337
|
+
) -> list[dict[str, Any]]:
|
|
338
|
+
self._ensure_authenticated()
|
|
339
|
+
|
|
340
|
+
def _do_fetch(p):
|
|
341
|
+
browser = p.chromium.launch(headless=True)
|
|
342
|
+
context = browser.new_context(storage_state=str(self.state_path))
|
|
343
|
+
page = context.new_page()
|
|
344
|
+
|
|
345
|
+
page.goto(
|
|
346
|
+
"https://substack.com/saved",
|
|
347
|
+
wait_until="domcontentloaded",
|
|
348
|
+
timeout=15000,
|
|
349
|
+
)
|
|
350
|
+
|
|
351
|
+
if "sign-in" in page.url or page.locator("text=Sign in").count() > 0:
|
|
352
|
+
browser.close()
|
|
353
|
+
raise AuthRequiredError(
|
|
354
|
+
"Substack session has expired or is invalid. Please run 'substack-saved-mcp login'."
|
|
355
|
+
)
|
|
356
|
+
|
|
357
|
+
# Extract post elements from DOM with infinite scrolling
|
|
358
|
+
seen_urls = set()
|
|
359
|
+
results: list[dict[str, Any]] = []
|
|
360
|
+
target_count = max(offset + limit, 1000)
|
|
361
|
+
max_stale_scrolls = 6
|
|
362
|
+
stale_scrolls = 0
|
|
363
|
+
|
|
364
|
+
while len(results) < target_count and stale_scrolls < max_stale_scrolls:
|
|
365
|
+
prev_count = len(results)
|
|
366
|
+
cards = page.locator("div.reader2-post-container").all()
|
|
367
|
+
|
|
368
|
+
for card in cards:
|
|
369
|
+
try:
|
|
370
|
+
link = card.locator("a[href*='/p/']").first
|
|
371
|
+
if link.count() == 0:
|
|
372
|
+
continue
|
|
373
|
+
href = link.get_attribute("href")
|
|
374
|
+
if not href or "/p/" not in href:
|
|
375
|
+
continue
|
|
376
|
+
|
|
377
|
+
clean = canonicalize_url(href)
|
|
378
|
+
if clean in seen_urls:
|
|
379
|
+
continue
|
|
380
|
+
seen_urls.add(clean)
|
|
381
|
+
|
|
382
|
+
parsed = urlparse(clean)
|
|
383
|
+
|
|
384
|
+
def _card_text(card: Any, selector: str) -> str | None:
|
|
385
|
+
loc = card.locator(selector).first
|
|
386
|
+
if loc.count() > 0:
|
|
387
|
+
text = loc.inner_text().strip()
|
|
388
|
+
return text or None
|
|
389
|
+
return None
|
|
390
|
+
|
|
391
|
+
title = _card_text(card, ".reader2-post-title")
|
|
392
|
+
pub_name = _card_text(card, ".pub-name")
|
|
393
|
+
excerpt = _card_text(card, ".reader2-paragraph")
|
|
394
|
+
# Localized relative display string (e.g. "1 de jul.", "3h");
|
|
395
|
+
# Substack does not expose a machine-readable ISO date here.
|
|
396
|
+
published_display = _card_text(card, ".inbox-item-timestamp")
|
|
397
|
+
|
|
398
|
+
# ".reader2-item-meta" reads like "Author∙8 min read"; take the author part.
|
|
399
|
+
meta_text = _card_text(card, ".reader2-item-meta")
|
|
400
|
+
author = meta_text.split("∙")[0].strip() if meta_text else None
|
|
401
|
+
|
|
402
|
+
fallback_pub = parsed.netloc.split(".")[0].capitalize()
|
|
403
|
+
|
|
404
|
+
results.append(
|
|
405
|
+
{
|
|
406
|
+
"_dom": True,
|
|
407
|
+
"canonical_url": clean,
|
|
408
|
+
"title": title or pub_name or fallback_pub,
|
|
409
|
+
"publication_name": pub_name or fallback_pub,
|
|
410
|
+
"publication_url": f"{parsed.scheme}://{parsed.netloc}",
|
|
411
|
+
"author_name": author,
|
|
412
|
+
"excerpt": excerpt,
|
|
413
|
+
"saved_at": None,
|
|
414
|
+
"published_at": published_display,
|
|
415
|
+
}
|
|
416
|
+
)
|
|
417
|
+
except Exception:
|
|
418
|
+
continue
|
|
419
|
+
|
|
420
|
+
if len(results) >= target_count:
|
|
421
|
+
break
|
|
422
|
+
|
|
423
|
+
if len(results) == prev_count:
|
|
424
|
+
stale_scrolls += 1
|
|
425
|
+
else:
|
|
426
|
+
stale_scrolls = 0
|
|
427
|
+
|
|
428
|
+
# Scroll down and attempt clicking any load/more buttons
|
|
429
|
+
try:
|
|
430
|
+
more_btn = page.locator(
|
|
431
|
+
"button:has-text('Load more'), button:has-text('Show more'), button:has-text('More')"
|
|
432
|
+
).first
|
|
433
|
+
if more_btn.count() > 0 and more_btn.is_visible():
|
|
434
|
+
more_btn.click(timeout=1000)
|
|
435
|
+
except Exception:
|
|
436
|
+
pass
|
|
437
|
+
|
|
438
|
+
page.evaluate("window.scrollBy(0, -100)")
|
|
439
|
+
page.evaluate("window.scrollTo(0, document.body.scrollHeight)")
|
|
440
|
+
page.wait_for_timeout(1500)
|
|
441
|
+
|
|
442
|
+
browser.close()
|
|
443
|
+
return results
|
|
444
|
+
|
|
445
|
+
if self._dom_cache is None:
|
|
446
|
+
if playwright_instance is not None:
|
|
447
|
+
self._dom_cache = _do_fetch(playwright_instance)
|
|
448
|
+
else:
|
|
449
|
+
from playwright.sync_api import sync_playwright
|
|
450
|
+
|
|
451
|
+
with sync_playwright() as p:
|
|
452
|
+
self._dom_cache = _do_fetch(p)
|
|
453
|
+
|
|
454
|
+
return self._dom_cache[offset : offset + limit]
|
|
455
|
+
|
|
456
|
+
def _click_bookmark_toggle(self, page: Any) -> str:
|
|
457
|
+
"""Click the save/bookmark toggle button on a post page and report confidence.
|
|
458
|
+
|
|
459
|
+
Substack's bookmark button markup isn't officially documented (unlike the
|
|
460
|
+
saved-list card markup captured from a real page for DOM extraction), so
|
|
461
|
+
this can only detect whether *something* about the button's rendered
|
|
462
|
+
state changed after the click — not that the intended direction (save
|
|
463
|
+
vs. unsave) is what actually happened.
|
|
464
|
+
|
|
465
|
+
Returns "confirmed" (button found, clicked, and its aria-label/aria-pressed/
|
|
466
|
+
class fingerprint changed), "unconfirmed" (found and clicked but no change
|
|
467
|
+
was detectable), "not_found" (no matching button on the page), or
|
|
468
|
+
"click_failed" (found but the click itself raised).
|
|
469
|
+
"""
|
|
470
|
+
btn = page.locator(
|
|
471
|
+
"button[aria-label*='bookmark' i], button[aria-label*='save' i]"
|
|
472
|
+
).first
|
|
473
|
+
if btn.count() == 0:
|
|
474
|
+
return "not_found"
|
|
475
|
+
|
|
476
|
+
def _fingerprint():
|
|
477
|
+
try:
|
|
478
|
+
return (
|
|
479
|
+
btn.get_attribute("aria-label"),
|
|
480
|
+
btn.get_attribute("aria-pressed"),
|
|
481
|
+
btn.get_attribute("class"),
|
|
482
|
+
)
|
|
483
|
+
except Exception:
|
|
484
|
+
return None
|
|
485
|
+
|
|
486
|
+
before = _fingerprint()
|
|
487
|
+
try:
|
|
488
|
+
btn.click(timeout=3000)
|
|
489
|
+
except Exception:
|
|
490
|
+
return "click_failed"
|
|
491
|
+
|
|
492
|
+
try:
|
|
493
|
+
page.wait_for_timeout(300)
|
|
494
|
+
except Exception:
|
|
495
|
+
pass
|
|
496
|
+
|
|
497
|
+
after = _fingerprint()
|
|
498
|
+
if before is not None and after is not None and before != after:
|
|
499
|
+
return "confirmed"
|
|
500
|
+
return "unconfirmed"
|
|
501
|
+
|
|
502
|
+
def save_post(self, url: str) -> tuple[SavedPost, str]:
|
|
503
|
+
"""Bookmark a post on Substack remotely and return (extracted metadata, confirmation status).
|
|
504
|
+
|
|
505
|
+
Extracts the post's numeric ID and rich metadata from ``window._preloads``
|
|
506
|
+
(server-rendered into every post page) and, when found, calls the real
|
|
507
|
+
endpoint captured via `inspect-network` (``POST
|
|
508
|
+
https://substack.com/api/v1/posts/saved`` with body ``{"post_id": ...}``)
|
|
509
|
+
directly. Falls back to the best-effort DOM click (see
|
|
510
|
+
``_click_bookmark_toggle``) if the ID can't be extracted or that call
|
|
511
|
+
doesn't confirm; confirmation is "confirmed", "unconfirmed", "not_found",
|
|
512
|
+
or "click_failed".
|
|
513
|
+
"""
|
|
514
|
+
return _run_playwright_sync(self._save_post_impl, url=url)
|
|
515
|
+
|
|
516
|
+
def _save_post_impl(
|
|
517
|
+
self, url: str, playwright_instance: Any = None
|
|
518
|
+
) -> tuple[SavedPost, str]:
|
|
519
|
+
self._ensure_authenticated()
|
|
520
|
+
clean_url = canonicalize_url(url)
|
|
521
|
+
|
|
522
|
+
def _do_save(p):
|
|
523
|
+
browser = p.chromium.launch(headless=True)
|
|
524
|
+
context = browser.new_context(storage_state=str(self.state_path))
|
|
525
|
+
page = context.new_page()
|
|
526
|
+
|
|
527
|
+
page.goto(clean_url, wait_until="domcontentloaded")
|
|
528
|
+
|
|
529
|
+
if "sign-in" in page.url:
|
|
530
|
+
browser.close()
|
|
531
|
+
raise AuthRequiredError(
|
|
532
|
+
"Session expired during save. Please run 'substack-saved-mcp login'."
|
|
533
|
+
)
|
|
534
|
+
|
|
535
|
+
preloads = None
|
|
536
|
+
try:
|
|
537
|
+
preloads = page.evaluate("() => window._preloads")
|
|
538
|
+
except Exception:
|
|
539
|
+
pass
|
|
540
|
+
|
|
541
|
+
post_obj = (preloads or {}).get("post") or {}
|
|
542
|
+
pub_obj = (preloads or {}).get("pub") or {}
|
|
543
|
+
post_id = post_obj.get("id")
|
|
544
|
+
|
|
545
|
+
parsed = urlparse(clean_url)
|
|
546
|
+
title = (
|
|
547
|
+
post_obj.get("title")
|
|
548
|
+
or page.title().split("|")[0].strip()
|
|
549
|
+
or "Substack Post"
|
|
550
|
+
)
|
|
551
|
+
pub_name = pub_obj.get("name") or parsed.netloc.split(".")[0].capitalize()
|
|
552
|
+
excerpt = post_obj.get("description") or post_obj.get("subtitle")
|
|
553
|
+
published_at = post_obj.get("post_date")
|
|
554
|
+
audience = post_obj.get("audience")
|
|
555
|
+
|
|
556
|
+
toggle_status = self._click_bookmark_toggle(page)
|
|
557
|
+
|
|
558
|
+
# Direct API call, keyed by the real numeric post_id, as an independent
|
|
559
|
+
# (and generally more reliable) confirmation channel than the DOM click.
|
|
560
|
+
api_confirmed = False
|
|
561
|
+
if post_id is not None:
|
|
562
|
+
try:
|
|
563
|
+
api_context = p.request.new_context(
|
|
564
|
+
storage_state=str(self.state_path)
|
|
565
|
+
)
|
|
566
|
+
api_response = api_context.post(
|
|
567
|
+
"https://substack.com/api/v1/posts/saved",
|
|
568
|
+
data={"post_id": post_id},
|
|
569
|
+
)
|
|
570
|
+
api_confirmed = bool(getattr(api_response, "ok", False))
|
|
571
|
+
except Exception:
|
|
572
|
+
pass
|
|
573
|
+
|
|
574
|
+
# Save updated storage state
|
|
575
|
+
context.storage_state(path=str(self.state_path))
|
|
576
|
+
browser.close()
|
|
577
|
+
|
|
578
|
+
confirmation = (
|
|
579
|
+
"confirmed"
|
|
580
|
+
if (api_confirmed or toggle_status == "confirmed")
|
|
581
|
+
else toggle_status
|
|
582
|
+
)
|
|
583
|
+
saved_post = SavedPost(
|
|
584
|
+
substack_post_id=str(post_id) if post_id is not None else None,
|
|
585
|
+
url=clean_url,
|
|
586
|
+
title=title,
|
|
587
|
+
publication_name=pub_name,
|
|
588
|
+
publication_url=f"{parsed.scheme}://{parsed.netloc}",
|
|
589
|
+
excerpt=excerpt,
|
|
590
|
+
published_at=published_at,
|
|
591
|
+
audience=audience,
|
|
592
|
+
is_paywalled=1 if audience == "only_paid" else 0,
|
|
593
|
+
is_saved=1,
|
|
594
|
+
)
|
|
595
|
+
return saved_post, confirmation
|
|
596
|
+
|
|
597
|
+
if playwright_instance is not None:
|
|
598
|
+
return _do_save(playwright_instance)
|
|
599
|
+
from playwright.sync_api import sync_playwright
|
|
600
|
+
|
|
601
|
+
with sync_playwright() as p:
|
|
602
|
+
return _do_save(p)
|
|
603
|
+
|
|
604
|
+
def fetch_post_content(self, url: str) -> dict[str, Any]:
|
|
605
|
+
"""Fetch a post's full body HTML by visiting its page.
|
|
606
|
+
|
|
607
|
+
Reads ``window._preloads.post.body_html`` (the same server-rendered
|
|
608
|
+
blob already relied on by ``save_post`` for title/audience/description),
|
|
609
|
+
which is Substack's field name for full post content — ``parse_remote_post``
|
|
610
|
+
already expects a ``body_html`` key from the API for this reason. Returns a
|
|
611
|
+
dict with ``body_html`` (``None`` if not found on the page, e.g. the embed
|
|
612
|
+
format changed or the post is paywalled beyond this account's access) plus
|
|
613
|
+
the post's ``title`` and ``audience`` as read from the same blob.
|
|
614
|
+
"""
|
|
615
|
+
return _run_playwright_sync(self._fetch_post_content_impl, url=url)
|
|
616
|
+
|
|
617
|
+
def _fetch_post_content_impl(
|
|
618
|
+
self, url: str, playwright_instance: Any = None
|
|
619
|
+
) -> dict[str, Any]:
|
|
620
|
+
self._ensure_authenticated()
|
|
621
|
+
clean_url = canonicalize_url(url)
|
|
622
|
+
|
|
623
|
+
def _do_fetch(p):
|
|
624
|
+
browser = p.chromium.launch(headless=True)
|
|
625
|
+
context = browser.new_context(storage_state=str(self.state_path))
|
|
626
|
+
page = context.new_page()
|
|
627
|
+
|
|
628
|
+
page.goto(clean_url, wait_until="domcontentloaded")
|
|
629
|
+
|
|
630
|
+
if "sign-in" in page.url:
|
|
631
|
+
browser.close()
|
|
632
|
+
raise AuthRequiredError(
|
|
633
|
+
"Session expired while fetching content. Please run 'substack-saved-mcp login'."
|
|
634
|
+
)
|
|
635
|
+
|
|
636
|
+
preloads = None
|
|
637
|
+
try:
|
|
638
|
+
preloads = page.evaluate("() => window._preloads")
|
|
639
|
+
except Exception:
|
|
640
|
+
pass
|
|
641
|
+
|
|
642
|
+
browser.close()
|
|
643
|
+
|
|
644
|
+
post_obj = (preloads or {}).get("post") or {}
|
|
645
|
+
return {
|
|
646
|
+
"body_html": post_obj.get("body_html"),
|
|
647
|
+
"title": post_obj.get("title"),
|
|
648
|
+
"audience": post_obj.get("audience"),
|
|
649
|
+
}
|
|
650
|
+
|
|
651
|
+
if playwright_instance is not None:
|
|
652
|
+
return _do_fetch(playwright_instance)
|
|
653
|
+
from playwright.sync_api import sync_playwright
|
|
654
|
+
|
|
655
|
+
with sync_playwright() as p:
|
|
656
|
+
return _do_fetch(p)
|
|
657
|
+
|
|
658
|
+
def unsave_post(self, url: str, post_id: int | None = None) -> str:
|
|
659
|
+
"""Unbookmark a post on Substack remotely; returns a confirmation status.
|
|
660
|
+
|
|
661
|
+
When ``post_id`` (Substack's numeric post ID, i.e. ``SavedPost.substack_post_id``)
|
|
662
|
+
is known — normally the case for any post that has been through a `sync` —
|
|
663
|
+
this calls the real endpoint captured via `inspect-network`
|
|
664
|
+
(``DELETE https://substack.com/api/v1/posts/saved`` with body
|
|
665
|
+
``{"post_id": ...}``) directly, without touching the DOM, and returns
|
|
666
|
+
"confirmed" on an ok response. If that's unavailable or fails, or
|
|
667
|
+
``post_id`` is unknown, this falls back to the previous best-effort DOM
|
|
668
|
+
click — see ``_click_bookmark_toggle`` for the meaning of its statuses.
|
|
669
|
+
"""
|
|
670
|
+
return _run_playwright_sync(self._unsave_post_impl, url=url, post_id=post_id)
|
|
671
|
+
|
|
672
|
+
def _unsave_post_impl(
|
|
673
|
+
self, url: str, post_id: int | None = None, playwright_instance: Any = None
|
|
674
|
+
) -> str:
|
|
675
|
+
self._ensure_authenticated()
|
|
676
|
+
clean_url = canonicalize_url(url)
|
|
677
|
+
|
|
678
|
+
def _do_unsave(p):
|
|
679
|
+
if post_id is not None:
|
|
680
|
+
try:
|
|
681
|
+
api_context = p.request.new_context(
|
|
682
|
+
storage_state=str(self.state_path)
|
|
683
|
+
)
|
|
684
|
+
response = api_context.delete(
|
|
685
|
+
"https://substack.com/api/v1/posts/saved",
|
|
686
|
+
data={"post_id": post_id},
|
|
687
|
+
)
|
|
688
|
+
if getattr(response, "ok", False):
|
|
689
|
+
return "confirmed"
|
|
690
|
+
except Exception:
|
|
691
|
+
pass
|
|
692
|
+
# Falls through to the DOM click below if the direct API call didn't confirm.
|
|
693
|
+
|
|
694
|
+
browser = p.chromium.launch(headless=True)
|
|
695
|
+
context = browser.new_context(storage_state=str(self.state_path))
|
|
696
|
+
page = context.new_page()
|
|
697
|
+
|
|
698
|
+
page.goto(clean_url, wait_until="domcontentloaded")
|
|
699
|
+
|
|
700
|
+
if "sign-in" in page.url:
|
|
701
|
+
browser.close()
|
|
702
|
+
raise AuthRequiredError(
|
|
703
|
+
"Session expired during unsave. Please run 'substack-saved-mcp login'."
|
|
704
|
+
)
|
|
705
|
+
|
|
706
|
+
toggle_status = self._click_bookmark_toggle(page)
|
|
707
|
+
|
|
708
|
+
# Save updated state
|
|
709
|
+
context.storage_state(path=str(self.state_path))
|
|
710
|
+
browser.close()
|
|
711
|
+
return toggle_status
|
|
712
|
+
|
|
713
|
+
if playwright_instance is not None:
|
|
714
|
+
return _do_unsave(playwright_instance)
|
|
715
|
+
from playwright.sync_api import sync_playwright
|
|
716
|
+
|
|
717
|
+
with sync_playwright() as p:
|
|
718
|
+
return _do_unsave(p)
|