substack-saved-mcp 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- substack_saved_mcp/__init__.py +3 -0
- substack_saved_mcp/cli.py +413 -0
- substack_saved_mcp/config.py +55 -0
- substack_saved_mcp/content_utils.py +151 -0
- substack_saved_mcp/database.py +550 -0
- substack_saved_mcp/mcp_server.py +307 -0
- substack_saved_mcp/models.py +92 -0
- substack_saved_mcp/substack_client.py +718 -0
- substack_saved_mcp/sync.py +267 -0
- substack_saved_mcp/url_utils.py +55 -0
- substack_saved_mcp-0.1.0.dist-info/METADATA +218 -0
- substack_saved_mcp-0.1.0.dist-info/RECORD +15 -0
- substack_saved_mcp-0.1.0.dist-info/WHEEL +4 -0
- substack_saved_mcp-0.1.0.dist-info/entry_points.txt +2 -0
- substack_saved_mcp-0.1.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,550 @@
|
|
|
1
|
+
"""SQLite database schema, FTS5 virtual table indexing, and query repository."""
|
|
2
|
+
|
|
3
|
+
import sqlite3
|
|
4
|
+
from collections.abc import Generator
|
|
5
|
+
from contextlib import contextmanager
|
|
6
|
+
from datetime import UTC, datetime
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
|
|
9
|
+
from substack_saved_mcp.config import get_db_path
|
|
10
|
+
from substack_saved_mcp.models import (
|
|
11
|
+
AudienceSummary,
|
|
12
|
+
PostSummary,
|
|
13
|
+
PublicationSummary,
|
|
14
|
+
SavedPost,
|
|
15
|
+
SavedPostsStatus,
|
|
16
|
+
)
|
|
17
|
+
from substack_saved_mcp.url_utils import canonicalize_url
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
@contextmanager
|
|
21
|
+
def get_db_connection(
|
|
22
|
+
db_path: Path | None = None,
|
|
23
|
+
) -> Generator[sqlite3.Connection, None, None]:
|
|
24
|
+
"""Context manager for SQLite database connection with WAL mode enabled."""
|
|
25
|
+
target_path = db_path or get_db_path()
|
|
26
|
+
target_path.parent.mkdir(parents=True, exist_ok=True)
|
|
27
|
+
|
|
28
|
+
conn = sqlite3.connect(str(target_path), timeout=30.0)
|
|
29
|
+
conn.row_factory = sqlite3.Row
|
|
30
|
+
conn.execute("PRAGMA journal_mode = WAL;")
|
|
31
|
+
conn.execute("PRAGMA foreign_keys = ON;")
|
|
32
|
+
try:
|
|
33
|
+
yield conn
|
|
34
|
+
conn.commit()
|
|
35
|
+
except Exception:
|
|
36
|
+
conn.rollback()
|
|
37
|
+
raise
|
|
38
|
+
finally:
|
|
39
|
+
conn.close()
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def init_db(db_path: Path | None = None) -> None:
|
|
43
|
+
"""Initialize SQLite database tables, indexes, FTS5 virtual table, and triggers."""
|
|
44
|
+
from substack_saved_mcp.config import ensure_app_dirs
|
|
45
|
+
|
|
46
|
+
ensure_app_dirs()
|
|
47
|
+
with get_db_connection(db_path) as conn:
|
|
48
|
+
# Add columns to a pre-existing posts table before the index below is
|
|
49
|
+
# created, since CREATE INDEX IF NOT EXISTS still fails on a missing column.
|
|
50
|
+
cursor = conn.cursor()
|
|
51
|
+
cursor.execute(
|
|
52
|
+
"SELECT name FROM sqlite_master WHERE type='table' AND name='posts'"
|
|
53
|
+
)
|
|
54
|
+
if cursor.fetchone():
|
|
55
|
+
cursor.execute("PRAGMA table_info(posts)")
|
|
56
|
+
existing_cols = {row["name"] for row in cursor.fetchall()}
|
|
57
|
+
if "audience" not in existing_cols:
|
|
58
|
+
conn.execute("ALTER TABLE posts ADD COLUMN audience TEXT")
|
|
59
|
+
|
|
60
|
+
conn.executescript("""
|
|
61
|
+
CREATE TABLE IF NOT EXISTS posts (
|
|
62
|
+
id INTEGER PRIMARY KEY AUTOINCREMENT,
|
|
63
|
+
substack_post_id TEXT UNIQUE,
|
|
64
|
+
url TEXT NOT NULL UNIQUE,
|
|
65
|
+
title TEXT NOT NULL,
|
|
66
|
+
publication_name TEXT NOT NULL,
|
|
67
|
+
publication_url TEXT,
|
|
68
|
+
author_name TEXT,
|
|
69
|
+
published_at TEXT,
|
|
70
|
+
saved_at TEXT,
|
|
71
|
+
unsaved_at TEXT,
|
|
72
|
+
is_saved INTEGER NOT NULL DEFAULT 1,
|
|
73
|
+
excerpt TEXT,
|
|
74
|
+
content_text TEXT,
|
|
75
|
+
image_url TEXT,
|
|
76
|
+
audience TEXT,
|
|
77
|
+
is_paywalled INTEGER DEFAULT 0,
|
|
78
|
+
reading_time_minutes INTEGER,
|
|
79
|
+
word_count INTEGER,
|
|
80
|
+
created_at TEXT NOT NULL,
|
|
81
|
+
updated_at TEXT NOT NULL
|
|
82
|
+
);
|
|
83
|
+
|
|
84
|
+
CREATE INDEX IF NOT EXISTS idx_posts_url ON posts(url);
|
|
85
|
+
CREATE INDEX IF NOT EXISTS idx_posts_published_at ON posts(published_at DESC);
|
|
86
|
+
CREATE INDEX IF NOT EXISTS idx_posts_saved_at ON posts(saved_at DESC);
|
|
87
|
+
CREATE INDEX IF NOT EXISTS idx_posts_is_saved ON posts(is_saved);
|
|
88
|
+
CREATE INDEX IF NOT EXISTS idx_posts_publication ON posts(publication_name);
|
|
89
|
+
CREATE INDEX IF NOT EXISTS idx_posts_audience ON posts(audience);
|
|
90
|
+
|
|
91
|
+
CREATE TABLE IF NOT EXISTS sync_runs (
|
|
92
|
+
id INTEGER PRIMARY KEY AUTOINCREMENT,
|
|
93
|
+
started_at TEXT NOT NULL,
|
|
94
|
+
completed_at TEXT,
|
|
95
|
+
status TEXT NOT NULL,
|
|
96
|
+
sync_mode TEXT NOT NULL DEFAULT 'incremental',
|
|
97
|
+
fetched_count INTEGER DEFAULT 0,
|
|
98
|
+
upserted_count INTEGER DEFAULT 0,
|
|
99
|
+
reconciled_count INTEGER DEFAULT 0,
|
|
100
|
+
error_message TEXT
|
|
101
|
+
);
|
|
102
|
+
|
|
103
|
+
-- FTS5 Full-Text Search Virtual Table
|
|
104
|
+
CREATE VIRTUAL TABLE IF NOT EXISTS posts_fts USING fts5(
|
|
105
|
+
title,
|
|
106
|
+
publication_name,
|
|
107
|
+
author_name,
|
|
108
|
+
excerpt,
|
|
109
|
+
content_text,
|
|
110
|
+
content='posts',
|
|
111
|
+
content_rowid='id'
|
|
112
|
+
);
|
|
113
|
+
|
|
114
|
+
-- Triggers to synchronize FTS5 index on INSERT, UPDATE, and DELETE
|
|
115
|
+
CREATE TRIGGER IF NOT EXISTS posts_ai AFTER INSERT ON posts BEGIN
|
|
116
|
+
INSERT INTO posts_fts(rowid, title, publication_name, author_name, excerpt, content_text)
|
|
117
|
+
VALUES (new.id, new.title, new.publication_name, new.author_name, new.excerpt, new.content_text);
|
|
118
|
+
END;
|
|
119
|
+
|
|
120
|
+
CREATE TRIGGER IF NOT EXISTS posts_ad AFTER DELETE ON posts BEGIN
|
|
121
|
+
INSERT INTO posts_fts(posts_fts, rowid, title, publication_name, author_name, excerpt, content_text)
|
|
122
|
+
VALUES('delete', old.id, old.title, old.publication_name, old.author_name, old.excerpt, old.content_text);
|
|
123
|
+
END;
|
|
124
|
+
|
|
125
|
+
CREATE TRIGGER IF NOT EXISTS posts_au AFTER UPDATE ON posts BEGIN
|
|
126
|
+
INSERT INTO posts_fts(posts_fts, rowid, title, publication_name, author_name, excerpt, content_text)
|
|
127
|
+
VALUES('delete', old.id, old.title, old.publication_name, old.author_name, old.excerpt, old.content_text);
|
|
128
|
+
INSERT INTO posts_fts(rowid, title, publication_name, author_name, excerpt, content_text)
|
|
129
|
+
VALUES (new.id, new.title, new.publication_name, new.author_name, new.excerpt, new.content_text);
|
|
130
|
+
END;
|
|
131
|
+
""")
|
|
132
|
+
# Additive migration for databases created before reconciliation tracking existed.
|
|
133
|
+
try:
|
|
134
|
+
conn.execute(
|
|
135
|
+
"ALTER TABLE sync_runs ADD COLUMN reconciled_count INTEGER DEFAULT 0"
|
|
136
|
+
)
|
|
137
|
+
except sqlite3.OperationalError:
|
|
138
|
+
pass # column already present
|
|
139
|
+
|
|
140
|
+
|
|
141
|
+
def upsert_post(post: SavedPost, db_path: Path | None = None) -> SavedPost:
|
|
142
|
+
"""Insert or update a post in the database. Returns the updated SavedPost object."""
|
|
143
|
+
now_iso = datetime.now(UTC).isoformat()
|
|
144
|
+
clean_url = canonicalize_url(post.url)
|
|
145
|
+
|
|
146
|
+
with get_db_connection(db_path) as conn:
|
|
147
|
+
cursor = conn.cursor()
|
|
148
|
+
# Check if record already exists by substack_post_id or URL
|
|
149
|
+
existing = None
|
|
150
|
+
if post.substack_post_id:
|
|
151
|
+
cursor.execute(
|
|
152
|
+
"SELECT id, created_at, saved_at FROM posts WHERE substack_post_id = ?",
|
|
153
|
+
(post.substack_post_id,),
|
|
154
|
+
)
|
|
155
|
+
existing = cursor.fetchone()
|
|
156
|
+
if not existing and clean_url:
|
|
157
|
+
cursor.execute(
|
|
158
|
+
"SELECT id, created_at, saved_at FROM posts WHERE url = ?", (clean_url,)
|
|
159
|
+
)
|
|
160
|
+
existing = cursor.fetchone()
|
|
161
|
+
|
|
162
|
+
created_at = existing["created_at"] if existing else now_iso
|
|
163
|
+
# Preserve a known save time; never fabricate one from the DB-insert moment.
|
|
164
|
+
# When the source does not expose the original Substack bookmark time, leave
|
|
165
|
+
# saved_at NULL rather than stamping now().
|
|
166
|
+
saved_at = post.saved_at or (existing["saved_at"] if existing else None)
|
|
167
|
+
|
|
168
|
+
cursor.execute(
|
|
169
|
+
"""
|
|
170
|
+
INSERT INTO posts (
|
|
171
|
+
substack_post_id, url, title, publication_name, publication_url,
|
|
172
|
+
author_name, published_at, saved_at, unsaved_at, is_saved,
|
|
173
|
+
excerpt, content_text, image_url, audience, is_paywalled, reading_time_minutes,
|
|
174
|
+
word_count, created_at, updated_at
|
|
175
|
+
) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)
|
|
176
|
+
ON CONFLICT(url) DO UPDATE SET
|
|
177
|
+
substack_post_id = COALESCE(excluded.substack_post_id, posts.substack_post_id),
|
|
178
|
+
title = excluded.title,
|
|
179
|
+
publication_name = excluded.publication_name,
|
|
180
|
+
publication_url = COALESCE(excluded.publication_url, posts.publication_url),
|
|
181
|
+
author_name = COALESCE(excluded.author_name, posts.author_name),
|
|
182
|
+
published_at = COALESCE(excluded.published_at, posts.published_at),
|
|
183
|
+
saved_at = COALESCE(excluded.saved_at, posts.saved_at),
|
|
184
|
+
unsaved_at = excluded.unsaved_at,
|
|
185
|
+
is_saved = excluded.is_saved,
|
|
186
|
+
excerpt = COALESCE(excluded.excerpt, posts.excerpt),
|
|
187
|
+
content_text = COALESCE(excluded.content_text, posts.content_text),
|
|
188
|
+
image_url = COALESCE(excluded.image_url, posts.image_url),
|
|
189
|
+
audience = COALESCE(excluded.audience, posts.audience),
|
|
190
|
+
is_paywalled = excluded.is_paywalled,
|
|
191
|
+
reading_time_minutes = COALESCE(excluded.reading_time_minutes, posts.reading_time_minutes),
|
|
192
|
+
word_count = COALESCE(excluded.word_count, posts.word_count),
|
|
193
|
+
updated_at = excluded.updated_at
|
|
194
|
+
""",
|
|
195
|
+
(
|
|
196
|
+
post.substack_post_id,
|
|
197
|
+
clean_url,
|
|
198
|
+
post.title,
|
|
199
|
+
post.publication_name,
|
|
200
|
+
post.publication_url,
|
|
201
|
+
post.author_name,
|
|
202
|
+
post.published_at,
|
|
203
|
+
saved_at,
|
|
204
|
+
post.unsaved_at,
|
|
205
|
+
post.is_saved,
|
|
206
|
+
post.excerpt,
|
|
207
|
+
post.content_text,
|
|
208
|
+
post.image_url,
|
|
209
|
+
post.audience,
|
|
210
|
+
post.is_paywalled,
|
|
211
|
+
post.reading_time_minutes,
|
|
212
|
+
post.word_count,
|
|
213
|
+
created_at,
|
|
214
|
+
now_iso,
|
|
215
|
+
),
|
|
216
|
+
)
|
|
217
|
+
|
|
218
|
+
post_id = cursor.lastrowid if not existing else existing["id"]
|
|
219
|
+
cursor.execute("SELECT * FROM posts WHERE id = ?", (post_id,))
|
|
220
|
+
row = cursor.fetchone()
|
|
221
|
+
return SavedPost(**dict(row))
|
|
222
|
+
|
|
223
|
+
|
|
224
|
+
def soft_delete_post(
|
|
225
|
+
url_or_id: str | int, db_path: Path | None = None
|
|
226
|
+
) -> SavedPost | None:
|
|
227
|
+
"""Mark a post as unsaved (is_saved = 0, unsaved_at = now)."""
|
|
228
|
+
now_iso = datetime.now(UTC).isoformat()
|
|
229
|
+
with get_db_connection(db_path) as conn:
|
|
230
|
+
cursor = conn.cursor()
|
|
231
|
+
|
|
232
|
+
if isinstance(url_or_id, int) or (
|
|
233
|
+
isinstance(url_or_id, str) and url_or_id.isdigit()
|
|
234
|
+
):
|
|
235
|
+
cursor.execute("SELECT * FROM posts WHERE id = ?", (int(url_or_id),))
|
|
236
|
+
else:
|
|
237
|
+
clean_url = canonicalize_url(str(url_or_id))
|
|
238
|
+
cursor.execute(
|
|
239
|
+
"SELECT * FROM posts WHERE url = ? OR substack_post_id = ?",
|
|
240
|
+
(clean_url, str(url_or_id)),
|
|
241
|
+
)
|
|
242
|
+
|
|
243
|
+
row = cursor.fetchone()
|
|
244
|
+
if not row:
|
|
245
|
+
return None
|
|
246
|
+
|
|
247
|
+
cursor.execute(
|
|
248
|
+
"""
|
|
249
|
+
UPDATE posts
|
|
250
|
+
SET is_saved = 0, unsaved_at = ?, updated_at = ?
|
|
251
|
+
WHERE id = ?
|
|
252
|
+
""",
|
|
253
|
+
(now_iso, now_iso, row["id"]),
|
|
254
|
+
)
|
|
255
|
+
|
|
256
|
+
cursor.execute("SELECT * FROM posts WHERE id = ?", (row["id"],))
|
|
257
|
+
return SavedPost(**dict(cursor.fetchone()))
|
|
258
|
+
|
|
259
|
+
|
|
260
|
+
def reconcile_unsaved_posts(remote_urls: list[str], db_path: Path | None = None) -> int:
|
|
261
|
+
"""Soft-delete locally-active posts absent from a complete remote saved-URL set.
|
|
262
|
+
|
|
263
|
+
Intended for use only after a full sync has enumerated every currently-saved
|
|
264
|
+
post on Substack (an incremental sync may stop early and would wrongly treat
|
|
265
|
+
un-refetched posts as removed). Skips entirely when remote_urls is empty, since
|
|
266
|
+
that's more likely a fetch problem than genuine mass-unsaving.
|
|
267
|
+
"""
|
|
268
|
+
if not remote_urls:
|
|
269
|
+
return 0
|
|
270
|
+
clean_urls = {canonicalize_url(u) for u in remote_urls if u}
|
|
271
|
+
|
|
272
|
+
now_iso = datetime.now(UTC).isoformat()
|
|
273
|
+
with get_db_connection(db_path) as conn:
|
|
274
|
+
cursor = conn.cursor()
|
|
275
|
+
cursor.execute("SELECT id, url FROM posts WHERE is_saved = 1")
|
|
276
|
+
stale_ids = [
|
|
277
|
+
row["id"] for row in cursor.fetchall() if row["url"] not in clean_urls
|
|
278
|
+
]
|
|
279
|
+
if not stale_ids:
|
|
280
|
+
return 0
|
|
281
|
+
|
|
282
|
+
cursor.executemany(
|
|
283
|
+
"UPDATE posts SET is_saved = 0, unsaved_at = ?, updated_at = ? WHERE id = ?",
|
|
284
|
+
[(now_iso, now_iso, post_id) for post_id in stale_ids],
|
|
285
|
+
)
|
|
286
|
+
return len(stale_ids)
|
|
287
|
+
|
|
288
|
+
|
|
289
|
+
def get_post(url_or_id: str | int, db_path: Path | None = None) -> SavedPost | None:
|
|
290
|
+
"""Retrieve full post record by local ID, Substack post ID, or URL."""
|
|
291
|
+
with get_db_connection(db_path) as conn:
|
|
292
|
+
cursor = conn.cursor()
|
|
293
|
+
if isinstance(url_or_id, int) or (
|
|
294
|
+
isinstance(url_or_id, str) and url_or_id.isdigit()
|
|
295
|
+
):
|
|
296
|
+
cursor.execute("SELECT * FROM posts WHERE id = ?", (int(url_or_id),))
|
|
297
|
+
else:
|
|
298
|
+
clean_url = canonicalize_url(str(url_or_id))
|
|
299
|
+
cursor.execute(
|
|
300
|
+
"SELECT * FROM posts WHERE url = ? OR substack_post_id = ?",
|
|
301
|
+
(clean_url, str(url_or_id)),
|
|
302
|
+
)
|
|
303
|
+
row = cursor.fetchone()
|
|
304
|
+
return SavedPost(**dict(row)) if row else None
|
|
305
|
+
|
|
306
|
+
|
|
307
|
+
def list_posts(
|
|
308
|
+
limit: int = 20,
|
|
309
|
+
offset: int = 0,
|
|
310
|
+
publication: str | None = None,
|
|
311
|
+
audience: str | None = None,
|
|
312
|
+
sort_by: str = "saved_at",
|
|
313
|
+
is_saved_only: bool = True,
|
|
314
|
+
db_path: Path | None = None,
|
|
315
|
+
) -> list[PostSummary]:
|
|
316
|
+
"""List posts with pagination, publication/audience filters, and sorting."""
|
|
317
|
+
order_col = "published_at" if sort_by == "published_at" else "saved_at"
|
|
318
|
+
where_clauses = []
|
|
319
|
+
params: list[str | int] = []
|
|
320
|
+
|
|
321
|
+
if is_saved_only:
|
|
322
|
+
where_clauses.append("is_saved = 1")
|
|
323
|
+
if publication:
|
|
324
|
+
where_clauses.append("LOWER(publication_name) LIKE LOWER(?)")
|
|
325
|
+
params.append(f"%{publication}%")
|
|
326
|
+
if audience:
|
|
327
|
+
where_clauses.append("LOWER(audience) = LOWER(?)")
|
|
328
|
+
params.append(audience)
|
|
329
|
+
|
|
330
|
+
where_sql = f"WHERE {' AND '.join(where_clauses)}" if where_clauses else ""
|
|
331
|
+
sql = f"""
|
|
332
|
+
SELECT id, substack_post_id, url, title, publication_name, author_name,
|
|
333
|
+
published_at, saved_at, is_saved, excerpt, image_url, audience, is_paywalled,
|
|
334
|
+
reading_time_minutes, word_count
|
|
335
|
+
FROM posts
|
|
336
|
+
{where_sql}
|
|
337
|
+
ORDER BY {order_col} DESC NULLS LAST
|
|
338
|
+
LIMIT ? OFFSET ?
|
|
339
|
+
"""
|
|
340
|
+
params.extend([limit, offset])
|
|
341
|
+
|
|
342
|
+
with get_db_connection(db_path) as conn:
|
|
343
|
+
cursor = conn.cursor()
|
|
344
|
+
cursor.execute(sql, params)
|
|
345
|
+
return [PostSummary(**dict(r)) for r in cursor.fetchall()]
|
|
346
|
+
|
|
347
|
+
|
|
348
|
+
def search_posts(
|
|
349
|
+
query: str,
|
|
350
|
+
publication: str | None = None,
|
|
351
|
+
audience: str | None = None,
|
|
352
|
+
published_after: str | None = None,
|
|
353
|
+
published_before: str | None = None,
|
|
354
|
+
saved_after: str | None = None,
|
|
355
|
+
saved_before: str | None = None,
|
|
356
|
+
limit: int = 20,
|
|
357
|
+
is_saved_only: bool = True,
|
|
358
|
+
db_path: Path | None = None,
|
|
359
|
+
) -> list[PostSummary]:
|
|
360
|
+
"""Full-text search over posts using FTS5 BM25 relevance ranking and metadata filters."""
|
|
361
|
+
where_clauses = ["posts_fts MATCH ?"]
|
|
362
|
+
params: list[str | int] = [query]
|
|
363
|
+
|
|
364
|
+
if is_saved_only:
|
|
365
|
+
where_clauses.append("p.is_saved = 1")
|
|
366
|
+
if publication:
|
|
367
|
+
where_clauses.append("LOWER(p.publication_name) LIKE LOWER(?)")
|
|
368
|
+
params.append(f"%{publication}%")
|
|
369
|
+
if audience:
|
|
370
|
+
where_clauses.append("LOWER(p.audience) = LOWER(?)")
|
|
371
|
+
params.append(audience)
|
|
372
|
+
if published_after:
|
|
373
|
+
where_clauses.append("p.published_at >= ?")
|
|
374
|
+
params.append(published_after)
|
|
375
|
+
if published_before:
|
|
376
|
+
where_clauses.append("p.published_at <= ?")
|
|
377
|
+
params.append(published_before)
|
|
378
|
+
if saved_after:
|
|
379
|
+
where_clauses.append("p.saved_at >= ?")
|
|
380
|
+
params.append(saved_after)
|
|
381
|
+
if saved_before:
|
|
382
|
+
where_clauses.append("p.saved_at <= ?")
|
|
383
|
+
params.append(saved_before)
|
|
384
|
+
|
|
385
|
+
sql = f"""
|
|
386
|
+
SELECT p.id, p.substack_post_id, p.url, p.title, p.publication_name, p.author_name,
|
|
387
|
+
p.published_at, p.saved_at, p.is_saved, p.excerpt, p.image_url, p.audience, p.is_paywalled,
|
|
388
|
+
p.reading_time_minutes, p.word_count
|
|
389
|
+
FROM posts_fts fts
|
|
390
|
+
JOIN posts p ON fts.rowid = p.id
|
|
391
|
+
WHERE {" AND ".join(where_clauses)}
|
|
392
|
+
ORDER BY fts.rank
|
|
393
|
+
LIMIT ?
|
|
394
|
+
"""
|
|
395
|
+
params.append(limit)
|
|
396
|
+
|
|
397
|
+
with get_db_connection(db_path) as conn:
|
|
398
|
+
cursor = conn.cursor()
|
|
399
|
+
try:
|
|
400
|
+
cursor.execute(sql, params)
|
|
401
|
+
return [PostSummary(**dict(r)) for r in cursor.fetchall()]
|
|
402
|
+
except sqlite3.OperationalError:
|
|
403
|
+
# Fallback to standard LIKE if FTS query syntax is special/malformed
|
|
404
|
+
like_query = f"%{query}%"
|
|
405
|
+
fallback_where = [
|
|
406
|
+
"(title LIKE ? OR excerpt LIKE ? OR publication_name LIKE ? OR author_name LIKE ?)",
|
|
407
|
+
"is_saved = 1",
|
|
408
|
+
]
|
|
409
|
+
fallback_params: list[str | int] = [
|
|
410
|
+
like_query,
|
|
411
|
+
like_query,
|
|
412
|
+
like_query,
|
|
413
|
+
like_query,
|
|
414
|
+
]
|
|
415
|
+
if audience:
|
|
416
|
+
fallback_where.append("LOWER(audience) = LOWER(?)")
|
|
417
|
+
fallback_params.append(audience)
|
|
418
|
+
fallback_sql = f"""
|
|
419
|
+
SELECT id, substack_post_id, url, title, publication_name, author_name,
|
|
420
|
+
published_at, saved_at, is_saved, excerpt, image_url, audience, is_paywalled,
|
|
421
|
+
reading_time_minutes, word_count
|
|
422
|
+
FROM posts
|
|
423
|
+
WHERE {" AND ".join(fallback_where)}
|
|
424
|
+
ORDER BY saved_at DESC
|
|
425
|
+
LIMIT ?
|
|
426
|
+
"""
|
|
427
|
+
fallback_params.append(limit)
|
|
428
|
+
cursor.execute(fallback_sql, fallback_params)
|
|
429
|
+
return [PostSummary(**dict(r)) for r in cursor.fetchall()]
|
|
430
|
+
|
|
431
|
+
|
|
432
|
+
def list_publications(db_path: Path | None = None) -> list[PublicationSummary]:
|
|
433
|
+
"""Return all unique publications in cache with active saved post count."""
|
|
434
|
+
sql = """
|
|
435
|
+
SELECT publication_name, MAX(publication_url) as publication_url, COUNT(*) as post_count
|
|
436
|
+
FROM posts
|
|
437
|
+
WHERE is_saved = 1
|
|
438
|
+
GROUP BY publication_name
|
|
439
|
+
ORDER BY post_count DESC, publication_name ASC
|
|
440
|
+
"""
|
|
441
|
+
with get_db_connection(db_path) as conn:
|
|
442
|
+
cursor = conn.cursor()
|
|
443
|
+
cursor.execute(sql)
|
|
444
|
+
return [PublicationSummary(**dict(r)) for r in cursor.fetchall()]
|
|
445
|
+
|
|
446
|
+
|
|
447
|
+
def list_audiences(db_path: Path | None = None) -> list[AudienceSummary]:
|
|
448
|
+
"""Return distinct audience tiers present in cache with active saved post counts.
|
|
449
|
+
|
|
450
|
+
Discovers the actual values in use (e.g. "everyone", "only_paid") rather than
|
|
451
|
+
hardcoding Substack's audience enum, since it's not officially documented and
|
|
452
|
+
may grow (e.g. "only_founding", "preview").
|
|
453
|
+
"""
|
|
454
|
+
sql = """
|
|
455
|
+
SELECT audience, COUNT(*) as post_count
|
|
456
|
+
FROM posts
|
|
457
|
+
WHERE is_saved = 1
|
|
458
|
+
GROUP BY audience
|
|
459
|
+
ORDER BY post_count DESC
|
|
460
|
+
"""
|
|
461
|
+
with get_db_connection(db_path) as conn:
|
|
462
|
+
cursor = conn.cursor()
|
|
463
|
+
cursor.execute(sql)
|
|
464
|
+
return [AudienceSummary(**dict(r)) for r in cursor.fetchall()]
|
|
465
|
+
|
|
466
|
+
|
|
467
|
+
def get_status(db_path: Path | None = None) -> SavedPostsStatus:
|
|
468
|
+
"""Return metrics and statistics for local SQLite database."""
|
|
469
|
+
target_path = db_path or get_db_path()
|
|
470
|
+
with get_db_connection(target_path) as conn:
|
|
471
|
+
cursor = conn.cursor()
|
|
472
|
+
|
|
473
|
+
cursor.execute("SELECT COUNT(*) FROM posts WHERE is_saved = 1")
|
|
474
|
+
total_saved = cursor.fetchone()[0]
|
|
475
|
+
|
|
476
|
+
cursor.execute("SELECT COUNT(*) FROM posts WHERE is_saved = 0")
|
|
477
|
+
total_unsaved = cursor.fetchone()[0]
|
|
478
|
+
|
|
479
|
+
cursor.execute(
|
|
480
|
+
"SELECT COUNT(DISTINCT publication_name) FROM posts WHERE is_saved = 1"
|
|
481
|
+
)
|
|
482
|
+
total_pubs = cursor.fetchone()[0]
|
|
483
|
+
|
|
484
|
+
cursor.execute("""
|
|
485
|
+
SELECT completed_at, status FROM sync_runs
|
|
486
|
+
WHERE status = 'success'
|
|
487
|
+
ORDER BY id DESC LIMIT 1
|
|
488
|
+
""")
|
|
489
|
+
last_success_row = cursor.fetchone()
|
|
490
|
+
last_success = last_success_row["completed_at"] if last_success_row else None
|
|
491
|
+
|
|
492
|
+
cursor.execute("SELECT status FROM sync_runs ORDER BY id DESC LIMIT 1")
|
|
493
|
+
last_status_row = cursor.fetchone()
|
|
494
|
+
last_status = last_status_row["status"] if last_status_row else None
|
|
495
|
+
|
|
496
|
+
return SavedPostsStatus(
|
|
497
|
+
total_saved_posts=total_saved,
|
|
498
|
+
total_unsaved_posts=total_unsaved,
|
|
499
|
+
total_publications=total_pubs,
|
|
500
|
+
last_successful_sync=last_success,
|
|
501
|
+
last_sync_status=last_status,
|
|
502
|
+
database_path=str(target_path),
|
|
503
|
+
)
|
|
504
|
+
|
|
505
|
+
|
|
506
|
+
def start_sync_run(sync_mode: str = "incremental", db_path: Path | None = None) -> int:
|
|
507
|
+
"""Create a sync_runs entry and return its ID."""
|
|
508
|
+
now_iso = datetime.now(UTC).isoformat()
|
|
509
|
+
with get_db_connection(db_path) as conn:
|
|
510
|
+
cursor = conn.cursor()
|
|
511
|
+
cursor.execute(
|
|
512
|
+
"""
|
|
513
|
+
INSERT INTO sync_runs (started_at, status, sync_mode)
|
|
514
|
+
VALUES (?, 'running', ?)
|
|
515
|
+
""",
|
|
516
|
+
(now_iso, sync_mode),
|
|
517
|
+
)
|
|
518
|
+
return cursor.lastrowid # type: ignore
|
|
519
|
+
|
|
520
|
+
|
|
521
|
+
def finish_sync_run(
|
|
522
|
+
sync_id: int,
|
|
523
|
+
status: str,
|
|
524
|
+
fetched_count: int,
|
|
525
|
+
upserted_count: int,
|
|
526
|
+
reconciled_count: int = 0,
|
|
527
|
+
error_message: str | None = None,
|
|
528
|
+
db_path: Path | None = None,
|
|
529
|
+
) -> None:
|
|
530
|
+
"""Update a sync_runs record upon completion or failure."""
|
|
531
|
+
now_iso = datetime.now(UTC).isoformat()
|
|
532
|
+
with get_db_connection(db_path) as conn:
|
|
533
|
+
cursor = conn.cursor()
|
|
534
|
+
cursor.execute(
|
|
535
|
+
"""
|
|
536
|
+
UPDATE sync_runs
|
|
537
|
+
SET completed_at = ?, status = ?, fetched_count = ?, upserted_count = ?,
|
|
538
|
+
reconciled_count = ?, error_message = ?
|
|
539
|
+
WHERE id = ?
|
|
540
|
+
""",
|
|
541
|
+
(
|
|
542
|
+
now_iso,
|
|
543
|
+
status,
|
|
544
|
+
fetched_count,
|
|
545
|
+
upserted_count,
|
|
546
|
+
reconciled_count,
|
|
547
|
+
error_message,
|
|
548
|
+
sync_id,
|
|
549
|
+
),
|
|
550
|
+
)
|