substack-saved-mcp 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,550 @@
1
+ """SQLite database schema, FTS5 virtual table indexing, and query repository."""
2
+
3
+ import sqlite3
4
+ from collections.abc import Generator
5
+ from contextlib import contextmanager
6
+ from datetime import UTC, datetime
7
+ from pathlib import Path
8
+
9
+ from substack_saved_mcp.config import get_db_path
10
+ from substack_saved_mcp.models import (
11
+ AudienceSummary,
12
+ PostSummary,
13
+ PublicationSummary,
14
+ SavedPost,
15
+ SavedPostsStatus,
16
+ )
17
+ from substack_saved_mcp.url_utils import canonicalize_url
18
+
19
+
20
+ @contextmanager
21
+ def get_db_connection(
22
+ db_path: Path | None = None,
23
+ ) -> Generator[sqlite3.Connection, None, None]:
24
+ """Context manager for SQLite database connection with WAL mode enabled."""
25
+ target_path = db_path or get_db_path()
26
+ target_path.parent.mkdir(parents=True, exist_ok=True)
27
+
28
+ conn = sqlite3.connect(str(target_path), timeout=30.0)
29
+ conn.row_factory = sqlite3.Row
30
+ conn.execute("PRAGMA journal_mode = WAL;")
31
+ conn.execute("PRAGMA foreign_keys = ON;")
32
+ try:
33
+ yield conn
34
+ conn.commit()
35
+ except Exception:
36
+ conn.rollback()
37
+ raise
38
+ finally:
39
+ conn.close()
40
+
41
+
42
+ def init_db(db_path: Path | None = None) -> None:
43
+ """Initialize SQLite database tables, indexes, FTS5 virtual table, and triggers."""
44
+ from substack_saved_mcp.config import ensure_app_dirs
45
+
46
+ ensure_app_dirs()
47
+ with get_db_connection(db_path) as conn:
48
+ # Add columns to a pre-existing posts table before the index below is
49
+ # created, since CREATE INDEX IF NOT EXISTS still fails on a missing column.
50
+ cursor = conn.cursor()
51
+ cursor.execute(
52
+ "SELECT name FROM sqlite_master WHERE type='table' AND name='posts'"
53
+ )
54
+ if cursor.fetchone():
55
+ cursor.execute("PRAGMA table_info(posts)")
56
+ existing_cols = {row["name"] for row in cursor.fetchall()}
57
+ if "audience" not in existing_cols:
58
+ conn.execute("ALTER TABLE posts ADD COLUMN audience TEXT")
59
+
60
+ conn.executescript("""
61
+ CREATE TABLE IF NOT EXISTS posts (
62
+ id INTEGER PRIMARY KEY AUTOINCREMENT,
63
+ substack_post_id TEXT UNIQUE,
64
+ url TEXT NOT NULL UNIQUE,
65
+ title TEXT NOT NULL,
66
+ publication_name TEXT NOT NULL,
67
+ publication_url TEXT,
68
+ author_name TEXT,
69
+ published_at TEXT,
70
+ saved_at TEXT,
71
+ unsaved_at TEXT,
72
+ is_saved INTEGER NOT NULL DEFAULT 1,
73
+ excerpt TEXT,
74
+ content_text TEXT,
75
+ image_url TEXT,
76
+ audience TEXT,
77
+ is_paywalled INTEGER DEFAULT 0,
78
+ reading_time_minutes INTEGER,
79
+ word_count INTEGER,
80
+ created_at TEXT NOT NULL,
81
+ updated_at TEXT NOT NULL
82
+ );
83
+
84
+ CREATE INDEX IF NOT EXISTS idx_posts_url ON posts(url);
85
+ CREATE INDEX IF NOT EXISTS idx_posts_published_at ON posts(published_at DESC);
86
+ CREATE INDEX IF NOT EXISTS idx_posts_saved_at ON posts(saved_at DESC);
87
+ CREATE INDEX IF NOT EXISTS idx_posts_is_saved ON posts(is_saved);
88
+ CREATE INDEX IF NOT EXISTS idx_posts_publication ON posts(publication_name);
89
+ CREATE INDEX IF NOT EXISTS idx_posts_audience ON posts(audience);
90
+
91
+ CREATE TABLE IF NOT EXISTS sync_runs (
92
+ id INTEGER PRIMARY KEY AUTOINCREMENT,
93
+ started_at TEXT NOT NULL,
94
+ completed_at TEXT,
95
+ status TEXT NOT NULL,
96
+ sync_mode TEXT NOT NULL DEFAULT 'incremental',
97
+ fetched_count INTEGER DEFAULT 0,
98
+ upserted_count INTEGER DEFAULT 0,
99
+ reconciled_count INTEGER DEFAULT 0,
100
+ error_message TEXT
101
+ );
102
+
103
+ -- FTS5 Full-Text Search Virtual Table
104
+ CREATE VIRTUAL TABLE IF NOT EXISTS posts_fts USING fts5(
105
+ title,
106
+ publication_name,
107
+ author_name,
108
+ excerpt,
109
+ content_text,
110
+ content='posts',
111
+ content_rowid='id'
112
+ );
113
+
114
+ -- Triggers to synchronize FTS5 index on INSERT, UPDATE, and DELETE
115
+ CREATE TRIGGER IF NOT EXISTS posts_ai AFTER INSERT ON posts BEGIN
116
+ INSERT INTO posts_fts(rowid, title, publication_name, author_name, excerpt, content_text)
117
+ VALUES (new.id, new.title, new.publication_name, new.author_name, new.excerpt, new.content_text);
118
+ END;
119
+
120
+ CREATE TRIGGER IF NOT EXISTS posts_ad AFTER DELETE ON posts BEGIN
121
+ INSERT INTO posts_fts(posts_fts, rowid, title, publication_name, author_name, excerpt, content_text)
122
+ VALUES('delete', old.id, old.title, old.publication_name, old.author_name, old.excerpt, old.content_text);
123
+ END;
124
+
125
+ CREATE TRIGGER IF NOT EXISTS posts_au AFTER UPDATE ON posts BEGIN
126
+ INSERT INTO posts_fts(posts_fts, rowid, title, publication_name, author_name, excerpt, content_text)
127
+ VALUES('delete', old.id, old.title, old.publication_name, old.author_name, old.excerpt, old.content_text);
128
+ INSERT INTO posts_fts(rowid, title, publication_name, author_name, excerpt, content_text)
129
+ VALUES (new.id, new.title, new.publication_name, new.author_name, new.excerpt, new.content_text);
130
+ END;
131
+ """)
132
+ # Additive migration for databases created before reconciliation tracking existed.
133
+ try:
134
+ conn.execute(
135
+ "ALTER TABLE sync_runs ADD COLUMN reconciled_count INTEGER DEFAULT 0"
136
+ )
137
+ except sqlite3.OperationalError:
138
+ pass # column already present
139
+
140
+
141
+ def upsert_post(post: SavedPost, db_path: Path | None = None) -> SavedPost:
142
+ """Insert or update a post in the database. Returns the updated SavedPost object."""
143
+ now_iso = datetime.now(UTC).isoformat()
144
+ clean_url = canonicalize_url(post.url)
145
+
146
+ with get_db_connection(db_path) as conn:
147
+ cursor = conn.cursor()
148
+ # Check if record already exists by substack_post_id or URL
149
+ existing = None
150
+ if post.substack_post_id:
151
+ cursor.execute(
152
+ "SELECT id, created_at, saved_at FROM posts WHERE substack_post_id = ?",
153
+ (post.substack_post_id,),
154
+ )
155
+ existing = cursor.fetchone()
156
+ if not existing and clean_url:
157
+ cursor.execute(
158
+ "SELECT id, created_at, saved_at FROM posts WHERE url = ?", (clean_url,)
159
+ )
160
+ existing = cursor.fetchone()
161
+
162
+ created_at = existing["created_at"] if existing else now_iso
163
+ # Preserve a known save time; never fabricate one from the DB-insert moment.
164
+ # When the source does not expose the original Substack bookmark time, leave
165
+ # saved_at NULL rather than stamping now().
166
+ saved_at = post.saved_at or (existing["saved_at"] if existing else None)
167
+
168
+ cursor.execute(
169
+ """
170
+ INSERT INTO posts (
171
+ substack_post_id, url, title, publication_name, publication_url,
172
+ author_name, published_at, saved_at, unsaved_at, is_saved,
173
+ excerpt, content_text, image_url, audience, is_paywalled, reading_time_minutes,
174
+ word_count, created_at, updated_at
175
+ ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)
176
+ ON CONFLICT(url) DO UPDATE SET
177
+ substack_post_id = COALESCE(excluded.substack_post_id, posts.substack_post_id),
178
+ title = excluded.title,
179
+ publication_name = excluded.publication_name,
180
+ publication_url = COALESCE(excluded.publication_url, posts.publication_url),
181
+ author_name = COALESCE(excluded.author_name, posts.author_name),
182
+ published_at = COALESCE(excluded.published_at, posts.published_at),
183
+ saved_at = COALESCE(excluded.saved_at, posts.saved_at),
184
+ unsaved_at = excluded.unsaved_at,
185
+ is_saved = excluded.is_saved,
186
+ excerpt = COALESCE(excluded.excerpt, posts.excerpt),
187
+ content_text = COALESCE(excluded.content_text, posts.content_text),
188
+ image_url = COALESCE(excluded.image_url, posts.image_url),
189
+ audience = COALESCE(excluded.audience, posts.audience),
190
+ is_paywalled = excluded.is_paywalled,
191
+ reading_time_minutes = COALESCE(excluded.reading_time_minutes, posts.reading_time_minutes),
192
+ word_count = COALESCE(excluded.word_count, posts.word_count),
193
+ updated_at = excluded.updated_at
194
+ """,
195
+ (
196
+ post.substack_post_id,
197
+ clean_url,
198
+ post.title,
199
+ post.publication_name,
200
+ post.publication_url,
201
+ post.author_name,
202
+ post.published_at,
203
+ saved_at,
204
+ post.unsaved_at,
205
+ post.is_saved,
206
+ post.excerpt,
207
+ post.content_text,
208
+ post.image_url,
209
+ post.audience,
210
+ post.is_paywalled,
211
+ post.reading_time_minutes,
212
+ post.word_count,
213
+ created_at,
214
+ now_iso,
215
+ ),
216
+ )
217
+
218
+ post_id = cursor.lastrowid if not existing else existing["id"]
219
+ cursor.execute("SELECT * FROM posts WHERE id = ?", (post_id,))
220
+ row = cursor.fetchone()
221
+ return SavedPost(**dict(row))
222
+
223
+
224
+ def soft_delete_post(
225
+ url_or_id: str | int, db_path: Path | None = None
226
+ ) -> SavedPost | None:
227
+ """Mark a post as unsaved (is_saved = 0, unsaved_at = now)."""
228
+ now_iso = datetime.now(UTC).isoformat()
229
+ with get_db_connection(db_path) as conn:
230
+ cursor = conn.cursor()
231
+
232
+ if isinstance(url_or_id, int) or (
233
+ isinstance(url_or_id, str) and url_or_id.isdigit()
234
+ ):
235
+ cursor.execute("SELECT * FROM posts WHERE id = ?", (int(url_or_id),))
236
+ else:
237
+ clean_url = canonicalize_url(str(url_or_id))
238
+ cursor.execute(
239
+ "SELECT * FROM posts WHERE url = ? OR substack_post_id = ?",
240
+ (clean_url, str(url_or_id)),
241
+ )
242
+
243
+ row = cursor.fetchone()
244
+ if not row:
245
+ return None
246
+
247
+ cursor.execute(
248
+ """
249
+ UPDATE posts
250
+ SET is_saved = 0, unsaved_at = ?, updated_at = ?
251
+ WHERE id = ?
252
+ """,
253
+ (now_iso, now_iso, row["id"]),
254
+ )
255
+
256
+ cursor.execute("SELECT * FROM posts WHERE id = ?", (row["id"],))
257
+ return SavedPost(**dict(cursor.fetchone()))
258
+
259
+
260
+ def reconcile_unsaved_posts(remote_urls: list[str], db_path: Path | None = None) -> int:
261
+ """Soft-delete locally-active posts absent from a complete remote saved-URL set.
262
+
263
+ Intended for use only after a full sync has enumerated every currently-saved
264
+ post on Substack (an incremental sync may stop early and would wrongly treat
265
+ un-refetched posts as removed). Skips entirely when remote_urls is empty, since
266
+ that's more likely a fetch problem than genuine mass-unsaving.
267
+ """
268
+ if not remote_urls:
269
+ return 0
270
+ clean_urls = {canonicalize_url(u) for u in remote_urls if u}
271
+
272
+ now_iso = datetime.now(UTC).isoformat()
273
+ with get_db_connection(db_path) as conn:
274
+ cursor = conn.cursor()
275
+ cursor.execute("SELECT id, url FROM posts WHERE is_saved = 1")
276
+ stale_ids = [
277
+ row["id"] for row in cursor.fetchall() if row["url"] not in clean_urls
278
+ ]
279
+ if not stale_ids:
280
+ return 0
281
+
282
+ cursor.executemany(
283
+ "UPDATE posts SET is_saved = 0, unsaved_at = ?, updated_at = ? WHERE id = ?",
284
+ [(now_iso, now_iso, post_id) for post_id in stale_ids],
285
+ )
286
+ return len(stale_ids)
287
+
288
+
289
+ def get_post(url_or_id: str | int, db_path: Path | None = None) -> SavedPost | None:
290
+ """Retrieve full post record by local ID, Substack post ID, or URL."""
291
+ with get_db_connection(db_path) as conn:
292
+ cursor = conn.cursor()
293
+ if isinstance(url_or_id, int) or (
294
+ isinstance(url_or_id, str) and url_or_id.isdigit()
295
+ ):
296
+ cursor.execute("SELECT * FROM posts WHERE id = ?", (int(url_or_id),))
297
+ else:
298
+ clean_url = canonicalize_url(str(url_or_id))
299
+ cursor.execute(
300
+ "SELECT * FROM posts WHERE url = ? OR substack_post_id = ?",
301
+ (clean_url, str(url_or_id)),
302
+ )
303
+ row = cursor.fetchone()
304
+ return SavedPost(**dict(row)) if row else None
305
+
306
+
307
+ def list_posts(
308
+ limit: int = 20,
309
+ offset: int = 0,
310
+ publication: str | None = None,
311
+ audience: str | None = None,
312
+ sort_by: str = "saved_at",
313
+ is_saved_only: bool = True,
314
+ db_path: Path | None = None,
315
+ ) -> list[PostSummary]:
316
+ """List posts with pagination, publication/audience filters, and sorting."""
317
+ order_col = "published_at" if sort_by == "published_at" else "saved_at"
318
+ where_clauses = []
319
+ params: list[str | int] = []
320
+
321
+ if is_saved_only:
322
+ where_clauses.append("is_saved = 1")
323
+ if publication:
324
+ where_clauses.append("LOWER(publication_name) LIKE LOWER(?)")
325
+ params.append(f"%{publication}%")
326
+ if audience:
327
+ where_clauses.append("LOWER(audience) = LOWER(?)")
328
+ params.append(audience)
329
+
330
+ where_sql = f"WHERE {' AND '.join(where_clauses)}" if where_clauses else ""
331
+ sql = f"""
332
+ SELECT id, substack_post_id, url, title, publication_name, author_name,
333
+ published_at, saved_at, is_saved, excerpt, image_url, audience, is_paywalled,
334
+ reading_time_minutes, word_count
335
+ FROM posts
336
+ {where_sql}
337
+ ORDER BY {order_col} DESC NULLS LAST
338
+ LIMIT ? OFFSET ?
339
+ """
340
+ params.extend([limit, offset])
341
+
342
+ with get_db_connection(db_path) as conn:
343
+ cursor = conn.cursor()
344
+ cursor.execute(sql, params)
345
+ return [PostSummary(**dict(r)) for r in cursor.fetchall()]
346
+
347
+
348
+ def search_posts(
349
+ query: str,
350
+ publication: str | None = None,
351
+ audience: str | None = None,
352
+ published_after: str | None = None,
353
+ published_before: str | None = None,
354
+ saved_after: str | None = None,
355
+ saved_before: str | None = None,
356
+ limit: int = 20,
357
+ is_saved_only: bool = True,
358
+ db_path: Path | None = None,
359
+ ) -> list[PostSummary]:
360
+ """Full-text search over posts using FTS5 BM25 relevance ranking and metadata filters."""
361
+ where_clauses = ["posts_fts MATCH ?"]
362
+ params: list[str | int] = [query]
363
+
364
+ if is_saved_only:
365
+ where_clauses.append("p.is_saved = 1")
366
+ if publication:
367
+ where_clauses.append("LOWER(p.publication_name) LIKE LOWER(?)")
368
+ params.append(f"%{publication}%")
369
+ if audience:
370
+ where_clauses.append("LOWER(p.audience) = LOWER(?)")
371
+ params.append(audience)
372
+ if published_after:
373
+ where_clauses.append("p.published_at >= ?")
374
+ params.append(published_after)
375
+ if published_before:
376
+ where_clauses.append("p.published_at <= ?")
377
+ params.append(published_before)
378
+ if saved_after:
379
+ where_clauses.append("p.saved_at >= ?")
380
+ params.append(saved_after)
381
+ if saved_before:
382
+ where_clauses.append("p.saved_at <= ?")
383
+ params.append(saved_before)
384
+
385
+ sql = f"""
386
+ SELECT p.id, p.substack_post_id, p.url, p.title, p.publication_name, p.author_name,
387
+ p.published_at, p.saved_at, p.is_saved, p.excerpt, p.image_url, p.audience, p.is_paywalled,
388
+ p.reading_time_minutes, p.word_count
389
+ FROM posts_fts fts
390
+ JOIN posts p ON fts.rowid = p.id
391
+ WHERE {" AND ".join(where_clauses)}
392
+ ORDER BY fts.rank
393
+ LIMIT ?
394
+ """
395
+ params.append(limit)
396
+
397
+ with get_db_connection(db_path) as conn:
398
+ cursor = conn.cursor()
399
+ try:
400
+ cursor.execute(sql, params)
401
+ return [PostSummary(**dict(r)) for r in cursor.fetchall()]
402
+ except sqlite3.OperationalError:
403
+ # Fallback to standard LIKE if FTS query syntax is special/malformed
404
+ like_query = f"%{query}%"
405
+ fallback_where = [
406
+ "(title LIKE ? OR excerpt LIKE ? OR publication_name LIKE ? OR author_name LIKE ?)",
407
+ "is_saved = 1",
408
+ ]
409
+ fallback_params: list[str | int] = [
410
+ like_query,
411
+ like_query,
412
+ like_query,
413
+ like_query,
414
+ ]
415
+ if audience:
416
+ fallback_where.append("LOWER(audience) = LOWER(?)")
417
+ fallback_params.append(audience)
418
+ fallback_sql = f"""
419
+ SELECT id, substack_post_id, url, title, publication_name, author_name,
420
+ published_at, saved_at, is_saved, excerpt, image_url, audience, is_paywalled,
421
+ reading_time_minutes, word_count
422
+ FROM posts
423
+ WHERE {" AND ".join(fallback_where)}
424
+ ORDER BY saved_at DESC
425
+ LIMIT ?
426
+ """
427
+ fallback_params.append(limit)
428
+ cursor.execute(fallback_sql, fallback_params)
429
+ return [PostSummary(**dict(r)) for r in cursor.fetchall()]
430
+
431
+
432
+ def list_publications(db_path: Path | None = None) -> list[PublicationSummary]:
433
+ """Return all unique publications in cache with active saved post count."""
434
+ sql = """
435
+ SELECT publication_name, MAX(publication_url) as publication_url, COUNT(*) as post_count
436
+ FROM posts
437
+ WHERE is_saved = 1
438
+ GROUP BY publication_name
439
+ ORDER BY post_count DESC, publication_name ASC
440
+ """
441
+ with get_db_connection(db_path) as conn:
442
+ cursor = conn.cursor()
443
+ cursor.execute(sql)
444
+ return [PublicationSummary(**dict(r)) for r in cursor.fetchall()]
445
+
446
+
447
+ def list_audiences(db_path: Path | None = None) -> list[AudienceSummary]:
448
+ """Return distinct audience tiers present in cache with active saved post counts.
449
+
450
+ Discovers the actual values in use (e.g. "everyone", "only_paid") rather than
451
+ hardcoding Substack's audience enum, since it's not officially documented and
452
+ may grow (e.g. "only_founding", "preview").
453
+ """
454
+ sql = """
455
+ SELECT audience, COUNT(*) as post_count
456
+ FROM posts
457
+ WHERE is_saved = 1
458
+ GROUP BY audience
459
+ ORDER BY post_count DESC
460
+ """
461
+ with get_db_connection(db_path) as conn:
462
+ cursor = conn.cursor()
463
+ cursor.execute(sql)
464
+ return [AudienceSummary(**dict(r)) for r in cursor.fetchall()]
465
+
466
+
467
+ def get_status(db_path: Path | None = None) -> SavedPostsStatus:
468
+ """Return metrics and statistics for local SQLite database."""
469
+ target_path = db_path or get_db_path()
470
+ with get_db_connection(target_path) as conn:
471
+ cursor = conn.cursor()
472
+
473
+ cursor.execute("SELECT COUNT(*) FROM posts WHERE is_saved = 1")
474
+ total_saved = cursor.fetchone()[0]
475
+
476
+ cursor.execute("SELECT COUNT(*) FROM posts WHERE is_saved = 0")
477
+ total_unsaved = cursor.fetchone()[0]
478
+
479
+ cursor.execute(
480
+ "SELECT COUNT(DISTINCT publication_name) FROM posts WHERE is_saved = 1"
481
+ )
482
+ total_pubs = cursor.fetchone()[0]
483
+
484
+ cursor.execute("""
485
+ SELECT completed_at, status FROM sync_runs
486
+ WHERE status = 'success'
487
+ ORDER BY id DESC LIMIT 1
488
+ """)
489
+ last_success_row = cursor.fetchone()
490
+ last_success = last_success_row["completed_at"] if last_success_row else None
491
+
492
+ cursor.execute("SELECT status FROM sync_runs ORDER BY id DESC LIMIT 1")
493
+ last_status_row = cursor.fetchone()
494
+ last_status = last_status_row["status"] if last_status_row else None
495
+
496
+ return SavedPostsStatus(
497
+ total_saved_posts=total_saved,
498
+ total_unsaved_posts=total_unsaved,
499
+ total_publications=total_pubs,
500
+ last_successful_sync=last_success,
501
+ last_sync_status=last_status,
502
+ database_path=str(target_path),
503
+ )
504
+
505
+
506
+ def start_sync_run(sync_mode: str = "incremental", db_path: Path | None = None) -> int:
507
+ """Create a sync_runs entry and return its ID."""
508
+ now_iso = datetime.now(UTC).isoformat()
509
+ with get_db_connection(db_path) as conn:
510
+ cursor = conn.cursor()
511
+ cursor.execute(
512
+ """
513
+ INSERT INTO sync_runs (started_at, status, sync_mode)
514
+ VALUES (?, 'running', ?)
515
+ """,
516
+ (now_iso, sync_mode),
517
+ )
518
+ return cursor.lastrowid # type: ignore
519
+
520
+
521
+ def finish_sync_run(
522
+ sync_id: int,
523
+ status: str,
524
+ fetched_count: int,
525
+ upserted_count: int,
526
+ reconciled_count: int = 0,
527
+ error_message: str | None = None,
528
+ db_path: Path | None = None,
529
+ ) -> None:
530
+ """Update a sync_runs record upon completion or failure."""
531
+ now_iso = datetime.now(UTC).isoformat()
532
+ with get_db_connection(db_path) as conn:
533
+ cursor = conn.cursor()
534
+ cursor.execute(
535
+ """
536
+ UPDATE sync_runs
537
+ SET completed_at = ?, status = ?, fetched_count = ?, upserted_count = ?,
538
+ reconciled_count = ?, error_message = ?
539
+ WHERE id = ?
540
+ """,
541
+ (
542
+ now_iso,
543
+ status,
544
+ fetched_count,
545
+ upserted_count,
546
+ reconciled_count,
547
+ error_message,
548
+ sync_id,
549
+ ),
550
+ )