youpdated 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,335 @@
1
+ """YouTube channels and playlists
2
+
3
+ YouTube's RSS endpoint (``/feeds/videos.xml``) is the easiest,
4
+ but its iffy and 404s, then serves fine later. (endpoints work, it's throttling)
5
+
6
+ Every fetch goes through a list:
7
+ 1. the official feed
8
+ 2. an Invidious instance (privacy-friendly, but flaky sometimes)
9
+ 3. the Data API, if YOUTUBE_API_KEY is set
10
+
11
+ Set ``privacy.proxy`` and step 1 generally works
12
+ """
13
+
14
+ from __future__ import annotations
15
+
16
+ import os
17
+ import re
18
+ from datetime import datetime, timezone
19
+ from typing import Any, ClassVar, Iterable
20
+
21
+ from ..http import Client, FetchError
22
+ from ..models import Target, Update
23
+ from ..registry import register
24
+ from .base import ConfigEntryError, entry_fields, require
25
+ from .feed import parse_feed
26
+
27
+ CHANNEL_ID_RE = re.compile(r"^UC[A-Za-z0-9_-]{22}$")
28
+ PLAYLIST_ID_RE = re.compile(r"^(?:PL|UU|LL|FL|OL)[A-Za-z0-9_-]{10,}$")
29
+ EXTERNAL_ID_RE = re.compile(r'"externalId"\s*:\s*"(UC[A-Za-z0-9_-]{22})"')
30
+
31
+ CHANNEL_CACHE = "youtube_channel_ids"
32
+ INSTANCE_CACHE = "youtube_invidious"
33
+ MAX_ITEMS = 15
34
+ MAX_INSTANCES = 4
35
+
36
+
37
+ @register
38
+ class YouTubeSource:
39
+ name: ClassVar[str] = "youtube"
40
+ summary: ClassVar[str] = "New videos on a channel or playlist"
41
+
42
+ def targets(self, entries: list[Any]) -> list[Target]:
43
+ targets = []
44
+ for entry in entries:
45
+ fields = entry_fields(entry, "channel", self.name)
46
+
47
+ if fields.get("playlist"):
48
+ playlist = str(fields["playlist"]).strip()
49
+ playlist = _extract_playlist_id(playlist)
50
+ if not PLAYLIST_ID_RE.match(playlist):
51
+ raise ConfigEntryError(
52
+ f"sources.{self.name}: `{playlist}` is not a playlist id or URL"
53
+ )
54
+ targets.append(
55
+ Target(
56
+ source=self.name,
57
+ key=f"playlist:{playlist}",
58
+ label=fields.get("name") or f"playlist {playlist}",
59
+ params={"kind": "playlist", "id": playlist},
60
+ )
61
+ )
62
+ continue
63
+
64
+ raw = str(require(fields, "channel", self.name)).strip()
65
+ ref = _extract_channel_ref(raw)
66
+ if ref is None:
67
+ raise ConfigEntryError(
68
+ f"sources.{self.name}: `{raw}` is not a channel id, @handle, or channel URL"
69
+ )
70
+ targets.append(
71
+ Target(
72
+ source=self.name,
73
+ key=ref,
74
+ label=fields.get("name"),
75
+ params={"kind": "channel", "ref": ref},
76
+ )
77
+ )
78
+ return targets
79
+
80
+ def fetch(self, target: Target, client: Client) -> Iterable[Update]:
81
+ if target.params["kind"] == "playlist":
82
+ feed_id, api_id = ("playlist_id", target.params["id"])
83
+ else:
84
+ channel_id = self._channel_id(target.params["ref"], client)
85
+ if channel_id is None:
86
+ raise FetchError(
87
+ f"could not resolve YouTube channel `{target.params['ref']}` to a channel id"
88
+ )
89
+ feed_id, api_id = ("channel_id", channel_id)
90
+
91
+ failures: list[str] = []
92
+
93
+ for attempt in (self._official_feed, self._invidious, self._data_api):
94
+ try:
95
+ updates = attempt(target, client, feed_id, api_id)
96
+ except FetchError as exc:
97
+ failures.append(str(exc))
98
+ continue
99
+ if updates is not None:
100
+ return updates
101
+
102
+ raise FetchError(
103
+ "every YouTube path failed (" + "; ".join(failures) + "). "
104
+ "The official feed throttles, retrying often works, or set "
105
+ "privacy.proxy, or YOUTUBE_API_KEY."
106
+ )
107
+
108
+ # yt official feed
109
+
110
+ def _official_feed(
111
+ self, target: Target, client: Client, feed_id: str, api_id: str
112
+ ) -> list[Update] | None:
113
+ url = f"https://www.youtube.com/feeds/videos.xml?{feed_id}={api_id}"
114
+ fetched = client.get(url, conditional=True, soft_statuses=(404, 403))
115
+ if fetched is None:
116
+ return []
117
+ if fetched.status in (403, 404):
118
+ raise FetchError(f"official feed returned {fetched.status} (likely blocked)")
119
+ return self._label_and_parse(target, fetched.content, via="feed")
120
+
121
+ # invidious
122
+
123
+ def _invidious(
124
+ self, target: Target, client: Client, feed_id: str, api_id: str
125
+ ) -> list[Update] | None:
126
+ errors = []
127
+ for base in self._instances(client):
128
+ if feed_id == "channel_id":
129
+ url = f"{base}/api/v1/channels/{api_id}/latest"
130
+ else:
131
+ url = f"{base}/api/v1/playlists/{api_id}"
132
+ try:
133
+ fetched = client.get(url, soft_statuses=(403, 404, 429), retries=0)
134
+ except FetchError as exc:
135
+ errors.append(str(exc))
136
+ continue
137
+ if fetched is None or fetched.status != 200:
138
+ errors.append(f"{base}: HTTP {fetched.status if fetched else 'none'}")
139
+ continue
140
+ try:
141
+ payload = fetched.json()
142
+ except ValueError:
143
+ errors.append(f"{base}: non-JSON response")
144
+ continue
145
+ videos = payload.get("videos") if isinstance(payload, dict) else payload
146
+ if not isinstance(videos, list):
147
+ errors.append(f"{base}: unexpected payload shape")
148
+ continue
149
+ client.note(f"youtube: by Invidious instance {base}")
150
+ return self._from_invidious(target, videos)
151
+ raise FetchError("no Invidious instance answered (" + "; ".join(errors[:3]) + ")")
152
+
153
+ def _instances(self, client: Client) -> list[str]:
154
+ cached = client.state.cache_get(INSTANCE_CACHE, "list") if client.state else None
155
+ if cached:
156
+ return cached.split(",")[:MAX_INSTANCES]
157
+ try:
158
+ fetched = client.get("https://api.invidious.io/instances.json", retries=1)
159
+ except FetchError:
160
+ return []
161
+ if fetched is None:
162
+ return []
163
+ try:
164
+ listing = fetched.json()
165
+ except ValueError:
166
+ return []
167
+
168
+ bases = []
169
+ for item in listing:
170
+ if not (isinstance(item, list) and len(item) == 2):
171
+ continue
172
+ meta = item[1] or {}
173
+ if meta.get("type") == "https" and meta.get("api") and meta.get("uri"):
174
+ bases.append(str(meta["uri"]).rstrip("/"))
175
+ if bases and client.state:
176
+ client.state.cache_set(INSTANCE_CACHE, "list", ",".join(bases[:8]))
177
+ return bases[:MAX_INSTANCES]
178
+
179
+ def _from_invidious(self, target: Target, videos: list[dict]) -> list[Update]:
180
+ updates = []
181
+ for video in videos[:MAX_ITEMS]:
182
+ vid = video.get("videoId")
183
+ if not vid:
184
+ continue
185
+ published = video.get("published")
186
+ updates.append(
187
+ Update(
188
+ source=self.name,
189
+ target=target.key,
190
+ uid=f"yt:video:{vid}",
191
+ title=video.get("title") or "(untitled)",
192
+ url=f"https://www.youtube.com/watch?v={vid}",
193
+ published=(
194
+ datetime.fromtimestamp(published, tz=timezone.utc)
195
+ if isinstance(published, (int, float))
196
+ else None
197
+ ),
198
+ body=(video.get("description") or None),
199
+ tags=("video",),
200
+ )
201
+ )
202
+ if not target.label and video.get("author"):
203
+ target.label = video["author"]
204
+ return updates
205
+
206
+ # Data API
207
+
208
+ def _data_api(
209
+ self, target: Target, client: Client, feed_id: str, api_id: str
210
+ ) -> list[Update] | None:
211
+ key = os.environ.get("YOUTUBE_API_KEY")
212
+ if not key:
213
+ raise FetchError("YOUTUBE_API_KEY not set")
214
+
215
+ # channel's uploads playlist is id with UC swapped for UU
216
+ playlist_id = api_id if feed_id == "playlist_id" else "UU" + api_id[2:]
217
+ url = (
218
+ "https://www.googleapis.com/youtube/v3/playlistItems"
219
+ f"?part=snippet&maxResults={MAX_ITEMS}&playlistId={playlist_id}&key={key}"
220
+ )
221
+ fetched = client.get(url)
222
+ if fetched is None:
223
+ return []
224
+ client.note("youtube: by Data API")
225
+
226
+ updates = []
227
+ for item in fetched.json().get("items", []):
228
+ snippet = item.get("snippet") or {}
229
+ vid = (snippet.get("resourceId") or {}).get("videoId")
230
+ if not vid:
231
+ continue
232
+ if not target.label and snippet.get("channelTitle"):
233
+ target.label = snippet["channelTitle"]
234
+ updates.append(
235
+ Update(
236
+ source=self.name,
237
+ target=target.key,
238
+ uid=f"yt:video:{vid}",
239
+ title=snippet.get("title") or "(untitled)",
240
+ url=f"https://www.youtube.com/watch?v={vid}",
241
+ published=_parse_iso(snippet.get("publishedAt")),
242
+ body=(snippet.get("description") or None)[:400] if snippet.get("description") else None,
243
+ tags=("video",),
244
+ )
245
+ )
246
+ return updates
247
+
248
+ # helpers
249
+
250
+ def _label_and_parse(self, target: Target, content: bytes, via: str) -> list[Update]:
251
+ updates = parse_feed(
252
+ content,
253
+ source=self.name,
254
+ target=target.key,
255
+ limit=MAX_ITEMS,
256
+ tags=("video",),
257
+ )
258
+ # Rewrite uids to the video id so the three paths dedupe
259
+ normalized = []
260
+ for update in updates:
261
+ vid = _video_id(update.url) or update.uid
262
+ normalized.append(
263
+ Update(
264
+ source=update.source,
265
+ target=update.target,
266
+ uid=f"yt:video:{vid}",
267
+ title=update.title,
268
+ url=update.url,
269
+ published=update.published,
270
+ version=update.version,
271
+ body=update.body,
272
+ tags=update.tags,
273
+ )
274
+ )
275
+ return normalized
276
+
277
+ def _channel_id(self, ref: str, client: Client) -> str | None:
278
+ if CHANNEL_ID_RE.match(ref):
279
+ return ref
280
+ state = client.state
281
+ if state is not None:
282
+ cached = state.cache_get(CHANNEL_CACHE, ref)
283
+ if cached:
284
+ return cached
285
+
286
+ page = f"https://www.youtube.com/{ref}" if ref.startswith("@") else ref
287
+ try:
288
+ fetched = client.get(page, soft_statuses=(404,))
289
+ except FetchError:
290
+ return None
291
+ if fetched is None or fetched.status != 200:
292
+ return None
293
+
294
+ match = EXTERNAL_ID_RE.search(fetched.text)
295
+ if not match:
296
+ return None
297
+ channel_id = match.group(1)
298
+ if state is not None:
299
+ state.cache_set(CHANNEL_CACHE, ref, channel_id)
300
+ return channel_id
301
+
302
+
303
+ def _extract_channel_ref(raw: str) -> str | None:
304
+ """Reduce input to channel id or @handle"""
305
+ if CHANNEL_ID_RE.match(raw):
306
+ return raw
307
+ if raw.startswith("@") and len(raw) > 1:
308
+ return raw
309
+ match = re.search(r"youtube\.com/channel/(UC[A-Za-z0-9_-]{22})", raw)
310
+ if match:
311
+ return match.group(1)
312
+ match = re.search(r"youtube\.com/(@[A-Za-z0-9._-]+)", raw)
313
+ if match:
314
+ return match.group(1)
315
+ return None
316
+
317
+
318
+ def _extract_playlist_id(raw: str) -> str:
319
+ match = re.search(r"[?&]list=([A-Za-z0-9_-]+)", raw)
320
+ return match.group(1) if match else raw
321
+
322
+
323
+ def _video_id(url: str) -> str | None:
324
+ match = re.search(r"[?&]v=([A-Za-z0-9_-]{11})", url or "")
325
+ return match.group(1) if match else None
326
+
327
+
328
+ def _parse_iso(value: str | None) -> datetime | None:
329
+ if not value:
330
+ return None
331
+ try:
332
+ parsed = datetime.fromisoformat(str(value).replace("Z", "+00:00"))
333
+ except ValueError:
334
+ return None
335
+ return parsed if parsed.tzinfo else parsed.replace(tzinfo=timezone.utc)
youpdated/state.py ADDED
@@ -0,0 +1,154 @@
1
+ """Local SQLite state: what's reported, HTTP validators, and resolved-name caches"""
2
+
3
+ from __future__ import annotations
4
+
5
+ import sqlite3
6
+ import threading
7
+ from datetime import datetime, timezone
8
+ from pathlib import Path
9
+ from typing import Iterable
10
+
11
+ from .models import Update
12
+
13
+ _SCHEMA = """
14
+ CREATE TABLE IF NOT EXISTS seen (
15
+ source TEXT NOT NULL,
16
+ target TEXT NOT NULL,
17
+ uid TEXT NOT NULL,
18
+ first_seen TEXT NOT NULL,
19
+ PRIMARY KEY (source, target, uid)
20
+ );
21
+
22
+ CREATE TABLE IF NOT EXISTS http_cache (
23
+ url TEXT PRIMARY KEY,
24
+ etag TEXT,
25
+ last_modified TEXT,
26
+ fetched_at TEXT NOT NULL
27
+ );
28
+
29
+ CREATE TABLE IF NOT EXISTS kv (
30
+ namespace TEXT NOT NULL,
31
+ key TEXT NOT NULL,
32
+ value TEXT NOT NULL,
33
+ PRIMARY KEY (namespace, key)
34
+ );
35
+ """
36
+
37
+
38
+ def _now() -> str:
39
+ return datetime.now(timezone.utc).isoformat()
40
+
41
+
42
+ class State:
43
+ """Thread-safe wrapper over SQLite file."""
44
+
45
+ def __init__(self, path: str | Path | None = None):
46
+ self.path = Path(path) if path is not None else None
47
+ if self.path is not None:
48
+ self.path.parent.mkdir(parents=True, exist_ok=True)
49
+ target = str(self.path) if self.path is not None else ":memory:"
50
+ self._lock = threading.Lock()
51
+ self._conn = sqlite3.connect(target, check_same_thread=False)
52
+ self._conn.row_factory = sqlite3.Row
53
+ with self._lock:
54
+ self._conn.executescript(_SCHEMA)
55
+ self._conn.commit()
56
+
57
+ def close(self) -> None:
58
+ with self._lock:
59
+ self._conn.close()
60
+
61
+ def __enter__(self) -> "State":
62
+ return self
63
+
64
+ def __exit__(self, *exc: object) -> None:
65
+ self.close()
66
+
67
+ # already seen items
68
+
69
+ def is_new(self, update: Update) -> bool:
70
+ with self._lock:
71
+ row = self._conn.execute(
72
+ "SELECT 1 FROM seen WHERE source=? AND target=? AND uid=?",
73
+ update.dedupe_key,
74
+ ).fetchone()
75
+ return row is None
76
+
77
+ def filter_new(self, updates: Iterable[Update]) -> list[Update]:
78
+ return [u for u in updates if self.is_new(u)]
79
+
80
+ def mark_seen(self, updates: Iterable[Update]) -> None:
81
+ rows = [(*u.dedupe_key, _now()) for u in updates]
82
+ if not rows:
83
+ return
84
+ with self._lock:
85
+ self._conn.executemany(
86
+ "INSERT OR IGNORE INTO seen (source, target, uid, first_seen) VALUES (?,?,?,?)",
87
+ rows,
88
+ )
89
+ self._conn.commit()
90
+
91
+ def seen_count(self) -> int:
92
+ with self._lock:
93
+ return self._conn.execute("SELECT COUNT(*) FROM seen").fetchone()[0]
94
+
95
+ # conditional get validators
96
+
97
+ def conditional_headers(self, url: str) -> dict[str, str]:
98
+ with self._lock:
99
+ row = self._conn.execute(
100
+ "SELECT etag, last_modified FROM http_cache WHERE url=?", (url,)
101
+ ).fetchone()
102
+ if row is None:
103
+ return {}
104
+ headers = {}
105
+ if row["etag"]:
106
+ headers["If-None-Match"] = row["etag"]
107
+ if row["last_modified"]:
108
+ headers["If-Modified-Since"] = row["last_modified"]
109
+ return headers
110
+
111
+ def remember_validators(self, url: str, etag: str | None, last_modified: str | None) -> None:
112
+ if not etag and not last_modified:
113
+ return
114
+ with self._lock:
115
+ self._conn.execute(
116
+ "INSERT INTO http_cache (url, etag, last_modified, fetched_at) VALUES (?,?,?,?) "
117
+ "ON CONFLICT(url) DO UPDATE SET etag=excluded.etag, "
118
+ "last_modified=excluded.last_modified, fetched_at=excluded.fetched_at",
119
+ (url, etag, last_modified, _now()),
120
+ )
121
+ self._conn.commit()
122
+
123
+ # small caches (resolved names, channel ids)
124
+
125
+ def cache_get(self, namespace: str, key: str) -> str | None:
126
+ with self._lock:
127
+ row = self._conn.execute(
128
+ "SELECT value FROM kv WHERE namespace=? AND key=?", (namespace, key)
129
+ ).fetchone()
130
+ return row["value"] if row else None
131
+
132
+ def cache_set(self, namespace: str, key: str, value: str) -> None:
133
+ with self._lock:
134
+ self._conn.execute(
135
+ "INSERT INTO kv (namespace, key, value) VALUES (?,?,?) "
136
+ "ON CONFLICT(namespace, key) DO UPDATE SET value=excluded.value",
137
+ (namespace, key, value),
138
+ )
139
+ self._conn.commit()
140
+
141
+ # upkeeping
142
+
143
+ def last_run(self) -> datetime | None:
144
+ raw = self.cache_get("meta", "last_run")
145
+ if not raw:
146
+ return None
147
+ try:
148
+ return datetime.fromisoformat(raw)
149
+ except ValueError:
150
+ return None
151
+
152
+ def set_last_run(self, when: datetime | None = None) -> None:
153
+ stamp = (when or datetime.now(timezone.utc)).isoformat()
154
+ self.cache_set("meta", "last_run", stamp)