youpdated 0.2.0__tar.gz → 0.2.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {youpdated-0.2.0 → youpdated-0.2.1}/CHANGELOG.md +41 -0
- {youpdated-0.2.0 → youpdated-0.2.1}/PKG-INFO +1 -1
- {youpdated-0.2.0 → youpdated-0.2.1}/pyproject.toml +1 -1
- {youpdated-0.2.0 → youpdated-0.2.1}/youpdated/__init__.py +1 -1
- {youpdated-0.2.0 → youpdated-0.2.1}/youpdated/http.py +60 -8
- {youpdated-0.2.0 → youpdated-0.2.1}/youpdated/runner.py +2 -1
- {youpdated-0.2.0 → youpdated-0.2.1}/youpdated/sources/feed.py +38 -4
- {youpdated-0.2.0 → youpdated-0.2.1}/youpdated/sources/generic.py +8 -8
- {youpdated-0.2.0 → youpdated-0.2.1}/youpdated/sources/youtube.py +10 -5
- {youpdated-0.2.0 → youpdated-0.2.1}/youpdated.egg-info/PKG-INFO +1 -1
- {youpdated-0.2.0 → youpdated-0.2.1}/ADDING_SOURCES.md +0 -0
- {youpdated-0.2.0 → youpdated-0.2.1}/CONTRIBUTING.md +0 -0
- {youpdated-0.2.0 → youpdated-0.2.1}/LICENSE +0 -0
- {youpdated-0.2.0 → youpdated-0.2.1}/MANIFEST.in +0 -0
- {youpdated-0.2.0 → youpdated-0.2.1}/README.md +0 -0
- {youpdated-0.2.0 → youpdated-0.2.1}/SECURITY.md +0 -0
- {youpdated-0.2.0 → youpdated-0.2.1}/setup.cfg +0 -0
- {youpdated-0.2.0 → youpdated-0.2.1}/youpdated/__main__.py +0 -0
- {youpdated-0.2.0 → youpdated-0.2.1}/youpdated/cleanup.py +0 -0
- {youpdated-0.2.0 → youpdated-0.2.1}/youpdated/cli.py +0 -0
- {youpdated-0.2.0 → youpdated-0.2.1}/youpdated/config.py +0 -0
- {youpdated-0.2.0 → youpdated-0.2.1}/youpdated/crypto.py +0 -0
- {youpdated-0.2.0 → youpdated-0.2.1}/youpdated/models.py +0 -0
- {youpdated-0.2.0 → youpdated-0.2.1}/youpdated/py.typed +0 -0
- {youpdated-0.2.0 → youpdated-0.2.1}/youpdated/registry.py +0 -0
- {youpdated-0.2.0 → youpdated-0.2.1}/youpdated/render/__init__.py +0 -0
- {youpdated-0.2.0 → youpdated-0.2.1}/youpdated/render/json_out.py +0 -0
- {youpdated-0.2.0 → youpdated-0.2.1}/youpdated/render/rss_out.py +0 -0
- {youpdated-0.2.0 → youpdated-0.2.1}/youpdated/render/terminal.py +0 -0
- {youpdated-0.2.0 → youpdated-0.2.1}/youpdated/sources/__init__.py +0 -0
- {youpdated-0.2.0 → youpdated-0.2.1}/youpdated/sources/base.py +0 -0
- {youpdated-0.2.0 → youpdated-0.2.1}/youpdated/sources/browser.py +0 -0
- {youpdated-0.2.0 → youpdated-0.2.1}/youpdated/sources/github.py +0 -0
- {youpdated-0.2.0 → youpdated-0.2.1}/youpdated/sources/itch.py +0 -0
- {youpdated-0.2.0 → youpdated-0.2.1}/youpdated/sources/npm.py +0 -0
- {youpdated-0.2.0 → youpdated-0.2.1}/youpdated/sources/steam.py +0 -0
- {youpdated-0.2.0 → youpdated-0.2.1}/youpdated/state.py +0 -0
- {youpdated-0.2.0 → youpdated-0.2.1}/youpdated.egg-info/SOURCES.txt +0 -0
- {youpdated-0.2.0 → youpdated-0.2.1}/youpdated.egg-info/dependency_links.txt +0 -0
- {youpdated-0.2.0 → youpdated-0.2.1}/youpdated.egg-info/entry_points.txt +0 -0
- {youpdated-0.2.0 → youpdated-0.2.1}/youpdated.egg-info/requires.txt +0 -0
- {youpdated-0.2.0 → youpdated-0.2.1}/youpdated.egg-info/top_level.txt +0 -0
|
@@ -4,6 +4,46 @@ All notable changes to this project are documented here. The format follows
|
|
|
4
4
|
[Keep a Changelog](https://keepachangelog.com/en/1.1.0/), and this project adheres to
|
|
5
5
|
[Semantic Versioning](https://semver.org/spec/v2.0.0.html).
|
|
6
6
|
|
|
7
|
+
## [0.2.1] — 2026-08-25
|
|
8
|
+
|
|
9
|
+
### Fixed
|
|
10
|
+
|
|
11
|
+
- **The release workflow's `attach` job failed on v0.2.0 and has been removed.** It copied the
|
|
12
|
+
built artifacts onto the GitHub Release, but this repo has immutable releases enabled, which
|
|
13
|
+
freezes assets when the release is published — and the job runs after that, so it could only ever
|
|
14
|
+
fail (`HTTP 422: Cannot upload assets to an immutable release`). Publishing to PyPI was
|
|
15
|
+
unaffected; 0.2.0 shipped correctly. Releases from now on will not carry attached `.whl`/`.tar.gz`
|
|
16
|
+
files. Get them from PyPI, where they are covered by a signed PEP 740 attestation binding each
|
|
17
|
+
digest to this repository and workflow — a stronger guarantee than the copy the job was making.
|
|
18
|
+
|
|
19
|
+
- **Targets sharing one upstream document lost their updates**
|
|
20
|
+
Every Brave channel is served by one GitHub releases document, and every Firefox channel by one
|
|
21
|
+
Mozilla JSON. Each channel fetched it separately, so the first stored an ETag and the rest were
|
|
22
|
+
answered `304 Not Modified` against that ETag moments later and reported *nothing*. With the
|
|
23
|
+
default `jitter`, per-host pacing serializes those requests, which is the case that loses: a
|
|
24
|
+
config watching five Firefox channels and three Brave channels reported 2 of 8. Under
|
|
25
|
+
concurrency it was non-deterministic: is a channel reported depended on thread timing, so a
|
|
26
|
+
loaded machine changed the result.
|
|
27
|
+
|
|
28
|
+
A run now reuses each document across the targets that share it (`Client.run_scope()`), so the
|
|
29
|
+
document is fetched once and every target sees it. The same config went from 8 requests
|
|
30
|
+
reporting 2 of 8 targets, to 2 requests reporting 8 of 8. (And 147 ms to 43 ms of wall time)
|
|
31
|
+
Outside a run the client is unchanged: one GET per call, and a 304 still means "nothing new".
|
|
32
|
+
|
|
33
|
+
- **The YouTube official-feed path never set the channel label.** `_label_and_parse` did not do
|
|
34
|
+
the labelling it was supposed to, so a channel that answered on the primary path was reported by
|
|
35
|
+
its raw `@handle` while the Invidious and Data API fallbacks named it properly. The label now
|
|
36
|
+
comes from the feed on all three paths.
|
|
37
|
+
|
|
38
|
+
- **`feed` sources parsed every feed twice**
|
|
39
|
+
A `feed:` entry given as a bare URL had no label, so `FeedSource.fetch` parsed the document once
|
|
40
|
+
to read the feed's own `<title>` and then handed the same bytes to `parse_feed`, which parsed
|
|
41
|
+
them again. Feed parsing is the most expensive part of a run, so this ~doubled the
|
|
42
|
+
CPU cost of every bare-URL feed. One parse now feeds both the label and the entries.
|
|
43
|
+
|
|
44
|
+
`parse_feed()` is unchanged for callers that only need entries; it is now a wrapper over the new
|
|
45
|
+
`parse_document()` / `parse_entries()` split in `youpdated.sources.feed`.
|
|
46
|
+
|
|
7
47
|
## [0.2.0] — 2026-08-19
|
|
8
48
|
|
|
9
49
|
### Added
|
|
@@ -106,6 +146,7 @@ First release.
|
|
|
106
146
|
- Firefox publishes current versions, so it reports one item per channel.
|
|
107
147
|
- Edge exposes release notes only for the stable and beta channels. (But like, it's Edge, why do you want to know when it updates?)
|
|
108
148
|
|
|
149
|
+
[0.2.1]: https://github.com/Void1-1/youpdated/releases/tag/v0.2.1
|
|
109
150
|
[0.2.0]: https://github.com/Void1-1/youpdated/releases/tag/v0.2.0
|
|
110
151
|
[0.1.1]: https://github.com/Void1-1/youpdated/releases/tag/v0.1.1
|
|
111
152
|
[0.1.0]: https://github.com/Void1-1/youpdated/releases/tag/v0.1.0
|
|
@@ -9,8 +9,9 @@ import random
|
|
|
9
9
|
import socket
|
|
10
10
|
import threading
|
|
11
11
|
import time
|
|
12
|
+
from contextlib import contextmanager
|
|
12
13
|
from dataclasses import dataclass, field
|
|
13
|
-
from typing import Any, Iterable, Sequence
|
|
14
|
+
from typing import Any, Iterable, Iterator, Sequence
|
|
14
15
|
from urllib.parse import urlsplit
|
|
15
16
|
|
|
16
17
|
import httpx
|
|
@@ -127,6 +128,16 @@ class Client:
|
|
|
127
128
|
self._host_last: dict[str, float] = {}
|
|
128
129
|
self._registry_lock = threading.Lock()
|
|
129
130
|
|
|
131
|
+
# One upstream document can back several targets: every Brave channel lives
|
|
132
|
+
# in one releases feed, every Firefox channel in one JSON. Bodies fetched
|
|
133
|
+
# during this run are reused so those targets neither re-download the
|
|
134
|
+
# document nor lose it to a validator this same run stored.
|
|
135
|
+
self._run_bodies: dict[tuple[str, tuple[tuple[str, str], ...]], Fetched] = {}
|
|
136
|
+
self._bodies_lock = threading.Lock()
|
|
137
|
+
#: Only reuse bodies inside :meth:`run_scope`. A client used directly,
|
|
138
|
+
#: outside a run, keeps the plain one-GET-per-call contract.
|
|
139
|
+
self._run_active = False
|
|
140
|
+
|
|
130
141
|
self._client = httpx.Client(
|
|
131
142
|
proxy=self.privacy.proxy,
|
|
132
143
|
timeout=self.privacy.timeout,
|
|
@@ -155,6 +166,22 @@ class Client:
|
|
|
155
166
|
f"concurrency={self.privacy.concurrency} timeout={self.privacy.timeout}s"
|
|
156
167
|
)
|
|
157
168
|
|
|
169
|
+
@contextmanager
|
|
170
|
+
def run_scope(self) -> "Iterator[Client]":
|
|
171
|
+
"""Reuse each fetched document for the duration of one run.
|
|
172
|
+
|
|
173
|
+
Bodies are dropped again on exit so a client reused for a later run never serves stale data.
|
|
174
|
+
"""
|
|
175
|
+
with self._bodies_lock:
|
|
176
|
+
self._run_bodies.clear()
|
|
177
|
+
self._run_active = True
|
|
178
|
+
try:
|
|
179
|
+
yield self
|
|
180
|
+
finally:
|
|
181
|
+
with self._bodies_lock:
|
|
182
|
+
self._run_bodies.clear()
|
|
183
|
+
self._run_active = False
|
|
184
|
+
|
|
158
185
|
# internals
|
|
159
186
|
|
|
160
187
|
def _user_agent(self) -> str:
|
|
@@ -176,6 +203,13 @@ class Client:
|
|
|
176
203
|
time.sleep(wait)
|
|
177
204
|
self._host_last[host] = time.monotonic()
|
|
178
205
|
|
|
206
|
+
def _cached_body(
|
|
207
|
+
self, key: tuple[str, tuple[tuple[str, str], ...]]
|
|
208
|
+
) -> "Fetched | None":
|
|
209
|
+
"""The 200 body already fetched for this key during this run, if any."""
|
|
210
|
+
with self._bodies_lock:
|
|
211
|
+
return self._run_bodies.get(key) if self._run_active else None
|
|
212
|
+
|
|
179
213
|
def note(self, message: str) -> None:
|
|
180
214
|
if self.verbose and self._log:
|
|
181
215
|
self._log(message)
|
|
@@ -204,6 +238,11 @@ class Client:
|
|
|
204
238
|
|
|
205
239
|
soft = frozenset(soft_statuses)
|
|
206
240
|
conditional = conditional and self.use_conditional
|
|
241
|
+
cache_key = (url, tuple(sorted((headers or {}).items())))
|
|
242
|
+
cached = self._cached_body(cache_key)
|
|
243
|
+
if cached is not None:
|
|
244
|
+
self.note(f"GET {url} -> reusing the copy already fetched this run")
|
|
245
|
+
return cached
|
|
207
246
|
request_headers = {"User-Agent": self._user_agent()}
|
|
208
247
|
if headers:
|
|
209
248
|
request_headers.update(headers)
|
|
@@ -230,21 +269,34 @@ class Client:
|
|
|
230
269
|
self.note(f"GET {url} -> {response.status_code}")
|
|
231
270
|
|
|
232
271
|
if response.status_code == 304:
|
|
233
|
-
|
|
272
|
+
# Normally "unchanged since your last run", so there is nothing to
|
|
273
|
+
# report. But if this run already holds the body, the validator we
|
|
274
|
+
# sent was our own from moments ago: hand back what we have. Only
|
|
275
|
+
# reachable when two threads race past the check above; the common
|
|
276
|
+
# case is served from the cache without a request at all.
|
|
277
|
+
return self._cached_body(cache_key)
|
|
234
278
|
|
|
235
279
|
if response.status_code == 200:
|
|
280
|
+
result = Fetched(
|
|
281
|
+
url=str(response.url),
|
|
282
|
+
status=response.status_code,
|
|
283
|
+
content=response.content,
|
|
284
|
+
headers=dict(response.headers),
|
|
285
|
+
)
|
|
286
|
+
# Cache the body before publishing validator. A thread racing
|
|
287
|
+
# can only be answered 304 once the validator is stored, so this
|
|
288
|
+
# ordering guarantees the body is already there for it to fall back
|
|
289
|
+
# on.
|
|
290
|
+
with self._bodies_lock:
|
|
291
|
+
if self._run_active:
|
|
292
|
+
self._run_bodies[cache_key] = result
|
|
236
293
|
if conditional and self.state is not None:
|
|
237
294
|
self.state.remember_validators(
|
|
238
295
|
url,
|
|
239
296
|
response.headers.get("etag"),
|
|
240
297
|
response.headers.get("last-modified"),
|
|
241
298
|
)
|
|
242
|
-
return
|
|
243
|
-
url=str(response.url),
|
|
244
|
-
status=response.status_code,
|
|
245
|
-
content=response.content,
|
|
246
|
-
headers=dict(response.headers),
|
|
247
|
-
)
|
|
299
|
+
return result
|
|
248
300
|
|
|
249
301
|
if response.status_code in soft:
|
|
250
302
|
return Fetched(
|
|
@@ -94,7 +94,8 @@ def run(
|
|
|
94
94
|
return target, exc
|
|
95
95
|
|
|
96
96
|
workers = max(1, min(config.privacy.concurrency, len(targets)))
|
|
97
|
-
|
|
97
|
+
# run_scope: targets that share an upstream document fetch it once between them
|
|
98
|
+
with client.run_scope(), ThreadPoolExecutor(max_workers=workers) as pool:
|
|
98
99
|
for target, outcome in pool.map(work, targets):
|
|
99
100
|
if isinstance(outcome, Exception):
|
|
100
101
|
result.errors.append(
|
|
@@ -75,8 +75,23 @@ def entry_body(entry: Any) -> str | None:
|
|
|
75
75
|
return html_to_text(entry.get("summary"))
|
|
76
76
|
|
|
77
77
|
|
|
78
|
-
def
|
|
79
|
-
|
|
78
|
+
def parse_document(content: bytes) -> Any:
|
|
79
|
+
"""Parse feed bytes once.
|
|
80
|
+
|
|
81
|
+
Anything that needs both a feed's metadata and entries parses once and passes the result to
|
|
82
|
+
:func:`parse_entries`
|
|
83
|
+
"""
|
|
84
|
+
return feedparser.parse(content)
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
def feed_title(parsed: Any) -> str | None:
|
|
88
|
+
"""The title the feed publishes for itself, or None if it has none."""
|
|
89
|
+
title = (parsed.feed or {}).get("title")
|
|
90
|
+
return title.strip() or None if title else None
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
def parse_entries(
|
|
94
|
+
parsed: Any,
|
|
80
95
|
*,
|
|
81
96
|
source: str,
|
|
82
97
|
target: str,
|
|
@@ -84,8 +99,7 @@ def parse_feed(
|
|
|
84
99
|
version_of: Callable[[Any], str | None] | None = None,
|
|
85
100
|
tags: tuple[str, ...] = (),
|
|
86
101
|
) -> list[Update]:
|
|
87
|
-
"""Turn feed
|
|
88
|
-
parsed = feedparser.parse(content)
|
|
102
|
+
"""Turn an already-parsed feed into Updates, newest first."""
|
|
89
103
|
updates: list[Update] = []
|
|
90
104
|
|
|
91
105
|
for entry in parsed.entries[: limit or None]:
|
|
@@ -109,3 +123,23 @@ def parse_feed(
|
|
|
109
123
|
# Undated entries sort last
|
|
110
124
|
updates.sort(key=lambda u: u.published or _UNDATED, reverse=True)
|
|
111
125
|
return updates
|
|
126
|
+
|
|
127
|
+
|
|
128
|
+
def parse_feed(
|
|
129
|
+
content: bytes,
|
|
130
|
+
*,
|
|
131
|
+
source: str,
|
|
132
|
+
target: str,
|
|
133
|
+
limit: int | None = 20,
|
|
134
|
+
version_of: Callable[[Any], str | None] | None = None,
|
|
135
|
+
tags: tuple[str, ...] = (),
|
|
136
|
+
) -> list[Update]:
|
|
137
|
+
"""Turn feed bytes into Updates, newest first."""
|
|
138
|
+
return parse_entries(
|
|
139
|
+
parse_document(content),
|
|
140
|
+
source=source,
|
|
141
|
+
target=target,
|
|
142
|
+
limit=limit,
|
|
143
|
+
version_of=version_of,
|
|
144
|
+
tags=tags,
|
|
145
|
+
)
|
|
@@ -13,13 +13,11 @@ import re
|
|
|
13
13
|
from typing import Any, ClassVar, Iterable
|
|
14
14
|
from urllib.parse import urlsplit
|
|
15
15
|
|
|
16
|
-
import feedparser
|
|
17
|
-
|
|
18
16
|
from ..http import Client
|
|
19
17
|
from ..models import Target, Update
|
|
20
18
|
from ..registry import register
|
|
21
19
|
from .base import ConfigEntryError, entry_fields, require
|
|
22
|
-
from .feed import
|
|
20
|
+
from .feed import feed_title, parse_document, parse_entries
|
|
23
21
|
|
|
24
22
|
MAX_ITEMS = 20
|
|
25
23
|
VERSION_RE = re.compile(r"\bv?(\d+\.\d+[\d.]*(?:-[A-Za-z0-9.]+)?)\b")
|
|
@@ -55,14 +53,16 @@ class FeedSource:
|
|
|
55
53
|
if fetched is None:
|
|
56
54
|
return []
|
|
57
55
|
|
|
56
|
+
# One parse feeds both the label and the entries; feedparser is the most
|
|
57
|
+
# expensive step in a run, and parsing the same bytes twice doubled it.
|
|
58
|
+
parsed = parse_document(fetched.content)
|
|
59
|
+
|
|
58
60
|
# Prefer the feed's title over raw URL
|
|
59
61
|
if not target.label:
|
|
60
|
-
|
|
61
|
-
title = (parsed.feed or {}).get("title")
|
|
62
|
-
target.label = title.strip() if title else urlsplit(target.key).netloc
|
|
62
|
+
target.label = feed_title(parsed) or urlsplit(target.key).netloc
|
|
63
63
|
|
|
64
|
-
return
|
|
65
|
-
|
|
64
|
+
return parse_entries(
|
|
65
|
+
parsed,
|
|
66
66
|
source=self.name,
|
|
67
67
|
target=target.key,
|
|
68
68
|
limit=target.params["limit"],
|
|
@@ -22,7 +22,7 @@ from ..http import Client, FetchError
|
|
|
22
22
|
from ..models import Target, Update
|
|
23
23
|
from ..registry import register
|
|
24
24
|
from .base import ConfigEntryError, entry_fields, require
|
|
25
|
-
from .feed import
|
|
25
|
+
from .feed import feed_title, parse_document, parse_entries
|
|
26
26
|
|
|
27
27
|
CHANNEL_ID_RE = re.compile(r"^UC[A-Za-z0-9_-]{22}$")
|
|
28
28
|
PLAYLIST_ID_RE = re.compile(r"^(?:PL|UU|LL|FL|OL)[A-Za-z0-9_-]{10,}$")
|
|
@@ -116,7 +116,7 @@ class YouTubeSource:
|
|
|
116
116
|
return []
|
|
117
117
|
if fetched.status in (403, 404):
|
|
118
118
|
raise FetchError(f"official feed returned {fetched.status} (likely blocked)")
|
|
119
|
-
return self._label_and_parse(target, fetched.content
|
|
119
|
+
return self._label_and_parse(target, fetched.content)
|
|
120
120
|
|
|
121
121
|
# invidious
|
|
122
122
|
|
|
@@ -247,9 +247,14 @@ class YouTubeSource:
|
|
|
247
247
|
|
|
248
248
|
# helpers
|
|
249
249
|
|
|
250
|
-
def _label_and_parse(self, target: Target, content: bytes
|
|
251
|
-
|
|
252
|
-
|
|
250
|
+
def _label_and_parse(self, target: Target, content: bytes) -> list[Update]:
|
|
251
|
+
parsed = parse_document(content)
|
|
252
|
+
# The Invidious and Data API paths both name the channel; take it from
|
|
253
|
+
# feed so the label does not depend on which path answered.
|
|
254
|
+
if not target.label:
|
|
255
|
+
target.label = feed_title(parsed)
|
|
256
|
+
updates = parse_entries(
|
|
257
|
+
parsed,
|
|
253
258
|
source=self.name,
|
|
254
259
|
target=target.key,
|
|
255
260
|
limit=MAX_ITEMS,
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|