youpdated 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- youpdated/__init__.py +3 -0
- youpdated/__main__.py +6 -0
- youpdated/cleanup.py +117 -0
- youpdated/cli.py +251 -0
- youpdated/config.py +214 -0
- youpdated/http.py +216 -0
- youpdated/models.py +63 -0
- youpdated/py.typed +0 -0
- youpdated/registry.py +49 -0
- youpdated/render/__init__.py +5 -0
- youpdated/render/json_out.py +29 -0
- youpdated/render/rss_out.py +58 -0
- youpdated/render/terminal.py +142 -0
- youpdated/runner.py +134 -0
- youpdated/sources/__init__.py +5 -0
- youpdated/sources/base.py +44 -0
- youpdated/sources/browser.py +397 -0
- youpdated/sources/feed.py +111 -0
- youpdated/sources/generic.py +76 -0
- youpdated/sources/github.py +158 -0
- youpdated/sources/itch.py +210 -0
- youpdated/sources/npm.py +88 -0
- youpdated/sources/steam.py +88 -0
- youpdated/sources/youtube.py +335 -0
- youpdated/state.py +154 -0
- youpdated-0.1.0.dist-info/METADATA +452 -0
- youpdated-0.1.0.dist-info/RECORD +31 -0
- youpdated-0.1.0.dist-info/WHEEL +5 -0
- youpdated-0.1.0.dist-info/entry_points.txt +2 -0
- youpdated-0.1.0.dist-info/licenses/LICENSE +21 -0
- youpdated-0.1.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,44 @@
|
|
|
1
|
+
"""Source plugin contract"""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from typing import Any, ClassVar, Iterable, Protocol, runtime_checkable
|
|
6
|
+
|
|
7
|
+
from ..http import Client
|
|
8
|
+
from ..models import Target, Update
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
class ConfigEntryError(Exception):
|
|
12
|
+
"""Unknown entry or formats"""
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
@runtime_checkable
|
|
16
|
+
class Source(Protocol):
|
|
17
|
+
name: ClassVar[str]
|
|
18
|
+
# shown by `youpdated sources`
|
|
19
|
+
summary: ClassVar[str]
|
|
20
|
+
|
|
21
|
+
def targets(self, entries: list[Any]) -> list[Target]:
|
|
22
|
+
"""Normalize raw config into targets"""
|
|
23
|
+
|
|
24
|
+
def fetch(self, target: Target, client: Client) -> Iterable[Update]:
|
|
25
|
+
"""Fetch current items for one target. May raise errs; collected."""
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def entry_fields(entry: Any, scalar_key: str, source: str) -> dict[str, Any]:
|
|
29
|
+
"""Accept both config shapes: a scalar or mapping.
|
|
30
|
+
|
|
31
|
+
``- python/cpython`` and ``- {repo: python/cpython, watch: [...]}`` both arrive and leave as dict
|
|
32
|
+
"""
|
|
33
|
+
if isinstance(entry, dict):
|
|
34
|
+
return dict(entry)
|
|
35
|
+
if isinstance(entry, (str, int, float)):
|
|
36
|
+
return {scalar_key: entry}
|
|
37
|
+
raise ConfigEntryError(f"sources.{source}: entry must be a value or a mapping, got {entry!r}")
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def require(fields: dict[str, Any], key: str, source: str) -> Any:
|
|
41
|
+
value = fields.get(key)
|
|
42
|
+
if value is None or (isinstance(value, str) and not value.strip()):
|
|
43
|
+
raise ConfigEntryError(f"sources.{source}: entry is missing required `{key}`")
|
|
44
|
+
return value
|
|
@@ -0,0 +1,397 @@
|
|
|
1
|
+
"""Browser releases: Chrome, Brave, Firefox, and Edge.
|
|
2
|
+
|
|
3
|
+
Each publishes version history public and unauthenticated;
|
|
4
|
+
This normalizes the four into one Update, and the platform names for the user
|
|
5
|
+
(``mac`` each translates custom)
|
|
6
|
+
|
|
7
|
+
sources:
|
|
8
|
+
browser:
|
|
9
|
+
- chrome
|
|
10
|
+
- brave
|
|
11
|
+
- browser: chrome
|
|
12
|
+
platform: windows
|
|
13
|
+
channel: beta
|
|
14
|
+
"""
|
|
15
|
+
|
|
16
|
+
from __future__ import annotations
|
|
17
|
+
|
|
18
|
+
import re
|
|
19
|
+
import sys
|
|
20
|
+
from datetime import datetime, timezone
|
|
21
|
+
from typing import Any, ClassVar, Iterable
|
|
22
|
+
|
|
23
|
+
from ..http import Client
|
|
24
|
+
from ..models import Target, Update
|
|
25
|
+
from ..registry import register
|
|
26
|
+
from .base import ConfigEntryError, entry_fields, require
|
|
27
|
+
from .feed import html_to_text, parse_feed
|
|
28
|
+
|
|
29
|
+
MAX_VERSIONS = 10
|
|
30
|
+
VERSION_RE = re.compile(r"v?(\d+(?:\.\d+)+)")
|
|
31
|
+
|
|
32
|
+
# user input -> vendor name
|
|
33
|
+
PLATFORM_ALIASES = {
|
|
34
|
+
"mac": "mac", "macos": "mac", "osx": "mac", "darwin": "mac",
|
|
35
|
+
"win": "windows", "windows": "windows", "win64": "windows", "win32": "windows",
|
|
36
|
+
"linux": "linux",
|
|
37
|
+
"android": "android",
|
|
38
|
+
"ios": "ios",
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
CHROME_PLATFORMS = {
|
|
42
|
+
"mac": "mac", "windows": "win64", "linux": "linux",
|
|
43
|
+
"android": "android", "ios": "ios",
|
|
44
|
+
}
|
|
45
|
+
CHROME_CHANNELS = ("stable", "beta", "dev", "canary", "extended")
|
|
46
|
+
|
|
47
|
+
EDGE_PLATFORMS = {
|
|
48
|
+
"mac": "MacOS", "windows": "Windows", "linux": "Linux",
|
|
49
|
+
"android": "Android", "ios": "iOS",
|
|
50
|
+
}
|
|
51
|
+
EDGE_CHANNELS = {"stable": "Stable", "beta": "Beta", "dev": "Dev", "canary": "Canary"}
|
|
52
|
+
# Microsoft only publishes release notes for stable and beta. (Again, who uses EDGE?)
|
|
53
|
+
EDGE_RELNOTES = {
|
|
54
|
+
"stable": "https://learn.microsoft.com/deployedge/microsoft-edge-relnote-stable-channel",
|
|
55
|
+
"beta": "https://learn.microsoft.com/deployedge/microsoft-edge-relnote-beta-channel",
|
|
56
|
+
}
|
|
57
|
+
EDGE_RELNOTE_FALLBACK = "https://www.microsoft.com/edge/download/insider"
|
|
58
|
+
|
|
59
|
+
# Brave publishes every channel into one repo feed, distinguished by title prefix ("Release v1.95.79", "Beta v1.94.112", "Nightly v1.96.0")
|
|
60
|
+
BRAVE_CHANNELS = {
|
|
61
|
+
"stable": "Release",
|
|
62
|
+
"release": "Release",
|
|
63
|
+
"beta": "Beta",
|
|
64
|
+
"nightly": "Nightly",
|
|
65
|
+
"dev": "Dev",
|
|
66
|
+
}
|
|
67
|
+
|
|
68
|
+
# Mozilla publishes one document
|
|
69
|
+
FIREFOX_CHANNELS = {
|
|
70
|
+
"stable": ("LATEST_FIREFOX_VERSION", "LAST_RELEASE_DATE"),
|
|
71
|
+
"release": ("LATEST_FIREFOX_VERSION", "LAST_RELEASE_DATE"),
|
|
72
|
+
"beta": ("LATEST_FIREFOX_DEVEL_VERSION", None),
|
|
73
|
+
"dev": ("FIREFOX_DEVEDITION", None),
|
|
74
|
+
"nightly": ("FIREFOX_NIGHTLY", None),
|
|
75
|
+
"esr": ("FIREFOX_ESR", None),
|
|
76
|
+
}
|
|
77
|
+
|
|
78
|
+
BROWSERS = ("chrome", "brave", "firefox", "edge")
|
|
79
|
+
|
|
80
|
+
def _host_platform() -> str:
|
|
81
|
+
if sys.platform.startswith("darwin"):
|
|
82
|
+
return "mac"
|
|
83
|
+
if sys.platform.startswith("win"):
|
|
84
|
+
return "windows"
|
|
85
|
+
return "linux"
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
@register
|
|
89
|
+
class BrowserSource:
|
|
90
|
+
name: ClassVar[str] = "browser"
|
|
91
|
+
summary: ClassVar[str] = "New releases of Chrome, Brave, Firefox, or Edge"
|
|
92
|
+
|
|
93
|
+
def targets(self, entries: list[Any]) -> list[Target]:
|
|
94
|
+
targets = []
|
|
95
|
+
for entry in entries:
|
|
96
|
+
fields = entry_fields(entry, "browser", self.name)
|
|
97
|
+
browser = str(require(fields, "browser", self.name)).strip().lower()
|
|
98
|
+
if browser not in BROWSERS:
|
|
99
|
+
raise ConfigEntryError(
|
|
100
|
+
f"sources.{self.name}: unknown browser `{browser}`; "
|
|
101
|
+
f"valid options are {list(BROWSERS)}"
|
|
102
|
+
)
|
|
103
|
+
|
|
104
|
+
raw_platform = str(fields.get("platform") or _host_platform()).lower()
|
|
105
|
+
platform = PLATFORM_ALIASES.get(raw_platform)
|
|
106
|
+
if platform is None:
|
|
107
|
+
raise ConfigEntryError(
|
|
108
|
+
f"sources.{self.name}: unknown platform `{raw_platform}`; "
|
|
109
|
+
f"valid options are {sorted(set(PLATFORM_ALIASES.values()))}"
|
|
110
|
+
)
|
|
111
|
+
|
|
112
|
+
channel = str(fields.get("channel") or "stable").lower()
|
|
113
|
+
self._validate_channel(browser, channel)
|
|
114
|
+
|
|
115
|
+
# Brave publishes one feed for every platform, so the platform is not part of identity but channel is.
|
|
116
|
+
key = (
|
|
117
|
+
f"brave/{channel}"
|
|
118
|
+
if browser == "brave"
|
|
119
|
+
else f"{browser}/{platform}/{channel}"
|
|
120
|
+
)
|
|
121
|
+
targets.append(
|
|
122
|
+
Target(
|
|
123
|
+
source=self.name,
|
|
124
|
+
key=key,
|
|
125
|
+
label=fields.get("name") or _pretty(browser, platform, channel),
|
|
126
|
+
params={"browser": browser, "platform": platform, "channel": channel},
|
|
127
|
+
)
|
|
128
|
+
)
|
|
129
|
+
return targets
|
|
130
|
+
|
|
131
|
+
def _validate_channel(self, browser: str, channel: str) -> None:
|
|
132
|
+
valid = {
|
|
133
|
+
"chrome": CHROME_CHANNELS,
|
|
134
|
+
"edge": tuple(EDGE_CHANNELS),
|
|
135
|
+
"firefox": tuple(FIREFOX_CHANNELS),
|
|
136
|
+
"brave": tuple(BRAVE_CHANNELS),
|
|
137
|
+
}[browser]
|
|
138
|
+
if channel not in valid:
|
|
139
|
+
raise ConfigEntryError(
|
|
140
|
+
f"sources.{self.name}: `{channel}` is not a {browser} channel; "
|
|
141
|
+
f"valid options are {list(valid)}"
|
|
142
|
+
)
|
|
143
|
+
|
|
144
|
+
def fetch(self, target: Target, client: Client) -> Iterable[Update]:
|
|
145
|
+
browser = target.params["browser"]
|
|
146
|
+
return {
|
|
147
|
+
"chrome": self._chrome,
|
|
148
|
+
"brave": self._brave,
|
|
149
|
+
"firefox": self._firefox,
|
|
150
|
+
"edge": self._edge,
|
|
151
|
+
}[browser](target, client)
|
|
152
|
+
|
|
153
|
+
# Chrome
|
|
154
|
+
|
|
155
|
+
def _chrome(self, target: Target, client: Client) -> list[Update]:
|
|
156
|
+
platform = CHROME_PLATFORMS[target.params["platform"]]
|
|
157
|
+
channel = target.params["channel"]
|
|
158
|
+
url = (
|
|
159
|
+
"https://versionhistory.googleapis.com/v1/chrome/platforms/"
|
|
160
|
+
f"{platform}/channels/{channel}/versions/all/releases"
|
|
161
|
+
)
|
|
162
|
+
fetched = client.get(url, conditional=True)
|
|
163
|
+
if fetched is None:
|
|
164
|
+
return []
|
|
165
|
+
|
|
166
|
+
# One version can appear several times as its rollout fraction grows: keeps the earliest start time per version.
|
|
167
|
+
starts: dict[str, datetime] = {}
|
|
168
|
+
for release in fetched.json().get("releases", []):
|
|
169
|
+
version = release.get("version")
|
|
170
|
+
started = _parse_iso((release.get("serving") or {}).get("startTime"))
|
|
171
|
+
if not version:
|
|
172
|
+
continue
|
|
173
|
+
if version not in starts or (started and started < starts[version]):
|
|
174
|
+
starts[version] = started
|
|
175
|
+
|
|
176
|
+
ordered = sorted(
|
|
177
|
+
starts.items(), key=lambda kv: kv[1] or datetime.min.replace(tzinfo=timezone.utc),
|
|
178
|
+
reverse=True,
|
|
179
|
+
)
|
|
180
|
+
return [
|
|
181
|
+
Update(
|
|
182
|
+
source=self.name,
|
|
183
|
+
target=target.key,
|
|
184
|
+
uid=f"chrome:{platform}:{channel}:{version}",
|
|
185
|
+
title=f"Chrome {version}",
|
|
186
|
+
url="https://chromereleases.googleblog.com/",
|
|
187
|
+
published=started,
|
|
188
|
+
version=version,
|
|
189
|
+
body=f"{channel} channel on {target.params['platform']}",
|
|
190
|
+
tags=("release", channel),
|
|
191
|
+
)
|
|
192
|
+
for version, started in ordered[:MAX_VERSIONS]
|
|
193
|
+
]
|
|
194
|
+
|
|
195
|
+
# Brave
|
|
196
|
+
|
|
197
|
+
def _brave(self, target: Target, client: Client) -> list[Update]:
|
|
198
|
+
"""Brave ships all channels as GitHub releases in one repo.
|
|
199
|
+
|
|
200
|
+
The ``.atom`` feed only carries the latest 10, and Brave has nightlies
|
|
201
|
+
The REST API is primary; the atom as fallback for rate-limits.
|
|
202
|
+
"""
|
|
203
|
+
channel = target.params["channel"]
|
|
204
|
+
prefix = BRAVE_CHANNELS[channel].lower()
|
|
205
|
+
|
|
206
|
+
fetched = client.get(
|
|
207
|
+
"https://api.github.com/repos/brave/brave-browser/releases?per_page=100",
|
|
208
|
+
conditional=True,
|
|
209
|
+
headers={"Accept": "application/vnd.github+json"},
|
|
210
|
+
soft_statuses=(403, 429),
|
|
211
|
+
)
|
|
212
|
+
if fetched is not None and fetched.status == 200:
|
|
213
|
+
return self._brave_from_api(target, fetched.json(), channel, prefix)
|
|
214
|
+
if fetched is None:
|
|
215
|
+
return [] # 304: nothing changed
|
|
216
|
+
|
|
217
|
+
client.note("brave: GitHub API unavailable, falling back to the atom feed")
|
|
218
|
+
return self._brave_from_atom(target, client, channel, prefix)
|
|
219
|
+
|
|
220
|
+
def _brave_from_api(
|
|
221
|
+
self, target: Target, releases: list[dict], channel: str, prefix: str
|
|
222
|
+
) -> list[Update]:
|
|
223
|
+
updates = []
|
|
224
|
+
for release in releases:
|
|
225
|
+
if release.get("draft"):
|
|
226
|
+
continue
|
|
227
|
+
# Release names can carry trailing whitespace and newlines
|
|
228
|
+
name = (release.get("name") or release.get("tag_name") or "").strip()
|
|
229
|
+
if not name.lower().startswith(prefix):
|
|
230
|
+
continue
|
|
231
|
+
tag = release.get("tag_name") or name
|
|
232
|
+
updates.append(
|
|
233
|
+
Update(
|
|
234
|
+
source=self.name,
|
|
235
|
+
target=target.key,
|
|
236
|
+
uid=f"brave:{tag}",
|
|
237
|
+
title=name,
|
|
238
|
+
url=release.get("html_url", ""),
|
|
239
|
+
published=_parse_iso(release.get("published_at")),
|
|
240
|
+
version=_version_from_title(tag) or _version_from_title(name),
|
|
241
|
+
body=html_to_text(release.get("body")),
|
|
242
|
+
tags=("release", channel),
|
|
243
|
+
)
|
|
244
|
+
)
|
|
245
|
+
if len(updates) >= MAX_VERSIONS:
|
|
246
|
+
break
|
|
247
|
+
return updates
|
|
248
|
+
|
|
249
|
+
def _brave_from_atom(
|
|
250
|
+
self, target: Target, client: Client, channel: str, prefix: str
|
|
251
|
+
) -> list[Update]:
|
|
252
|
+
fetched = client.get("https://github.com/brave/brave-browser/releases.atom")
|
|
253
|
+
if fetched is None:
|
|
254
|
+
return []
|
|
255
|
+
updates = parse_feed(
|
|
256
|
+
fetched.content,
|
|
257
|
+
source=self.name,
|
|
258
|
+
target=target.key,
|
|
259
|
+
limit=None, # filtered below, don't truncate before
|
|
260
|
+
tags=("release", channel),
|
|
261
|
+
)
|
|
262
|
+
return [
|
|
263
|
+
Update(
|
|
264
|
+
source=u.source,
|
|
265
|
+
target=u.target,
|
|
266
|
+
# Keyed on the tag so the two paths dedupe
|
|
267
|
+
uid=f"brave:{_tag_from_url(u.url) or u.uid}",
|
|
268
|
+
title=u.title,
|
|
269
|
+
url=u.url,
|
|
270
|
+
published=u.published,
|
|
271
|
+
version=_version_from_title(u.title),
|
|
272
|
+
body=u.body,
|
|
273
|
+
tags=u.tags,
|
|
274
|
+
)
|
|
275
|
+
for u in updates
|
|
276
|
+
if u.title.strip().lower().startswith(prefix)
|
|
277
|
+
][:MAX_VERSIONS]
|
|
278
|
+
|
|
279
|
+
# Firefox
|
|
280
|
+
|
|
281
|
+
def _firefox(self, target: Target, client: Client) -> list[Update]:
|
|
282
|
+
fetched = client.get(
|
|
283
|
+
"https://product-details.mozilla.org/1.0/firefox_versions.json", conditional=True
|
|
284
|
+
)
|
|
285
|
+
if fetched is None:
|
|
286
|
+
return []
|
|
287
|
+
|
|
288
|
+
doc = fetched.json()
|
|
289
|
+
channel = target.params["channel"]
|
|
290
|
+
version_key, date_key = FIREFOX_CHANNELS[channel]
|
|
291
|
+
version = doc.get(version_key)
|
|
292
|
+
if not version:
|
|
293
|
+
return []
|
|
294
|
+
|
|
295
|
+
# Mozilla publishes only current versions
|
|
296
|
+
# one item per channel; the uid carries the version so it reports once.
|
|
297
|
+
return [
|
|
298
|
+
Update(
|
|
299
|
+
source=self.name,
|
|
300
|
+
target=target.key,
|
|
301
|
+
uid=f"firefox:{channel}:{version}",
|
|
302
|
+
title=f"Firefox {version}",
|
|
303
|
+
url="https://www.mozilla.org/firefox/releases/",
|
|
304
|
+
published=_parse_date(doc.get(date_key)) if date_key else None,
|
|
305
|
+
version=version,
|
|
306
|
+
body=f"{channel} channel",
|
|
307
|
+
tags=("release", channel),
|
|
308
|
+
)
|
|
309
|
+
]
|
|
310
|
+
|
|
311
|
+
# Edge
|
|
312
|
+
|
|
313
|
+
def _edge(self, target: Target, client: Client) -> list[Update]:
|
|
314
|
+
fetched = client.get("https://edgeupdates.microsoft.com/api/products", conditional=True)
|
|
315
|
+
if fetched is None:
|
|
316
|
+
return []
|
|
317
|
+
|
|
318
|
+
product_name = EDGE_CHANNELS[target.params["channel"]]
|
|
319
|
+
platform = EDGE_PLATFORMS[target.params["platform"]]
|
|
320
|
+
|
|
321
|
+
product = next(
|
|
322
|
+
(p for p in fetched.json() if p.get("Product") == product_name), None
|
|
323
|
+
)
|
|
324
|
+
if product is None:
|
|
325
|
+
return []
|
|
326
|
+
|
|
327
|
+
seen: dict[str, datetime | None] = {}
|
|
328
|
+
for release in product.get("Releases", []):
|
|
329
|
+
if release.get("Platform") != platform:
|
|
330
|
+
continue
|
|
331
|
+
version = release.get("ProductVersion")
|
|
332
|
+
published = _parse_iso(release.get("PublishedTime"))
|
|
333
|
+
if version and version not in seen:
|
|
334
|
+
seen[version] = published
|
|
335
|
+
|
|
336
|
+
ordered = sorted(
|
|
337
|
+
seen.items(), key=lambda kv: kv[1] or datetime.min.replace(tzinfo=timezone.utc),
|
|
338
|
+
reverse=True,
|
|
339
|
+
)
|
|
340
|
+
return [
|
|
341
|
+
Update(
|
|
342
|
+
source=self.name,
|
|
343
|
+
target=target.key,
|
|
344
|
+
uid=f"edge:{product_name}:{platform}:{version}",
|
|
345
|
+
title=f"Edge {version}",
|
|
346
|
+
url=EDGE_RELNOTES.get(target.params["channel"], EDGE_RELNOTE_FALLBACK),
|
|
347
|
+
published=published,
|
|
348
|
+
version=version,
|
|
349
|
+
body=f"{product_name} channel on {platform}",
|
|
350
|
+
tags=("release", target.params["channel"]),
|
|
351
|
+
)
|
|
352
|
+
for version, published in ordered[:MAX_VERSIONS]
|
|
353
|
+
]
|
|
354
|
+
|
|
355
|
+
|
|
356
|
+
def _pretty(browser: str, platform: str, channel: str) -> str:
|
|
357
|
+
names = {"chrome": "Chrome", "brave": "Brave", "firefox": "Firefox", "edge": "Edge"}
|
|
358
|
+
suffix = "" if channel in ("stable", "release") else f" {channel}"
|
|
359
|
+
if browser == "brave":
|
|
360
|
+
return f"Brave{suffix}"
|
|
361
|
+
return f"{names[browser]}{suffix} ({platform})"
|
|
362
|
+
|
|
363
|
+
|
|
364
|
+
def _tag_from_url(url: str) -> str | None:
|
|
365
|
+
match = re.search(r"/releases/tag/([^/?#]+)", url or "")
|
|
366
|
+
return match.group(1) if match else None
|
|
367
|
+
|
|
368
|
+
|
|
369
|
+
def _version_from_title(title: str) -> str | None:
|
|
370
|
+
match = VERSION_RE.search(title or "")
|
|
371
|
+
return match.group(1) if match else None
|
|
372
|
+
|
|
373
|
+
|
|
374
|
+
def _parse_iso(value: str | None) -> datetime | None:
|
|
375
|
+
if not value:
|
|
376
|
+
return None
|
|
377
|
+
text = str(value).replace("Z", "+00:00")
|
|
378
|
+
# Some have more digits than fromisoformat accepts
|
|
379
|
+
if "." in text:
|
|
380
|
+
head, _, tail = text.partition(".")
|
|
381
|
+
digits = "".join(c for c in tail if c.isdigit())[:6]
|
|
382
|
+
rest = tail[len(digits):] if tail[len(digits):].startswith(("+", "-")) else ""
|
|
383
|
+
text = f"{head}.{digits or '0'}{rest}"
|
|
384
|
+
try:
|
|
385
|
+
parsed = datetime.fromisoformat(text)
|
|
386
|
+
except ValueError:
|
|
387
|
+
return None
|
|
388
|
+
return parsed if parsed.tzinfo else parsed.replace(tzinfo=timezone.utc)
|
|
389
|
+
|
|
390
|
+
|
|
391
|
+
def _parse_date(value: str | None) -> datetime | None:
|
|
392
|
+
if not value:
|
|
393
|
+
return None
|
|
394
|
+
try:
|
|
395
|
+
return datetime.strptime(str(value), "%Y-%m-%d").replace(tzinfo=timezone.utc)
|
|
396
|
+
except ValueError:
|
|
397
|
+
return _parse_iso(value)
|
|
@@ -0,0 +1,111 @@
|
|
|
1
|
+
"""Shared RSS/Atom handling
|
|
2
|
+
|
|
3
|
+
Steam, itch, GitHub, and YouTube all publish feeds, so entry to update mapping is here
|
|
4
|
+
"""
|
|
5
|
+
|
|
6
|
+
from __future__ import annotations
|
|
7
|
+
|
|
8
|
+
import re
|
|
9
|
+
from datetime import datetime, timezone
|
|
10
|
+
from html.parser import HTMLParser
|
|
11
|
+
from typing import Any, Callable
|
|
12
|
+
|
|
13
|
+
import feedparser
|
|
14
|
+
|
|
15
|
+
from ..models import Update
|
|
16
|
+
|
|
17
|
+
BODY_LIMIT = 400
|
|
18
|
+
_UNDATED = datetime(1970, 1, 1, tzinfo=timezone.utc)
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
class _TextExtractor(HTMLParser):
|
|
22
|
+
def __init__(self) -> None:
|
|
23
|
+
super().__init__(convert_charrefs=True)
|
|
24
|
+
self.parts: list[str] = []
|
|
25
|
+
self._skip = 0
|
|
26
|
+
|
|
27
|
+
def handle_starttag(self, tag: str, attrs: Any) -> None:
|
|
28
|
+
if tag in {"script", "style"}:
|
|
29
|
+
self._skip += 1
|
|
30
|
+
elif tag in {"br", "p", "div", "li", "tr", "h1", "h2", "h3"}:
|
|
31
|
+
self.parts.append("\n")
|
|
32
|
+
|
|
33
|
+
def handle_endtag(self, tag: str) -> None:
|
|
34
|
+
if tag in {"script", "style"} and self._skip:
|
|
35
|
+
self._skip -= 1
|
|
36
|
+
|
|
37
|
+
def handle_data(self, data: str) -> None:
|
|
38
|
+
if not self._skip:
|
|
39
|
+
self.parts.append(data)
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def html_to_text(html: str | None, limit: int | None = BODY_LIMIT) -> str | None:
|
|
43
|
+
"""Flatten feed HTML into plain-text summary"""
|
|
44
|
+
if not html:
|
|
45
|
+
return None
|
|
46
|
+
parser = _TextExtractor()
|
|
47
|
+
try:
|
|
48
|
+
parser.feed(html)
|
|
49
|
+
parser.close()
|
|
50
|
+
except Exception:
|
|
51
|
+
parser.parts = [re.sub(r"<[^>]+>", " ", html)]
|
|
52
|
+
text = "".join(parser.parts)
|
|
53
|
+
text = re.sub(r"[ \t\r\f\v]+", " ", text)
|
|
54
|
+
text = re.sub(r"\n\s*\n+", "\n", text).strip()
|
|
55
|
+
if not text:
|
|
56
|
+
return None
|
|
57
|
+
if limit and len(text) > limit:
|
|
58
|
+
text = text[: limit - 1].rstrip() + "…"
|
|
59
|
+
return text
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
def entry_datetime(entry: Any) -> datetime | None:
|
|
63
|
+
for key in ("published_parsed", "updated_parsed", "created_parsed"):
|
|
64
|
+
parsed = entry.get(key)
|
|
65
|
+
if parsed:
|
|
66
|
+
# feedparser normalizes struct_time to UTC
|
|
67
|
+
return datetime(*parsed[:6], tzinfo=timezone.utc)
|
|
68
|
+
return None
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def entry_body(entry: Any) -> str | None:
|
|
72
|
+
content = entry.get("content")
|
|
73
|
+
if content and isinstance(content, list) and content[0].get("value"):
|
|
74
|
+
return html_to_text(content[0]["value"])
|
|
75
|
+
return html_to_text(entry.get("summary"))
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def parse_feed(
|
|
79
|
+
content: bytes,
|
|
80
|
+
*,
|
|
81
|
+
source: str,
|
|
82
|
+
target: str,
|
|
83
|
+
limit: int | None = 20,
|
|
84
|
+
version_of: Callable[[Any], str | None] | None = None,
|
|
85
|
+
tags: tuple[str, ...] = (),
|
|
86
|
+
) -> list[Update]:
|
|
87
|
+
"""Turn feed bytes into Updates, newest first."""
|
|
88
|
+
parsed = feedparser.parse(content)
|
|
89
|
+
updates: list[Update] = []
|
|
90
|
+
|
|
91
|
+
for entry in parsed.entries[: limit or None]:
|
|
92
|
+
link = entry.get("link") or ""
|
|
93
|
+
title = (entry.get("title") or "").strip() or "(untitled)"
|
|
94
|
+
uid = entry.get("id") or link or f"{title}|{entry.get('published', '')}"
|
|
95
|
+
updates.append(
|
|
96
|
+
Update(
|
|
97
|
+
source=source,
|
|
98
|
+
target=target,
|
|
99
|
+
uid=str(uid),
|
|
100
|
+
title=title,
|
|
101
|
+
url=link,
|
|
102
|
+
published=entry_datetime(entry),
|
|
103
|
+
version=version_of(entry) if version_of else None,
|
|
104
|
+
body=entry_body(entry),
|
|
105
|
+
tags=tags,
|
|
106
|
+
)
|
|
107
|
+
)
|
|
108
|
+
|
|
109
|
+
# Undated entries sort last
|
|
110
|
+
updates.sort(key=lambda u: u.published or _UNDATED, reverse=True)
|
|
111
|
+
return updates
|
|
@@ -0,0 +1,76 @@
|
|
|
1
|
+
"""Any RSS or Atom feed
|
|
2
|
+
|
|
3
|
+
sources:
|
|
4
|
+
feed:
|
|
5
|
+
- https://blog.rust-lang.org/feed.xml
|
|
6
|
+
- url: https://github.com/obsidianmd/obsidian-releases/releases.atom
|
|
7
|
+
name: Obsidian
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
import re
|
|
13
|
+
from typing import Any, ClassVar, Iterable
|
|
14
|
+
from urllib.parse import urlsplit
|
|
15
|
+
|
|
16
|
+
import feedparser
|
|
17
|
+
|
|
18
|
+
from ..http import Client
|
|
19
|
+
from ..models import Target, Update
|
|
20
|
+
from ..registry import register
|
|
21
|
+
from .base import ConfigEntryError, entry_fields, require
|
|
22
|
+
from .feed import parse_feed
|
|
23
|
+
|
|
24
|
+
MAX_ITEMS = 20
|
|
25
|
+
VERSION_RE = re.compile(r"\bv?(\d+\.\d+[\d.]*(?:-[A-Za-z0-9.]+)?)\b")
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
@register
|
|
29
|
+
class FeedSource:
|
|
30
|
+
name: ClassVar[str] = "feed"
|
|
31
|
+
summary: ClassVar[str] = "Any RSS or Atom feed, for apps without a dedicated source"
|
|
32
|
+
|
|
33
|
+
def targets(self, entries: list[Any]) -> list[Target]:
|
|
34
|
+
targets = []
|
|
35
|
+
for entry in entries:
|
|
36
|
+
fields = entry_fields(entry, "url", self.name)
|
|
37
|
+
url = str(require(fields, "url", self.name)).strip()
|
|
38
|
+
parts = urlsplit(url)
|
|
39
|
+
if parts.scheme not in ("http", "https") or not parts.netloc:
|
|
40
|
+
raise ConfigEntryError(
|
|
41
|
+
f"sources.{self.name}: `{url}` is not an http(s) feed URL"
|
|
42
|
+
)
|
|
43
|
+
targets.append(
|
|
44
|
+
Target(
|
|
45
|
+
source=self.name,
|
|
46
|
+
key=url,
|
|
47
|
+
label=fields.get("name"),
|
|
48
|
+
params={"limit": int(fields.get("limit") or MAX_ITEMS)},
|
|
49
|
+
)
|
|
50
|
+
)
|
|
51
|
+
return targets
|
|
52
|
+
|
|
53
|
+
def fetch(self, target: Target, client: Client) -> Iterable[Update]:
|
|
54
|
+
fetched = client.get(target.key, conditional=True)
|
|
55
|
+
if fetched is None:
|
|
56
|
+
return []
|
|
57
|
+
|
|
58
|
+
# Prefer the feed's title over raw URL
|
|
59
|
+
if not target.label:
|
|
60
|
+
parsed = feedparser.parse(fetched.content)
|
|
61
|
+
title = (parsed.feed or {}).get("title")
|
|
62
|
+
target.label = title.strip() if title else urlsplit(target.key).netloc
|
|
63
|
+
|
|
64
|
+
return parse_feed(
|
|
65
|
+
fetched.content,
|
|
66
|
+
source=self.name,
|
|
67
|
+
target=target.key,
|
|
68
|
+
limit=target.params["limit"],
|
|
69
|
+
version_of=_version_from_title,
|
|
70
|
+
tags=("item",),
|
|
71
|
+
)
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def _version_from_title(entry: Any) -> str | None:
|
|
75
|
+
match = VERSION_RE.search(entry.get("title") or "")
|
|
76
|
+
return match.group(1) if match else None
|