rssfeed 0.4.2__tar.gz → 0.4.4__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {rssfeed-0.4.2 → rssfeed-0.4.4}/PKG-INFO +2 -2
- {rssfeed-0.4.2 → rssfeed-0.4.4}/pyproject.toml +2 -4
- {rssfeed-0.4.2 → rssfeed-0.4.4}/rssfeed/lib.py +38 -20
- {rssfeed-0.4.2 → rssfeed-0.4.4}/LICENSE +0 -0
- {rssfeed-0.4.2 → rssfeed-0.4.4}/README.md +0 -0
- {rssfeed-0.4.2 → rssfeed-0.4.4}/rssfeed/__init__.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.1
|
|
2
2
|
Name: rssfeed
|
|
3
|
-
Version: 0.4.
|
|
3
|
+
Version: 0.4.4
|
|
4
4
|
Summary: A simple rss/atom feed parser
|
|
5
5
|
Keywords: rssfeed,rss,feed,atom,opml
|
|
6
6
|
Author: p7e4
|
|
@@ -8,12 +8,12 @@ License: GPLv3
|
|
|
8
8
|
Classifier: Programming Language :: Python :: 3
|
|
9
9
|
Classifier: Programming Language :: Python :: 3.12
|
|
10
10
|
Classifier: Programming Language :: Python :: 3.13
|
|
11
|
+
Classifier: Programming Language :: Python :: 3.14
|
|
11
12
|
Classifier: License :: OSI Approved :: GNU General Public License v3 (GPLv3)
|
|
12
13
|
Classifier: Topic :: Text Processing :: Markup :: XML
|
|
13
14
|
Project-URL: Repository, https://github.com/p7e4/rssfeed
|
|
14
15
|
Project-URL: Issues, https://github.com/p7e4/rssfeed/issues
|
|
15
16
|
Requires-Python: >=3.12
|
|
16
|
-
Requires-Dist: python-dateutil>=2.9
|
|
17
17
|
Description-Content-Type: text/markdown
|
|
18
18
|
|
|
19
19
|
# rssfeed
|
|
@@ -11,9 +11,6 @@ authors = [
|
|
|
11
11
|
{ name = "p7e4" },
|
|
12
12
|
]
|
|
13
13
|
requires-python = ">= 3.12"
|
|
14
|
-
dependencies = [
|
|
15
|
-
"python-dateutil >= 2.9",
|
|
16
|
-
]
|
|
17
14
|
keywords = [
|
|
18
15
|
"rssfeed",
|
|
19
16
|
"rss",
|
|
@@ -25,10 +22,11 @@ classifiers = [
|
|
|
25
22
|
"Programming Language :: Python :: 3",
|
|
26
23
|
"Programming Language :: Python :: 3.12",
|
|
27
24
|
"Programming Language :: Python :: 3.13",
|
|
25
|
+
"Programming Language :: Python :: 3.14",
|
|
28
26
|
"License :: OSI Approved :: GNU General Public License v3 (GPLv3)",
|
|
29
27
|
"Topic :: Text Processing :: Markup :: XML",
|
|
30
28
|
]
|
|
31
|
-
version = "0.4.
|
|
29
|
+
version = "0.4.4"
|
|
32
30
|
|
|
33
31
|
[project.license]
|
|
34
32
|
text = "GPLv3"
|
|
@@ -1,7 +1,8 @@
|
|
|
1
|
-
from
|
|
1
|
+
from email.utils import parsedate_to_datetime
|
|
2
2
|
from xml.etree import ElementTree
|
|
3
|
+
from datetime import datetime
|
|
3
4
|
|
|
4
|
-
__version__ = "0.4.
|
|
5
|
+
__version__ = "0.4.4"
|
|
5
6
|
|
|
6
7
|
class ParseError(Exception):
|
|
7
8
|
pass
|
|
@@ -17,14 +18,26 @@ def _parse(data):
|
|
|
17
18
|
raise ParseError("xml parse fail") from e
|
|
18
19
|
return parser
|
|
19
20
|
|
|
21
|
+
def timeParse(s):
|
|
22
|
+
if not s: return 0
|
|
23
|
+
try:
|
|
24
|
+
if s.isdigit():
|
|
25
|
+
return int(s)
|
|
26
|
+
if len(s) > 4 and s[4] == "-":
|
|
27
|
+
t = datetime.fromisoformat(s.replace("Z", "+00:00"))
|
|
28
|
+
else:
|
|
29
|
+
t = parsedate_to_datetime(s)
|
|
30
|
+
return int(t.timestamp())
|
|
31
|
+
except (TypeError, ValueError):
|
|
32
|
+
return 0
|
|
33
|
+
|
|
20
34
|
def parse(data, url=None):
|
|
21
|
-
if url: url = url[:8] + url[8:].
|
|
35
|
+
if url: url = url[:8] + url[8:].split("/")[0]
|
|
22
36
|
items = list()
|
|
23
37
|
for event, elem in _parse(data).read_events():
|
|
24
|
-
tag = elem.tag.
|
|
25
|
-
text = elem.text.strip() if elem.text else str()
|
|
38
|
+
tag = elem.tag.rsplit("}", 1)[-1]
|
|
26
39
|
if event == "start":
|
|
27
|
-
if tag in ("channel", "
|
|
40
|
+
if tag in ("channel", "feed", "item", "entry"):
|
|
28
41
|
items.append({
|
|
29
42
|
"title": str(),
|
|
30
43
|
"author": str(),
|
|
@@ -33,31 +46,36 @@ def parse(data, url=None):
|
|
|
33
46
|
"content": str()
|
|
34
47
|
})
|
|
35
48
|
else:
|
|
49
|
+
if not (elem.text and (text:=elem.text.strip())) and tag != "link":
|
|
50
|
+
continue
|
|
36
51
|
i = items[-1]
|
|
37
52
|
match tag:
|
|
38
|
-
case "
|
|
53
|
+
case "content" | "encoded":
|
|
39
54
|
i["content"] = text
|
|
40
|
-
case "
|
|
41
|
-
if text
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
raise ParseError("time parse fail") from e
|
|
55
|
+
case "summary" | "description":
|
|
56
|
+
if not i["content"]: i["content"] = text
|
|
57
|
+
case "pubDate" | "published" | "date":
|
|
58
|
+
try:
|
|
59
|
+
if not i["timestamp"]: i["timestamp"] = timeParse(text)
|
|
60
|
+
except Exception as e:
|
|
61
|
+
raise ParseError("time parse fail") from e
|
|
48
62
|
case "link":
|
|
49
|
-
i["url"] =
|
|
50
|
-
if
|
|
63
|
+
if not i["url"]: i["url"] = elem.get("href") or text
|
|
64
|
+
if not i["url"].startswith(("http://", "https://")) and url:
|
|
51
65
|
i["url"] = f"{url}/{i["url"].lstrip("/")}"
|
|
52
|
-
case "
|
|
53
|
-
i[
|
|
66
|
+
case "author" | "name" | "creator":
|
|
67
|
+
if not i["author"]:
|
|
68
|
+
i["author"] = text
|
|
69
|
+
case "title":
|
|
70
|
+
if not i["title"]:
|
|
71
|
+
i[title] = text
|
|
54
72
|
|
|
55
73
|
if not items:
|
|
56
74
|
raise ParseError("not valid result")
|
|
57
75
|
|
|
58
76
|
feed = {
|
|
59
77
|
"name": items[0]["title"],
|
|
60
|
-
"lastupdate":
|
|
78
|
+
"lastupdate": max(i["timestamp"] for i in items[1:]),
|
|
61
79
|
"items": items[1:]
|
|
62
80
|
}
|
|
63
81
|
|
|
File without changes
|
|
File without changes
|
|
File without changes
|