rssfeed 0.4.2__tar.gz → 0.4.4__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.1
2
2
  Name: rssfeed
3
- Version: 0.4.2
3
+ Version: 0.4.4
4
4
  Summary: A simple rss/atom feed parser
5
5
  Keywords: rssfeed,rss,feed,atom,opml
6
6
  Author: p7e4
@@ -8,12 +8,12 @@ License: GPLv3
8
8
  Classifier: Programming Language :: Python :: 3
9
9
  Classifier: Programming Language :: Python :: 3.12
10
10
  Classifier: Programming Language :: Python :: 3.13
11
+ Classifier: Programming Language :: Python :: 3.14
11
12
  Classifier: License :: OSI Approved :: GNU General Public License v3 (GPLv3)
12
13
  Classifier: Topic :: Text Processing :: Markup :: XML
13
14
  Project-URL: Repository, https://github.com/p7e4/rssfeed
14
15
  Project-URL: Issues, https://github.com/p7e4/rssfeed/issues
15
16
  Requires-Python: >=3.12
16
- Requires-Dist: python-dateutil>=2.9
17
17
  Description-Content-Type: text/markdown
18
18
 
19
19
  # rssfeed
@@ -11,9 +11,6 @@ authors = [
11
11
  { name = "p7e4" },
12
12
  ]
13
13
  requires-python = ">= 3.12"
14
- dependencies = [
15
- "python-dateutil >= 2.9",
16
- ]
17
14
  keywords = [
18
15
  "rssfeed",
19
16
  "rss",
@@ -25,10 +22,11 @@ classifiers = [
25
22
  "Programming Language :: Python :: 3",
26
23
  "Programming Language :: Python :: 3.12",
27
24
  "Programming Language :: Python :: 3.13",
25
+ "Programming Language :: Python :: 3.14",
28
26
  "License :: OSI Approved :: GNU General Public License v3 (GPLv3)",
29
27
  "Topic :: Text Processing :: Markup :: XML",
30
28
  ]
31
- version = "0.4.2"
29
+ version = "0.4.4"
32
30
 
33
31
  [project.license]
34
32
  text = "GPLv3"
@@ -1,7 +1,8 @@
1
- from dateutil.parser import parse as timeParse
1
+ from email.utils import parsedate_to_datetime
2
2
  from xml.etree import ElementTree
3
+ from datetime import datetime
3
4
 
4
- __version__ = "0.4.2"
5
+ __version__ = "0.4.4"
5
6
 
6
7
  class ParseError(Exception):
7
8
  pass
@@ -17,14 +18,26 @@ def _parse(data):
17
18
  raise ParseError("xml parse fail") from e
18
19
  return parser
19
20
 
21
+ def timeParse(s):
22
+ if not s: return 0
23
+ try:
24
+ if s.isdigit():
25
+ return int(s)
26
+ if len(s) > 4 and s[4] == "-":
27
+ t = datetime.fromisoformat(s.replace("Z", "+00:00"))
28
+ else:
29
+ t = parsedate_to_datetime(s)
30
+ return int(t.timestamp())
31
+ except (TypeError, ValueError):
32
+ return 0
33
+
20
34
  def parse(data, url=None):
21
- if url: url = url[:8] + url[8:].rsplit("/")[0]
35
+ if url: url = url[:8] + url[8:].split("/")[0]
22
36
  items = list()
23
37
  for event, elem in _parse(data).read_events():
24
- tag = elem.tag.split("}", 1)[1] if elem.tag.startswith("{") else elem.tag
25
- text = elem.text.strip() if elem.text else str()
38
+ tag = elem.tag.rsplit("}", 1)[-1]
26
39
  if event == "start":
27
- if tag in ("channel", "RDF", "feed", "item", "entry"):
40
+ if tag in ("channel", "feed", "item", "entry"):
28
41
  items.append({
29
42
  "title": str(),
30
43
  "author": str(),
@@ -33,31 +46,36 @@ def parse(data, url=None):
33
46
  "content": str()
34
47
  })
35
48
  else:
49
+ if not (elem.text and (text:=elem.text.strip())) and tag != "link":
50
+ continue
36
51
  i = items[-1]
37
52
  match tag:
38
- case "description" | "encoded" | "summary" | "content":
53
+ case "content" | "encoded":
39
54
  i["content"] = text
40
- case "pubDate" | "updated" | "published" | "lastBuildDate":
41
- if text.isdigit():
42
- i["timestamp"] = int(text)
43
- elif text:
44
- try:
45
- i["timestamp"] = int(timeParse(text).timestamp())
46
- except Exception as e:
47
- raise ParseError("time parse fail") from e
55
+ case "summary" | "description":
56
+ if not i["content"]: i["content"] = text
57
+ case "pubDate" | "published" | "date":
58
+ try:
59
+ if not i["timestamp"]: i["timestamp"] = timeParse(text)
60
+ except Exception as e:
61
+ raise ParseError("time parse fail") from e
48
62
  case "link":
49
- i["url"] = text or elem.get("href")
50
- if url and not i["url"].startswith("http"):
63
+ if not i["url"]: i["url"] = elem.get("href") or text
64
+ if not i["url"].startswith(("http://", "https://")) and url:
51
65
  i["url"] = f"{url}/{i["url"].lstrip("/")}"
52
- case "title" | "author":
53
- i[tag] = text
66
+ case "author" | "name" | "creator":
67
+ if not i["author"]:
68
+ i["author"] = text
69
+ case "title":
70
+ if not i["title"]:
71
+ i[title] = text
54
72
 
55
73
  if not items:
56
74
  raise ParseError("not valid result")
57
75
 
58
76
  feed = {
59
77
  "name": items[0]["title"],
60
- "lastupdate": items[0]["timestamp"],
78
+ "lastupdate": max(i["timestamp"] for i in items[1:]),
61
79
  "items": items[1:]
62
80
  }
63
81
 
File without changes
File without changes
File without changes