rssfeed 0.2__tar.gz → 0.4.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.1
2
2
  Name: rssfeed
3
- Version: 0.2
3
+ Version: 0.4.0
4
4
  Summary: A simple rss/atom feed parser
5
5
  Keywords: rssfeed,rss,feed,atom,feedparser
6
6
  Author: p7e4
@@ -27,7 +27,7 @@ classifiers = [
27
27
  "License :: OSI Approved :: GNU General Public License v3 (GPLv3)",
28
28
  "Topic :: Text Processing :: Markup :: XML",
29
29
  ]
30
- version = "0.2"
30
+ version = "0.4.0"
31
31
 
32
32
  [project.license]
33
33
  text = "GPLv3"
@@ -0,0 +1,62 @@
1
+ from dateutil.parser import parse as timeParse
2
+ from xml.etree import ElementTree
3
+
4
+ __version__ = "0.4.0"
5
+
6
+ class ParseError(Exception):
7
+ pass
8
+
9
+ def parse(data):
10
+ parser = ElementTree.XMLPullParser(("start", "end"))
11
+ try:
12
+ parser.feed(data)
13
+ parser.close()
14
+ except ElementTree.ParseError as e:
15
+ raise ParseError("xml parse fail") from e
16
+
17
+ items = list()
18
+ lastTag = str()
19
+ for event, elem in parser.read_events():
20
+ tag = elem.tag.split("}", 1)[1] if elem.tag.startswith("{") else elem.tag
21
+ text = elem.text.strip() if elem.text else str()
22
+ if event == "start":
23
+ if tag in ("channel", "RDF", "feed", "item", "entry"):
24
+ items.append({
25
+ "title": str(),
26
+ "author": str(),
27
+ "timestamp": 0,
28
+ "url": str(),
29
+ "content": str()
30
+ })
31
+ else:
32
+ i = items[-1]
33
+ match tag:
34
+ case "content" | "description" | "encoded" | "summary":
35
+ i["content"] = text
36
+ case "updated" | "pubDate" | "published" | "lastBuildDate":
37
+ if text.isdigit():
38
+ i["timestamp"] = int(text)
39
+ elif text:
40
+ try:
41
+ i["timestamp"] = int(timeParse(text).timestamp())
42
+ except Exception as e:
43
+ raise ParseError("time parse fail") from e
44
+ case "link":
45
+ i["url"] = text or elem.get("href")
46
+ case "name" if lastTag == "author":
47
+ i["author"] = text
48
+ case "title":
49
+ i[tag] = text
50
+
51
+ lastTag = tag
52
+
53
+ if not items:
54
+ raise ParseError("not valid result")
55
+
56
+ feed = {
57
+ "name": items[0]["title"],
58
+ "lastupdate": items[0]["timestamp"],
59
+ "items": items[1:]
60
+ }
61
+
62
+ return feed
@@ -1,81 +0,0 @@
1
- from dateutil.parser import parse as timeParse
2
- from xml.etree import ElementTree
3
-
4
- __version__ = "0.2"
5
-
6
- def parse(data):
7
- if not data or not (data:=data.lstrip()):
8
- return
9
- if not any((data.startswith(i) for i in ("<?xml ", "<rss ", "<feed "))):
10
- return
11
- parser = ElementTree.XMLPullParser(("start", "end"), _parser=ElementTree.XMLParser(encoding='utf-8'))
12
- try:
13
- parser.feed(data)
14
- parser.close()
15
- except ElementTree.ParseError:
16
- return
17
-
18
- items = list()
19
- authorTag = False
20
- for event, elem in parser.read_events():
21
- tag = elem.tag.split("}", 1)[1] if elem.tag.startswith("{") else elem.tag
22
- text = elem.text.strip() if elem.text else str()
23
- if event == "start":
24
- if tag in ("channel", "RDF", "feed", "item", "entry"):
25
- items.append({
26
- "title": str(),
27
- "author": str(),
28
- "timestamp": 0,
29
- "url": str(),
30
- "content": str()
31
- })
32
- elif tag == "author":
33
- authorTag = True
34
- else:
35
- match tag:
36
- case "guid":
37
- tag = "id"
38
- case "summary" | "description" | "encoded":
39
- tag = "content"
40
- case "updated" | "pubDate" | "published" | "lastBuildDate":
41
- if text.isdigit():
42
- items[-1]["timestamp"] = int(text)
43
- elif text:
44
- try:
45
- items[-1]["timestamp"] = int(timeParse(text).timestamp())
46
- except:
47
- pass
48
- continue
49
- case "link":
50
- items[-1]["url"] = elem.get("href") if text and elem.get("href") else text
51
- continue
52
- case "author":
53
- authorTag = False
54
- continue
55
- case "name" if authorTag:
56
- tag = "author"
57
- case "title" | "id" | "content":
58
- pass
59
- case _:
60
- continue
61
-
62
- items[-1][tag] = text
63
-
64
- if not items: return
65
- feed = items.pop(0)
66
- feed = {
67
- "name": feed["title"],
68
- "lastupdate": feed["timestamp"],
69
- "items": items
70
- }
71
- for item in items:
72
- if item.get("id"):
73
- if not item["url"] and item["id"].startswith("http"):
74
- item["url"] = item["id"]
75
- del item["id"]
76
-
77
- if feed["lastupdate"] < item["timestamp"]:
78
- feed["lastupdate"] = item["timestamp"]
79
-
80
- return feed
81
-
File without changes
File without changes
File without changes