rssfeed 0.2__tar.gz → 0.4.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {rssfeed-0.2 → rssfeed-0.4.0}/PKG-INFO +1 -1
- {rssfeed-0.2 → rssfeed-0.4.0}/pyproject.toml +1 -1
- rssfeed-0.4.0/rssfeed/lib.py +62 -0
- rssfeed-0.2/rssfeed/lib.py +0 -81
- {rssfeed-0.2 → rssfeed-0.4.0}/LICENSE +0 -0
- {rssfeed-0.2 → rssfeed-0.4.0}/README.md +0 -0
- {rssfeed-0.2 → rssfeed-0.4.0}/rssfeed/__init__.py +0 -0
|
@@ -0,0 +1,62 @@
|
|
|
1
|
+
from dateutil.parser import parse as timeParse
|
|
2
|
+
from xml.etree import ElementTree
|
|
3
|
+
|
|
4
|
+
__version__ = "0.4.0"
|
|
5
|
+
|
|
6
|
+
class ParseError(Exception):
|
|
7
|
+
pass
|
|
8
|
+
|
|
9
|
+
def parse(data):
|
|
10
|
+
parser = ElementTree.XMLPullParser(("start", "end"))
|
|
11
|
+
try:
|
|
12
|
+
parser.feed(data)
|
|
13
|
+
parser.close()
|
|
14
|
+
except ElementTree.ParseError as e:
|
|
15
|
+
raise ParseError("xml parse fail") from e
|
|
16
|
+
|
|
17
|
+
items = list()
|
|
18
|
+
lastTag = str()
|
|
19
|
+
for event, elem in parser.read_events():
|
|
20
|
+
tag = elem.tag.split("}", 1)[1] if elem.tag.startswith("{") else elem.tag
|
|
21
|
+
text = elem.text.strip() if elem.text else str()
|
|
22
|
+
if event == "start":
|
|
23
|
+
if tag in ("channel", "RDF", "feed", "item", "entry"):
|
|
24
|
+
items.append({
|
|
25
|
+
"title": str(),
|
|
26
|
+
"author": str(),
|
|
27
|
+
"timestamp": 0,
|
|
28
|
+
"url": str(),
|
|
29
|
+
"content": str()
|
|
30
|
+
})
|
|
31
|
+
else:
|
|
32
|
+
i = items[-1]
|
|
33
|
+
match tag:
|
|
34
|
+
case "content" | "description" | "encoded" | "summary":
|
|
35
|
+
i["content"] = text
|
|
36
|
+
case "updated" | "pubDate" | "published" | "lastBuildDate":
|
|
37
|
+
if text.isdigit():
|
|
38
|
+
i["timestamp"] = int(text)
|
|
39
|
+
elif text:
|
|
40
|
+
try:
|
|
41
|
+
i["timestamp"] = int(timeParse(text).timestamp())
|
|
42
|
+
except Exception as e:
|
|
43
|
+
raise ParseError("time parse fail") from e
|
|
44
|
+
case "link":
|
|
45
|
+
i["url"] = text or elem.get("href")
|
|
46
|
+
case "name" if lastTag == "author":
|
|
47
|
+
i["author"] = text
|
|
48
|
+
case "title":
|
|
49
|
+
i[tag] = text
|
|
50
|
+
|
|
51
|
+
lastTag = tag
|
|
52
|
+
|
|
53
|
+
if not items:
|
|
54
|
+
raise ParseError("not valid result")
|
|
55
|
+
|
|
56
|
+
feed = {
|
|
57
|
+
"name": items[0]["title"],
|
|
58
|
+
"lastupdate": items[0]["timestamp"],
|
|
59
|
+
"items": items[1:]
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
return feed
|
rssfeed-0.2/rssfeed/lib.py
DELETED
|
@@ -1,81 +0,0 @@
|
|
|
1
|
-
from dateutil.parser import parse as timeParse
|
|
2
|
-
from xml.etree import ElementTree
|
|
3
|
-
|
|
4
|
-
__version__ = "0.2"
|
|
5
|
-
|
|
6
|
-
def parse(data):
|
|
7
|
-
if not data or not (data:=data.lstrip()):
|
|
8
|
-
return
|
|
9
|
-
if not any((data.startswith(i) for i in ("<?xml ", "<rss ", "<feed "))):
|
|
10
|
-
return
|
|
11
|
-
parser = ElementTree.XMLPullParser(("start", "end"), _parser=ElementTree.XMLParser(encoding='utf-8'))
|
|
12
|
-
try:
|
|
13
|
-
parser.feed(data)
|
|
14
|
-
parser.close()
|
|
15
|
-
except ElementTree.ParseError:
|
|
16
|
-
return
|
|
17
|
-
|
|
18
|
-
items = list()
|
|
19
|
-
authorTag = False
|
|
20
|
-
for event, elem in parser.read_events():
|
|
21
|
-
tag = elem.tag.split("}", 1)[1] if elem.tag.startswith("{") else elem.tag
|
|
22
|
-
text = elem.text.strip() if elem.text else str()
|
|
23
|
-
if event == "start":
|
|
24
|
-
if tag in ("channel", "RDF", "feed", "item", "entry"):
|
|
25
|
-
items.append({
|
|
26
|
-
"title": str(),
|
|
27
|
-
"author": str(),
|
|
28
|
-
"timestamp": 0,
|
|
29
|
-
"url": str(),
|
|
30
|
-
"content": str()
|
|
31
|
-
})
|
|
32
|
-
elif tag == "author":
|
|
33
|
-
authorTag = True
|
|
34
|
-
else:
|
|
35
|
-
match tag:
|
|
36
|
-
case "guid":
|
|
37
|
-
tag = "id"
|
|
38
|
-
case "summary" | "description" | "encoded":
|
|
39
|
-
tag = "content"
|
|
40
|
-
case "updated" | "pubDate" | "published" | "lastBuildDate":
|
|
41
|
-
if text.isdigit():
|
|
42
|
-
items[-1]["timestamp"] = int(text)
|
|
43
|
-
elif text:
|
|
44
|
-
try:
|
|
45
|
-
items[-1]["timestamp"] = int(timeParse(text).timestamp())
|
|
46
|
-
except:
|
|
47
|
-
pass
|
|
48
|
-
continue
|
|
49
|
-
case "link":
|
|
50
|
-
items[-1]["url"] = elem.get("href") if text and elem.get("href") else text
|
|
51
|
-
continue
|
|
52
|
-
case "author":
|
|
53
|
-
authorTag = False
|
|
54
|
-
continue
|
|
55
|
-
case "name" if authorTag:
|
|
56
|
-
tag = "author"
|
|
57
|
-
case "title" | "id" | "content":
|
|
58
|
-
pass
|
|
59
|
-
case _:
|
|
60
|
-
continue
|
|
61
|
-
|
|
62
|
-
items[-1][tag] = text
|
|
63
|
-
|
|
64
|
-
if not items: return
|
|
65
|
-
feed = items.pop(0)
|
|
66
|
-
feed = {
|
|
67
|
-
"name": feed["title"],
|
|
68
|
-
"lastupdate": feed["timestamp"],
|
|
69
|
-
"items": items
|
|
70
|
-
}
|
|
71
|
-
for item in items:
|
|
72
|
-
if item.get("id"):
|
|
73
|
-
if not item["url"] and item["id"].startswith("http"):
|
|
74
|
-
item["url"] = item["id"]
|
|
75
|
-
del item["id"]
|
|
76
|
-
|
|
77
|
-
if feed["lastupdate"] < item["timestamp"]:
|
|
78
|
-
feed["lastupdate"] = item["timestamp"]
|
|
79
|
-
|
|
80
|
-
return feed
|
|
81
|
-
|
|
File without changes
|
|
File without changes
|
|
File without changes
|