rssfeed 0.2__tar.gz → 0.3__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.1
2
2
  Name: rssfeed
3
- Version: 0.2
3
+ Version: 0.3
4
4
  Summary: A simple rss/atom feed parser
5
5
  Keywords: rssfeed,rss,feed,atom,feedparser
6
6
  Author: p7e4
@@ -27,7 +27,7 @@ classifiers = [
27
27
  "License :: OSI Approved :: GNU General Public License v3 (GPLv3)",
28
28
  "Topic :: Text Processing :: Markup :: XML",
29
29
  ]
30
- version = "0.2"
30
+ version = "0.3"
31
31
 
32
32
  [project.license]
33
33
  text = "GPLv3"
@@ -1,14 +1,15 @@
1
1
  from dateutil.parser import parse as timeParse
2
2
  from xml.etree import ElementTree
3
3
 
4
- __version__ = "0.2"
4
+ __version__ = "0.3"
5
5
 
6
6
  def parse(data):
7
+ assert type(data) == str, "data argument must be a string"
7
8
  if not data or not (data:=data.lstrip()):
8
9
  return
9
10
  if not any((data.startswith(i) for i in ("<?xml ", "<rss ", "<feed "))):
10
11
  return
11
- parser = ElementTree.XMLPullParser(("start", "end"), _parser=ElementTree.XMLParser(encoding='utf-8'))
12
+ parser = ElementTree.XMLPullParser(("start", "end"))
12
13
  try:
13
14
  parser.feed(data)
14
15
  parser.close()
@@ -16,7 +17,7 @@ def parse(data):
16
17
  return
17
18
 
18
19
  items = list()
19
- authorTag = False
20
+ path = list()
20
21
  for event, elem in parser.read_events():
21
22
  tag = elem.tag.split("}", 1)[1] if elem.tag.startswith("{") else elem.tag
22
23
  text = elem.text.strip() if elem.text else str()
@@ -29,12 +30,9 @@ def parse(data):
29
30
  "url": str(),
30
31
  "content": str()
31
32
  })
32
- elif tag == "author":
33
- authorTag = True
33
+ path.append(tag)
34
34
  else:
35
35
  match tag:
36
- case "guid":
37
- tag = "id"
38
36
  case "summary" | "description" | "encoded":
39
37
  tag = "content"
40
38
  case "updated" | "pubDate" | "published" | "lastBuildDate":
@@ -47,35 +45,30 @@ def parse(data):
47
45
  pass
48
46
  continue
49
47
  case "link":
50
- items[-1]["url"] = elem.get("href") if text and elem.get("href") else text
48
+ items[-1]["url"] = text or elem.get("href")
51
49
  continue
52
50
  case "author":
53
51
  authorTag = False
54
52
  continue
55
- case "name" if authorTag:
53
+ case "name" if path[-2] == "author":
56
54
  tag = "author"
57
- case "title" | "id" | "content":
55
+ case "title" | "content":
58
56
  pass
59
57
  case _:
60
58
  continue
61
59
 
62
60
  items[-1][tag] = text
61
+ path.pop()
63
62
 
64
63
  if not items: return
65
- feed = items.pop(0)
66
64
  feed = {
67
- "name": feed["title"],
68
- "lastupdate": feed["timestamp"],
69
- "items": items
65
+ "name": items[0]["title"],
66
+ "lastupdate": items[0]["timestamp"],
67
+ "items": items[1:]
70
68
  }
71
- for item in items:
72
- if item.get("id"):
73
- if not item["url"] and item["id"].startswith("http"):
74
- item["url"] = item["id"]
75
- del item["id"]
76
-
77
- if feed["lastupdate"] < item["timestamp"]:
78
- feed["lastupdate"] = item["timestamp"]
69
+ # for item in items:
70
+ # if feed["lastupdate"] < item["timestamp"]:
71
+ # feed["lastupdate"] = item["timestamp"]
79
72
 
80
73
  return feed
81
74
 
File without changes
File without changes
File without changes