rssfeed 0.1__tar.gz → 0.3__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
rssfeed-0.3/PKG-INFO ADDED
@@ -0,0 +1,73 @@
1
+ Metadata-Version: 2.1
2
+ Name: rssfeed
3
+ Version: 0.3
4
+ Summary: A simple rss/atom feed parser
5
+ Keywords: rssfeed,rss,feed,atom,feedparser
6
+ Author: p7e4
7
+ License: GPLv3
8
+ Classifier: Programming Language :: Python :: 3
9
+ Classifier: Programming Language :: Python :: 3.12
10
+ Classifier: License :: OSI Approved :: GNU General Public License v3 (GPLv3)
11
+ Classifier: Topic :: Text Processing :: Markup :: XML
12
+ Project-URL: Repository, https://github.com/p7e4/rssfeed
13
+ Project-URL: Issues, https://github.com/p7e4/rssfeed/issues
14
+ Requires-Python: >=3.12
15
+ Requires-Dist: python-dateutil>=2.9
16
+ Description-Content-Type: text/markdown
17
+
18
+ # rssfeed
19
+
20
+ A simple rss/atom feed parser
21
+
22
+ ## Installation
23
+
24
+ `pip install rssfeed`
25
+
26
+ ## Get Started
27
+
28
+ ``` python
29
+ import requests
30
+ import rssfeed
31
+
32
+ feed = rssfeed.parse(requests.get("https://www.solidot.org/index.rss").text)
33
+ print(feed)
34
+ ```
35
+ ```
36
+ {
37
+ "name": "奇客Solidot–传递最新科技情报",
38
+ "lastupdate": 1717423475,
39
+ "items": [
40
+ {
41
+ "title": "中国科学家使用细胞疗法治愈一名患者的糖尿病",
42
+ "author": "",
43
+ "timestamp": 1717410594,
44
+ "url": "https://www.solidot.org/story?sid=78338",
45
+ "content": "《南华早报》报道,中国科学家利用细胞疗法成功治愈了一名患者的糖尿病。研究报告发表在《Cell Discovery》期刊 ..."
46
+ },
47
+ {
48
+ "title": "Steam 平台 Linux 玩家四分之三使用 AMD CPU",
49
+ "author": "",
50
+ "timestamp": 1717404736,
51
+ "url": "https://www.solidot.org/story?sid=78337",
52
+ "content": "根据 Valve 公布的 Steam 硬件和软件调查,Linux 份额在过去的五月增长了 0.42% 至 2.32%,macOS 增至 1.47% ..."
53
+ },
54
+ {
55
+ "title": "Hugging Face 称黑客窃取了 Spaces 平台的身份验证令牌",
56
+ "author": "",
57
+ "timestamp": 1717400574,
58
+ "url": "https://www.solidot.org/story?sid=78336",
59
+ "content": "Hugging Face 官方博客披露黑客窃取了其 Spaces 平台的身份验证令牌。Spaces 是社区用户创建和递交 AI 应用的库 ..."
60
+ }
61
+ ...
62
+ ]
63
+ }
64
+ ```
65
+
66
+ ## Warning
67
+
68
+ rssfeed **does not** escape any HTML tags, which mean if you does not check the content and display it somewhere html can be rendered, it may lead to [Cross-site scripting](https://developer.mozilla.org/en-US/docs/Glossary/Cross-site_scripting) attacks.
69
+
70
+ ## Changelog
71
+
72
+ [Changelog.md](/Changelog.md)
73
+
rssfeed-0.3/README.md ADDED
@@ -0,0 +1,56 @@
1
+ # rssfeed
2
+
3
+ A simple rss/atom feed parser
4
+
5
+ ## Installation
6
+
7
+ `pip install rssfeed`
8
+
9
+ ## Get Started
10
+
11
+ ``` python
12
+ import requests
13
+ import rssfeed
14
+
15
+ feed = rssfeed.parse(requests.get("https://www.solidot.org/index.rss").text)
16
+ print(feed)
17
+ ```
18
+ ```
19
+ {
20
+ "name": "奇客Solidot–传递最新科技情报",
21
+ "lastupdate": 1717423475,
22
+ "items": [
23
+ {
24
+ "title": "中国科学家使用细胞疗法治愈一名患者的糖尿病",
25
+ "author": "",
26
+ "timestamp": 1717410594,
27
+ "url": "https://www.solidot.org/story?sid=78338",
28
+ "content": "《南华早报》报道,中国科学家利用细胞疗法成功治愈了一名患者的糖尿病。研究报告发表在《Cell Discovery》期刊 ..."
29
+ },
30
+ {
31
+ "title": "Steam 平台 Linux 玩家四分之三使用 AMD CPU",
32
+ "author": "",
33
+ "timestamp": 1717404736,
34
+ "url": "https://www.solidot.org/story?sid=78337",
35
+ "content": "根据 Valve 公布的 Steam 硬件和软件调查,Linux 份额在过去的五月增长了 0.42% 至 2.32%,macOS 增至 1.47% ..."
36
+ },
37
+ {
38
+ "title": "Hugging Face 称黑客窃取了 Spaces 平台的身份验证令牌",
39
+ "author": "",
40
+ "timestamp": 1717400574,
41
+ "url": "https://www.solidot.org/story?sid=78336",
42
+ "content": "Hugging Face 官方博客披露黑客窃取了其 Spaces 平台的身份验证令牌。Spaces 是社区用户创建和递交 AI 应用的库 ..."
43
+ }
44
+ ...
45
+ ]
46
+ }
47
+ ```
48
+
49
+ ## Warning
50
+
51
+ rssfeed **does not** escape any HTML tags, which mean if you does not check the content and display it somewhere html can be rendered, it may lead to [Cross-site scripting](https://developer.mozilla.org/en-US/docs/Glossary/Cross-site_scripting) attacks.
52
+
53
+ ## Changelog
54
+
55
+ [Changelog.md](/Changelog.md)
56
+
@@ -1,7 +1,5 @@
1
1
  [build-system]
2
- requires = [
3
- "pdm-backend",
4
- ]
2
+ requires = []
5
3
  build-backend = "pdm.backend"
6
4
 
7
5
  [project]
@@ -21,8 +19,7 @@ keywords = [
21
19
  "rss",
22
20
  "feed",
23
21
  "atom",
24
- "rssparse",
25
- "feedparse",
22
+ "feedparser",
26
23
  ]
27
24
  classifiers = [
28
25
  "Programming Language :: Python :: 3",
@@ -30,7 +27,7 @@ classifiers = [
30
27
  "License :: OSI Approved :: GNU General Public License v3 (GPLv3)",
31
28
  "Topic :: Text Processing :: Markup :: XML",
32
29
  ]
33
- version = "0.1"
30
+ version = "0.3"
34
31
 
35
32
  [project.license]
36
33
  text = "GPLv3"
@@ -38,7 +35,6 @@ text = "GPLv3"
38
35
  [project.urls]
39
36
  Repository = "https://github.com/p7e4/rssfeed"
40
37
  Issues = "https://github.com/p7e4/rssfeed/issues"
41
- Changelog = "https://github.com/p7e4/rssfeed/blob/main/changelog.md"
42
38
 
43
39
  [tool.pdm.version]
44
40
  source = "file"
@@ -1,14 +1,15 @@
1
1
  from dateutil.parser import parse as timeParse
2
2
  from xml.etree import ElementTree
3
3
 
4
- __version__ = "0.1"
4
+ __version__ = "0.3"
5
5
 
6
6
  def parse(data):
7
+ assert type(data) == str, "data argument must be a string"
7
8
  if not data or not (data:=data.lstrip()):
8
9
  return
9
10
  if not any((data.startswith(i) for i in ("<?xml ", "<rss ", "<feed "))):
10
11
  return
11
- parser = ElementTree.XMLPullParser(("start", "end"), _parser=ElementTree.XMLParser(encoding='utf-8'))
12
+ parser = ElementTree.XMLPullParser(("start", "end"))
12
13
  try:
13
14
  parser.feed(data)
14
15
  parser.close()
@@ -16,7 +17,7 @@ def parse(data):
16
17
  return
17
18
 
18
19
  items = list()
19
- authorTag = False
20
+ path = list()
20
21
  for event, elem in parser.read_events():
21
22
  tag = elem.tag.split("}", 1)[1] if elem.tag.startswith("{") else elem.tag
22
23
  text = elem.text.strip() if elem.text else str()
@@ -29,49 +30,45 @@ def parse(data):
29
30
  "url": str(),
30
31
  "content": str()
31
32
  })
32
- elif tag == "author":
33
- authorTag = True
33
+ path.append(tag)
34
34
  else:
35
35
  match tag:
36
- case "guid":
37
- tag = "id"
38
36
  case "summary" | "description" | "encoded":
39
37
  tag = "content"
40
38
  case "updated" | "pubDate" | "published" | "lastBuildDate":
41
- items[-1]["timestamp"] = int(text) if text.isdigit() else int(timeParse(text).timestamp())
39
+ if text.isdigit():
40
+ items[-1]["timestamp"] = int(text)
41
+ elif text:
42
+ try:
43
+ items[-1]["timestamp"] = int(timeParse(text).timestamp())
44
+ except:
45
+ pass
42
46
  continue
43
47
  case "link":
44
- items[-1]["url"] = elem.get("href") if text and elem.get("href") else text
48
+ items[-1]["url"] = text or elem.get("href")
45
49
  continue
46
50
  case "author":
47
51
  authorTag = False
48
52
  continue
49
- case "name" if authorTag:
53
+ case "name" if path[-2] == "author":
50
54
  tag = "author"
51
- case "title" | "id" | "content":
55
+ case "title" | "content":
52
56
  pass
53
57
  case _:
54
58
  continue
55
59
 
56
60
  items[-1][tag] = text
61
+ path.pop()
57
62
 
58
63
  if not items: return
59
- feed = items.pop(0)
60
64
  feed = {
61
- "name": feed["title"],
62
- "lastupdate": feed["timestamp"],
63
- "items": items
65
+ "name": items[0]["title"],
66
+ "lastupdate": items[0]["timestamp"],
67
+ "items": items[1:]
64
68
  }
65
- for item in items:
66
- if item.get("id"):
67
- if not item["url"] and item["id"].startswith("http"):
68
- item["url"] = item["id"]
69
- # if not item["url"].startswith("http"):
70
- # item["url"] = f"https://{item["url"]}"
71
- del item["id"]
72
-
73
- if feed["lastupdate"] < item["timestamp"]:
74
- feed["lastupdate"] = item["timestamp"]
69
+ # for item in items:
70
+ # if feed["lastupdate"] < item["timestamp"]:
71
+ # feed["lastupdate"] = item["timestamp"]
75
72
 
76
73
  return feed
77
74
 
rssfeed-0.1/PKG-INFO DELETED
@@ -1,27 +0,0 @@
1
- Metadata-Version: 2.1
2
- Name: rssfeed
3
- Version: 0.1
4
- Summary: A simple rss/atom feed parser
5
- Keywords: rssfeed,rss,feed,atom,rssparse,feedparse
6
- Author: p7e4
7
- License: GPLv3
8
- Classifier: Programming Language :: Python :: 3
9
- Classifier: Programming Language :: Python :: 3.12
10
- Classifier: License :: OSI Approved :: GNU General Public License v3 (GPLv3)
11
- Classifier: Topic :: Text Processing :: Markup :: XML
12
- Project-URL: Repository, https://github.com/p7e4/rssfeed
13
- Project-URL: Issues, https://github.com/p7e4/rssfeed/issues
14
- Project-URL: Changelog, https://github.com/p7e4/rssfeed/blob/main/changelog.md
15
- Requires-Python: >=3.12
16
- Requires-Dist: python-dateutil>=2.9
17
- Description-Content-Type: text/markdown
18
-
19
- # rssfeed
20
-
21
- A simple rss/atom feed parser
22
-
23
- ## Installation
24
-
25
- `$ pip install --upgrade rssfeed`
26
-
27
-
rssfeed-0.1/README.md DELETED
@@ -1,9 +0,0 @@
1
- # rssfeed
2
-
3
- A simple rss/atom feed parser
4
-
5
- ## Installation
6
-
7
- `$ pip install --upgrade rssfeed`
8
-
9
-
File without changes
File without changes