fastfeedparser 0.5.5__tar.gz → 0.5.6__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {fastfeedparser-0.5.5/src/fastfeedparser.egg-info → fastfeedparser-0.5.6}/PKG-INFO +3 -2
- {fastfeedparser-0.5.5 → fastfeedparser-0.5.6}/README.md +1 -1
- {fastfeedparser-0.5.5 → fastfeedparser-0.5.6}/setup.cfg +4 -1
- {fastfeedparser-0.5.5 → fastfeedparser-0.5.6}/src/fastfeedparser/main.py +515 -255
- {fastfeedparser-0.5.5 → fastfeedparser-0.5.6/src/fastfeedparser.egg-info}/PKG-INFO +3 -2
- {fastfeedparser-0.5.5 → fastfeedparser-0.5.6}/src/fastfeedparser.egg-info/requires.txt +4 -0
- {fastfeedparser-0.5.5 → fastfeedparser-0.5.6}/tests/test_encoding.py +46 -0
- {fastfeedparser-0.5.5 → fastfeedparser-0.5.6}/LICENSE +0 -0
- {fastfeedparser-0.5.5 → fastfeedparser-0.5.6}/pyproject.toml +0 -0
- {fastfeedparser-0.5.5 → fastfeedparser-0.5.6}/src/fastfeedparser/__init__.py +0 -0
- {fastfeedparser-0.5.5 → fastfeedparser-0.5.6}/src/fastfeedparser.egg-info/SOURCES.txt +0 -0
- {fastfeedparser-0.5.5 → fastfeedparser-0.5.6}/src/fastfeedparser.egg-info/dependency_links.txt +0 -0
- {fastfeedparser-0.5.5 → fastfeedparser-0.5.6}/src/fastfeedparser.egg-info/top_level.txt +0 -0
- {fastfeedparser-0.5.5 → fastfeedparser-0.5.6}/tests/test_integration.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.1
|
|
2
2
|
Name: fastfeedparser
|
|
3
|
-
Version: 0.5.
|
|
3
|
+
Version: 0.5.6
|
|
4
4
|
Summary: High performance RSS, Atom, JSON and RDF feed parser in Python
|
|
5
5
|
Home-page: https://github.com/kagisearch/fastfeedparser
|
|
6
6
|
Author: Vladimir Prelovac
|
|
@@ -25,6 +25,7 @@ Requires-Python: >=3.7
|
|
|
25
25
|
Description-Content-Type: text/markdown
|
|
26
26
|
Provides-Extra: dateparser
|
|
27
27
|
Provides-Extra: brotli
|
|
28
|
+
Provides-Extra: orjson
|
|
28
29
|
Provides-Extra: full
|
|
29
30
|
License-File: LICENSE
|
|
30
31
|
|
|
@@ -146,7 +147,7 @@ Speedup: 50.1x
|
|
|
146
147
|
|
|
147
148
|
### Main Functions
|
|
148
149
|
|
|
149
|
-
- `parse(source)`: Parse feed from a source
|
|
150
|
+
- `parse(source, *, include_content=True, include_tags=True, include_media=True, include_enclosures=True)`: Parse feed from a URL/XML/JSON source, with optional field extraction toggles for faster parsing.
|
|
150
151
|
|
|
151
152
|
|
|
152
153
|
### Feed Object Structure
|
|
@@ -116,7 +116,7 @@ Speedup: 50.1x
|
|
|
116
116
|
|
|
117
117
|
### Main Functions
|
|
118
118
|
|
|
119
|
-
- `parse(source)`: Parse feed from a source
|
|
119
|
+
- `parse(source, *, include_content=True, include_tags=True, include_media=True, include_enclosures=True)`: Parse feed from a URL/XML/JSON source, with optional field extraction toggles for faster parsing.
|
|
120
120
|
|
|
121
121
|
|
|
122
122
|
### Feed Object Structure
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
[metadata]
|
|
2
2
|
name = fastfeedparser
|
|
3
|
-
version = 0.5.
|
|
3
|
+
version = 0.5.6
|
|
4
4
|
author = Vladimir Prelovac
|
|
5
5
|
author_email = vlad@kagi.com
|
|
6
6
|
description = High performance RSS, Atom, JSON and RDF feed parser in Python
|
|
@@ -40,9 +40,12 @@ dateparser =
|
|
|
40
40
|
dateparser
|
|
41
41
|
brotli =
|
|
42
42
|
brotli
|
|
43
|
+
orjson =
|
|
44
|
+
orjson
|
|
43
45
|
full =
|
|
44
46
|
dateparser
|
|
45
47
|
brotli
|
|
48
|
+
orjson
|
|
46
49
|
|
|
47
50
|
[options.packages.find]
|
|
48
51
|
where = src
|
|
@@ -15,7 +15,14 @@ try:
|
|
|
15
15
|
HAS_BROTLI = True
|
|
16
16
|
except ImportError:
|
|
17
17
|
HAS_BROTLI = False
|
|
18
|
-
|
|
18
|
+
|
|
19
|
+
try:
|
|
20
|
+
import orjson
|
|
21
|
+
|
|
22
|
+
_json_loads = orjson.loads
|
|
23
|
+
except ImportError:
|
|
24
|
+
_json_loads = json.loads
|
|
25
|
+
from typing import Any, Callable, Optional, Protocol, TYPE_CHECKING, Literal
|
|
19
26
|
from urllib.parse import urljoin
|
|
20
27
|
from urllib.request import (
|
|
21
28
|
HTTPErrorProcessor,
|
|
@@ -81,6 +88,44 @@ _MONTHS_RFC822: dict[str, int] = {
|
|
|
81
88
|
"dec": 12,
|
|
82
89
|
}
|
|
83
90
|
|
|
91
|
+
_XML_NS = "{http://www.w3.org/XML/1998/namespace}"
|
|
92
|
+
_XML_LANG_ATTR = _XML_NS + "lang"
|
|
93
|
+
_XML_BASE_ATTR = _XML_NS + "base"
|
|
94
|
+
_RDF_ABOUT_ATTR = "{http://www.w3.org/1999/02/22-rdf-syntax-ns#}about"
|
|
95
|
+
_RSS_CONTENT_ENCODED_TAG = "{http://purl.org/rss/1.0/modules/content/}encoded"
|
|
96
|
+
_DC_SUBJECT_TAG = "{http://purl.org/dc/elements/1.1/}subject"
|
|
97
|
+
_MEDIA_CONTENT_TAG = "{http://search.yahoo.com/mrss/}content"
|
|
98
|
+
_MEDIA_THUMBNAIL_TAG = "{http://search.yahoo.com/mrss/}thumbnail"
|
|
99
|
+
_MEDIA_TITLE_TAG = "{http://search.yahoo.com/mrss/}title"
|
|
100
|
+
_MEDIA_TEXT_TAG = "{http://search.yahoo.com/mrss/}text"
|
|
101
|
+
_MEDIA_DESCRIPTION_TAG = "{http://search.yahoo.com/mrss/}description"
|
|
102
|
+
_MEDIA_CREDIT_TAG = "{http://search.yahoo.com/mrss/}credit"
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
@lru_cache(maxsize=4)
|
|
106
|
+
def _atom_ns_tags(atom_ns: str) -> dict[str, str]:
|
|
107
|
+
"""Pre-compute namespace-prefixed tag strings once per unique namespace.
|
|
108
|
+
|
|
109
|
+
Avoids thousands of redundant f-string / concatenation operations when
|
|
110
|
+
parsing feeds with many entries.
|
|
111
|
+
"""
|
|
112
|
+
ns = f"{{{atom_ns}}}"
|
|
113
|
+
is_atom_03 = atom_ns == "http://purl.org/atom/ns#"
|
|
114
|
+
return {
|
|
115
|
+
"ns": ns,
|
|
116
|
+
"id": ns + "id",
|
|
117
|
+
"title": ns + "title",
|
|
118
|
+
"summary": ns + "summary",
|
|
119
|
+
"link": ns + "link",
|
|
120
|
+
"content": ns + "content",
|
|
121
|
+
"author_name": ns + "author/" + ns + "name",
|
|
122
|
+
"category": ns + "category",
|
|
123
|
+
"published": ns + ("issued" if is_atom_03 else "published"),
|
|
124
|
+
"updated": ns + ("modified" if is_atom_03 else "updated"),
|
|
125
|
+
"pub_fallback": ns + ("published" if is_atom_03 else "issued"),
|
|
126
|
+
"upd_fallback": ns + ("updated" if is_atom_03 else "modified"),
|
|
127
|
+
}
|
|
128
|
+
|
|
84
129
|
|
|
85
130
|
class FastFeedParserDict(dict):
|
|
86
131
|
"""A dictionary that allows access to its keys as attributes."""
|
|
@@ -154,11 +199,18 @@ def _clean_feed_bytes(content: bytes) -> bytes:
|
|
|
154
199
|
b"<?xml-stylesheet",
|
|
155
200
|
)
|
|
156
201
|
|
|
157
|
-
|
|
158
|
-
|
|
159
|
-
|
|
160
|
-
|
|
161
|
-
|
|
202
|
+
# Search for XML start patterns without splitting entire content into lines.
|
|
203
|
+
# For large feeds (multi-MB), splitlines() creates thousands of byte string
|
|
204
|
+
# objects; find() scans in-place with zero allocations.
|
|
205
|
+
search_limit = min(len(content), 8192)
|
|
206
|
+
search_chunk = content[:search_limit].lower()
|
|
207
|
+
earliest = -1
|
|
208
|
+
for pattern in xml_start_patterns:
|
|
209
|
+
idx = search_chunk.find(pattern)
|
|
210
|
+
if idx != -1 and (earliest == -1 or idx < earliest):
|
|
211
|
+
earliest = idx
|
|
212
|
+
if earliest != -1:
|
|
213
|
+
return content[earliest:]
|
|
162
214
|
|
|
163
215
|
if b"<script>" in preview_lower or b"<body>" in preview_lower:
|
|
164
216
|
raise ValueError("Content appears to be HTML, not a valid RSS/Atom feed")
|
|
@@ -167,21 +219,30 @@ def _clean_feed_bytes(content: bytes) -> bytes:
|
|
|
167
219
|
|
|
168
220
|
|
|
169
221
|
def _fix_malformed_xml_bytes(content: bytes, actual_encoding: str = "utf-8") -> bytes:
|
|
222
|
+
# XML declarations and encoding definitions live at the top of the file.
|
|
223
|
+
# Run declaration-fixing regexes only on the first 2 KB to avoid scanning
|
|
224
|
+
# multi-megabyte payloads with patterns that can only match the header.
|
|
225
|
+
header = content[:2048]
|
|
226
|
+
tail = content[2048:]
|
|
227
|
+
|
|
170
228
|
# Fix double XML declarations like "<?xml?xml version="1.0"?>"
|
|
171
|
-
|
|
229
|
+
header = _RE_DOUBLE_XML_DECL_BYTES.sub(b"<?xml ", header)
|
|
172
230
|
|
|
173
231
|
# Fix double closing ?> in XML declaration like "??>>"
|
|
174
|
-
|
|
175
|
-
|
|
176
|
-
# Fix malformed attribute syntax like rss:version=2.0 (missing quotes)
|
|
177
|
-
content = _RE_UNQUOTED_ATTR_BYTES.sub(rb'\1="\2"', content)
|
|
232
|
+
header = _RE_DOUBLE_CLOSE_BYTES.sub(b"?>", header)
|
|
178
233
|
|
|
179
234
|
# Update encoding in XML declaration to match actual encoding when a feed was transcoded.
|
|
180
235
|
if actual_encoding.lower() != "utf-16":
|
|
181
236
|
replacement = (
|
|
182
237
|
rb"\1" + actual_encoding.encode("ascii", errors="replace") + rb"\3"
|
|
183
238
|
)
|
|
184
|
-
|
|
239
|
+
header = _RE_UTF16_ENCODING_BYTES.sub(replacement, header)
|
|
240
|
+
|
|
241
|
+
# Reassemble before running body-wide fixes
|
|
242
|
+
content = header + tail
|
|
243
|
+
|
|
244
|
+
# Fix malformed attribute syntax like rss:version=2.0 (missing quotes)
|
|
245
|
+
content = _RE_UNQUOTED_ATTR_BYTES.sub(rb'\1="\2"', content)
|
|
185
246
|
|
|
186
247
|
# Fix unclosed link tags - common in Atom feeds
|
|
187
248
|
content = _RE_UNCLOSED_LINK_BYTES.sub(rb"<link\1/>", content)
|
|
@@ -225,7 +286,13 @@ def _prepare_xml_bytes(xml_content: str | bytes) -> bytes:
|
|
|
225
286
|
return _prepare_xml_bytes(xml_content.encode("utf-8", errors="replace"))
|
|
226
287
|
|
|
227
288
|
|
|
228
|
-
def _parse_json_feed(
|
|
289
|
+
def _parse_json_feed(
|
|
290
|
+
json_data: dict,
|
|
291
|
+
*,
|
|
292
|
+
include_content: bool = True,
|
|
293
|
+
include_tags: bool = True,
|
|
294
|
+
include_enclosures: bool = True,
|
|
295
|
+
) -> FastFeedParserDict:
|
|
229
296
|
"""Parse a JSON Feed and convert to FastFeedParserDict format.
|
|
230
297
|
|
|
231
298
|
JSON Feed spec: https://jsonfeed.org/
|
|
@@ -283,10 +350,12 @@ def _parse_json_feed(json_data: dict) -> FastFeedParserDict:
|
|
|
283
350
|
summary = item.get("summary", "")
|
|
284
351
|
|
|
285
352
|
if content_html:
|
|
286
|
-
|
|
353
|
+
if include_content:
|
|
354
|
+
entry["content"] = [{"type": "text/html", "value": content_html}]
|
|
287
355
|
entry["description"] = summary
|
|
288
356
|
elif content_text:
|
|
289
|
-
|
|
357
|
+
if include_content:
|
|
358
|
+
entry["content"] = [{"type": "text/plain", "value": content_text}]
|
|
290
359
|
entry["description"] = summary or content_text[:512]
|
|
291
360
|
else:
|
|
292
361
|
entry["description"] = summary
|
|
@@ -319,14 +388,14 @@ def _parse_json_feed(json_data: dict) -> FastFeedParserDict:
|
|
|
319
388
|
|
|
320
389
|
# Add tags
|
|
321
390
|
tags = item.get("tags")
|
|
322
|
-
if tags:
|
|
391
|
+
if include_tags and tags:
|
|
323
392
|
entry["tags"] = [
|
|
324
393
|
{"term": tag, "scheme": None, "label": None} for tag in tags
|
|
325
394
|
]
|
|
326
395
|
|
|
327
396
|
# Add attachments as enclosures
|
|
328
397
|
attachments = item.get("attachments")
|
|
329
|
-
if attachments:
|
|
398
|
+
if include_enclosures and attachments:
|
|
330
399
|
enclosures = []
|
|
331
400
|
for attachment in attachments:
|
|
332
401
|
url = attachment.get("url", "")
|
|
@@ -389,19 +458,23 @@ def _fetch_url_content(url: str) -> str | bytes:
|
|
|
389
458
|
return content.decode(content_charset) if content_charset else content
|
|
390
459
|
|
|
391
460
|
|
|
392
|
-
def _maybe_parse_json_feed(
|
|
461
|
+
def _maybe_parse_json_feed(
|
|
462
|
+
content: str | bytes,
|
|
463
|
+
*,
|
|
464
|
+
include_content: bool = True,
|
|
465
|
+
include_tags: bool = True,
|
|
466
|
+
include_enclosures: bool = True,
|
|
467
|
+
) -> FastFeedParserDict | None:
|
|
393
468
|
if isinstance(content, bytes):
|
|
394
469
|
if not content.lstrip().startswith(b"{"):
|
|
395
470
|
return None
|
|
396
|
-
json_str = content.decode("utf-8", errors="replace")
|
|
397
471
|
else:
|
|
398
472
|
if not content.lstrip().startswith("{"):
|
|
399
473
|
return None
|
|
400
|
-
json_str = content
|
|
401
474
|
|
|
402
475
|
try:
|
|
403
|
-
json_data =
|
|
404
|
-
except
|
|
476
|
+
json_data = _json_loads(content)
|
|
477
|
+
except Exception:
|
|
405
478
|
return None
|
|
406
479
|
|
|
407
480
|
if not isinstance(json_data, dict):
|
|
@@ -409,10 +482,20 @@ def _maybe_parse_json_feed(content: str | bytes) -> FastFeedParserDict | None:
|
|
|
409
482
|
|
|
410
483
|
version = json_data.get("version")
|
|
411
484
|
if isinstance(version, str) and "jsonfeed.org" in version:
|
|
412
|
-
return _parse_json_feed(
|
|
485
|
+
return _parse_json_feed(
|
|
486
|
+
json_data,
|
|
487
|
+
include_content=include_content,
|
|
488
|
+
include_tags=include_tags,
|
|
489
|
+
include_enclosures=include_enclosures,
|
|
490
|
+
)
|
|
413
491
|
|
|
414
492
|
if isinstance(json_data.get("items"), list):
|
|
415
|
-
return _parse_json_feed(
|
|
493
|
+
return _parse_json_feed(
|
|
494
|
+
json_data,
|
|
495
|
+
include_content=include_content,
|
|
496
|
+
include_tags=include_tags,
|
|
497
|
+
include_enclosures=include_enclosures,
|
|
498
|
+
)
|
|
416
499
|
|
|
417
500
|
return None
|
|
418
501
|
|
|
@@ -667,9 +750,21 @@ def _detect_feed_structure(
|
|
|
667
750
|
raise ValueError(f"Unknown feed type: {root.tag}")
|
|
668
751
|
|
|
669
752
|
|
|
670
|
-
def _parse_content(
|
|
753
|
+
def _parse_content(
|
|
754
|
+
xml_content: str | bytes,
|
|
755
|
+
*,
|
|
756
|
+
include_content: bool = True,
|
|
757
|
+
include_tags: bool = True,
|
|
758
|
+
include_media: bool = True,
|
|
759
|
+
include_enclosures: bool = True,
|
|
760
|
+
) -> FastFeedParserDict:
|
|
671
761
|
"""Parse feed content (XML or JSON) that has already been fetched."""
|
|
672
|
-
json_feed = _maybe_parse_json_feed(
|
|
762
|
+
json_feed = _maybe_parse_json_feed(
|
|
763
|
+
xml_content,
|
|
764
|
+
include_content=include_content,
|
|
765
|
+
include_tags=include_tags,
|
|
766
|
+
include_enclosures=include_enclosures,
|
|
767
|
+
)
|
|
673
768
|
if json_feed is not None:
|
|
674
769
|
return json_feed
|
|
675
770
|
|
|
@@ -682,7 +777,9 @@ def _parse_content(xml_content: str | bytes) -> FastFeedParserDict:
|
|
|
682
777
|
root, xml_content, root_tag_local
|
|
683
778
|
)
|
|
684
779
|
|
|
685
|
-
feed = _parse_feed_info(
|
|
780
|
+
feed = _parse_feed_info(
|
|
781
|
+
channel, feed_type, atom_namespace, include_tags=include_tags
|
|
782
|
+
)
|
|
686
783
|
|
|
687
784
|
# Detect once whether media namespace is used anywhere in the document
|
|
688
785
|
has_media_ns = (
|
|
@@ -694,34 +791,41 @@ def _parse_content(xml_content: str | bytes) -> FastFeedParserDict:
|
|
|
694
791
|
# Parse entries — resolve parser once per feed instead of per entry
|
|
695
792
|
entries: list[FastFeedParserDict] = []
|
|
696
793
|
feed["entries"] = entries
|
|
697
|
-
|
|
698
|
-
|
|
699
|
-
|
|
700
|
-
|
|
701
|
-
|
|
702
|
-
|
|
703
|
-
|
|
704
|
-
|
|
705
|
-
|
|
706
|
-
|
|
707
|
-
|
|
708
|
-
|
|
709
|
-
|
|
710
|
-
|
|
711
|
-
|
|
712
|
-
entry = _parse_feed_entry(item, feed_type, atom_namespace, has_media_ns)
|
|
713
|
-
entry["title"] = entry.get("title", "").strip()
|
|
714
|
-
entry["description"] = entry.get("description", "").strip()
|
|
715
|
-
entries.append(entry)
|
|
794
|
+
for item in items:
|
|
795
|
+
entry = _parse_feed_entry(
|
|
796
|
+
item,
|
|
797
|
+
feed_type,
|
|
798
|
+
atom_namespace,
|
|
799
|
+
has_media_ns,
|
|
800
|
+
include_content=include_content,
|
|
801
|
+
include_tags=include_tags,
|
|
802
|
+
include_media=include_media,
|
|
803
|
+
include_enclosures=include_enclosures,
|
|
804
|
+
)
|
|
805
|
+
# Ensure that titles and descriptions are always present
|
|
806
|
+
entry["title"] = entry.get("title", "").strip()
|
|
807
|
+
entry["description"] = entry.get("description", "").strip()
|
|
808
|
+
entries.append(entry)
|
|
716
809
|
|
|
717
810
|
return feed
|
|
718
811
|
|
|
719
812
|
|
|
720
|
-
def parse(
|
|
813
|
+
def parse(
|
|
814
|
+
source: str | bytes,
|
|
815
|
+
*,
|
|
816
|
+
include_content: bool = True,
|
|
817
|
+
include_tags: bool = True,
|
|
818
|
+
include_media: bool = True,
|
|
819
|
+
include_enclosures: bool = True,
|
|
820
|
+
) -> FastFeedParserDict:
|
|
721
821
|
"""Parse a feed from a URL or XML content.
|
|
722
822
|
|
|
723
823
|
Args:
|
|
724
824
|
source: URL string or XML content string/bytes
|
|
825
|
+
include_content: Include per-entry content blobs and synthesized descriptions
|
|
826
|
+
include_tags: Include feed and entry tags/categories
|
|
827
|
+
include_media: Include media namespace content (media:content/media:thumbnail)
|
|
828
|
+
include_enclosures: Include RSS enclosures and JSON-feed attachments
|
|
725
829
|
|
|
726
830
|
Returns:
|
|
727
831
|
FastFeedParserDict containing parsed feed data
|
|
@@ -738,7 +842,13 @@ def parse(source: str | bytes) -> FastFeedParserDict:
|
|
|
738
842
|
content = source
|
|
739
843
|
|
|
740
844
|
try:
|
|
741
|
-
return _parse_content(
|
|
845
|
+
return _parse_content(
|
|
846
|
+
content,
|
|
847
|
+
include_content=include_content,
|
|
848
|
+
include_tags=include_tags,
|
|
849
|
+
include_media=include_media,
|
|
850
|
+
include_enclosures=include_enclosures,
|
|
851
|
+
)
|
|
742
852
|
except ValueError as e:
|
|
743
853
|
if not is_url:
|
|
744
854
|
raise
|
|
@@ -749,11 +859,21 @@ def parse(source: str | bytes) -> FastFeedParserDict:
|
|
|
749
859
|
redirect_url = _extract_meta_refresh_url(content, source)
|
|
750
860
|
if redirect_url is None:
|
|
751
861
|
raise
|
|
752
|
-
return parse(
|
|
862
|
+
return parse(
|
|
863
|
+
redirect_url,
|
|
864
|
+
include_content=include_content,
|
|
865
|
+
include_tags=include_tags,
|
|
866
|
+
include_media=include_media,
|
|
867
|
+
include_enclosures=include_enclosures,
|
|
868
|
+
)
|
|
753
869
|
|
|
754
870
|
|
|
755
871
|
def _parse_feed_info(
|
|
756
|
-
channel: _Element,
|
|
872
|
+
channel: _Element,
|
|
873
|
+
feed_type: _FeedType,
|
|
874
|
+
atom_namespace: Optional[str] = None,
|
|
875
|
+
*,
|
|
876
|
+
include_tags: bool = True,
|
|
757
877
|
) -> FastFeedParserDict:
|
|
758
878
|
# Use dynamic atom namespace or fallback to default
|
|
759
879
|
atom_ns = atom_namespace or "http://www.w3.org/2005/Atom"
|
|
@@ -824,8 +944,8 @@ def _parse_feed_info(
|
|
|
824
944
|
if value:
|
|
825
945
|
feed[field[0]] = value
|
|
826
946
|
|
|
827
|
-
feed_lang = channel.get(
|
|
828
|
-
feed_base = channel.get(
|
|
947
|
+
feed_lang = channel.get(_XML_LANG_ATTR)
|
|
948
|
+
feed_base = channel.get(_XML_BASE_ATTR)
|
|
829
949
|
feed["language"] = feed_lang
|
|
830
950
|
|
|
831
951
|
# Add title_detail and subtitle_detail
|
|
@@ -896,9 +1016,10 @@ def _parse_feed_info(
|
|
|
896
1016
|
feed["author"] = managing_editor
|
|
897
1017
|
|
|
898
1018
|
# Parse feed-level tags/categories
|
|
899
|
-
|
|
900
|
-
|
|
901
|
-
|
|
1019
|
+
if include_tags:
|
|
1020
|
+
tags = _parse_tags(channel, feed_type, atom_ns)
|
|
1021
|
+
if tags:
|
|
1022
|
+
feed["tags"] = tags
|
|
902
1023
|
|
|
903
1024
|
return FastFeedParserDict(feed=feed)
|
|
904
1025
|
|
|
@@ -917,14 +1038,14 @@ def _parse_tags(
|
|
|
917
1038
|
{"term": term, "scheme": cat.get("domain"), "label": None}
|
|
918
1039
|
)
|
|
919
1040
|
# RSS might also use <dc:subject>
|
|
920
|
-
for subject in element.findall(
|
|
1041
|
+
for subject in element.findall(_DC_SUBJECT_TAG):
|
|
921
1042
|
term = subject.text.strip() if subject.text else None
|
|
922
1043
|
if term:
|
|
923
1044
|
tags_list.append({"term": term, "scheme": None, "label": None})
|
|
924
1045
|
elif feed_type == "atom":
|
|
925
1046
|
# Atom uses <category> elements with attributes
|
|
926
1047
|
atom_ns = atom_namespace or "http://www.w3.org/2005/Atom"
|
|
927
|
-
for cat in element.findall(
|
|
1048
|
+
for cat in element.findall(_atom_ns_tags(atom_ns)["category"]):
|
|
928
1049
|
term = cat.get("term")
|
|
929
1050
|
if term:
|
|
930
1051
|
tags_list.append(
|
|
@@ -936,7 +1057,7 @@ def _parse_tags(
|
|
|
936
1057
|
)
|
|
937
1058
|
elif feed_type == "rdf":
|
|
938
1059
|
# RDF uses <dc:subject> or <taxo:topic>
|
|
939
|
-
for subject in element.findall(
|
|
1060
|
+
for subject in element.findall(_DC_SUBJECT_TAG):
|
|
940
1061
|
term = subject.text.strip() if subject.text else None
|
|
941
1062
|
if term:
|
|
942
1063
|
tags_list.append({"term": term, "scheme": None, "label": None})
|
|
@@ -972,12 +1093,16 @@ def _coerce_int_fields(mapping: dict[str, Any], fields: tuple[str, ...]) -> None
|
|
|
972
1093
|
mapping.pop(field, None)
|
|
973
1094
|
|
|
974
1095
|
|
|
975
|
-
def
|
|
976
|
-
entry: FastFeedParserDict,
|
|
1096
|
+
def _populate_entry_links_from_elements(
|
|
1097
|
+
entry: FastFeedParserDict,
|
|
1098
|
+
atom_links: list[_Element],
|
|
1099
|
+
*,
|
|
1100
|
+
guid_text: Optional[str] = None,
|
|
1101
|
+
guid_is_permalink: bool = False,
|
|
977
1102
|
) -> None:
|
|
978
1103
|
entry_links: list[dict[str, Optional[str]]] = []
|
|
979
1104
|
alternate_link: Optional[dict[str, Optional[str]]] = None
|
|
980
|
-
for link in
|
|
1105
|
+
for link in atom_links:
|
|
981
1106
|
rel = link.get("rel")
|
|
982
1107
|
href = link.get("href") or link.get("link")
|
|
983
1108
|
if not href:
|
|
@@ -993,8 +1118,6 @@ def _populate_entry_links(
|
|
|
993
1118
|
elif rel not in {"edit", "self"}:
|
|
994
1119
|
entry_links.append(link_dict)
|
|
995
1120
|
|
|
996
|
-
guid = item.find("guid")
|
|
997
|
-
guid_text = guid.text.strip() if guid is not None and guid.text else None
|
|
998
1121
|
is_guid_url = guid_text is not None and guid_text.startswith(
|
|
999
1122
|
("http://", "https://")
|
|
1000
1123
|
)
|
|
@@ -1008,44 +1131,55 @@ def _populate_entry_links(
|
|
|
1008
1131
|
elif alternate_link:
|
|
1009
1132
|
entry["link"] = alternate_link["href"]
|
|
1010
1133
|
entry_links.insert(0, alternate_link)
|
|
1011
|
-
elif (
|
|
1012
|
-
("link" not in entry)
|
|
1013
|
-
and (guid is not None)
|
|
1014
|
-
and guid.get("isPermaLink") == "true"
|
|
1015
|
-
):
|
|
1134
|
+
elif ("link" not in entry) and guid_is_permalink:
|
|
1016
1135
|
entry["link"] = guid_text
|
|
1017
1136
|
|
|
1018
1137
|
entry["links"] = entry_links
|
|
1019
1138
|
|
|
1020
1139
|
|
|
1021
|
-
def
|
|
1022
|
-
entry: FastFeedParserDict,
|
|
1023
|
-
item: _Element,
|
|
1024
|
-
feed_type: _FeedType,
|
|
1025
|
-
atom_ns: str,
|
|
1026
|
-
rss_text_by_full: Optional[dict[str, Optional[str]]] = None,
|
|
1140
|
+
def _populate_entry_links(
|
|
1141
|
+
entry: FastFeedParserDict, item: _Element, atom_ns: str
|
|
1027
1142
|
) -> None:
|
|
1028
|
-
|
|
1029
|
-
|
|
1030
|
-
|
|
1031
|
-
|
|
1032
|
-
|
|
1033
|
-
|
|
1034
|
-
|
|
1035
|
-
|
|
1036
|
-
|
|
1037
|
-
|
|
1038
|
-
|
|
1039
|
-
|
|
1040
|
-
|
|
1041
|
-
|
|
1143
|
+
tags = _atom_ns_tags(atom_ns)
|
|
1144
|
+
guid = item.find("guid")
|
|
1145
|
+
guid_text = guid.text.strip() if guid is not None and guid.text else None
|
|
1146
|
+
_populate_entry_links_from_elements(
|
|
1147
|
+
entry,
|
|
1148
|
+
item.findall(tags["link"]),
|
|
1149
|
+
guid_text=guid_text,
|
|
1150
|
+
guid_is_permalink=guid is not None and guid.get("isPermaLink") == "true",
|
|
1151
|
+
)
|
|
1152
|
+
|
|
1153
|
+
|
|
1154
|
+
def _synthesize_entry_description(entry: FastFeedParserDict) -> None:
|
|
1155
|
+
if "description" in entry or "content" not in entry:
|
|
1156
|
+
return
|
|
1157
|
+
|
|
1158
|
+
content_value = entry["content"][0]["value"]
|
|
1159
|
+
if content_value:
|
|
1160
|
+
if "<" in content_value and ">" in content_value:
|
|
1161
|
+
content_value = _RE_HTML_TAGS.sub(" ", content_value[:2048])
|
|
1162
|
+
if "&" in content_value:
|
|
1163
|
+
content_value = _html_mod.unescape(content_value)
|
|
1164
|
+
if (
|
|
1165
|
+
" " in content_value
|
|
1166
|
+
or "\n" in content_value
|
|
1167
|
+
or "\t" in content_value
|
|
1168
|
+
or "\r" in content_value
|
|
1169
|
+
):
|
|
1170
|
+
content_value = _RE_WHITESPACE.sub(" ", content_value).strip()
|
|
1042
1171
|
else:
|
|
1043
|
-
|
|
1044
|
-
|
|
1045
|
-
content_el = item.find("content")
|
|
1046
|
-
elif feed_type == "atom":
|
|
1047
|
-
content_el = item.find(f"{{{atom_ns}}}content")
|
|
1172
|
+
content_value = content_value.strip()
|
|
1173
|
+
entry["description"] = content_value[:512]
|
|
1048
1174
|
|
|
1175
|
+
|
|
1176
|
+
def _populate_entry_content_preparsed(
|
|
1177
|
+
entry: FastFeedParserDict,
|
|
1178
|
+
item: _Element,
|
|
1179
|
+
*,
|
|
1180
|
+
content_el: Optional[_Element],
|
|
1181
|
+
rss_description_text: Optional[str],
|
|
1182
|
+
) -> None:
|
|
1049
1183
|
if content_el is not None:
|
|
1050
1184
|
content_type = content_el.get("type", "text/html")
|
|
1051
1185
|
if content_type in {"xhtml", "application/xhtml+xml"}:
|
|
@@ -1055,48 +1189,51 @@ def _populate_entry_content(
|
|
|
1055
1189
|
entry["content"] = [
|
|
1056
1190
|
{
|
|
1057
1191
|
"type": content_type,
|
|
1058
|
-
"language": content_el.get(
|
|
1059
|
-
|
|
1060
|
-
),
|
|
1061
|
-
"base": content_el.get("{http://www.w3.org/XML/1998/namespace}base"),
|
|
1192
|
+
"language": content_el.get(_XML_LANG_ATTR),
|
|
1193
|
+
"base": content_el.get(_XML_BASE_ATTR),
|
|
1062
1194
|
"value": content_value,
|
|
1063
1195
|
}
|
|
1064
1196
|
]
|
|
1197
|
+
elif rss_description_text:
|
|
1198
|
+
entry["content"] = [
|
|
1199
|
+
{
|
|
1200
|
+
"type": "text/html",
|
|
1201
|
+
"language": item.get(_XML_LANG_ATTR),
|
|
1202
|
+
"base": item.get(_XML_BASE_ATTR),
|
|
1203
|
+
"value": rss_description_text,
|
|
1204
|
+
}
|
|
1205
|
+
]
|
|
1206
|
+
|
|
1207
|
+
_synthesize_entry_description(entry)
|
|
1208
|
+
|
|
1065
1209
|
|
|
1066
|
-
|
|
1210
|
+
def _populate_entry_content(
|
|
1211
|
+
entry: FastFeedParserDict, item: _Element, feed_type: _FeedType, atom_ns: str
|
|
1212
|
+
) -> None:
|
|
1213
|
+
content_el: Optional[_Element] = None
|
|
1214
|
+
rss_description_text: Optional[str] = None
|
|
1215
|
+
if feed_type == "rss":
|
|
1216
|
+
content_el = item.find(_RSS_CONTENT_ENCODED_TAG)
|
|
1217
|
+
if content_el is None:
|
|
1218
|
+
content_el = item.find("content")
|
|
1067
1219
|
description = item.find("description")
|
|
1068
|
-
if description is not None
|
|
1069
|
-
|
|
1070
|
-
|
|
1071
|
-
|
|
1072
|
-
"language": item.get("{http://www.w3.org/XML/1998/namespace}lang"),
|
|
1073
|
-
"base": item.get("{http://www.w3.org/XML/1998/namespace}base"),
|
|
1074
|
-
"value": description.text,
|
|
1075
|
-
}
|
|
1076
|
-
]
|
|
1220
|
+
if description is not None:
|
|
1221
|
+
rss_description_text = description.text
|
|
1222
|
+
elif feed_type == "atom":
|
|
1223
|
+
content_el = item.find(_atom_ns_tags(atom_ns)["content"])
|
|
1077
1224
|
|
|
1078
|
-
|
|
1079
|
-
|
|
1080
|
-
|
|
1081
|
-
|
|
1082
|
-
|
|
1083
|
-
|
|
1084
|
-
if (
|
|
1085
|
-
" " in content_value
|
|
1086
|
-
or "\n" in content_value
|
|
1087
|
-
or "\t" in content_value
|
|
1088
|
-
or "\r" in content_value
|
|
1089
|
-
):
|
|
1090
|
-
content_value = _RE_WHITESPACE.sub(" ", content_value).strip()
|
|
1091
|
-
else:
|
|
1092
|
-
content_value = content_value.strip()
|
|
1093
|
-
entry["description"] = content_value[:512]
|
|
1225
|
+
_populate_entry_content_preparsed(
|
|
1226
|
+
entry,
|
|
1227
|
+
item,
|
|
1228
|
+
content_el=content_el,
|
|
1229
|
+
rss_description_text=rss_description_text,
|
|
1230
|
+
)
|
|
1094
1231
|
|
|
1095
1232
|
|
|
1096
1233
|
def _parse_media_content(item: _Element) -> list[dict[str, Any]] | None:
|
|
1097
1234
|
media_contents: list[dict[str, Any]] = []
|
|
1098
1235
|
|
|
1099
|
-
for media in item.findall(".//{
|
|
1236
|
+
for media in item.findall(f".//{_MEDIA_CONTENT_TAG}"):
|
|
1100
1237
|
media_item: dict[str, str | int | None] = {
|
|
1101
1238
|
"url": media.get("url"),
|
|
1102
1239
|
"type": media.get("type"),
|
|
@@ -1106,32 +1243,32 @@ def _parse_media_content(item: _Element) -> list[dict[str, Any]] | None:
|
|
|
1106
1243
|
}
|
|
1107
1244
|
_coerce_int_fields(media_item, ("width", "height"))
|
|
1108
1245
|
|
|
1109
|
-
title = media.find(
|
|
1246
|
+
title = media.find(_MEDIA_TITLE_TAG)
|
|
1110
1247
|
if title is not None and title.text:
|
|
1111
1248
|
media_item["title"] = title.text.strip()
|
|
1112
1249
|
|
|
1113
|
-
text = media.find(
|
|
1250
|
+
text = media.find(_MEDIA_TEXT_TAG)
|
|
1114
1251
|
if text is not None and text.text:
|
|
1115
1252
|
media_item["text"] = text.text.strip()
|
|
1116
1253
|
|
|
1117
|
-
desc = media.find(
|
|
1254
|
+
desc = media.find(_MEDIA_DESCRIPTION_TAG)
|
|
1118
1255
|
if desc is None:
|
|
1119
1256
|
parent = media.getparent()
|
|
1120
1257
|
if parent is not None:
|
|
1121
|
-
desc = parent.find(
|
|
1258
|
+
desc = parent.find(_MEDIA_DESCRIPTION_TAG)
|
|
1122
1259
|
if desc is not None and desc.text:
|
|
1123
1260
|
media_item["description"] = desc.text.strip()
|
|
1124
1261
|
|
|
1125
|
-
credit = media.find(
|
|
1262
|
+
credit = media.find(_MEDIA_CREDIT_TAG)
|
|
1126
1263
|
if credit is None:
|
|
1127
1264
|
parent = media.getparent()
|
|
1128
1265
|
if parent is not None:
|
|
1129
|
-
credit = parent.find(
|
|
1266
|
+
credit = parent.find(_MEDIA_CREDIT_TAG)
|
|
1130
1267
|
if credit is not None and credit.text:
|
|
1131
1268
|
media_item["credit"] = credit.text.strip()
|
|
1132
1269
|
media_item["credit_scheme"] = credit.get("scheme")
|
|
1133
1270
|
|
|
1134
|
-
thumbnail = media.find(
|
|
1271
|
+
thumbnail = media.find(_MEDIA_THUMBNAIL_TAG)
|
|
1135
1272
|
if thumbnail is not None:
|
|
1136
1273
|
media_item["thumbnail_url"] = thumbnail.get("url")
|
|
1137
1274
|
|
|
@@ -1140,9 +1277,9 @@ def _parse_media_content(item: _Element) -> list[dict[str, Any]] | None:
|
|
|
1140
1277
|
media_contents.append(cleaned)
|
|
1141
1278
|
|
|
1142
1279
|
if not media_contents:
|
|
1143
|
-
for thumbnail in item.findall(".//{
|
|
1280
|
+
for thumbnail in item.findall(f".//{_MEDIA_THUMBNAIL_TAG}"):
|
|
1144
1281
|
parent = thumbnail.getparent()
|
|
1145
|
-
if parent is None or parent.tag ==
|
|
1282
|
+
if parent is None or parent.tag == _MEDIA_CONTENT_TAG:
|
|
1146
1283
|
continue
|
|
1147
1284
|
thumb_item: dict[str, str | int | None] = {
|
|
1148
1285
|
"url": thumbnail.get("url"),
|
|
@@ -1161,47 +1298,26 @@ def _parse_media_content(item: _Element) -> list[dict[str, Any]] | None:
|
|
|
1161
1298
|
def _parse_enclosures(item: _Element) -> list[dict[str, Any]] | None:
|
|
1162
1299
|
enclosures: list[dict[str, Any]] = []
|
|
1163
1300
|
for enclosure in item.findall("enclosure"):
|
|
1164
|
-
|
|
1165
|
-
"url": enclosure.get("url"),
|
|
1166
|
-
"type": enclosure.get("type"),
|
|
1167
|
-
"length": enclosure.get("length"),
|
|
1168
|
-
}
|
|
1169
|
-
length = enc_item.get("length")
|
|
1170
|
-
if length:
|
|
1171
|
-
try:
|
|
1172
|
-
enc_item["length"] = int(length)
|
|
1173
|
-
except (ValueError, TypeError):
|
|
1174
|
-
enc_item.pop("length", None)
|
|
1175
|
-
|
|
1176
|
-
cleaned = _drop_none_values(enc_item)
|
|
1301
|
+
cleaned = _parse_enclosure_element(enclosure)
|
|
1177
1302
|
if cleaned.get("url"):
|
|
1178
1303
|
enclosures.append(cleaned)
|
|
1179
1304
|
|
|
1180
1305
|
return enclosures or None
|
|
1181
1306
|
|
|
1182
1307
|
|
|
1183
|
-
def
|
|
1184
|
-
|
|
1185
|
-
)
|
|
1186
|
-
|
|
1187
|
-
|
|
1188
|
-
|
|
1189
|
-
|
|
1190
|
-
|
|
1191
|
-
|
|
1192
|
-
|
|
1193
|
-
|
|
1194
|
-
|
|
1195
|
-
|
|
1196
|
-
if "{" in tag:
|
|
1197
|
-
local = tag.rsplit("}", 1)[1].lower()
|
|
1198
|
-
elif ":" in tag:
|
|
1199
|
-
local = tag.split(":", 1)[1].lower()
|
|
1200
|
-
else:
|
|
1201
|
-
local = tag.lower()
|
|
1202
|
-
if local not in by_local:
|
|
1203
|
-
by_local[local] = text_value
|
|
1204
|
-
return by_local, by_full
|
|
1308
|
+
def _parse_enclosure_element(enclosure: _Element) -> dict[str, Any]:
|
|
1309
|
+
enc_item: dict[str, str | int | None] = {
|
|
1310
|
+
"url": enclosure.get("url"),
|
|
1311
|
+
"type": enclosure.get("type"),
|
|
1312
|
+
"length": enclosure.get("length"),
|
|
1313
|
+
}
|
|
1314
|
+
length = enc_item.get("length")
|
|
1315
|
+
if length:
|
|
1316
|
+
try:
|
|
1317
|
+
enc_item["length"] = int(length)
|
|
1318
|
+
except (ValueError, TypeError):
|
|
1319
|
+
enc_item.pop("length", None)
|
|
1320
|
+
return _drop_none_values(enc_item)
|
|
1205
1321
|
|
|
1206
1322
|
|
|
1207
1323
|
def _first_non_empty(
|
|
@@ -1218,13 +1334,78 @@ def _parse_rss_feed_entry_fast(
|
|
|
1218
1334
|
item: _Element,
|
|
1219
1335
|
atom_ns: str,
|
|
1220
1336
|
has_media_ns: bool = True,
|
|
1337
|
+
*,
|
|
1338
|
+
include_content: bool = True,
|
|
1339
|
+
include_tags: bool = True,
|
|
1340
|
+
include_media: bool = True,
|
|
1341
|
+
include_enclosures: bool = True,
|
|
1221
1342
|
) -> FastFeedParserDict:
|
|
1222
|
-
|
|
1343
|
+
atom_tags = _atom_ns_tags(atom_ns)
|
|
1344
|
+
text_by_local: dict[str, Optional[str]] = {}
|
|
1345
|
+
text_by_full: dict[str, Optional[str]] = {}
|
|
1346
|
+
atom_links: list[_Element] = []
|
|
1347
|
+
guid_element: Optional[_Element] = None
|
|
1348
|
+
encoded_content_el: Optional[_Element] = None
|
|
1349
|
+
raw_content_el: Optional[_Element] = None
|
|
1350
|
+
rss_description_text: Optional[str] = None
|
|
1351
|
+
tag_categories: list[dict[str, str | None]] = []
|
|
1352
|
+
tag_subjects: list[dict[str, str | None]] = []
|
|
1353
|
+
enclosures: list[dict[str, Any]] = []
|
|
1354
|
+
|
|
1355
|
+
for child in item:
|
|
1356
|
+
tag = child.tag
|
|
1357
|
+
if not isinstance(tag, str):
|
|
1358
|
+
continue
|
|
1359
|
+
|
|
1360
|
+
text_value = child.text or None
|
|
1361
|
+
if tag not in text_by_full:
|
|
1362
|
+
text_by_full[tag] = text_value
|
|
1363
|
+
|
|
1364
|
+
if "{" in tag:
|
|
1365
|
+
local = tag.rsplit("}", 1)[1].lower()
|
|
1366
|
+
elif ":" in tag:
|
|
1367
|
+
local = tag.split(":", 1)[1].lower()
|
|
1368
|
+
else:
|
|
1369
|
+
local = tag.lower()
|
|
1370
|
+
if local not in text_by_local:
|
|
1371
|
+
text_by_local[local] = text_value
|
|
1372
|
+
|
|
1373
|
+
if tag == atom_tags["link"]:
|
|
1374
|
+
atom_links.append(child)
|
|
1375
|
+
elif tag == "guid":
|
|
1376
|
+
if guid_element is None:
|
|
1377
|
+
guid_element = child
|
|
1378
|
+
elif tag == _RSS_CONTENT_ENCODED_TAG:
|
|
1379
|
+
if encoded_content_el is None:
|
|
1380
|
+
encoded_content_el = child
|
|
1381
|
+
elif tag == "content":
|
|
1382
|
+
if raw_content_el is None:
|
|
1383
|
+
raw_content_el = child
|
|
1384
|
+
elif tag == "description":
|
|
1385
|
+
if rss_description_text is None:
|
|
1386
|
+
rss_description_text = text_value
|
|
1387
|
+
|
|
1388
|
+
if include_enclosures and tag == "enclosure":
|
|
1389
|
+
cleaned = _parse_enclosure_element(child)
|
|
1390
|
+
if cleaned.get("url"):
|
|
1391
|
+
enclosures.append(cleaned)
|
|
1392
|
+
|
|
1393
|
+
if include_tags:
|
|
1394
|
+
if local == "category":
|
|
1395
|
+
term = text_value.strip() if text_value else None
|
|
1396
|
+
if term:
|
|
1397
|
+
tag_categories.append(
|
|
1398
|
+
{"term": term, "scheme": child.get("domain"), "label": None}
|
|
1399
|
+
)
|
|
1400
|
+
elif tag == _DC_SUBJECT_TAG:
|
|
1401
|
+
term = text_value.strip() if text_value else None
|
|
1402
|
+
if term:
|
|
1403
|
+
tag_subjects.append({"term": term, "scheme": None, "label": None})
|
|
1223
1404
|
|
|
1224
1405
|
entry = FastFeedParserDict()
|
|
1225
|
-
atom_id = text_by_full.get(
|
|
1406
|
+
atom_id = text_by_full.get(atom_tags["id"])
|
|
1226
1407
|
rss_guid = text_by_local.get("guid")
|
|
1227
|
-
rdf_about = item.get(
|
|
1408
|
+
rdf_about = item.get(_RDF_ABOUT_ATTR)
|
|
1228
1409
|
entry_id: Optional[str] = atom_id or rss_guid or rdf_about
|
|
1229
1410
|
if entry_id:
|
|
1230
1411
|
entry["id"] = entry_id.strip()
|
|
@@ -1269,13 +1450,20 @@ def _parse_rss_feed_entry_fast(
|
|
|
1269
1450
|
if "updated" in entry and "published" not in entry:
|
|
1270
1451
|
entry["published"] = entry["updated"]
|
|
1271
1452
|
|
|
1272
|
-
# Inline link population for RSS (avoids redundant findall/find for 98.8% of entries)
|
|
1273
|
-
atom_links = item.findall(f"{{{atom_ns}}}link")
|
|
1274
1453
|
if atom_links:
|
|
1275
|
-
|
|
1276
|
-
|
|
1454
|
+
guid_text = (
|
|
1455
|
+
guid_element.text.strip()
|
|
1456
|
+
if guid_element is not None and guid_element.text
|
|
1457
|
+
else None
|
|
1458
|
+
)
|
|
1459
|
+
_populate_entry_links_from_elements(
|
|
1460
|
+
entry,
|
|
1461
|
+
atom_links,
|
|
1462
|
+
guid_text=guid_text,
|
|
1463
|
+
guid_is_permalink=guid_element is not None
|
|
1464
|
+
and guid_element.get("isPermaLink") == "true",
|
|
1465
|
+
)
|
|
1277
1466
|
else:
|
|
1278
|
-
# Common RSS case: no atom:link elements
|
|
1279
1467
|
entry["links"] = []
|
|
1280
1468
|
if (
|
|
1281
1469
|
"link" not in entry
|
|
@@ -1287,20 +1475,27 @@ def _parse_rss_feed_entry_fast(
|
|
|
1287
1475
|
if "id" not in entry and "link" in entry:
|
|
1288
1476
|
entry["id"] = entry["link"]
|
|
1289
1477
|
|
|
1290
|
-
|
|
1478
|
+
if include_content:
|
|
1479
|
+
_populate_entry_content_preparsed(
|
|
1480
|
+
entry,
|
|
1481
|
+
item,
|
|
1482
|
+
content_el=(
|
|
1483
|
+
encoded_content_el if encoded_content_el is not None else raw_content_el
|
|
1484
|
+
),
|
|
1485
|
+
rss_description_text=rss_description_text,
|
|
1486
|
+
)
|
|
1291
1487
|
|
|
1292
|
-
if has_media_ns:
|
|
1488
|
+
if include_media and has_media_ns:
|
|
1293
1489
|
media_contents = _parse_media_content(item)
|
|
1294
1490
|
if media_contents:
|
|
1295
1491
|
entry["media_content"] = media_contents
|
|
1296
1492
|
|
|
1297
|
-
|
|
1298
|
-
if enclosures:
|
|
1493
|
+
if include_enclosures and enclosures:
|
|
1299
1494
|
entry["enclosures"] = enclosures
|
|
1300
1495
|
|
|
1301
1496
|
author = _first_non_empty(text_by_local, ("author", "creator"))
|
|
1302
1497
|
if not author:
|
|
1303
|
-
atom_author = item.find(
|
|
1498
|
+
atom_author = item.find(atom_tags["author_name"])
|
|
1304
1499
|
author = (
|
|
1305
1500
|
atom_author.text.strip()
|
|
1306
1501
|
if atom_author is not None and atom_author.text
|
|
@@ -1313,9 +1508,8 @@ def _parse_rss_feed_entry_fast(
|
|
|
1313
1508
|
if comments:
|
|
1314
1509
|
entry["comments"] = comments.strip()
|
|
1315
1510
|
|
|
1316
|
-
|
|
1317
|
-
|
|
1318
|
-
entry["tags"] = tags
|
|
1511
|
+
if include_tags and (tag_categories or tag_subjects):
|
|
1512
|
+
entry["tags"] = tag_categories + tag_subjects
|
|
1319
1513
|
|
|
1320
1514
|
return entry
|
|
1321
1515
|
|
|
@@ -1324,93 +1518,134 @@ def _parse_atom_feed_entry_fast(
|
|
|
1324
1518
|
item: _Element,
|
|
1325
1519
|
atom_ns: str,
|
|
1326
1520
|
has_media_ns: bool = True,
|
|
1521
|
+
*,
|
|
1522
|
+
include_content: bool = True,
|
|
1523
|
+
include_tags: bool = True,
|
|
1524
|
+
include_media: bool = True,
|
|
1525
|
+
include_enclosures: bool = True,
|
|
1327
1526
|
) -> FastFeedParserDict:
|
|
1328
|
-
|
|
1527
|
+
t = _atom_ns_tags(atom_ns)
|
|
1528
|
+
atom_link_tag = t["link"]
|
|
1529
|
+
atom_author_tag = t["ns"] + "author"
|
|
1530
|
+
atom_name_tag = t["ns"] + "name"
|
|
1531
|
+
atom_links: list[_Element] = []
|
|
1532
|
+
atom_categories: list[dict[str, str | None]] = []
|
|
1533
|
+
enclosures: list[dict[str, Any]] = []
|
|
1534
|
+
content_el: Optional[_Element] = None
|
|
1535
|
+
author_name: Optional[str] = None
|
|
1536
|
+
first_link_href: Optional[str] = None
|
|
1537
|
+
published_source: Optional[str] = None
|
|
1538
|
+
updated_source: Optional[str] = None
|
|
1539
|
+
published_fallback_source: Optional[str] = None
|
|
1540
|
+
updated_fallback_source: Optional[str] = None
|
|
1541
|
+
|
|
1329
1542
|
entry = FastFeedParserDict()
|
|
1543
|
+
for child in item:
|
|
1544
|
+
tag = child.tag
|
|
1545
|
+
if not isinstance(tag, str):
|
|
1546
|
+
continue
|
|
1330
1547
|
|
|
1331
|
-
|
|
1332
|
-
|
|
1333
|
-
|
|
1334
|
-
|
|
1335
|
-
|
|
1336
|
-
|
|
1337
|
-
|
|
1338
|
-
|
|
1339
|
-
|
|
1340
|
-
|
|
1341
|
-
|
|
1342
|
-
|
|
1343
|
-
|
|
1344
|
-
|
|
1345
|
-
|
|
1346
|
-
|
|
1347
|
-
|
|
1348
|
-
|
|
1349
|
-
|
|
1350
|
-
|
|
1351
|
-
|
|
1352
|
-
|
|
1353
|
-
|
|
1354
|
-
|
|
1355
|
-
|
|
1356
|
-
|
|
1357
|
-
|
|
1358
|
-
|
|
1359
|
-
|
|
1360
|
-
|
|
1361
|
-
|
|
1362
|
-
|
|
1548
|
+
text_value = child.text
|
|
1549
|
+
if tag == t["id"] and "id" not in entry and text_value:
|
|
1550
|
+
entry["id"] = text_value.strip()
|
|
1551
|
+
elif tag == t["title"] and "title" not in entry and text_value:
|
|
1552
|
+
entry["title"] = text_value.strip()
|
|
1553
|
+
elif tag == t["summary"] and "description" not in entry and text_value:
|
|
1554
|
+
entry["description"] = text_value.strip()
|
|
1555
|
+
elif tag == t["published"] and published_source is None and text_value:
|
|
1556
|
+
published_source = text_value
|
|
1557
|
+
elif tag == t["updated"] and updated_source is None and text_value:
|
|
1558
|
+
updated_source = text_value
|
|
1559
|
+
elif (
|
|
1560
|
+
tag == t["pub_fallback"]
|
|
1561
|
+
and published_fallback_source is None
|
|
1562
|
+
and text_value
|
|
1563
|
+
):
|
|
1564
|
+
published_fallback_source = text_value
|
|
1565
|
+
elif (
|
|
1566
|
+
tag == t["upd_fallback"] and updated_fallback_source is None and text_value
|
|
1567
|
+
):
|
|
1568
|
+
updated_fallback_source = text_value
|
|
1569
|
+
elif tag == atom_link_tag:
|
|
1570
|
+
atom_links.append(child)
|
|
1571
|
+
href = child.get("href")
|
|
1572
|
+
if href and first_link_href is None:
|
|
1573
|
+
first_link_href = href.strip()
|
|
1574
|
+
elif include_content and tag == t["content"] and content_el is None:
|
|
1575
|
+
content_el = child
|
|
1576
|
+
elif tag == atom_author_tag and author_name is None:
|
|
1577
|
+
author_name_el = child.find(atom_name_tag)
|
|
1578
|
+
if author_name_el is not None and author_name_el.text:
|
|
1579
|
+
author_name = author_name_el.text.strip()
|
|
1580
|
+
|
|
1581
|
+
if include_tags and tag == t["category"]:
|
|
1582
|
+
term = child.get("term")
|
|
1583
|
+
if term:
|
|
1584
|
+
atom_categories.append(
|
|
1585
|
+
{
|
|
1586
|
+
"term": term,
|
|
1587
|
+
"scheme": child.get("scheme"),
|
|
1588
|
+
"label": child.get("label"),
|
|
1589
|
+
}
|
|
1590
|
+
)
|
|
1591
|
+
|
|
1592
|
+
if include_enclosures and tag == "enclosure":
|
|
1593
|
+
cleaned = _parse_enclosure_element(child)
|
|
1594
|
+
if cleaned.get("url"):
|
|
1595
|
+
enclosures.append(cleaned)
|
|
1596
|
+
|
|
1597
|
+
if first_link_href:
|
|
1598
|
+
entry["link"] = first_link_href
|
|
1599
|
+
|
|
1600
|
+
if published_source:
|
|
1601
|
+
published = _parse_date(published_source)
|
|
1363
1602
|
if published:
|
|
1364
1603
|
entry["published"] = published
|
|
1365
1604
|
|
|
1366
|
-
|
|
1367
|
-
|
|
1368
|
-
updated = _parse_date(el.text)
|
|
1605
|
+
if updated_source:
|
|
1606
|
+
updated = _parse_date(updated_source)
|
|
1369
1607
|
if updated:
|
|
1370
1608
|
entry["updated"] = updated
|
|
1371
1609
|
|
|
1372
|
-
|
|
1373
|
-
|
|
1374
|
-
|
|
1375
|
-
|
|
1376
|
-
published = _parse_date(el.text)
|
|
1377
|
-
if published:
|
|
1378
|
-
entry["published"] = published
|
|
1610
|
+
if "published" not in entry and published_fallback_source:
|
|
1611
|
+
published = _parse_date(published_fallback_source)
|
|
1612
|
+
if published:
|
|
1613
|
+
entry["published"] = published
|
|
1379
1614
|
|
|
1380
|
-
if "updated" not in entry:
|
|
1381
|
-
|
|
1382
|
-
if
|
|
1383
|
-
updated =
|
|
1384
|
-
if updated:
|
|
1385
|
-
entry["updated"] = updated
|
|
1615
|
+
if "updated" not in entry and updated_fallback_source:
|
|
1616
|
+
updated = _parse_date(updated_fallback_source)
|
|
1617
|
+
if updated:
|
|
1618
|
+
entry["updated"] = updated
|
|
1386
1619
|
|
|
1387
1620
|
if "updated" in entry and "published" not in entry:
|
|
1388
1621
|
entry["published"] = entry["updated"]
|
|
1389
1622
|
|
|
1390
|
-
|
|
1623
|
+
_populate_entry_links_from_elements(entry, atom_links)
|
|
1391
1624
|
|
|
1392
1625
|
if "id" not in entry and "link" in entry:
|
|
1393
1626
|
entry["id"] = entry["link"]
|
|
1394
1627
|
|
|
1395
|
-
|
|
1628
|
+
if include_content:
|
|
1629
|
+
_populate_entry_content_preparsed(
|
|
1630
|
+
entry,
|
|
1631
|
+
item,
|
|
1632
|
+
content_el=content_el,
|
|
1633
|
+
rss_description_text=None,
|
|
1634
|
+
)
|
|
1396
1635
|
|
|
1397
|
-
if has_media_ns:
|
|
1636
|
+
if include_media and has_media_ns:
|
|
1398
1637
|
media_contents = _parse_media_content(item)
|
|
1399
1638
|
if media_contents:
|
|
1400
1639
|
entry["media_content"] = media_contents
|
|
1401
1640
|
|
|
1402
|
-
|
|
1403
|
-
if enclosures:
|
|
1641
|
+
if include_enclosures and enclosures:
|
|
1404
1642
|
entry["enclosures"] = enclosures
|
|
1405
1643
|
|
|
1406
|
-
|
|
1407
|
-
|
|
1408
|
-
if el is not None and el.text:
|
|
1409
|
-
entry["author"] = el.text.strip()
|
|
1644
|
+
if author_name:
|
|
1645
|
+
entry["author"] = author_name
|
|
1410
1646
|
|
|
1411
|
-
|
|
1412
|
-
|
|
1413
|
-
entry["tags"] = tags
|
|
1647
|
+
if include_tags and atom_categories:
|
|
1648
|
+
entry["tags"] = atom_categories
|
|
1414
1649
|
|
|
1415
1650
|
return entry
|
|
1416
1651
|
|
|
@@ -1420,15 +1655,36 @@ def _parse_feed_entry(
|
|
|
1420
1655
|
feed_type: _FeedType,
|
|
1421
1656
|
atom_namespace: Optional[str] = None,
|
|
1422
1657
|
has_media_ns: bool = True,
|
|
1658
|
+
*,
|
|
1659
|
+
include_content: bool = True,
|
|
1660
|
+
include_tags: bool = True,
|
|
1661
|
+
include_media: bool = True,
|
|
1662
|
+
include_enclosures: bool = True,
|
|
1423
1663
|
) -> FastFeedParserDict:
|
|
1424
1664
|
# Use dynamic atom namespace or fallback to default
|
|
1425
1665
|
atom_ns = atom_namespace or "http://www.w3.org/2005/Atom"
|
|
1426
1666
|
|
|
1427
1667
|
if feed_type == "rss":
|
|
1428
|
-
return _parse_rss_feed_entry_fast(
|
|
1668
|
+
return _parse_rss_feed_entry_fast(
|
|
1669
|
+
item,
|
|
1670
|
+
atom_ns,
|
|
1671
|
+
has_media_ns,
|
|
1672
|
+
include_content=include_content,
|
|
1673
|
+
include_tags=include_tags,
|
|
1674
|
+
include_media=include_media,
|
|
1675
|
+
include_enclosures=include_enclosures,
|
|
1676
|
+
)
|
|
1429
1677
|
|
|
1430
1678
|
if feed_type == "atom":
|
|
1431
|
-
return _parse_atom_feed_entry_fast(
|
|
1679
|
+
return _parse_atom_feed_entry_fast(
|
|
1680
|
+
item,
|
|
1681
|
+
atom_ns,
|
|
1682
|
+
has_media_ns,
|
|
1683
|
+
include_content=include_content,
|
|
1684
|
+
include_tags=include_tags,
|
|
1685
|
+
include_media=include_media,
|
|
1686
|
+
include_enclosures=include_enclosures,
|
|
1687
|
+
)
|
|
1432
1688
|
|
|
1433
1689
|
# RDF path uses the generic field machinery
|
|
1434
1690
|
# Check if this is Atom 0.3 to use different date field names
|
|
@@ -1544,16 +1800,18 @@ def _parse_feed_entry(
|
|
|
1544
1800
|
if "id" not in entry and "link" in entry:
|
|
1545
1801
|
entry["id"] = entry["link"]
|
|
1546
1802
|
|
|
1547
|
-
|
|
1803
|
+
if include_content:
|
|
1804
|
+
_populate_entry_content(entry, item, feed_type, atom_ns)
|
|
1548
1805
|
|
|
1549
|
-
if has_media_ns:
|
|
1806
|
+
if include_media and has_media_ns:
|
|
1550
1807
|
media_contents = _parse_media_content(item)
|
|
1551
1808
|
if media_contents:
|
|
1552
1809
|
entry["media_content"] = media_contents
|
|
1553
1810
|
|
|
1554
|
-
|
|
1555
|
-
|
|
1556
|
-
|
|
1811
|
+
if include_enclosures:
|
|
1812
|
+
enclosures = _parse_enclosures(item)
|
|
1813
|
+
if enclosures:
|
|
1814
|
+
entry["enclosures"] = enclosures
|
|
1557
1815
|
|
|
1558
1816
|
author = get_field_value(
|
|
1559
1817
|
"author",
|
|
@@ -1569,9 +1827,10 @@ def _parse_feed_entry(
|
|
|
1569
1827
|
entry["author"] = author
|
|
1570
1828
|
|
|
1571
1829
|
# Parse entry-level tags/categories
|
|
1572
|
-
|
|
1573
|
-
|
|
1574
|
-
|
|
1830
|
+
if include_tags:
|
|
1831
|
+
tags = _parse_tags(item, feed_type, atom_ns)
|
|
1832
|
+
if tags:
|
|
1833
|
+
entry["tags"] = tags
|
|
1575
1834
|
|
|
1576
1835
|
return entry
|
|
1577
1836
|
|
|
@@ -1902,6 +2161,7 @@ def _slow_dateparser(value: str) -> Optional[datetime.datetime]:
|
|
|
1902
2161
|
return None
|
|
1903
2162
|
|
|
1904
2163
|
|
|
2164
|
+
@lru_cache(maxsize=8192)
|
|
1905
2165
|
def _parse_date(date_str: str) -> Optional[str]:
|
|
1906
2166
|
"""Parse date string and return as an ISO 8601 formatted UTC string.
|
|
1907
2167
|
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.1
|
|
2
2
|
Name: fastfeedparser
|
|
3
|
-
Version: 0.5.
|
|
3
|
+
Version: 0.5.6
|
|
4
4
|
Summary: High performance RSS, Atom, JSON and RDF feed parser in Python
|
|
5
5
|
Home-page: https://github.com/kagisearch/fastfeedparser
|
|
6
6
|
Author: Vladimir Prelovac
|
|
@@ -25,6 +25,7 @@ Requires-Python: >=3.7
|
|
|
25
25
|
Description-Content-Type: text/markdown
|
|
26
26
|
Provides-Extra: dateparser
|
|
27
27
|
Provides-Extra: brotli
|
|
28
|
+
Provides-Extra: orjson
|
|
28
29
|
Provides-Extra: full
|
|
29
30
|
License-File: LICENSE
|
|
30
31
|
|
|
@@ -146,7 +147,7 @@ Speedup: 50.1x
|
|
|
146
147
|
|
|
147
148
|
### Main Functions
|
|
148
149
|
|
|
149
|
-
- `parse(source)`: Parse feed from a source
|
|
150
|
+
- `parse(source, *, include_content=True, include_tags=True, include_media=True, include_enclosures=True)`: Parse feed from a URL/XML/JSON source, with optional field extraction toggles for faster parsing.
|
|
150
151
|
|
|
151
152
|
|
|
152
153
|
### Feed Object Structure
|
|
@@ -56,3 +56,49 @@ def test_meta_refresh_none_when_missing():
|
|
|
56
56
|
def test_meta_refresh_none_when_same_url():
|
|
57
57
|
html = '<html><head><meta http-equiv="refresh" content="0; url=https://example.com/"></head></html>'
|
|
58
58
|
assert _extract_meta_refresh_url(html, "https://example.com/") is None
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
def test_parse_optional_field_flags():
|
|
62
|
+
xml = """<?xml version="1.0" encoding="utf-8"?>
|
|
63
|
+
<rss version="2.0"
|
|
64
|
+
xmlns:content="http://purl.org/rss/1.0/modules/content/"
|
|
65
|
+
xmlns:media="http://search.yahoo.com/mrss/">
|
|
66
|
+
<channel>
|
|
67
|
+
<title>Example Feed</title>
|
|
68
|
+
<category>feed-tag</category>
|
|
69
|
+
<item>
|
|
70
|
+
<title>Example Item</title>
|
|
71
|
+
<link>https://example.com/item</link>
|
|
72
|
+
<pubDate>Mon, 01 Jan 2024 00:00:00 GMT</pubDate>
|
|
73
|
+
<description>Summary</description>
|
|
74
|
+
<content:encoded><![CDATA[<p>Body</p>]]></content:encoded>
|
|
75
|
+
<category>entry-tag</category>
|
|
76
|
+
<enclosure url="https://example.com/audio.mp3" type="audio/mpeg" length="123" />
|
|
77
|
+
<media:content url="https://example.com/image.jpg" type="image/jpeg" />
|
|
78
|
+
</item>
|
|
79
|
+
</channel>
|
|
80
|
+
</rss>
|
|
81
|
+
"""
|
|
82
|
+
full = parse(xml)
|
|
83
|
+
entry_full = full.entries[0]
|
|
84
|
+
assert "content" in entry_full
|
|
85
|
+
assert "tags" in entry_full
|
|
86
|
+
assert "enclosures" in entry_full
|
|
87
|
+
assert "media_content" in entry_full
|
|
88
|
+
assert "tags" in full.feed
|
|
89
|
+
|
|
90
|
+
trimmed = parse(
|
|
91
|
+
xml,
|
|
92
|
+
include_content=False,
|
|
93
|
+
include_tags=False,
|
|
94
|
+
include_media=False,
|
|
95
|
+
include_enclosures=False,
|
|
96
|
+
)
|
|
97
|
+
entry_trimmed = trimmed.entries[0]
|
|
98
|
+
assert "content" not in entry_trimmed
|
|
99
|
+
assert "tags" not in entry_trimmed
|
|
100
|
+
assert "enclosures" not in entry_trimmed
|
|
101
|
+
assert "media_content" not in entry_trimmed
|
|
102
|
+
assert "tags" not in trimmed.feed
|
|
103
|
+
assert entry_trimmed["title"] == "Example Item"
|
|
104
|
+
assert entry_trimmed["link"] == "https://example.com/item"
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{fastfeedparser-0.5.5 → fastfeedparser-0.5.6}/src/fastfeedparser.egg-info/dependency_links.txt
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|