fastfeedparser 0.5.5__tar.gz → 0.5.6__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.1
2
2
  Name: fastfeedparser
3
- Version: 0.5.5
3
+ Version: 0.5.6
4
4
  Summary: High performance RSS, Atom, JSON and RDF feed parser in Python
5
5
  Home-page: https://github.com/kagisearch/fastfeedparser
6
6
  Author: Vladimir Prelovac
@@ -25,6 +25,7 @@ Requires-Python: >=3.7
25
25
  Description-Content-Type: text/markdown
26
26
  Provides-Extra: dateparser
27
27
  Provides-Extra: brotli
28
+ Provides-Extra: orjson
28
29
  Provides-Extra: full
29
30
  License-File: LICENSE
30
31
 
@@ -146,7 +147,7 @@ Speedup: 50.1x
146
147
 
147
148
  ### Main Functions
148
149
 
149
- - `parse(source)`: Parse feed from a source that can be URL or a string
150
+ - `parse(source, *, include_content=True, include_tags=True, include_media=True, include_enclosures=True)`: Parse feed from a URL/XML/JSON source, with optional field extraction toggles for faster parsing.
150
151
 
151
152
 
152
153
  ### Feed Object Structure
@@ -116,7 +116,7 @@ Speedup: 50.1x
116
116
 
117
117
  ### Main Functions
118
118
 
119
- - `parse(source)`: Parse feed from a source that can be URL or a string
119
+ - `parse(source, *, include_content=True, include_tags=True, include_media=True, include_enclosures=True)`: Parse feed from a URL/XML/JSON source, with optional field extraction toggles for faster parsing.
120
120
 
121
121
 
122
122
  ### Feed Object Structure
@@ -1,6 +1,6 @@
1
1
  [metadata]
2
2
  name = fastfeedparser
3
- version = 0.5.5
3
+ version = 0.5.6
4
4
  author = Vladimir Prelovac
5
5
  author_email = vlad@kagi.com
6
6
  description = High performance RSS, Atom, JSON and RDF feed parser in Python
@@ -40,9 +40,12 @@ dateparser =
40
40
  dateparser
41
41
  brotli =
42
42
  brotli
43
+ orjson =
44
+ orjson
43
45
  full =
44
46
  dateparser
45
47
  brotli
48
+ orjson
46
49
 
47
50
  [options.packages.find]
48
51
  where = src
@@ -15,7 +15,14 @@ try:
15
15
  HAS_BROTLI = True
16
16
  except ImportError:
17
17
  HAS_BROTLI = False
18
- from typing import Any, Callable, Optional, TYPE_CHECKING, Literal
18
+
19
+ try:
20
+ import orjson
21
+
22
+ _json_loads = orjson.loads
23
+ except ImportError:
24
+ _json_loads = json.loads
25
+ from typing import Any, Callable, Optional, Protocol, TYPE_CHECKING, Literal
19
26
  from urllib.parse import urljoin
20
27
  from urllib.request import (
21
28
  HTTPErrorProcessor,
@@ -81,6 +88,44 @@ _MONTHS_RFC822: dict[str, int] = {
81
88
  "dec": 12,
82
89
  }
83
90
 
91
+ _XML_NS = "{http://www.w3.org/XML/1998/namespace}"
92
+ _XML_LANG_ATTR = _XML_NS + "lang"
93
+ _XML_BASE_ATTR = _XML_NS + "base"
94
+ _RDF_ABOUT_ATTR = "{http://www.w3.org/1999/02/22-rdf-syntax-ns#}about"
95
+ _RSS_CONTENT_ENCODED_TAG = "{http://purl.org/rss/1.0/modules/content/}encoded"
96
+ _DC_SUBJECT_TAG = "{http://purl.org/dc/elements/1.1/}subject"
97
+ _MEDIA_CONTENT_TAG = "{http://search.yahoo.com/mrss/}content"
98
+ _MEDIA_THUMBNAIL_TAG = "{http://search.yahoo.com/mrss/}thumbnail"
99
+ _MEDIA_TITLE_TAG = "{http://search.yahoo.com/mrss/}title"
100
+ _MEDIA_TEXT_TAG = "{http://search.yahoo.com/mrss/}text"
101
+ _MEDIA_DESCRIPTION_TAG = "{http://search.yahoo.com/mrss/}description"
102
+ _MEDIA_CREDIT_TAG = "{http://search.yahoo.com/mrss/}credit"
103
+
104
+
105
+ @lru_cache(maxsize=4)
106
+ def _atom_ns_tags(atom_ns: str) -> dict[str, str]:
107
+ """Pre-compute namespace-prefixed tag strings once per unique namespace.
108
+
109
+ Avoids thousands of redundant f-string / concatenation operations when
110
+ parsing feeds with many entries.
111
+ """
112
+ ns = f"{{{atom_ns}}}"
113
+ is_atom_03 = atom_ns == "http://purl.org/atom/ns#"
114
+ return {
115
+ "ns": ns,
116
+ "id": ns + "id",
117
+ "title": ns + "title",
118
+ "summary": ns + "summary",
119
+ "link": ns + "link",
120
+ "content": ns + "content",
121
+ "author_name": ns + "author/" + ns + "name",
122
+ "category": ns + "category",
123
+ "published": ns + ("issued" if is_atom_03 else "published"),
124
+ "updated": ns + ("modified" if is_atom_03 else "updated"),
125
+ "pub_fallback": ns + ("published" if is_atom_03 else "issued"),
126
+ "upd_fallback": ns + ("updated" if is_atom_03 else "modified"),
127
+ }
128
+
84
129
 
85
130
  class FastFeedParserDict(dict):
86
131
  """A dictionary that allows access to its keys as attributes."""
@@ -154,11 +199,18 @@ def _clean_feed_bytes(content: bytes) -> bytes:
154
199
  b"<?xml-stylesheet",
155
200
  )
156
201
 
157
- lines = content.splitlines()
158
- for i, line in enumerate(lines):
159
- line_stripped = line.strip().lower()
160
- if any(line_stripped.startswith(pattern) for pattern in xml_start_patterns):
161
- return b"\n".join(lines[i:])
202
+ # Search for XML start patterns without splitting entire content into lines.
203
+ # For large feeds (multi-MB), splitlines() creates thousands of byte string
204
+ # objects; find() scans in-place with zero allocations.
205
+ search_limit = min(len(content), 8192)
206
+ search_chunk = content[:search_limit].lower()
207
+ earliest = -1
208
+ for pattern in xml_start_patterns:
209
+ idx = search_chunk.find(pattern)
210
+ if idx != -1 and (earliest == -1 or idx < earliest):
211
+ earliest = idx
212
+ if earliest != -1:
213
+ return content[earliest:]
162
214
 
163
215
  if b"<script>" in preview_lower or b"<body>" in preview_lower:
164
216
  raise ValueError("Content appears to be HTML, not a valid RSS/Atom feed")
@@ -167,21 +219,30 @@ def _clean_feed_bytes(content: bytes) -> bytes:
167
219
 
168
220
 
169
221
  def _fix_malformed_xml_bytes(content: bytes, actual_encoding: str = "utf-8") -> bytes:
222
+ # XML declarations and encoding definitions live at the top of the file.
223
+ # Run declaration-fixing regexes only on the first 2 KB to avoid scanning
224
+ # multi-megabyte payloads with patterns that can only match the header.
225
+ header = content[:2048]
226
+ tail = content[2048:]
227
+
170
228
  # Fix double XML declarations like "<?xml?xml version="1.0"?>"
171
- content = _RE_DOUBLE_XML_DECL_BYTES.sub(b"<?xml ", content)
229
+ header = _RE_DOUBLE_XML_DECL_BYTES.sub(b"<?xml ", header)
172
230
 
173
231
  # Fix double closing ?> in XML declaration like "??>>"
174
- content = _RE_DOUBLE_CLOSE_BYTES.sub(b"?>", content)
175
-
176
- # Fix malformed attribute syntax like rss:version=2.0 (missing quotes)
177
- content = _RE_UNQUOTED_ATTR_BYTES.sub(rb'\1="\2"', content)
232
+ header = _RE_DOUBLE_CLOSE_BYTES.sub(b"?>", header)
178
233
 
179
234
  # Update encoding in XML declaration to match actual encoding when a feed was transcoded.
180
235
  if actual_encoding.lower() != "utf-16":
181
236
  replacement = (
182
237
  rb"\1" + actual_encoding.encode("ascii", errors="replace") + rb"\3"
183
238
  )
184
- content = _RE_UTF16_ENCODING_BYTES.sub(replacement, content)
239
+ header = _RE_UTF16_ENCODING_BYTES.sub(replacement, header)
240
+
241
+ # Reassemble before running body-wide fixes
242
+ content = header + tail
243
+
244
+ # Fix malformed attribute syntax like rss:version=2.0 (missing quotes)
245
+ content = _RE_UNQUOTED_ATTR_BYTES.sub(rb'\1="\2"', content)
185
246
 
186
247
  # Fix unclosed link tags - common in Atom feeds
187
248
  content = _RE_UNCLOSED_LINK_BYTES.sub(rb"<link\1/>", content)
@@ -225,7 +286,13 @@ def _prepare_xml_bytes(xml_content: str | bytes) -> bytes:
225
286
  return _prepare_xml_bytes(xml_content.encode("utf-8", errors="replace"))
226
287
 
227
288
 
228
- def _parse_json_feed(json_data: dict) -> FastFeedParserDict:
289
+ def _parse_json_feed(
290
+ json_data: dict,
291
+ *,
292
+ include_content: bool = True,
293
+ include_tags: bool = True,
294
+ include_enclosures: bool = True,
295
+ ) -> FastFeedParserDict:
229
296
  """Parse a JSON Feed and convert to FastFeedParserDict format.
230
297
 
231
298
  JSON Feed spec: https://jsonfeed.org/
@@ -283,10 +350,12 @@ def _parse_json_feed(json_data: dict) -> FastFeedParserDict:
283
350
  summary = item.get("summary", "")
284
351
 
285
352
  if content_html:
286
- entry["content"] = [{"type": "text/html", "value": content_html}]
353
+ if include_content:
354
+ entry["content"] = [{"type": "text/html", "value": content_html}]
287
355
  entry["description"] = summary
288
356
  elif content_text:
289
- entry["content"] = [{"type": "text/plain", "value": content_text}]
357
+ if include_content:
358
+ entry["content"] = [{"type": "text/plain", "value": content_text}]
290
359
  entry["description"] = summary or content_text[:512]
291
360
  else:
292
361
  entry["description"] = summary
@@ -319,14 +388,14 @@ def _parse_json_feed(json_data: dict) -> FastFeedParserDict:
319
388
 
320
389
  # Add tags
321
390
  tags = item.get("tags")
322
- if tags:
391
+ if include_tags and tags:
323
392
  entry["tags"] = [
324
393
  {"term": tag, "scheme": None, "label": None} for tag in tags
325
394
  ]
326
395
 
327
396
  # Add attachments as enclosures
328
397
  attachments = item.get("attachments")
329
- if attachments:
398
+ if include_enclosures and attachments:
330
399
  enclosures = []
331
400
  for attachment in attachments:
332
401
  url = attachment.get("url", "")
@@ -389,19 +458,23 @@ def _fetch_url_content(url: str) -> str | bytes:
389
458
  return content.decode(content_charset) if content_charset else content
390
459
 
391
460
 
392
- def _maybe_parse_json_feed(content: str | bytes) -> FastFeedParserDict | None:
461
+ def _maybe_parse_json_feed(
462
+ content: str | bytes,
463
+ *,
464
+ include_content: bool = True,
465
+ include_tags: bool = True,
466
+ include_enclosures: bool = True,
467
+ ) -> FastFeedParserDict | None:
393
468
  if isinstance(content, bytes):
394
469
  if not content.lstrip().startswith(b"{"):
395
470
  return None
396
- json_str = content.decode("utf-8", errors="replace")
397
471
  else:
398
472
  if not content.lstrip().startswith("{"):
399
473
  return None
400
- json_str = content
401
474
 
402
475
  try:
403
- json_data = json.loads(json_str)
404
- except (json.JSONDecodeError, ValueError):
476
+ json_data = _json_loads(content)
477
+ except Exception:
405
478
  return None
406
479
 
407
480
  if not isinstance(json_data, dict):
@@ -409,10 +482,20 @@ def _maybe_parse_json_feed(content: str | bytes) -> FastFeedParserDict | None:
409
482
 
410
483
  version = json_data.get("version")
411
484
  if isinstance(version, str) and "jsonfeed.org" in version:
412
- return _parse_json_feed(json_data)
485
+ return _parse_json_feed(
486
+ json_data,
487
+ include_content=include_content,
488
+ include_tags=include_tags,
489
+ include_enclosures=include_enclosures,
490
+ )
413
491
 
414
492
  if isinstance(json_data.get("items"), list):
415
- return _parse_json_feed(json_data)
493
+ return _parse_json_feed(
494
+ json_data,
495
+ include_content=include_content,
496
+ include_tags=include_tags,
497
+ include_enclosures=include_enclosures,
498
+ )
416
499
 
417
500
  return None
418
501
 
@@ -667,9 +750,21 @@ def _detect_feed_structure(
667
750
  raise ValueError(f"Unknown feed type: {root.tag}")
668
751
 
669
752
 
670
- def _parse_content(xml_content: str | bytes) -> FastFeedParserDict:
753
+ def _parse_content(
754
+ xml_content: str | bytes,
755
+ *,
756
+ include_content: bool = True,
757
+ include_tags: bool = True,
758
+ include_media: bool = True,
759
+ include_enclosures: bool = True,
760
+ ) -> FastFeedParserDict:
671
761
  """Parse feed content (XML or JSON) that has already been fetched."""
672
- json_feed = _maybe_parse_json_feed(xml_content)
762
+ json_feed = _maybe_parse_json_feed(
763
+ xml_content,
764
+ include_content=include_content,
765
+ include_tags=include_tags,
766
+ include_enclosures=include_enclosures,
767
+ )
673
768
  if json_feed is not None:
674
769
  return json_feed
675
770
 
@@ -682,7 +777,9 @@ def _parse_content(xml_content: str | bytes) -> FastFeedParserDict:
682
777
  root, xml_content, root_tag_local
683
778
  )
684
779
 
685
- feed = _parse_feed_info(channel, feed_type, atom_namespace)
780
+ feed = _parse_feed_info(
781
+ channel, feed_type, atom_namespace, include_tags=include_tags
782
+ )
686
783
 
687
784
  # Detect once whether media namespace is used anywhere in the document
688
785
  has_media_ns = (
@@ -694,34 +791,41 @@ def _parse_content(xml_content: str | bytes) -> FastFeedParserDict:
694
791
  # Parse entries — resolve parser once per feed instead of per entry
695
792
  entries: list[FastFeedParserDict] = []
696
793
  feed["entries"] = entries
697
- atom_ns = atom_namespace or "http://www.w3.org/2005/Atom"
698
- if feed_type == "rss":
699
- for item in items:
700
- entry = _parse_rss_feed_entry_fast(item, atom_ns, has_media_ns)
701
- entry.setdefault("title", "")
702
- entry.setdefault("description", "")
703
- entries.append(entry)
704
- elif feed_type == "atom":
705
- for item in items:
706
- entry = _parse_atom_feed_entry_fast(item, atom_ns, has_media_ns)
707
- entry.setdefault("title", "")
708
- entry.setdefault("description", "")
709
- entries.append(entry)
710
- else:
711
- for item in items:
712
- entry = _parse_feed_entry(item, feed_type, atom_namespace, has_media_ns)
713
- entry["title"] = entry.get("title", "").strip()
714
- entry["description"] = entry.get("description", "").strip()
715
- entries.append(entry)
794
+ for item in items:
795
+ entry = _parse_feed_entry(
796
+ item,
797
+ feed_type,
798
+ atom_namespace,
799
+ has_media_ns,
800
+ include_content=include_content,
801
+ include_tags=include_tags,
802
+ include_media=include_media,
803
+ include_enclosures=include_enclosures,
804
+ )
805
+ # Ensure that titles and descriptions are always present
806
+ entry["title"] = entry.get("title", "").strip()
807
+ entry["description"] = entry.get("description", "").strip()
808
+ entries.append(entry)
716
809
 
717
810
  return feed
718
811
 
719
812
 
720
- def parse(source: str | bytes) -> FastFeedParserDict:
813
+ def parse(
814
+ source: str | bytes,
815
+ *,
816
+ include_content: bool = True,
817
+ include_tags: bool = True,
818
+ include_media: bool = True,
819
+ include_enclosures: bool = True,
820
+ ) -> FastFeedParserDict:
721
821
  """Parse a feed from a URL or XML content.
722
822
 
723
823
  Args:
724
824
  source: URL string or XML content string/bytes
825
+ include_content: Include per-entry content blobs and synthesized descriptions
826
+ include_tags: Include feed and entry tags/categories
827
+ include_media: Include media namespace content (media:content/media:thumbnail)
828
+ include_enclosures: Include RSS enclosures and JSON-feed attachments
725
829
 
726
830
  Returns:
727
831
  FastFeedParserDict containing parsed feed data
@@ -738,7 +842,13 @@ def parse(source: str | bytes) -> FastFeedParserDict:
738
842
  content = source
739
843
 
740
844
  try:
741
- return _parse_content(content)
845
+ return _parse_content(
846
+ content,
847
+ include_content=include_content,
848
+ include_tags=include_tags,
849
+ include_media=include_media,
850
+ include_enclosures=include_enclosures,
851
+ )
742
852
  except ValueError as e:
743
853
  if not is_url:
744
854
  raise
@@ -749,11 +859,21 @@ def parse(source: str | bytes) -> FastFeedParserDict:
749
859
  redirect_url = _extract_meta_refresh_url(content, source)
750
860
  if redirect_url is None:
751
861
  raise
752
- return parse(redirect_url)
862
+ return parse(
863
+ redirect_url,
864
+ include_content=include_content,
865
+ include_tags=include_tags,
866
+ include_media=include_media,
867
+ include_enclosures=include_enclosures,
868
+ )
753
869
 
754
870
 
755
871
  def _parse_feed_info(
756
- channel: _Element, feed_type: _FeedType, atom_namespace: Optional[str] = None
872
+ channel: _Element,
873
+ feed_type: _FeedType,
874
+ atom_namespace: Optional[str] = None,
875
+ *,
876
+ include_tags: bool = True,
757
877
  ) -> FastFeedParserDict:
758
878
  # Use dynamic atom namespace or fallback to default
759
879
  atom_ns = atom_namespace or "http://www.w3.org/2005/Atom"
@@ -824,8 +944,8 @@ def _parse_feed_info(
824
944
  if value:
825
945
  feed[field[0]] = value
826
946
 
827
- feed_lang = channel.get("{http://www.w3.org/XML/1998/namespace}lang")
828
- feed_base = channel.get("{http://www.w3.org/XML/1998/namespace}base")
947
+ feed_lang = channel.get(_XML_LANG_ATTR)
948
+ feed_base = channel.get(_XML_BASE_ATTR)
829
949
  feed["language"] = feed_lang
830
950
 
831
951
  # Add title_detail and subtitle_detail
@@ -896,9 +1016,10 @@ def _parse_feed_info(
896
1016
  feed["author"] = managing_editor
897
1017
 
898
1018
  # Parse feed-level tags/categories
899
- tags = _parse_tags(channel, feed_type, atom_ns)
900
- if tags:
901
- feed["tags"] = tags
1019
+ if include_tags:
1020
+ tags = _parse_tags(channel, feed_type, atom_ns)
1021
+ if tags:
1022
+ feed["tags"] = tags
902
1023
 
903
1024
  return FastFeedParserDict(feed=feed)
904
1025
 
@@ -917,14 +1038,14 @@ def _parse_tags(
917
1038
  {"term": term, "scheme": cat.get("domain"), "label": None}
918
1039
  )
919
1040
  # RSS might also use <dc:subject>
920
- for subject in element.findall("{http://purl.org/dc/elements/1.1/}subject"):
1041
+ for subject in element.findall(_DC_SUBJECT_TAG):
921
1042
  term = subject.text.strip() if subject.text else None
922
1043
  if term:
923
1044
  tags_list.append({"term": term, "scheme": None, "label": None})
924
1045
  elif feed_type == "atom":
925
1046
  # Atom uses <category> elements with attributes
926
1047
  atom_ns = atom_namespace or "http://www.w3.org/2005/Atom"
927
- for cat in element.findall(f"{{{atom_ns}}}category"):
1048
+ for cat in element.findall(_atom_ns_tags(atom_ns)["category"]):
928
1049
  term = cat.get("term")
929
1050
  if term:
930
1051
  tags_list.append(
@@ -936,7 +1057,7 @@ def _parse_tags(
936
1057
  )
937
1058
  elif feed_type == "rdf":
938
1059
  # RDF uses <dc:subject> or <taxo:topic>
939
- for subject in element.findall("{http://purl.org/dc/elements/1.1/}subject"):
1060
+ for subject in element.findall(_DC_SUBJECT_TAG):
940
1061
  term = subject.text.strip() if subject.text else None
941
1062
  if term:
942
1063
  tags_list.append({"term": term, "scheme": None, "label": None})
@@ -972,12 +1093,16 @@ def _coerce_int_fields(mapping: dict[str, Any], fields: tuple[str, ...]) -> None
972
1093
  mapping.pop(field, None)
973
1094
 
974
1095
 
975
- def _populate_entry_links(
976
- entry: FastFeedParserDict, item: _Element, atom_ns: str
1096
+ def _populate_entry_links_from_elements(
1097
+ entry: FastFeedParserDict,
1098
+ atom_links: list[_Element],
1099
+ *,
1100
+ guid_text: Optional[str] = None,
1101
+ guid_is_permalink: bool = False,
977
1102
  ) -> None:
978
1103
  entry_links: list[dict[str, Optional[str]]] = []
979
1104
  alternate_link: Optional[dict[str, Optional[str]]] = None
980
- for link in item.findall(f"{{{atom_ns}}}link"):
1105
+ for link in atom_links:
981
1106
  rel = link.get("rel")
982
1107
  href = link.get("href") or link.get("link")
983
1108
  if not href:
@@ -993,8 +1118,6 @@ def _populate_entry_links(
993
1118
  elif rel not in {"edit", "self"}:
994
1119
  entry_links.append(link_dict)
995
1120
 
996
- guid = item.find("guid")
997
- guid_text = guid.text.strip() if guid is not None and guid.text else None
998
1121
  is_guid_url = guid_text is not None and guid_text.startswith(
999
1122
  ("http://", "https://")
1000
1123
  )
@@ -1008,44 +1131,55 @@ def _populate_entry_links(
1008
1131
  elif alternate_link:
1009
1132
  entry["link"] = alternate_link["href"]
1010
1133
  entry_links.insert(0, alternate_link)
1011
- elif (
1012
- ("link" not in entry)
1013
- and (guid is not None)
1014
- and guid.get("isPermaLink") == "true"
1015
- ):
1134
+ elif ("link" not in entry) and guid_is_permalink:
1016
1135
  entry["link"] = guid_text
1017
1136
 
1018
1137
  entry["links"] = entry_links
1019
1138
 
1020
1139
 
1021
- def _populate_entry_content(
1022
- entry: FastFeedParserDict,
1023
- item: _Element,
1024
- feed_type: _FeedType,
1025
- atom_ns: str,
1026
- rss_text_by_full: Optional[dict[str, Optional[str]]] = None,
1140
+ def _populate_entry_links(
1141
+ entry: FastFeedParserDict, item: _Element, atom_ns: str
1027
1142
  ) -> None:
1028
- content_el = None
1029
- if feed_type == "rss":
1030
- # Fast path: check pre-built text map before doing tree searches
1031
- if rss_text_by_full is not None:
1032
- content_encoded_text = rss_text_by_full.get(
1033
- "{http://purl.org/rss/1.0/modules/content/}encoded"
1034
- )
1035
- if content_encoded_text is not None:
1036
- # We have content:encoded text — still need the element for type/lang/base attrs
1037
- content_el = item.find("{http://purl.org/rss/1.0/modules/content/}encoded")
1038
- else:
1039
- content_text = rss_text_by_full.get("content")
1040
- if content_text is not None:
1041
- content_el = item.find("content")
1143
+ tags = _atom_ns_tags(atom_ns)
1144
+ guid = item.find("guid")
1145
+ guid_text = guid.text.strip() if guid is not None and guid.text else None
1146
+ _populate_entry_links_from_elements(
1147
+ entry,
1148
+ item.findall(tags["link"]),
1149
+ guid_text=guid_text,
1150
+ guid_is_permalink=guid is not None and guid.get("isPermaLink") == "true",
1151
+ )
1152
+
1153
+
1154
+ def _synthesize_entry_description(entry: FastFeedParserDict) -> None:
1155
+ if "description" in entry or "content" not in entry:
1156
+ return
1157
+
1158
+ content_value = entry["content"][0]["value"]
1159
+ if content_value:
1160
+ if "<" in content_value and ">" in content_value:
1161
+ content_value = _RE_HTML_TAGS.sub(" ", content_value[:2048])
1162
+ if "&" in content_value:
1163
+ content_value = _html_mod.unescape(content_value)
1164
+ if (
1165
+ " " in content_value
1166
+ or "\n" in content_value
1167
+ or "\t" in content_value
1168
+ or "\r" in content_value
1169
+ ):
1170
+ content_value = _RE_WHITESPACE.sub(" ", content_value).strip()
1042
1171
  else:
1043
- content_el = item.find("{http://purl.org/rss/1.0/modules/content/}encoded")
1044
- if content_el is None:
1045
- content_el = item.find("content")
1046
- elif feed_type == "atom":
1047
- content_el = item.find(f"{{{atom_ns}}}content")
1172
+ content_value = content_value.strip()
1173
+ entry["description"] = content_value[:512]
1048
1174
 
1175
+
1176
+ def _populate_entry_content_preparsed(
1177
+ entry: FastFeedParserDict,
1178
+ item: _Element,
1179
+ *,
1180
+ content_el: Optional[_Element],
1181
+ rss_description_text: Optional[str],
1182
+ ) -> None:
1049
1183
  if content_el is not None:
1050
1184
  content_type = content_el.get("type", "text/html")
1051
1185
  if content_type in {"xhtml", "application/xhtml+xml"}:
@@ -1055,48 +1189,51 @@ def _populate_entry_content(
1055
1189
  entry["content"] = [
1056
1190
  {
1057
1191
  "type": content_type,
1058
- "language": content_el.get(
1059
- "{http://www.w3.org/XML/1998/namespace}lang"
1060
- ),
1061
- "base": content_el.get("{http://www.w3.org/XML/1998/namespace}base"),
1192
+ "language": content_el.get(_XML_LANG_ATTR),
1193
+ "base": content_el.get(_XML_BASE_ATTR),
1062
1194
  "value": content_value,
1063
1195
  }
1064
1196
  ]
1197
+ elif rss_description_text:
1198
+ entry["content"] = [
1199
+ {
1200
+ "type": "text/html",
1201
+ "language": item.get(_XML_LANG_ATTR),
1202
+ "base": item.get(_XML_BASE_ATTR),
1203
+ "value": rss_description_text,
1204
+ }
1205
+ ]
1206
+
1207
+ _synthesize_entry_description(entry)
1208
+
1065
1209
 
1066
- if "content" not in entry:
1210
+ def _populate_entry_content(
1211
+ entry: FastFeedParserDict, item: _Element, feed_type: _FeedType, atom_ns: str
1212
+ ) -> None:
1213
+ content_el: Optional[_Element] = None
1214
+ rss_description_text: Optional[str] = None
1215
+ if feed_type == "rss":
1216
+ content_el = item.find(_RSS_CONTENT_ENCODED_TAG)
1217
+ if content_el is None:
1218
+ content_el = item.find("content")
1067
1219
  description = item.find("description")
1068
- if description is not None and description.text:
1069
- entry["content"] = [
1070
- {
1071
- "type": "text/html",
1072
- "language": item.get("{http://www.w3.org/XML/1998/namespace}lang"),
1073
- "base": item.get("{http://www.w3.org/XML/1998/namespace}base"),
1074
- "value": description.text,
1075
- }
1076
- ]
1220
+ if description is not None:
1221
+ rss_description_text = description.text
1222
+ elif feed_type == "atom":
1223
+ content_el = item.find(_atom_ns_tags(atom_ns)["content"])
1077
1224
 
1078
- if "description" not in entry and "content" in entry:
1079
- content_value = entry["content"][0]["value"]
1080
- if content_value:
1081
- if "<" in content_value:
1082
- content_value = _RE_HTML_TAGS.sub(" ", content_value[:1024])
1083
- content_value = _html_mod.unescape(content_value)
1084
- if (
1085
- " " in content_value
1086
- or "\n" in content_value
1087
- or "\t" in content_value
1088
- or "\r" in content_value
1089
- ):
1090
- content_value = _RE_WHITESPACE.sub(" ", content_value).strip()
1091
- else:
1092
- content_value = content_value.strip()
1093
- entry["description"] = content_value[:512]
1225
+ _populate_entry_content_preparsed(
1226
+ entry,
1227
+ item,
1228
+ content_el=content_el,
1229
+ rss_description_text=rss_description_text,
1230
+ )
1094
1231
 
1095
1232
 
1096
1233
  def _parse_media_content(item: _Element) -> list[dict[str, Any]] | None:
1097
1234
  media_contents: list[dict[str, Any]] = []
1098
1235
 
1099
- for media in item.findall(".//{http://search.yahoo.com/mrss/}content"):
1236
+ for media in item.findall(f".//{_MEDIA_CONTENT_TAG}"):
1100
1237
  media_item: dict[str, str | int | None] = {
1101
1238
  "url": media.get("url"),
1102
1239
  "type": media.get("type"),
@@ -1106,32 +1243,32 @@ def _parse_media_content(item: _Element) -> list[dict[str, Any]] | None:
1106
1243
  }
1107
1244
  _coerce_int_fields(media_item, ("width", "height"))
1108
1245
 
1109
- title = media.find("{http://search.yahoo.com/mrss/}title")
1246
+ title = media.find(_MEDIA_TITLE_TAG)
1110
1247
  if title is not None and title.text:
1111
1248
  media_item["title"] = title.text.strip()
1112
1249
 
1113
- text = media.find("{http://search.yahoo.com/mrss/}text")
1250
+ text = media.find(_MEDIA_TEXT_TAG)
1114
1251
  if text is not None and text.text:
1115
1252
  media_item["text"] = text.text.strip()
1116
1253
 
1117
- desc = media.find("{http://search.yahoo.com/mrss/}description")
1254
+ desc = media.find(_MEDIA_DESCRIPTION_TAG)
1118
1255
  if desc is None:
1119
1256
  parent = media.getparent()
1120
1257
  if parent is not None:
1121
- desc = parent.find("{http://search.yahoo.com/mrss/}description")
1258
+ desc = parent.find(_MEDIA_DESCRIPTION_TAG)
1122
1259
  if desc is not None and desc.text:
1123
1260
  media_item["description"] = desc.text.strip()
1124
1261
 
1125
- credit = media.find("{http://search.yahoo.com/mrss/}credit")
1262
+ credit = media.find(_MEDIA_CREDIT_TAG)
1126
1263
  if credit is None:
1127
1264
  parent = media.getparent()
1128
1265
  if parent is not None:
1129
- credit = parent.find("{http://search.yahoo.com/mrss/}credit")
1266
+ credit = parent.find(_MEDIA_CREDIT_TAG)
1130
1267
  if credit is not None and credit.text:
1131
1268
  media_item["credit"] = credit.text.strip()
1132
1269
  media_item["credit_scheme"] = credit.get("scheme")
1133
1270
 
1134
- thumbnail = media.find("{http://search.yahoo.com/mrss/}thumbnail")
1271
+ thumbnail = media.find(_MEDIA_THUMBNAIL_TAG)
1135
1272
  if thumbnail is not None:
1136
1273
  media_item["thumbnail_url"] = thumbnail.get("url")
1137
1274
 
@@ -1140,9 +1277,9 @@ def _parse_media_content(item: _Element) -> list[dict[str, Any]] | None:
1140
1277
  media_contents.append(cleaned)
1141
1278
 
1142
1279
  if not media_contents:
1143
- for thumbnail in item.findall(".//{http://search.yahoo.com/mrss/}thumbnail"):
1280
+ for thumbnail in item.findall(f".//{_MEDIA_THUMBNAIL_TAG}"):
1144
1281
  parent = thumbnail.getparent()
1145
- if parent is None or parent.tag == "{http://search.yahoo.com/mrss/}content":
1282
+ if parent is None or parent.tag == _MEDIA_CONTENT_TAG:
1146
1283
  continue
1147
1284
  thumb_item: dict[str, str | int | None] = {
1148
1285
  "url": thumbnail.get("url"),
@@ -1161,47 +1298,26 @@ def _parse_media_content(item: _Element) -> list[dict[str, Any]] | None:
1161
1298
  def _parse_enclosures(item: _Element) -> list[dict[str, Any]] | None:
1162
1299
  enclosures: list[dict[str, Any]] = []
1163
1300
  for enclosure in item.findall("enclosure"):
1164
- enc_item: dict[str, str | int | None] = {
1165
- "url": enclosure.get("url"),
1166
- "type": enclosure.get("type"),
1167
- "length": enclosure.get("length"),
1168
- }
1169
- length = enc_item.get("length")
1170
- if length:
1171
- try:
1172
- enc_item["length"] = int(length)
1173
- except (ValueError, TypeError):
1174
- enc_item.pop("length", None)
1175
-
1176
- cleaned = _drop_none_values(enc_item)
1301
+ cleaned = _parse_enclosure_element(enclosure)
1177
1302
  if cleaned.get("url"):
1178
1303
  enclosures.append(cleaned)
1179
1304
 
1180
1305
  return enclosures or None
1181
1306
 
1182
1307
 
1183
- def _build_rss_item_text_maps(
1184
- item: _Element,
1185
- ) -> tuple[dict[str, Optional[str]], dict[str, Optional[str]]]:
1186
- by_local: dict[str, Optional[str]] = {}
1187
- by_full: dict[str, Optional[str]] = {}
1188
- for child in item:
1189
- tag = child.tag
1190
- if not isinstance(tag, str):
1191
- continue
1192
- text_value = child.text or None
1193
- if tag not in by_full:
1194
- by_full[tag] = text_value
1195
- # Fast path: ~80% of RSS tags have no namespace or colon prefix
1196
- if "{" in tag:
1197
- local = tag.rsplit("}", 1)[1].lower()
1198
- elif ":" in tag:
1199
- local = tag.split(":", 1)[1].lower()
1200
- else:
1201
- local = tag.lower()
1202
- if local not in by_local:
1203
- by_local[local] = text_value
1204
- return by_local, by_full
1308
+ def _parse_enclosure_element(enclosure: _Element) -> dict[str, Any]:
1309
+ enc_item: dict[str, str | int | None] = {
1310
+ "url": enclosure.get("url"),
1311
+ "type": enclosure.get("type"),
1312
+ "length": enclosure.get("length"),
1313
+ }
1314
+ length = enc_item.get("length")
1315
+ if length:
1316
+ try:
1317
+ enc_item["length"] = int(length)
1318
+ except (ValueError, TypeError):
1319
+ enc_item.pop("length", None)
1320
+ return _drop_none_values(enc_item)
1205
1321
 
1206
1322
 
1207
1323
  def _first_non_empty(
@@ -1218,13 +1334,78 @@ def _parse_rss_feed_entry_fast(
1218
1334
  item: _Element,
1219
1335
  atom_ns: str,
1220
1336
  has_media_ns: bool = True,
1337
+ *,
1338
+ include_content: bool = True,
1339
+ include_tags: bool = True,
1340
+ include_media: bool = True,
1341
+ include_enclosures: bool = True,
1221
1342
  ) -> FastFeedParserDict:
1222
- text_by_local, text_by_full = _build_rss_item_text_maps(item)
1343
+ atom_tags = _atom_ns_tags(atom_ns)
1344
+ text_by_local: dict[str, Optional[str]] = {}
1345
+ text_by_full: dict[str, Optional[str]] = {}
1346
+ atom_links: list[_Element] = []
1347
+ guid_element: Optional[_Element] = None
1348
+ encoded_content_el: Optional[_Element] = None
1349
+ raw_content_el: Optional[_Element] = None
1350
+ rss_description_text: Optional[str] = None
1351
+ tag_categories: list[dict[str, str | None]] = []
1352
+ tag_subjects: list[dict[str, str | None]] = []
1353
+ enclosures: list[dict[str, Any]] = []
1354
+
1355
+ for child in item:
1356
+ tag = child.tag
1357
+ if not isinstance(tag, str):
1358
+ continue
1359
+
1360
+ text_value = child.text or None
1361
+ if tag not in text_by_full:
1362
+ text_by_full[tag] = text_value
1363
+
1364
+ if "{" in tag:
1365
+ local = tag.rsplit("}", 1)[1].lower()
1366
+ elif ":" in tag:
1367
+ local = tag.split(":", 1)[1].lower()
1368
+ else:
1369
+ local = tag.lower()
1370
+ if local not in text_by_local:
1371
+ text_by_local[local] = text_value
1372
+
1373
+ if tag == atom_tags["link"]:
1374
+ atom_links.append(child)
1375
+ elif tag == "guid":
1376
+ if guid_element is None:
1377
+ guid_element = child
1378
+ elif tag == _RSS_CONTENT_ENCODED_TAG:
1379
+ if encoded_content_el is None:
1380
+ encoded_content_el = child
1381
+ elif tag == "content":
1382
+ if raw_content_el is None:
1383
+ raw_content_el = child
1384
+ elif tag == "description":
1385
+ if rss_description_text is None:
1386
+ rss_description_text = text_value
1387
+
1388
+ if include_enclosures and tag == "enclosure":
1389
+ cleaned = _parse_enclosure_element(child)
1390
+ if cleaned.get("url"):
1391
+ enclosures.append(cleaned)
1392
+
1393
+ if include_tags:
1394
+ if local == "category":
1395
+ term = text_value.strip() if text_value else None
1396
+ if term:
1397
+ tag_categories.append(
1398
+ {"term": term, "scheme": child.get("domain"), "label": None}
1399
+ )
1400
+ elif tag == _DC_SUBJECT_TAG:
1401
+ term = text_value.strip() if text_value else None
1402
+ if term:
1403
+ tag_subjects.append({"term": term, "scheme": None, "label": None})
1223
1404
 
1224
1405
  entry = FastFeedParserDict()
1225
- atom_id = text_by_full.get(f"{{{atom_ns}}}id")
1406
+ atom_id = text_by_full.get(atom_tags["id"])
1226
1407
  rss_guid = text_by_local.get("guid")
1227
- rdf_about = item.get("{http://www.w3.org/1999/02/22-rdf-syntax-ns#}about")
1408
+ rdf_about = item.get(_RDF_ABOUT_ATTR)
1228
1409
  entry_id: Optional[str] = atom_id or rss_guid or rdf_about
1229
1410
  if entry_id:
1230
1411
  entry["id"] = entry_id.strip()
@@ -1269,13 +1450,20 @@ def _parse_rss_feed_entry_fast(
1269
1450
  if "updated" in entry and "published" not in entry:
1270
1451
  entry["published"] = entry["updated"]
1271
1452
 
1272
- # Inline link population for RSS (avoids redundant findall/find for 98.8% of entries)
1273
- atom_links = item.findall(f"{{{atom_ns}}}link")
1274
1453
  if atom_links:
1275
- # Has atom:link elements - use full logic
1276
- _populate_entry_links(entry, item, atom_ns)
1454
+ guid_text = (
1455
+ guid_element.text.strip()
1456
+ if guid_element is not None and guid_element.text
1457
+ else None
1458
+ )
1459
+ _populate_entry_links_from_elements(
1460
+ entry,
1461
+ atom_links,
1462
+ guid_text=guid_text,
1463
+ guid_is_permalink=guid_element is not None
1464
+ and guid_element.get("isPermaLink") == "true",
1465
+ )
1277
1466
  else:
1278
- # Common RSS case: no atom:link elements
1279
1467
  entry["links"] = []
1280
1468
  if (
1281
1469
  "link" not in entry
@@ -1287,20 +1475,27 @@ def _parse_rss_feed_entry_fast(
1287
1475
  if "id" not in entry and "link" in entry:
1288
1476
  entry["id"] = entry["link"]
1289
1477
 
1290
- _populate_entry_content(entry, item, "rss", atom_ns, rss_text_by_full=text_by_full)
1478
+ if include_content:
1479
+ _populate_entry_content_preparsed(
1480
+ entry,
1481
+ item,
1482
+ content_el=(
1483
+ encoded_content_el if encoded_content_el is not None else raw_content_el
1484
+ ),
1485
+ rss_description_text=rss_description_text,
1486
+ )
1291
1487
 
1292
- if has_media_ns:
1488
+ if include_media and has_media_ns:
1293
1489
  media_contents = _parse_media_content(item)
1294
1490
  if media_contents:
1295
1491
  entry["media_content"] = media_contents
1296
1492
 
1297
- enclosures = _parse_enclosures(item)
1298
- if enclosures:
1493
+ if include_enclosures and enclosures:
1299
1494
  entry["enclosures"] = enclosures
1300
1495
 
1301
1496
  author = _first_non_empty(text_by_local, ("author", "creator"))
1302
1497
  if not author:
1303
- atom_author = item.find(f"{{{atom_ns}}}author/{{{atom_ns}}}name")
1498
+ atom_author = item.find(atom_tags["author_name"])
1304
1499
  author = (
1305
1500
  atom_author.text.strip()
1306
1501
  if atom_author is not None and atom_author.text
@@ -1313,9 +1508,8 @@ def _parse_rss_feed_entry_fast(
1313
1508
  if comments:
1314
1509
  entry["comments"] = comments.strip()
1315
1510
 
1316
- tags = _parse_tags(item, "rss", atom_ns)
1317
- if tags:
1318
- entry["tags"] = tags
1511
+ if include_tags and (tag_categories or tag_subjects):
1512
+ entry["tags"] = tag_categories + tag_subjects
1319
1513
 
1320
1514
  return entry
1321
1515
 
@@ -1324,93 +1518,134 @@ def _parse_atom_feed_entry_fast(
1324
1518
  item: _Element,
1325
1519
  atom_ns: str,
1326
1520
  has_media_ns: bool = True,
1521
+ *,
1522
+ include_content: bool = True,
1523
+ include_tags: bool = True,
1524
+ include_media: bool = True,
1525
+ include_enclosures: bool = True,
1327
1526
  ) -> FastFeedParserDict:
1328
- ns = f"{{{atom_ns}}}"
1527
+ t = _atom_ns_tags(atom_ns)
1528
+ atom_link_tag = t["link"]
1529
+ atom_author_tag = t["ns"] + "author"
1530
+ atom_name_tag = t["ns"] + "name"
1531
+ atom_links: list[_Element] = []
1532
+ atom_categories: list[dict[str, str | None]] = []
1533
+ enclosures: list[dict[str, Any]] = []
1534
+ content_el: Optional[_Element] = None
1535
+ author_name: Optional[str] = None
1536
+ first_link_href: Optional[str] = None
1537
+ published_source: Optional[str] = None
1538
+ updated_source: Optional[str] = None
1539
+ published_fallback_source: Optional[str] = None
1540
+ updated_fallback_source: Optional[str] = None
1541
+
1329
1542
  entry = FastFeedParserDict()
1543
+ for child in item:
1544
+ tag = child.tag
1545
+ if not isinstance(tag, str):
1546
+ continue
1330
1547
 
1331
- # ID
1332
- el = item.find(ns + "id")
1333
- if el is not None and el.text:
1334
- entry["id"] = el.text.strip()
1335
-
1336
- # Title
1337
- el = item.find(ns + "title")
1338
- if el is not None and el.text:
1339
- entry["title"] = el.text.strip()
1340
-
1341
- # Description (summary)
1342
- el = item.find(ns + "summary")
1343
- if el is not None and el.text:
1344
- entry["description"] = el.text.strip()
1345
-
1346
- # Link (href attribute)
1347
- el = item.find(ns + "link")
1348
- if el is not None:
1349
- href = el.get("href")
1350
- if href:
1351
- entry["link"] = href.strip()
1352
-
1353
- # Dates: Atom 1.0 uses published/updated, Atom 0.3 uses issued/modified
1354
- is_atom_03 = atom_ns == "http://purl.org/atom/ns#"
1355
- pub_tag = "issued" if is_atom_03 else "published"
1356
- upd_tag = "modified" if is_atom_03 else "updated"
1357
- pub_fallback_tag = "published" if is_atom_03 else "issued"
1358
- upd_fallback_tag = "updated" if is_atom_03 else "modified"
1359
-
1360
- el = item.find(ns + pub_tag)
1361
- if el is not None and el.text:
1362
- published = _parse_date(el.text)
1548
+ text_value = child.text
1549
+ if tag == t["id"] and "id" not in entry and text_value:
1550
+ entry["id"] = text_value.strip()
1551
+ elif tag == t["title"] and "title" not in entry and text_value:
1552
+ entry["title"] = text_value.strip()
1553
+ elif tag == t["summary"] and "description" not in entry and text_value:
1554
+ entry["description"] = text_value.strip()
1555
+ elif tag == t["published"] and published_source is None and text_value:
1556
+ published_source = text_value
1557
+ elif tag == t["updated"] and updated_source is None and text_value:
1558
+ updated_source = text_value
1559
+ elif (
1560
+ tag == t["pub_fallback"]
1561
+ and published_fallback_source is None
1562
+ and text_value
1563
+ ):
1564
+ published_fallback_source = text_value
1565
+ elif (
1566
+ tag == t["upd_fallback"] and updated_fallback_source is None and text_value
1567
+ ):
1568
+ updated_fallback_source = text_value
1569
+ elif tag == atom_link_tag:
1570
+ atom_links.append(child)
1571
+ href = child.get("href")
1572
+ if href and first_link_href is None:
1573
+ first_link_href = href.strip()
1574
+ elif include_content and tag == t["content"] and content_el is None:
1575
+ content_el = child
1576
+ elif tag == atom_author_tag and author_name is None:
1577
+ author_name_el = child.find(atom_name_tag)
1578
+ if author_name_el is not None and author_name_el.text:
1579
+ author_name = author_name_el.text.strip()
1580
+
1581
+ if include_tags and tag == t["category"]:
1582
+ term = child.get("term")
1583
+ if term:
1584
+ atom_categories.append(
1585
+ {
1586
+ "term": term,
1587
+ "scheme": child.get("scheme"),
1588
+ "label": child.get("label"),
1589
+ }
1590
+ )
1591
+
1592
+ if include_enclosures and tag == "enclosure":
1593
+ cleaned = _parse_enclosure_element(child)
1594
+ if cleaned.get("url"):
1595
+ enclosures.append(cleaned)
1596
+
1597
+ if first_link_href:
1598
+ entry["link"] = first_link_href
1599
+
1600
+ if published_source:
1601
+ published = _parse_date(published_source)
1363
1602
  if published:
1364
1603
  entry["published"] = published
1365
1604
 
1366
- el = item.find(ns + upd_tag)
1367
- if el is not None and el.text:
1368
- updated = _parse_date(el.text)
1605
+ if updated_source:
1606
+ updated = _parse_date(updated_source)
1369
1607
  if updated:
1370
1608
  entry["updated"] = updated
1371
1609
 
1372
- # Fallback date fields for mixed namespace scenarios
1373
- if "published" not in entry:
1374
- el = item.find(ns + pub_fallback_tag)
1375
- if el is not None and el.text:
1376
- published = _parse_date(el.text)
1377
- if published:
1378
- entry["published"] = published
1610
+ if "published" not in entry and published_fallback_source:
1611
+ published = _parse_date(published_fallback_source)
1612
+ if published:
1613
+ entry["published"] = published
1379
1614
 
1380
- if "updated" not in entry:
1381
- el = item.find(ns + upd_fallback_tag)
1382
- if el is not None and el.text:
1383
- updated = _parse_date(el.text)
1384
- if updated:
1385
- entry["updated"] = updated
1615
+ if "updated" not in entry and updated_fallback_source:
1616
+ updated = _parse_date(updated_fallback_source)
1617
+ if updated:
1618
+ entry["updated"] = updated
1386
1619
 
1387
1620
  if "updated" in entry and "published" not in entry:
1388
1621
  entry["published"] = entry["updated"]
1389
1622
 
1390
- _populate_entry_links(entry, item, atom_ns)
1623
+ _populate_entry_links_from_elements(entry, atom_links)
1391
1624
 
1392
1625
  if "id" not in entry and "link" in entry:
1393
1626
  entry["id"] = entry["link"]
1394
1627
 
1395
- _populate_entry_content(entry, item, "atom", atom_ns)
1628
+ if include_content:
1629
+ _populate_entry_content_preparsed(
1630
+ entry,
1631
+ item,
1632
+ content_el=content_el,
1633
+ rss_description_text=None,
1634
+ )
1396
1635
 
1397
- if has_media_ns:
1636
+ if include_media and has_media_ns:
1398
1637
  media_contents = _parse_media_content(item)
1399
1638
  if media_contents:
1400
1639
  entry["media_content"] = media_contents
1401
1640
 
1402
- enclosures = _parse_enclosures(item)
1403
- if enclosures:
1641
+ if include_enclosures and enclosures:
1404
1642
  entry["enclosures"] = enclosures
1405
1643
 
1406
- # Author
1407
- el = item.find(ns + "author/" + ns + "name")
1408
- if el is not None and el.text:
1409
- entry["author"] = el.text.strip()
1644
+ if author_name:
1645
+ entry["author"] = author_name
1410
1646
 
1411
- tags = _parse_tags(item, "atom", atom_ns)
1412
- if tags:
1413
- entry["tags"] = tags
1647
+ if include_tags and atom_categories:
1648
+ entry["tags"] = atom_categories
1414
1649
 
1415
1650
  return entry
1416
1651
 
@@ -1420,15 +1655,36 @@ def _parse_feed_entry(
1420
1655
  feed_type: _FeedType,
1421
1656
  atom_namespace: Optional[str] = None,
1422
1657
  has_media_ns: bool = True,
1658
+ *,
1659
+ include_content: bool = True,
1660
+ include_tags: bool = True,
1661
+ include_media: bool = True,
1662
+ include_enclosures: bool = True,
1423
1663
  ) -> FastFeedParserDict:
1424
1664
  # Use dynamic atom namespace or fallback to default
1425
1665
  atom_ns = atom_namespace or "http://www.w3.org/2005/Atom"
1426
1666
 
1427
1667
  if feed_type == "rss":
1428
- return _parse_rss_feed_entry_fast(item, atom_ns, has_media_ns)
1668
+ return _parse_rss_feed_entry_fast(
1669
+ item,
1670
+ atom_ns,
1671
+ has_media_ns,
1672
+ include_content=include_content,
1673
+ include_tags=include_tags,
1674
+ include_media=include_media,
1675
+ include_enclosures=include_enclosures,
1676
+ )
1429
1677
 
1430
1678
  if feed_type == "atom":
1431
- return _parse_atom_feed_entry_fast(item, atom_ns, has_media_ns)
1679
+ return _parse_atom_feed_entry_fast(
1680
+ item,
1681
+ atom_ns,
1682
+ has_media_ns,
1683
+ include_content=include_content,
1684
+ include_tags=include_tags,
1685
+ include_media=include_media,
1686
+ include_enclosures=include_enclosures,
1687
+ )
1432
1688
 
1433
1689
  # RDF path uses the generic field machinery
1434
1690
  # Check if this is Atom 0.3 to use different date field names
@@ -1544,16 +1800,18 @@ def _parse_feed_entry(
1544
1800
  if "id" not in entry and "link" in entry:
1545
1801
  entry["id"] = entry["link"]
1546
1802
 
1547
- _populate_entry_content(entry, item, feed_type, atom_ns)
1803
+ if include_content:
1804
+ _populate_entry_content(entry, item, feed_type, atom_ns)
1548
1805
 
1549
- if has_media_ns:
1806
+ if include_media and has_media_ns:
1550
1807
  media_contents = _parse_media_content(item)
1551
1808
  if media_contents:
1552
1809
  entry["media_content"] = media_contents
1553
1810
 
1554
- enclosures = _parse_enclosures(item)
1555
- if enclosures:
1556
- entry["enclosures"] = enclosures
1811
+ if include_enclosures:
1812
+ enclosures = _parse_enclosures(item)
1813
+ if enclosures:
1814
+ entry["enclosures"] = enclosures
1557
1815
 
1558
1816
  author = get_field_value(
1559
1817
  "author",
@@ -1569,9 +1827,10 @@ def _parse_feed_entry(
1569
1827
  entry["author"] = author
1570
1828
 
1571
1829
  # Parse entry-level tags/categories
1572
- tags = _parse_tags(item, feed_type, atom_ns)
1573
- if tags:
1574
- entry["tags"] = tags
1830
+ if include_tags:
1831
+ tags = _parse_tags(item, feed_type, atom_ns)
1832
+ if tags:
1833
+ entry["tags"] = tags
1575
1834
 
1576
1835
  return entry
1577
1836
 
@@ -1902,6 +2161,7 @@ def _slow_dateparser(value: str) -> Optional[datetime.datetime]:
1902
2161
  return None
1903
2162
 
1904
2163
 
2164
+ @lru_cache(maxsize=8192)
1905
2165
  def _parse_date(date_str: str) -> Optional[str]:
1906
2166
  """Parse date string and return as an ISO 8601 formatted UTC string.
1907
2167
 
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.1
2
2
  Name: fastfeedparser
3
- Version: 0.5.5
3
+ Version: 0.5.6
4
4
  Summary: High performance RSS, Atom, JSON and RDF feed parser in Python
5
5
  Home-page: https://github.com/kagisearch/fastfeedparser
6
6
  Author: Vladimir Prelovac
@@ -25,6 +25,7 @@ Requires-Python: >=3.7
25
25
  Description-Content-Type: text/markdown
26
26
  Provides-Extra: dateparser
27
27
  Provides-Extra: brotli
28
+ Provides-Extra: orjson
28
29
  Provides-Extra: full
29
30
  License-File: LICENSE
30
31
 
@@ -146,7 +147,7 @@ Speedup: 50.1x
146
147
 
147
148
  ### Main Functions
148
149
 
149
- - `parse(source)`: Parse feed from a source that can be URL or a string
150
+ - `parse(source, *, include_content=True, include_tags=True, include_media=True, include_enclosures=True)`: Parse feed from a URL/XML/JSON source, with optional field extraction toggles for faster parsing.
150
151
 
151
152
 
152
153
  ### Feed Object Structure
@@ -10,3 +10,7 @@ dateparser
10
10
  [full]
11
11
  dateparser
12
12
  brotli
13
+ orjson
14
+
15
+ [orjson]
16
+ orjson
@@ -56,3 +56,49 @@ def test_meta_refresh_none_when_missing():
56
56
  def test_meta_refresh_none_when_same_url():
57
57
  html = '<html><head><meta http-equiv="refresh" content="0; url=https://example.com/"></head></html>'
58
58
  assert _extract_meta_refresh_url(html, "https://example.com/") is None
59
+
60
+
61
+ def test_parse_optional_field_flags():
62
+ xml = """<?xml version="1.0" encoding="utf-8"?>
63
+ <rss version="2.0"
64
+ xmlns:content="http://purl.org/rss/1.0/modules/content/"
65
+ xmlns:media="http://search.yahoo.com/mrss/">
66
+ <channel>
67
+ <title>Example Feed</title>
68
+ <category>feed-tag</category>
69
+ <item>
70
+ <title>Example Item</title>
71
+ <link>https://example.com/item</link>
72
+ <pubDate>Mon, 01 Jan 2024 00:00:00 GMT</pubDate>
73
+ <description>Summary</description>
74
+ <content:encoded><![CDATA[<p>Body</p>]]></content:encoded>
75
+ <category>entry-tag</category>
76
+ <enclosure url="https://example.com/audio.mp3" type="audio/mpeg" length="123" />
77
+ <media:content url="https://example.com/image.jpg" type="image/jpeg" />
78
+ </item>
79
+ </channel>
80
+ </rss>
81
+ """
82
+ full = parse(xml)
83
+ entry_full = full.entries[0]
84
+ assert "content" in entry_full
85
+ assert "tags" in entry_full
86
+ assert "enclosures" in entry_full
87
+ assert "media_content" in entry_full
88
+ assert "tags" in full.feed
89
+
90
+ trimmed = parse(
91
+ xml,
92
+ include_content=False,
93
+ include_tags=False,
94
+ include_media=False,
95
+ include_enclosures=False,
96
+ )
97
+ entry_trimmed = trimmed.entries[0]
98
+ assert "content" not in entry_trimmed
99
+ assert "tags" not in entry_trimmed
100
+ assert "enclosures" not in entry_trimmed
101
+ assert "media_content" not in entry_trimmed
102
+ assert "tags" not in trimmed.feed
103
+ assert entry_trimmed["title"] == "Example Item"
104
+ assert entry_trimmed["link"] == "https://example.com/item"
File without changes