fastfeedparser 0.5.3__tar.gz → 0.5.5__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.1
2
2
  Name: fastfeedparser
3
- Version: 0.5.3
3
+ Version: 0.5.5
4
4
  Summary: High performance RSS, Atom, JSON and RDF feed parser in Python
5
5
  Home-page: https://github.com/kagisearch/fastfeedparser
6
6
  Author: Vladimir Prelovac
@@ -34,7 +34,7 @@ A high-performance feed parser for Python that handles RSS, Atom, and RDF. Built
34
34
 
35
35
  ### Why FastFeedParser?
36
36
 
37
- It's about 10x faster (check included `benchmark.py`) than popular feedparser
37
+ It's about 25x faster (check included `benchmark.py`) than popular feedparser
38
38
  library while keeping a familiar API. This speed comes from:
39
39
 
40
40
  - lxml for efficient XML parsing
@@ -4,7 +4,7 @@ A high-performance feed parser for Python that handles RSS, Atom, and RDF. Built
4
4
 
5
5
  ### Why FastFeedParser?
6
6
 
7
- It's about 10x faster (check included `benchmark.py`) than popular feedparser
7
+ It's about 25x faster (check included `benchmark.py`) than popular feedparser
8
8
  library while keeping a familiar API. This speed comes from:
9
9
 
10
10
  - lxml for efficient XML parsing
@@ -0,0 +1,23 @@
1
+ [build-system]
2
+ requires = ["setuptools~=67.0", "wheel"]
3
+ build-backend = "setuptools.build_meta"
4
+
5
+ [tool.ruff]
6
+ extend-exclude = [
7
+ "1.py",
8
+ "debug_*.py",
9
+ "test_*.py",
10
+ "benchmark.py",
11
+ "investigate_failures.py",
12
+ "show_error_messages.py",
13
+ "check_oh4_dates.py",
14
+ "comprehensive_debug.py",
15
+ "profile_dylanharris.py",
16
+ ]
17
+
18
+ [tool.ty.rules]
19
+ unresolved-import = "ignore"
20
+
21
+ [tool.pytest.ini_options]
22
+ testpaths = ["tests"]
23
+
@@ -1,6 +1,6 @@
1
1
  [metadata]
2
2
  name = fastfeedparser
3
- version = 0.5.3
3
+ version = 0.5.5
4
4
  author = Vladimir Prelovac
5
5
  author_email = vlad@kagi.com
6
6
  description = High performance RSS, Atom, JSON and RDF feed parser in Python
@@ -28,10 +28,16 @@ from dateutil import parser as dateutil_parser
28
28
  from lxml import etree
29
29
 
30
30
  if TYPE_CHECKING:
31
+ from typing import Protocol
32
+
31
33
  from lxml.etree import _Element
32
34
 
35
+ class _ElementValueGetter(Protocol):
36
+ def __call__(self, path: str, attribute: Optional[str] = None) -> Optional[str]: ...
37
+
33
38
  _FeedType = Literal["rss", "atom", "rdf"]
34
39
 
40
+
35
41
  _UTC = datetime.timezone.utc
36
42
 
37
43
  # Pre-compiled regex patterns for performance
@@ -39,16 +45,16 @@ _RE_XML_DECL_ENCODING = re.compile(
39
45
  r'(<\?xml[^>]*encoding=["\'])([^"\']+)(["\'][^>]*\?>)', re.IGNORECASE
40
46
  )
41
47
  _RE_XML_DECL_ENCODING_BYTES = re.compile(
42
- br'(<\?xml[^>]*encoding=["\'])([^"\']+)(["\'][^>]*\?>)', re.IGNORECASE
48
+ rb'(<\?xml[^>]*encoding=["\'])([^"\']+)(["\'][^>]*\?>)', re.IGNORECASE
43
49
  )
44
- _RE_DOUBLE_XML_DECL_BYTES = re.compile(br"<\?xml\?xml\s+", re.IGNORECASE)
45
- _RE_DOUBLE_CLOSE_BYTES = re.compile(br"\?\?>\s*")
46
- _RE_UNQUOTED_ATTR_BYTES = re.compile(br'(\s+[\w:]+)=([^\s>"\']+)')
50
+ _RE_DOUBLE_XML_DECL_BYTES = re.compile(rb"<\?xml\?xml\s+", re.IGNORECASE)
51
+ _RE_DOUBLE_CLOSE_BYTES = re.compile(rb"\?\?>\s*")
52
+ _RE_UNQUOTED_ATTR_BYTES = re.compile(rb'(\s+[\w:]+)=([^\s>"\']+)')
47
53
  _RE_UTF16_ENCODING_BYTES = re.compile(
48
- br'(<\?xml[^>]*encoding=["\'])utf-16(-le|-be)?(["\'][^>]*\?>)', re.IGNORECASE
54
+ rb'(<\?xml[^>]*encoding=["\'])utf-16(-le|-be)?(["\'][^>]*\?>)', re.IGNORECASE
49
55
  )
50
56
  _RE_UNCLOSED_LINK_BYTES = re.compile(
51
- br"<link([^>]*[^/])>\s*(?=\n\s*<(?!/link\s*>))", re.MULTILINE
57
+ rb"<link([^>]*[^/])>\s*(?=\n\s*<(?!/link\s*>))", re.MULTILINE
52
58
  )
53
59
  _RE_FEB29 = re.compile(r"(\d{4})-02-29")
54
60
  _RE_HTML_TAGS = re.compile(r"<[^>]+>")
@@ -59,14 +65,20 @@ _RE_ISO_FRACTION = re.compile(r"\.(\d{7,})(?=(?:[+-]\d{2}:?\d{2}|Z|$))", re.IGNO
59
65
  _RE_RFC822 = re.compile(
60
66
  r"(?:\w{3},\s+)?(\d{1,2})\s+(\w{3})\s+(\d{4})\s+(\d{2}):(\d{2}):(\d{2})\s+([+-]\d{4}|[A-Z]{2,5})"
61
67
  )
68
+ _RE_HOUR24 = re.compile(r"(\d{4}-\d{2}-\d{2})[T ]24:(\d{2}):(\d{2})")
62
69
  _MONTHS_RFC822: dict[str, int] = {
63
- "jan": 1, "feb": 2, "mar": 3, "apr": 4, "may": 5, "jun": 6,
64
- "jul": 7, "aug": 8, "sep": 9, "oct": 10, "nov": 11, "dec": 12,
65
- }
66
- _TZ_OFFSETS_RFC822: dict[str, int] = {
67
- "GMT": 0, "UTC": 0, "UT": 0,
68
- "EST": -18000, "EDT": -14400, "CST": -21600, "CDT": -18000,
69
- "MST": -25200, "MDT": -21600, "PST": -28800, "PDT": -25200,
70
+ "jan": 1,
71
+ "feb": 2,
72
+ "mar": 3,
73
+ "apr": 4,
74
+ "may": 5,
75
+ "jun": 6,
76
+ "jul": 7,
77
+ "aug": 8,
78
+ "sep": 9,
79
+ "oct": 10,
80
+ "nov": 11,
81
+ "dec": 12,
70
82
  }
71
83
 
72
84
 
@@ -129,7 +141,9 @@ def _clean_feed_bytes(content: bytes) -> bytes:
129
141
  if preview_lower.startswith((b"<?xml", b"<rss", b"<feed", b"<rdf")):
130
142
  return stripped_content
131
143
 
132
- if preview_lower.startswith(b"<!doctype html") or preview_lower.startswith(b"<html"):
144
+ if preview_lower.startswith(b"<!doctype html") or preview_lower.startswith(
145
+ b"<html"
146
+ ):
133
147
  raise ValueError("Content appears to be HTML, not a valid RSS/Atom feed")
134
148
 
135
149
  xml_start_patterns = (
@@ -160,15 +174,17 @@ def _fix_malformed_xml_bytes(content: bytes, actual_encoding: str = "utf-8") ->
160
174
  content = _RE_DOUBLE_CLOSE_BYTES.sub(b"?>", content)
161
175
 
162
176
  # Fix malformed attribute syntax like rss:version=2.0 (missing quotes)
163
- content = _RE_UNQUOTED_ATTR_BYTES.sub(br'\1="\2"', content)
177
+ content = _RE_UNQUOTED_ATTR_BYTES.sub(rb'\1="\2"', content)
164
178
 
165
179
  # Update encoding in XML declaration to match actual encoding when a feed was transcoded.
166
180
  if actual_encoding.lower() != "utf-16":
167
- replacement = br"\1" + actual_encoding.encode("ascii", errors="replace") + br"\3"
181
+ replacement = (
182
+ rb"\1" + actual_encoding.encode("ascii", errors="replace") + rb"\3"
183
+ )
168
184
  content = _RE_UTF16_ENCODING_BYTES.sub(replacement, content)
169
185
 
170
186
  # Fix unclosed link tags - common in Atom feeds
171
- content = _RE_UNCLOSED_LINK_BYTES.sub(br"<link\1/>", content)
187
+ content = _RE_UNCLOSED_LINK_BYTES.sub(rb"<link\1/>", content)
172
188
 
173
189
  return content
174
190
 
@@ -179,6 +195,13 @@ def _prepare_xml_bytes(xml_content: str | bytes) -> bytes:
179
195
  if not cleaned.strip():
180
196
  raise ValueError("Empty content")
181
197
 
198
+ # Replace Unicode LINE SEPARATOR (U+2028) and PARAGRAPH SEPARATOR (U+2029)
199
+ # with regular newlines — these are invalid in XML 1.0 and cause lxml to fail.
200
+ if b"\xe2\x80\xa8" in cleaned or b"\xe2\x80\xa9" in cleaned:
201
+ cleaned = cleaned.replace(b"\xe2\x80\xa8", b"\n").replace(
202
+ b"\xe2\x80\xa9", b"\n"
203
+ )
204
+
182
205
  detected_encoding = _detect_xml_encoding(cleaned)
183
206
  actual_encoding = detected_encoding
184
207
  if detected_encoding.startswith("utf-16") and b"\x00" not in cleaned[:200]:
@@ -500,16 +523,16 @@ def _raise_for_non_feed_root(
500
523
  if base_msg is None:
501
524
  return
502
525
 
503
- error_msg = _extract_error_message(root, raw_bytes).strip()[:300] or "No error message"
526
+ error_msg = (
527
+ _extract_error_message(root, raw_bytes).strip()[:300] or "No error message"
528
+ )
504
529
 
505
530
  if error_msg != "No error message" and len(error_msg) > 10:
506
531
  raise ValueError(f"{base_msg}: {error_msg[:150]}")
507
532
  raise ValueError(base_msg)
508
533
 
509
534
 
510
- _RE_META_REFRESH_URL = re.compile(
511
- r'url\s*=\s*["\']?\s*([^"\'>\s]+)', re.IGNORECASE
512
- )
535
+ _RE_META_REFRESH_URL = re.compile(r'url\s*=\s*["\']?\s*([^"\'>\s]+)', re.IGNORECASE)
513
536
 
514
537
 
515
538
  def _extract_meta_refresh_url(content: str | bytes, base_url: str) -> str | None:
@@ -558,7 +581,8 @@ def _detect_feed_structure(
558
581
  if channel is None:
559
582
  has_atom_elements = any(
560
583
  isinstance(child.tag, str)
561
- and child.tag in {"entry", "title", "subtitle", "updated", "id", "author", "link"}
584
+ and child.tag
585
+ in {"entry", "title", "subtitle", "updated", "id", "author", "link"}
562
586
  for child in root
563
587
  )
564
588
  if has_atom_elements:
@@ -586,7 +610,9 @@ def _detect_feed_structure(
586
610
  items = []
587
611
  items.append(child)
588
612
  if not items:
589
- items = channel.xpath(".//item") or channel.xpath(".//*[local-name()='item']")
613
+ items = channel.xpath(".//item") or channel.xpath(
614
+ ".//*[local-name()='item']"
615
+ )
590
616
 
591
617
  if not items:
592
618
  items = channel.findall("entry")
@@ -633,7 +659,9 @@ def _detect_feed_structure(
633
659
  if root.tag == "{http://www.w3.org/1999/02/22-rdf-syntax-ns#}RDF":
634
660
  feed_type = "rdf"
635
661
  channel = root
636
- items = channel.findall(".//{http://purl.org/rss/1.0/}item") or channel.findall("item")
662
+ items = channel.findall(".//{http://purl.org/rss/1.0/}item") or channel.findall(
663
+ "item"
664
+ )
637
665
  return feed_type, channel, items, atom_namespace
638
666
 
639
667
  raise ValueError(f"Unknown feed type: {root.tag}")
@@ -657,22 +685,34 @@ def _parse_content(xml_content: str | bytes) -> FastFeedParserDict:
657
685
  feed = _parse_feed_info(channel, feed_type, atom_namespace)
658
686
 
659
687
  # Detect once whether media namespace is used anywhere in the document
660
- has_media_ns = b"search.yahoo.com/mrss" in xml_content if isinstance(xml_content, bytes) else "search.yahoo.com/mrss" in xml_content
688
+ has_media_ns = (
689
+ b"search.yahoo.com/mrss" in xml_content
690
+ if isinstance(xml_content, bytes)
691
+ else "search.yahoo.com/mrss" in xml_content
692
+ )
661
693
 
662
- # Parse entries
694
+ # Parse entries — resolve parser once per feed instead of per entry
663
695
  entries: list[FastFeedParserDict] = []
664
696
  feed["entries"] = entries
665
- for item in items:
666
- entry = _parse_feed_entry(
667
- item,
668
- feed_type,
669
- atom_namespace,
670
- has_media_ns,
671
- )
672
- # Ensure that titles and descriptions are always present
673
- entry["title"] = entry.get("title", "").strip()
674
- entry["description"] = entry.get("description", "").strip()
675
- entries.append(entry)
697
+ atom_ns = atom_namespace or "http://www.w3.org/2005/Atom"
698
+ if feed_type == "rss":
699
+ for item in items:
700
+ entry = _parse_rss_feed_entry_fast(item, atom_ns, has_media_ns)
701
+ entry.setdefault("title", "")
702
+ entry.setdefault("description", "")
703
+ entries.append(entry)
704
+ elif feed_type == "atom":
705
+ for item in items:
706
+ entry = _parse_atom_feed_entry_fast(item, atom_ns, has_media_ns)
707
+ entry.setdefault("title", "")
708
+ entry.setdefault("description", "")
709
+ entries.append(entry)
710
+ else:
711
+ for item in items:
712
+ entry = _parse_feed_entry(item, feed_type, atom_namespace, has_media_ns)
713
+ entry["title"] = entry.get("title", "").strip()
714
+ entry["description"] = entry.get("description", "").strip()
715
+ entries.append(entry)
676
716
 
677
717
  return feed
678
718
 
@@ -692,6 +732,7 @@ def parse(source: str | bytes) -> FastFeedParserDict:
692
732
  """
693
733
  is_url = isinstance(source, str) and source.startswith(("http://", "https://"))
694
734
  if is_url:
735
+ assert isinstance(source, str)
695
736
  content = _fetch_url_content(source)
696
737
  else:
697
738
  content = source
@@ -701,6 +742,7 @@ def parse(source: str | bytes) -> FastFeedParserDict:
701
742
  except ValueError as e:
702
743
  if not is_url:
703
744
  raise
745
+ assert isinstance(source, str)
704
746
  err_msg = str(e)
705
747
  if "HTML" not in err_msg and "not a valid RSS/Atom feed" not in err_msg:
706
748
  raise
@@ -930,7 +972,9 @@ def _coerce_int_fields(mapping: dict[str, Any], fields: tuple[str, ...]) -> None
930
972
  mapping.pop(field, None)
931
973
 
932
974
 
933
- def _populate_entry_links(entry: FastFeedParserDict, item: _Element, atom_ns: str) -> None:
975
+ def _populate_entry_links(
976
+ entry: FastFeedParserDict, item: _Element, atom_ns: str
977
+ ) -> None:
934
978
  entry_links: list[dict[str, Optional[str]]] = []
935
979
  alternate_link: Optional[dict[str, Optional[str]]] = None
936
980
  for link in item.findall(f"{{{atom_ns}}}link"):
@@ -951,7 +995,9 @@ def _populate_entry_links(entry: FastFeedParserDict, item: _Element, atom_ns: st
951
995
 
952
996
  guid = item.find("guid")
953
997
  guid_text = guid.text.strip() if guid is not None and guid.text else None
954
- is_guid_url = guid_text is not None and guid_text.startswith(("http://", "https://"))
998
+ is_guid_url = guid_text is not None and guid_text.startswith(
999
+ ("http://", "https://")
1000
+ )
955
1001
 
956
1002
  if is_guid_url and "link" not in entry:
957
1003
  entry["link"] = guid_text
@@ -973,13 +1019,30 @@ def _populate_entry_links(entry: FastFeedParserDict, item: _Element, atom_ns: st
973
1019
 
974
1020
 
975
1021
  def _populate_entry_content(
976
- entry: FastFeedParserDict, item: _Element, feed_type: _FeedType, atom_ns: str
1022
+ entry: FastFeedParserDict,
1023
+ item: _Element,
1024
+ feed_type: _FeedType,
1025
+ atom_ns: str,
1026
+ rss_text_by_full: Optional[dict[str, Optional[str]]] = None,
977
1027
  ) -> None:
978
1028
  content_el = None
979
1029
  if feed_type == "rss":
980
- content_el = item.find("{http://purl.org/rss/1.0/modules/content/}encoded")
981
- if content_el is None:
982
- content_el = item.find("content")
1030
+ # Fast path: check pre-built text map before doing tree searches
1031
+ if rss_text_by_full is not None:
1032
+ content_encoded_text = rss_text_by_full.get(
1033
+ "{http://purl.org/rss/1.0/modules/content/}encoded"
1034
+ )
1035
+ if content_encoded_text is not None:
1036
+ # We have content:encoded text — still need the element for type/lang/base attrs
1037
+ content_el = item.find("{http://purl.org/rss/1.0/modules/content/}encoded")
1038
+ else:
1039
+ content_text = rss_text_by_full.get("content")
1040
+ if content_text is not None:
1041
+ content_el = item.find("content")
1042
+ else:
1043
+ content_el = item.find("{http://purl.org/rss/1.0/modules/content/}encoded")
1044
+ if content_el is None:
1045
+ content_el = item.find("content")
983
1046
  elif feed_type == "atom":
984
1047
  content_el = item.find(f"{{{atom_ns}}}content")
985
1048
 
@@ -992,7 +1055,9 @@ def _populate_entry_content(
992
1055
  entry["content"] = [
993
1056
  {
994
1057
  "type": content_type,
995
- "language": content_el.get("{http://www.w3.org/XML/1998/namespace}lang"),
1058
+ "language": content_el.get(
1059
+ "{http://www.w3.org/XML/1998/namespace}lang"
1060
+ ),
996
1061
  "base": content_el.get("{http://www.w3.org/XML/1998/namespace}base"),
997
1062
  "value": content_value,
998
1063
  }
@@ -1014,9 +1079,14 @@ def _populate_entry_content(
1014
1079
  content_value = entry["content"][0]["value"]
1015
1080
  if content_value:
1016
1081
  if "<" in content_value:
1017
- content_value = _RE_HTML_TAGS.sub(" ", content_value[:2048])
1082
+ content_value = _RE_HTML_TAGS.sub(" ", content_value[:1024])
1018
1083
  content_value = _html_mod.unescape(content_value)
1019
- if " " in content_value or "\n" in content_value or "\t" in content_value or "\r" in content_value:
1084
+ if (
1085
+ " " in content_value
1086
+ or "\n" in content_value
1087
+ or "\t" in content_value
1088
+ or "\r" in content_value
1089
+ ):
1020
1090
  content_value = _RE_WHITESPACE.sub(" ", content_value).strip()
1021
1091
  else:
1022
1092
  content_value = content_value.strip()
@@ -1110,7 +1180,9 @@ def _parse_enclosures(item: _Element) -> list[dict[str, Any]] | None:
1110
1180
  return enclosures or None
1111
1181
 
1112
1182
 
1113
- def _build_rss_item_text_maps(item: _Element) -> tuple[dict[str, Optional[str]], dict[str, Optional[str]]]:
1183
+ def _build_rss_item_text_maps(
1184
+ item: _Element,
1185
+ ) -> tuple[dict[str, Optional[str]], dict[str, Optional[str]]]:
1114
1186
  by_local: dict[str, Optional[str]] = {}
1115
1187
  by_full: dict[str, Optional[str]] = {}
1116
1188
  for child in item:
@@ -1132,7 +1204,9 @@ def _build_rss_item_text_maps(item: _Element) -> tuple[dict[str, Optional[str]],
1132
1204
  return by_local, by_full
1133
1205
 
1134
1206
 
1135
- def _first_non_empty(mapping: dict[str, Optional[str]], keys: tuple[str, ...]) -> Optional[str]:
1207
+ def _first_non_empty(
1208
+ mapping: dict[str, Optional[str]], keys: tuple[str, ...]
1209
+ ) -> Optional[str]:
1136
1210
  for key in keys:
1137
1211
  value = mapping.get(key)
1138
1212
  if value:
@@ -1157,29 +1231,37 @@ def _parse_rss_feed_entry_fast(
1157
1231
 
1158
1232
  title = text_by_local.get("title")
1159
1233
  if title:
1160
- entry["title"] = title
1234
+ entry["title"] = title.strip()
1161
1235
 
1162
1236
  description = _first_non_empty(text_by_local, ("description", "summary"))
1163
1237
  if description:
1164
- entry["description"] = description
1238
+ entry["description"] = description.strip()
1165
1239
 
1166
1240
  link = text_by_local.get("link")
1167
1241
  if link:
1168
1242
  entry["link"] = link.strip()
1169
1243
 
1170
- published_source = _first_non_empty(text_by_local, ("pubdate", "published", "issued", "date"))
1244
+ published_source = _first_non_empty(
1245
+ text_by_local, ("pubdate", "published", "issued", "date")
1246
+ )
1171
1247
  if published_source:
1172
1248
  published = _parse_date(published_source)
1173
1249
  if published:
1174
1250
  entry["published"] = published
1175
1251
 
1176
- updated_source = _first_non_empty(text_by_local, ("lastbuilddate", "updated", "modified"))
1252
+ updated_source = _first_non_empty(
1253
+ text_by_local, ("lastbuilddate", "updated", "modified")
1254
+ )
1177
1255
  if updated_source:
1178
1256
  updated = _parse_date(updated_source)
1179
1257
  if updated:
1180
1258
  entry["updated"] = updated
1181
1259
 
1182
- if "published" not in entry and rss_guid and not rss_guid.startswith(("http://", "https://")):
1260
+ if (
1261
+ "published" not in entry
1262
+ and rss_guid
1263
+ and not rss_guid.startswith(("http://", "https://"))
1264
+ ):
1183
1265
  guid_date = _parse_date(rss_guid)
1184
1266
  if guid_date:
1185
1267
  entry["published"] = guid_date
@@ -1195,13 +1277,17 @@ def _parse_rss_feed_entry_fast(
1195
1277
  else:
1196
1278
  # Common RSS case: no atom:link elements
1197
1279
  entry["links"] = []
1198
- if "link" not in entry and rss_guid and rss_guid.startswith(("http://", "https://")):
1280
+ if (
1281
+ "link" not in entry
1282
+ and rss_guid
1283
+ and rss_guid.startswith(("http://", "https://"))
1284
+ ):
1199
1285
  entry["link"] = rss_guid
1200
1286
 
1201
1287
  if "id" not in entry and "link" in entry:
1202
1288
  entry["id"] = entry["link"]
1203
1289
 
1204
- _populate_entry_content(entry, item, "rss", atom_ns)
1290
+ _populate_entry_content(entry, item, "rss", atom_ns, rss_text_by_full=text_by_full)
1205
1291
 
1206
1292
  if has_media_ns:
1207
1293
  media_contents = _parse_media_content(item)
@@ -1215,7 +1301,11 @@ def _parse_rss_feed_entry_fast(
1215
1301
  author = _first_non_empty(text_by_local, ("author", "creator"))
1216
1302
  if not author:
1217
1303
  atom_author = item.find(f"{{{atom_ns}}}author/{{{atom_ns}}}name")
1218
- author = atom_author.text.strip() if atom_author is not None and atom_author.text else None
1304
+ author = (
1305
+ atom_author.text.strip()
1306
+ if atom_author is not None and atom_author.text
1307
+ else None
1308
+ )
1219
1309
  if author:
1220
1310
  entry["author"] = author.strip()
1221
1311
 
@@ -1432,7 +1522,11 @@ def _parse_feed_entry(
1432
1522
  entry["updated"] = _parse_date(fallback_updated)
1433
1523
 
1434
1524
  # Try to extract date from GUID as final fallback
1435
- if "published" not in entry and rss_guid and not rss_guid.startswith(("http://", "https://")):
1525
+ if (
1526
+ "published" not in entry
1527
+ and rss_guid
1528
+ and not rss_guid.startswith(("http://", "https://"))
1529
+ ):
1436
1530
  guid_date = _parse_date(rss_guid)
1437
1531
  if guid_date:
1438
1532
  entry["published"] = guid_date
@@ -1468,7 +1562,9 @@ def _parse_feed_entry(
1468
1562
  False,
1469
1563
  )
1470
1564
  if not author:
1471
- author = element_get("{http://purl.org/dc/elements/1.1/}creator") or element_get("author")
1565
+ author = element_get(
1566
+ "{http://purl.org/dc/elements/1.1/}creator"
1567
+ ) or element_get("author")
1472
1568
  if author:
1473
1569
  entry["author"] = author
1474
1570
 
@@ -1483,9 +1579,9 @@ def _parse_feed_entry(
1483
1579
  def _field_value_getter(
1484
1580
  root: _Element,
1485
1581
  feed_type: _FeedType,
1486
- cached_get: Optional[Callable[[str, Optional[str]], Optional[str]]] = None,
1582
+ cached_get: Optional[_ElementValueGetter] = None,
1487
1583
  ) -> Callable[[str, str, str, bool], str | None]:
1488
- get_value = cached_get or _cached_element_value_factory(root)
1584
+ get_value: _ElementValueGetter = cached_get or _cached_element_value_factory(root)
1489
1585
 
1490
1586
  if feed_type == "rss":
1491
1587
 
@@ -1574,7 +1670,11 @@ def _get_element_value(
1574
1670
  el = found
1575
1671
  break
1576
1672
  else:
1577
- prefixed_paths = [f"rss:{path_lower}", f"atom:{path_lower}", f"dc:{path_lower}"]
1673
+ prefixed_paths = [
1674
+ f"rss:{path_lower}",
1675
+ f"atom:{path_lower}",
1676
+ f"dc:{path_lower}",
1677
+ ]
1578
1678
  for child in root:
1579
1679
  if not isinstance(child.tag, str):
1580
1680
  continue
@@ -1594,7 +1694,7 @@ def _get_element_value(
1594
1694
 
1595
1695
  def _cached_element_value_factory(
1596
1696
  root: _Element,
1597
- ) -> Callable[[str, Optional[str]], Optional[str]]:
1697
+ ) -> _ElementValueGetter:
1598
1698
  """Create a closure with a child tag index for fast namespace-prefix lookups."""
1599
1699
  # Build child tag index once: O(children) instead of O(children × misses)
1600
1700
  child_index: dict[str, _Element] = {}
@@ -1603,7 +1703,9 @@ def _cached_element_value_factory(
1603
1703
  child_index[child.tag.lower()] = child
1604
1704
 
1605
1705
  def getter(path: str, attribute: Optional[str] = None) -> Optional[str]:
1606
- return _get_element_value(root, path, attribute=attribute, child_index=child_index)
1706
+ return _get_element_value(
1707
+ root, path, attribute=attribute, child_index=child_index
1708
+ )
1607
1709
 
1608
1710
  return getter
1609
1711
 
@@ -1632,7 +1734,13 @@ def _normalize_iso_datetime_string(value: str) -> str:
1632
1734
  if cleaned.endswith(("Z", "z")):
1633
1735
  cleaned = cleaned[:-1] + "+00:00"
1634
1736
 
1635
- if " " in cleaned and "T" not in cleaned[:11] and len(cleaned) >= 10 and cleaned[4] == "-" and cleaned[0:4].isdigit():
1737
+ if (
1738
+ " " in cleaned
1739
+ and "T" not in cleaned[:11]
1740
+ and len(cleaned) >= 10
1741
+ and cleaned[4] == "-"
1742
+ and cleaned[0:4].isdigit()
1743
+ ):
1636
1744
  date_part, rest = cleaned.split(" ", 1)
1637
1745
  if rest and rest[0].isdigit():
1638
1746
  cleaned = f"{date_part}T{rest}"
@@ -1671,7 +1779,7 @@ def _fast_rfc822_to_iso(value: str) -> Optional[str]:
1671
1779
  1 if tz[0] == "+" else -1
1672
1780
  )
1673
1781
  else:
1674
- tz_offset_seconds = _TZ_OFFSETS_RFC822.get(tz)
1782
+ tz_offset_seconds = _custom_tzinfos.get(tz)
1675
1783
  if tz_offset_seconds is None:
1676
1784
  return None # Unknown tz name, fall through to full parser
1677
1785
  # Python requires offset strictly between -24h and +24h
@@ -1688,7 +1796,12 @@ def _fast_rfc822_to_iso(value: str) -> Optional[str]:
1688
1796
  if tz_offset_seconds == 0:
1689
1797
  return f"{base.year:04d}-{base.month:02d}-{base.day:02d}T{h:02d}:{mi:02d}:{s:02d}+00:00"
1690
1798
  dt = datetime.datetime(
1691
- base.year, base.month, base.day, h, mi, s,
1799
+ base.year,
1800
+ base.month,
1801
+ base.day,
1802
+ h,
1803
+ mi,
1804
+ s,
1692
1805
  tzinfo=datetime.timezone(datetime.timedelta(seconds=tz_offset_seconds)),
1693
1806
  )
1694
1807
  utc = dt.astimezone(_UTC)
@@ -1696,7 +1809,12 @@ def _fast_rfc822_to_iso(value: str) -> Optional[str]:
1696
1809
  if tz_offset_seconds == 0:
1697
1810
  return f"{year}-{month:02d}-{d:02d}T{hour}:{minute}:{second}+00:00"
1698
1811
  dt = datetime.datetime(
1699
- int(year), month, d, h, mi, s,
1812
+ int(year),
1813
+ month,
1814
+ d,
1815
+ h,
1816
+ mi,
1817
+ s,
1700
1818
  tzinfo=datetime.timezone(datetime.timedelta(seconds=tz_offset_seconds)),
1701
1819
  )
1702
1820
  utc = dt.astimezone(_UTC)
@@ -1777,7 +1895,9 @@ def _slow_dateparser(value: str) -> Optional[datetime.datetime]:
1777
1895
  except ImportError:
1778
1896
  return None
1779
1897
  try:
1780
- return _dateparser.parse(value, languages=["en"], settings={**_DATEPARSER_SETTINGS})
1898
+ return _dateparser.parse(
1899
+ value, languages=["en"], settings=_DATEPARSER_SETTINGS
1900
+ )
1781
1901
  except (ValueError, TypeError):
1782
1902
  return None
1783
1903
 
@@ -1834,16 +1954,22 @@ def _parse_date(date_str: str) -> Optional[str]:
1834
1954
  candidate = candidate.replace(f"{year}-02-29", f"{year}-02-28")
1835
1955
 
1836
1956
  if "T24:" in candidate or " 24:" in candidate:
1837
- m24 = re.search(r"(\d{4}-\d{2}-\d{2})[T ]24:(\d{2}):(\d{2})", candidate)
1957
+ m24 = _RE_HOUR24.search(candidate)
1838
1958
  if m24:
1839
1959
  base = datetime.date.fromisoformat(m24.group(1))
1840
1960
  mins, secs = int(m24.group(2)), int(m24.group(3))
1841
1961
  next_day = base + datetime.timedelta(days=1)
1842
- candidate = candidate[:m24.start()] + f"{next_day}T00:{mins:02d}:{secs:02d}" + candidate[m24.end():]
1962
+ candidate = (
1963
+ candidate[: m24.start()]
1964
+ + f"{next_day}T00:{mins:02d}:{secs:02d}"
1965
+ + candidate[m24.end() :]
1966
+ )
1843
1967
 
1844
1968
  dt: Optional[datetime.datetime] = None
1845
1969
 
1846
- is_iso_like = len(candidate) >= 10 and candidate[4] == "-" and candidate[0:4].isdigit()
1970
+ is_iso_like = (
1971
+ len(candidate) >= 10 and candidate[4] == "-" and candidate[0:4].isdigit()
1972
+ )
1847
1973
  if is_iso_like:
1848
1974
  iso_candidate = _normalize_iso_datetime_string(candidate)
1849
1975
  try:
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.1
2
2
  Name: fastfeedparser
3
- Version: 0.5.3
3
+ Version: 0.5.5
4
4
  Summary: High performance RSS, Atom, JSON and RDF feed parser in Python
5
5
  Home-page: https://github.com/kagisearch/fastfeedparser
6
6
  Author: Vladimir Prelovac
@@ -34,7 +34,7 @@ A high-performance feed parser for Python that handles RSS, Atom, and RDF. Built
34
34
 
35
35
  ### Why FastFeedParser?
36
36
 
37
- It's about 10x faster (check included `benchmark.py`) than popular feedparser
37
+ It's about 25x faster (check included `benchmark.py`) than popular feedparser
38
38
  library while keeping a familiar API. This speed comes from:
39
39
 
40
40
  - lxml for efficient XML parsing
@@ -34,12 +34,18 @@ def test_parse_bytes_with_non_utf8_encoding():
34
34
 
35
35
  def test_meta_refresh_extraction():
36
36
  html = '<!doctype html><html><head><meta http-equiv=refresh content="0; url=https://example.com/feed.xml"></head></html>'
37
- assert _extract_meta_refresh_url(html, "https://example.com/feed/") == "https://example.com/feed.xml"
37
+ assert (
38
+ _extract_meta_refresh_url(html, "https://example.com/feed/")
39
+ == "https://example.com/feed.xml"
40
+ )
38
41
 
39
42
 
40
43
  def test_meta_refresh_relative_url():
41
44
  html = b'<html><head><meta http-equiv="refresh" content="0;url=/index.xml"></head></html>'
42
- assert _extract_meta_refresh_url(html, "https://example.com/feed/") == "https://example.com/index.xml"
45
+ assert (
46
+ _extract_meta_refresh_url(html, "https://example.com/feed/")
47
+ == "https://example.com/index.xml"
48
+ )
43
49
 
44
50
 
45
51
  def test_meta_refresh_none_when_missing():
@@ -50,4 +56,3 @@ def test_meta_refresh_none_when_missing():
50
56
  def test_meta_refresh_none_when_same_url():
51
57
  html = '<html><head><meta http-equiv="refresh" content="0; url=https://example.com/"></head></html>'
52
58
  assert _extract_meta_refresh_url(html, "https://example.com/") is None
53
-
@@ -1,7 +0,0 @@
1
- [build-system]
2
- requires = ["setuptools~=67.0", "wheel"]
3
- build-backend = "setuptools.build_meta"
4
-
5
- [tool.pytest.ini_options]
6
- testpaths = ["tests"]
7
-
File without changes