fastfeedparser 0.5.2__tar.gz → 0.5.4__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.1
2
2
  Name: fastfeedparser
3
- Version: 0.5.2
3
+ Version: 0.5.4
4
4
  Summary: High performance RSS, Atom, JSON and RDF feed parser in Python
5
5
  Home-page: https://github.com/kagisearch/fastfeedparser
6
6
  Author: Vladimir Prelovac
@@ -0,0 +1,23 @@
1
+ [build-system]
2
+ requires = ["setuptools~=67.0", "wheel"]
3
+ build-backend = "setuptools.build_meta"
4
+
5
+ [tool.ruff]
6
+ extend-exclude = [
7
+ "1.py",
8
+ "debug_*.py",
9
+ "test_*.py",
10
+ "benchmark.py",
11
+ "investigate_failures.py",
12
+ "show_error_messages.py",
13
+ "check_oh4_dates.py",
14
+ "comprehensive_debug.py",
15
+ "profile_dylanharris.py",
16
+ ]
17
+
18
+ [tool.ty.rules]
19
+ unresolved-import = "ignore"
20
+
21
+ [tool.pytest.ini_options]
22
+ testpaths = ["tests"]
23
+
@@ -1,6 +1,6 @@
1
1
  [metadata]
2
2
  name = fastfeedparser
3
- version = 0.5.2
3
+ version = 0.5.4
4
4
  author = Vladimir Prelovac
5
5
  author_email = vlad@kagi.com
6
6
  description = High performance RSS, Atom, JSON and RDF feed parser in Python
@@ -15,7 +15,7 @@ try:
15
15
  HAS_BROTLI = True
16
16
  except ImportError:
17
17
  HAS_BROTLI = False
18
- from typing import Any, Callable, Optional, TYPE_CHECKING, Literal
18
+ from typing import Any, Callable, Optional, Protocol, TYPE_CHECKING, Literal
19
19
  from urllib.parse import urljoin
20
20
  from urllib.request import (
21
21
  HTTPErrorProcessor,
@@ -32,6 +32,11 @@ if TYPE_CHECKING:
32
32
 
33
33
  _FeedType = Literal["rss", "atom", "rdf"]
34
34
 
35
+
36
+ class _ElementValueGetter(Protocol):
37
+ def __call__(self, path: str, attribute: Optional[str] = None) -> Optional[str]: ...
38
+
39
+
35
40
  _UTC = datetime.timezone.utc
36
41
 
37
42
  # Pre-compiled regex patterns for performance
@@ -39,16 +44,16 @@ _RE_XML_DECL_ENCODING = re.compile(
39
44
  r'(<\?xml[^>]*encoding=["\'])([^"\']+)(["\'][^>]*\?>)', re.IGNORECASE
40
45
  )
41
46
  _RE_XML_DECL_ENCODING_BYTES = re.compile(
42
- br'(<\?xml[^>]*encoding=["\'])([^"\']+)(["\'][^>]*\?>)', re.IGNORECASE
47
+ rb'(<\?xml[^>]*encoding=["\'])([^"\']+)(["\'][^>]*\?>)', re.IGNORECASE
43
48
  )
44
- _RE_DOUBLE_XML_DECL_BYTES = re.compile(br"<\?xml\?xml\s+", re.IGNORECASE)
45
- _RE_DOUBLE_CLOSE_BYTES = re.compile(br"\?\?>\s*")
46
- _RE_UNQUOTED_ATTR_BYTES = re.compile(br'(\s+[\w:]+)=([^\s>"\']+)')
49
+ _RE_DOUBLE_XML_DECL_BYTES = re.compile(rb"<\?xml\?xml\s+", re.IGNORECASE)
50
+ _RE_DOUBLE_CLOSE_BYTES = re.compile(rb"\?\?>\s*")
51
+ _RE_UNQUOTED_ATTR_BYTES = re.compile(rb'(\s+[\w:]+)=([^\s>"\']+)')
47
52
  _RE_UTF16_ENCODING_BYTES = re.compile(
48
- br'(<\?xml[^>]*encoding=["\'])utf-16(-le|-be)?(["\'][^>]*\?>)', re.IGNORECASE
53
+ rb'(<\?xml[^>]*encoding=["\'])utf-16(-le|-be)?(["\'][^>]*\?>)', re.IGNORECASE
49
54
  )
50
55
  _RE_UNCLOSED_LINK_BYTES = re.compile(
51
- br"<link([^>]*[^/])>\s*(?=\n\s*<(?!/link\s*>))", re.MULTILINE
56
+ rb"<link([^>]*[^/])>\s*(?=\n\s*<(?!/link\s*>))", re.MULTILINE
52
57
  )
53
58
  _RE_FEB29 = re.compile(r"(\d{4})-02-29")
54
59
  _RE_HTML_TAGS = re.compile(r"<[^>]+>")
@@ -56,6 +61,36 @@ _RE_WHITESPACE = re.compile(r"\s+")
56
61
  _RE_ISO_TZ_NO_COLON = re.compile(r"([+-]\d{2})(\d{2})$")
57
62
  _RE_ISO_TZ_HOUR_ONLY = re.compile(r"([+-]\d{2})$")
58
63
  _RE_ISO_FRACTION = re.compile(r"\.(\d{7,})(?=(?:[+-]\d{2}:?\d{2}|Z|$))", re.IGNORECASE)
64
+ _RE_RFC822 = re.compile(
65
+ r"(?:\w{3},\s+)?(\d{1,2})\s+(\w{3})\s+(\d{4})\s+(\d{2}):(\d{2}):(\d{2})\s+([+-]\d{4}|[A-Z]{2,5})"
66
+ )
67
+ _MONTHS_RFC822: dict[str, int] = {
68
+ "jan": 1,
69
+ "feb": 2,
70
+ "mar": 3,
71
+ "apr": 4,
72
+ "may": 5,
73
+ "jun": 6,
74
+ "jul": 7,
75
+ "aug": 8,
76
+ "sep": 9,
77
+ "oct": 10,
78
+ "nov": 11,
79
+ "dec": 12,
80
+ }
81
+ _TZ_OFFSETS_RFC822: dict[str, int] = {
82
+ "GMT": 0,
83
+ "UTC": 0,
84
+ "UT": 0,
85
+ "EST": -18000,
86
+ "EDT": -14400,
87
+ "CST": -21600,
88
+ "CDT": -18000,
89
+ "MST": -25200,
90
+ "MDT": -21600,
91
+ "PST": -28800,
92
+ "PDT": -25200,
93
+ }
59
94
 
60
95
 
61
96
  class FastFeedParserDict(dict):
@@ -117,7 +152,9 @@ def _clean_feed_bytes(content: bytes) -> bytes:
117
152
  if preview_lower.startswith((b"<?xml", b"<rss", b"<feed", b"<rdf")):
118
153
  return stripped_content
119
154
 
120
- if preview_lower.startswith(b"<!doctype html") or preview_lower.startswith(b"<html"):
155
+ if preview_lower.startswith(b"<!doctype html") or preview_lower.startswith(
156
+ b"<html"
157
+ ):
121
158
  raise ValueError("Content appears to be HTML, not a valid RSS/Atom feed")
122
159
 
123
160
  xml_start_patterns = (
@@ -148,15 +185,17 @@ def _fix_malformed_xml_bytes(content: bytes, actual_encoding: str = "utf-8") ->
148
185
  content = _RE_DOUBLE_CLOSE_BYTES.sub(b"?>", content)
149
186
 
150
187
  # Fix malformed attribute syntax like rss:version=2.0 (missing quotes)
151
- content = _RE_UNQUOTED_ATTR_BYTES.sub(br'\1="\2"', content)
188
+ content = _RE_UNQUOTED_ATTR_BYTES.sub(rb'\1="\2"', content)
152
189
 
153
190
  # Update encoding in XML declaration to match actual encoding when a feed was transcoded.
154
191
  if actual_encoding.lower() != "utf-16":
155
- replacement = br"\1" + actual_encoding.encode("ascii", errors="replace") + br"\3"
192
+ replacement = (
193
+ rb"\1" + actual_encoding.encode("ascii", errors="replace") + rb"\3"
194
+ )
156
195
  content = _RE_UTF16_ENCODING_BYTES.sub(replacement, content)
157
196
 
158
197
  # Fix unclosed link tags - common in Atom feeds
159
- content = _RE_UNCLOSED_LINK_BYTES.sub(br"<link\1/>", content)
198
+ content = _RE_UNCLOSED_LINK_BYTES.sub(rb"<link\1/>", content)
160
199
 
161
200
  return content
162
201
 
@@ -382,24 +421,26 @@ def _maybe_parse_json_feed(content: str | bytes) -> FastFeedParserDict | None:
382
421
  return None
383
422
 
384
423
 
424
+ _STRICT_XML_PARSER = etree.XMLParser(
425
+ ns_clean=True,
426
+ recover=False,
427
+ collect_ids=False,
428
+ resolve_entities=False,
429
+ )
430
+ _RECOVER_XML_PARSER = etree.XMLParser(
431
+ ns_clean=True,
432
+ recover=True,
433
+ collect_ids=False,
434
+ resolve_entities=False,
435
+ )
436
+
437
+
385
438
  def _parse_xml_root(xml_content: bytes) -> _Element:
386
439
  try:
387
- strict_parser = etree.XMLParser(
388
- ns_clean=True,
389
- recover=False,
390
- collect_ids=False,
391
- resolve_entities=False,
392
- )
393
- root = etree.fromstring(xml_content, parser=strict_parser)
440
+ root = etree.fromstring(xml_content, parser=_STRICT_XML_PARSER)
394
441
  except etree.XMLSyntaxError:
395
- recover_parser = etree.XMLParser(
396
- ns_clean=True,
397
- recover=True,
398
- collect_ids=False,
399
- resolve_entities=False,
400
- )
401
442
  try:
402
- root = etree.fromstring(xml_content, parser=recover_parser)
443
+ root = etree.fromstring(xml_content, parser=_RECOVER_XML_PARSER)
403
444
  except etree.XMLSyntaxError as e:
404
445
  raise ValueError(f"Failed to parse XML content: {str(e)}")
405
446
 
@@ -486,16 +527,16 @@ def _raise_for_non_feed_root(
486
527
  if base_msg is None:
487
528
  return
488
529
 
489
- error_msg = _extract_error_message(root, raw_bytes).strip()[:300] or "No error message"
530
+ error_msg = (
531
+ _extract_error_message(root, raw_bytes).strip()[:300] or "No error message"
532
+ )
490
533
 
491
534
  if error_msg != "No error message" and len(error_msg) > 10:
492
535
  raise ValueError(f"{base_msg}: {error_msg[:150]}")
493
536
  raise ValueError(base_msg)
494
537
 
495
538
 
496
- _RE_META_REFRESH_URL = re.compile(
497
- r'url\s*=\s*["\']?\s*([^"\'>\s]+)', re.IGNORECASE
498
- )
539
+ _RE_META_REFRESH_URL = re.compile(r'url\s*=\s*["\']?\s*([^"\'>\s]+)', re.IGNORECASE)
499
540
 
500
541
 
501
542
  def _extract_meta_refresh_url(content: str | bytes, base_url: str) -> str | None:
@@ -544,7 +585,8 @@ def _detect_feed_structure(
544
585
  if channel is None:
545
586
  has_atom_elements = any(
546
587
  isinstance(child.tag, str)
547
- and child.tag in {"entry", "title", "subtitle", "updated", "id", "author", "link"}
588
+ and child.tag
589
+ in {"entry", "title", "subtitle", "updated", "id", "author", "link"}
548
590
  for child in root
549
591
  )
550
592
  if has_atom_elements:
@@ -572,7 +614,9 @@ def _detect_feed_structure(
572
614
  items = []
573
615
  items.append(child)
574
616
  if not items:
575
- items = channel.xpath(".//item") or channel.xpath(".//*[local-name()='item']")
617
+ items = channel.xpath(".//item") or channel.xpath(
618
+ ".//*[local-name()='item']"
619
+ )
576
620
 
577
621
  if not items:
578
622
  items = channel.findall("entry")
@@ -619,7 +663,9 @@ def _detect_feed_structure(
619
663
  if root.tag == "{http://www.w3.org/1999/02/22-rdf-syntax-ns#}RDF":
620
664
  feed_type = "rdf"
621
665
  channel = root
622
- items = channel.findall(".//{http://purl.org/rss/1.0/}item") or channel.findall("item")
666
+ items = channel.findall(".//{http://purl.org/rss/1.0/}item") or channel.findall(
667
+ "item"
668
+ )
623
669
  return feed_type, channel, items, atom_namespace
624
670
 
625
671
  raise ValueError(f"Unknown feed type: {root.tag}")
@@ -642,6 +688,13 @@ def _parse_content(xml_content: str | bytes) -> FastFeedParserDict:
642
688
 
643
689
  feed = _parse_feed_info(channel, feed_type, atom_namespace)
644
690
 
691
+ # Detect once whether media namespace is used anywhere in the document
692
+ has_media_ns = (
693
+ b"search.yahoo.com/mrss" in xml_content
694
+ if isinstance(xml_content, bytes)
695
+ else "search.yahoo.com/mrss" in xml_content
696
+ )
697
+
645
698
  # Parse entries
646
699
  entries: list[FastFeedParserDict] = []
647
700
  feed["entries"] = entries
@@ -650,6 +703,7 @@ def _parse_content(xml_content: str | bytes) -> FastFeedParserDict:
650
703
  item,
651
704
  feed_type,
652
705
  atom_namespace,
706
+ has_media_ns,
653
707
  )
654
708
  # Ensure that titles and descriptions are always present
655
709
  entry["title"] = entry.get("title", "").strip()
@@ -674,6 +728,7 @@ def parse(source: str | bytes) -> FastFeedParserDict:
674
728
  """
675
729
  is_url = isinstance(source, str) and source.startswith(("http://", "https://"))
676
730
  if is_url:
731
+ assert isinstance(source, str)
677
732
  content = _fetch_url_content(source)
678
733
  else:
679
734
  content = source
@@ -683,6 +738,7 @@ def parse(source: str | bytes) -> FastFeedParserDict:
683
738
  except ValueError as e:
684
739
  if not is_url:
685
740
  raise
741
+ assert isinstance(source, str)
686
742
  err_msg = str(e)
687
743
  if "HTML" not in err_msg and "not a valid RSS/Atom feed" not in err_msg:
688
744
  raise
@@ -912,7 +968,9 @@ def _coerce_int_fields(mapping: dict[str, Any], fields: tuple[str, ...]) -> None
912
968
  mapping.pop(field, None)
913
969
 
914
970
 
915
- def _populate_entry_links(entry: FastFeedParserDict, item: _Element, atom_ns: str) -> None:
971
+ def _populate_entry_links(
972
+ entry: FastFeedParserDict, item: _Element, atom_ns: str
973
+ ) -> None:
916
974
  entry_links: list[dict[str, Optional[str]]] = []
917
975
  alternate_link: Optional[dict[str, Optional[str]]] = None
918
976
  for link in item.findall(f"{{{atom_ns}}}link"):
@@ -933,7 +991,9 @@ def _populate_entry_links(entry: FastFeedParserDict, item: _Element, atom_ns: st
933
991
 
934
992
  guid = item.find("guid")
935
993
  guid_text = guid.text.strip() if guid is not None and guid.text else None
936
- is_guid_url = guid_text is not None and guid_text.startswith(("http://", "https://"))
994
+ is_guid_url = guid_text is not None and guid_text.startswith(
995
+ ("http://", "https://")
996
+ )
937
997
 
938
998
  if is_guid_url and "link" not in entry:
939
999
  entry["link"] = guid_text
@@ -974,7 +1034,9 @@ def _populate_entry_content(
974
1034
  entry["content"] = [
975
1035
  {
976
1036
  "type": content_type,
977
- "language": content_el.get("{http://www.w3.org/XML/1998/namespace}lang"),
1037
+ "language": content_el.get(
1038
+ "{http://www.w3.org/XML/1998/namespace}lang"
1039
+ ),
978
1040
  "base": content_el.get("{http://www.w3.org/XML/1998/namespace}base"),
979
1041
  "value": content_value,
980
1042
  }
@@ -998,7 +1060,15 @@ def _populate_entry_content(
998
1060
  if "<" in content_value:
999
1061
  content_value = _RE_HTML_TAGS.sub(" ", content_value[:2048])
1000
1062
  content_value = _html_mod.unescape(content_value)
1001
- content_value = _RE_WHITESPACE.sub(" ", content_value).strip()
1063
+ if (
1064
+ " " in content_value
1065
+ or "\n" in content_value
1066
+ or "\t" in content_value
1067
+ or "\r" in content_value
1068
+ ):
1069
+ content_value = _RE_WHITESPACE.sub(" ", content_value).strip()
1070
+ else:
1071
+ content_value = content_value.strip()
1002
1072
  entry["description"] = content_value[:512]
1003
1073
 
1004
1074
 
@@ -1089,7 +1159,9 @@ def _parse_enclosures(item: _Element) -> list[dict[str, Any]] | None:
1089
1159
  return enclosures or None
1090
1160
 
1091
1161
 
1092
- def _build_rss_item_text_maps(item: _Element) -> tuple[dict[str, Optional[str]], dict[str, Optional[str]]]:
1162
+ def _build_rss_item_text_maps(
1163
+ item: _Element,
1164
+ ) -> tuple[dict[str, Optional[str]], dict[str, Optional[str]]]:
1093
1165
  by_local: dict[str, Optional[str]] = {}
1094
1166
  by_full: dict[str, Optional[str]] = {}
1095
1167
  for child in item:
@@ -1099,15 +1171,21 @@ def _build_rss_item_text_maps(item: _Element) -> tuple[dict[str, Optional[str]],
1099
1171
  text_value = child.text or None
1100
1172
  if tag not in by_full:
1101
1173
  by_full[tag] = text_value
1102
- local = tag.rsplit("}", 1)[-1].lower()
1103
- if ":" in local:
1104
- local = local.split(":", 1)[1]
1174
+ # Fast path: ~80% of RSS tags have no namespace or colon prefix
1175
+ if "{" in tag:
1176
+ local = tag.rsplit("}", 1)[1].lower()
1177
+ elif ":" in tag:
1178
+ local = tag.split(":", 1)[1].lower()
1179
+ else:
1180
+ local = tag.lower()
1105
1181
  if local not in by_local:
1106
1182
  by_local[local] = text_value
1107
1183
  return by_local, by_full
1108
1184
 
1109
1185
 
1110
- def _first_non_empty(mapping: dict[str, Optional[str]], keys: tuple[str, ...]) -> Optional[str]:
1186
+ def _first_non_empty(
1187
+ mapping: dict[str, Optional[str]], keys: tuple[str, ...]
1188
+ ) -> Optional[str]:
1111
1189
  for key in keys:
1112
1190
  value = mapping.get(key)
1113
1191
  if value:
@@ -1118,6 +1196,7 @@ def _first_non_empty(mapping: dict[str, Optional[str]], keys: tuple[str, ...]) -
1118
1196
  def _parse_rss_feed_entry_fast(
1119
1197
  item: _Element,
1120
1198
  atom_ns: str,
1199
+ has_media_ns: bool = True,
1121
1200
  ) -> FastFeedParserDict:
1122
1201
  text_by_local, text_by_full = _build_rss_item_text_maps(item)
1123
1202
 
@@ -1141,19 +1220,27 @@ def _parse_rss_feed_entry_fast(
1141
1220
  if link:
1142
1221
  entry["link"] = link.strip()
1143
1222
 
1144
- published_source = _first_non_empty(text_by_local, ("pubdate", "published", "issued", "date"))
1223
+ published_source = _first_non_empty(
1224
+ text_by_local, ("pubdate", "published", "issued", "date")
1225
+ )
1145
1226
  if published_source:
1146
1227
  published = _parse_date(published_source)
1147
1228
  if published:
1148
1229
  entry["published"] = published
1149
1230
 
1150
- updated_source = _first_non_empty(text_by_local, ("lastbuilddate", "updated", "modified"))
1231
+ updated_source = _first_non_empty(
1232
+ text_by_local, ("lastbuilddate", "updated", "modified")
1233
+ )
1151
1234
  if updated_source:
1152
1235
  updated = _parse_date(updated_source)
1153
1236
  if updated:
1154
1237
  entry["updated"] = updated
1155
1238
 
1156
- if "published" not in entry and rss_guid and not rss_guid.startswith(("http://", "https://")):
1239
+ if (
1240
+ "published" not in entry
1241
+ and rss_guid
1242
+ and not rss_guid.startswith(("http://", "https://"))
1243
+ ):
1157
1244
  guid_date = _parse_date(rss_guid)
1158
1245
  if guid_date:
1159
1246
  entry["published"] = guid_date
@@ -1161,15 +1248,30 @@ def _parse_rss_feed_entry_fast(
1161
1248
  if "updated" in entry and "published" not in entry:
1162
1249
  entry["published"] = entry["updated"]
1163
1250
 
1164
- _populate_entry_links(entry, item, atom_ns)
1251
+ # Inline link population for RSS (avoids redundant findall/find for 98.8% of entries)
1252
+ atom_links = item.findall(f"{{{atom_ns}}}link")
1253
+ if atom_links:
1254
+ # Has atom:link elements - use full logic
1255
+ _populate_entry_links(entry, item, atom_ns)
1256
+ else:
1257
+ # Common RSS case: no atom:link elements
1258
+ entry["links"] = []
1259
+ if (
1260
+ "link" not in entry
1261
+ and rss_guid
1262
+ and rss_guid.startswith(("http://", "https://"))
1263
+ ):
1264
+ entry["link"] = rss_guid
1265
+
1165
1266
  if "id" not in entry and "link" in entry:
1166
1267
  entry["id"] = entry["link"]
1167
1268
 
1168
1269
  _populate_entry_content(entry, item, "rss", atom_ns)
1169
1270
 
1170
- media_contents = _parse_media_content(item)
1171
- if media_contents:
1172
- entry["media_content"] = media_contents
1271
+ if has_media_ns:
1272
+ media_contents = _parse_media_content(item)
1273
+ if media_contents:
1274
+ entry["media_content"] = media_contents
1173
1275
 
1174
1276
  enclosures = _parse_enclosures(item)
1175
1277
  if enclosures:
@@ -1178,7 +1280,11 @@ def _parse_rss_feed_entry_fast(
1178
1280
  author = _first_non_empty(text_by_local, ("author", "creator"))
1179
1281
  if not author:
1180
1282
  atom_author = item.find(f"{{{atom_ns}}}author/{{{atom_ns}}}name")
1181
- author = atom_author.text.strip() if atom_author is not None and atom_author.text else None
1283
+ author = (
1284
+ atom_author.text.strip()
1285
+ if atom_author is not None and atom_author.text
1286
+ else None
1287
+ )
1182
1288
  if author:
1183
1289
  entry["author"] = author.strip()
1184
1290
 
@@ -1196,6 +1302,7 @@ def _parse_rss_feed_entry_fast(
1196
1302
  def _parse_atom_feed_entry_fast(
1197
1303
  item: _Element,
1198
1304
  atom_ns: str,
1305
+ has_media_ns: bool = True,
1199
1306
  ) -> FastFeedParserDict:
1200
1307
  ns = f"{{{atom_ns}}}"
1201
1308
  entry = FastFeedParserDict()
@@ -1266,9 +1373,10 @@ def _parse_atom_feed_entry_fast(
1266
1373
 
1267
1374
  _populate_entry_content(entry, item, "atom", atom_ns)
1268
1375
 
1269
- media_contents = _parse_media_content(item)
1270
- if media_contents:
1271
- entry["media_content"] = media_contents
1376
+ if has_media_ns:
1377
+ media_contents = _parse_media_content(item)
1378
+ if media_contents:
1379
+ entry["media_content"] = media_contents
1272
1380
 
1273
1381
  enclosures = _parse_enclosures(item)
1274
1382
  if enclosures:
@@ -1290,15 +1398,16 @@ def _parse_feed_entry(
1290
1398
  item: _Element,
1291
1399
  feed_type: _FeedType,
1292
1400
  atom_namespace: Optional[str] = None,
1401
+ has_media_ns: bool = True,
1293
1402
  ) -> FastFeedParserDict:
1294
1403
  # Use dynamic atom namespace or fallback to default
1295
1404
  atom_ns = atom_namespace or "http://www.w3.org/2005/Atom"
1296
1405
 
1297
1406
  if feed_type == "rss":
1298
- return _parse_rss_feed_entry_fast(item, atom_ns)
1407
+ return _parse_rss_feed_entry_fast(item, atom_ns, has_media_ns)
1299
1408
 
1300
1409
  if feed_type == "atom":
1301
- return _parse_atom_feed_entry_fast(item, atom_ns)
1410
+ return _parse_atom_feed_entry_fast(item, atom_ns, has_media_ns)
1302
1411
 
1303
1412
  # RDF path uses the generic field machinery
1304
1413
  # Check if this is Atom 0.3 to use different date field names
@@ -1392,7 +1501,11 @@ def _parse_feed_entry(
1392
1501
  entry["updated"] = _parse_date(fallback_updated)
1393
1502
 
1394
1503
  # Try to extract date from GUID as final fallback
1395
- if "published" not in entry and rss_guid and not rss_guid.startswith(("http://", "https://")):
1504
+ if (
1505
+ "published" not in entry
1506
+ and rss_guid
1507
+ and not rss_guid.startswith(("http://", "https://"))
1508
+ ):
1396
1509
  guid_date = _parse_date(rss_guid)
1397
1510
  if guid_date:
1398
1511
  entry["published"] = guid_date
@@ -1412,9 +1525,10 @@ def _parse_feed_entry(
1412
1525
 
1413
1526
  _populate_entry_content(entry, item, feed_type, atom_ns)
1414
1527
 
1415
- media_contents = _parse_media_content(item)
1416
- if media_contents:
1417
- entry["media_content"] = media_contents
1528
+ if has_media_ns:
1529
+ media_contents = _parse_media_content(item)
1530
+ if media_contents:
1531
+ entry["media_content"] = media_contents
1418
1532
 
1419
1533
  enclosures = _parse_enclosures(item)
1420
1534
  if enclosures:
@@ -1427,7 +1541,9 @@ def _parse_feed_entry(
1427
1541
  False,
1428
1542
  )
1429
1543
  if not author:
1430
- author = element_get("{http://purl.org/dc/elements/1.1/}creator") or element_get("author")
1544
+ author = element_get(
1545
+ "{http://purl.org/dc/elements/1.1/}creator"
1546
+ ) or element_get("author")
1431
1547
  if author:
1432
1548
  entry["author"] = author
1433
1549
 
@@ -1442,9 +1558,9 @@ def _parse_feed_entry(
1442
1558
  def _field_value_getter(
1443
1559
  root: _Element,
1444
1560
  feed_type: _FeedType,
1445
- cached_get: Optional[Callable[[str, Optional[str]], Optional[str]]] = None,
1561
+ cached_get: Optional[_ElementValueGetter] = None,
1446
1562
  ) -> Callable[[str, str, str, bool], str | None]:
1447
- get_value = cached_get or _cached_element_value_factory(root)
1563
+ get_value: _ElementValueGetter = cached_get or _cached_element_value_factory(root)
1448
1564
 
1449
1565
  if feed_type == "rss":
1450
1566
 
@@ -1533,7 +1649,11 @@ def _get_element_value(
1533
1649
  el = found
1534
1650
  break
1535
1651
  else:
1536
- prefixed_paths = [f"rss:{path_lower}", f"atom:{path_lower}", f"dc:{path_lower}"]
1652
+ prefixed_paths = [
1653
+ f"rss:{path_lower}",
1654
+ f"atom:{path_lower}",
1655
+ f"dc:{path_lower}",
1656
+ ]
1537
1657
  for child in root:
1538
1658
  if not isinstance(child.tag, str):
1539
1659
  continue
@@ -1553,7 +1673,7 @@ def _get_element_value(
1553
1673
 
1554
1674
  def _cached_element_value_factory(
1555
1675
  root: _Element,
1556
- ) -> Callable[[str, Optional[str]], Optional[str]]:
1676
+ ) -> _ElementValueGetter:
1557
1677
  """Create a closure with a child tag index for fast namespace-prefix lookups."""
1558
1678
  # Build child tag index once: O(children) instead of O(children × misses)
1559
1679
  child_index: dict[str, _Element] = {}
@@ -1562,7 +1682,9 @@ def _cached_element_value_factory(
1562
1682
  child_index[child.tag.lower()] = child
1563
1683
 
1564
1684
  def getter(path: str, attribute: Optional[str] = None) -> Optional[str]:
1565
- return _get_element_value(root, path, attribute=attribute, child_index=child_index)
1685
+ return _get_element_value(
1686
+ root, path, attribute=attribute, child_index=child_index
1687
+ )
1566
1688
 
1567
1689
  return getter
1568
1690
 
@@ -1591,7 +1713,13 @@ def _normalize_iso_datetime_string(value: str) -> str:
1591
1713
  if cleaned.endswith(("Z", "z")):
1592
1714
  cleaned = cleaned[:-1] + "+00:00"
1593
1715
 
1594
- if " " in cleaned and "T" not in cleaned[:11] and len(cleaned) >= 10 and cleaned[4] == "-" and cleaned[0:4].isdigit():
1716
+ if (
1717
+ " " in cleaned
1718
+ and "T" not in cleaned[:11]
1719
+ and len(cleaned) >= 10
1720
+ and cleaned[4] == "-"
1721
+ and cleaned[0:4].isdigit()
1722
+ ):
1595
1723
  date_part, rest = cleaned.split(" ", 1)
1596
1724
  if rest and rest[0].isdigit():
1597
1725
  cleaned = f"{date_part}T{rest}"
@@ -1616,8 +1744,64 @@ def _ensure_utc(dt: datetime.datetime) -> Optional[datetime.datetime]:
1616
1744
  return None
1617
1745
 
1618
1746
 
1747
+ def _fast_rfc822_to_iso(value: str) -> Optional[str]:
1748
+ """Fast RFC-822 date to ISO string, bypassing datetime objects for UTC dates."""
1749
+ m = _RE_RFC822.match(value)
1750
+ if not m:
1751
+ return None
1752
+ day, mon_str, year, hour, minute, second, tz = m.groups()
1753
+ month = _MONTHS_RFC822.get(mon_str.lower())
1754
+ if month is None:
1755
+ return None
1756
+ if tz[0] in "+-":
1757
+ tz_offset_seconds = (int(tz[1:3]) * 3600 + int(tz[3:5]) * 60) * (
1758
+ 1 if tz[0] == "+" else -1
1759
+ )
1760
+ else:
1761
+ tz_offset_seconds = _TZ_OFFSETS_RFC822.get(tz)
1762
+ if tz_offset_seconds is None:
1763
+ return None # Unknown tz name, fall through to full parser
1764
+ # Python requires offset strictly between -24h and +24h
1765
+ if not (-86400 < tz_offset_seconds < 86400):
1766
+ return None
1767
+ d = int(day)
1768
+ h = int(hour)
1769
+ mi = int(minute)
1770
+ s = int(second)
1771
+ # Hour 24 is invalid (even ISO only allows 24:00:00); roll to next day at 00:mm:ss
1772
+ if h == 24:
1773
+ base = datetime.date(int(year), month, d) + datetime.timedelta(days=1)
1774
+ h = 0
1775
+ if tz_offset_seconds == 0:
1776
+ return f"{base.year:04d}-{base.month:02d}-{base.day:02d}T{h:02d}:{mi:02d}:{s:02d}+00:00"
1777
+ dt = datetime.datetime(
1778
+ base.year,
1779
+ base.month,
1780
+ base.day,
1781
+ h,
1782
+ mi,
1783
+ s,
1784
+ tzinfo=datetime.timezone(datetime.timedelta(seconds=tz_offset_seconds)),
1785
+ )
1786
+ utc = dt.astimezone(_UTC)
1787
+ return f"{utc.year:04d}-{utc.month:02d}-{utc.day:02d}T{utc.hour:02d}:{utc.minute:02d}:{utc.second:02d}+00:00"
1788
+ if tz_offset_seconds == 0:
1789
+ return f"{year}-{month:02d}-{d:02d}T{hour}:{minute}:{second}+00:00"
1790
+ dt = datetime.datetime(
1791
+ int(year),
1792
+ month,
1793
+ d,
1794
+ h,
1795
+ mi,
1796
+ s,
1797
+ tzinfo=datetime.timezone(datetime.timedelta(seconds=tz_offset_seconds)),
1798
+ )
1799
+ utc = dt.astimezone(_UTC)
1800
+ return f"{utc.year:04d}-{utc.month:02d}-{utc.day:02d}T{utc.hour:02d}:{utc.minute:02d}:{utc.second:02d}+00:00"
1801
+
1802
+
1619
1803
  def _parsedate_to_utc(value: str) -> Optional[datetime.datetime]:
1620
- """Fast RFC-822 / RFC-2822 parsing via email.utils."""
1804
+ """RFC-822 / RFC-2822 parsing via email.utils (fallback)."""
1621
1805
  try:
1622
1806
  parsed = parsedate_to_datetime(value)
1623
1807
  except (TypeError, ValueError, IndexError):
@@ -1690,7 +1874,9 @@ def _slow_dateparser(value: str) -> Optional[datetime.datetime]:
1690
1874
  except ImportError:
1691
1875
  return None
1692
1876
  try:
1693
- return _dateparser.parse(value, languages=["en"], settings={**_DATEPARSER_SETTINGS})
1877
+ return _dateparser.parse(
1878
+ value, languages=["en"], settings={**_DATEPARSER_SETTINGS}
1879
+ )
1694
1880
  except (ValueError, TypeError):
1695
1881
  return None
1696
1882
 
@@ -1717,8 +1903,9 @@ def _parse_date(date_str: str) -> Optional[str]:
1717
1903
  last = candidate[-1]
1718
1904
  # Most common: ends with 'Z' (e.g., 2024-01-15T10:30:00Z)
1719
1905
  if last in ("Z", "z"):
1906
+ iso = candidate[:-1] + "+00:00"
1720
1907
  try:
1721
- dt = datetime.datetime.fromisoformat(candidate[:-1] + "+00:00")
1908
+ dt = datetime.datetime.fromisoformat(iso)
1722
1909
  return dt.isoformat()
1723
1910
  except ValueError:
1724
1911
  pass # Fall through to full parsing
@@ -1726,7 +1913,9 @@ def _parse_date(date_str: str) -> Optional[str]:
1726
1913
  elif clen > 6 and candidate[-6] in ("+", "-") and candidate[-3] == ":":
1727
1914
  try:
1728
1915
  dt = datetime.datetime.fromisoformat(candidate)
1729
- utc_dt = dt.replace(tzinfo=_UTC) if dt.tzinfo is None else dt.astimezone(_UTC)
1916
+ if dt.tzinfo is _UTC:
1917
+ return dt.isoformat()
1918
+ utc_dt = dt.astimezone(_UTC)
1730
1919
  return utc_dt.isoformat()
1731
1920
  except (ValueError, OverflowError):
1732
1921
  pass # Fall through to full parsing
@@ -1743,14 +1932,23 @@ def _parse_date(date_str: str) -> Optional[str]:
1743
1932
  if not ((year % 4 == 0 and year % 100 != 0) or (year % 400 == 0)):
1744
1933
  candidate = candidate.replace(f"{year}-02-29", f"{year}-02-28")
1745
1934
 
1746
- if "24:00" in candidate:
1747
- candidate = candidate.replace("24:00:00", "00:00:00").replace(
1748
- " 24:00", " 00:00"
1749
- )
1935
+ if "T24:" in candidate or " 24:" in candidate:
1936
+ m24 = re.search(r"(\d{4}-\d{2}-\d{2})[T ]24:(\d{2}):(\d{2})", candidate)
1937
+ if m24:
1938
+ base = datetime.date.fromisoformat(m24.group(1))
1939
+ mins, secs = int(m24.group(2)), int(m24.group(3))
1940
+ next_day = base + datetime.timedelta(days=1)
1941
+ candidate = (
1942
+ candidate[: m24.start()]
1943
+ + f"{next_day}T00:{mins:02d}:{secs:02d}"
1944
+ + candidate[m24.end() :]
1945
+ )
1750
1946
 
1751
1947
  dt: Optional[datetime.datetime] = None
1752
1948
 
1753
- is_iso_like = len(candidate) >= 10 and candidate[4] == "-" and candidate[0:4].isdigit()
1949
+ is_iso_like = (
1950
+ len(candidate) >= 10 and candidate[4] == "-" and candidate[0:4].isdigit()
1951
+ )
1754
1952
  if is_iso_like:
1755
1953
  iso_candidate = _normalize_iso_datetime_string(candidate)
1756
1954
  try:
@@ -1762,6 +1960,10 @@ def _parse_date(date_str: str) -> Optional[str]:
1762
1960
  if utc_dt is not None:
1763
1961
  return utc_dt.isoformat()
1764
1962
 
1963
+ rfc822_result = _fast_rfc822_to_iso(candidate)
1964
+ if rfc822_result is not None:
1965
+ return rfc822_result
1966
+
1765
1967
  dt = _parsedate_to_utc(candidate)
1766
1968
  if dt is not None:
1767
1969
  return dt.isoformat()
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.1
2
2
  Name: fastfeedparser
3
- Version: 0.5.2
3
+ Version: 0.5.4
4
4
  Summary: High performance RSS, Atom, JSON and RDF feed parser in Python
5
5
  Home-page: https://github.com/kagisearch/fastfeedparser
6
6
  Author: Vladimir Prelovac
@@ -34,12 +34,18 @@ def test_parse_bytes_with_non_utf8_encoding():
34
34
 
35
35
  def test_meta_refresh_extraction():
36
36
  html = '<!doctype html><html><head><meta http-equiv=refresh content="0; url=https://example.com/feed.xml"></head></html>'
37
- assert _extract_meta_refresh_url(html, "https://example.com/feed/") == "https://example.com/feed.xml"
37
+ assert (
38
+ _extract_meta_refresh_url(html, "https://example.com/feed/")
39
+ == "https://example.com/feed.xml"
40
+ )
38
41
 
39
42
 
40
43
  def test_meta_refresh_relative_url():
41
44
  html = b'<html><head><meta http-equiv="refresh" content="0;url=/index.xml"></head></html>'
42
- assert _extract_meta_refresh_url(html, "https://example.com/feed/") == "https://example.com/index.xml"
45
+ assert (
46
+ _extract_meta_refresh_url(html, "https://example.com/feed/")
47
+ == "https://example.com/index.xml"
48
+ )
43
49
 
44
50
 
45
51
  def test_meta_refresh_none_when_missing():
@@ -50,4 +56,3 @@ def test_meta_refresh_none_when_missing():
50
56
  def test_meta_refresh_none_when_same_url():
51
57
  html = '<html><head><meta http-equiv="refresh" content="0; url=https://example.com/"></head></html>'
52
58
  assert _extract_meta_refresh_url(html, "https://example.com/") is None
53
-
@@ -1,7 +0,0 @@
1
- [build-system]
2
- requires = ["setuptools~=67.0", "wheel"]
3
- build-backend = "setuptools.build_meta"
4
-
5
- [tool.pytest.ini_options]
6
- testpaths = ["tests"]
7
-
File without changes
File without changes