fastfeedparser 0.5.0__tar.gz → 0.5.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.1
2
2
  Name: fastfeedparser
3
- Version: 0.5.0
3
+ Version: 0.5.2
4
4
  Summary: High performance RSS, Atom, JSON and RDF feed parser in Python
5
5
  Home-page: https://github.com/kagisearch/fastfeedparser
6
6
  Author: Vladimir Prelovac
@@ -2,3 +2,6 @@
2
2
  requires = ["setuptools~=67.0", "wheel"]
3
3
  build-backend = "setuptools.build_meta"
4
4
 
5
+ [tool.pytest.ini_options]
6
+ testpaths = ["tests"]
7
+
@@ -1,6 +1,6 @@
1
1
  [metadata]
2
2
  name = fastfeedparser
3
- version = 0.5.0
3
+ version = 0.5.2
4
4
  author = Vladimir Prelovac
5
5
  author_email = vlad@kagi.com
6
6
  description = High performance RSS, Atom, JSON and RDF feed parser in Python
@@ -1,4 +1,4 @@
1
1
  from .main import parse, FastFeedParserDict
2
2
 
3
- __version__ = "0.5.0"
3
+ __version__ = "0.5.1"
4
4
  __all__ = ["parse", "FastFeedParserDict"]
@@ -41,21 +41,12 @@ _RE_XML_DECL_ENCODING = re.compile(
41
41
  _RE_XML_DECL_ENCODING_BYTES = re.compile(
42
42
  br'(<\?xml[^>]*encoding=["\'])([^"\']+)(["\'][^>]*\?>)', re.IGNORECASE
43
43
  )
44
- _RE_DOUBLE_XML_DECL = re.compile(r"<\?xml\?xml\s+", re.IGNORECASE)
45
44
  _RE_DOUBLE_XML_DECL_BYTES = re.compile(br"<\?xml\?xml\s+", re.IGNORECASE)
46
- _RE_DOUBLE_CLOSE = re.compile(r"\?\?>\s*")
47
45
  _RE_DOUBLE_CLOSE_BYTES = re.compile(br"\?\?>\s*")
48
- _RE_UNQUOTED_ATTR = re.compile(r'(\s+[\w:]+)=([^\s>"\']+)')
49
46
  _RE_UNQUOTED_ATTR_BYTES = re.compile(br'(\s+[\w:]+)=([^\s>"\']+)')
50
- _RE_UTF16_ENCODING = re.compile(
51
- r'(<\?xml[^>]*encoding=["\'])utf-16(-le|-be)?(["\'][^>]*\?>)', re.IGNORECASE
52
- )
53
47
  _RE_UTF16_ENCODING_BYTES = re.compile(
54
48
  br'(<\?xml[^>]*encoding=["\'])utf-16(-le|-be)?(["\'][^>]*\?>)', re.IGNORECASE
55
49
  )
56
- _RE_UNCLOSED_LINK = re.compile(
57
- r"<link([^>]*[^/])>\s*(?=\n\s*<(?!/link\s*>))", re.MULTILINE
58
- )
59
50
  _RE_UNCLOSED_LINK_BYTES = re.compile(
60
51
  br"<link([^>]*[^/])>\s*(?=\n\s*<(?!/link\s*>))", re.MULTILINE
61
52
  )
@@ -112,36 +103,6 @@ def _ensure_utf8_xml_declaration(content: str) -> str:
112
103
  return _RE_XML_DECL_ENCODING.sub(r"\1utf-8\3", content, count=1)
113
104
 
114
105
 
115
- def _clean_feed_text(content: str) -> str:
116
- """Clean feed text by extracting the XML document (if it's embedded in junk)."""
117
- stripped_content = content.lstrip()
118
- stripped_lower = stripped_content[:2000].lower()
119
- if stripped_lower.startswith(("<?xml", "<rss", "<feed", "<rdf")):
120
- return stripped_content
121
-
122
- if stripped_lower.startswith("<!doctype html") or stripped_lower.startswith("<html"):
123
- raise ValueError("Content appears to be HTML, not a valid RSS/Atom feed")
124
-
125
- xml_start_patterns = (
126
- "<?xml",
127
- "<rss",
128
- "<feed",
129
- "<rdf:rdf",
130
- "<?xml-stylesheet",
131
- )
132
-
133
- content_lines = content.splitlines()
134
- for i, line in enumerate(content_lines):
135
- line_stripped = line.strip().lower()
136
- if any(line_stripped.startswith(pattern) for pattern in xml_start_patterns):
137
- return "\n".join(content_lines[i:])
138
-
139
- if "<script>" in stripped_lower or "<body>" in stripped_lower:
140
- raise ValueError("Content appears to be HTML, not a valid RSS/Atom feed")
141
-
142
- return content
143
-
144
-
145
106
  def _clean_feed_bytes(content: bytes) -> bytes:
146
107
  """Clean feed bytes by extracting the XML document (if it's embedded in junk)."""
147
108
  stripped_content = content.lstrip()
@@ -224,60 +185,9 @@ def _prepare_xml_bytes(xml_content: str | bytes) -> bytes:
224
185
  cleaned = _fix_malformed_xml_bytes(cleaned, actual_encoding=actual_encoding)
225
186
  return cleaned
226
187
 
227
- cleaned_text = _clean_feed_text(xml_content)
228
- if not cleaned_text.strip():
229
- raise ValueError("Empty content")
230
-
231
- needs_fixing = (
232
- "?xml?xml" in cleaned_text[:200]
233
- or "??>" in cleaned_text[:200]
234
- or (
235
- "rss:" in cleaned_text[:500] and "xmlns:rss" not in cleaned_text[:1000]
236
- )
237
- or ("utf-16" in cleaned_text[:200].lower())
238
- )
239
- if needs_fixing:
240
- cleaned_text = _fix_malformed_xml(cleaned_text, actual_encoding="utf-8")
241
-
242
- cleaned_text = _ensure_utf8_xml_declaration(cleaned_text)
243
- return cleaned_text.encode("utf-8", errors="replace")
244
-
245
-
246
- def _fix_malformed_xml(content: str, actual_encoding: str = "utf-8") -> str:
247
- """Fix common malformed XML issues in feeds.
248
-
249
- Some feeds have malformed XML like unclosed link tags or other issues
250
- that can be automatically corrected.
251
-
252
- Args:
253
- content: The XML content as a string
254
- actual_encoding: The actual encoding used (default: utf-8)
255
- """
256
- # Fix double XML declarations like "<?xml?xml version="1.0"?>"
257
- # This is found in dylanharris.org feed
258
- content = _RE_DOUBLE_XML_DECL.sub(r"<?xml ", content)
259
-
260
- # Fix double closing ?> in XML declaration like "??>>"
261
- content = _RE_DOUBLE_CLOSE.sub(r"?>", content)
262
-
263
- # Fix malformed attribute syntax like rss:version=2.0 (missing quotes)
264
- # This is found in dylanharris.org feed
265
- content = _RE_UNQUOTED_ATTR.sub(r'\1="\2"', content)
266
-
267
- # Update encoding in XML declaration to match actual encoding
268
- # This handles cases where content was transcoded from UTF-16 to UTF-8
269
- if actual_encoding.lower() != "utf-16":
270
- content = _RE_UTF16_ENCODING.sub(rf"\1{actual_encoding}\3", content)
271
-
272
- # Fix unclosed link tags - common in Atom feeds
273
- # Pattern: <link ...> followed by whitespace and another tag (not </link>)
274
- # should be <link .../>
275
- # Only fix link tags that are clearly malformed:
276
- # - End with > instead of />
277
- # - Are followed by whitespace and another tag (not a closing </link>)
278
- content = _RE_UNCLOSED_LINK.sub(r"<link\1/>", content)
279
-
280
- return content
188
+ # Str input: fix encoding declaration, encode to bytes, then use bytes path.
189
+ xml_content = _ensure_utf8_xml_declaration(xml_content)
190
+ return _prepare_xml_bytes(xml_content.encode("utf-8", errors="replace"))
281
191
 
282
192
 
283
193
  def _parse_json_feed(json_data: dict) -> FastFeedParserDict:
@@ -556,46 +466,31 @@ def _extract_error_message(root: _Element, raw_bytes: Optional[bytes] = None) ->
556
466
  return error_msg
557
467
 
558
468
 
469
+ _NON_FEED_MESSAGES: dict[str, str] = {
470
+ "html": "Received HTML page instead of feed",
471
+ "div": "Received HTML fragment instead of feed",
472
+ "body": "Received HTML fragment instead of feed",
473
+ "br": "Received HTML fragment instead of feed",
474
+ "status": "Feed server returned status message",
475
+ "error": "Feed server returned error",
476
+ "opml": "Received OPML document instead of feed (OPML is an outline format, not a feed)",
477
+ "urlset": "Received XML sitemap instead of feed (sitemap is for search engines, not a feed)",
478
+ "sitemapindex": "Received XML sitemap instead of feed (sitemap is for search engines, not a feed)",
479
+ }
480
+
481
+
559
482
  def _raise_for_non_feed_root(
560
483
  root: _Element, root_tag_local: str, raw_bytes: Optional[bytes] = None
561
484
  ) -> None:
562
- non_feed_tags = {
563
- "status", "error", "html", "opml", "br", "div", "body",
564
- "urlset", "sitemapindex",
565
- }
566
- if root_tag_local not in non_feed_tags:
485
+ base_msg = _NON_FEED_MESSAGES.get(root_tag_local)
486
+ if base_msg is None:
567
487
  return
568
488
 
569
489
  error_msg = _extract_error_message(root, raw_bytes).strip()[:300] or "No error message"
570
490
 
571
- if root_tag_local == "html":
572
- if error_msg != "No error message" and len(error_msg) > 10:
573
- raise ValueError(f"Received HTML page instead of feed: {error_msg[:150]}")
574
- raise ValueError(
575
- "Received HTML page instead of feed (possible redirect, 404, or server error)"
576
- )
577
- if root_tag_local in {"div", "body"}:
578
- if error_msg != "No error message" and len(error_msg) > 10:
579
- raise ValueError(f"Received HTML fragment instead of feed: {error_msg[:150]}")
580
- raise ValueError("Received HTML fragment instead of feed")
581
- if root_tag_local == "br":
582
- if error_msg != "No error message" and len(error_msg) > 10:
583
- raise ValueError(f"Received HTML error instead of feed: {error_msg[:150]}")
584
- raise ValueError("Received HTML fragment instead of feed")
585
- if root_tag_local == "status":
586
- raise ValueError(f"Feed server returned status message: {error_msg}")
587
- if root_tag_local == "error":
588
- if error_msg != "No error message":
589
- raise ValueError(f"Feed server returned error: {error_msg}")
590
- raise ValueError("Feed server returned error (no details provided)")
591
- if root_tag_local == "opml":
592
- raise ValueError(
593
- "Received OPML document instead of feed (OPML is an outline format, not a feed)"
594
- )
595
- if root_tag_local in {"urlset", "sitemapindex"}:
596
- raise ValueError(
597
- "Received XML sitemap instead of feed (sitemap is for search engines, not a feed)"
598
- )
491
+ if error_msg != "No error message" and len(error_msg) > 10:
492
+ raise ValueError(f"{base_msg}: {error_msg[:150]}")
493
+ raise ValueError(base_msg)
599
494
 
600
495
 
601
496
  _RE_META_REFRESH_URL = re.compile(
@@ -730,24 +625,6 @@ def _detect_feed_structure(
730
625
  raise ValueError(f"Unknown feed type: {root.tag}")
731
626
 
732
627
 
733
- def _should_parse_media_content(root: _Element, xml_content: bytes) -> bool:
734
- """Check if feed likely contains Media RSS fields."""
735
- ns_values = root.nsmap.values() if root.nsmap else ()
736
- for ns_value in ns_values:
737
- if not ns_value:
738
- continue
739
- if "search.yahoo.com/mrss" in ns_value:
740
- return True
741
-
742
- # Fallback for feeds with undeclared/late namespace usage.
743
- return b"search.yahoo.com/mrss" in xml_content or b"<media:" in xml_content
744
-
745
-
746
- def _should_parse_enclosures(feed_type: _FeedType, xml_content: bytes) -> bool:
747
- """Check if feed likely contains RSS enclosure elements."""
748
- return feed_type == "rss" and b"<enclosure" in xml_content
749
-
750
-
751
628
  def _parse_content(xml_content: str | bytes) -> FastFeedParserDict:
752
629
  """Parse feed content (XML or JSON) that has already been fetched."""
753
630
  json_feed = _maybe_parse_json_feed(xml_content)
@@ -762,8 +639,6 @@ def _parse_content(xml_content: str | bytes) -> FastFeedParserDict:
762
639
  feed_type, channel, items, atom_namespace = _detect_feed_structure(
763
640
  root, xml_content, root_tag_local
764
641
  )
765
- parse_media_content = _should_parse_media_content(root, xml_content)
766
- parse_enclosures = _should_parse_enclosures(feed_type, xml_content)
767
642
 
768
643
  feed = _parse_feed_info(channel, feed_type, atom_namespace)
769
644
 
@@ -775,8 +650,6 @@ def _parse_content(xml_content: str | bytes) -> FastFeedParserDict:
775
650
  item,
776
651
  feed_type,
777
652
  atom_namespace,
778
- parse_media_content=parse_media_content,
779
- parse_enclosures=parse_enclosures,
780
653
  )
781
654
  # Ensure that titles and descriptions are always present
782
655
  entry["title"] = entry.get("title", "").strip()
@@ -1223,7 +1096,7 @@ def _build_rss_item_text_maps(item: _Element) -> tuple[dict[str, Optional[str]],
1223
1096
  tag = child.tag
1224
1097
  if not isinstance(tag, str):
1225
1098
  continue
1226
- text_value = child.text.strip() if child.text else None
1099
+ text_value = child.text or None
1227
1100
  if tag not in by_full:
1228
1101
  by_full[tag] = text_value
1229
1102
  local = tag.rsplit("}", 1)[-1].lower()
@@ -1245,8 +1118,6 @@ def _first_non_empty(mapping: dict[str, Optional[str]], keys: tuple[str, ...]) -
1245
1118
  def _parse_rss_feed_entry_fast(
1246
1119
  item: _Element,
1247
1120
  atom_ns: str,
1248
- parse_media_content: bool = True,
1249
- parse_enclosures: bool = True,
1250
1121
  ) -> FastFeedParserDict:
1251
1122
  text_by_local, text_by_full = _build_rss_item_text_maps(item)
1252
1123
 
@@ -1268,7 +1139,7 @@ def _parse_rss_feed_entry_fast(
1268
1139
 
1269
1140
  link = text_by_local.get("link")
1270
1141
  if link:
1271
- entry["link"] = link
1142
+ entry["link"] = link.strip()
1272
1143
 
1273
1144
  published_source = _first_non_empty(text_by_local, ("pubdate", "published", "issued", "date"))
1274
1145
  if published_source:
@@ -1282,7 +1153,7 @@ def _parse_rss_feed_entry_fast(
1282
1153
  if updated:
1283
1154
  entry["updated"] = updated
1284
1155
 
1285
- if "published" not in entry and rss_guid:
1156
+ if "published" not in entry and rss_guid and not rss_guid.startswith(("http://", "https://")):
1286
1157
  guid_date = _parse_date(rss_guid)
1287
1158
  if guid_date:
1288
1159
  entry["published"] = guid_date
@@ -1296,26 +1167,24 @@ def _parse_rss_feed_entry_fast(
1296
1167
 
1297
1168
  _populate_entry_content(entry, item, "rss", atom_ns)
1298
1169
 
1299
- if parse_media_content:
1300
- media_contents = _parse_media_content(item)
1301
- if media_contents:
1302
- entry["media_content"] = media_contents
1170
+ media_contents = _parse_media_content(item)
1171
+ if media_contents:
1172
+ entry["media_content"] = media_contents
1303
1173
 
1304
- if parse_enclosures:
1305
- enclosures = _parse_enclosures(item)
1306
- if enclosures:
1307
- entry["enclosures"] = enclosures
1174
+ enclosures = _parse_enclosures(item)
1175
+ if enclosures:
1176
+ entry["enclosures"] = enclosures
1308
1177
 
1309
1178
  author = _first_non_empty(text_by_local, ("author", "creator"))
1310
1179
  if not author:
1311
1180
  atom_author = item.find(f"{{{atom_ns}}}author/{{{atom_ns}}}name")
1312
1181
  author = atom_author.text.strip() if atom_author is not None and atom_author.text else None
1313
1182
  if author:
1314
- entry["author"] = author
1183
+ entry["author"] = author.strip()
1315
1184
 
1316
1185
  comments = text_by_local.get("comments")
1317
1186
  if comments:
1318
- entry["comments"] = comments
1187
+ entry["comments"] = comments.strip()
1319
1188
 
1320
1189
  tags = _parse_tags(item, "rss", atom_ns)
1321
1190
  if tags:
@@ -1324,25 +1193,114 @@ def _parse_rss_feed_entry_fast(
1324
1193
  return entry
1325
1194
 
1326
1195
 
1196
+ def _parse_atom_feed_entry_fast(
1197
+ item: _Element,
1198
+ atom_ns: str,
1199
+ ) -> FastFeedParserDict:
1200
+ ns = f"{{{atom_ns}}}"
1201
+ entry = FastFeedParserDict()
1202
+
1203
+ # ID
1204
+ el = item.find(ns + "id")
1205
+ if el is not None and el.text:
1206
+ entry["id"] = el.text.strip()
1207
+
1208
+ # Title
1209
+ el = item.find(ns + "title")
1210
+ if el is not None and el.text:
1211
+ entry["title"] = el.text.strip()
1212
+
1213
+ # Description (summary)
1214
+ el = item.find(ns + "summary")
1215
+ if el is not None and el.text:
1216
+ entry["description"] = el.text.strip()
1217
+
1218
+ # Link (href attribute)
1219
+ el = item.find(ns + "link")
1220
+ if el is not None:
1221
+ href = el.get("href")
1222
+ if href:
1223
+ entry["link"] = href.strip()
1224
+
1225
+ # Dates: Atom 1.0 uses published/updated, Atom 0.3 uses issued/modified
1226
+ is_atom_03 = atom_ns == "http://purl.org/atom/ns#"
1227
+ pub_tag = "issued" if is_atom_03 else "published"
1228
+ upd_tag = "modified" if is_atom_03 else "updated"
1229
+ pub_fallback_tag = "published" if is_atom_03 else "issued"
1230
+ upd_fallback_tag = "updated" if is_atom_03 else "modified"
1231
+
1232
+ el = item.find(ns + pub_tag)
1233
+ if el is not None and el.text:
1234
+ published = _parse_date(el.text)
1235
+ if published:
1236
+ entry["published"] = published
1237
+
1238
+ el = item.find(ns + upd_tag)
1239
+ if el is not None and el.text:
1240
+ updated = _parse_date(el.text)
1241
+ if updated:
1242
+ entry["updated"] = updated
1243
+
1244
+ # Fallback date fields for mixed namespace scenarios
1245
+ if "published" not in entry:
1246
+ el = item.find(ns + pub_fallback_tag)
1247
+ if el is not None and el.text:
1248
+ published = _parse_date(el.text)
1249
+ if published:
1250
+ entry["published"] = published
1251
+
1252
+ if "updated" not in entry:
1253
+ el = item.find(ns + upd_fallback_tag)
1254
+ if el is not None and el.text:
1255
+ updated = _parse_date(el.text)
1256
+ if updated:
1257
+ entry["updated"] = updated
1258
+
1259
+ if "updated" in entry and "published" not in entry:
1260
+ entry["published"] = entry["updated"]
1261
+
1262
+ _populate_entry_links(entry, item, atom_ns)
1263
+
1264
+ if "id" not in entry and "link" in entry:
1265
+ entry["id"] = entry["link"]
1266
+
1267
+ _populate_entry_content(entry, item, "atom", atom_ns)
1268
+
1269
+ media_contents = _parse_media_content(item)
1270
+ if media_contents:
1271
+ entry["media_content"] = media_contents
1272
+
1273
+ enclosures = _parse_enclosures(item)
1274
+ if enclosures:
1275
+ entry["enclosures"] = enclosures
1276
+
1277
+ # Author
1278
+ el = item.find(ns + "author/" + ns + "name")
1279
+ if el is not None and el.text:
1280
+ entry["author"] = el.text.strip()
1281
+
1282
+ tags = _parse_tags(item, "atom", atom_ns)
1283
+ if tags:
1284
+ entry["tags"] = tags
1285
+
1286
+ return entry
1287
+
1288
+
1327
1289
  def _parse_feed_entry(
1328
1290
  item: _Element,
1329
1291
  feed_type: _FeedType,
1330
1292
  atom_namespace: Optional[str] = None,
1331
- *,
1332
- parse_media_content: bool = True,
1333
- parse_enclosures: bool = True,
1334
1293
  ) -> FastFeedParserDict:
1335
1294
  # Use dynamic atom namespace or fallback to default
1336
1295
  atom_ns = atom_namespace or "http://www.w3.org/2005/Atom"
1337
1296
 
1338
1297
  if feed_type == "rss":
1339
- return _parse_rss_feed_entry_fast(
1340
- item,
1341
- atom_ns,
1342
- parse_media_content=parse_media_content,
1343
- parse_enclosures=parse_enclosures,
1344
- )
1298
+ return _parse_rss_feed_entry_fast(item, atom_ns)
1345
1299
 
1300
+ if feed_type == "atom":
1301
+ return _parse_atom_feed_entry_fast(item, atom_ns)
1302
+
1303
+ # RDF path uses the generic field machinery
1346
1304
  # Check if this is Atom 0.3 to use different date field names
1347
1305
  is_atom_03 = atom_ns == "http://purl.org/atom/ns#"
1348
1306
 
@@ -1434,8 +1392,7 @@ def _parse_feed_entry(
1434
1392
  entry["updated"] = _parse_date(fallback_updated)
1435
1393
 
1436
1394
  # Try to extract date from GUID as final fallback
1437
- if "published" not in entry and rss_guid:
1438
- # Check if GUID contains date information
1395
+ if "published" not in entry and rss_guid and not rss_guid.startswith(("http://", "https://")):
1439
1396
  guid_date = _parse_date(rss_guid)
1440
1397
  if guid_date:
1441
1398
  entry["published"] = guid_date
@@ -1455,15 +1412,13 @@ def _parse_feed_entry(
1455
1412
 
1456
1413
  _populate_entry_content(entry, item, feed_type, atom_ns)
1457
1414
 
1458
- if parse_media_content:
1459
- media_contents = _parse_media_content(item)
1460
- if media_contents:
1461
- entry["media_content"] = media_contents
1415
+ media_contents = _parse_media_content(item)
1416
+ if media_contents:
1417
+ entry["media_content"] = media_contents
1462
1418
 
1463
- if parse_enclosures:
1464
- enclosures = _parse_enclosures(item)
1465
- if enclosures:
1466
- entry["enclosures"] = enclosures
1419
+ enclosures = _parse_enclosures(item)
1420
+ if enclosures:
1421
+ entry["enclosures"] = enclosures
1467
1422
 
1468
1423
  author = get_field_value(
1469
1424
  "author",
@@ -1618,6 +1573,14 @@ def _normalize_iso_datetime_string(value: str) -> str:
1618
1573
  if not cleaned:
1619
1574
  return cleaned
1620
1575
 
1576
+ # Fast path: 'Z' suffix (most common in Atom feeds)
1577
+ if cleaned[-1] in ("Z", "z"):
1578
+ return cleaned[:-1] + "+00:00"
1579
+
1580
+ # Fast path: already has proper +HH:MM or -HH:MM timezone
1581
+ if len(cleaned) > 6 and cleaned[-6] in ("+", "-") and cleaned[-3] == ":":
1582
+ return cleaned
1583
+
1621
1584
  upper_cleaned = cleaned.upper()
1622
1585
  for suffix in (" UTC", " GMT", " Z"):
1623
1586
  if upper_cleaned.endswith(suffix):
@@ -1664,7 +1627,7 @@ def _parsedate_to_utc(value: str) -> Optional[datetime.datetime]:
1664
1627
  return _ensure_utc(parsed)
1665
1628
 
1666
1629
 
1667
- custom_tzinfos: dict[str, int] = {
1630
+ _custom_tzinfos: dict[str, int] = {
1668
1631
  "UTC": 0,
1669
1632
  "UT": 0,
1670
1633
  "GMT": 0,
@@ -1715,7 +1678,7 @@ _DATEPARSER_SETTINGS = {
1715
1678
  @lru_cache(maxsize=512)
1716
1679
  def _slow_dateutil_parse(value: str) -> Optional[datetime.datetime]:
1717
1680
  try:
1718
- return dateutil_parser.parse(value, tzinfos=custom_tzinfos, ignoretz=False)
1681
+ return dateutil_parser.parse(value, tzinfos=_custom_tzinfos, ignoretz=False)
1719
1682
  except (ValueError, TypeError, OverflowError):
1720
1683
  return None
1721
1684
 
@@ -1727,7 +1690,7 @@ def _slow_dateparser(value: str) -> Optional[datetime.datetime]:
1727
1690
  except ImportError:
1728
1691
  return None
1729
1692
  try:
1730
- return _dateparser.parse(value, languages=["en"], settings=_DATEPARSER_SETTINGS)
1693
+ return _dateparser.parse(value, languages=["en"], settings={**_DATEPARSER_SETTINGS})
1731
1694
  except (ValueError, TypeError):
1732
1695
  return None
1733
1696
 
@@ -1747,6 +1710,27 @@ def _parse_date(date_str: str) -> Optional[str]:
1747
1710
  candidate = date_str.strip()
1748
1711
  if not candidate:
1749
1712
  return None
1713
+
1714
+ # Fast path: clean ISO-8601 (covers >90% of Atom/modern RSS dates)
1715
+ clen = len(candidate)
1716
+ if clen >= 20 and candidate[4] == "-" and candidate[0:4].isdigit():
1717
+ last = candidate[-1]
1718
+ # Most common: ends with 'Z' (e.g., 2024-01-15T10:30:00Z)
1719
+ if last in ("Z", "z"):
1720
+ try:
1721
+ dt = datetime.datetime.fromisoformat(candidate[:-1] + "+00:00")
1722
+ return dt.isoformat()
1723
+ except ValueError:
1724
+ pass # Fall through to full parsing
1725
+ # Second most common: ends with +HH:MM (e.g., 2024-01-15T10:30:00+00:00)
1726
+ elif clen > 6 and candidate[-6] in ("+", "-") and candidate[-3] == ":":
1727
+ try:
1728
+ dt = datetime.datetime.fromisoformat(candidate)
1729
+ utc_dt = dt.replace(tzinfo=_UTC) if dt.tzinfo is None else dt.astimezone(_UTC)
1730
+ return utc_dt.isoformat()
1731
+ except (ValueError, OverflowError):
1732
+ pass # Fall through to full parsing
1733
+
1750
1734
  if "\n" in candidate or "\r" in candidate or "\t" in candidate or " " in candidate:
1751
1735
  candidate = _RE_WHITESPACE.sub(" ", candidate)
1752
1736
 
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.1
2
2
  Name: fastfeedparser
3
- Version: 0.5.0
3
+ Version: 0.5.2
4
4
  Summary: High performance RSS, Atom, JSON and RDF feed parser in Python
5
5
  Home-page: https://github.com/kagisearch/fastfeedparser
6
6
  Author: Vladimir Prelovac
File without changes
File without changes