fastfeedparser 0.4.9__tar.gz → 0.5.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.1
2
2
  Name: fastfeedparser
3
- Version: 0.4.9
3
+ Version: 0.5.1
4
4
  Summary: High performance RSS, Atom, JSON and RDF feed parser in Python
5
5
  Home-page: https://github.com/kagisearch/fastfeedparser
6
6
  Author: Vladimir Prelovac
@@ -2,3 +2,6 @@
2
2
  requires = ["setuptools~=67.0", "wheel"]
3
3
  build-backend = "setuptools.build_meta"
4
4
 
5
+ [tool.pytest.ini_options]
6
+ testpaths = ["tests"]
7
+
@@ -1,6 +1,6 @@
1
1
  [metadata]
2
2
  name = fastfeedparser
3
- version = 0.4.9
3
+ version = 0.5.1
4
4
  author = Vladimir Prelovac
5
5
  author_email = vlad@kagi.com
6
6
  description = High performance RSS, Atom, JSON and RDF feed parser in Python
@@ -1,4 +1,4 @@
1
1
  from .main import parse, FastFeedParserDict
2
2
 
3
- __version__ = "0.4.9"
3
+ __version__ = "0.5.1"
4
4
  __all__ = ["parse", "FastFeedParserDict"]
@@ -3,6 +3,7 @@ from __future__ import annotations
3
3
  import datetime
4
4
  from email.utils import parsedate_to_datetime
5
5
  import gzip
6
+ import html as _html_mod
6
7
  import json
7
8
  import re
8
9
  import zlib
@@ -40,27 +41,18 @@ _RE_XML_DECL_ENCODING = re.compile(
40
41
  _RE_XML_DECL_ENCODING_BYTES = re.compile(
41
42
  br'(<\?xml[^>]*encoding=["\'])([^"\']+)(["\'][^>]*\?>)', re.IGNORECASE
42
43
  )
43
- _RE_DOUBLE_XML_DECL = re.compile(r"<\?xml\?xml\s+", re.IGNORECASE)
44
44
  _RE_DOUBLE_XML_DECL_BYTES = re.compile(br"<\?xml\?xml\s+", re.IGNORECASE)
45
- _RE_DOUBLE_CLOSE = re.compile(r"\?\?>\s*")
46
45
  _RE_DOUBLE_CLOSE_BYTES = re.compile(br"\?\?>\s*")
47
- _RE_UNQUOTED_ATTR = re.compile(r'(\s+[\w:]+)=([^\s>"\']+)')
48
46
  _RE_UNQUOTED_ATTR_BYTES = re.compile(br'(\s+[\w:]+)=([^\s>"\']+)')
49
- _RE_UTF16_ENCODING = re.compile(
50
- r'(<\?xml[^>]*encoding=["\'])utf-16(-le|-be)?(["\'][^>]*\?>)', re.IGNORECASE
51
- )
52
47
  _RE_UTF16_ENCODING_BYTES = re.compile(
53
48
  br'(<\?xml[^>]*encoding=["\'])utf-16(-le|-be)?(["\'][^>]*\?>)', re.IGNORECASE
54
49
  )
55
- _RE_UNCLOSED_LINK = re.compile(
56
- r"<link([^>]*[^/])>\s*(?=\n\s*<(?!/link\s*>))", re.MULTILINE
57
- )
58
50
  _RE_UNCLOSED_LINK_BYTES = re.compile(
59
51
  br"<link([^>]*[^/])>\s*(?=\n\s*<(?!/link\s*>))", re.MULTILINE
60
52
  )
61
53
  _RE_FEB29 = re.compile(r"(\d{4})-02-29")
54
+ _RE_HTML_TAGS = re.compile(r"<[^>]+>")
62
55
  _RE_WHITESPACE = re.compile(r"\s+")
63
- _RE_ISO_LIKE = re.compile(r"^\d{4}-\d{2}-\d{2}")
64
56
  _RE_ISO_TZ_NO_COLON = re.compile(r"([+-]\d{2})(\d{2})$")
65
57
  _RE_ISO_TZ_HOUR_ONLY = re.compile(r"([+-]\d{2})$")
66
58
  _RE_ISO_FRACTION = re.compile(r"\.(\d{7,})(?=(?:[+-]\d{2}:?\d{2}|Z|$))", re.IGNORECASE)
@@ -111,36 +103,6 @@ def _ensure_utf8_xml_declaration(content: str) -> str:
111
103
  return _RE_XML_DECL_ENCODING.sub(r"\1utf-8\3", content, count=1)
112
104
 
113
105
 
114
- def _clean_feed_text(content: str) -> str:
115
- """Clean feed text by extracting the XML document (if it's embedded in junk)."""
116
- stripped_content = content.lstrip()
117
- stripped_lower = stripped_content[:2000].lower()
118
- if stripped_lower.startswith(("<?xml", "<rss", "<feed", "<rdf")):
119
- return stripped_content
120
-
121
- if stripped_lower.startswith("<!doctype html") or stripped_lower.startswith("<html"):
122
- raise ValueError("Content appears to be HTML, not a valid RSS/Atom feed")
123
-
124
- xml_start_patterns = (
125
- "<?xml",
126
- "<rss",
127
- "<feed",
128
- "<rdf:rdf",
129
- "<?xml-stylesheet",
130
- )
131
-
132
- content_lines = content.splitlines()
133
- for i, line in enumerate(content_lines):
134
- line_stripped = line.strip().lower()
135
- if any(line_stripped.startswith(pattern) for pattern in xml_start_patterns):
136
- return "\n".join(content_lines[i:])
137
-
138
- if "<script>" in stripped_lower or "<body>" in stripped_lower:
139
- raise ValueError("Content appears to be HTML, not a valid RSS/Atom feed")
140
-
141
- return content
142
-
143
-
144
106
  def _clean_feed_bytes(content: bytes) -> bytes:
145
107
  """Clean feed bytes by extracting the XML document (if it's embedded in junk)."""
146
108
  stripped_content = content.lstrip()
@@ -223,60 +185,9 @@ def _prepare_xml_bytes(xml_content: str | bytes) -> bytes:
223
185
  cleaned = _fix_malformed_xml_bytes(cleaned, actual_encoding=actual_encoding)
224
186
  return cleaned
225
187
 
226
- cleaned_text = _clean_feed_text(xml_content)
227
- if not cleaned_text.strip():
228
- raise ValueError("Empty content")
229
-
230
- needs_fixing = (
231
- "?xml?xml" in cleaned_text[:200]
232
- or "??>" in cleaned_text[:200]
233
- or (
234
- "rss:" in cleaned_text[:500] and "xmlns:rss" not in cleaned_text[:1000]
235
- )
236
- or ("utf-16" in cleaned_text[:200].lower())
237
- )
238
- if needs_fixing:
239
- cleaned_text = _fix_malformed_xml(cleaned_text, actual_encoding="utf-8")
240
-
241
- cleaned_text = _ensure_utf8_xml_declaration(cleaned_text)
242
- return cleaned_text.encode("utf-8", errors="replace")
243
-
244
-
245
- def _fix_malformed_xml(content: str, actual_encoding: str = "utf-8") -> str:
246
- """Fix common malformed XML issues in feeds.
247
-
248
- Some feeds have malformed XML like unclosed link tags or other issues
249
- that can be automatically corrected.
250
-
251
- Args:
252
- content: The XML content as a string
253
- actual_encoding: The actual encoding used (default: utf-8)
254
- """
255
- # Fix double XML declarations like "<?xml?xml version="1.0"?>"
256
- # This is found in dylanharris.org feed
257
- content = _RE_DOUBLE_XML_DECL.sub(r"<?xml ", content)
258
-
259
- # Fix double closing ?> in XML declaration like "??>>"
260
- content = _RE_DOUBLE_CLOSE.sub(r"?>", content)
261
-
262
- # Fix malformed attribute syntax like rss:version=2.0 (missing quotes)
263
- # This is found in dylanharris.org feed
264
- content = _RE_UNQUOTED_ATTR.sub(r'\1="\2"', content)
265
-
266
- # Update encoding in XML declaration to match actual encoding
267
- # This handles cases where content was transcoded from UTF-16 to UTF-8
268
- if actual_encoding.lower() != "utf-16":
269
- content = _RE_UTF16_ENCODING.sub(rf"\1{actual_encoding}\3", content)
270
-
271
- # Fix unclosed link tags - common in Atom feeds
272
- # Pattern: <link ...> followed by whitespace and another tag (not </link>)
273
- # should be <link .../>
274
- # Only fix link tags that are clearly malformed:
275
- # - End with > instead of />
276
- # - Are followed by whitespace and another tag (not a closing </link>)
277
- content = _RE_UNCLOSED_LINK.sub(r"<link\1/>", content)
278
-
279
- return content
188
+ # Str input: fix encoding declaration, encode to bytes, then use bytes path.
189
+ xml_content = _ensure_utf8_xml_declaration(xml_content)
190
+ return _prepare_xml_bytes(xml_content.encode("utf-8", errors="replace"))
280
191
 
281
192
 
282
193
  def _parse_json_feed(json_data: dict) -> FastFeedParserDict:
@@ -555,47 +466,31 @@ def _extract_error_message(root: _Element, raw_bytes: Optional[bytes] = None) ->
555
466
  return error_msg
556
467
 
557
468
 
469
+ _NON_FEED_MESSAGES: dict[str, str] = {
470
+ "html": "Received HTML page instead of feed",
471
+ "div": "Received HTML fragment instead of feed",
472
+ "body": "Received HTML fragment instead of feed",
473
+ "br": "Received HTML fragment instead of feed",
474
+ "status": "Feed server returned status message",
475
+ "error": "Feed server returned error",
476
+ "opml": "Received OPML document instead of feed (OPML is an outline format, not a feed)",
477
+ "urlset": "Received XML sitemap instead of feed (sitemap is for search engines, not a feed)",
478
+ "sitemapindex": "Received XML sitemap instead of feed (sitemap is for search engines, not a feed)",
479
+ }
480
+
481
+
558
482
  def _raise_for_non_feed_root(
559
483
  root: _Element, root_tag_local: str, raw_bytes: Optional[bytes] = None
560
484
  ) -> None:
561
- non_feed_tags = {
562
- "status", "error", "html", "opml", "br", "div", "body",
563
- "urlset", "sitemapindex",
564
- }
565
- if root_tag_local not in non_feed_tags:
485
+ base_msg = _NON_FEED_MESSAGES.get(root_tag_local)
486
+ if base_msg is None:
566
487
  return
567
488
 
568
489
  error_msg = _extract_error_message(root, raw_bytes).strip()[:300] or "No error message"
569
490
 
570
- if root_tag_local == "html":
571
- if error_msg != "No error message" and len(error_msg) > 10:
572
- raise ValueError(f"Received HTML page instead of feed: {error_msg[:150]}")
573
- raise ValueError(
574
- "Received HTML page instead of feed (possible redirect, 404, or server error)"
575
- )
576
- if root_tag_local in {"div", "body"}:
577
- if error_msg != "No error message" and len(error_msg) > 10:
578
- raise ValueError(f"Received HTML fragment instead of feed: {error_msg[:150]}")
579
- raise ValueError("Received HTML fragment instead of feed")
580
- if root_tag_local == "br":
581
- if error_msg != "No error message" and len(error_msg) > 10:
582
- raise ValueError(f"Received HTML error instead of feed: {error_msg[:150]}")
583
- raise ValueError("Received HTML fragment instead of feed")
584
- if root_tag_local == "status":
585
- raise ValueError(f"Feed server returned status message: {error_msg}")
586
- if root_tag_local == "error":
587
- if error_msg != "No error message":
588
- raise ValueError(f"Feed server returned error: {error_msg}")
589
- raise ValueError("Feed server returned error (no details provided)")
590
- if root_tag_local == "opml":
591
- raise ValueError(
592
- "Received OPML document instead of feed (OPML is an outline format, not a feed)"
593
- )
594
- if root_tag_local in {"urlset", "sitemapindex"}:
595
- raise ValueError(
596
- "Received XML sitemap instead of feed (sitemap is for search engines, not a feed)"
597
- )
598
- raise ValueError(f"Not a valid feed: {root_tag_local} element found - {error_msg[:100]}")
491
+ if error_msg != "No error message" and len(error_msg) > 10:
492
+ raise ValueError(f"{base_msg}: {error_msg[:150]}")
493
+ raise ValueError(base_msg)
599
494
 
600
495
 
601
496
  _RE_META_REFRESH_URL = re.compile(
@@ -730,24 +625,6 @@ def _detect_feed_structure(
730
625
  raise ValueError(f"Unknown feed type: {root.tag}")
731
626
 
732
627
 
733
- def _should_parse_media_content(root: _Element, xml_content: bytes) -> bool:
734
- """Check if feed likely contains Media RSS fields."""
735
- ns_values = root.nsmap.values() if root.nsmap else ()
736
- for ns_value in ns_values:
737
- if not ns_value:
738
- continue
739
- if "search.yahoo.com/mrss" in ns_value:
740
- return True
741
-
742
- # Fallback for feeds with undeclared/late namespace usage.
743
- return b"search.yahoo.com/mrss" in xml_content or b"<media:" in xml_content
744
-
745
-
746
- def _should_parse_enclosures(feed_type: _FeedType, xml_content: bytes) -> bool:
747
- """Check if feed likely contains RSS enclosure elements."""
748
- return feed_type == "rss" and b"<enclosure" in xml_content
749
-
750
-
751
628
  def _parse_content(xml_content: str | bytes) -> FastFeedParserDict:
752
629
  """Parse feed content (XML or JSON) that has already been fetched."""
753
630
  json_feed = _maybe_parse_json_feed(xml_content)
@@ -762,8 +639,6 @@ def _parse_content(xml_content: str | bytes) -> FastFeedParserDict:
762
639
  feed_type, channel, items, atom_namespace = _detect_feed_structure(
763
640
  root, xml_content, root_tag_local
764
641
  )
765
- parse_media_content = _should_parse_media_content(root, xml_content)
766
- parse_enclosures = _should_parse_enclosures(feed_type, xml_content)
767
642
 
768
643
  feed = _parse_feed_info(channel, feed_type, atom_namespace)
769
644
 
@@ -775,8 +650,6 @@ def _parse_content(xml_content: str | bytes) -> FastFeedParserDict:
775
650
  item,
776
651
  feed_type,
777
652
  atom_namespace,
778
- parse_media_content=parse_media_content,
779
- parse_enclosures=parse_enclosures,
780
653
  )
781
654
  # Ensure that titles and descriptions are always present
782
655
  entry["title"] = entry.get("title", "").strip()
@@ -1123,16 +996,9 @@ def _populate_entry_content(
1123
996
  content_value = entry["content"][0]["value"]
1124
997
  if content_value:
1125
998
  if "<" in content_value:
1126
- try:
1127
- html_content = etree.HTML(content_value)
1128
- if html_content is not None:
1129
- content_text = html_content.xpath("string()")
1130
- if isinstance(content_text, str):
1131
- content_value = _RE_WHITESPACE.sub(" ", content_text)
1132
- except etree.ParserError:
1133
- pass
1134
- else:
1135
- content_value = _RE_WHITESPACE.sub(" ", content_value)
999
+ content_value = _RE_HTML_TAGS.sub(" ", content_value[:2048])
1000
+ content_value = _html_mod.unescape(content_value)
1001
+ content_value = _RE_WHITESPACE.sub(" ", content_value).strip()
1136
1002
  entry["description"] = content_value[:512]
1137
1003
 
1138
1004
 
@@ -1223,13 +1089,6 @@ def _parse_enclosures(item: _Element) -> list[dict[str, Any]] | None:
1223
1089
  return enclosures or None
1224
1090
 
1225
1091
 
1226
- def _normalize_local_tag_name(tag: str) -> str:
1227
- local = tag.rsplit("}", 1)[-1].lower()
1228
- if ":" in local:
1229
- local = local.split(":", 1)[1]
1230
- return local
1231
-
1232
-
1233
1092
  def _build_rss_item_text_maps(item: _Element) -> tuple[dict[str, Optional[str]], dict[str, Optional[str]]]:
1234
1093
  by_local: dict[str, Optional[str]] = {}
1235
1094
  by_full: dict[str, Optional[str]] = {}
@@ -1240,7 +1099,9 @@ def _build_rss_item_text_maps(item: _Element) -> tuple[dict[str, Optional[str]],
1240
1099
  text_value = child.text.strip() if child.text else None
1241
1100
  if tag not in by_full:
1242
1101
  by_full[tag] = text_value
1243
- local = _normalize_local_tag_name(tag)
1102
+ local = tag.rsplit("}", 1)[-1].lower()
1103
+ if ":" in local:
1104
+ local = local.split(":", 1)[1]
1244
1105
  if local not in by_local:
1245
1106
  by_local[local] = text_value
1246
1107
  return by_local, by_full
@@ -1257,8 +1118,6 @@ def _first_non_empty(mapping: dict[str, Optional[str]], keys: tuple[str, ...]) -
1257
1118
  def _parse_rss_feed_entry_fast(
1258
1119
  item: _Element,
1259
1120
  atom_ns: str,
1260
- parse_media_content: bool = True,
1261
- parse_enclosures: bool = True,
1262
1121
  ) -> FastFeedParserDict:
1263
1122
  text_by_local, text_by_full = _build_rss_item_text_maps(item)
1264
1123
 
@@ -1294,7 +1153,7 @@ def _parse_rss_feed_entry_fast(
1294
1153
  if updated:
1295
1154
  entry["updated"] = updated
1296
1155
 
1297
- if "published" not in entry and rss_guid:
1156
+ if "published" not in entry and rss_guid and not rss_guid.startswith(("http://", "https://")):
1298
1157
  guid_date = _parse_date(rss_guid)
1299
1158
  if guid_date:
1300
1159
  entry["published"] = guid_date
@@ -1308,15 +1167,13 @@ def _parse_rss_feed_entry_fast(
1308
1167
 
1309
1168
  _populate_entry_content(entry, item, "rss", atom_ns)
1310
1169
 
1311
- if parse_media_content:
1312
- media_contents = _parse_media_content(item)
1313
- if media_contents:
1314
- entry["media_content"] = media_contents
1170
+ media_contents = _parse_media_content(item)
1171
+ if media_contents:
1172
+ entry["media_content"] = media_contents
1315
1173
 
1316
- if parse_enclosures:
1317
- enclosures = _parse_enclosures(item)
1318
- if enclosures:
1319
- entry["enclosures"] = enclosures
1174
+ enclosures = _parse_enclosures(item)
1175
+ if enclosures:
1176
+ entry["enclosures"] = enclosures
1320
1177
 
1321
1178
  author = _first_non_empty(text_by_local, ("author", "creator"))
1322
1179
  if not author:
@@ -1340,20 +1197,12 @@ def _parse_feed_entry(
1340
1197
  item: _Element,
1341
1198
  feed_type: _FeedType,
1342
1199
  atom_namespace: Optional[str] = None,
1343
- *,
1344
- parse_media_content: bool = True,
1345
- parse_enclosures: bool = True,
1346
1200
  ) -> FastFeedParserDict:
1347
1201
  # Use dynamic atom namespace or fallback to default
1348
1202
  atom_ns = atom_namespace or "http://www.w3.org/2005/Atom"
1349
1203
 
1350
1204
  if feed_type == "rss":
1351
- return _parse_rss_feed_entry_fast(
1352
- item,
1353
- atom_ns,
1354
- parse_media_content=parse_media_content,
1355
- parse_enclosures=parse_enclosures,
1356
- )
1205
+ return _parse_rss_feed_entry_fast(item, atom_ns)
1357
1206
 
1358
1207
  # Check if this is Atom 0.3 to use different date field names
1359
1208
  is_atom_03 = atom_ns == "http://purl.org/atom/ns#"
@@ -1446,8 +1295,7 @@ def _parse_feed_entry(
1446
1295
  entry["updated"] = _parse_date(fallback_updated)
1447
1296
 
1448
1297
  # Try to extract date from GUID as final fallback
1449
- if "published" not in entry and rss_guid:
1450
- # Check if GUID contains date information
1298
+ if "published" not in entry and rss_guid and not rss_guid.startswith(("http://", "https://")):
1451
1299
  guid_date = _parse_date(rss_guid)
1452
1300
  if guid_date:
1453
1301
  entry["published"] = guid_date
@@ -1467,32 +1315,22 @@ def _parse_feed_entry(
1467
1315
 
1468
1316
  _populate_entry_content(entry, item, feed_type, atom_ns)
1469
1317
 
1470
- if parse_media_content:
1471
- media_contents = _parse_media_content(item)
1472
- if media_contents:
1473
- entry["media_content"] = media_contents
1318
+ media_contents = _parse_media_content(item)
1319
+ if media_contents:
1320
+ entry["media_content"] = media_contents
1474
1321
 
1475
- if parse_enclosures:
1476
- enclosures = _parse_enclosures(item)
1477
- if enclosures:
1478
- entry["enclosures"] = enclosures
1322
+ enclosures = _parse_enclosures(item)
1323
+ if enclosures:
1324
+ entry["enclosures"] = enclosures
1479
1325
 
1480
- author = (
1481
- get_field_value(
1482
- "author",
1483
- f"{{{atom_ns}}}author/{{{atom_ns}}}name",
1484
- "{http://purl.org/dc/elements/1.1/}creator",
1485
- False,
1486
- )
1487
- or get_field_value(
1488
- "{http://purl.org/dc/elements/1.1/}creator",
1489
- "{http://purl.org/dc/elements/1.1/}creator",
1490
- "{http://purl.org/dc/elements/1.1/}creator",
1491
- False,
1492
- )
1493
- or element_get("{http://purl.org/dc/elements/1.1/}creator")
1494
- or element_get("author")
1326
+ author = get_field_value(
1327
+ "author",
1328
+ f"{{{atom_ns}}}author/{{{atom_ns}}}name",
1329
+ "{http://purl.org/dc/elements/1.1/}creator",
1330
+ False,
1495
1331
  )
1332
+ if not author:
1333
+ author = element_get("{http://purl.org/dc/elements/1.1/}creator") or element_get("author")
1496
1334
  if author:
1497
1335
  entry["author"] = author
1498
1336
 
@@ -1648,7 +1486,7 @@ def _normalize_iso_datetime_string(value: str) -> str:
1648
1486
  if cleaned.endswith(("Z", "z")):
1649
1487
  cleaned = cleaned[:-1] + "+00:00"
1650
1488
 
1651
- if " " in cleaned and "T" not in cleaned[:11] and _RE_ISO_LIKE.match(cleaned):
1489
+ if " " in cleaned and "T" not in cleaned[:11] and len(cleaned) >= 10 and cleaned[4] == "-" and cleaned[0:4].isdigit():
1652
1490
  date_part, rest = cleaned.split(" ", 1)
1653
1491
  if rest and rest[0].isdigit():
1654
1492
  cleaned = f"{date_part}T{rest}"
@@ -1684,7 +1522,7 @@ def _parsedate_to_utc(value: str) -> Optional[datetime.datetime]:
1684
1522
  return _ensure_utc(parsed)
1685
1523
 
1686
1524
 
1687
- custom_tzinfos: dict[str, int] = {
1525
+ _custom_tzinfos: dict[str, int] = {
1688
1526
  "UTC": 0,
1689
1527
  "UT": 0,
1690
1528
  "GMT": 0,
@@ -1735,7 +1573,7 @@ _DATEPARSER_SETTINGS = {
1735
1573
  @lru_cache(maxsize=512)
1736
1574
  def _slow_dateutil_parse(value: str) -> Optional[datetime.datetime]:
1737
1575
  try:
1738
- return dateutil_parser.parse(value, tzinfos=custom_tzinfos, ignoretz=False)
1576
+ return dateutil_parser.parse(value, tzinfos=_custom_tzinfos, ignoretz=False)
1739
1577
  except (ValueError, TypeError, OverflowError):
1740
1578
  return None
1741
1579
 
@@ -1772,12 +1610,12 @@ def _parse_date(date_str: str) -> Optional[str]:
1772
1610
 
1773
1611
  # Fix invalid leap year dates (Feb 29 in non-leap years)
1774
1612
  # This handles feeds with incorrect dates like "2023-02-29"
1775
- year_match = _RE_FEB29.match(candidate)
1776
- if year_match:
1777
- year = int(year_match.group(1))
1778
- if not ((year % 4 == 0 and year % 100 != 0) or (year % 400 == 0)):
1779
- # Not a leap year, change Feb 29 to Feb 28
1780
- candidate = candidate.replace(f"{year}-02-29", f"{year}-02-28")
1613
+ if "-02-29" in candidate:
1614
+ year_match = _RE_FEB29.match(candidate)
1615
+ if year_match:
1616
+ year = int(year_match.group(1))
1617
+ if not ((year % 4 == 0 and year % 100 != 0) or (year % 400 == 0)):
1618
+ candidate = candidate.replace(f"{year}-02-29", f"{year}-02-28")
1781
1619
 
1782
1620
  if "24:00" in candidate:
1783
1621
  candidate = candidate.replace("24:00:00", "00:00:00").replace(
@@ -1786,7 +1624,7 @@ def _parse_date(date_str: str) -> Optional[str]:
1786
1624
 
1787
1625
  dt: Optional[datetime.datetime] = None
1788
1626
 
1789
- is_iso_like = _RE_ISO_LIKE.match(candidate) is not None
1627
+ is_iso_like = len(candidate) >= 10 and candidate[4] == "-" and candidate[0:4].isdigit()
1790
1628
  if is_iso_like:
1791
1629
  iso_candidate = _normalize_iso_datetime_string(candidate)
1792
1630
  try:
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.1
2
2
  Name: fastfeedparser
3
- Version: 0.4.9
3
+ Version: 0.5.1
4
4
  Summary: High performance RSS, Atom, JSON and RDF feed parser in Python
5
5
  Home-page: https://github.com/kagisearch/fastfeedparser
6
6
  Author: Vladimir Prelovac
File without changes
File without changes