fastfeedparser 0.5.1__tar.gz → 0.5.3__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {fastfeedparser-0.5.1/src/fastfeedparser.egg-info → fastfeedparser-0.5.3}/PKG-INFO +1 -1
- {fastfeedparser-0.5.1 → fastfeedparser-0.5.3}/setup.cfg +1 -1
- {fastfeedparser-0.5.1 → fastfeedparser-0.5.3}/src/fastfeedparser/main.py +259 -36
- {fastfeedparser-0.5.1 → fastfeedparser-0.5.3/src/fastfeedparser.egg-info}/PKG-INFO +1 -1
- {fastfeedparser-0.5.1 → fastfeedparser-0.5.3}/LICENSE +0 -0
- {fastfeedparser-0.5.1 → fastfeedparser-0.5.3}/README.md +0 -0
- {fastfeedparser-0.5.1 → fastfeedparser-0.5.3}/pyproject.toml +0 -0
- {fastfeedparser-0.5.1 → fastfeedparser-0.5.3}/src/fastfeedparser/__init__.py +0 -0
- {fastfeedparser-0.5.1 → fastfeedparser-0.5.3}/src/fastfeedparser.egg-info/SOURCES.txt +0 -0
- {fastfeedparser-0.5.1 → fastfeedparser-0.5.3}/src/fastfeedparser.egg-info/dependency_links.txt +0 -0
- {fastfeedparser-0.5.1 → fastfeedparser-0.5.3}/src/fastfeedparser.egg-info/requires.txt +0 -0
- {fastfeedparser-0.5.1 → fastfeedparser-0.5.3}/src/fastfeedparser.egg-info/top_level.txt +0 -0
- {fastfeedparser-0.5.1 → fastfeedparser-0.5.3}/tests/test_encoding.py +0 -0
- {fastfeedparser-0.5.1 → fastfeedparser-0.5.3}/tests/test_integration.py +0 -0
|
@@ -56,6 +56,18 @@ _RE_WHITESPACE = re.compile(r"\s+")
|
|
|
56
56
|
_RE_ISO_TZ_NO_COLON = re.compile(r"([+-]\d{2})(\d{2})$")
|
|
57
57
|
_RE_ISO_TZ_HOUR_ONLY = re.compile(r"([+-]\d{2})$")
|
|
58
58
|
_RE_ISO_FRACTION = re.compile(r"\.(\d{7,})(?=(?:[+-]\d{2}:?\d{2}|Z|$))", re.IGNORECASE)
|
|
59
|
+
_RE_RFC822 = re.compile(
|
|
60
|
+
r"(?:\w{3},\s+)?(\d{1,2})\s+(\w{3})\s+(\d{4})\s+(\d{2}):(\d{2}):(\d{2})\s+([+-]\d{4}|[A-Z]{2,5})"
|
|
61
|
+
)
|
|
62
|
+
_MONTHS_RFC822: dict[str, int] = {
|
|
63
|
+
"jan": 1, "feb": 2, "mar": 3, "apr": 4, "may": 5, "jun": 6,
|
|
64
|
+
"jul": 7, "aug": 8, "sep": 9, "oct": 10, "nov": 11, "dec": 12,
|
|
65
|
+
}
|
|
66
|
+
_TZ_OFFSETS_RFC822: dict[str, int] = {
|
|
67
|
+
"GMT": 0, "UTC": 0, "UT": 0,
|
|
68
|
+
"EST": -18000, "EDT": -14400, "CST": -21600, "CDT": -18000,
|
|
69
|
+
"MST": -25200, "MDT": -21600, "PST": -28800, "PDT": -25200,
|
|
70
|
+
}
|
|
59
71
|
|
|
60
72
|
|
|
61
73
|
class FastFeedParserDict(dict):
|
|
@@ -382,24 +394,26 @@ def _maybe_parse_json_feed(content: str | bytes) -> FastFeedParserDict | None:
|
|
|
382
394
|
return None
|
|
383
395
|
|
|
384
396
|
|
|
397
|
+
_STRICT_XML_PARSER = etree.XMLParser(
|
|
398
|
+
ns_clean=True,
|
|
399
|
+
recover=False,
|
|
400
|
+
collect_ids=False,
|
|
401
|
+
resolve_entities=False,
|
|
402
|
+
)
|
|
403
|
+
_RECOVER_XML_PARSER = etree.XMLParser(
|
|
404
|
+
ns_clean=True,
|
|
405
|
+
recover=True,
|
|
406
|
+
collect_ids=False,
|
|
407
|
+
resolve_entities=False,
|
|
408
|
+
)
|
|
409
|
+
|
|
410
|
+
|
|
385
411
|
def _parse_xml_root(xml_content: bytes) -> _Element:
|
|
386
412
|
try:
|
|
387
|
-
|
|
388
|
-
ns_clean=True,
|
|
389
|
-
recover=False,
|
|
390
|
-
collect_ids=False,
|
|
391
|
-
resolve_entities=False,
|
|
392
|
-
)
|
|
393
|
-
root = etree.fromstring(xml_content, parser=strict_parser)
|
|
413
|
+
root = etree.fromstring(xml_content, parser=_STRICT_XML_PARSER)
|
|
394
414
|
except etree.XMLSyntaxError:
|
|
395
|
-
recover_parser = etree.XMLParser(
|
|
396
|
-
ns_clean=True,
|
|
397
|
-
recover=True,
|
|
398
|
-
collect_ids=False,
|
|
399
|
-
resolve_entities=False,
|
|
400
|
-
)
|
|
401
415
|
try:
|
|
402
|
-
root = etree.fromstring(xml_content, parser=
|
|
416
|
+
root = etree.fromstring(xml_content, parser=_RECOVER_XML_PARSER)
|
|
403
417
|
except etree.XMLSyntaxError as e:
|
|
404
418
|
raise ValueError(f"Failed to parse XML content: {str(e)}")
|
|
405
419
|
|
|
@@ -642,6 +656,9 @@ def _parse_content(xml_content: str | bytes) -> FastFeedParserDict:
|
|
|
642
656
|
|
|
643
657
|
feed = _parse_feed_info(channel, feed_type, atom_namespace)
|
|
644
658
|
|
|
659
|
+
# Detect once whether media namespace is used anywhere in the document
|
|
660
|
+
has_media_ns = b"search.yahoo.com/mrss" in xml_content if isinstance(xml_content, bytes) else "search.yahoo.com/mrss" in xml_content
|
|
661
|
+
|
|
645
662
|
# Parse entries
|
|
646
663
|
entries: list[FastFeedParserDict] = []
|
|
647
664
|
feed["entries"] = entries
|
|
@@ -650,6 +667,7 @@ def _parse_content(xml_content: str | bytes) -> FastFeedParserDict:
|
|
|
650
667
|
item,
|
|
651
668
|
feed_type,
|
|
652
669
|
atom_namespace,
|
|
670
|
+
has_media_ns,
|
|
653
671
|
)
|
|
654
672
|
# Ensure that titles and descriptions are always present
|
|
655
673
|
entry["title"] = entry.get("title", "").strip()
|
|
@@ -998,7 +1016,10 @@ def _populate_entry_content(
|
|
|
998
1016
|
if "<" in content_value:
|
|
999
1017
|
content_value = _RE_HTML_TAGS.sub(" ", content_value[:2048])
|
|
1000
1018
|
content_value = _html_mod.unescape(content_value)
|
|
1001
|
-
content_value
|
|
1019
|
+
if " " in content_value or "\n" in content_value or "\t" in content_value or "\r" in content_value:
|
|
1020
|
+
content_value = _RE_WHITESPACE.sub(" ", content_value).strip()
|
|
1021
|
+
else:
|
|
1022
|
+
content_value = content_value.strip()
|
|
1002
1023
|
entry["description"] = content_value[:512]
|
|
1003
1024
|
|
|
1004
1025
|
|
|
@@ -1096,12 +1117,16 @@ def _build_rss_item_text_maps(item: _Element) -> tuple[dict[str, Optional[str]],
|
|
|
1096
1117
|
tag = child.tag
|
|
1097
1118
|
if not isinstance(tag, str):
|
|
1098
1119
|
continue
|
|
1099
|
-
text_value = child.text
|
|
1120
|
+
text_value = child.text or None
|
|
1100
1121
|
if tag not in by_full:
|
|
1101
1122
|
by_full[tag] = text_value
|
|
1102
|
-
|
|
1103
|
-
if "
|
|
1104
|
-
local =
|
|
1123
|
+
# Fast path: ~80% of RSS tags have no namespace or colon prefix
|
|
1124
|
+
if "{" in tag:
|
|
1125
|
+
local = tag.rsplit("}", 1)[1].lower()
|
|
1126
|
+
elif ":" in tag:
|
|
1127
|
+
local = tag.split(":", 1)[1].lower()
|
|
1128
|
+
else:
|
|
1129
|
+
local = tag.lower()
|
|
1105
1130
|
if local not in by_local:
|
|
1106
1131
|
by_local[local] = text_value
|
|
1107
1132
|
return by_local, by_full
|
|
@@ -1118,6 +1143,7 @@ def _first_non_empty(mapping: dict[str, Optional[str]], keys: tuple[str, ...]) -
|
|
|
1118
1143
|
def _parse_rss_feed_entry_fast(
|
|
1119
1144
|
item: _Element,
|
|
1120
1145
|
atom_ns: str,
|
|
1146
|
+
has_media_ns: bool = True,
|
|
1121
1147
|
) -> FastFeedParserDict:
|
|
1122
1148
|
text_by_local, text_by_full = _build_rss_item_text_maps(item)
|
|
1123
1149
|
|
|
@@ -1139,7 +1165,7 @@ def _parse_rss_feed_entry_fast(
|
|
|
1139
1165
|
|
|
1140
1166
|
link = text_by_local.get("link")
|
|
1141
1167
|
if link:
|
|
1142
|
-
entry["link"] = link
|
|
1168
|
+
entry["link"] = link.strip()
|
|
1143
1169
|
|
|
1144
1170
|
published_source = _first_non_empty(text_by_local, ("pubdate", "published", "issued", "date"))
|
|
1145
1171
|
if published_source:
|
|
@@ -1161,15 +1187,26 @@ def _parse_rss_feed_entry_fast(
|
|
|
1161
1187
|
if "updated" in entry and "published" not in entry:
|
|
1162
1188
|
entry["published"] = entry["updated"]
|
|
1163
1189
|
|
|
1164
|
-
|
|
1190
|
+
# Inline link population for RSS (avoids redundant findall/find for 98.8% of entries)
|
|
1191
|
+
atom_links = item.findall(f"{{{atom_ns}}}link")
|
|
1192
|
+
if atom_links:
|
|
1193
|
+
# Has atom:link elements - use full logic
|
|
1194
|
+
_populate_entry_links(entry, item, atom_ns)
|
|
1195
|
+
else:
|
|
1196
|
+
# Common RSS case: no atom:link elements
|
|
1197
|
+
entry["links"] = []
|
|
1198
|
+
if "link" not in entry and rss_guid and rss_guid.startswith(("http://", "https://")):
|
|
1199
|
+
entry["link"] = rss_guid
|
|
1200
|
+
|
|
1165
1201
|
if "id" not in entry and "link" in entry:
|
|
1166
1202
|
entry["id"] = entry["link"]
|
|
1167
1203
|
|
|
1168
1204
|
_populate_entry_content(entry, item, "rss", atom_ns)
|
|
1169
1205
|
|
|
1170
|
-
|
|
1171
|
-
|
|
1172
|
-
|
|
1206
|
+
if has_media_ns:
|
|
1207
|
+
media_contents = _parse_media_content(item)
|
|
1208
|
+
if media_contents:
|
|
1209
|
+
entry["media_content"] = media_contents
|
|
1173
1210
|
|
|
1174
1211
|
enclosures = _parse_enclosures(item)
|
|
1175
1212
|
if enclosures:
|
|
@@ -1180,11 +1217,11 @@ def _parse_rss_feed_entry_fast(
|
|
|
1180
1217
|
atom_author = item.find(f"{{{atom_ns}}}author/{{{atom_ns}}}name")
|
|
1181
1218
|
author = atom_author.text.strip() if atom_author is not None and atom_author.text else None
|
|
1182
1219
|
if author:
|
|
1183
|
-
entry["author"] = author
|
|
1220
|
+
entry["author"] = author.strip()
|
|
1184
1221
|
|
|
1185
1222
|
comments = text_by_local.get("comments")
|
|
1186
1223
|
if comments:
|
|
1187
|
-
entry["comments"] = comments
|
|
1224
|
+
entry["comments"] = comments.strip()
|
|
1188
1225
|
|
|
1189
1226
|
tags = _parse_tags(item, "rss", atom_ns)
|
|
1190
1227
|
if tags:
|
|
@@ -1193,17 +1230,117 @@ def _parse_rss_feed_entry_fast(
|
|
|
1193
1230
|
return entry
|
|
1194
1231
|
|
|
1195
1232
|
|
|
1233
|
+
def _parse_atom_feed_entry_fast(
|
|
1234
|
+
item: _Element,
|
|
1235
|
+
atom_ns: str,
|
|
1236
|
+
has_media_ns: bool = True,
|
|
1237
|
+
) -> FastFeedParserDict:
|
|
1238
|
+
ns = f"{{{atom_ns}}}"
|
|
1239
|
+
entry = FastFeedParserDict()
|
|
1240
|
+
|
|
1241
|
+
# ID
|
|
1242
|
+
el = item.find(ns + "id")
|
|
1243
|
+
if el is not None and el.text:
|
|
1244
|
+
entry["id"] = el.text.strip()
|
|
1245
|
+
|
|
1246
|
+
# Title
|
|
1247
|
+
el = item.find(ns + "title")
|
|
1248
|
+
if el is not None and el.text:
|
|
1249
|
+
entry["title"] = el.text.strip()
|
|
1250
|
+
|
|
1251
|
+
# Description (summary)
|
|
1252
|
+
el = item.find(ns + "summary")
|
|
1253
|
+
if el is not None and el.text:
|
|
1254
|
+
entry["description"] = el.text.strip()
|
|
1255
|
+
|
|
1256
|
+
# Link (href attribute)
|
|
1257
|
+
el = item.find(ns + "link")
|
|
1258
|
+
if el is not None:
|
|
1259
|
+
href = el.get("href")
|
|
1260
|
+
if href:
|
|
1261
|
+
entry["link"] = href.strip()
|
|
1262
|
+
|
|
1263
|
+
# Dates: Atom 1.0 uses published/updated, Atom 0.3 uses issued/modified
|
|
1264
|
+
is_atom_03 = atom_ns == "http://purl.org/atom/ns#"
|
|
1265
|
+
pub_tag = "issued" if is_atom_03 else "published"
|
|
1266
|
+
upd_tag = "modified" if is_atom_03 else "updated"
|
|
1267
|
+
pub_fallback_tag = "published" if is_atom_03 else "issued"
|
|
1268
|
+
upd_fallback_tag = "updated" if is_atom_03 else "modified"
|
|
1269
|
+
|
|
1270
|
+
el = item.find(ns + pub_tag)
|
|
1271
|
+
if el is not None and el.text:
|
|
1272
|
+
published = _parse_date(el.text)
|
|
1273
|
+
if published:
|
|
1274
|
+
entry["published"] = published
|
|
1275
|
+
|
|
1276
|
+
el = item.find(ns + upd_tag)
|
|
1277
|
+
if el is not None and el.text:
|
|
1278
|
+
updated = _parse_date(el.text)
|
|
1279
|
+
if updated:
|
|
1280
|
+
entry["updated"] = updated
|
|
1281
|
+
|
|
1282
|
+
# Fallback date fields for mixed namespace scenarios
|
|
1283
|
+
if "published" not in entry:
|
|
1284
|
+
el = item.find(ns + pub_fallback_tag)
|
|
1285
|
+
if el is not None and el.text:
|
|
1286
|
+
published = _parse_date(el.text)
|
|
1287
|
+
if published:
|
|
1288
|
+
entry["published"] = published
|
|
1289
|
+
|
|
1290
|
+
if "updated" not in entry:
|
|
1291
|
+
el = item.find(ns + upd_fallback_tag)
|
|
1292
|
+
if el is not None and el.text:
|
|
1293
|
+
updated = _parse_date(el.text)
|
|
1294
|
+
if updated:
|
|
1295
|
+
entry["updated"] = updated
|
|
1296
|
+
|
|
1297
|
+
if "updated" in entry and "published" not in entry:
|
|
1298
|
+
entry["published"] = entry["updated"]
|
|
1299
|
+
|
|
1300
|
+
_populate_entry_links(entry, item, atom_ns)
|
|
1301
|
+
|
|
1302
|
+
if "id" not in entry and "link" in entry:
|
|
1303
|
+
entry["id"] = entry["link"]
|
|
1304
|
+
|
|
1305
|
+
_populate_entry_content(entry, item, "atom", atom_ns)
|
|
1306
|
+
|
|
1307
|
+
if has_media_ns:
|
|
1308
|
+
media_contents = _parse_media_content(item)
|
|
1309
|
+
if media_contents:
|
|
1310
|
+
entry["media_content"] = media_contents
|
|
1311
|
+
|
|
1312
|
+
enclosures = _parse_enclosures(item)
|
|
1313
|
+
if enclosures:
|
|
1314
|
+
entry["enclosures"] = enclosures
|
|
1315
|
+
|
|
1316
|
+
# Author
|
|
1317
|
+
el = item.find(ns + "author/" + ns + "name")
|
|
1318
|
+
if el is not None and el.text:
|
|
1319
|
+
entry["author"] = el.text.strip()
|
|
1320
|
+
|
|
1321
|
+
tags = _parse_tags(item, "atom", atom_ns)
|
|
1322
|
+
if tags:
|
|
1323
|
+
entry["tags"] = tags
|
|
1324
|
+
|
|
1325
|
+
return entry
|
|
1326
|
+
|
|
1327
|
+
|
|
1196
1328
|
def _parse_feed_entry(
|
|
1197
1329
|
item: _Element,
|
|
1198
1330
|
feed_type: _FeedType,
|
|
1199
1331
|
atom_namespace: Optional[str] = None,
|
|
1332
|
+
has_media_ns: bool = True,
|
|
1200
1333
|
) -> FastFeedParserDict:
|
|
1201
1334
|
# Use dynamic atom namespace or fallback to default
|
|
1202
1335
|
atom_ns = atom_namespace or "http://www.w3.org/2005/Atom"
|
|
1203
1336
|
|
|
1204
1337
|
if feed_type == "rss":
|
|
1205
|
-
return _parse_rss_feed_entry_fast(item, atom_ns)
|
|
1338
|
+
return _parse_rss_feed_entry_fast(item, atom_ns, has_media_ns)
|
|
1206
1339
|
|
|
1340
|
+
if feed_type == "atom":
|
|
1341
|
+
return _parse_atom_feed_entry_fast(item, atom_ns, has_media_ns)
|
|
1342
|
+
|
|
1343
|
+
# RDF path uses the generic field machinery
|
|
1207
1344
|
# Check if this is Atom 0.3 to use different date field names
|
|
1208
1345
|
is_atom_03 = atom_ns == "http://purl.org/atom/ns#"
|
|
1209
1346
|
|
|
@@ -1315,9 +1452,10 @@ def _parse_feed_entry(
|
|
|
1315
1452
|
|
|
1316
1453
|
_populate_entry_content(entry, item, feed_type, atom_ns)
|
|
1317
1454
|
|
|
1318
|
-
|
|
1319
|
-
|
|
1320
|
-
|
|
1455
|
+
if has_media_ns:
|
|
1456
|
+
media_contents = _parse_media_content(item)
|
|
1457
|
+
if media_contents:
|
|
1458
|
+
entry["media_content"] = media_contents
|
|
1321
1459
|
|
|
1322
1460
|
enclosures = _parse_enclosures(item)
|
|
1323
1461
|
if enclosures:
|
|
@@ -1476,6 +1614,14 @@ def _normalize_iso_datetime_string(value: str) -> str:
|
|
|
1476
1614
|
if not cleaned:
|
|
1477
1615
|
return cleaned
|
|
1478
1616
|
|
|
1617
|
+
# Fast path: 'Z' suffix (most common in Atom feeds)
|
|
1618
|
+
if cleaned[-1] in ("Z", "z"):
|
|
1619
|
+
return cleaned[:-1] + "+00:00"
|
|
1620
|
+
|
|
1621
|
+
# Fast path: already has proper +HH:MM or -HH:MM timezone
|
|
1622
|
+
if len(cleaned) > 6 and cleaned[-6] in ("+", "-") and cleaned[-3] == ":":
|
|
1623
|
+
return cleaned
|
|
1624
|
+
|
|
1479
1625
|
upper_cleaned = cleaned.upper()
|
|
1480
1626
|
for suffix in (" UTC", " GMT", " Z"):
|
|
1481
1627
|
if upper_cleaned.endswith(suffix):
|
|
@@ -1511,8 +1657,54 @@ def _ensure_utc(dt: datetime.datetime) -> Optional[datetime.datetime]:
|
|
|
1511
1657
|
return None
|
|
1512
1658
|
|
|
1513
1659
|
|
|
1660
|
+
def _fast_rfc822_to_iso(value: str) -> Optional[str]:
|
|
1661
|
+
"""Fast RFC-822 date to ISO string, bypassing datetime objects for UTC dates."""
|
|
1662
|
+
m = _RE_RFC822.match(value)
|
|
1663
|
+
if not m:
|
|
1664
|
+
return None
|
|
1665
|
+
day, mon_str, year, hour, minute, second, tz = m.groups()
|
|
1666
|
+
month = _MONTHS_RFC822.get(mon_str.lower())
|
|
1667
|
+
if month is None:
|
|
1668
|
+
return None
|
|
1669
|
+
if tz[0] in "+-":
|
|
1670
|
+
tz_offset_seconds = (int(tz[1:3]) * 3600 + int(tz[3:5]) * 60) * (
|
|
1671
|
+
1 if tz[0] == "+" else -1
|
|
1672
|
+
)
|
|
1673
|
+
else:
|
|
1674
|
+
tz_offset_seconds = _TZ_OFFSETS_RFC822.get(tz)
|
|
1675
|
+
if tz_offset_seconds is None:
|
|
1676
|
+
return None # Unknown tz name, fall through to full parser
|
|
1677
|
+
# Python requires offset strictly between -24h and +24h
|
|
1678
|
+
if not (-86400 < tz_offset_seconds < 86400):
|
|
1679
|
+
return None
|
|
1680
|
+
d = int(day)
|
|
1681
|
+
h = int(hour)
|
|
1682
|
+
mi = int(minute)
|
|
1683
|
+
s = int(second)
|
|
1684
|
+
# Hour 24 is invalid (even ISO only allows 24:00:00); roll to next day at 00:mm:ss
|
|
1685
|
+
if h == 24:
|
|
1686
|
+
base = datetime.date(int(year), month, d) + datetime.timedelta(days=1)
|
|
1687
|
+
h = 0
|
|
1688
|
+
if tz_offset_seconds == 0:
|
|
1689
|
+
return f"{base.year:04d}-{base.month:02d}-{base.day:02d}T{h:02d}:{mi:02d}:{s:02d}+00:00"
|
|
1690
|
+
dt = datetime.datetime(
|
|
1691
|
+
base.year, base.month, base.day, h, mi, s,
|
|
1692
|
+
tzinfo=datetime.timezone(datetime.timedelta(seconds=tz_offset_seconds)),
|
|
1693
|
+
)
|
|
1694
|
+
utc = dt.astimezone(_UTC)
|
|
1695
|
+
return f"{utc.year:04d}-{utc.month:02d}-{utc.day:02d}T{utc.hour:02d}:{utc.minute:02d}:{utc.second:02d}+00:00"
|
|
1696
|
+
if tz_offset_seconds == 0:
|
|
1697
|
+
return f"{year}-{month:02d}-{d:02d}T{hour}:{minute}:{second}+00:00"
|
|
1698
|
+
dt = datetime.datetime(
|
|
1699
|
+
int(year), month, d, h, mi, s,
|
|
1700
|
+
tzinfo=datetime.timezone(datetime.timedelta(seconds=tz_offset_seconds)),
|
|
1701
|
+
)
|
|
1702
|
+
utc = dt.astimezone(_UTC)
|
|
1703
|
+
return f"{utc.year:04d}-{utc.month:02d}-{utc.day:02d}T{utc.hour:02d}:{utc.minute:02d}:{utc.second:02d}+00:00"
|
|
1704
|
+
|
|
1705
|
+
|
|
1514
1706
|
def _parsedate_to_utc(value: str) -> Optional[datetime.datetime]:
|
|
1515
|
-
"""
|
|
1707
|
+
"""RFC-822 / RFC-2822 parsing via email.utils (fallback)."""
|
|
1516
1708
|
try:
|
|
1517
1709
|
parsed = parsedate_to_datetime(value)
|
|
1518
1710
|
except (TypeError, ValueError, IndexError):
|
|
@@ -1585,7 +1777,7 @@ def _slow_dateparser(value: str) -> Optional[datetime.datetime]:
|
|
|
1585
1777
|
except ImportError:
|
|
1586
1778
|
return None
|
|
1587
1779
|
try:
|
|
1588
|
-
return _dateparser.parse(value, languages=["en"], settings=_DATEPARSER_SETTINGS)
|
|
1780
|
+
return _dateparser.parse(value, languages=["en"], settings={**_DATEPARSER_SETTINGS})
|
|
1589
1781
|
except (ValueError, TypeError):
|
|
1590
1782
|
return None
|
|
1591
1783
|
|
|
@@ -1605,6 +1797,30 @@ def _parse_date(date_str: str) -> Optional[str]:
|
|
|
1605
1797
|
candidate = date_str.strip()
|
|
1606
1798
|
if not candidate:
|
|
1607
1799
|
return None
|
|
1800
|
+
|
|
1801
|
+
# Fast path: clean ISO-8601 (covers >90% of Atom/modern RSS dates)
|
|
1802
|
+
clen = len(candidate)
|
|
1803
|
+
if clen >= 20 and candidate[4] == "-" and candidate[0:4].isdigit():
|
|
1804
|
+
last = candidate[-1]
|
|
1805
|
+
# Most common: ends with 'Z' (e.g., 2024-01-15T10:30:00Z)
|
|
1806
|
+
if last in ("Z", "z"):
|
|
1807
|
+
iso = candidate[:-1] + "+00:00"
|
|
1808
|
+
try:
|
|
1809
|
+
dt = datetime.datetime.fromisoformat(iso)
|
|
1810
|
+
return dt.isoformat()
|
|
1811
|
+
except ValueError:
|
|
1812
|
+
pass # Fall through to full parsing
|
|
1813
|
+
# Second most common: ends with +HH:MM (e.g., 2024-01-15T10:30:00+00:00)
|
|
1814
|
+
elif clen > 6 and candidate[-6] in ("+", "-") and candidate[-3] == ":":
|
|
1815
|
+
try:
|
|
1816
|
+
dt = datetime.datetime.fromisoformat(candidate)
|
|
1817
|
+
if dt.tzinfo is _UTC:
|
|
1818
|
+
return dt.isoformat()
|
|
1819
|
+
utc_dt = dt.astimezone(_UTC)
|
|
1820
|
+
return utc_dt.isoformat()
|
|
1821
|
+
except (ValueError, OverflowError):
|
|
1822
|
+
pass # Fall through to full parsing
|
|
1823
|
+
|
|
1608
1824
|
if "\n" in candidate or "\r" in candidate or "\t" in candidate or " " in candidate:
|
|
1609
1825
|
candidate = _RE_WHITESPACE.sub(" ", candidate)
|
|
1610
1826
|
|
|
@@ -1617,10 +1833,13 @@ def _parse_date(date_str: str) -> Optional[str]:
|
|
|
1617
1833
|
if not ((year % 4 == 0 and year % 100 != 0) or (year % 400 == 0)):
|
|
1618
1834
|
candidate = candidate.replace(f"{year}-02-29", f"{year}-02-28")
|
|
1619
1835
|
|
|
1620
|
-
if "24:
|
|
1621
|
-
|
|
1622
|
-
|
|
1623
|
-
|
|
1836
|
+
if "T24:" in candidate or " 24:" in candidate:
|
|
1837
|
+
m24 = re.search(r"(\d{4}-\d{2}-\d{2})[T ]24:(\d{2}):(\d{2})", candidate)
|
|
1838
|
+
if m24:
|
|
1839
|
+
base = datetime.date.fromisoformat(m24.group(1))
|
|
1840
|
+
mins, secs = int(m24.group(2)), int(m24.group(3))
|
|
1841
|
+
next_day = base + datetime.timedelta(days=1)
|
|
1842
|
+
candidate = candidate[:m24.start()] + f"{next_day}T00:{mins:02d}:{secs:02d}" + candidate[m24.end():]
|
|
1624
1843
|
|
|
1625
1844
|
dt: Optional[datetime.datetime] = None
|
|
1626
1845
|
|
|
@@ -1636,6 +1855,10 @@ def _parse_date(date_str: str) -> Optional[str]:
|
|
|
1636
1855
|
if utc_dt is not None:
|
|
1637
1856
|
return utc_dt.isoformat()
|
|
1638
1857
|
|
|
1858
|
+
rfc822_result = _fast_rfc822_to_iso(candidate)
|
|
1859
|
+
if rfc822_result is not None:
|
|
1860
|
+
return rfc822_result
|
|
1861
|
+
|
|
1639
1862
|
dt = _parsedate_to_utc(candidate)
|
|
1640
1863
|
if dt is not None:
|
|
1641
1864
|
return dt.isoformat()
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{fastfeedparser-0.5.1 → fastfeedparser-0.5.3}/src/fastfeedparser.egg-info/dependency_links.txt
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|