fastfeedparser 0.5.1__tar.gz → 0.5.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {fastfeedparser-0.5.1/src/fastfeedparser.egg-info → fastfeedparser-0.5.2}/PKG-INFO +1 -1
- {fastfeedparser-0.5.1 → fastfeedparser-0.5.2}/setup.cfg +1 -1
- {fastfeedparser-0.5.1 → fastfeedparser-0.5.2}/src/fastfeedparser/main.py +131 -5
- {fastfeedparser-0.5.1 → fastfeedparser-0.5.2/src/fastfeedparser.egg-info}/PKG-INFO +1 -1
- {fastfeedparser-0.5.1 → fastfeedparser-0.5.2}/LICENSE +0 -0
- {fastfeedparser-0.5.1 → fastfeedparser-0.5.2}/README.md +0 -0
- {fastfeedparser-0.5.1 → fastfeedparser-0.5.2}/pyproject.toml +0 -0
- {fastfeedparser-0.5.1 → fastfeedparser-0.5.2}/src/fastfeedparser/__init__.py +0 -0
- {fastfeedparser-0.5.1 → fastfeedparser-0.5.2}/src/fastfeedparser.egg-info/SOURCES.txt +0 -0
- {fastfeedparser-0.5.1 → fastfeedparser-0.5.2}/src/fastfeedparser.egg-info/dependency_links.txt +0 -0
- {fastfeedparser-0.5.1 → fastfeedparser-0.5.2}/src/fastfeedparser.egg-info/requires.txt +0 -0
- {fastfeedparser-0.5.1 → fastfeedparser-0.5.2}/src/fastfeedparser.egg-info/top_level.txt +0 -0
- {fastfeedparser-0.5.1 → fastfeedparser-0.5.2}/tests/test_encoding.py +0 -0
- {fastfeedparser-0.5.1 → fastfeedparser-0.5.2}/tests/test_integration.py +0 -0
|
@@ -1096,7 +1096,7 @@ def _build_rss_item_text_maps(item: _Element) -> tuple[dict[str, Optional[str]],
|
|
|
1096
1096
|
tag = child.tag
|
|
1097
1097
|
if not isinstance(tag, str):
|
|
1098
1098
|
continue
|
|
1099
|
-
text_value = child.text
|
|
1099
|
+
text_value = child.text or None
|
|
1100
1100
|
if tag not in by_full:
|
|
1101
1101
|
by_full[tag] = text_value
|
|
1102
1102
|
local = tag.rsplit("}", 1)[-1].lower()
|
|
@@ -1139,7 +1139,7 @@ def _parse_rss_feed_entry_fast(
|
|
|
1139
1139
|
|
|
1140
1140
|
link = text_by_local.get("link")
|
|
1141
1141
|
if link:
|
|
1142
|
-
entry["link"] = link
|
|
1142
|
+
entry["link"] = link.strip()
|
|
1143
1143
|
|
|
1144
1144
|
published_source = _first_non_empty(text_by_local, ("pubdate", "published", "issued", "date"))
|
|
1145
1145
|
if published_source:
|
|
@@ -1180,11 +1180,11 @@ def _parse_rss_feed_entry_fast(
|
|
|
1180
1180
|
atom_author = item.find(f"{{{atom_ns}}}author/{{{atom_ns}}}name")
|
|
1181
1181
|
author = atom_author.text.strip() if atom_author is not None and atom_author.text else None
|
|
1182
1182
|
if author:
|
|
1183
|
-
entry["author"] = author
|
|
1183
|
+
entry["author"] = author.strip()
|
|
1184
1184
|
|
|
1185
1185
|
comments = text_by_local.get("comments")
|
|
1186
1186
|
if comments:
|
|
1187
|
-
entry["comments"] = comments
|
|
1187
|
+
entry["comments"] = comments.strip()
|
|
1188
1188
|
|
|
1189
1189
|
tags = _parse_tags(item, "rss", atom_ns)
|
|
1190
1190
|
if tags:
|
|
@@ -1193,6 +1193,99 @@ def _parse_rss_feed_entry_fast(
|
|
|
1193
1193
|
return entry
|
|
1194
1194
|
|
|
1195
1195
|
|
|
1196
|
+
def _parse_atom_feed_entry_fast(
|
|
1197
|
+
item: _Element,
|
|
1198
|
+
atom_ns: str,
|
|
1199
|
+
) -> FastFeedParserDict:
|
|
1200
|
+
ns = f"{{{atom_ns}}}"
|
|
1201
|
+
entry = FastFeedParserDict()
|
|
1202
|
+
|
|
1203
|
+
# ID
|
|
1204
|
+
el = item.find(ns + "id")
|
|
1205
|
+
if el is not None and el.text:
|
|
1206
|
+
entry["id"] = el.text.strip()
|
|
1207
|
+
|
|
1208
|
+
# Title
|
|
1209
|
+
el = item.find(ns + "title")
|
|
1210
|
+
if el is not None and el.text:
|
|
1211
|
+
entry["title"] = el.text.strip()
|
|
1212
|
+
|
|
1213
|
+
# Description (summary)
|
|
1214
|
+
el = item.find(ns + "summary")
|
|
1215
|
+
if el is not None and el.text:
|
|
1216
|
+
entry["description"] = el.text.strip()
|
|
1217
|
+
|
|
1218
|
+
# Link (href attribute)
|
|
1219
|
+
el = item.find(ns + "link")
|
|
1220
|
+
if el is not None:
|
|
1221
|
+
href = el.get("href")
|
|
1222
|
+
if href:
|
|
1223
|
+
entry["link"] = href.strip()
|
|
1224
|
+
|
|
1225
|
+
# Dates: Atom 1.0 uses published/updated, Atom 0.3 uses issued/modified
|
|
1226
|
+
is_atom_03 = atom_ns == "http://purl.org/atom/ns#"
|
|
1227
|
+
pub_tag = "issued" if is_atom_03 else "published"
|
|
1228
|
+
upd_tag = "modified" if is_atom_03 else "updated"
|
|
1229
|
+
pub_fallback_tag = "published" if is_atom_03 else "issued"
|
|
1230
|
+
upd_fallback_tag = "updated" if is_atom_03 else "modified"
|
|
1231
|
+
|
|
1232
|
+
el = item.find(ns + pub_tag)
|
|
1233
|
+
if el is not None and el.text:
|
|
1234
|
+
published = _parse_date(el.text)
|
|
1235
|
+
if published:
|
|
1236
|
+
entry["published"] = published
|
|
1237
|
+
|
|
1238
|
+
el = item.find(ns + upd_tag)
|
|
1239
|
+
if el is not None and el.text:
|
|
1240
|
+
updated = _parse_date(el.text)
|
|
1241
|
+
if updated:
|
|
1242
|
+
entry["updated"] = updated
|
|
1243
|
+
|
|
1244
|
+
# Fallback date fields for mixed namespace scenarios
|
|
1245
|
+
if "published" not in entry:
|
|
1246
|
+
el = item.find(ns + pub_fallback_tag)
|
|
1247
|
+
if el is not None and el.text:
|
|
1248
|
+
published = _parse_date(el.text)
|
|
1249
|
+
if published:
|
|
1250
|
+
entry["published"] = published
|
|
1251
|
+
|
|
1252
|
+
if "updated" not in entry:
|
|
1253
|
+
el = item.find(ns + upd_fallback_tag)
|
|
1254
|
+
if el is not None and el.text:
|
|
1255
|
+
updated = _parse_date(el.text)
|
|
1256
|
+
if updated:
|
|
1257
|
+
entry["updated"] = updated
|
|
1258
|
+
|
|
1259
|
+
if "updated" in entry and "published" not in entry:
|
|
1260
|
+
entry["published"] = entry["updated"]
|
|
1261
|
+
|
|
1262
|
+
_populate_entry_links(entry, item, atom_ns)
|
|
1263
|
+
|
|
1264
|
+
if "id" not in entry and "link" in entry:
|
|
1265
|
+
entry["id"] = entry["link"]
|
|
1266
|
+
|
|
1267
|
+
_populate_entry_content(entry, item, "atom", atom_ns)
|
|
1268
|
+
|
|
1269
|
+
media_contents = _parse_media_content(item)
|
|
1270
|
+
if media_contents:
|
|
1271
|
+
entry["media_content"] = media_contents
|
|
1272
|
+
|
|
1273
|
+
enclosures = _parse_enclosures(item)
|
|
1274
|
+
if enclosures:
|
|
1275
|
+
entry["enclosures"] = enclosures
|
|
1276
|
+
|
|
1277
|
+
# Author
|
|
1278
|
+
el = item.find(ns + "author/" + ns + "name")
|
|
1279
|
+
if el is not None and el.text:
|
|
1280
|
+
entry["author"] = el.text.strip()
|
|
1281
|
+
|
|
1282
|
+
tags = _parse_tags(item, "atom", atom_ns)
|
|
1283
|
+
if tags:
|
|
1284
|
+
entry["tags"] = tags
|
|
1285
|
+
|
|
1286
|
+
return entry
|
|
1287
|
+
|
|
1288
|
+
|
|
1196
1289
|
def _parse_feed_entry(
|
|
1197
1290
|
item: _Element,
|
|
1198
1291
|
feed_type: _FeedType,
|
|
@@ -1204,6 +1297,10 @@ def _parse_feed_entry(
|
|
|
1204
1297
|
if feed_type == "rss":
|
|
1205
1298
|
return _parse_rss_feed_entry_fast(item, atom_ns)
|
|
1206
1299
|
|
|
1300
|
+
if feed_type == "atom":
|
|
1301
|
+
return _parse_atom_feed_entry_fast(item, atom_ns)
|
|
1302
|
+
|
|
1303
|
+
# RDF path uses the generic field machinery
|
|
1207
1304
|
# Check if this is Atom 0.3 to use different date field names
|
|
1208
1305
|
is_atom_03 = atom_ns == "http://purl.org/atom/ns#"
|
|
1209
1306
|
|
|
@@ -1476,6 +1573,14 @@ def _normalize_iso_datetime_string(value: str) -> str:
|
|
|
1476
1573
|
if not cleaned:
|
|
1477
1574
|
return cleaned
|
|
1478
1575
|
|
|
1576
|
+
# Fast path: 'Z' suffix (most common in Atom feeds)
|
|
1577
|
+
if cleaned[-1] in ("Z", "z"):
|
|
1578
|
+
return cleaned[:-1] + "+00:00"
|
|
1579
|
+
|
|
1580
|
+
# Fast path: already has proper +HH:MM or -HH:MM timezone
|
|
1581
|
+
if len(cleaned) > 6 and cleaned[-6] in ("+", "-") and cleaned[-3] == ":":
|
|
1582
|
+
return cleaned
|
|
1583
|
+
|
|
1479
1584
|
upper_cleaned = cleaned.upper()
|
|
1480
1585
|
for suffix in (" UTC", " GMT", " Z"):
|
|
1481
1586
|
if upper_cleaned.endswith(suffix):
|
|
@@ -1585,7 +1690,7 @@ def _slow_dateparser(value: str) -> Optional[datetime.datetime]:
|
|
|
1585
1690
|
except ImportError:
|
|
1586
1691
|
return None
|
|
1587
1692
|
try:
|
|
1588
|
-
return _dateparser.parse(value, languages=["en"], settings=_DATEPARSER_SETTINGS)
|
|
1693
|
+
return _dateparser.parse(value, languages=["en"], settings={**_DATEPARSER_SETTINGS})
|
|
1589
1694
|
except (ValueError, TypeError):
|
|
1590
1695
|
return None
|
|
1591
1696
|
|
|
@@ -1605,6 +1710,27 @@ def _parse_date(date_str: str) -> Optional[str]:
|
|
|
1605
1710
|
candidate = date_str.strip()
|
|
1606
1711
|
if not candidate:
|
|
1607
1712
|
return None
|
|
1713
|
+
|
|
1714
|
+
# Fast path: clean ISO-8601 (covers >90% of Atom/modern RSS dates)
|
|
1715
|
+
clen = len(candidate)
|
|
1716
|
+
if clen >= 20 and candidate[4] == "-" and candidate[0:4].isdigit():
|
|
1717
|
+
last = candidate[-1]
|
|
1718
|
+
# Most common: ends with 'Z' (e.g., 2024-01-15T10:30:00Z)
|
|
1719
|
+
if last in ("Z", "z"):
|
|
1720
|
+
try:
|
|
1721
|
+
dt = datetime.datetime.fromisoformat(candidate[:-1] + "+00:00")
|
|
1722
|
+
return dt.isoformat()
|
|
1723
|
+
except ValueError:
|
|
1724
|
+
pass # Fall through to full parsing
|
|
1725
|
+
# Second most common: ends with +HH:MM (e.g., 2024-01-15T10:30:00+00:00)
|
|
1726
|
+
elif clen > 6 and candidate[-6] in ("+", "-") and candidate[-3] == ":":
|
|
1727
|
+
try:
|
|
1728
|
+
dt = datetime.datetime.fromisoformat(candidate)
|
|
1729
|
+
utc_dt = dt.replace(tzinfo=_UTC) if dt.tzinfo is None else dt.astimezone(_UTC)
|
|
1730
|
+
return utc_dt.isoformat()
|
|
1731
|
+
except (ValueError, OverflowError):
|
|
1732
|
+
pass # Fall through to full parsing
|
|
1733
|
+
|
|
1608
1734
|
if "\n" in candidate or "\r" in candidate or "\t" in candidate or " " in candidate:
|
|
1609
1735
|
candidate = _RE_WHITESPACE.sub(" ", candidate)
|
|
1610
1736
|
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{fastfeedparser-0.5.1 → fastfeedparser-0.5.2}/src/fastfeedparser.egg-info/dependency_links.txt
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|