fastfeedparser 0.5.1__tar.gz → 0.5.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.1
2
2
  Name: fastfeedparser
3
- Version: 0.5.1
3
+ Version: 0.5.2
4
4
  Summary: High performance RSS, Atom, JSON and RDF feed parser in Python
5
5
  Home-page: https://github.com/kagisearch/fastfeedparser
6
6
  Author: Vladimir Prelovac
@@ -1,6 +1,6 @@
1
1
  [metadata]
2
2
  name = fastfeedparser
3
- version = 0.5.1
3
+ version = 0.5.2
4
4
  author = Vladimir Prelovac
5
5
  author_email = vlad@kagi.com
6
6
  description = High performance RSS, Atom, JSON and RDF feed parser in Python
@@ -1096,7 +1096,7 @@ def _build_rss_item_text_maps(item: _Element) -> tuple[dict[str, Optional[str]],
1096
1096
  tag = child.tag
1097
1097
  if not isinstance(tag, str):
1098
1098
  continue
1099
- text_value = child.text.strip() if child.text else None
1099
+ text_value = child.text or None
1100
1100
  if tag not in by_full:
1101
1101
  by_full[tag] = text_value
1102
1102
  local = tag.rsplit("}", 1)[-1].lower()
@@ -1139,7 +1139,7 @@ def _parse_rss_feed_entry_fast(
1139
1139
 
1140
1140
  link = text_by_local.get("link")
1141
1141
  if link:
1142
- entry["link"] = link
1142
+ entry["link"] = link.strip()
1143
1143
 
1144
1144
  published_source = _first_non_empty(text_by_local, ("pubdate", "published", "issued", "date"))
1145
1145
  if published_source:
@@ -1180,11 +1180,11 @@ def _parse_rss_feed_entry_fast(
1180
1180
  atom_author = item.find(f"{{{atom_ns}}}author/{{{atom_ns}}}name")
1181
1181
  author = atom_author.text.strip() if atom_author is not None and atom_author.text else None
1182
1182
  if author:
1183
- entry["author"] = author
1183
+ entry["author"] = author.strip()
1184
1184
 
1185
1185
  comments = text_by_local.get("comments")
1186
1186
  if comments:
1187
- entry["comments"] = comments
1187
+ entry["comments"] = comments.strip()
1188
1188
 
1189
1189
  tags = _parse_tags(item, "rss", atom_ns)
1190
1190
  if tags:
@@ -1193,6 +1193,99 @@ def _parse_rss_feed_entry_fast(
1193
1193
  return entry
1194
1194
 
1195
1195
 
1196
+ def _parse_atom_feed_entry_fast(
1197
+ item: _Element,
1198
+ atom_ns: str,
1199
+ ) -> FastFeedParserDict:
1200
+ ns = f"{{{atom_ns}}}"
1201
+ entry = FastFeedParserDict()
1202
+
1203
+ # ID
1204
+ el = item.find(ns + "id")
1205
+ if el is not None and el.text:
1206
+ entry["id"] = el.text.strip()
1207
+
1208
+ # Title
1209
+ el = item.find(ns + "title")
1210
+ if el is not None and el.text:
1211
+ entry["title"] = el.text.strip()
1212
+
1213
+ # Description (summary)
1214
+ el = item.find(ns + "summary")
1215
+ if el is not None and el.text:
1216
+ entry["description"] = el.text.strip()
1217
+
1218
+ # Link (href attribute)
1219
+ el = item.find(ns + "link")
1220
+ if el is not None:
1221
+ href = el.get("href")
1222
+ if href:
1223
+ entry["link"] = href.strip()
1224
+
1225
+ # Dates: Atom 1.0 uses published/updated, Atom 0.3 uses issued/modified
1226
+ is_atom_03 = atom_ns == "http://purl.org/atom/ns#"
1227
+ pub_tag = "issued" if is_atom_03 else "published"
1228
+ upd_tag = "modified" if is_atom_03 else "updated"
1229
+ pub_fallback_tag = "published" if is_atom_03 else "issued"
1230
+ upd_fallback_tag = "updated" if is_atom_03 else "modified"
1231
+
1232
+ el = item.find(ns + pub_tag)
1233
+ if el is not None and el.text:
1234
+ published = _parse_date(el.text)
1235
+ if published:
1236
+ entry["published"] = published
1237
+
1238
+ el = item.find(ns + upd_tag)
1239
+ if el is not None and el.text:
1240
+ updated = _parse_date(el.text)
1241
+ if updated:
1242
+ entry["updated"] = updated
1243
+
1244
+ # Fallback date fields for mixed namespace scenarios
1245
+ if "published" not in entry:
1246
+ el = item.find(ns + pub_fallback_tag)
1247
+ if el is not None and el.text:
1248
+ published = _parse_date(el.text)
1249
+ if published:
1250
+ entry["published"] = published
1251
+
1252
+ if "updated" not in entry:
1253
+ el = item.find(ns + upd_fallback_tag)
1254
+ if el is not None and el.text:
1255
+ updated = _parse_date(el.text)
1256
+ if updated:
1257
+ entry["updated"] = updated
1258
+
1259
+ if "updated" in entry and "published" not in entry:
1260
+ entry["published"] = entry["updated"]
1261
+
1262
+ _populate_entry_links(entry, item, atom_ns)
1263
+
1264
+ if "id" not in entry and "link" in entry:
1265
+ entry["id"] = entry["link"]
1266
+
1267
+ _populate_entry_content(entry, item, "atom", atom_ns)
1268
+
1269
+ media_contents = _parse_media_content(item)
1270
+ if media_contents:
1271
+ entry["media_content"] = media_contents
1272
+
1273
+ enclosures = _parse_enclosures(item)
1274
+ if enclosures:
1275
+ entry["enclosures"] = enclosures
1276
+
1277
+ # Author
1278
+ el = item.find(ns + "author/" + ns + "name")
1279
+ if el is not None and el.text:
1280
+ entry["author"] = el.text.strip()
1281
+
1282
+ tags = _parse_tags(item, "atom", atom_ns)
1283
+ if tags:
1284
+ entry["tags"] = tags
1285
+
1286
+ return entry
1287
+
1288
+
1196
1289
  def _parse_feed_entry(
1197
1290
  item: _Element,
1198
1291
  feed_type: _FeedType,
@@ -1204,6 +1297,10 @@ def _parse_feed_entry(
1204
1297
  if feed_type == "rss":
1205
1298
  return _parse_rss_feed_entry_fast(item, atom_ns)
1206
1299
 
1300
+ if feed_type == "atom":
1301
+ return _parse_atom_feed_entry_fast(item, atom_ns)
1302
+
1303
+ # RDF path uses the generic field machinery
1207
1304
  # Check if this is Atom 0.3 to use different date field names
1208
1305
  is_atom_03 = atom_ns == "http://purl.org/atom/ns#"
1209
1306
 
@@ -1476,6 +1573,14 @@ def _normalize_iso_datetime_string(value: str) -> str:
1476
1573
  if not cleaned:
1477
1574
  return cleaned
1478
1575
 
1576
+ # Fast path: 'Z' suffix (most common in Atom feeds)
1577
+ if cleaned[-1] in ("Z", "z"):
1578
+ return cleaned[:-1] + "+00:00"
1579
+
1580
+ # Fast path: already has proper +HH:MM or -HH:MM timezone
1581
+ if len(cleaned) > 6 and cleaned[-6] in ("+", "-") and cleaned[-3] == ":":
1582
+ return cleaned
1583
+
1479
1584
  upper_cleaned = cleaned.upper()
1480
1585
  for suffix in (" UTC", " GMT", " Z"):
1481
1586
  if upper_cleaned.endswith(suffix):
@@ -1585,7 +1690,7 @@ def _slow_dateparser(value: str) -> Optional[datetime.datetime]:
1585
1690
  except ImportError:
1586
1691
  return None
1587
1692
  try:
1588
- return _dateparser.parse(value, languages=["en"], settings=_DATEPARSER_SETTINGS)
1693
+ return _dateparser.parse(value, languages=["en"], settings={**_DATEPARSER_SETTINGS})
1589
1694
  except (ValueError, TypeError):
1590
1695
  return None
1591
1696
 
@@ -1605,6 +1710,27 @@ def _parse_date(date_str: str) -> Optional[str]:
1605
1710
  candidate = date_str.strip()
1606
1711
  if not candidate:
1607
1712
  return None
1713
+
1714
+ # Fast path: clean ISO-8601 (covers >90% of Atom/modern RSS dates)
1715
+ clen = len(candidate)
1716
+ if clen >= 20 and candidate[4] == "-" and candidate[0:4].isdigit():
1717
+ last = candidate[-1]
1718
+ # Most common: ends with 'Z' (e.g., 2024-01-15T10:30:00Z)
1719
+ if last in ("Z", "z"):
1720
+ try:
1721
+ dt = datetime.datetime.fromisoformat(candidate[:-1] + "+00:00")
1722
+ return dt.isoformat()
1723
+ except ValueError:
1724
+ pass # Fall through to full parsing
1725
+ # Second most common: ends with +HH:MM (e.g., 2024-01-15T10:30:00+00:00)
1726
+ elif clen > 6 and candidate[-6] in ("+", "-") and candidate[-3] == ":":
1727
+ try:
1728
+ dt = datetime.datetime.fromisoformat(candidate)
1729
+ utc_dt = dt.replace(tzinfo=_UTC) if dt.tzinfo is None else dt.astimezone(_UTC)
1730
+ return utc_dt.isoformat()
1731
+ except (ValueError, OverflowError):
1732
+ pass # Fall through to full parsing
1733
+
1608
1734
  if "\n" in candidate or "\r" in candidate or "\t" in candidate or " " in candidate:
1609
1735
  candidate = _RE_WHITESPACE.sub(" ", candidate)
1610
1736
 
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.1
2
2
  Name: fastfeedparser
3
- Version: 0.5.1
3
+ Version: 0.5.2
4
4
  Summary: High performance RSS, Atom, JSON and RDF feed parser in Python
5
5
  Home-page: https://github.com/kagisearch/fastfeedparser
6
6
  Author: Vladimir Prelovac
File without changes
File without changes