fastfeedparser 0.6.0__tar.gz → 0.6.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {fastfeedparser-0.6.0/src/fastfeedparser.egg-info → fastfeedparser-0.6.1}/PKG-INFO +33 -1
- {fastfeedparser-0.6.0 → fastfeedparser-0.6.1}/README.md +32 -0
- {fastfeedparser-0.6.0 → fastfeedparser-0.6.1}/setup.cfg +1 -1
- {fastfeedparser-0.6.0 → fastfeedparser-0.6.1}/src/fastfeedparser/__init__.py +1 -1
- {fastfeedparser-0.6.0 → fastfeedparser-0.6.1}/src/fastfeedparser/main.py +108 -8
- {fastfeedparser-0.6.0 → fastfeedparser-0.6.1/src/fastfeedparser.egg-info}/PKG-INFO +33 -1
- {fastfeedparser-0.6.0 → fastfeedparser-0.6.1}/src/fastfeedparser.egg-info/SOURCES.txt +3 -1
- fastfeedparser-0.6.1/tests/test_decompression_limits.py +172 -0
- fastfeedparser-0.6.1/tests/test_url_scheme.py +108 -0
- {fastfeedparser-0.6.0 → fastfeedparser-0.6.1}/LICENSE +0 -0
- {fastfeedparser-0.6.0 → fastfeedparser-0.6.1}/pyproject.toml +0 -0
- {fastfeedparser-0.6.0 → fastfeedparser-0.6.1}/src/fastfeedparser.egg-info/dependency_links.txt +0 -0
- {fastfeedparser-0.6.0 → fastfeedparser-0.6.1}/src/fastfeedparser.egg-info/requires.txt +0 -0
- {fastfeedparser-0.6.0 → fastfeedparser-0.6.1}/src/fastfeedparser.egg-info/top_level.txt +0 -0
- {fastfeedparser-0.6.0 → fastfeedparser-0.6.1}/tests/test_encoding.py +0 -0
- {fastfeedparser-0.6.0 → fastfeedparser-0.6.1}/tests/test_feed_image.py +0 -0
- {fastfeedparser-0.6.0 → fastfeedparser-0.6.1}/tests/test_integration.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.1
|
|
2
2
|
Name: fastfeedparser
|
|
3
|
-
Version: 0.6.
|
|
3
|
+
Version: 0.6.1
|
|
4
4
|
Summary: High performance RSS, Atom, JSON and RDF feed parser in Python
|
|
5
5
|
Home-page: https://github.com/kagisearch/fastfeedparser
|
|
6
6
|
Author: Vladimir Prelovac
|
|
@@ -122,6 +122,38 @@ Feedparser: 11 entries in 0.339s
|
|
|
122
122
|
Speedup: 50.1x
|
|
123
123
|
```
|
|
124
124
|
|
|
125
|
+
And publish a full report looking like this
|
|
126
|
+
```
|
|
127
|
+
Summary:
|
|
128
|
+
--------------------------------------------------
|
|
129
|
+
Total wall-clock time: 38.70s
|
|
130
|
+
Successfully tested 200/200 feeds
|
|
131
|
+
|
|
132
|
+
FastFeedParser:
|
|
133
|
+
Total entries: 6600
|
|
134
|
+
Total parsing time: 0.46s
|
|
135
|
+
Average per feed: 0.002s
|
|
136
|
+
Feeds/sec: 439.0
|
|
137
|
+
|
|
138
|
+
Feedparser:
|
|
139
|
+
Total entries: 6555
|
|
140
|
+
Total parsing time: 12.31s
|
|
141
|
+
Average per feed: 0.062s
|
|
142
|
+
Feeds/sec: 16.2
|
|
143
|
+
|
|
144
|
+
Speedup: FastFeedParser is 27.0x faster
|
|
145
|
+
|
|
146
|
+
OUTLIERS: Entry Count Mismatches (2 feeds)
|
|
147
|
+
--------------------------------------------------
|
|
148
|
+
https://dylanharris.org/feed-me.rss
|
|
149
|
+
FastFeedParser: 35 entries
|
|
150
|
+
Feedparser: 0 entries
|
|
151
|
+
Difference: +35
|
|
152
|
+
https://humanwhocodes.com/feeds/all.json
|
|
153
|
+
FastFeedParser: 10 entries
|
|
154
|
+
Feedparser: 0 entries
|
|
155
|
+
Difference: +10
|
|
156
|
+
```
|
|
125
157
|
|
|
126
158
|
## Key Features
|
|
127
159
|
|
|
@@ -91,6 +91,38 @@ Feedparser: 11 entries in 0.339s
|
|
|
91
91
|
Speedup: 50.1x
|
|
92
92
|
```
|
|
93
93
|
|
|
94
|
+
And publish a full report looking like this
|
|
95
|
+
```
|
|
96
|
+
Summary:
|
|
97
|
+
--------------------------------------------------
|
|
98
|
+
Total wall-clock time: 38.70s
|
|
99
|
+
Successfully tested 200/200 feeds
|
|
100
|
+
|
|
101
|
+
FastFeedParser:
|
|
102
|
+
Total entries: 6600
|
|
103
|
+
Total parsing time: 0.46s
|
|
104
|
+
Average per feed: 0.002s
|
|
105
|
+
Feeds/sec: 439.0
|
|
106
|
+
|
|
107
|
+
Feedparser:
|
|
108
|
+
Total entries: 6555
|
|
109
|
+
Total parsing time: 12.31s
|
|
110
|
+
Average per feed: 0.062s
|
|
111
|
+
Feeds/sec: 16.2
|
|
112
|
+
|
|
113
|
+
Speedup: FastFeedParser is 27.0x faster
|
|
114
|
+
|
|
115
|
+
OUTLIERS: Entry Count Mismatches (2 feeds)
|
|
116
|
+
--------------------------------------------------
|
|
117
|
+
https://dylanharris.org/feed-me.rss
|
|
118
|
+
FastFeedParser: 35 entries
|
|
119
|
+
Feedparser: 0 entries
|
|
120
|
+
Difference: +35
|
|
121
|
+
https://humanwhocodes.com/feeds/all.json
|
|
122
|
+
FastFeedParser: 10 entries
|
|
123
|
+
Feedparser: 0 entries
|
|
124
|
+
Difference: +10
|
|
125
|
+
```
|
|
94
126
|
|
|
95
127
|
## Key Features
|
|
96
128
|
|
|
@@ -2,7 +2,6 @@ from __future__ import annotations
|
|
|
2
2
|
|
|
3
3
|
import datetime
|
|
4
4
|
from email.utils import parsedate_to_datetime
|
|
5
|
-
import gzip
|
|
6
5
|
import html as _html_mod
|
|
7
6
|
import json
|
|
8
7
|
import re
|
|
@@ -23,7 +22,7 @@ try:
|
|
|
23
22
|
except ImportError:
|
|
24
23
|
_json_loads = json.loads
|
|
25
24
|
from typing import Any, Callable, Optional, Protocol, TYPE_CHECKING, Literal
|
|
26
|
-
from urllib.parse import urljoin
|
|
25
|
+
from urllib.parse import urljoin, urlsplit
|
|
27
26
|
from urllib.request import (
|
|
28
27
|
HTTPErrorProcessor,
|
|
29
28
|
HTTPRedirectHandler,
|
|
@@ -440,7 +439,100 @@ def _parse_json_feed(
|
|
|
440
439
|
return feed
|
|
441
440
|
|
|
442
441
|
|
|
442
|
+
_ALLOWED_URL_SCHEMES = frozenset(("http", "https"))
|
|
443
|
+
|
|
444
|
+
|
|
445
|
+
def _is_http_url(url: str) -> bool:
|
|
446
|
+
"""True only for http(s) URLs.
|
|
447
|
+
|
|
448
|
+
urlsplit lowercases the scheme and rejects one that is not
|
|
449
|
+
``[a-zA-Z][a-zA-Z0-9+.-]*``, so "FILE://" and "+file://" both fail here.
|
|
450
|
+
urllib.request derives Request.type the same way (lowercased text before
|
|
451
|
+
the first colon), so a URL that passes this check cannot dispatch to
|
|
452
|
+
FileHandler, FTPHandler or DataHandler.
|
|
453
|
+
"""
|
|
454
|
+
return urlsplit(url).scheme in _ALLOWED_URL_SCHEMES
|
|
455
|
+
|
|
456
|
+
|
|
457
|
+
class _SchemeRestrictedRedirectHandler(HTTPRedirectHandler):
|
|
458
|
+
"""Refuse 30x redirects that leave http(s).
|
|
459
|
+
|
|
460
|
+
urllib's default handler permits redirecting to ftp:// as well, which
|
|
461
|
+
would let a server reach a non-http scheme through the same fetch.
|
|
462
|
+
"""
|
|
463
|
+
|
|
464
|
+
def redirect_request(
|
|
465
|
+
self,
|
|
466
|
+
req: Any,
|
|
467
|
+
fp: Any,
|
|
468
|
+
code: int,
|
|
469
|
+
msg: str,
|
|
470
|
+
headers: Any,
|
|
471
|
+
newurl: str,
|
|
472
|
+
) -> Any:
|
|
473
|
+
if not _is_http_url(newurl):
|
|
474
|
+
raise ValueError(f"refusing redirect to non-http(s) URL: {newurl[:100]}")
|
|
475
|
+
return super().redirect_request(req, fp, code, msg, headers, newurl)
|
|
476
|
+
|
|
477
|
+
|
|
478
|
+
# Cap on both the bytes read off the wire and the bytes a compressed response
|
|
479
|
+
# is allowed to expand into. A ~1.7KB brotli body can otherwise inflate to 1GB.
|
|
480
|
+
_MAX_CONTENT_BYTES = 32 * 1024 * 1024
|
|
481
|
+
_BROTLI_INPUT_CHUNK = 64 * 1024
|
|
482
|
+
|
|
483
|
+
|
|
484
|
+
def _inflate_bounded(data: bytes, wbits: int, limit: int) -> bytes:
|
|
485
|
+
"""Inflate `data`, stopping once more than `limit` bytes are produced.
|
|
486
|
+
|
|
487
|
+
Returns up to limit + 1 bytes so the caller can detect the overflow.
|
|
488
|
+
`wbits` selects the container: 16 + MAX_WBITS for gzip, -MAX_WBITS for raw
|
|
489
|
+
deflate. gzip bodies may concatenate members, which gzip.decompress joined,
|
|
490
|
+
so a finished stream with trailing bytes is resumed rather than truncated.
|
|
491
|
+
"""
|
|
492
|
+
parts: list[bytes] = []
|
|
493
|
+
produced = 0
|
|
494
|
+
while True:
|
|
495
|
+
# max_length of 0 means "unlimited" to zlib; the produced > limit break
|
|
496
|
+
# below keeps limit + 1 - produced at 1 or more.
|
|
497
|
+
decompressor = zlib.decompressobj(wbits)
|
|
498
|
+
parts.append(decompressor.decompress(data, limit + 1 - produced))
|
|
499
|
+
produced += len(parts[-1])
|
|
500
|
+
if produced > limit:
|
|
501
|
+
break
|
|
502
|
+
if not decompressor.eof:
|
|
503
|
+
# Under the limit with the stream unfinished means it ended early.
|
|
504
|
+
# The one-shot decompressors raised on truncated input (gzip with
|
|
505
|
+
# EOFError, zlib with zlib.error); keep failing rather than
|
|
506
|
+
# returning a partial body as if it were the whole feed.
|
|
507
|
+
raise zlib.error("incomplete or truncated stream")
|
|
508
|
+
data = decompressor.unused_data
|
|
509
|
+
if not data:
|
|
510
|
+
break
|
|
511
|
+
return b"".join(parts)
|
|
512
|
+
|
|
513
|
+
|
|
514
|
+
def _brotli_decompress_bounded(data: bytes, limit: int) -> bytes:
|
|
515
|
+
"""Brotli-decompress `data`, producing at most limit + 1 bytes."""
|
|
516
|
+
try:
|
|
517
|
+
return brotli.Decompressor().process(data, output_buffer_limit=limit + 1)
|
|
518
|
+
except TypeError:
|
|
519
|
+
# brotli < 1.2.0 has no output cap, so feed the input in slices and
|
|
520
|
+
# check the running total instead.
|
|
521
|
+
pass
|
|
522
|
+
decompressor = brotli.Decompressor()
|
|
523
|
+
parts: list[bytes] = []
|
|
524
|
+
produced = 0
|
|
525
|
+
for start in range(0, len(data), _BROTLI_INPUT_CHUNK):
|
|
526
|
+
parts.append(decompressor.process(data[start : start + _BROTLI_INPUT_CHUNK]))
|
|
527
|
+
produced += len(parts[-1])
|
|
528
|
+
if produced > limit:
|
|
529
|
+
break
|
|
530
|
+
return b"".join(parts)
|
|
531
|
+
|
|
532
|
+
|
|
443
533
|
def _fetch_url_content(url: str) -> str | bytes:
|
|
534
|
+
if not _is_http_url(url):
|
|
535
|
+
raise ValueError(f"refusing to fetch non-http(s) URL: {url[:100]}")
|
|
444
536
|
accept_encoding = "gzip, deflate, br" if HAS_BROTLI else "gzip, deflate"
|
|
445
537
|
request = Request(
|
|
446
538
|
url,
|
|
@@ -450,20 +542,25 @@ def _fetch_url_content(url: str) -> str | bytes:
|
|
|
450
542
|
"User-Agent": "fastfeedparser (+https://github.com/kagisearch/fastfeedparser)",
|
|
451
543
|
},
|
|
452
544
|
)
|
|
453
|
-
opener = build_opener(
|
|
545
|
+
opener = build_opener(_SchemeRestrictedRedirectHandler(), HTTPErrorProcessor())
|
|
454
546
|
with opener.open(request, timeout=30) as response:
|
|
455
|
-
|
|
547
|
+
limit = _MAX_CONTENT_BYTES
|
|
548
|
+
content: bytes = response.read(limit + 1)
|
|
549
|
+
if len(content) > limit:
|
|
550
|
+
raise ValueError(f"response body exceeds {limit} bytes")
|
|
456
551
|
content_encoding = response.headers.get("Content-Encoding")
|
|
457
552
|
if content_encoding == "gzip":
|
|
458
|
-
content =
|
|
553
|
+
content = _inflate_bounded(content, 16 + zlib.MAX_WBITS, limit)
|
|
459
554
|
elif content_encoding == "deflate":
|
|
460
|
-
content =
|
|
555
|
+
content = _inflate_bounded(content, -zlib.MAX_WBITS, limit)
|
|
461
556
|
elif content_encoding == "br":
|
|
462
557
|
if not HAS_BROTLI:
|
|
463
558
|
raise ValueError(
|
|
464
559
|
"Received brotli-compressed response but 'brotli' is not installed"
|
|
465
560
|
)
|
|
466
|
-
content =
|
|
561
|
+
content = _brotli_decompress_bounded(content, limit)
|
|
562
|
+
if len(content) > limit:
|
|
563
|
+
raise ValueError(f"decompressed response exceeds {limit} bytes")
|
|
467
564
|
content_charset = response.headers.get_content_charset()
|
|
468
565
|
if content_charset:
|
|
469
566
|
try:
|
|
@@ -653,7 +750,10 @@ def _extract_meta_refresh_url(content: str | bytes, base_url: str) -> str | None
|
|
|
653
750
|
match = _RE_META_REFRESH_URL.search(meta.get("content", ""))
|
|
654
751
|
if match:
|
|
655
752
|
url = urljoin(base_url, match.group(1))
|
|
656
|
-
|
|
753
|
+
# urljoin keeps an absolute reference's own scheme, so an
|
|
754
|
+
# attacker page can name file:// or ftp:// here. Apply the
|
|
755
|
+
# same allow-list parse() applies to its source.
|
|
756
|
+
if url != base_url and _is_http_url(url):
|
|
657
757
|
return url
|
|
658
758
|
return None
|
|
659
759
|
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.1
|
|
2
2
|
Name: fastfeedparser
|
|
3
|
-
Version: 0.6.
|
|
3
|
+
Version: 0.6.1
|
|
4
4
|
Summary: High performance RSS, Atom, JSON and RDF feed parser in Python
|
|
5
5
|
Home-page: https://github.com/kagisearch/fastfeedparser
|
|
6
6
|
Author: Vladimir Prelovac
|
|
@@ -122,6 +122,38 @@ Feedparser: 11 entries in 0.339s
|
|
|
122
122
|
Speedup: 50.1x
|
|
123
123
|
```
|
|
124
124
|
|
|
125
|
+
And publish a full report looking like this
|
|
126
|
+
```
|
|
127
|
+
Summary:
|
|
128
|
+
--------------------------------------------------
|
|
129
|
+
Total wall-clock time: 38.70s
|
|
130
|
+
Successfully tested 200/200 feeds
|
|
131
|
+
|
|
132
|
+
FastFeedParser:
|
|
133
|
+
Total entries: 6600
|
|
134
|
+
Total parsing time: 0.46s
|
|
135
|
+
Average per feed: 0.002s
|
|
136
|
+
Feeds/sec: 439.0
|
|
137
|
+
|
|
138
|
+
Feedparser:
|
|
139
|
+
Total entries: 6555
|
|
140
|
+
Total parsing time: 12.31s
|
|
141
|
+
Average per feed: 0.062s
|
|
142
|
+
Feeds/sec: 16.2
|
|
143
|
+
|
|
144
|
+
Speedup: FastFeedParser is 27.0x faster
|
|
145
|
+
|
|
146
|
+
OUTLIERS: Entry Count Mismatches (2 feeds)
|
|
147
|
+
--------------------------------------------------
|
|
148
|
+
https://dylanharris.org/feed-me.rss
|
|
149
|
+
FastFeedParser: 35 entries
|
|
150
|
+
Feedparser: 0 entries
|
|
151
|
+
Difference: +35
|
|
152
|
+
https://humanwhocodes.com/feeds/all.json
|
|
153
|
+
FastFeedParser: 10 entries
|
|
154
|
+
Feedparser: 0 entries
|
|
155
|
+
Difference: +10
|
|
156
|
+
```
|
|
125
157
|
|
|
126
158
|
## Key Features
|
|
127
159
|
|
|
@@ -9,6 +9,8 @@ src/fastfeedparser.egg-info/SOURCES.txt
|
|
|
9
9
|
src/fastfeedparser.egg-info/dependency_links.txt
|
|
10
10
|
src/fastfeedparser.egg-info/requires.txt
|
|
11
11
|
src/fastfeedparser.egg-info/top_level.txt
|
|
12
|
+
tests/test_decompression_limits.py
|
|
12
13
|
tests/test_encoding.py
|
|
13
14
|
tests/test_feed_image.py
|
|
14
|
-
tests/test_integration.py
|
|
15
|
+
tests/test_integration.py
|
|
16
|
+
tests/test_url_scheme.py
|
|
@@ -0,0 +1,172 @@
|
|
|
1
|
+
"""Bounded-decompression tests for _fetch_url_content.
|
|
2
|
+
|
|
3
|
+
A compressed response body must not be allowed to expand without limit:
|
|
4
|
+
~1.7KB of brotli inflates to 1GB, which is a cheap remote memory-exhaustion.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
import gzip
|
|
8
|
+
import tracemalloc
|
|
9
|
+
import zlib
|
|
10
|
+
|
|
11
|
+
import pytest
|
|
12
|
+
|
|
13
|
+
import fastfeedparser.main as main
|
|
14
|
+
from fastfeedparser.main import HAS_BROTLI, _fetch_url_content, _inflate_bounded
|
|
15
|
+
|
|
16
|
+
FEED = (
|
|
17
|
+
b'<?xml version="1.0"?><rss version="2.0"><channel><title>ok</title>'
|
|
18
|
+
b"<item><title>one</title></item></channel></rss>"
|
|
19
|
+
)
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
class _FakeHeaders(dict):
|
|
23
|
+
def get_content_charset(self):
|
|
24
|
+
return self.get("charset")
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
class _FakeResponse:
|
|
28
|
+
def __init__(self, body: bytes, headers: dict):
|
|
29
|
+
self._body = body
|
|
30
|
+
self.headers = _FakeHeaders(headers)
|
|
31
|
+
|
|
32
|
+
def read(self, amt=None):
|
|
33
|
+
return self._body if amt is None else self._body[:amt]
|
|
34
|
+
|
|
35
|
+
def __enter__(self):
|
|
36
|
+
return self
|
|
37
|
+
|
|
38
|
+
def __exit__(self, *exc):
|
|
39
|
+
return False
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
class _FakeOpener:
|
|
43
|
+
def __init__(self, response):
|
|
44
|
+
self._response = response
|
|
45
|
+
|
|
46
|
+
def open(self, request, timeout=None):
|
|
47
|
+
return self._response
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def _serve(monkeypatch, body: bytes, headers: dict):
|
|
51
|
+
"""Point _fetch_url_content at a canned response instead of the network."""
|
|
52
|
+
monkeypatch.setattr(
|
|
53
|
+
main, "build_opener", lambda *a, **kw: _FakeOpener(_FakeResponse(body, headers))
|
|
54
|
+
)
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def _raw_deflate(data: bytes) -> bytes:
|
|
58
|
+
compressor = zlib.compressobj(9, zlib.DEFLATED, -zlib.MAX_WBITS)
|
|
59
|
+
return compressor.compress(data) + compressor.flush()
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
def test_gzip_bomb_is_refused(monkeypatch):
|
|
63
|
+
monkeypatch.setattr(main, "_MAX_CONTENT_BYTES", 1024 * 1024)
|
|
64
|
+
_serve(monkeypatch, gzip.compress(b"\0" * (64 * 1024 * 1024)), {"Content-Encoding": "gzip"})
|
|
65
|
+
|
|
66
|
+
with pytest.raises(ValueError, match="decompressed response exceeds"):
|
|
67
|
+
_fetch_url_content("https://attacker.example/feed")
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def test_deflate_bomb_is_refused(monkeypatch):
|
|
71
|
+
monkeypatch.setattr(main, "_MAX_CONTENT_BYTES", 1024 * 1024)
|
|
72
|
+
_serve(monkeypatch, _raw_deflate(b"\0" * (64 * 1024 * 1024)), {"Content-Encoding": "deflate"})
|
|
73
|
+
|
|
74
|
+
with pytest.raises(ValueError, match="decompressed response exceeds"):
|
|
75
|
+
_fetch_url_content("https://attacker.example/feed")
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
@pytest.mark.skipif(not HAS_BROTLI, reason="brotli not installed")
|
|
79
|
+
def test_brotli_bomb_is_refused(monkeypatch):
|
|
80
|
+
import brotli
|
|
81
|
+
|
|
82
|
+
monkeypatch.setattr(main, "_MAX_CONTENT_BYTES", 1024 * 1024)
|
|
83
|
+
_serve(
|
|
84
|
+
monkeypatch,
|
|
85
|
+
brotli.compress(b"\0" * (128 * 1024 * 1024), quality=5),
|
|
86
|
+
{"Content-Encoding": "br"},
|
|
87
|
+
)
|
|
88
|
+
|
|
89
|
+
with pytest.raises(ValueError, match="decompressed response exceeds"):
|
|
90
|
+
_fetch_url_content("https://attacker.example/feed")
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
@pytest.mark.skipif(not HAS_BROTLI, reason="brotli not installed")
|
|
94
|
+
def test_brotli_bomb_does_not_materialize_in_memory(monkeypatch):
|
|
95
|
+
"""The point of the cap: refusing must not require inflating the bomb first."""
|
|
96
|
+
import brotli
|
|
97
|
+
|
|
98
|
+
cap = 1024 * 1024
|
|
99
|
+
monkeypatch.setattr(main, "_MAX_CONTENT_BYTES", cap)
|
|
100
|
+
bomb = brotli.compress(b"\0" * (512 * 1024 * 1024), quality=5)
|
|
101
|
+
_serve(monkeypatch, bomb, {"Content-Encoding": "br"})
|
|
102
|
+
|
|
103
|
+
tracemalloc.start()
|
|
104
|
+
try:
|
|
105
|
+
with pytest.raises(ValueError, match="decompressed response exceeds"):
|
|
106
|
+
_fetch_url_content("https://attacker.example/feed")
|
|
107
|
+
_, peak = tracemalloc.get_traced_memory()
|
|
108
|
+
finally:
|
|
109
|
+
tracemalloc.stop()
|
|
110
|
+
|
|
111
|
+
# 512MB bomb, 1MB cap: peak must stay near the cap, not near the payload.
|
|
112
|
+
assert peak < 16 * 1024 * 1024, f"peak allocation was {peak} bytes"
|
|
113
|
+
|
|
114
|
+
|
|
115
|
+
def test_uncompressed_body_over_limit_is_refused(monkeypatch):
|
|
116
|
+
monkeypatch.setattr(main, "_MAX_CONTENT_BYTES", 4096)
|
|
117
|
+
_serve(monkeypatch, b"x" * 8192, {})
|
|
118
|
+
|
|
119
|
+
with pytest.raises(ValueError, match="response body exceeds"):
|
|
120
|
+
_fetch_url_content("https://attacker.example/feed")
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
def test_body_exactly_at_limit_is_allowed(monkeypatch):
|
|
124
|
+
monkeypatch.setattr(main, "_MAX_CONTENT_BYTES", 4096)
|
|
125
|
+
_serve(monkeypatch, b"x" * 4096, {})
|
|
126
|
+
|
|
127
|
+
assert _fetch_url_content("https://example.com/feed") == b"x" * 4096
|
|
128
|
+
|
|
129
|
+
|
|
130
|
+
@pytest.mark.parametrize("encoding", ["gzip", "deflate", "br", None])
|
|
131
|
+
def test_normal_feed_still_round_trips(monkeypatch, encoding):
|
|
132
|
+
if encoding == "gzip":
|
|
133
|
+
body = gzip.compress(FEED)
|
|
134
|
+
elif encoding == "deflate":
|
|
135
|
+
body = _raw_deflate(FEED)
|
|
136
|
+
elif encoding == "br":
|
|
137
|
+
if not HAS_BROTLI:
|
|
138
|
+
pytest.skip("brotli not installed")
|
|
139
|
+
import brotli
|
|
140
|
+
|
|
141
|
+
body = brotli.compress(FEED)
|
|
142
|
+
else:
|
|
143
|
+
body = FEED
|
|
144
|
+
|
|
145
|
+
_serve(monkeypatch, body, {"Content-Encoding": encoding} if encoding else {})
|
|
146
|
+
assert _fetch_url_content("https://example.com/feed") == FEED
|
|
147
|
+
|
|
148
|
+
|
|
149
|
+
def test_multi_member_gzip_is_not_truncated(monkeypatch):
|
|
150
|
+
"""gzip.decompress joined concatenated members; the bounded path must too."""
|
|
151
|
+
_serve(
|
|
152
|
+
monkeypatch,
|
|
153
|
+
gzip.compress(b"<rss>one</rss>") + gzip.compress(b"<rss>two</rss>"),
|
|
154
|
+
{"Content-Encoding": "gzip"},
|
|
155
|
+
)
|
|
156
|
+
assert _fetch_url_content("https://example.com/feed") == b"<rss>one</rss><rss>two</rss>"
|
|
157
|
+
|
|
158
|
+
|
|
159
|
+
def test_truncated_gzip_is_not_accepted_as_partial_body(monkeypatch):
|
|
160
|
+
"""A cut-off stream must fail, not return a truncated feed as if complete."""
|
|
161
|
+
_serve(monkeypatch, gzip.compress(FEED)[:-8], {"Content-Encoding": "gzip"})
|
|
162
|
+
|
|
163
|
+
# gzip.decompress raised EOFError here; the bounded path raises zlib.error.
|
|
164
|
+
with pytest.raises((zlib.error, EOFError)):
|
|
165
|
+
_fetch_url_content("https://example.com/feed")
|
|
166
|
+
|
|
167
|
+
|
|
168
|
+
def test_inflate_bounded_returns_one_byte_over_limit():
|
|
169
|
+
"""The overflow signal the caller checks: limit+1 bytes, never a silent cut."""
|
|
170
|
+
data = gzip.compress(b"\0" * 5000)
|
|
171
|
+
assert len(_inflate_bounded(data, 16 + zlib.MAX_WBITS, 1000)) == 1001
|
|
172
|
+
assert len(_inflate_bounded(data, 16 + zlib.MAX_WBITS, 5000)) == 5000
|
|
@@ -0,0 +1,108 @@
|
|
|
1
|
+
"""Regression tests for GHSA-xwxr-vfq4-mq54.
|
|
2
|
+
|
|
3
|
+
A meta-refresh tag in an attacker-controlled HTML response must not be able
|
|
4
|
+
to steer the follow-up fetch at a non-http(s) scheme (file://, ftp://, data:).
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
import pytest
|
|
8
|
+
|
|
9
|
+
import fastfeedparser.main as main
|
|
10
|
+
from fastfeedparser import parse
|
|
11
|
+
from fastfeedparser.main import (
|
|
12
|
+
_extract_meta_refresh_url,
|
|
13
|
+
_fetch_url_content,
|
|
14
|
+
_SchemeRestrictedRedirectHandler,
|
|
15
|
+
)
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def _meta_refresh_html(target: str) -> str:
|
|
19
|
+
return (
|
|
20
|
+
"<!doctype html><html><head>"
|
|
21
|
+
f'<meta http-equiv="refresh" content="0; url={target}">'
|
|
22
|
+
"</head><body>not a feed</body></html>"
|
|
23
|
+
)
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
@pytest.mark.parametrize(
|
|
27
|
+
"target",
|
|
28
|
+
[
|
|
29
|
+
"file:///etc/hosts",
|
|
30
|
+
"FILE:///etc/hosts",
|
|
31
|
+
"ftp://attacker.example/payload",
|
|
32
|
+
"data:text/html,<html></html>",
|
|
33
|
+
"jar:file:///etc/hosts!/",
|
|
34
|
+
],
|
|
35
|
+
)
|
|
36
|
+
def test_meta_refresh_rejects_non_http_scheme(target):
|
|
37
|
+
html = _meta_refresh_html(target)
|
|
38
|
+
assert _extract_meta_refresh_url(html, "http://attacker.example/x") is None
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def test_meta_refresh_still_follows_http_targets():
|
|
42
|
+
html = _meta_refresh_html("https://example.com/feed.xml")
|
|
43
|
+
assert (
|
|
44
|
+
_extract_meta_refresh_url(html, "http://attacker.example/x")
|
|
45
|
+
== "https://example.com/feed.xml"
|
|
46
|
+
)
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def test_meta_refresh_still_follows_relative_targets():
|
|
50
|
+
html = _meta_refresh_html("/index.xml")
|
|
51
|
+
assert (
|
|
52
|
+
_extract_meta_refresh_url(html, "https://example.com/feed/")
|
|
53
|
+
== "https://example.com/index.xml"
|
|
54
|
+
)
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def test_fetch_url_content_refuses_file_scheme(tmp_path):
|
|
58
|
+
secret = tmp_path / "secret.txt"
|
|
59
|
+
secret.write_text("SENTINEL-do-not-leak")
|
|
60
|
+
|
|
61
|
+
with pytest.raises(ValueError, match="non-http"):
|
|
62
|
+
_fetch_url_content(secret.as_uri())
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
@pytest.mark.parametrize(
|
|
66
|
+
"url",
|
|
67
|
+
["ftp://attacker.example/payload", "data:text/plain,hello", "/etc/hosts"],
|
|
68
|
+
)
|
|
69
|
+
def test_fetch_url_content_refuses_other_non_http_schemes(url):
|
|
70
|
+
with pytest.raises(ValueError, match="non-http"):
|
|
71
|
+
_fetch_url_content(url)
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def test_redirect_handler_refuses_non_http_location():
|
|
75
|
+
handler = _SchemeRestrictedRedirectHandler()
|
|
76
|
+
with pytest.raises(ValueError, match="non-http"):
|
|
77
|
+
handler.redirect_request(
|
|
78
|
+
None, None, 302, "Found", {}, "ftp://attacker.example/payload"
|
|
79
|
+
)
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def test_parse_does_not_leak_local_file_via_meta_refresh(tmp_path, monkeypatch):
|
|
83
|
+
"""End-to-end PoC from the advisory: the second fetch must not read the file."""
|
|
84
|
+
secret = tmp_path / "secret.txt"
|
|
85
|
+
secret.write_text("SENTINEL-do-not-leak")
|
|
86
|
+
|
|
87
|
+
real_fetch = main._fetch_url_content
|
|
88
|
+
calls = []
|
|
89
|
+
|
|
90
|
+
def fake_fetch(url: str):
|
|
91
|
+
calls.append(url)
|
|
92
|
+
if len(calls) == 1:
|
|
93
|
+
# Stands in for the attacker's web server, not for the library.
|
|
94
|
+
return _meta_refresh_html(secret.as_uri())
|
|
95
|
+
return real_fetch(url)
|
|
96
|
+
|
|
97
|
+
monkeypatch.setattr(main, "_fetch_url_content", fake_fetch)
|
|
98
|
+
|
|
99
|
+
with pytest.raises(ValueError) as excinfo:
|
|
100
|
+
parse("http://attacker.example/feed")
|
|
101
|
+
|
|
102
|
+
assert calls == ["http://attacker.example/feed"]
|
|
103
|
+
chain = []
|
|
104
|
+
exc = excinfo.value
|
|
105
|
+
while exc is not None:
|
|
106
|
+
chain.append(str(exc))
|
|
107
|
+
exc = exc.__cause__ or exc.__context__
|
|
108
|
+
assert "SENTINEL-do-not-leak" not in "\n".join(chain)
|
|
File without changes
|
|
File without changes
|
{fastfeedparser-0.6.0 → fastfeedparser-0.6.1}/src/fastfeedparser.egg-info/dependency_links.txt
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|