redturtle.rssservice 2.2.2__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- redturtle/rssservice/__init__.py +6 -0
- redturtle/rssservice/configure.zcml +20 -0
- redturtle/rssservice/interfaces.py +44 -0
- redturtle/rssservice/proxycacheserver/__init__.py +1 -0
- redturtle/rssservice/proxycacheserver/main.py +328 -0
- redturtle/rssservice/rss_mixer.py +378 -0
- redturtle/rssservice/testing.py +78 -0
- redturtle/rssservice/tests/__init__.py +0 -0
- redturtle/rssservice/tests/test_proxycache.py +91 -0
- redturtle/rssservice/tests/test_rss_mixer.py +334 -0
- redturtle.rssservice-2.2.2-py3.11-nspkg.pth +1 -0
- redturtle_rssservice-2.2.2.dist-info/METADATA +293 -0
- redturtle_rssservice-2.2.2.dist-info/RECORD +19 -0
- redturtle_rssservice-2.2.2.dist-info/WHEEL +5 -0
- redturtle_rssservice-2.2.2.dist-info/entry_points.txt +5 -0
- redturtle_rssservice-2.2.2.dist-info/licenses/LICENSE.GPL +339 -0
- redturtle_rssservice-2.2.2.dist-info/licenses/LICENSE.rst +15 -0
- redturtle_rssservice-2.2.2.dist-info/namespace_packages.txt +1 -0
- redturtle_rssservice-2.2.2.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,20 @@
|
|
|
1
|
+
<configure
|
|
2
|
+
xmlns="http://namespaces.zope.org/zope"
|
|
3
|
+
xmlns:plone="http://namespaces.plone.org/plone"
|
|
4
|
+
i18n_domain="redturtle.rssservice"
|
|
5
|
+
>
|
|
6
|
+
|
|
7
|
+
<include
|
|
8
|
+
package="plone.restapi"
|
|
9
|
+
file="configure.zcml"
|
|
10
|
+
/>
|
|
11
|
+
|
|
12
|
+
<plone:service
|
|
13
|
+
method="GET"
|
|
14
|
+
factory=".rss_mixer.RSSMixerService"
|
|
15
|
+
for="*"
|
|
16
|
+
permission="zope.Public"
|
|
17
|
+
name="@rss_mixer_data"
|
|
18
|
+
/>
|
|
19
|
+
|
|
20
|
+
</configure>
|
|
@@ -0,0 +1,44 @@
|
|
|
1
|
+
# -*- coding: utf-8 -*-
|
|
2
|
+
from zope.interface import Interface
|
|
3
|
+
|
|
4
|
+
|
|
5
|
+
class IRSSMixerFeed(Interface):
|
|
6
|
+
def __init__(url, source, timeout):
|
|
7
|
+
"""Initialize the feed with the given url. will not automatically load
|
|
8
|
+
if timeout defines the time between updates in minutes.
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
def loaded():
|
|
12
|
+
"""Return if this feed is in a loaded state."""
|
|
13
|
+
|
|
14
|
+
def title():
|
|
15
|
+
"""Return the title of the feed."""
|
|
16
|
+
|
|
17
|
+
def items():
|
|
18
|
+
"""Return the items of the feed."""
|
|
19
|
+
|
|
20
|
+
def feed_link():
|
|
21
|
+
"""Return the url of this feed in feed:// format."""
|
|
22
|
+
|
|
23
|
+
def site_url():
|
|
24
|
+
"""Return the URL of the site."""
|
|
25
|
+
|
|
26
|
+
def last_update_time_in_minutes():
|
|
27
|
+
"""Return the time this feed was last updated in minutes since epoch."""
|
|
28
|
+
|
|
29
|
+
def last_update_time():
|
|
30
|
+
"""Return the time the feed was last updated as DateTime object."""
|
|
31
|
+
|
|
32
|
+
def needs_update():
|
|
33
|
+
"""return if this feed needs to be updated."""
|
|
34
|
+
|
|
35
|
+
def update():
|
|
36
|
+
"""Update this feed. will automatically check failure state etc.
|
|
37
|
+
returns True or False whether it succeeded or not.
|
|
38
|
+
"""
|
|
39
|
+
|
|
40
|
+
def update_failed():
|
|
41
|
+
"""Return if the last update failed or not."""
|
|
42
|
+
|
|
43
|
+
def ok():
|
|
44
|
+
"""Is this feed ok to display?"""
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
from .main import main # NOQA
|
|
@@ -0,0 +1,328 @@
|
|
|
1
|
+
"""
|
|
2
|
+
This code implements a caching proxy server that stores and serves web content.
|
|
3
|
+
|
|
4
|
+
Key Components:
|
|
5
|
+
* Creates unique filenames for cached content using SHA256 hashing
|
|
6
|
+
* Stores both the content and metadata (URL information) in separate files
|
|
7
|
+
* Background refresh content periodically
|
|
8
|
+
|
|
9
|
+
The Proxy Server:
|
|
10
|
+
|
|
11
|
+
* Listens for incoming requests (protected against SSRF and concurrent requests)
|
|
12
|
+
* Checks if requested content is in cache
|
|
13
|
+
* If found, serves from cache
|
|
14
|
+
* If not found, fetches it, saves it, then serves it
|
|
15
|
+
|
|
16
|
+
Background Refresh:
|
|
17
|
+
|
|
18
|
+
* Automatically updates cached content periodically
|
|
19
|
+
* Runs in separate threads to not block the main server (deduplicated per URL)
|
|
20
|
+
* Time between updates is configurable (TTL - Time To Live)
|
|
21
|
+
|
|
22
|
+
Command Line Interface: Uses Click library to accept parameters like:
|
|
23
|
+
|
|
24
|
+
Host address (default: 127.0.0.1)
|
|
25
|
+
Port number (default: 8080)
|
|
26
|
+
Cache directory location (default: ./var/cache)
|
|
27
|
+
TTL for cache refresh (default: 3600 seconds)
|
|
28
|
+
|
|
29
|
+
Usage Example:
|
|
30
|
+
|
|
31
|
+
```
|
|
32
|
+
rssmixer-proxy --host 127.0.0.1 --port 8080 --cache-dir ./var/cache --ttl 3600
|
|
33
|
+
```
|
|
34
|
+
|
|
35
|
+
XXX: this is not actually a real HTTP/HTTPS proxy because needs to act as man-in-the-middle
|
|
36
|
+
|
|
37
|
+
Usage:
|
|
38
|
+
|
|
39
|
+
```python
|
|
40
|
+
import requests
|
|
41
|
+
|
|
42
|
+
RSSMIXER_PROXY = "http://127.0.0.1:8080"
|
|
43
|
+
url = "https://abcnews.go.com/abcnews/usheadlines"
|
|
44
|
+
res = requests.get(f"{RSSMIXER_PROXY}/{url}")
|
|
45
|
+
```
|
|
46
|
+
|
|
47
|
+
This is particularly useful for:
|
|
48
|
+
|
|
49
|
+
* Reducing load on original servers
|
|
50
|
+
* Improving response times
|
|
51
|
+
* Working with content even when the original source is temporarily unavailable
|
|
52
|
+
* Saving bandwidth by not repeatedly downloading the same content
|
|
53
|
+
"""
|
|
54
|
+
|
|
55
|
+
import click
|
|
56
|
+
import hashlib
|
|
57
|
+
import http.server
|
|
58
|
+
import json
|
|
59
|
+
import logging
|
|
60
|
+
import os
|
|
61
|
+
import re
|
|
62
|
+
import requests
|
|
63
|
+
import socketserver
|
|
64
|
+
import threading
|
|
65
|
+
import time
|
|
66
|
+
from urllib.parse import urlparse
|
|
67
|
+
|
|
68
|
+
LOCK = threading.Lock()
|
|
69
|
+
LAST_ACCESS_TIMES = {}
|
|
70
|
+
ACTIVE_REFRESH_THREADS = set()
|
|
71
|
+
MAX_TTL_IN_CACHE = 7 * 24 * 3600 # 1 week
|
|
72
|
+
|
|
73
|
+
logger = logging.getLogger("rssmixer-proxy")
|
|
74
|
+
logger.setLevel(logging.INFO)
|
|
75
|
+
if not logger.handlers:
|
|
76
|
+
formatter = logging.Formatter(
|
|
77
|
+
"%(asctime)s - %(name)s - %(levelname)s - %(message)s",
|
|
78
|
+
datefmt="%Y-%m-%d %H:%M:%S",
|
|
79
|
+
)
|
|
80
|
+
stream_handler = logging.StreamHandler()
|
|
81
|
+
stream_handler.setFormatter(formatter)
|
|
82
|
+
logger.addHandler(stream_handler)
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
def cache_path(url, cache_dir):
|
|
86
|
+
hash_url = hashlib.sha256(url.encode("utf-8")).hexdigest()
|
|
87
|
+
return os.path.join(cache_dir, f"{hash_url}.json")
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
def safe_atomic_write_json(file_path, data):
|
|
91
|
+
tmp_path = f"{file_path}.tmp.{threading.get_ident()}_{time.time_ns()}"
|
|
92
|
+
try:
|
|
93
|
+
with open(tmp_path, "w", encoding="utf-8") as f:
|
|
94
|
+
json.dump(data, f, indent=2)
|
|
95
|
+
os.replace(tmp_path, file_path)
|
|
96
|
+
except Exception as e:
|
|
97
|
+
logger.error("Error writing atomically to %s: %s", file_path, e)
|
|
98
|
+
if os.path.exists(tmp_path):
|
|
99
|
+
try:
|
|
100
|
+
os.remove(tmp_path)
|
|
101
|
+
except OSError:
|
|
102
|
+
pass
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
def load_json(cache_file):
|
|
106
|
+
try:
|
|
107
|
+
if os.path.exists(cache_file):
|
|
108
|
+
with open(cache_file, "r", encoding="utf-8") as f:
|
|
109
|
+
return json.load(f)
|
|
110
|
+
except Exception as e:
|
|
111
|
+
logger.warning("Error reading cache file %s: %s", cache_file, e)
|
|
112
|
+
return {}
|
|
113
|
+
|
|
114
|
+
|
|
115
|
+
def is_valid_url(url):
|
|
116
|
+
"""Validate URL format and prevent basic SSRF targets."""
|
|
117
|
+
if not re.match(r"^https?:\/\/", url):
|
|
118
|
+
return False
|
|
119
|
+
parsed = urlparse(url)
|
|
120
|
+
hostname = parsed.hostname
|
|
121
|
+
if not hostname:
|
|
122
|
+
return False
|
|
123
|
+
forbidden_hosts = {"localhost", "127.0.0.1", "0.0.0.0", "169.254.169.254", "::1"}
|
|
124
|
+
if hostname.lower() in forbidden_hosts:
|
|
125
|
+
return False
|
|
126
|
+
return True
|
|
127
|
+
|
|
128
|
+
|
|
129
|
+
def fetch_and_cache(url, cache_dir, client_headers=None, timeout=(3, 10)):
|
|
130
|
+
cache_file = cache_path(url, cache_dir)
|
|
131
|
+
headers = {}
|
|
132
|
+
|
|
133
|
+
if client_headers is None:
|
|
134
|
+
data = load_json(cache_file)
|
|
135
|
+
headers = data.get("request_headers", {})
|
|
136
|
+
else:
|
|
137
|
+
headers = dict(client_headers)
|
|
138
|
+
|
|
139
|
+
headers.setdefault("User-Agent", "RSSMixerProxy/1.0")
|
|
140
|
+
headers.pop("Host", None)
|
|
141
|
+
|
|
142
|
+
if not is_valid_url(url):
|
|
143
|
+
logger.error("Invalid or restricted URL path: %s", url)
|
|
144
|
+
return {
|
|
145
|
+
"url": url,
|
|
146
|
+
"request_headers": headers,
|
|
147
|
+
"response_headers": {},
|
|
148
|
+
"status_code": 400,
|
|
149
|
+
"body": f"Invalid or restricted URL: {url}",
|
|
150
|
+
}
|
|
151
|
+
|
|
152
|
+
try:
|
|
153
|
+
response = requests.get(url, headers=headers, timeout=timeout)
|
|
154
|
+
cache_content = {
|
|
155
|
+
"url": url,
|
|
156
|
+
"request_headers": headers,
|
|
157
|
+
"response_headers": dict(response.headers),
|
|
158
|
+
"status_code": response.status_code,
|
|
159
|
+
"body": response.text,
|
|
160
|
+
}
|
|
161
|
+
|
|
162
|
+
if response.status_code == 200:
|
|
163
|
+
safe_atomic_write_json(cache_file, cache_content)
|
|
164
|
+
logger.info("Cached %s: %s in %s", response.status_code, url, cache_dir)
|
|
165
|
+
else:
|
|
166
|
+
logger.error("Failed to fetch %s: status %s", url, response.status_code)
|
|
167
|
+
if not os.path.exists(cache_file):
|
|
168
|
+
safe_atomic_write_json(cache_file, cache_content)
|
|
169
|
+
logger.info(
|
|
170
|
+
"Cached error %s: %s in %s", response.status_code, url, cache_dir
|
|
171
|
+
)
|
|
172
|
+
except Exception as e:
|
|
173
|
+
logger.error("Error fetching %s: %s", url, e)
|
|
174
|
+
cache_content = {
|
|
175
|
+
"url": url,
|
|
176
|
+
"request_headers": headers,
|
|
177
|
+
"response_headers": {},
|
|
178
|
+
"status_code": 502,
|
|
179
|
+
"body": str(e),
|
|
180
|
+
}
|
|
181
|
+
# Do not persist transient connection errors permanently to disk
|
|
182
|
+
return cache_content
|
|
183
|
+
|
|
184
|
+
|
|
185
|
+
def refresh_cache(url, cache_dir, ttl):
|
|
186
|
+
logger.info("Refresh cache for %s every %s seconds", url, ttl)
|
|
187
|
+
try:
|
|
188
|
+
while True:
|
|
189
|
+
time.sleep(ttl)
|
|
190
|
+
with LOCK:
|
|
191
|
+
last_access = LAST_ACCESS_TIMES.get(url, time.time())
|
|
192
|
+
if last_access + MAX_TTL_IN_CACHE < time.time():
|
|
193
|
+
cache_file = cache_path(url, cache_dir)
|
|
194
|
+
if os.path.exists(cache_file):
|
|
195
|
+
try:
|
|
196
|
+
os.remove(cache_file)
|
|
197
|
+
except OSError:
|
|
198
|
+
pass
|
|
199
|
+
LAST_ACCESS_TIMES.pop(url, None)
|
|
200
|
+
logger.warning("Remove %s from cached files due to inactivity", url)
|
|
201
|
+
return
|
|
202
|
+
|
|
203
|
+
logger.info("Refresh cache for %s", url)
|
|
204
|
+
fetch_and_cache(url, cache_dir)
|
|
205
|
+
finally:
|
|
206
|
+
with LOCK:
|
|
207
|
+
ACTIVE_REFRESH_THREADS.discard(url)
|
|
208
|
+
|
|
209
|
+
|
|
210
|
+
def ensure_refresh_thread(url, cache_dir, ttl):
|
|
211
|
+
"""Ensure at most one background refresh thread runs per URL."""
|
|
212
|
+
with LOCK:
|
|
213
|
+
if url not in ACTIVE_REFRESH_THREADS:
|
|
214
|
+
ACTIVE_REFRESH_THREADS.add(url)
|
|
215
|
+
threading.Thread(
|
|
216
|
+
target=refresh_cache, args=(url, cache_dir, ttl), daemon=True
|
|
217
|
+
).start()
|
|
218
|
+
|
|
219
|
+
|
|
220
|
+
def load_urls_from_cache(cache_dir):
|
|
221
|
+
urls = []
|
|
222
|
+
if not os.path.exists(cache_dir):
|
|
223
|
+
return urls
|
|
224
|
+
for file in os.listdir(cache_dir):
|
|
225
|
+
if file.endswith(".json"):
|
|
226
|
+
hash_file = os.path.join(cache_dir, file)
|
|
227
|
+
data = load_json(hash_file)
|
|
228
|
+
url = data.get("url", "")
|
|
229
|
+
if url:
|
|
230
|
+
logger.info("Load: %s from cache %s", url, hash_file)
|
|
231
|
+
urls.append(url)
|
|
232
|
+
return urls
|
|
233
|
+
|
|
234
|
+
|
|
235
|
+
class ThreadingHTTPServer(socketserver.ThreadingMixIn, http.server.HTTPServer):
|
|
236
|
+
daemon_threads = True
|
|
237
|
+
allow_reuse_address = True
|
|
238
|
+
|
|
239
|
+
|
|
240
|
+
class CachingProxyHandler(http.server.BaseHTTPRequestHandler):
|
|
241
|
+
def __init__(self, *args, cache_dir=None, ttl=None, **kwargs):
|
|
242
|
+
self.cache_dir = cache_dir
|
|
243
|
+
self.ttl = ttl
|
|
244
|
+
super().__init__(*args, **kwargs)
|
|
245
|
+
|
|
246
|
+
def do_GET(self):
|
|
247
|
+
url = self.path.lstrip("/").replace("\n", "").replace("\r", "")
|
|
248
|
+
with LOCK:
|
|
249
|
+
LAST_ACCESS_TIMES[url] = time.time()
|
|
250
|
+
|
|
251
|
+
cache_file = cache_path(url, self.cache_dir)
|
|
252
|
+
cache_content = load_json(cache_file)
|
|
253
|
+
|
|
254
|
+
if cache_content:
|
|
255
|
+
logger.info("Serving from cache: %s", url)
|
|
256
|
+
else:
|
|
257
|
+
logger.info("Fetching and caching: %s", url)
|
|
258
|
+
client_headers = dict(self.headers)
|
|
259
|
+
cache_content = fetch_and_cache(url, self.cache_dir, client_headers)
|
|
260
|
+
ensure_refresh_thread(url, self.cache_dir, self.ttl)
|
|
261
|
+
|
|
262
|
+
body_str = cache_content.get("body", "")
|
|
263
|
+
body_bytes = body_str.encode("utf-8")
|
|
264
|
+
status_code = cache_content.get("status_code", 500)
|
|
265
|
+
|
|
266
|
+
self.send_response(status_code)
|
|
267
|
+
response_headers = cache_content.get("response_headers", {})
|
|
268
|
+
for header, value in response_headers.items():
|
|
269
|
+
header_lower = header.lower()
|
|
270
|
+
if header_lower in ("set-cookie", "content-length"):
|
|
271
|
+
continue
|
|
272
|
+
if header_lower in (
|
|
273
|
+
"content-type",
|
|
274
|
+
"cache-control",
|
|
275
|
+
"etag",
|
|
276
|
+
"last-modified",
|
|
277
|
+
):
|
|
278
|
+
self.send_header(header, value)
|
|
279
|
+
|
|
280
|
+
self.send_header("Content-Length", str(len(body_bytes)))
|
|
281
|
+
self.end_headers()
|
|
282
|
+
self.wfile.write(body_bytes)
|
|
283
|
+
|
|
284
|
+
def log_message(self, format, *args):
|
|
285
|
+
logger.debug(
|
|
286
|
+
"%s - - [%s] %s",
|
|
287
|
+
self.address_string(),
|
|
288
|
+
self.log_date_time_string(),
|
|
289
|
+
format % args,
|
|
290
|
+
)
|
|
291
|
+
|
|
292
|
+
|
|
293
|
+
def start_server(host, port, cache_dir, ttl):
|
|
294
|
+
def handler(*args, **kwargs):
|
|
295
|
+
return CachingProxyHandler(*args, cache_dir=cache_dir, ttl=ttl, **kwargs)
|
|
296
|
+
|
|
297
|
+
with ThreadingHTTPServer((host, port), handler) as httpd:
|
|
298
|
+
try:
|
|
299
|
+
logger.info("Serving on http://%s:%s", host, port)
|
|
300
|
+
httpd.serve_forever()
|
|
301
|
+
finally:
|
|
302
|
+
logger.info("Closing connection")
|
|
303
|
+
httpd.shutdown()
|
|
304
|
+
|
|
305
|
+
|
|
306
|
+
@click.command()
|
|
307
|
+
@click.option("--host", default="127.0.0.1", help="Ip address to run the server on.")
|
|
308
|
+
@click.option("--port", default=8080, help="Port to run the server on.")
|
|
309
|
+
@click.option(
|
|
310
|
+
"--cache-dir", default="./var/cache", help="Directory to store cached files."
|
|
311
|
+
)
|
|
312
|
+
@click.option("--ttl", default=3600, help="TTL for cache refresh in seconds.")
|
|
313
|
+
def main(host, port, cache_dir, ttl):
|
|
314
|
+
os.makedirs(cache_dir, exist_ok=True)
|
|
315
|
+
cached_urls = load_urls_from_cache(cache_dir)
|
|
316
|
+
try:
|
|
317
|
+
for url in cached_urls:
|
|
318
|
+
ensure_refresh_thread(url, cache_dir, ttl)
|
|
319
|
+
|
|
320
|
+
start_server(host, port, cache_dir, ttl)
|
|
321
|
+
except KeyboardInterrupt:
|
|
322
|
+
logger.info("Server stopped.")
|
|
323
|
+
finally:
|
|
324
|
+
logger.info("Closing connection")
|
|
325
|
+
|
|
326
|
+
|
|
327
|
+
if __name__ == "__main__":
|
|
328
|
+
main()
|