redturtle.rssservice 2.2.2__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,6 @@
1
+ # -*- coding: utf-8 -*-
2
+ """Init and utils."""
3
+
4
+ from zope.i18nmessageid import MessageFactory
5
+
6
+ _ = MessageFactory("design.plone.rssservice")
@@ -0,0 +1,20 @@
1
+ <configure
2
+ xmlns="http://namespaces.zope.org/zope"
3
+ xmlns:plone="http://namespaces.plone.org/plone"
4
+ i18n_domain="redturtle.rssservice"
5
+ >
6
+
7
+ <include
8
+ package="plone.restapi"
9
+ file="configure.zcml"
10
+ />
11
+
12
+ <plone:service
13
+ method="GET"
14
+ factory=".rss_mixer.RSSMixerService"
15
+ for="*"
16
+ permission="zope.Public"
17
+ name="@rss_mixer_data"
18
+ />
19
+
20
+ </configure>
@@ -0,0 +1,44 @@
1
+ # -*- coding: utf-8 -*-
2
+ from zope.interface import Interface
3
+
4
+
5
+ class IRSSMixerFeed(Interface):
6
+ def __init__(url, source, timeout):
7
+ """Initialize the feed with the given url. will not automatically load
8
+ if timeout defines the time between updates in minutes.
9
+ """
10
+
11
+ def loaded():
12
+ """Return if this feed is in a loaded state."""
13
+
14
+ def title():
15
+ """Return the title of the feed."""
16
+
17
+ def items():
18
+ """Return the items of the feed."""
19
+
20
+ def feed_link():
21
+ """Return the url of this feed in feed:// format."""
22
+
23
+ def site_url():
24
+ """Return the URL of the site."""
25
+
26
+ def last_update_time_in_minutes():
27
+ """Return the time this feed was last updated in minutes since epoch."""
28
+
29
+ def last_update_time():
30
+ """Return the time the feed was last updated as DateTime object."""
31
+
32
+ def needs_update():
33
+ """return if this feed needs to be updated."""
34
+
35
+ def update():
36
+ """Update this feed. will automatically check failure state etc.
37
+ returns True or False whether it succeeded or not.
38
+ """
39
+
40
+ def update_failed():
41
+ """Return if the last update failed or not."""
42
+
43
+ def ok():
44
+ """Is this feed ok to display?"""
@@ -0,0 +1 @@
1
+ from .main import main # NOQA
@@ -0,0 +1,328 @@
1
+ """
2
+ This code implements a caching proxy server that stores and serves web content.
3
+
4
+ Key Components:
5
+ * Creates unique filenames for cached content using SHA256 hashing
6
+ * Stores both the content and metadata (URL information) in separate files
7
+ * Background refresh content periodically
8
+
9
+ The Proxy Server:
10
+
11
+ * Listens for incoming requests (protected against SSRF and concurrent requests)
12
+ * Checks if requested content is in cache
13
+ * If found, serves from cache
14
+ * If not found, fetches it, saves it, then serves it
15
+
16
+ Background Refresh:
17
+
18
+ * Automatically updates cached content periodically
19
+ * Runs in separate threads to not block the main server (deduplicated per URL)
20
+ * Time between updates is configurable (TTL - Time To Live)
21
+
22
+ Command Line Interface: Uses Click library to accept parameters like:
23
+
24
+ Host address (default: 127.0.0.1)
25
+ Port number (default: 8080)
26
+ Cache directory location (default: ./var/cache)
27
+ TTL for cache refresh (default: 3600 seconds)
28
+
29
+ Usage Example:
30
+
31
+ ```
32
+ rssmixer-proxy --host 127.0.0.1 --port 8080 --cache-dir ./var/cache --ttl 3600
33
+ ```
34
+
35
+ XXX: this is not actually a real HTTP/HTTPS proxy because needs to act as man-in-the-middle
36
+
37
+ Usage:
38
+
39
+ ```python
40
+ import requests
41
+
42
+ RSSMIXER_PROXY = "http://127.0.0.1:8080"
43
+ url = "https://abcnews.go.com/abcnews/usheadlines"
44
+ res = requests.get(f"{RSSMIXER_PROXY}/{url}")
45
+ ```
46
+
47
+ This is particularly useful for:
48
+
49
+ * Reducing load on original servers
50
+ * Improving response times
51
+ * Working with content even when the original source is temporarily unavailable
52
+ * Saving bandwidth by not repeatedly downloading the same content
53
+ """
54
+
55
+ import click
56
+ import hashlib
57
+ import http.server
58
+ import json
59
+ import logging
60
+ import os
61
+ import re
62
+ import requests
63
+ import socketserver
64
+ import threading
65
+ import time
66
+ from urllib.parse import urlparse
67
+
68
+ LOCK = threading.Lock()
69
+ LAST_ACCESS_TIMES = {}
70
+ ACTIVE_REFRESH_THREADS = set()
71
+ MAX_TTL_IN_CACHE = 7 * 24 * 3600 # 1 week
72
+
73
+ logger = logging.getLogger("rssmixer-proxy")
74
+ logger.setLevel(logging.INFO)
75
+ if not logger.handlers:
76
+ formatter = logging.Formatter(
77
+ "%(asctime)s - %(name)s - %(levelname)s - %(message)s",
78
+ datefmt="%Y-%m-%d %H:%M:%S",
79
+ )
80
+ stream_handler = logging.StreamHandler()
81
+ stream_handler.setFormatter(formatter)
82
+ logger.addHandler(stream_handler)
83
+
84
+
85
+ def cache_path(url, cache_dir):
86
+ hash_url = hashlib.sha256(url.encode("utf-8")).hexdigest()
87
+ return os.path.join(cache_dir, f"{hash_url}.json")
88
+
89
+
90
+ def safe_atomic_write_json(file_path, data):
91
+ tmp_path = f"{file_path}.tmp.{threading.get_ident()}_{time.time_ns()}"
92
+ try:
93
+ with open(tmp_path, "w", encoding="utf-8") as f:
94
+ json.dump(data, f, indent=2)
95
+ os.replace(tmp_path, file_path)
96
+ except Exception as e:
97
+ logger.error("Error writing atomically to %s: %s", file_path, e)
98
+ if os.path.exists(tmp_path):
99
+ try:
100
+ os.remove(tmp_path)
101
+ except OSError:
102
+ pass
103
+
104
+
105
+ def load_json(cache_file):
106
+ try:
107
+ if os.path.exists(cache_file):
108
+ with open(cache_file, "r", encoding="utf-8") as f:
109
+ return json.load(f)
110
+ except Exception as e:
111
+ logger.warning("Error reading cache file %s: %s", cache_file, e)
112
+ return {}
113
+
114
+
115
+ def is_valid_url(url):
116
+ """Validate URL format and prevent basic SSRF targets."""
117
+ if not re.match(r"^https?:\/\/", url):
118
+ return False
119
+ parsed = urlparse(url)
120
+ hostname = parsed.hostname
121
+ if not hostname:
122
+ return False
123
+ forbidden_hosts = {"localhost", "127.0.0.1", "0.0.0.0", "169.254.169.254", "::1"}
124
+ if hostname.lower() in forbidden_hosts:
125
+ return False
126
+ return True
127
+
128
+
129
+ def fetch_and_cache(url, cache_dir, client_headers=None, timeout=(3, 10)):
130
+ cache_file = cache_path(url, cache_dir)
131
+ headers = {}
132
+
133
+ if client_headers is None:
134
+ data = load_json(cache_file)
135
+ headers = data.get("request_headers", {})
136
+ else:
137
+ headers = dict(client_headers)
138
+
139
+ headers.setdefault("User-Agent", "RSSMixerProxy/1.0")
140
+ headers.pop("Host", None)
141
+
142
+ if not is_valid_url(url):
143
+ logger.error("Invalid or restricted URL path: %s", url)
144
+ return {
145
+ "url": url,
146
+ "request_headers": headers,
147
+ "response_headers": {},
148
+ "status_code": 400,
149
+ "body": f"Invalid or restricted URL: {url}",
150
+ }
151
+
152
+ try:
153
+ response = requests.get(url, headers=headers, timeout=timeout)
154
+ cache_content = {
155
+ "url": url,
156
+ "request_headers": headers,
157
+ "response_headers": dict(response.headers),
158
+ "status_code": response.status_code,
159
+ "body": response.text,
160
+ }
161
+
162
+ if response.status_code == 200:
163
+ safe_atomic_write_json(cache_file, cache_content)
164
+ logger.info("Cached %s: %s in %s", response.status_code, url, cache_dir)
165
+ else:
166
+ logger.error("Failed to fetch %s: status %s", url, response.status_code)
167
+ if not os.path.exists(cache_file):
168
+ safe_atomic_write_json(cache_file, cache_content)
169
+ logger.info(
170
+ "Cached error %s: %s in %s", response.status_code, url, cache_dir
171
+ )
172
+ except Exception as e:
173
+ logger.error("Error fetching %s: %s", url, e)
174
+ cache_content = {
175
+ "url": url,
176
+ "request_headers": headers,
177
+ "response_headers": {},
178
+ "status_code": 502,
179
+ "body": str(e),
180
+ }
181
+ # Do not persist transient connection errors permanently to disk
182
+ return cache_content
183
+
184
+
185
+ def refresh_cache(url, cache_dir, ttl):
186
+ logger.info("Refresh cache for %s every %s seconds", url, ttl)
187
+ try:
188
+ while True:
189
+ time.sleep(ttl)
190
+ with LOCK:
191
+ last_access = LAST_ACCESS_TIMES.get(url, time.time())
192
+ if last_access + MAX_TTL_IN_CACHE < time.time():
193
+ cache_file = cache_path(url, cache_dir)
194
+ if os.path.exists(cache_file):
195
+ try:
196
+ os.remove(cache_file)
197
+ except OSError:
198
+ pass
199
+ LAST_ACCESS_TIMES.pop(url, None)
200
+ logger.warning("Remove %s from cached files due to inactivity", url)
201
+ return
202
+
203
+ logger.info("Refresh cache for %s", url)
204
+ fetch_and_cache(url, cache_dir)
205
+ finally:
206
+ with LOCK:
207
+ ACTIVE_REFRESH_THREADS.discard(url)
208
+
209
+
210
+ def ensure_refresh_thread(url, cache_dir, ttl):
211
+ """Ensure at most one background refresh thread runs per URL."""
212
+ with LOCK:
213
+ if url not in ACTIVE_REFRESH_THREADS:
214
+ ACTIVE_REFRESH_THREADS.add(url)
215
+ threading.Thread(
216
+ target=refresh_cache, args=(url, cache_dir, ttl), daemon=True
217
+ ).start()
218
+
219
+
220
+ def load_urls_from_cache(cache_dir):
221
+ urls = []
222
+ if not os.path.exists(cache_dir):
223
+ return urls
224
+ for file in os.listdir(cache_dir):
225
+ if file.endswith(".json"):
226
+ hash_file = os.path.join(cache_dir, file)
227
+ data = load_json(hash_file)
228
+ url = data.get("url", "")
229
+ if url:
230
+ logger.info("Load: %s from cache %s", url, hash_file)
231
+ urls.append(url)
232
+ return urls
233
+
234
+
235
+ class ThreadingHTTPServer(socketserver.ThreadingMixIn, http.server.HTTPServer):
236
+ daemon_threads = True
237
+ allow_reuse_address = True
238
+
239
+
240
+ class CachingProxyHandler(http.server.BaseHTTPRequestHandler):
241
+ def __init__(self, *args, cache_dir=None, ttl=None, **kwargs):
242
+ self.cache_dir = cache_dir
243
+ self.ttl = ttl
244
+ super().__init__(*args, **kwargs)
245
+
246
+ def do_GET(self):
247
+ url = self.path.lstrip("/").replace("\n", "").replace("\r", "")
248
+ with LOCK:
249
+ LAST_ACCESS_TIMES[url] = time.time()
250
+
251
+ cache_file = cache_path(url, self.cache_dir)
252
+ cache_content = load_json(cache_file)
253
+
254
+ if cache_content:
255
+ logger.info("Serving from cache: %s", url)
256
+ else:
257
+ logger.info("Fetching and caching: %s", url)
258
+ client_headers = dict(self.headers)
259
+ cache_content = fetch_and_cache(url, self.cache_dir, client_headers)
260
+ ensure_refresh_thread(url, self.cache_dir, self.ttl)
261
+
262
+ body_str = cache_content.get("body", "")
263
+ body_bytes = body_str.encode("utf-8")
264
+ status_code = cache_content.get("status_code", 500)
265
+
266
+ self.send_response(status_code)
267
+ response_headers = cache_content.get("response_headers", {})
268
+ for header, value in response_headers.items():
269
+ header_lower = header.lower()
270
+ if header_lower in ("set-cookie", "content-length"):
271
+ continue
272
+ if header_lower in (
273
+ "content-type",
274
+ "cache-control",
275
+ "etag",
276
+ "last-modified",
277
+ ):
278
+ self.send_header(header, value)
279
+
280
+ self.send_header("Content-Length", str(len(body_bytes)))
281
+ self.end_headers()
282
+ self.wfile.write(body_bytes)
283
+
284
+ def log_message(self, format, *args):
285
+ logger.debug(
286
+ "%s - - [%s] %s",
287
+ self.address_string(),
288
+ self.log_date_time_string(),
289
+ format % args,
290
+ )
291
+
292
+
293
+ def start_server(host, port, cache_dir, ttl):
294
+ def handler(*args, **kwargs):
295
+ return CachingProxyHandler(*args, cache_dir=cache_dir, ttl=ttl, **kwargs)
296
+
297
+ with ThreadingHTTPServer((host, port), handler) as httpd:
298
+ try:
299
+ logger.info("Serving on http://%s:%s", host, port)
300
+ httpd.serve_forever()
301
+ finally:
302
+ logger.info("Closing connection")
303
+ httpd.shutdown()
304
+
305
+
306
+ @click.command()
307
+ @click.option("--host", default="127.0.0.1", help="Ip address to run the server on.")
308
+ @click.option("--port", default=8080, help="Port to run the server on.")
309
+ @click.option(
310
+ "--cache-dir", default="./var/cache", help="Directory to store cached files."
311
+ )
312
+ @click.option("--ttl", default=3600, help="TTL for cache refresh in seconds.")
313
+ def main(host, port, cache_dir, ttl):
314
+ os.makedirs(cache_dir, exist_ok=True)
315
+ cached_urls = load_urls_from_cache(cache_dir)
316
+ try:
317
+ for url in cached_urls:
318
+ ensure_refresh_thread(url, cache_dir, ttl)
319
+
320
+ start_server(host, port, cache_dir, ttl)
321
+ except KeyboardInterrupt:
322
+ logger.info("Server stopped.")
323
+ finally:
324
+ logger.info("Closing connection")
325
+
326
+
327
+ if __name__ == "__main__":
328
+ main()