boardmail 0.4.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- boardmail/__init__.py +2 -0
- boardmail/__main__.py +3 -0
- boardmail/adapter_clawdchat.py +259 -0
- boardmail/adapter_fourclaw.py +156 -0
- boardmail/adapter_fruitflies.py +152 -0
- boardmail/adapters.py +110 -0
- boardmail/cli.py +67 -0
- boardmail/commands.py +63 -0
- boardmail/config.py +84 -0
- boardmail/mcp.py +158 -0
- boardmail/providers.py +370 -0
- boardmail/store.py +231 -0
- boardmail-0.4.0.dist-info/METADATA +173 -0
- boardmail-0.4.0.dist-info/RECORD +18 -0
- boardmail-0.4.0.dist-info/WHEEL +5 -0
- boardmail-0.4.0.dist-info/entry_points.txt +3 -0
- boardmail-0.4.0.dist-info/licenses/LICENSE +21 -0
- boardmail-0.4.0.dist-info/top_level.txt +1 -0
boardmail/__init__.py
ADDED
boardmail/__main__.py
ADDED
|
@@ -0,0 +1,259 @@
|
|
|
1
|
+
"""ClawdChat notifications, confirmed through anonymous public originals."""
|
|
2
|
+
from datetime import datetime
|
|
3
|
+
from http.client import HTTPException
|
|
4
|
+
import json
|
|
5
|
+
from pathlib import Path
|
|
6
|
+
import time
|
|
7
|
+
from urllib.error import HTTPError, URLError
|
|
8
|
+
from urllib.parse import urlencode, urlsplit
|
|
9
|
+
from urllib.request import HTTPRedirectHandler, Request, build_opener
|
|
10
|
+
|
|
11
|
+
from boardmail.adapters import Batch
|
|
12
|
+
from boardmail.config import MailError, uuid
|
|
13
|
+
|
|
14
|
+
API_VERSION = 1
|
|
15
|
+
ORIGIN = "https://clawdchat.cn"
|
|
16
|
+
PAGE_SIZE = 8
|
|
17
|
+
MAX_PENDING = 256
|
|
18
|
+
MAX_REQUESTS = 40
|
|
19
|
+
MAX_RESPONSE_BYTES = 1024 * 1024
|
|
20
|
+
SOURCE_SECONDS = 45
|
|
21
|
+
KINDS = {"comment": "reply_to_post", "reply": "reply_to_comment",
|
|
22
|
+
"mention_post": "mention", "mention_comment": "mention"}
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
class NoRedirect(HTTPRedirectHandler):
|
|
26
|
+
def redirect_request(self, req, fp, code, msg, headers, newurl):
|
|
27
|
+
fp.close()
|
|
28
|
+
raise MailError("redirect_refused")
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
class Client:
|
|
32
|
+
def __init__(self, settings):
|
|
33
|
+
try:
|
|
34
|
+
self.owner = uuid(settings["account_id"])
|
|
35
|
+
except (KeyError, ValueError, TypeError, AttributeError):
|
|
36
|
+
raise MailError("invalid_config") from None
|
|
37
|
+
try:
|
|
38
|
+
with Path(settings.get("api_key_file")).open() as stream:
|
|
39
|
+
self.key = stream.read(4097).strip()
|
|
40
|
+
if not self.key or len(self.key) > 4096 or any(ord(c) < 33 or ord(c) > 126 for c in self.key):
|
|
41
|
+
raise ValueError()
|
|
42
|
+
except (OSError, UnicodeError, ValueError, TypeError):
|
|
43
|
+
raise MailError("credentials_unavailable") from None
|
|
44
|
+
self.end = time.monotonic() + SOURCE_SECONDS
|
|
45
|
+
self.deadline = self.end
|
|
46
|
+
self.requests = 0
|
|
47
|
+
self.opener = build_opener(NoRedirect())
|
|
48
|
+
|
|
49
|
+
def phase(self, seconds):
|
|
50
|
+
self.deadline = min(self.end, time.monotonic() + seconds)
|
|
51
|
+
|
|
52
|
+
def get(self, path, params=None, *, authenticated=False):
|
|
53
|
+
headers = {"Accept": "application/json", "User-Agent": "boardmail/0.2"}
|
|
54
|
+
if authenticated:
|
|
55
|
+
headers["Authorization"] = "Bearer " + self.key
|
|
56
|
+
for attempt in range(3): # At most two retries, all inside this phase's budget.
|
|
57
|
+
remaining = self.deadline - time.monotonic()
|
|
58
|
+
if remaining <= 0 or self.requests >= MAX_REQUESTS:
|
|
59
|
+
raise MailError("budget_exhausted")
|
|
60
|
+
self.requests += 1
|
|
61
|
+
request = Request(ORIGIN + "/api/v1" + path + ("?" + urlencode(params) if params else ""), headers=headers)
|
|
62
|
+
try:
|
|
63
|
+
with self.opener.open(request, timeout=min(4, remaining)) as response:
|
|
64
|
+
chunks, size = [], 0
|
|
65
|
+
while True:
|
|
66
|
+
if time.monotonic() >= self.deadline:
|
|
67
|
+
raise MailError("budget_exhausted")
|
|
68
|
+
chunk = response.read1(65536)
|
|
69
|
+
if not chunk:
|
|
70
|
+
result = json.loads(b"".join(chunks))
|
|
71
|
+
if not isinstance(result, dict) or result.get("success") is False:
|
|
72
|
+
raise MailError("invalid_response")
|
|
73
|
+
return result
|
|
74
|
+
size += len(chunk)
|
|
75
|
+
if size > MAX_RESPONSE_BYTES:
|
|
76
|
+
raise MailError("response_too_large")
|
|
77
|
+
chunks.append(chunk)
|
|
78
|
+
except HTTPError as exc:
|
|
79
|
+
code = exc.code
|
|
80
|
+
exc.close()
|
|
81
|
+
if code not in (408, 500, 502, 503, 504) or attempt == 2:
|
|
82
|
+
raise MailError("http_" + str(code)) from None
|
|
83
|
+
except (URLError, OSError, HTTPException):
|
|
84
|
+
if attempt == 2:
|
|
85
|
+
raise MailError("network_error") from None
|
|
86
|
+
raise MailError("network_error")
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
def _text(value):
|
|
90
|
+
if not isinstance(value, str):
|
|
91
|
+
raise ValueError()
|
|
92
|
+
value.encode("utf-8")
|
|
93
|
+
return value
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
def _timestamp(value):
|
|
97
|
+
parsed = datetime.fromisoformat(_text(value).replace("Z", "+00:00"))
|
|
98
|
+
if parsed.tzinfo is None:
|
|
99
|
+
raise ValueError()
|
|
100
|
+
return int(parsed.timestamp())
|
|
101
|
+
|
|
102
|
+
|
|
103
|
+
def _url(value, fallback):
|
|
104
|
+
if isinstance(value, str) and len(value) <= 2048 and "\\" not in value and not any(ord(c) < 33 or ord(c) == 127 for c in value):
|
|
105
|
+
try:
|
|
106
|
+
url = urlsplit(value)
|
|
107
|
+
if (url.scheme == "https" and url.hostname == "clawdchat.cn" and url.port in (None, 443)
|
|
108
|
+
and url.username is None and url.password is None):
|
|
109
|
+
return value
|
|
110
|
+
except ValueError:
|
|
111
|
+
pass
|
|
112
|
+
return ORIGIN + "/api/v1" + fallback
|
|
113
|
+
|
|
114
|
+
|
|
115
|
+
def _reference(item):
|
|
116
|
+
kind = KINDS.get(item["type"])
|
|
117
|
+
if kind is None:
|
|
118
|
+
return None
|
|
119
|
+
post = uuid(item["post_id"]) if item.get("post_id") else None
|
|
120
|
+
mid = uuid(item["post_id"] if item["type"] == "mention_post" else item["comment_id"])
|
|
121
|
+
return {"id": mid, "post": post, "kind": kind, "is_post": item["type"] == "mention_post"}
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
def _original(client, entry):
|
|
125
|
+
path = ("/posts/" if entry["is_post"] else "/comments/") + entry["id"]
|
|
126
|
+
original = client.get(path) # Never attach credentials to public-original requests.
|
|
127
|
+
if uuid(original["id"]) != entry["id"]:
|
|
128
|
+
raise ValueError()
|
|
129
|
+
post_id = entry["id"] if entry["is_post"] else uuid(original["post_id"])
|
|
130
|
+
if entry["post"] is not None and entry["post"] != post_id:
|
|
131
|
+
raise ValueError()
|
|
132
|
+
context = original if entry["is_post"] else original.get("post") or {}
|
|
133
|
+
if context.get("id") and uuid(context["id"]) != post_id:
|
|
134
|
+
raise ValueError()
|
|
135
|
+
for obj in (original, context):
|
|
136
|
+
if obj.get("is_deleted") or obj.get("is_hidden") or obj.get("visibility", "public") != "public":
|
|
137
|
+
raise MailError("original_unavailable")
|
|
138
|
+
author = original["author"]
|
|
139
|
+
if uuid(author["id"]) == client.owner:
|
|
140
|
+
return None
|
|
141
|
+
body = original.get("content")
|
|
142
|
+
if body is None and entry["is_post"]:
|
|
143
|
+
body = "" # Link posts may have no text body.
|
|
144
|
+
return {"id": entry["id"], "thread_id": post_id, "kind": entry["kind"],
|
|
145
|
+
"parent_id": uuid(original["parent_id"]) if original.get("parent_id") else None,
|
|
146
|
+
"author": _text(author["name"]), "title": _text(context.get("title", "Public reply")),
|
|
147
|
+
"body": _text(body), "url": _url(original.get("web_url"), path),
|
|
148
|
+
"created_at": _timestamp(original["created_at"])}
|
|
149
|
+
|
|
150
|
+
|
|
151
|
+
def _error(batch, exc):
|
|
152
|
+
code = str(exc) if isinstance(exc, MailError) else "invalid_response"
|
|
153
|
+
if code != "budget_exhausted" and (batch.error is None or code == "http_429"):
|
|
154
|
+
batch.error = code
|
|
155
|
+
batch.complete = False
|
|
156
|
+
return code
|
|
157
|
+
|
|
158
|
+
|
|
159
|
+
FAILURES = (MailError, ValueError, KeyError, TypeError, AttributeError, OverflowError)
|
|
160
|
+
|
|
161
|
+
|
|
162
|
+
def collect(settings, state, known):
|
|
163
|
+
"""Rotate retries, read fresh and backfill pages, then confirm new references.
|
|
164
|
+
|
|
165
|
+
State retains references only. Overflow evicts the oldest reference with an
|
|
166
|
+
explicit error; cyclic notification scans may rediscover it while retained.
|
|
167
|
+
"""
|
|
168
|
+
batch = Batch(state={"offset": 0, "pending": []})
|
|
169
|
+
pending = {}
|
|
170
|
+
try:
|
|
171
|
+
offset = state.get("offset", 0)
|
|
172
|
+
if type(offset) is not int or not 0 <= offset < 2**63:
|
|
173
|
+
raise ValueError()
|
|
174
|
+
batch.state["offset"] = offset
|
|
175
|
+
entries = state.get("pending", [])
|
|
176
|
+
if not isinstance(entries, list) or len(entries) > MAX_PENDING:
|
|
177
|
+
raise ValueError()
|
|
178
|
+
for entry in entries:
|
|
179
|
+
mid = uuid(entry["id"])
|
|
180
|
+
post = uuid(entry["post"]) if entry["post"] is not None else None
|
|
181
|
+
if entry["kind"] not in KINDS.values() or type(entry["is_post"]) is not bool:
|
|
182
|
+
raise ValueError()
|
|
183
|
+
if mid not in known:
|
|
184
|
+
pending[mid] = {"id": mid, "post": post, "kind": entry["kind"], "is_post": entry["is_post"]}
|
|
185
|
+
client = Client(settings)
|
|
186
|
+
client.phase(5)
|
|
187
|
+
if uuid(client.get("/agents/me", authenticated=True)["id"]) != client.owner:
|
|
188
|
+
raise MailError("account_mismatch")
|
|
189
|
+
except FAILURES as exc:
|
|
190
|
+
_error(batch, exc)
|
|
191
|
+
batch.state["pending"] = list(pending.values())
|
|
192
|
+
return batch
|
|
193
|
+
|
|
194
|
+
seen = set(known)
|
|
195
|
+
|
|
196
|
+
def resolve(ids, seconds):
|
|
197
|
+
client.phase(seconds)
|
|
198
|
+
for mid in ids:
|
|
199
|
+
if mid not in pending:
|
|
200
|
+
continue
|
|
201
|
+
entry = pending.pop(mid)
|
|
202
|
+
pending[mid] = entry # A failed original must not monopolize retries.
|
|
203
|
+
try:
|
|
204
|
+
message = _original(client, entry)
|
|
205
|
+
if message is not None:
|
|
206
|
+
batch.messages.append(message)
|
|
207
|
+
seen.add(mid)
|
|
208
|
+
del pending[mid]
|
|
209
|
+
except FAILURES as exc:
|
|
210
|
+
if isinstance(exc, MailError) and str(exc) in ("http_403", "http_404", "http_410", "original_unavailable"):
|
|
211
|
+
batch.unavailable += 1
|
|
212
|
+
else:
|
|
213
|
+
code = _error(batch, exc)
|
|
214
|
+
if code == "http_429":
|
|
215
|
+
return False
|
|
216
|
+
if code == "budget_exhausted":
|
|
217
|
+
break
|
|
218
|
+
return True
|
|
219
|
+
|
|
220
|
+
if resolve(list(pending)[:8], 15):
|
|
221
|
+
fresh = []
|
|
222
|
+
# Always inspect the head. The second request resumes an independent sweep.
|
|
223
|
+
for page_offset in dict.fromkeys((0, batch.state["offset"])):
|
|
224
|
+
client.phase(5)
|
|
225
|
+
try:
|
|
226
|
+
raw = client.get("/notifications", {"limit": PAGE_SIZE, "offset": page_offset}, authenticated=True)
|
|
227
|
+
items, total = raw["items"], raw["total"]
|
|
228
|
+
if not isinstance(items, list) or len(items) > PAGE_SIZE or type(total) is not int or not 0 <= total < 2**63:
|
|
229
|
+
raise ValueError()
|
|
230
|
+
if not items and page_offset < total:
|
|
231
|
+
raise MailError("pagination_no_progress")
|
|
232
|
+
overflow = False
|
|
233
|
+
for item in items:
|
|
234
|
+
try:
|
|
235
|
+
entry = _reference(item)
|
|
236
|
+
if entry is None or entry["id"] in seen or entry["id"] in pending:
|
|
237
|
+
continue
|
|
238
|
+
if len(pending) == MAX_PENDING:
|
|
239
|
+
del pending[next(iter(pending))]
|
|
240
|
+
overflow = True
|
|
241
|
+
_error(batch, MailError("pending_overflow"))
|
|
242
|
+
pending[entry["id"]] = entry
|
|
243
|
+
fresh.append(entry["id"])
|
|
244
|
+
except FAILURES as exc:
|
|
245
|
+
_error(batch, exc)
|
|
246
|
+
if page_offset == batch.state["offset"]:
|
|
247
|
+
following = page_offset + len(items)
|
|
248
|
+
batch.state["offset"] = page_offset if overflow else following if following < total else 0
|
|
249
|
+
except FAILURES as exc:
|
|
250
|
+
code = _error(batch, exc)
|
|
251
|
+
if page_offset and code in ("http_400", "http_422", "pagination_no_progress"):
|
|
252
|
+
batch.state["offset"] = 0
|
|
253
|
+
if code == "http_429":
|
|
254
|
+
break
|
|
255
|
+
if batch.error != "http_429":
|
|
256
|
+
resolve(fresh[:8], 15)
|
|
257
|
+
batch.state["pending"] = list(pending.values())
|
|
258
|
+
batch.complete = batch.complete and not pending and batch.state["offset"] == 0
|
|
259
|
+
return batch
|
|
@@ -0,0 +1,156 @@
|
|
|
1
|
+
"""Collect personal text from explicitly watched anonymous 4claw pages."""
|
|
2
|
+
from datetime import datetime
|
|
3
|
+
from html.parser import HTMLParser
|
|
4
|
+
from http.client import HTTPException
|
|
5
|
+
import json
|
|
6
|
+
import re
|
|
7
|
+
import time
|
|
8
|
+
from urllib.error import HTTPError, URLError
|
|
9
|
+
from urllib.request import HTTPRedirectHandler, Request, build_opener
|
|
10
|
+
from uuid import UUID
|
|
11
|
+
|
|
12
|
+
from .adapters import Batch
|
|
13
|
+
|
|
14
|
+
API_VERSION = 1
|
|
15
|
+
HOST = "https://www.4claw.org"
|
|
16
|
+
MAX_BYTES = 2_000_000
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
class _NoRedirect(HTTPRedirectHandler):
|
|
20
|
+
def redirect_request(self, req, fp, code, msg, headers, newurl):
|
|
21
|
+
return None
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
class _Page(HTMLParser):
|
|
25
|
+
"""Read visible post fields only; never evaluate scripts or embedded data."""
|
|
26
|
+
def __init__(self):
|
|
27
|
+
super().__init__(convert_charrefs=True)
|
|
28
|
+
self.stack = []
|
|
29
|
+
self.posts = []
|
|
30
|
+
self.post = None
|
|
31
|
+
self.title = ""
|
|
32
|
+
|
|
33
|
+
def handle_starttag(self, tag, attrs):
|
|
34
|
+
attrs = dict(attrs)
|
|
35
|
+
classes = (attrs.get("class") or "").split()
|
|
36
|
+
inherited = self.stack[-1][1] if self.stack else None
|
|
37
|
+
field = inherited
|
|
38
|
+
if tag in ("script", "style", "svg", "form"):
|
|
39
|
+
field = "ignore"
|
|
40
|
+
elif inherited != "ignore":
|
|
41
|
+
if "claw-post" in classes:
|
|
42
|
+
self.post = {"op": "op" in classes, "author": "", "body": "", "date": ""}
|
|
43
|
+
self.posts.append(self.post)
|
|
44
|
+
if "claw-section-title" in classes: field = "title"
|
|
45
|
+
if "claw-post-name" in classes: field = "author"
|
|
46
|
+
if "claw-post-body" in classes: field = "body"
|
|
47
|
+
if tag == "time" and self.post is not None:
|
|
48
|
+
self.post["date"] = attrs.get("datetime", "")
|
|
49
|
+
if tag == "br": self.handle_data("\n")
|
|
50
|
+
if tag not in ("br", "img", "input", "meta", "link", "hr", "wbr", "source"):
|
|
51
|
+
self.stack.append((tag, field))
|
|
52
|
+
|
|
53
|
+
def handle_endtag(self, tag):
|
|
54
|
+
for i in range(len(self.stack) - 1, -1, -1):
|
|
55
|
+
if self.stack[i][0] == tag:
|
|
56
|
+
del self.stack[i:]
|
|
57
|
+
break
|
|
58
|
+
|
|
59
|
+
def handle_data(self, data):
|
|
60
|
+
field = self.stack[-1][1] if self.stack else None
|
|
61
|
+
if field == "title": self.title += data
|
|
62
|
+
elif field in ("author", "body") and self.post is not None:
|
|
63
|
+
self.post[field] += data
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
def _fetch(thread):
|
|
67
|
+
request = Request(f"{HOST}/t/{thread}", headers={"Accept": "text/html", "User-Agent": "boardmail/1"})
|
|
68
|
+
deadline = time.monotonic() + 10
|
|
69
|
+
with build_opener(_NoRedirect).open(request, timeout=5) as response:
|
|
70
|
+
if response.headers.get_content_type() != "text/html":
|
|
71
|
+
raise ValueError("format")
|
|
72
|
+
content = bytearray()
|
|
73
|
+
while True:
|
|
74
|
+
if time.monotonic() >= deadline: raise TimeoutError()
|
|
75
|
+
chunk = response.read1(min(65536, MAX_BYTES + 1 - len(content)))
|
|
76
|
+
if time.monotonic() >= deadline: raise TimeoutError()
|
|
77
|
+
if not chunk: break
|
|
78
|
+
content.extend(chunk)
|
|
79
|
+
if len(content) > MAX_BYTES: raise ValueError("size")
|
|
80
|
+
return content.decode("utf-8")
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
def _messages(html, thread, account, aliases):
|
|
84
|
+
page = _Page()
|
|
85
|
+
page.feed(html)
|
|
86
|
+
if not page.title or not page.posts or not page.posts[0]["op"]:
|
|
87
|
+
raise ValueError("format")
|
|
88
|
+
# Next.js emits public reply UUIDs as React node keys, absent from the DOM.
|
|
89
|
+
chunks = []
|
|
90
|
+
for raw in re.findall(r'<script>self\.__next_f\.push\((.*?)\)</script>', html, re.S):
|
|
91
|
+
value = json.loads(raw)
|
|
92
|
+
if isinstance(value, list) and len(value) == 2 and value[0] == 1 and isinstance(value[1], str):
|
|
93
|
+
chunks.append(value[1])
|
|
94
|
+
ids = re.findall(r'\["\$","div","([0-9a-f-]{36})",\{"className":"claw-post reply"', ''.join(chunks))
|
|
95
|
+
if len(ids) != len(page.posts) - 1 or len(set(ids)) != len(ids):
|
|
96
|
+
raise ValueError("reply_ids")
|
|
97
|
+
if any(str(UUID(reply)) != reply for reply in ids): raise ValueError("reply_ids")
|
|
98
|
+
result = []
|
|
99
|
+
own_thread = page.posts[0]["author"].strip().casefold() == account.casefold()
|
|
100
|
+
for position, post in enumerate(page.posts):
|
|
101
|
+
author = post["author"].strip()
|
|
102
|
+
if not author or not post["date"]: raise ValueError("format")
|
|
103
|
+
date = datetime.fromisoformat(post["date"].replace("Z", "+00:00"))
|
|
104
|
+
if date.tzinfo is None: raise ValueError("date")
|
|
105
|
+
created = int(date.timestamp())
|
|
106
|
+
if not -(2**63) <= created < 2**63: raise ValueError("date")
|
|
107
|
+
if author.casefold() == account.casefold(): continue
|
|
108
|
+
mention = any(re.search(r"(?<![\w@])@" + re.escape(alias) + r"(?!\w)", post["body"], re.I) for alias in aliases)
|
|
109
|
+
if not mention and not (own_thread and not post["op"]): continue
|
|
110
|
+
result.append({"id": thread if post["op"] else ids[position - 1],
|
|
111
|
+
"thread_id": thread, "kind": "reply_to_post" if own_thread and not post["op"] else "mention",
|
|
112
|
+
"parent_id": thread if own_thread and not post["op"] else None,
|
|
113
|
+
"author": author, "title": page.title, "body": post["body"],
|
|
114
|
+
"url": f"{HOST}/t/{thread}", "created_at": created})
|
|
115
|
+
return result
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
def collect(settings, state, known):
|
|
119
|
+
"""One round-robin window, including failures, with constant-size progress."""
|
|
120
|
+
try:
|
|
121
|
+
account = settings["account_id"]
|
|
122
|
+
threads = settings["watched_threads"]
|
|
123
|
+
aliases = settings.get("mention_aliases", [account])
|
|
124
|
+
if not isinstance(account, str) or not re.fullmatch(r"[A-Za-z0-9_]{2,64}", account): raise ValueError()
|
|
125
|
+
if not isinstance(threads, list) or not 1 <= len(threads) <= 100: raise ValueError()
|
|
126
|
+
if not all(isinstance(t, str) and str(UUID(t)) == t for t in threads): raise ValueError()
|
|
127
|
+
threads = list(dict.fromkeys(threads))
|
|
128
|
+
if not isinstance(aliases, list) or len(aliases) > 20: raise ValueError()
|
|
129
|
+
if not all(isinstance(a, str) and re.fullmatch(r"[A-Za-z0-9_]{2,64}", a) for a in aliases): raise ValueError()
|
|
130
|
+
except (KeyError, ValueError, TypeError, AttributeError):
|
|
131
|
+
return Batch(state=state, complete=False, error="invalid_config")
|
|
132
|
+
offset = state.get("next_thread", 0)
|
|
133
|
+
if type(offset) is not int or not 0 <= offset < len(threads): offset = 0
|
|
134
|
+
batch = Batch(complete=False)
|
|
135
|
+
started = time.monotonic()
|
|
136
|
+
checked = 0
|
|
137
|
+
for step in range(min(4, len(threads))):
|
|
138
|
+
if step and time.monotonic() - started >= 15: break
|
|
139
|
+
thread = threads[(offset + step) % len(threads)]
|
|
140
|
+
try:
|
|
141
|
+
messages = _messages(_fetch(thread), thread, account, aliases)
|
|
142
|
+
batch.messages.extend(m for m in messages if m["id"] not in known)
|
|
143
|
+
except HTTPError as exc:
|
|
144
|
+
exc.close()
|
|
145
|
+
batch.error = f"http_{exc.code}" if exc.code in (401, 403, 404, 429, 500, 502, 503, 504) else "fourclaw_http_error"
|
|
146
|
+
batch.unavailable += 1
|
|
147
|
+
except (URLError, TimeoutError, OSError, HTTPException):
|
|
148
|
+
batch.error = "fourclaw_network_error"
|
|
149
|
+
batch.unavailable += 1
|
|
150
|
+
except (ValueError, OverflowError, UnicodeError):
|
|
151
|
+
batch.error = "fourclaw_invalid_public_page"
|
|
152
|
+
batch.unavailable += 1
|
|
153
|
+
checked += 1
|
|
154
|
+
batch.state = {"next_thread": (offset + checked) % len(threads)}
|
|
155
|
+
batch.complete = offset + checked >= len(threads) and batch.error is None
|
|
156
|
+
return batch
|
|
@@ -0,0 +1,152 @@
|
|
|
1
|
+
"""Public Fruitflies mentions and replies within a documented parent window."""
|
|
2
|
+
from datetime import datetime
|
|
3
|
+
from http.client import HTTPException
|
|
4
|
+
import json
|
|
5
|
+
import re
|
|
6
|
+
from time import monotonic
|
|
7
|
+
from urllib.error import HTTPError, URLError
|
|
8
|
+
from urllib.parse import urlencode
|
|
9
|
+
from urllib.request import HTTPRedirectHandler, Request, build_opener
|
|
10
|
+
from uuid import UUID
|
|
11
|
+
|
|
12
|
+
from boardmail.adapters import Batch
|
|
13
|
+
|
|
14
|
+
API_VERSION = 1
|
|
15
|
+
BASE = 'https://api.fruitflies.ai/v1/feed'
|
|
16
|
+
PAGE = 100
|
|
17
|
+
MAX_OFFSET = 100000
|
|
18
|
+
MAX_BYTES = 2 * 1024 * 1024
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
class FetchError(Exception):
|
|
22
|
+
pass
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
class NoRedirect(HTTPRedirectHandler):
|
|
26
|
+
def redirect_request(self, req, fp, code, msg, headers, newurl):
|
|
27
|
+
return None
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def _fetch(params):
|
|
31
|
+
# Public requests never read configured credentials or follow redirects.
|
|
32
|
+
request = Request(BASE + '?' + urlencode(params), headers={'Accept': 'application/json',
|
|
33
|
+
'User-Agent': 'boardmail/fruitflies'})
|
|
34
|
+
try:
|
|
35
|
+
deadline = monotonic() + 8
|
|
36
|
+
with build_opener(NoRedirect()).open(request, timeout=8) as response:
|
|
37
|
+
chunks = []
|
|
38
|
+
size = 0
|
|
39
|
+
while size <= MAX_BYTES:
|
|
40
|
+
if monotonic() >= deadline:
|
|
41
|
+
raise FetchError('network_timeout')
|
|
42
|
+
chunk = response.read1(min(65536, MAX_BYTES + 1 - size))
|
|
43
|
+
if not chunk:
|
|
44
|
+
break
|
|
45
|
+
chunks.append(chunk)
|
|
46
|
+
size += len(chunk)
|
|
47
|
+
raw = b''.join(chunks)
|
|
48
|
+
if len(raw) > MAX_BYTES:
|
|
49
|
+
raise FetchError('response_too_large')
|
|
50
|
+
data = json.loads(raw)
|
|
51
|
+
if not isinstance(data, dict) or not isinstance(data.get('posts'), list):
|
|
52
|
+
raise FetchError('invalid_response')
|
|
53
|
+
if len(data['posts']) > PAGE:
|
|
54
|
+
raise FetchError('invalid_response')
|
|
55
|
+
return data['posts']
|
|
56
|
+
except HTTPError as exc:
|
|
57
|
+
exc.close()
|
|
58
|
+
raise FetchError('http_' + str(exc.code)) from None
|
|
59
|
+
except (URLError, OSError, TimeoutError, HTTPException):
|
|
60
|
+
raise FetchError('network_error') from None
|
|
61
|
+
except (ValueError, UnicodeError):
|
|
62
|
+
raise FetchError('invalid_response') from None
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
def _uuid(value):
|
|
66
|
+
if not isinstance(value, str) or str(UUID(value)) != value:
|
|
67
|
+
raise ValueError()
|
|
68
|
+
return value
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def _post(raw):
|
|
72
|
+
if not isinstance(raw, dict):
|
|
73
|
+
raise ValueError()
|
|
74
|
+
ident = _uuid(raw['id'])
|
|
75
|
+
parent = raw.get('parent_id')
|
|
76
|
+
if parent is not None:
|
|
77
|
+
parent = _uuid(parent)
|
|
78
|
+
kind = raw['post_type']
|
|
79
|
+
if kind not in ('post', 'question', 'answer'):
|
|
80
|
+
raise ValueError()
|
|
81
|
+
body = raw['content']
|
|
82
|
+
author = raw['agents']['handle']
|
|
83
|
+
if not isinstance(body, str) or not isinstance(author, str):
|
|
84
|
+
raise ValueError()
|
|
85
|
+
body.encode('utf-8'); author.encode('utf-8')
|
|
86
|
+
date = datetime.fromisoformat(raw['created_at'].replace('Z', '+00:00'))
|
|
87
|
+
if date.tzinfo is None:
|
|
88
|
+
raise ValueError()
|
|
89
|
+
return dict(id=ident, parent_id=parent, post_type=kind, body=body,
|
|
90
|
+
author=author, created_at=int(date.timestamp()))
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
def collect(settings, state, known):
|
|
94
|
+
"""Read newest and one historical page; retry failed history next call."""
|
|
95
|
+
handle = settings.get('account_id')
|
|
96
|
+
if not isinstance(handle, str) or not re.fullmatch(r'[A-Za-z0-9_-]{1,64}', handle):
|
|
97
|
+
return Batch(state={}, complete=False, error='invalid_config')
|
|
98
|
+
handle = handle.lower()
|
|
99
|
+
offset = state.get('offset', PAGE)
|
|
100
|
+
if type(offset) is not int or not PAGE <= offset <= MAX_OFFSET or offset % PAGE:
|
|
101
|
+
offset = PAGE
|
|
102
|
+
result = Batch(state={'offset': offset}, complete=False)
|
|
103
|
+
mention = re.compile(r'(?<![\w@])@' + re.escape(handle) + r'(?![\w-])', re.IGNORECASE)
|
|
104
|
+
own = {}
|
|
105
|
+
own_ok = True
|
|
106
|
+
try:
|
|
107
|
+
for raw in _fetch({'agent': handle, 'limit': PAGE, 'offset': 0}):
|
|
108
|
+
try:
|
|
109
|
+
post = _post(raw)
|
|
110
|
+
if post['author'].casefold() == handle.casefold():
|
|
111
|
+
own[post['id']] = post['post_type']
|
|
112
|
+
except (KeyError, TypeError, ValueError, AttributeError, OverflowError):
|
|
113
|
+
own_ok = False
|
|
114
|
+
result.error = 'invalid_response'
|
|
115
|
+
except FetchError as exc:
|
|
116
|
+
own_ok = False
|
|
117
|
+
result.error = str(exc)
|
|
118
|
+
|
|
119
|
+
emitted = set(known)
|
|
120
|
+
for position in (0, offset):
|
|
121
|
+
try:
|
|
122
|
+
rows = _fetch({'limit': PAGE, 'offset': position})
|
|
123
|
+
except FetchError as exc:
|
|
124
|
+
result.error = str(exc)
|
|
125
|
+
continue
|
|
126
|
+
for raw in rows:
|
|
127
|
+
try:
|
|
128
|
+
post = _post(raw)
|
|
129
|
+
except (KeyError, TypeError, ValueError, AttributeError, OverflowError):
|
|
130
|
+
result.unavailable += 1
|
|
131
|
+
result.error = 'invalid_response'
|
|
132
|
+
continue
|
|
133
|
+
if post['id'] in emitted or post['author'].casefold() == handle.casefold():
|
|
134
|
+
continue
|
|
135
|
+
parent_kind = own.get(post['parent_id'])
|
|
136
|
+
if parent_kind:
|
|
137
|
+
kind = 'reply_to_comment' if parent_kind == 'answer' else 'reply_to_post'
|
|
138
|
+
elif mention.search(post['body']):
|
|
139
|
+
kind = 'mention'
|
|
140
|
+
else:
|
|
141
|
+
continue
|
|
142
|
+
result.messages.append(dict(id=post['id'], parent_id=post['parent_id'],
|
|
143
|
+
thread_id=post['parent_id'] or post['id'], kind=kind, author=post['author'],
|
|
144
|
+
title='', body=post['body'], url='https://fruitflies.ai/feed',
|
|
145
|
+
created_at=post['created_at']))
|
|
146
|
+
emitted.add(post['id'])
|
|
147
|
+
if position == offset and own_ok:
|
|
148
|
+
# Cycle the finite scan window. Re-visits also retry malformed originals.
|
|
149
|
+
end = len(rows) < PAGE or offset >= MAX_OFFSET
|
|
150
|
+
result.state['offset'] = PAGE if end else offset + PAGE
|
|
151
|
+
result.complete = end and result.error is None
|
|
152
|
+
return result
|