@novacraft-engineering/mailbox 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.env.example +110 -0
- package/LICENSE +21 -0
- package/README.md +75 -0
- package/app/api/mail/accessors/route.ts +197 -0
- package/app/api/mail/attachments/route.ts +54 -0
- package/app/api/mail/company-signature/route.ts +36 -0
- package/app/api/mail/contacts/route.ts +13 -0
- package/app/api/mail/emails/[id]/attachments/download/route.ts +69 -0
- package/app/api/mail/emails/[id]/attachments/forward/route.ts +65 -0
- package/app/api/mail/emails/[id]/route.ts +127 -0
- package/app/api/mail/emails/route.ts +116 -0
- package/app/api/mail/events/route.ts +12 -0
- package/app/api/mail/fonts/route.ts +44 -0
- package/app/api/mail/inbox/attachments/download/route.ts +50 -0
- package/app/api/mail/inbox/attachments/forward/route.ts +54 -0
- package/app/api/mail/inbox/attachments/route.ts +99 -0
- package/app/api/mail/inbox/body/route.ts +14 -0
- package/app/api/mail/inbox/counts/route.ts +25 -0
- package/app/api/mail/inbox/route.ts +506 -0
- package/app/api/mail/link-check/route.ts +95 -0
- package/app/api/mail/login/route.ts +38 -0
- package/app/api/mail/logout/route.ts +9 -0
- package/app/api/mail/maintenance/attachments/route.ts +26 -0
- package/app/api/mail/maintenance/bodies/route.ts +30 -0
- package/app/api/mail/maintenance/list-columns/route.ts +25 -0
- package/app/api/mail/maintenance/threads/route.ts +32 -0
- package/app/api/mail/me/route.ts +13 -0
- package/app/api/mail/outgoing-upload/route.ts +37 -0
- package/app/api/mail/password/route.ts +64 -0
- package/app/api/mail/pixel/[id]/route.ts +20 -0
- package/app/api/mail/push/route.ts +42 -0
- package/app/api/mail/render-template/route.ts +58 -0
- package/app/api/mail/request-reset/route.ts +65 -0
- package/app/api/mail/reset/route.ts +46 -0
- package/app/api/mail/send/route.ts +205 -0
- package/app/api/mail/settings/route.ts +35 -0
- package/app/api/mail/share/attachment/route.ts +91 -0
- package/app/api/mail/share/route.ts +130 -0
- package/app/api/mail/signature-logo/route.ts +67 -0
- package/app/api/mail/stash/route.ts +55 -0
- package/app/api/mail/threads/route.ts +18 -0
- package/app/api/mail/upload/route.ts +43 -0
- package/app/api/share/[id]/route.ts +85 -0
- package/app/apple-icon.png +0 -0
- package/app/brand/[file]/route.ts +37 -0
- package/app/globals.css +14 -0
- package/app/icon.png +0 -0
- package/app/layout.tsx +30 -0
- package/app/mail/AccessCheck.tsx +45 -0
- package/app/mail/AttachmentLightbox.tsx +241 -0
- package/app/mail/ConfirmDialog.tsx +88 -0
- package/app/mail/MailSelect.tsx +138 -0
- package/app/mail/RichEditor.tsx +859 -0
- package/app/mail/layout.tsx +20 -0
- package/app/mail/page.module.css +6320 -0
- package/app/mail/page.tsx +7971 -0
- package/app/mail/pwa.ts +191 -0
- package/app/mail/reset/page.tsx +132 -0
- package/app/mail/search.ts +172 -0
- package/app/manifest.ts +21 -0
- package/app/page.tsx +5 -0
- package/app/robots.ts +22 -0
- package/app/share/[id]/page.tsx +166 -0
- package/app/share/[id]/share.module.css +148 -0
- package/eslint.config.mjs +9 -0
- package/lib/accent-ramp.ts +54 -0
- package/lib/attachments.ts +52 -0
- package/lib/brand.client.ts +52 -0
- package/lib/brand.ts +87 -0
- package/lib/d1.ts +64 -0
- package/lib/default-signature.ts +68 -0
- package/lib/dev-auth.ts +222 -0
- package/lib/email-html.test.ts +74 -0
- package/lib/email-html.ts +169 -0
- package/lib/emails/index.ts +281 -0
- package/lib/emails/templates/academy-followup.html +68 -0
- package/lib/emails/templates/auto-reply.html +33 -0
- package/lib/emails/templates/contact-followup.html +58 -0
- package/lib/emails/templates/field-row.html +4 -0
- package/lib/emails/templates/notification.html +23 -0
- package/lib/fonts.ts +23 -0
- package/lib/link-safety.ts +81 -0
- package/lib/mail-provider.ts +183 -0
- package/lib/mailbox.ts +1888 -0
- package/lib/password.ts +71 -0
- package/lib/public-url.ts +14 -0
- package/lib/push.ts +37 -0
- package/lib/r2.ts +157 -0
- package/lib/rate-limit.ts +43 -0
- package/lib/session.test.ts +43 -0
- package/lib/session.ts +80 -0
- package/lib/signature.ts +28 -0
- package/lib/threads.test.ts +96 -0
- package/lib/threads.ts +72 -0
- package/lib/turso.ts +53 -0
- package/next.config.ts +51 -0
- package/package.json +69 -0
- package/public/brand/README.md +35 -0
- package/public/icon-192.png +0 -0
- package/public/icon-512.png +0 -0
- package/public/icon-maskable-512.png +0 -0
- package/public/sw.js +41 -0
- package/scripts/backfill-attachments.mjs +96 -0
- package/scripts/backfill-content-ids.mjs +101 -0
- package/scripts/backfill-list-columns.mjs +80 -0
- package/scripts/offload-bodies.mjs +152 -0
- package/scripts/sample-files.mjs +119 -0
- package/scripts/seed-dev.mjs +240 -0
- package/tools/README.md +18 -0
- package/tools/db-backup.py +158 -0
- package/tools/doh.py +46 -0
- package/tools/import-mbox.py +703 -0
- package/tools/import-takeout.sh +49 -0
- package/tsconfig.json +35 -0
|
@@ -0,0 +1,703 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""Stream an mbox archive into the mailbox tables.
|
|
3
|
+
|
|
4
|
+
Runs in constant memory: messages are split off the file one at a time and written in
|
|
5
|
+
small chunks. Row ids derive from Message-ID, and every chunk checks which ids already
|
|
6
|
+
exist before uploading anything, so a re-run — or a crash and restart — never duplicates
|
|
7
|
+
a row or a blob.
|
|
8
|
+
|
|
9
|
+
Gmail labels drive routing: Spam and Drafts are skipped, Sent goes to the sent folder,
|
|
10
|
+
Trash and Starred become flags, and anything without an Inbox label is archived.
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
import argparse
|
|
14
|
+
import concurrent.futures
|
|
15
|
+
import email
|
|
16
|
+
import email.utils
|
|
17
|
+
import datetime
|
|
18
|
+
import hashlib
|
|
19
|
+
import hmac
|
|
20
|
+
import json
|
|
21
|
+
import secrets
|
|
22
|
+
import os
|
|
23
|
+
import re
|
|
24
|
+
import socket
|
|
25
|
+
import sys
|
|
26
|
+
import threading
|
|
27
|
+
import time
|
|
28
|
+
import urllib.error
|
|
29
|
+
import urllib.parse
|
|
30
|
+
import urllib.request
|
|
31
|
+
from datetime import timezone
|
|
32
|
+
from email.header import decode_header, make_header
|
|
33
|
+
|
|
34
|
+
BLOB_LIMIT_BYTES = 20 * 1024 * 1024
|
|
35
|
+
|
|
36
|
+
# A home router whose resolver flaps takes every request down with it while raw IP
|
|
37
|
+
# connectivity is fine. When the system resolver fails, ask 1.1.1.1 by address over
|
|
38
|
+
# HTTPS instead — scoped to this process, so nothing on the machine is changed.
|
|
39
|
+
_system_getaddrinfo = socket.getaddrinfo
|
|
40
|
+
_doh_cache = {}
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def _doh_lookup(host):
|
|
44
|
+
request = urllib.request.Request(
|
|
45
|
+
f'https://1.1.1.1/dns-query?name={urllib.parse.quote(host)}&type=A',
|
|
46
|
+
headers={'accept': 'application/dns-json'},
|
|
47
|
+
)
|
|
48
|
+
with urllib.request.urlopen(request, timeout=10) as response:
|
|
49
|
+
answers = json.load(response).get('Answer') or []
|
|
50
|
+
addresses = [entry['data'] for entry in answers if entry.get('type') == 1]
|
|
51
|
+
if not addresses:
|
|
52
|
+
raise socket.gaierror(f'no A record for {host} via DoH')
|
|
53
|
+
return addresses
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def _resilient_getaddrinfo(host, port, family=0, kind=0, proto=0, flags=0):
|
|
57
|
+
try:
|
|
58
|
+
return _system_getaddrinfo(host, port, family, kind, proto, flags)
|
|
59
|
+
except socket.gaierror:
|
|
60
|
+
if host in ('1.1.1.1',) or not isinstance(host, str):
|
|
61
|
+
raise
|
|
62
|
+
if host not in _doh_cache:
|
|
63
|
+
_doh_cache[host] = _doh_lookup(host)
|
|
64
|
+
print(f' ~ system DNS failed for {host}; using DoH answer {_doh_cache[host][0]}', flush=True)
|
|
65
|
+
return [(socket.AF_INET, socket.SOCK_STREAM, socket.IPPROTO_TCP, '', (address, port))
|
|
66
|
+
for address in _doh_cache[host]]
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
socket.getaddrinfo = _resilient_getaddrinfo
|
|
70
|
+
BODY_CAP = 4_000_000
|
|
71
|
+
INBOX_COLUMNS = ('id, from_addr, to_addrs, cc, bcc, reply_to, subject, html, body_text, headers, '
|
|
72
|
+
'received_at, read, attachments, owner, starred, archived, trashed, labels')
|
|
73
|
+
INSERT_INBOX = (f'INSERT INTO mail_inbox ({INBOX_COLUMNS}) VALUES ({",".join("?" * 18)}) '
|
|
74
|
+
'ON CONFLICT (id) DO NOTHING')
|
|
75
|
+
INSERT_SENT = ('INSERT INTO mail_sent (id, from_addr, to_addrs, cc, bcc, reply_to, subject, html, body_text, '
|
|
76
|
+
'created_at, last_event, provider) VALUES (?,?,?,?,?,?,?,?,?,?,?,?) ON CONFLICT (id) DO NOTHING')
|
|
77
|
+
INSERT_SENT_META = ('INSERT INTO mail_sent_meta (email_id, owner, is_auto, created_at) VALUES (?,?,0,?) '
|
|
78
|
+
'ON CONFLICT (email_id) DO NOTHING')
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
def messages(path):
|
|
82
|
+
"""Yield raw message bytes. Takeout escapes body 'From ' lines, so the bare form is a boundary.
|
|
83
|
+
A path of '-' reads the mbox from stdin, so a zip can be streamed in without extracting it."""
|
|
84
|
+
buffer = bytearray()
|
|
85
|
+
handle = sys.stdin.buffer if path == '-' else open(path, 'rb')
|
|
86
|
+
with handle:
|
|
87
|
+
for line in handle:
|
|
88
|
+
if line.startswith(b'From '):
|
|
89
|
+
if buffer:
|
|
90
|
+
yield bytes(buffer)
|
|
91
|
+
buffer = bytearray()
|
|
92
|
+
continue
|
|
93
|
+
buffer += line
|
|
94
|
+
if buffer:
|
|
95
|
+
yield bytes(buffer)
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
def decoded(value):
|
|
99
|
+
if not value:
|
|
100
|
+
return ''
|
|
101
|
+
try:
|
|
102
|
+
return str(make_header(decode_header(str(value))))
|
|
103
|
+
except Exception:
|
|
104
|
+
return str(value)
|
|
105
|
+
|
|
106
|
+
|
|
107
|
+
def addresses(message, header):
|
|
108
|
+
found = []
|
|
109
|
+
for raw in message.get_all(header, []):
|
|
110
|
+
for _, address in email.utils.getaddresses([str(raw)]):
|
|
111
|
+
if address:
|
|
112
|
+
found.append(address.lower())
|
|
113
|
+
return found
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
def received_at(message):
|
|
117
|
+
raw = message.get('Date')
|
|
118
|
+
if not raw:
|
|
119
|
+
return None
|
|
120
|
+
try:
|
|
121
|
+
parsed = email.utils.parsedate_to_datetime(str(raw))
|
|
122
|
+
except Exception:
|
|
123
|
+
return None
|
|
124
|
+
if parsed.tzinfo is None:
|
|
125
|
+
parsed = parsed.replace(tzinfo=timezone.utc)
|
|
126
|
+
return parsed.astimezone(timezone.utc).isoformat(timespec='milliseconds').replace('+00:00', 'Z')
|
|
127
|
+
|
|
128
|
+
|
|
129
|
+
def bodies(message, keep_bytes=True):
|
|
130
|
+
html, text, attachments = None, None, []
|
|
131
|
+
for part in (message.walk() if message.is_multipart() else [message]):
|
|
132
|
+
if part.get_content_maintype() == 'multipart':
|
|
133
|
+
continue
|
|
134
|
+
filename = part.get_filename()
|
|
135
|
+
disposition = str(part.get('Content-Disposition') or '')
|
|
136
|
+
if filename or 'attachment' in disposition.lower():
|
|
137
|
+
raw = b''
|
|
138
|
+
if keep_bytes:
|
|
139
|
+
try:
|
|
140
|
+
raw = part.get_payload(decode=True) or b''
|
|
141
|
+
except Exception:
|
|
142
|
+
raw = b''
|
|
143
|
+
attachments.append({
|
|
144
|
+
'filename': decoded(filename) or 'attachment',
|
|
145
|
+
'contentType': part.get_content_type(),
|
|
146
|
+
'size': len(raw),
|
|
147
|
+
'bytes': raw,
|
|
148
|
+
})
|
|
149
|
+
continue
|
|
150
|
+
try:
|
|
151
|
+
payload = part.get_payload(decode=True)
|
|
152
|
+
except Exception:
|
|
153
|
+
continue
|
|
154
|
+
if payload is None:
|
|
155
|
+
continue
|
|
156
|
+
charset = part.get_content_charset() or 'utf-8'
|
|
157
|
+
try:
|
|
158
|
+
body = payload.decode(charset, errors='replace')
|
|
159
|
+
except LookupError:
|
|
160
|
+
body = payload.decode('utf-8', errors='replace')
|
|
161
|
+
if part.get_content_type() == 'text/html' and html is None:
|
|
162
|
+
html = body[:BODY_CAP]
|
|
163
|
+
elif part.get_content_type() == 'text/plain' and text is None:
|
|
164
|
+
text = body[:BODY_CAP]
|
|
165
|
+
return html, text, attachments
|
|
166
|
+
|
|
167
|
+
|
|
168
|
+
def row_for(message, owner, force_read, keep_bytes=True):
|
|
169
|
+
message_id = str(message.get('Message-ID') or '').strip()
|
|
170
|
+
seed = message_id or '|'.join([
|
|
171
|
+
str(message.get('Date') or ''), str(message.get('From') or ''), str(message.get('Subject') or ''),
|
|
172
|
+
])
|
|
173
|
+
labels = [label.strip() for label in str(message.get('X-Gmail-Labels') or '').split(',') if label.strip()]
|
|
174
|
+
sender = addresses(message, 'From')
|
|
175
|
+
from_addr = sender[0] if sender else decoded(message.get('From'))
|
|
176
|
+
is_sent = 'Sent' in labels or (bool(sender) and sender[0] == owner)
|
|
177
|
+
html, text, attachments = bodies(message, keep_bytes)
|
|
178
|
+
primary_id = 'mbox-' + hashlib.sha256(seed.encode('utf-8', 'replace')).hexdigest()[:28]
|
|
179
|
+
return {
|
|
180
|
+
'id': primary_id,
|
|
181
|
+
'primary_id': primary_id,
|
|
182
|
+
'alt_id': 'mbox-' + hashlib.sha256((seed + '|' + owner).encode('utf-8', 'replace')).hexdigest()[:28],
|
|
183
|
+
'kind': 'skip' if ('Spam' in labels or 'Drafts' in labels) else ('sent' if is_sent else 'inbox'),
|
|
184
|
+
'from': from_addr,
|
|
185
|
+
'to': addresses(message, 'To') or addresses(message, 'Delivered-To'),
|
|
186
|
+
'cc': addresses(message, 'Cc'),
|
|
187
|
+
'bcc': addresses(message, 'Bcc'),
|
|
188
|
+
'replyTo': addresses(message, 'Reply-To'),
|
|
189
|
+
'subject': decoded(message.get('Subject')),
|
|
190
|
+
'html': html,
|
|
191
|
+
'text': text,
|
|
192
|
+
'headers': {'message-id': message_id, 'x-gmail-labels': ','.join(labels),
|
|
193
|
+
'date': str(message.get('Date') or ''), 'imported-from': 'mbox'},
|
|
194
|
+
'receivedAt': received_at(message),
|
|
195
|
+
'read': True if force_read else ('Unread' not in labels),
|
|
196
|
+
'attachments': attachments,
|
|
197
|
+
'owner': owner,
|
|
198
|
+
'labels': labels,
|
|
199
|
+
'starred': 'Starred' in labels,
|
|
200
|
+
'trashed': 'Trash' in labels,
|
|
201
|
+
'archived': 'Inbox' not in labels and 'Trash' not in labels and not is_sent,
|
|
202
|
+
}
|
|
203
|
+
|
|
204
|
+
|
|
205
|
+
RETRY_DELAYS = (5, 10, 20, 40, 60, 90, 120, 180, 300)
|
|
206
|
+
|
|
207
|
+
|
|
208
|
+
def transient(err):
|
|
209
|
+
"""A dropped network, DNS failure, timeout, or a 429/5xx from the service — not a bad statement."""
|
|
210
|
+
if isinstance(err, urllib.error.HTTPError):
|
|
211
|
+
return err.code == 429 or err.code >= 500
|
|
212
|
+
return isinstance(err, (urllib.error.URLError, OSError, TimeoutError))
|
|
213
|
+
|
|
214
|
+
|
|
215
|
+
def with_retry(action, label):
|
|
216
|
+
for attempt, delay in enumerate(RETRY_DELAYS + (None,)):
|
|
217
|
+
try:
|
|
218
|
+
return action()
|
|
219
|
+
except Exception as err:
|
|
220
|
+
if delay is None or not transient(err):
|
|
221
|
+
raise
|
|
222
|
+
print(f' ~ {label}: {str(err)[:80]} — retry {attempt + 1} in {delay}s', flush=True)
|
|
223
|
+
time.sleep(delay)
|
|
224
|
+
|
|
225
|
+
|
|
226
|
+
class UploadBudget:
|
|
227
|
+
"""Leaky-bucket schedule shared by every upload thread: each upload reserves the time
|
|
228
|
+
its bytes take at the cap and waits for its slot, so the aggregate never exceeds the cap
|
|
229
|
+
and downloads (the importer's own database replies included) are not starved."""
|
|
230
|
+
|
|
231
|
+
def __init__(self, mbps):
|
|
232
|
+
self.bytes_per_second = mbps * 1_000_000 / 8 if mbps else None
|
|
233
|
+
self.next_slot = time.monotonic()
|
|
234
|
+
self.lock = threading.Lock()
|
|
235
|
+
|
|
236
|
+
def reserve(self, size):
|
|
237
|
+
if not self.bytes_per_second:
|
|
238
|
+
return
|
|
239
|
+
with self.lock:
|
|
240
|
+
now = time.monotonic()
|
|
241
|
+
# Never let the schedule drift more than a minute ahead of the clock: failed
|
|
242
|
+
# attempts and clock hiccups would otherwise park every uploader asleep.
|
|
243
|
+
start = min(max(now, self.next_slot), now + 60)
|
|
244
|
+
self.next_slot = start + size / self.bytes_per_second
|
|
245
|
+
wait = start - now
|
|
246
|
+
if wait > 0:
|
|
247
|
+
time.sleep(wait)
|
|
248
|
+
|
|
249
|
+
|
|
250
|
+
UPLOAD_BUDGET = UploadBudget(None)
|
|
251
|
+
|
|
252
|
+
|
|
253
|
+
class S3Store:
|
|
254
|
+
"""S3-API object store (Cloudflare R2 or AWS S3) signed with SigV4 by hand — the same
|
|
255
|
+
scheme lib/r2.ts uses, so the app can presign downloads for whatever this uploads."""
|
|
256
|
+
|
|
257
|
+
def __init__(self):
|
|
258
|
+
env = os.environ.get
|
|
259
|
+
self.endpoint = (env('S3_ENDPOINT') or env('R2_S3_ENDPOINT') or '').rstrip('/')
|
|
260
|
+
self.bucket = env('S3_BUCKET') or env('R2_BUCKET') or ''
|
|
261
|
+
self.region = env('S3_REGION') or env('R2_REGION') or 'auto'
|
|
262
|
+
self.access_key = env('AWS_ACCESS_KEY_ID') or env('R2_ACCESS_KEY_ID') or ''
|
|
263
|
+
self.secret_key = env('AWS_SECRET_ACCESS_KEY') or env('R2_SECRET_ACCESS_KEY') or ''
|
|
264
|
+
|
|
265
|
+
def configured(self):
|
|
266
|
+
return all([self.endpoint, self.bucket, self.access_key, self.secret_key])
|
|
267
|
+
|
|
268
|
+
def put(self, key, body, content_type):
|
|
269
|
+
host = urllib.parse.urlparse(self.endpoint).netloc
|
|
270
|
+
path = '/' + self.bucket + '/' + '/'.join(urllib.parse.quote(part, safe='') for part in key.split('/'))
|
|
271
|
+
now = datetime.datetime.now(datetime.timezone.utc)
|
|
272
|
+
amz_date = now.strftime('%Y%m%dT%H%M%SZ')
|
|
273
|
+
stamp = now.strftime('%Y%m%d')
|
|
274
|
+
payload_hash = hashlib.sha256(body).hexdigest()
|
|
275
|
+
headers = {
|
|
276
|
+
'content-type': content_type or 'application/octet-stream',
|
|
277
|
+
'host': host,
|
|
278
|
+
'x-amz-content-sha256': payload_hash,
|
|
279
|
+
'x-amz-date': amz_date,
|
|
280
|
+
}
|
|
281
|
+
signed = ';'.join(sorted(headers))
|
|
282
|
+
canonical = '\n'.join(['PUT', path, '', *(f'{k}:{headers[k]}' for k in sorted(headers)), '', signed, payload_hash])
|
|
283
|
+
scope = f'{stamp}/{self.region}/s3/aws4_request'
|
|
284
|
+
to_sign = '\n'.join(['AWS4-HMAC-SHA256', amz_date, scope, hashlib.sha256(canonical.encode()).hexdigest()])
|
|
285
|
+
key_bytes = ('AWS4' + self.secret_key).encode()
|
|
286
|
+
for part in (stamp, self.region, 's3', 'aws4_request'):
|
|
287
|
+
key_bytes = hmac.new(key_bytes, part.encode(), hashlib.sha256).digest()
|
|
288
|
+
signature = hmac.new(key_bytes, to_sign.encode(), hashlib.sha256).hexdigest()
|
|
289
|
+
headers['authorization'] = (f'AWS4-HMAC-SHA256 Credential={self.access_key}/{scope}, '
|
|
290
|
+
f'SignedHeaders={signed}, Signature={signature}')
|
|
291
|
+
del headers['host']
|
|
292
|
+
request = urllib.request.Request(self.endpoint + path, data=body, headers=headers, method='PUT')
|
|
293
|
+
with urllib.request.urlopen(request, timeout=120):
|
|
294
|
+
return key
|
|
295
|
+
|
|
296
|
+
|
|
297
|
+
S3 = S3Store()
|
|
298
|
+
|
|
299
|
+
|
|
300
|
+
def upload_object(message_id, attachment):
|
|
301
|
+
"""Store to the S3-API bucket and return the object key the app will presign."""
|
|
302
|
+
safe = re.sub(r'[^\w.\- ]+', '_', attachment['filename'] or 'attachment')[:120]
|
|
303
|
+
key = f'mail/{message_id}/{secrets.token_hex(6)}-{safe}'
|
|
304
|
+
return S3.put(key, attachment['bytes'], attachment.get('contentType'))
|
|
305
|
+
|
|
306
|
+
|
|
307
|
+
def upload_blob(token, message_id, attachment):
|
|
308
|
+
safe = re.sub(r'[^\w.\- ]+', '_', attachment['filename'] or 'attachment')
|
|
309
|
+
request = urllib.request.Request(
|
|
310
|
+
f'https://blob.vercel-storage.com/mail/{message_id}/{urllib.parse.quote(safe)}',
|
|
311
|
+
data=attachment['bytes'],
|
|
312
|
+
headers={'authorization': f'Bearer {token}',
|
|
313
|
+
'x-content-type': attachment.get('contentType') or 'application/octet-stream',
|
|
314
|
+
'x-add-random-suffix': '1', 'x-api-version': '7'},
|
|
315
|
+
method='PUT',
|
|
316
|
+
)
|
|
317
|
+
with urllib.request.urlopen(request, timeout=120) as response:
|
|
318
|
+
return json.load(response)['url']
|
|
319
|
+
|
|
320
|
+
|
|
321
|
+
def store_attachment(token, message_id, attachment):
|
|
322
|
+
"""Upload one attachment; returns the fields to merge into its stored entry."""
|
|
323
|
+
UPLOAD_BUDGET.reserve(attachment['size'])
|
|
324
|
+
if STORE == 's3':
|
|
325
|
+
key = with_retry(lambda: upload_object(message_id, attachment), attachment['filename'][:30])
|
|
326
|
+
return {'key': key}
|
|
327
|
+
link = with_retry(lambda: upload_blob(token, message_id, attachment), attachment['filename'][:30])
|
|
328
|
+
return {'url': link}
|
|
329
|
+
|
|
330
|
+
|
|
331
|
+
STORE = 'blob'
|
|
332
|
+
|
|
333
|
+
|
|
334
|
+
def stored_attachments(url, token, ids):
|
|
335
|
+
"""id -> (owner, attachments) for inbox rows already stored."""
|
|
336
|
+
if not ids:
|
|
337
|
+
return {}
|
|
338
|
+
marks = ','.join('?' * len(ids))
|
|
339
|
+
results = execute(url, token, [
|
|
340
|
+
{'sql': f'SELECT id, owner, attachments FROM mail_inbox WHERE id IN ({marks})', 'args': [arg(i) for i in ids]},
|
|
341
|
+
])
|
|
342
|
+
found = {}
|
|
343
|
+
for row in results[0]['response']['result']['rows']:
|
|
344
|
+
try:
|
|
345
|
+
entries = json.loads(row[2]['value'] or '[]')
|
|
346
|
+
except Exception:
|
|
347
|
+
entries = []
|
|
348
|
+
found[row[0]['value']] = ((row[1]['value'] or '').lower(), entries)
|
|
349
|
+
return found
|
|
350
|
+
|
|
351
|
+
|
|
352
|
+
def arg(value, kind='text'):
|
|
353
|
+
if value is None:
|
|
354
|
+
return {'type': 'null', 'value': None}
|
|
355
|
+
return {'type': kind, 'value': str(value)}
|
|
356
|
+
|
|
357
|
+
|
|
358
|
+
def execute(url, token, statements):
|
|
359
|
+
return with_retry(lambda: execute_once(url, token, statements), 'turso')
|
|
360
|
+
|
|
361
|
+
|
|
362
|
+
def execute_once(url, token, statements):
|
|
363
|
+
requests = [{'type': 'execute', 'stmt': stmt} for stmt in statements] + [{'type': 'close'}]
|
|
364
|
+
request = urllib.request.Request(
|
|
365
|
+
url.replace('libsql:', 'https:').rstrip('/') + '/v2/pipeline',
|
|
366
|
+
data=json.dumps({'requests': requests}).encode(),
|
|
367
|
+
headers={'Authorization': f'Bearer {token}', 'Content-Type': 'application/json'}, method='POST',
|
|
368
|
+
)
|
|
369
|
+
with urllib.request.urlopen(request, timeout=120) as response:
|
|
370
|
+
payload = json.load(response)
|
|
371
|
+
results = payload.get('results', [])
|
|
372
|
+
for index, result in enumerate(results):
|
|
373
|
+
if result.get('type') == 'error' or 'response' not in result:
|
|
374
|
+
raise RuntimeError(f'statement {index} failed: {json.dumps(result)[:400]}')
|
|
375
|
+
return results
|
|
376
|
+
|
|
377
|
+
|
|
378
|
+
def existing_owners(url, token, ids):
|
|
379
|
+
"""id -> owner for rows already stored, in either table."""
|
|
380
|
+
if not ids:
|
|
381
|
+
return {}
|
|
382
|
+
marks = ','.join('?' * len(ids))
|
|
383
|
+
results = execute(url, token, [
|
|
384
|
+
{'sql': f'SELECT id, owner FROM mail_inbox WHERE id IN ({marks})', 'args': [arg(i) for i in ids]},
|
|
385
|
+
{'sql': f'SELECT s.id, m.owner FROM mail_sent s LEFT JOIN mail_sent_meta m ON m.email_id = s.id '
|
|
386
|
+
f'WHERE s.id IN ({marks})', 'args': [arg(i) for i in ids]},
|
|
387
|
+
])
|
|
388
|
+
found = {}
|
|
389
|
+
for result in results[:2]:
|
|
390
|
+
for row in result['response']['result']['rows']:
|
|
391
|
+
found[row[0]['value']] = (row[1]['value'] or '').lower()
|
|
392
|
+
return found
|
|
393
|
+
|
|
394
|
+
|
|
395
|
+
def inbox_statement(row):
|
|
396
|
+
return {'sql': INSERT_INBOX, 'args': [
|
|
397
|
+
arg(row['id']), arg(row['from']), arg(json.dumps(row['to'])), arg(json.dumps(row['cc'])),
|
|
398
|
+
arg(json.dumps(row['bcc'])), arg(json.dumps(row['replyTo'])), arg(row['subject']),
|
|
399
|
+
arg(row['html']), arg(row['text']), arg(json.dumps(row['headers'])), arg(row['receivedAt']),
|
|
400
|
+
arg(1 if row['read'] else 0, 'integer'), arg(json.dumps(row['attachments'])), arg(row['owner']),
|
|
401
|
+
arg(1 if row['starred'] else 0, 'integer'), arg(1 if row['archived'] else 0, 'integer'),
|
|
402
|
+
arg(1 if row['trashed'] else 0, 'integer'), arg(json.dumps(row['labels'])),
|
|
403
|
+
]}
|
|
404
|
+
|
|
405
|
+
|
|
406
|
+
def sent_statements(row):
|
|
407
|
+
return [
|
|
408
|
+
{'sql': INSERT_SENT, 'args': [
|
|
409
|
+
arg(row['id']), arg(row['from']), arg(json.dumps(row['to'])), arg(json.dumps(row['cc'])),
|
|
410
|
+
arg(json.dumps(row['bcc'])), arg(json.dumps(row['replyTo'])), arg(row['subject']),
|
|
411
|
+
arg(row['html']), arg(row['text']), arg(row['receivedAt']), arg('imported'), arg('mbox'),
|
|
412
|
+
]},
|
|
413
|
+
{'sql': INSERT_SENT_META, 'args': [arg(row['id']), arg(row['owner']), arg(row['receivedAt'])]},
|
|
414
|
+
]
|
|
415
|
+
|
|
416
|
+
|
|
417
|
+
def main():
|
|
418
|
+
parser = argparse.ArgumentParser()
|
|
419
|
+
parser.add_argument('mbox')
|
|
420
|
+
parser.add_argument('--owner', required=True)
|
|
421
|
+
parser.add_argument('--dry-run', action='store_true')
|
|
422
|
+
parser.add_argument('--limit', type=int, help='stop after this many messages')
|
|
423
|
+
parser.add_argument('--start', type=int, default=0, help='skip this many messages first (resume)')
|
|
424
|
+
parser.add_argument('--batch', type=int, default=10)
|
|
425
|
+
parser.add_argument('--mark-read', action='store_true')
|
|
426
|
+
parser.add_argument('--skip-attachments', action='store_true')
|
|
427
|
+
parser.add_argument('--attach-only', action='store_true',
|
|
428
|
+
help='second pass: upload attachments for rows already imported and write the URLs back')
|
|
429
|
+
parser.add_argument('--workers', type=int, default=8, help='parallel attachment uploads')
|
|
430
|
+
parser.add_argument('--writers', type=int, default=6, help='database batches in flight at once')
|
|
431
|
+
parser.add_argument('--max-mbps', type=float, help='cap aggregate upload bandwidth, e.g. 4.5')
|
|
432
|
+
parser.add_argument('--chunks', type=int, default=4, help='attachment chunks in flight at once')
|
|
433
|
+
parser.add_argument('--replace-blob', action='store_true',
|
|
434
|
+
help='treat entries hosted on Vercel Blob as missing so they are re-uploaded to the bucket')
|
|
435
|
+
parser.add_argument('--store', choices=['blob', 's3'], default='s3' if S3.configured() else 'blob',
|
|
436
|
+
help='where attachment bytes go: Vercel Blob, or an S3-API bucket (R2/S3)')
|
|
437
|
+
args = parser.parse_args()
|
|
438
|
+
if args.attach_only and args.batch == 10:
|
|
439
|
+
args.batch = 5
|
|
440
|
+
|
|
441
|
+
global STORE
|
|
442
|
+
STORE = args.store
|
|
443
|
+
owner = args.owner.lower()
|
|
444
|
+
if STORE == 's3' and not S3.configured() and not args.dry_run and not args.skip_attachments:
|
|
445
|
+
sys.exit('--store s3 needs S3_/R2_ endpoint, bucket and keys in the environment')
|
|
446
|
+
if STORE == 's3':
|
|
447
|
+
print(f'attachment store: s3 bucket {S3.bucket} at {S3.endpoint} (region {S3.region})', flush=True)
|
|
448
|
+
if args.max_mbps:
|
|
449
|
+
UPLOAD_BUDGET.__init__(args.max_mbps)
|
|
450
|
+
print(f'upload cap: {args.max_mbps} Mbit/s', flush=True)
|
|
451
|
+
url, db_token = os.environ.get('TURSO_DATABASE_URL'), os.environ.get('TURSO_AUTH_TOKEN')
|
|
452
|
+
blob_token = None if args.skip_attachments else (os.environ.get('BLOB_READ_WRITE_TOKEN') if STORE == 'blob' else 's3')
|
|
453
|
+
if not args.dry_run and not (url and db_token):
|
|
454
|
+
sys.exit('TURSO_DATABASE_URL and TURSO_AUTH_TOKEN must be set')
|
|
455
|
+
if not args.dry_run and not args.skip_attachments and not blob_token:
|
|
456
|
+
print('BLOB_READ_WRITE_TOKEN unset: attachments will be metadata only', flush=True)
|
|
457
|
+
|
|
458
|
+
stats = {'seen': 0, 'inbox': 0, 'sent': 0, 'skipped': 0, 'existing': 0, 'undated': 0, 'missing': 0,
|
|
459
|
+
'updated': 0, 'files': 0, 'file_bytes': 0, 'files_skipped': 0, 'file_errors': 0}
|
|
460
|
+
chunk = []
|
|
461
|
+
pool = concurrent.futures.ThreadPoolExecutor(max_workers=args.workers) if args.attach_only else None
|
|
462
|
+
|
|
463
|
+
def process_attach_chunk(rows):
|
|
464
|
+
local = {'updated': 0, 'existing': 0, 'missing': 0, 'files': 0, 'file_bytes': 0,
|
|
465
|
+
'files_skipped': 0, 'file_errors': 0}
|
|
466
|
+
if args.dry_run:
|
|
467
|
+
for row in rows:
|
|
468
|
+
local['files'] += len(row['attachments'])
|
|
469
|
+
local['file_bytes'] += sum(a['size'] for a in row['attachments'])
|
|
470
|
+
return local
|
|
471
|
+
stored = stored_attachments(url, db_token, [i for row in rows for i in (row['primary_id'], row['alt_id'])])
|
|
472
|
+
jobs = []
|
|
473
|
+
for row in rows:
|
|
474
|
+
mine = next((i for i in (row['primary_id'], row['alt_id'])
|
|
475
|
+
if i in stored and stored[i][0] == row['owner']), None)
|
|
476
|
+
if mine is None:
|
|
477
|
+
local['missing'] += 1
|
|
478
|
+
continue
|
|
479
|
+
row['id'] = mine
|
|
480
|
+
entries = stored[mine][1]
|
|
481
|
+
want = len(row['attachments'])
|
|
482
|
+
complete = lambda e: bool(e.get('key')) or (bool(e.get('url')) and not args.replace_blob)
|
|
483
|
+
# Completeness is tested before reuse. The other order rewrote every shared copy
|
|
484
|
+
# whose primary was keyed on every pass — even ones already finished — so a
|
|
485
|
+
# mailbox of shared copies burned whole slices without ever reaching what was
|
|
486
|
+
# actually missing. Compare >= and slice: a stored row can hold more entries
|
|
487
|
+
# than the message parses out, and such a row yields no work either way.
|
|
488
|
+
if len(entries) >= want and all(complete(e) for e in entries[:want]):
|
|
489
|
+
local['existing'] += 1
|
|
490
|
+
continue
|
|
491
|
+
if mine == row['alt_id'] and row['primary_id'] in stored:
|
|
492
|
+
primary_entries = stored[row['primary_id']][1]
|
|
493
|
+
if len(primary_entries) >= want and all(e.get('key') for e in primary_entries[:want]):
|
|
494
|
+
row['stored'] = [dict(e) for e in primary_entries[:want]]
|
|
495
|
+
row['reuse'] = True
|
|
496
|
+
local['reused'] = local.get('reused', 0) + 1
|
|
497
|
+
continue
|
|
498
|
+
row['stored'] = entries
|
|
499
|
+
for index, attachment in enumerate(row['attachments']):
|
|
500
|
+
if index < len(entries) and complete(entries[index]):
|
|
501
|
+
continue
|
|
502
|
+
if attachment['size'] > BLOB_LIMIT_BYTES:
|
|
503
|
+
local['files_skipped'] += 1
|
|
504
|
+
continue
|
|
505
|
+
jobs.append((row, index, attachment))
|
|
506
|
+
futures = {pool.submit(store_attachment, blob_token, row['id'], attachment): (row, index, attachment)
|
|
507
|
+
for row, index, attachment in jobs}
|
|
508
|
+
touched = set()
|
|
509
|
+
for future in concurrent.futures.as_completed(futures):
|
|
510
|
+
row, index, attachment = futures[future]
|
|
511
|
+
try:
|
|
512
|
+
stored_fields = future.result()
|
|
513
|
+
except Exception as err:
|
|
514
|
+
local['file_errors'] += 1
|
|
515
|
+
print(f' ! {attachment["filename"]}: {str(err)[:100]}', flush=True)
|
|
516
|
+
continue
|
|
517
|
+
entries = row['stored']
|
|
518
|
+
while len(entries) <= index:
|
|
519
|
+
entries.append({})
|
|
520
|
+
entries[index] = {'filename': attachment['filename'], 'contentType': attachment['contentType'],
|
|
521
|
+
'size': attachment['size'], **stored_fields}
|
|
522
|
+
local['files'] += 1
|
|
523
|
+
local['file_bytes'] += attachment['size']
|
|
524
|
+
touched.add(row['id'])
|
|
525
|
+
for row in rows:
|
|
526
|
+
if row['id'] in touched or row.get('reuse'):
|
|
527
|
+
execute(url, db_token, [{'sql': 'UPDATE mail_inbox SET attachments = ? WHERE id = ?',
|
|
528
|
+
'args': [arg(json.dumps(row['stored'])), arg(row['id'])]}])
|
|
529
|
+
local['updated'] += 1
|
|
530
|
+
for attachment in row['attachments']:
|
|
531
|
+
attachment.pop('bytes', None)
|
|
532
|
+
return local
|
|
533
|
+
|
|
534
|
+
chunk_pool = concurrent.futures.ThreadPoolExecutor(max_workers=args.chunks) if args.attach_only else None
|
|
535
|
+
attach_in_flight = []
|
|
536
|
+
|
|
537
|
+
def collect_attach(done_only=True):
|
|
538
|
+
remaining = []
|
|
539
|
+
for future in attach_in_flight:
|
|
540
|
+
if done_only and not future.done():
|
|
541
|
+
remaining.append(future)
|
|
542
|
+
continue
|
|
543
|
+
try:
|
|
544
|
+
counters = future.result()
|
|
545
|
+
except Exception as err:
|
|
546
|
+
stats['batch_errors'] = stats.get('batch_errors', 0) + 1
|
|
547
|
+
print(f' ! chunk lost: {str(err)[:140]}', flush=True)
|
|
548
|
+
continue
|
|
549
|
+
for key, value in counters.items():
|
|
550
|
+
stats[key] = stats.get(key, 0) + value
|
|
551
|
+
attach_in_flight[:] = remaining
|
|
552
|
+
|
|
553
|
+
def flush_attach():
|
|
554
|
+
if not chunk:
|
|
555
|
+
return
|
|
556
|
+
rows = chunk[:]
|
|
557
|
+
chunk.clear()
|
|
558
|
+
while len(attach_in_flight) >= args.chunks * 2:
|
|
559
|
+
concurrent.futures.wait(attach_in_flight, return_when=concurrent.futures.FIRST_COMPLETED)
|
|
560
|
+
collect_attach()
|
|
561
|
+
attach_in_flight.append(chunk_pool.submit(process_attach_chunk, rows))
|
|
562
|
+
collect_attach()
|
|
563
|
+
|
|
564
|
+
def write_rows(rows):
|
|
565
|
+
local = {'inbox': 0, 'sent': 0, 'existing': 0, 'files': 0, 'file_bytes': 0,
|
|
566
|
+
'files_skipped': 0, 'file_errors': 0, 'row_errors': 0}
|
|
567
|
+
if args.dry_run:
|
|
568
|
+
for row in rows:
|
|
569
|
+
local[row['kind']] += 1
|
|
570
|
+
local['files'] += len(row['attachments'])
|
|
571
|
+
local['file_bytes'] += sum(a['size'] for a in row['attachments'])
|
|
572
|
+
return local
|
|
573
|
+
owners = existing_owners(url, db_token, [i for row in rows for i in (row['primary_id'], row['alt_id'])])
|
|
574
|
+
statements = []
|
|
575
|
+
for row in rows:
|
|
576
|
+
if row['alt_id'] in owners or owners.get(row['primary_id']) == row['owner']:
|
|
577
|
+
local['existing'] += 1
|
|
578
|
+
continue
|
|
579
|
+
if row['primary_id'] in owners:
|
|
580
|
+
row['id'] = row['alt_id']
|
|
581
|
+
row['headers']['shared-copy-of'] = row['primary_id']
|
|
582
|
+
local['shared'] = local.get('shared', 0) + 1
|
|
583
|
+
for attachment in row['attachments']:
|
|
584
|
+
if blob_token and attachment['size'] <= BLOB_LIMIT_BYTES:
|
|
585
|
+
try:
|
|
586
|
+
attachment.update(store_attachment(blob_token, row['id'], attachment))
|
|
587
|
+
local['files'] += 1
|
|
588
|
+
local['file_bytes'] += attachment['size']
|
|
589
|
+
except Exception as err:
|
|
590
|
+
local['file_errors'] += 1
|
|
591
|
+
print(f' ! {attachment["filename"]}: {err}', flush=True)
|
|
592
|
+
elif blob_token:
|
|
593
|
+
local['files_skipped'] += 1
|
|
594
|
+
attachment.pop('bytes', None)
|
|
595
|
+
if row['kind'] == 'sent':
|
|
596
|
+
statements += sent_statements(row)
|
|
597
|
+
else:
|
|
598
|
+
statements.append(inbox_statement(row))
|
|
599
|
+
local[row['kind']] += 1
|
|
600
|
+
if not statements:
|
|
601
|
+
return local
|
|
602
|
+
try:
|
|
603
|
+
execute(url, db_token, statements)
|
|
604
|
+
except Exception as err:
|
|
605
|
+
print(f' batch failed ({str(err)[:120]}); retrying rows one at a time', flush=True)
|
|
606
|
+
for statement in statements:
|
|
607
|
+
try:
|
|
608
|
+
execute(url, db_token, [statement])
|
|
609
|
+
except Exception as inner:
|
|
610
|
+
local['row_errors'] += 1
|
|
611
|
+
print(f' ! row failed: {str(inner)[:160]}', flush=True)
|
|
612
|
+
return local
|
|
613
|
+
|
|
614
|
+
writers = concurrent.futures.ThreadPoolExecutor(max_workers=args.writers)
|
|
615
|
+
in_flight = []
|
|
616
|
+
|
|
617
|
+
def collect(done_only=True):
|
|
618
|
+
remaining = []
|
|
619
|
+
for future in in_flight:
|
|
620
|
+
if done_only and not future.done():
|
|
621
|
+
remaining.append(future)
|
|
622
|
+
continue
|
|
623
|
+
try:
|
|
624
|
+
counters = future.result()
|
|
625
|
+
except Exception as err:
|
|
626
|
+
stats['batch_errors'] = stats.get('batch_errors', 0) + 1
|
|
627
|
+
print(f' ! batch lost: {str(err)[:140]}', flush=True)
|
|
628
|
+
continue
|
|
629
|
+
for key, value in counters.items():
|
|
630
|
+
stats[key] = stats.get(key, 0) + value
|
|
631
|
+
in_flight[:] = remaining
|
|
632
|
+
|
|
633
|
+
def flush():
|
|
634
|
+
if not chunk:
|
|
635
|
+
return
|
|
636
|
+
rows = chunk[:]
|
|
637
|
+
chunk.clear()
|
|
638
|
+
while len(in_flight) >= args.writers * 2:
|
|
639
|
+
concurrent.futures.wait(in_flight, return_when=concurrent.futures.FIRST_COMPLETED)
|
|
640
|
+
collect()
|
|
641
|
+
in_flight.append(writers.submit(write_rows, rows))
|
|
642
|
+
collect()
|
|
643
|
+
|
|
644
|
+
for index, raw in enumerate(messages(args.mbox)):
|
|
645
|
+
if index < args.start:
|
|
646
|
+
continue
|
|
647
|
+
if args.limit and stats['seen'] >= args.limit:
|
|
648
|
+
break
|
|
649
|
+
stats['seen'] += 1
|
|
650
|
+
row = row_for(email.message_from_bytes(raw), owner, args.mark_read, keep_bytes=not args.skip_attachments)
|
|
651
|
+
if row['kind'] == 'skip':
|
|
652
|
+
stats['skipped'] += 1
|
|
653
|
+
continue
|
|
654
|
+
if not row['receivedAt']:
|
|
655
|
+
stats['undated'] += 1
|
|
656
|
+
continue
|
|
657
|
+
if stats['seen'] <= 12:
|
|
658
|
+
flag = 'read ' if row['read'] else 'UNRD '
|
|
659
|
+
print(f' {row["receivedAt"][:10]} {row["kind"]:5} {flag} {row["from"][:30]:30} {row["subject"][:44]}'
|
|
660
|
+
f' [{len(row["attachments"])} files]', flush=True)
|
|
661
|
+
if args.attach_only:
|
|
662
|
+
if row['kind'] != 'inbox' or not row['attachments']:
|
|
663
|
+
continue
|
|
664
|
+
chunk.append(row)
|
|
665
|
+
if len(chunk) >= args.batch:
|
|
666
|
+
flush_attach()
|
|
667
|
+
if stats['seen'] % 500 == 0:
|
|
668
|
+
print(f'... {stats["seen"]} seen, {stats["updated"]} rows updated, {stats["existing"]} already done, '
|
|
669
|
+
f'{stats.get("reused", 0)} reused, {stats.get("missing", 0)} missing, '
|
|
670
|
+
f'{stats["files"]} files ({stats["file_bytes"]/1e9:.2f} GB), {stats["file_errors"]} errors', flush=True)
|
|
671
|
+
continue
|
|
672
|
+
chunk.append(row)
|
|
673
|
+
if len(chunk) >= args.batch:
|
|
674
|
+
flush()
|
|
675
|
+
if stats['seen'] % 500 == 0:
|
|
676
|
+
print(f'... {stats["seen"]} seen, {stats["inbox"]} inbox, {stats["sent"]} sent, '
|
|
677
|
+
f'{stats["existing"]} existing, {stats["files"]} files ({stats["file_bytes"]/1e9:.2f} GB)', flush=True)
|
|
678
|
+
if args.attach_only:
|
|
679
|
+
flush_attach()
|
|
680
|
+
concurrent.futures.wait(attach_in_flight)
|
|
681
|
+
collect_attach(done_only=False)
|
|
682
|
+
chunk_pool.shutdown(wait=True)
|
|
683
|
+
pool.shutdown(wait=True)
|
|
684
|
+
else:
|
|
685
|
+
flush()
|
|
686
|
+
concurrent.futures.wait(in_flight)
|
|
687
|
+
collect(done_only=False)
|
|
688
|
+
writers.shutdown(wait=True)
|
|
689
|
+
|
|
690
|
+
mode = 'dry run — nothing written' if args.dry_run else ('attachments backfilled' if args.attach_only else 'imported')
|
|
691
|
+
print(f'\n{mode} for {owner}')
|
|
692
|
+
if args.attach_only:
|
|
693
|
+
print(f' seen {stats["seen"]} rows updated {stats["updated"]} already complete {stats["existing"]} '
|
|
694
|
+
f'not in db {stats["missing"]} keys reused from shared copy {stats.get("reused", 0)}')
|
|
695
|
+
else:
|
|
696
|
+
print(f' seen {stats["seen"]} inbox {stats["inbox"]} sent {stats["sent"]} spam/drafts skipped {stats["skipped"]}'
|
|
697
|
+
f' already present {stats["existing"]} shared copies {stats.get("shared", 0)} undated {stats["undated"]}')
|
|
698
|
+
print(f' attachments {stats["files"]} ({stats["file_bytes"]/1e9:.2f} GB)'
|
|
699
|
+
f' over 20MB skipped {stats["files_skipped"]} upload errors {stats["file_errors"]} row errors {stats.get("row_errors", 0)} batches lost {stats.get("batch_errors", 0)}')
|
|
700
|
+
|
|
701
|
+
|
|
702
|
+
if __name__ == '__main__':
|
|
703
|
+
main()
|