@novacraft-engineering/mailbox 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (114) hide show
  1. package/.env.example +110 -0
  2. package/LICENSE +21 -0
  3. package/README.md +75 -0
  4. package/app/api/mail/accessors/route.ts +197 -0
  5. package/app/api/mail/attachments/route.ts +54 -0
  6. package/app/api/mail/company-signature/route.ts +36 -0
  7. package/app/api/mail/contacts/route.ts +13 -0
  8. package/app/api/mail/emails/[id]/attachments/download/route.ts +69 -0
  9. package/app/api/mail/emails/[id]/attachments/forward/route.ts +65 -0
  10. package/app/api/mail/emails/[id]/route.ts +127 -0
  11. package/app/api/mail/emails/route.ts +116 -0
  12. package/app/api/mail/events/route.ts +12 -0
  13. package/app/api/mail/fonts/route.ts +44 -0
  14. package/app/api/mail/inbox/attachments/download/route.ts +50 -0
  15. package/app/api/mail/inbox/attachments/forward/route.ts +54 -0
  16. package/app/api/mail/inbox/attachments/route.ts +99 -0
  17. package/app/api/mail/inbox/body/route.ts +14 -0
  18. package/app/api/mail/inbox/counts/route.ts +25 -0
  19. package/app/api/mail/inbox/route.ts +506 -0
  20. package/app/api/mail/link-check/route.ts +95 -0
  21. package/app/api/mail/login/route.ts +38 -0
  22. package/app/api/mail/logout/route.ts +9 -0
  23. package/app/api/mail/maintenance/attachments/route.ts +26 -0
  24. package/app/api/mail/maintenance/bodies/route.ts +30 -0
  25. package/app/api/mail/maintenance/list-columns/route.ts +25 -0
  26. package/app/api/mail/maintenance/threads/route.ts +32 -0
  27. package/app/api/mail/me/route.ts +13 -0
  28. package/app/api/mail/outgoing-upload/route.ts +37 -0
  29. package/app/api/mail/password/route.ts +64 -0
  30. package/app/api/mail/pixel/[id]/route.ts +20 -0
  31. package/app/api/mail/push/route.ts +42 -0
  32. package/app/api/mail/render-template/route.ts +58 -0
  33. package/app/api/mail/request-reset/route.ts +65 -0
  34. package/app/api/mail/reset/route.ts +46 -0
  35. package/app/api/mail/send/route.ts +205 -0
  36. package/app/api/mail/settings/route.ts +35 -0
  37. package/app/api/mail/share/attachment/route.ts +91 -0
  38. package/app/api/mail/share/route.ts +130 -0
  39. package/app/api/mail/signature-logo/route.ts +67 -0
  40. package/app/api/mail/stash/route.ts +55 -0
  41. package/app/api/mail/threads/route.ts +18 -0
  42. package/app/api/mail/upload/route.ts +43 -0
  43. package/app/api/share/[id]/route.ts +85 -0
  44. package/app/apple-icon.png +0 -0
  45. package/app/brand/[file]/route.ts +37 -0
  46. package/app/globals.css +14 -0
  47. package/app/icon.png +0 -0
  48. package/app/layout.tsx +30 -0
  49. package/app/mail/AccessCheck.tsx +45 -0
  50. package/app/mail/AttachmentLightbox.tsx +241 -0
  51. package/app/mail/ConfirmDialog.tsx +88 -0
  52. package/app/mail/MailSelect.tsx +138 -0
  53. package/app/mail/RichEditor.tsx +859 -0
  54. package/app/mail/layout.tsx +20 -0
  55. package/app/mail/page.module.css +6320 -0
  56. package/app/mail/page.tsx +7971 -0
  57. package/app/mail/pwa.ts +191 -0
  58. package/app/mail/reset/page.tsx +132 -0
  59. package/app/mail/search.ts +172 -0
  60. package/app/manifest.ts +21 -0
  61. package/app/page.tsx +5 -0
  62. package/app/robots.ts +22 -0
  63. package/app/share/[id]/page.tsx +166 -0
  64. package/app/share/[id]/share.module.css +148 -0
  65. package/eslint.config.mjs +9 -0
  66. package/lib/accent-ramp.ts +54 -0
  67. package/lib/attachments.ts +52 -0
  68. package/lib/brand.client.ts +52 -0
  69. package/lib/brand.ts +87 -0
  70. package/lib/d1.ts +64 -0
  71. package/lib/default-signature.ts +68 -0
  72. package/lib/dev-auth.ts +222 -0
  73. package/lib/email-html.test.ts +74 -0
  74. package/lib/email-html.ts +169 -0
  75. package/lib/emails/index.ts +281 -0
  76. package/lib/emails/templates/academy-followup.html +68 -0
  77. package/lib/emails/templates/auto-reply.html +33 -0
  78. package/lib/emails/templates/contact-followup.html +58 -0
  79. package/lib/emails/templates/field-row.html +4 -0
  80. package/lib/emails/templates/notification.html +23 -0
  81. package/lib/fonts.ts +23 -0
  82. package/lib/link-safety.ts +81 -0
  83. package/lib/mail-provider.ts +183 -0
  84. package/lib/mailbox.ts +1888 -0
  85. package/lib/password.ts +71 -0
  86. package/lib/public-url.ts +14 -0
  87. package/lib/push.ts +37 -0
  88. package/lib/r2.ts +157 -0
  89. package/lib/rate-limit.ts +43 -0
  90. package/lib/session.test.ts +43 -0
  91. package/lib/session.ts +80 -0
  92. package/lib/signature.ts +28 -0
  93. package/lib/threads.test.ts +96 -0
  94. package/lib/threads.ts +72 -0
  95. package/lib/turso.ts +53 -0
  96. package/next.config.ts +51 -0
  97. package/package.json +69 -0
  98. package/public/brand/README.md +35 -0
  99. package/public/icon-192.png +0 -0
  100. package/public/icon-512.png +0 -0
  101. package/public/icon-maskable-512.png +0 -0
  102. package/public/sw.js +41 -0
  103. package/scripts/backfill-attachments.mjs +96 -0
  104. package/scripts/backfill-content-ids.mjs +101 -0
  105. package/scripts/backfill-list-columns.mjs +80 -0
  106. package/scripts/offload-bodies.mjs +152 -0
  107. package/scripts/sample-files.mjs +119 -0
  108. package/scripts/seed-dev.mjs +240 -0
  109. package/tools/README.md +18 -0
  110. package/tools/db-backup.py +158 -0
  111. package/tools/doh.py +46 -0
  112. package/tools/import-mbox.py +703 -0
  113. package/tools/import-takeout.sh +49 -0
  114. package/tsconfig.json +35 -0
@@ -0,0 +1,703 @@
1
+ #!/usr/bin/env python3
2
+ """Stream an mbox archive into the mailbox tables.
3
+
4
+ Runs in constant memory: messages are split off the file one at a time and written in
5
+ small chunks. Row ids derive from Message-ID, and every chunk checks which ids already
6
+ exist before uploading anything, so a re-run — or a crash and restart — never duplicates
7
+ a row or a blob.
8
+
9
+ Gmail labels drive routing: Spam and Drafts are skipped, Sent goes to the sent folder,
10
+ Trash and Starred become flags, and anything without an Inbox label is archived.
11
+ """
12
+
13
+ import argparse
14
+ import concurrent.futures
15
+ import email
16
+ import email.utils
17
+ import datetime
18
+ import hashlib
19
+ import hmac
20
+ import json
21
+ import secrets
22
+ import os
23
+ import re
24
+ import socket
25
+ import sys
26
+ import threading
27
+ import time
28
+ import urllib.error
29
+ import urllib.parse
30
+ import urllib.request
31
+ from datetime import timezone
32
+ from email.header import decode_header, make_header
33
+
34
+ BLOB_LIMIT_BYTES = 20 * 1024 * 1024
35
+
36
+ # A home router whose resolver flaps takes every request down with it while raw IP
37
+ # connectivity is fine. When the system resolver fails, ask 1.1.1.1 by address over
38
+ # HTTPS instead — scoped to this process, so nothing on the machine is changed.
39
+ _system_getaddrinfo = socket.getaddrinfo
40
+ _doh_cache = {}
41
+
42
+
43
+ def _doh_lookup(host):
44
+ request = urllib.request.Request(
45
+ f'https://1.1.1.1/dns-query?name={urllib.parse.quote(host)}&type=A',
46
+ headers={'accept': 'application/dns-json'},
47
+ )
48
+ with urllib.request.urlopen(request, timeout=10) as response:
49
+ answers = json.load(response).get('Answer') or []
50
+ addresses = [entry['data'] for entry in answers if entry.get('type') == 1]
51
+ if not addresses:
52
+ raise socket.gaierror(f'no A record for {host} via DoH')
53
+ return addresses
54
+
55
+
56
+ def _resilient_getaddrinfo(host, port, family=0, kind=0, proto=0, flags=0):
57
+ try:
58
+ return _system_getaddrinfo(host, port, family, kind, proto, flags)
59
+ except socket.gaierror:
60
+ if host in ('1.1.1.1',) or not isinstance(host, str):
61
+ raise
62
+ if host not in _doh_cache:
63
+ _doh_cache[host] = _doh_lookup(host)
64
+ print(f' ~ system DNS failed for {host}; using DoH answer {_doh_cache[host][0]}', flush=True)
65
+ return [(socket.AF_INET, socket.SOCK_STREAM, socket.IPPROTO_TCP, '', (address, port))
66
+ for address in _doh_cache[host]]
67
+
68
+
69
+ socket.getaddrinfo = _resilient_getaddrinfo
70
+ BODY_CAP = 4_000_000
71
+ INBOX_COLUMNS = ('id, from_addr, to_addrs, cc, bcc, reply_to, subject, html, body_text, headers, '
72
+ 'received_at, read, attachments, owner, starred, archived, trashed, labels')
73
+ INSERT_INBOX = (f'INSERT INTO mail_inbox ({INBOX_COLUMNS}) VALUES ({",".join("?" * 18)}) '
74
+ 'ON CONFLICT (id) DO NOTHING')
75
+ INSERT_SENT = ('INSERT INTO mail_sent (id, from_addr, to_addrs, cc, bcc, reply_to, subject, html, body_text, '
76
+ 'created_at, last_event, provider) VALUES (?,?,?,?,?,?,?,?,?,?,?,?) ON CONFLICT (id) DO NOTHING')
77
+ INSERT_SENT_META = ('INSERT INTO mail_sent_meta (email_id, owner, is_auto, created_at) VALUES (?,?,0,?) '
78
+ 'ON CONFLICT (email_id) DO NOTHING')
79
+
80
+
81
+ def messages(path):
82
+ """Yield raw message bytes. Takeout escapes body 'From ' lines, so the bare form is a boundary.
83
+ A path of '-' reads the mbox from stdin, so a zip can be streamed in without extracting it."""
84
+ buffer = bytearray()
85
+ handle = sys.stdin.buffer if path == '-' else open(path, 'rb')
86
+ with handle:
87
+ for line in handle:
88
+ if line.startswith(b'From '):
89
+ if buffer:
90
+ yield bytes(buffer)
91
+ buffer = bytearray()
92
+ continue
93
+ buffer += line
94
+ if buffer:
95
+ yield bytes(buffer)
96
+
97
+
98
+ def decoded(value):
99
+ if not value:
100
+ return ''
101
+ try:
102
+ return str(make_header(decode_header(str(value))))
103
+ except Exception:
104
+ return str(value)
105
+
106
+
107
+ def addresses(message, header):
108
+ found = []
109
+ for raw in message.get_all(header, []):
110
+ for _, address in email.utils.getaddresses([str(raw)]):
111
+ if address:
112
+ found.append(address.lower())
113
+ return found
114
+
115
+
116
+ def received_at(message):
117
+ raw = message.get('Date')
118
+ if not raw:
119
+ return None
120
+ try:
121
+ parsed = email.utils.parsedate_to_datetime(str(raw))
122
+ except Exception:
123
+ return None
124
+ if parsed.tzinfo is None:
125
+ parsed = parsed.replace(tzinfo=timezone.utc)
126
+ return parsed.astimezone(timezone.utc).isoformat(timespec='milliseconds').replace('+00:00', 'Z')
127
+
128
+
129
+ def bodies(message, keep_bytes=True):
130
+ html, text, attachments = None, None, []
131
+ for part in (message.walk() if message.is_multipart() else [message]):
132
+ if part.get_content_maintype() == 'multipart':
133
+ continue
134
+ filename = part.get_filename()
135
+ disposition = str(part.get('Content-Disposition') or '')
136
+ if filename or 'attachment' in disposition.lower():
137
+ raw = b''
138
+ if keep_bytes:
139
+ try:
140
+ raw = part.get_payload(decode=True) or b''
141
+ except Exception:
142
+ raw = b''
143
+ attachments.append({
144
+ 'filename': decoded(filename) or 'attachment',
145
+ 'contentType': part.get_content_type(),
146
+ 'size': len(raw),
147
+ 'bytes': raw,
148
+ })
149
+ continue
150
+ try:
151
+ payload = part.get_payload(decode=True)
152
+ except Exception:
153
+ continue
154
+ if payload is None:
155
+ continue
156
+ charset = part.get_content_charset() or 'utf-8'
157
+ try:
158
+ body = payload.decode(charset, errors='replace')
159
+ except LookupError:
160
+ body = payload.decode('utf-8', errors='replace')
161
+ if part.get_content_type() == 'text/html' and html is None:
162
+ html = body[:BODY_CAP]
163
+ elif part.get_content_type() == 'text/plain' and text is None:
164
+ text = body[:BODY_CAP]
165
+ return html, text, attachments
166
+
167
+
168
+ def row_for(message, owner, force_read, keep_bytes=True):
169
+ message_id = str(message.get('Message-ID') or '').strip()
170
+ seed = message_id or '|'.join([
171
+ str(message.get('Date') or ''), str(message.get('From') or ''), str(message.get('Subject') or ''),
172
+ ])
173
+ labels = [label.strip() for label in str(message.get('X-Gmail-Labels') or '').split(',') if label.strip()]
174
+ sender = addresses(message, 'From')
175
+ from_addr = sender[0] if sender else decoded(message.get('From'))
176
+ is_sent = 'Sent' in labels or (bool(sender) and sender[0] == owner)
177
+ html, text, attachments = bodies(message, keep_bytes)
178
+ primary_id = 'mbox-' + hashlib.sha256(seed.encode('utf-8', 'replace')).hexdigest()[:28]
179
+ return {
180
+ 'id': primary_id,
181
+ 'primary_id': primary_id,
182
+ 'alt_id': 'mbox-' + hashlib.sha256((seed + '|' + owner).encode('utf-8', 'replace')).hexdigest()[:28],
183
+ 'kind': 'skip' if ('Spam' in labels or 'Drafts' in labels) else ('sent' if is_sent else 'inbox'),
184
+ 'from': from_addr,
185
+ 'to': addresses(message, 'To') or addresses(message, 'Delivered-To'),
186
+ 'cc': addresses(message, 'Cc'),
187
+ 'bcc': addresses(message, 'Bcc'),
188
+ 'replyTo': addresses(message, 'Reply-To'),
189
+ 'subject': decoded(message.get('Subject')),
190
+ 'html': html,
191
+ 'text': text,
192
+ 'headers': {'message-id': message_id, 'x-gmail-labels': ','.join(labels),
193
+ 'date': str(message.get('Date') or ''), 'imported-from': 'mbox'},
194
+ 'receivedAt': received_at(message),
195
+ 'read': True if force_read else ('Unread' not in labels),
196
+ 'attachments': attachments,
197
+ 'owner': owner,
198
+ 'labels': labels,
199
+ 'starred': 'Starred' in labels,
200
+ 'trashed': 'Trash' in labels,
201
+ 'archived': 'Inbox' not in labels and 'Trash' not in labels and not is_sent,
202
+ }
203
+
204
+
205
+ RETRY_DELAYS = (5, 10, 20, 40, 60, 90, 120, 180, 300)
206
+
207
+
208
+ def transient(err):
209
+ """A dropped network, DNS failure, timeout, or a 429/5xx from the service — not a bad statement."""
210
+ if isinstance(err, urllib.error.HTTPError):
211
+ return err.code == 429 or err.code >= 500
212
+ return isinstance(err, (urllib.error.URLError, OSError, TimeoutError))
213
+
214
+
215
+ def with_retry(action, label):
216
+ for attempt, delay in enumerate(RETRY_DELAYS + (None,)):
217
+ try:
218
+ return action()
219
+ except Exception as err:
220
+ if delay is None or not transient(err):
221
+ raise
222
+ print(f' ~ {label}: {str(err)[:80]} — retry {attempt + 1} in {delay}s', flush=True)
223
+ time.sleep(delay)
224
+
225
+
226
+ class UploadBudget:
227
+ """Leaky-bucket schedule shared by every upload thread: each upload reserves the time
228
+ its bytes take at the cap and waits for its slot, so the aggregate never exceeds the cap
229
+ and downloads (the importer's own database replies included) are not starved."""
230
+
231
+ def __init__(self, mbps):
232
+ self.bytes_per_second = mbps * 1_000_000 / 8 if mbps else None
233
+ self.next_slot = time.monotonic()
234
+ self.lock = threading.Lock()
235
+
236
+ def reserve(self, size):
237
+ if not self.bytes_per_second:
238
+ return
239
+ with self.lock:
240
+ now = time.monotonic()
241
+ # Never let the schedule drift more than a minute ahead of the clock: failed
242
+ # attempts and clock hiccups would otherwise park every uploader asleep.
243
+ start = min(max(now, self.next_slot), now + 60)
244
+ self.next_slot = start + size / self.bytes_per_second
245
+ wait = start - now
246
+ if wait > 0:
247
+ time.sleep(wait)
248
+
249
+
250
+ UPLOAD_BUDGET = UploadBudget(None)
251
+
252
+
253
+ class S3Store:
254
+ """S3-API object store (Cloudflare R2 or AWS S3) signed with SigV4 by hand — the same
255
+ scheme lib/r2.ts uses, so the app can presign downloads for whatever this uploads."""
256
+
257
+ def __init__(self):
258
+ env = os.environ.get
259
+ self.endpoint = (env('S3_ENDPOINT') or env('R2_S3_ENDPOINT') or '').rstrip('/')
260
+ self.bucket = env('S3_BUCKET') or env('R2_BUCKET') or ''
261
+ self.region = env('S3_REGION') or env('R2_REGION') or 'auto'
262
+ self.access_key = env('AWS_ACCESS_KEY_ID') or env('R2_ACCESS_KEY_ID') or ''
263
+ self.secret_key = env('AWS_SECRET_ACCESS_KEY') or env('R2_SECRET_ACCESS_KEY') or ''
264
+
265
+ def configured(self):
266
+ return all([self.endpoint, self.bucket, self.access_key, self.secret_key])
267
+
268
+ def put(self, key, body, content_type):
269
+ host = urllib.parse.urlparse(self.endpoint).netloc
270
+ path = '/' + self.bucket + '/' + '/'.join(urllib.parse.quote(part, safe='') for part in key.split('/'))
271
+ now = datetime.datetime.now(datetime.timezone.utc)
272
+ amz_date = now.strftime('%Y%m%dT%H%M%SZ')
273
+ stamp = now.strftime('%Y%m%d')
274
+ payload_hash = hashlib.sha256(body).hexdigest()
275
+ headers = {
276
+ 'content-type': content_type or 'application/octet-stream',
277
+ 'host': host,
278
+ 'x-amz-content-sha256': payload_hash,
279
+ 'x-amz-date': amz_date,
280
+ }
281
+ signed = ';'.join(sorted(headers))
282
+ canonical = '\n'.join(['PUT', path, '', *(f'{k}:{headers[k]}' for k in sorted(headers)), '', signed, payload_hash])
283
+ scope = f'{stamp}/{self.region}/s3/aws4_request'
284
+ to_sign = '\n'.join(['AWS4-HMAC-SHA256', amz_date, scope, hashlib.sha256(canonical.encode()).hexdigest()])
285
+ key_bytes = ('AWS4' + self.secret_key).encode()
286
+ for part in (stamp, self.region, 's3', 'aws4_request'):
287
+ key_bytes = hmac.new(key_bytes, part.encode(), hashlib.sha256).digest()
288
+ signature = hmac.new(key_bytes, to_sign.encode(), hashlib.sha256).hexdigest()
289
+ headers['authorization'] = (f'AWS4-HMAC-SHA256 Credential={self.access_key}/{scope}, '
290
+ f'SignedHeaders={signed}, Signature={signature}')
291
+ del headers['host']
292
+ request = urllib.request.Request(self.endpoint + path, data=body, headers=headers, method='PUT')
293
+ with urllib.request.urlopen(request, timeout=120):
294
+ return key
295
+
296
+
297
+ S3 = S3Store()
298
+
299
+
300
+ def upload_object(message_id, attachment):
301
+ """Store to the S3-API bucket and return the object key the app will presign."""
302
+ safe = re.sub(r'[^\w.\- ]+', '_', attachment['filename'] or 'attachment')[:120]
303
+ key = f'mail/{message_id}/{secrets.token_hex(6)}-{safe}'
304
+ return S3.put(key, attachment['bytes'], attachment.get('contentType'))
305
+
306
+
307
+ def upload_blob(token, message_id, attachment):
308
+ safe = re.sub(r'[^\w.\- ]+', '_', attachment['filename'] or 'attachment')
309
+ request = urllib.request.Request(
310
+ f'https://blob.vercel-storage.com/mail/{message_id}/{urllib.parse.quote(safe)}',
311
+ data=attachment['bytes'],
312
+ headers={'authorization': f'Bearer {token}',
313
+ 'x-content-type': attachment.get('contentType') or 'application/octet-stream',
314
+ 'x-add-random-suffix': '1', 'x-api-version': '7'},
315
+ method='PUT',
316
+ )
317
+ with urllib.request.urlopen(request, timeout=120) as response:
318
+ return json.load(response)['url']
319
+
320
+
321
+ def store_attachment(token, message_id, attachment):
322
+ """Upload one attachment; returns the fields to merge into its stored entry."""
323
+ UPLOAD_BUDGET.reserve(attachment['size'])
324
+ if STORE == 's3':
325
+ key = with_retry(lambda: upload_object(message_id, attachment), attachment['filename'][:30])
326
+ return {'key': key}
327
+ link = with_retry(lambda: upload_blob(token, message_id, attachment), attachment['filename'][:30])
328
+ return {'url': link}
329
+
330
+
331
+ STORE = 'blob'
332
+
333
+
334
+ def stored_attachments(url, token, ids):
335
+ """id -> (owner, attachments) for inbox rows already stored."""
336
+ if not ids:
337
+ return {}
338
+ marks = ','.join('?' * len(ids))
339
+ results = execute(url, token, [
340
+ {'sql': f'SELECT id, owner, attachments FROM mail_inbox WHERE id IN ({marks})', 'args': [arg(i) for i in ids]},
341
+ ])
342
+ found = {}
343
+ for row in results[0]['response']['result']['rows']:
344
+ try:
345
+ entries = json.loads(row[2]['value'] or '[]')
346
+ except Exception:
347
+ entries = []
348
+ found[row[0]['value']] = ((row[1]['value'] or '').lower(), entries)
349
+ return found
350
+
351
+
352
+ def arg(value, kind='text'):
353
+ if value is None:
354
+ return {'type': 'null', 'value': None}
355
+ return {'type': kind, 'value': str(value)}
356
+
357
+
358
+ def execute(url, token, statements):
359
+ return with_retry(lambda: execute_once(url, token, statements), 'turso')
360
+
361
+
362
+ def execute_once(url, token, statements):
363
+ requests = [{'type': 'execute', 'stmt': stmt} for stmt in statements] + [{'type': 'close'}]
364
+ request = urllib.request.Request(
365
+ url.replace('libsql:', 'https:').rstrip('/') + '/v2/pipeline',
366
+ data=json.dumps({'requests': requests}).encode(),
367
+ headers={'Authorization': f'Bearer {token}', 'Content-Type': 'application/json'}, method='POST',
368
+ )
369
+ with urllib.request.urlopen(request, timeout=120) as response:
370
+ payload = json.load(response)
371
+ results = payload.get('results', [])
372
+ for index, result in enumerate(results):
373
+ if result.get('type') == 'error' or 'response' not in result:
374
+ raise RuntimeError(f'statement {index} failed: {json.dumps(result)[:400]}')
375
+ return results
376
+
377
+
378
+ def existing_owners(url, token, ids):
379
+ """id -> owner for rows already stored, in either table."""
380
+ if not ids:
381
+ return {}
382
+ marks = ','.join('?' * len(ids))
383
+ results = execute(url, token, [
384
+ {'sql': f'SELECT id, owner FROM mail_inbox WHERE id IN ({marks})', 'args': [arg(i) for i in ids]},
385
+ {'sql': f'SELECT s.id, m.owner FROM mail_sent s LEFT JOIN mail_sent_meta m ON m.email_id = s.id '
386
+ f'WHERE s.id IN ({marks})', 'args': [arg(i) for i in ids]},
387
+ ])
388
+ found = {}
389
+ for result in results[:2]:
390
+ for row in result['response']['result']['rows']:
391
+ found[row[0]['value']] = (row[1]['value'] or '').lower()
392
+ return found
393
+
394
+
395
+ def inbox_statement(row):
396
+ return {'sql': INSERT_INBOX, 'args': [
397
+ arg(row['id']), arg(row['from']), arg(json.dumps(row['to'])), arg(json.dumps(row['cc'])),
398
+ arg(json.dumps(row['bcc'])), arg(json.dumps(row['replyTo'])), arg(row['subject']),
399
+ arg(row['html']), arg(row['text']), arg(json.dumps(row['headers'])), arg(row['receivedAt']),
400
+ arg(1 if row['read'] else 0, 'integer'), arg(json.dumps(row['attachments'])), arg(row['owner']),
401
+ arg(1 if row['starred'] else 0, 'integer'), arg(1 if row['archived'] else 0, 'integer'),
402
+ arg(1 if row['trashed'] else 0, 'integer'), arg(json.dumps(row['labels'])),
403
+ ]}
404
+
405
+
406
+ def sent_statements(row):
407
+ return [
408
+ {'sql': INSERT_SENT, 'args': [
409
+ arg(row['id']), arg(row['from']), arg(json.dumps(row['to'])), arg(json.dumps(row['cc'])),
410
+ arg(json.dumps(row['bcc'])), arg(json.dumps(row['replyTo'])), arg(row['subject']),
411
+ arg(row['html']), arg(row['text']), arg(row['receivedAt']), arg('imported'), arg('mbox'),
412
+ ]},
413
+ {'sql': INSERT_SENT_META, 'args': [arg(row['id']), arg(row['owner']), arg(row['receivedAt'])]},
414
+ ]
415
+
416
+
417
+ def main():
418
+ parser = argparse.ArgumentParser()
419
+ parser.add_argument('mbox')
420
+ parser.add_argument('--owner', required=True)
421
+ parser.add_argument('--dry-run', action='store_true')
422
+ parser.add_argument('--limit', type=int, help='stop after this many messages')
423
+ parser.add_argument('--start', type=int, default=0, help='skip this many messages first (resume)')
424
+ parser.add_argument('--batch', type=int, default=10)
425
+ parser.add_argument('--mark-read', action='store_true')
426
+ parser.add_argument('--skip-attachments', action='store_true')
427
+ parser.add_argument('--attach-only', action='store_true',
428
+ help='second pass: upload attachments for rows already imported and write the URLs back')
429
+ parser.add_argument('--workers', type=int, default=8, help='parallel attachment uploads')
430
+ parser.add_argument('--writers', type=int, default=6, help='database batches in flight at once')
431
+ parser.add_argument('--max-mbps', type=float, help='cap aggregate upload bandwidth, e.g. 4.5')
432
+ parser.add_argument('--chunks', type=int, default=4, help='attachment chunks in flight at once')
433
+ parser.add_argument('--replace-blob', action='store_true',
434
+ help='treat entries hosted on Vercel Blob as missing so they are re-uploaded to the bucket')
435
+ parser.add_argument('--store', choices=['blob', 's3'], default='s3' if S3.configured() else 'blob',
436
+ help='where attachment bytes go: Vercel Blob, or an S3-API bucket (R2/S3)')
437
+ args = parser.parse_args()
438
+ if args.attach_only and args.batch == 10:
439
+ args.batch = 5
440
+
441
+ global STORE
442
+ STORE = args.store
443
+ owner = args.owner.lower()
444
+ if STORE == 's3' and not S3.configured() and not args.dry_run and not args.skip_attachments:
445
+ sys.exit('--store s3 needs S3_/R2_ endpoint, bucket and keys in the environment')
446
+ if STORE == 's3':
447
+ print(f'attachment store: s3 bucket {S3.bucket} at {S3.endpoint} (region {S3.region})', flush=True)
448
+ if args.max_mbps:
449
+ UPLOAD_BUDGET.__init__(args.max_mbps)
450
+ print(f'upload cap: {args.max_mbps} Mbit/s', flush=True)
451
+ url, db_token = os.environ.get('TURSO_DATABASE_URL'), os.environ.get('TURSO_AUTH_TOKEN')
452
+ blob_token = None if args.skip_attachments else (os.environ.get('BLOB_READ_WRITE_TOKEN') if STORE == 'blob' else 's3')
453
+ if not args.dry_run and not (url and db_token):
454
+ sys.exit('TURSO_DATABASE_URL and TURSO_AUTH_TOKEN must be set')
455
+ if not args.dry_run and not args.skip_attachments and not blob_token:
456
+ print('BLOB_READ_WRITE_TOKEN unset: attachments will be metadata only', flush=True)
457
+
458
+ stats = {'seen': 0, 'inbox': 0, 'sent': 0, 'skipped': 0, 'existing': 0, 'undated': 0, 'missing': 0,
459
+ 'updated': 0, 'files': 0, 'file_bytes': 0, 'files_skipped': 0, 'file_errors': 0}
460
+ chunk = []
461
+ pool = concurrent.futures.ThreadPoolExecutor(max_workers=args.workers) if args.attach_only else None
462
+
463
+ def process_attach_chunk(rows):
464
+ local = {'updated': 0, 'existing': 0, 'missing': 0, 'files': 0, 'file_bytes': 0,
465
+ 'files_skipped': 0, 'file_errors': 0}
466
+ if args.dry_run:
467
+ for row in rows:
468
+ local['files'] += len(row['attachments'])
469
+ local['file_bytes'] += sum(a['size'] for a in row['attachments'])
470
+ return local
471
+ stored = stored_attachments(url, db_token, [i for row in rows for i in (row['primary_id'], row['alt_id'])])
472
+ jobs = []
473
+ for row in rows:
474
+ mine = next((i for i in (row['primary_id'], row['alt_id'])
475
+ if i in stored and stored[i][0] == row['owner']), None)
476
+ if mine is None:
477
+ local['missing'] += 1
478
+ continue
479
+ row['id'] = mine
480
+ entries = stored[mine][1]
481
+ want = len(row['attachments'])
482
+ complete = lambda e: bool(e.get('key')) or (bool(e.get('url')) and not args.replace_blob)
483
+ # Completeness is tested before reuse. The other order rewrote every shared copy
484
+ # whose primary was keyed on every pass — even ones already finished — so a
485
+ # mailbox of shared copies burned whole slices without ever reaching what was
486
+ # actually missing. Compare >= and slice: a stored row can hold more entries
487
+ # than the message parses out, and such a row yields no work either way.
488
+ if len(entries) >= want and all(complete(e) for e in entries[:want]):
489
+ local['existing'] += 1
490
+ continue
491
+ if mine == row['alt_id'] and row['primary_id'] in stored:
492
+ primary_entries = stored[row['primary_id']][1]
493
+ if len(primary_entries) >= want and all(e.get('key') for e in primary_entries[:want]):
494
+ row['stored'] = [dict(e) for e in primary_entries[:want]]
495
+ row['reuse'] = True
496
+ local['reused'] = local.get('reused', 0) + 1
497
+ continue
498
+ row['stored'] = entries
499
+ for index, attachment in enumerate(row['attachments']):
500
+ if index < len(entries) and complete(entries[index]):
501
+ continue
502
+ if attachment['size'] > BLOB_LIMIT_BYTES:
503
+ local['files_skipped'] += 1
504
+ continue
505
+ jobs.append((row, index, attachment))
506
+ futures = {pool.submit(store_attachment, blob_token, row['id'], attachment): (row, index, attachment)
507
+ for row, index, attachment in jobs}
508
+ touched = set()
509
+ for future in concurrent.futures.as_completed(futures):
510
+ row, index, attachment = futures[future]
511
+ try:
512
+ stored_fields = future.result()
513
+ except Exception as err:
514
+ local['file_errors'] += 1
515
+ print(f' ! {attachment["filename"]}: {str(err)[:100]}', flush=True)
516
+ continue
517
+ entries = row['stored']
518
+ while len(entries) <= index:
519
+ entries.append({})
520
+ entries[index] = {'filename': attachment['filename'], 'contentType': attachment['contentType'],
521
+ 'size': attachment['size'], **stored_fields}
522
+ local['files'] += 1
523
+ local['file_bytes'] += attachment['size']
524
+ touched.add(row['id'])
525
+ for row in rows:
526
+ if row['id'] in touched or row.get('reuse'):
527
+ execute(url, db_token, [{'sql': 'UPDATE mail_inbox SET attachments = ? WHERE id = ?',
528
+ 'args': [arg(json.dumps(row['stored'])), arg(row['id'])]}])
529
+ local['updated'] += 1
530
+ for attachment in row['attachments']:
531
+ attachment.pop('bytes', None)
532
+ return local
533
+
534
+ chunk_pool = concurrent.futures.ThreadPoolExecutor(max_workers=args.chunks) if args.attach_only else None
535
+ attach_in_flight = []
536
+
537
+ def collect_attach(done_only=True):
538
+ remaining = []
539
+ for future in attach_in_flight:
540
+ if done_only and not future.done():
541
+ remaining.append(future)
542
+ continue
543
+ try:
544
+ counters = future.result()
545
+ except Exception as err:
546
+ stats['batch_errors'] = stats.get('batch_errors', 0) + 1
547
+ print(f' ! chunk lost: {str(err)[:140]}', flush=True)
548
+ continue
549
+ for key, value in counters.items():
550
+ stats[key] = stats.get(key, 0) + value
551
+ attach_in_flight[:] = remaining
552
+
553
+ def flush_attach():
554
+ if not chunk:
555
+ return
556
+ rows = chunk[:]
557
+ chunk.clear()
558
+ while len(attach_in_flight) >= args.chunks * 2:
559
+ concurrent.futures.wait(attach_in_flight, return_when=concurrent.futures.FIRST_COMPLETED)
560
+ collect_attach()
561
+ attach_in_flight.append(chunk_pool.submit(process_attach_chunk, rows))
562
+ collect_attach()
563
+
564
+ def write_rows(rows):
565
+ local = {'inbox': 0, 'sent': 0, 'existing': 0, 'files': 0, 'file_bytes': 0,
566
+ 'files_skipped': 0, 'file_errors': 0, 'row_errors': 0}
567
+ if args.dry_run:
568
+ for row in rows:
569
+ local[row['kind']] += 1
570
+ local['files'] += len(row['attachments'])
571
+ local['file_bytes'] += sum(a['size'] for a in row['attachments'])
572
+ return local
573
+ owners = existing_owners(url, db_token, [i for row in rows for i in (row['primary_id'], row['alt_id'])])
574
+ statements = []
575
+ for row in rows:
576
+ if row['alt_id'] in owners or owners.get(row['primary_id']) == row['owner']:
577
+ local['existing'] += 1
578
+ continue
579
+ if row['primary_id'] in owners:
580
+ row['id'] = row['alt_id']
581
+ row['headers']['shared-copy-of'] = row['primary_id']
582
+ local['shared'] = local.get('shared', 0) + 1
583
+ for attachment in row['attachments']:
584
+ if blob_token and attachment['size'] <= BLOB_LIMIT_BYTES:
585
+ try:
586
+ attachment.update(store_attachment(blob_token, row['id'], attachment))
587
+ local['files'] += 1
588
+ local['file_bytes'] += attachment['size']
589
+ except Exception as err:
590
+ local['file_errors'] += 1
591
+ print(f' ! {attachment["filename"]}: {err}', flush=True)
592
+ elif blob_token:
593
+ local['files_skipped'] += 1
594
+ attachment.pop('bytes', None)
595
+ if row['kind'] == 'sent':
596
+ statements += sent_statements(row)
597
+ else:
598
+ statements.append(inbox_statement(row))
599
+ local[row['kind']] += 1
600
+ if not statements:
601
+ return local
602
+ try:
603
+ execute(url, db_token, statements)
604
+ except Exception as err:
605
+ print(f' batch failed ({str(err)[:120]}); retrying rows one at a time', flush=True)
606
+ for statement in statements:
607
+ try:
608
+ execute(url, db_token, [statement])
609
+ except Exception as inner:
610
+ local['row_errors'] += 1
611
+ print(f' ! row failed: {str(inner)[:160]}', flush=True)
612
+ return local
613
+
614
+ writers = concurrent.futures.ThreadPoolExecutor(max_workers=args.writers)
615
+ in_flight = []
616
+
617
+ def collect(done_only=True):
618
+ remaining = []
619
+ for future in in_flight:
620
+ if done_only and not future.done():
621
+ remaining.append(future)
622
+ continue
623
+ try:
624
+ counters = future.result()
625
+ except Exception as err:
626
+ stats['batch_errors'] = stats.get('batch_errors', 0) + 1
627
+ print(f' ! batch lost: {str(err)[:140]}', flush=True)
628
+ continue
629
+ for key, value in counters.items():
630
+ stats[key] = stats.get(key, 0) + value
631
+ in_flight[:] = remaining
632
+
633
+ def flush():
634
+ if not chunk:
635
+ return
636
+ rows = chunk[:]
637
+ chunk.clear()
638
+ while len(in_flight) >= args.writers * 2:
639
+ concurrent.futures.wait(in_flight, return_when=concurrent.futures.FIRST_COMPLETED)
640
+ collect()
641
+ in_flight.append(writers.submit(write_rows, rows))
642
+ collect()
643
+
644
+ for index, raw in enumerate(messages(args.mbox)):
645
+ if index < args.start:
646
+ continue
647
+ if args.limit and stats['seen'] >= args.limit:
648
+ break
649
+ stats['seen'] += 1
650
+ row = row_for(email.message_from_bytes(raw), owner, args.mark_read, keep_bytes=not args.skip_attachments)
651
+ if row['kind'] == 'skip':
652
+ stats['skipped'] += 1
653
+ continue
654
+ if not row['receivedAt']:
655
+ stats['undated'] += 1
656
+ continue
657
+ if stats['seen'] <= 12:
658
+ flag = 'read ' if row['read'] else 'UNRD '
659
+ print(f' {row["receivedAt"][:10]} {row["kind"]:5} {flag} {row["from"][:30]:30} {row["subject"][:44]}'
660
+ f' [{len(row["attachments"])} files]', flush=True)
661
+ if args.attach_only:
662
+ if row['kind'] != 'inbox' or not row['attachments']:
663
+ continue
664
+ chunk.append(row)
665
+ if len(chunk) >= args.batch:
666
+ flush_attach()
667
+ if stats['seen'] % 500 == 0:
668
+ print(f'... {stats["seen"]} seen, {stats["updated"]} rows updated, {stats["existing"]} already done, '
669
+ f'{stats.get("reused", 0)} reused, {stats.get("missing", 0)} missing, '
670
+ f'{stats["files"]} files ({stats["file_bytes"]/1e9:.2f} GB), {stats["file_errors"]} errors', flush=True)
671
+ continue
672
+ chunk.append(row)
673
+ if len(chunk) >= args.batch:
674
+ flush()
675
+ if stats['seen'] % 500 == 0:
676
+ print(f'... {stats["seen"]} seen, {stats["inbox"]} inbox, {stats["sent"]} sent, '
677
+ f'{stats["existing"]} existing, {stats["files"]} files ({stats["file_bytes"]/1e9:.2f} GB)', flush=True)
678
+ if args.attach_only:
679
+ flush_attach()
680
+ concurrent.futures.wait(attach_in_flight)
681
+ collect_attach(done_only=False)
682
+ chunk_pool.shutdown(wait=True)
683
+ pool.shutdown(wait=True)
684
+ else:
685
+ flush()
686
+ concurrent.futures.wait(in_flight)
687
+ collect(done_only=False)
688
+ writers.shutdown(wait=True)
689
+
690
+ mode = 'dry run — nothing written' if args.dry_run else ('attachments backfilled' if args.attach_only else 'imported')
691
+ print(f'\n{mode} for {owner}')
692
+ if args.attach_only:
693
+ print(f' seen {stats["seen"]} rows updated {stats["updated"]} already complete {stats["existing"]} '
694
+ f'not in db {stats["missing"]} keys reused from shared copy {stats.get("reused", 0)}')
695
+ else:
696
+ print(f' seen {stats["seen"]} inbox {stats["inbox"]} sent {stats["sent"]} spam/drafts skipped {stats["skipped"]}'
697
+ f' already present {stats["existing"]} shared copies {stats.get("shared", 0)} undated {stats["undated"]}')
698
+ print(f' attachments {stats["files"]} ({stats["file_bytes"]/1e9:.2f} GB)'
699
+ f' over 20MB skipped {stats["files_skipped"]} upload errors {stats["file_errors"]} row errors {stats.get("row_errors", 0)} batches lost {stats.get("batch_errors", 0)}')
700
+
701
+
702
+ if __name__ == '__main__':
703
+ main()