@novacraft-engineering/mailbox 0.4.28 → 0.4.30
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/app/api/mail/inbox/counts/route.ts +13 -3
- package/app/api/mail/inbox/route.ts +6 -2
- package/app/api/mail/maintenance/rejudge/route.ts +1 -0
- package/app/api/mail/threads/route.ts +4 -3
- package/app/mail/InboxFilters.tsx +154 -0
- package/app/mail/page.module.css +182 -0
- package/app/mail/page.tsx +86 -9
- package/lib/inbox-filters.test.ts +54 -0
- package/lib/inbox-filters.ts +111 -0
- package/lib/mailbox.ts +120 -217
- package/lib/receive.ts +11 -2
- package/lib/risk.test.ts +35 -0
- package/lib/risk.ts +203 -0
- package/package.json +1 -1
package/lib/receive.ts
CHANGED
|
@@ -2,7 +2,8 @@ import { BRAND, ADDRESS_DOMAINS } from '@/lib/brand'
|
|
|
2
2
|
import { sendPush } from '@/lib/push'
|
|
3
3
|
import { stripOwnPixel } from '@/lib/email-html'
|
|
4
4
|
import { FORWARDING_ENABLED, FORWARD_RECIPIENTS, MAIL_DOMAIN } from '@/lib/dev-auth'
|
|
5
|
-
import { ADDRESS_ALIASES, appendInbound, classifyAddressed, isDmarcAggregateReport, judgeMessage, noteSender, senderStanding, senderDomainOf, getAccountByAddress, inboundExists, recordContact, recordSentMessage, recordSentMeta, repairInbound } from '@/lib/mailbox'
|
|
5
|
+
import { ADDRESS_ALIASES, appendInbound, classifyAddressed, isDmarcAggregateReport, judgeMessage, noteSender, senderStanding, senderDomainOf, getAccountByAddress, inboundExists, inboxFiltersFor, recordContact, recordSentMessage, recordSentMeta, repairInbound } from '@/lib/mailbox'
|
|
6
|
+
import { matchesInboxFilters } from '@/lib/inbox-filters'
|
|
6
7
|
import { archiveAddress, sendMail } from '@/lib/mail-provider'
|
|
7
8
|
|
|
8
9
|
/**
|
|
@@ -322,6 +323,7 @@ export async function ingestReceived(
|
|
|
322
323
|
replyTo: full?.replyTo ?? (Array.isArray(data.reply_to) ? (data.reply_to as string[]) : []),
|
|
323
324
|
subject: full?.subject || String(data.subject ?? ''),
|
|
324
325
|
text: full?.text ?? (data.text ? String(data.text) : null),
|
|
326
|
+
receivedAt: full?.createdAt || String(data.created_at ?? new Date().toISOString()),
|
|
325
327
|
}, standing)
|
|
326
328
|
if (verdicts.risk !== 'clean') {
|
|
327
329
|
console.warn(`[mail] ${verdicts.risk} inbound ${emailId}:`, verdicts.reasons.join('; '))
|
|
@@ -401,7 +403,7 @@ export async function ingestReceived(
|
|
|
401
403
|
await noteSender(owner, senderDomain, 'received').catch(() => {})
|
|
402
404
|
const sender = parseSender(inbound.from)
|
|
403
405
|
await recordContact(sender.email, sender.name)
|
|
404
|
-
if (!isDmarcAggregateReport(inbound.subject)) {
|
|
406
|
+
if (!isDmarcAggregateReport(inbound.subject) && !(await filteredOut(inbound))) {
|
|
405
407
|
await sendPush(inbound.owner, {
|
|
406
408
|
title: sender.name || sender.email || 'New mail',
|
|
407
409
|
body: inbound.subject,
|
|
@@ -429,6 +431,13 @@ export async function ingestReceived(
|
|
|
429
431
|
return { owner: inbound.owner, subject: inbound.subject, from: inbound.from }
|
|
430
432
|
}
|
|
431
433
|
|
|
434
|
+
async function filteredOut(inbound: { owner: string | null; from: string; subject: string; text: string | null }): Promise<boolean> {
|
|
435
|
+
const login = inbound.owner ? (await getAccountByAddress(inbound.owner).catch(() => null))?.email : null
|
|
436
|
+
if (!login) return false
|
|
437
|
+
const filters = await inboxFiltersFor(login).catch(() => [])
|
|
438
|
+
return matchesInboxFilters(filters, { senders: [inbound.from], subject: inbound.subject, text: inbound.text })
|
|
439
|
+
}
|
|
440
|
+
|
|
432
441
|
/** A copy of our own outgoing mail, BCC'd to the archive seat: it belongs in that seat's Sent, not its inbox. */
|
|
433
442
|
export function isArchiveCopy(owner: string | null | undefined, to: string[], cc: string[]): boolean {
|
|
434
443
|
const archive = archiveAddress()
|
package/lib/risk.test.ts
ADDED
|
@@ -0,0 +1,35 @@
|
|
|
1
|
+
import assert from 'node:assert/strict'
|
|
2
|
+
|
|
3
|
+
process.env.MAIL_ADDRESS_DOMAIN = 'example.com'
|
|
4
|
+
const { judgeMessage, FIRST_MESSAGE_REASON } = await import('./risk.ts')
|
|
5
|
+
type Standing = Parameters<typeof judgeMessage>[1]
|
|
6
|
+
|
|
7
|
+
const stranger: Standing = { received: 0, trashed: 0, markedSpam: 0, replied: 0, trusted: false, firstSeen: null }
|
|
8
|
+
const known = (firstSeen: string, extra: Partial<Standing> = {}): Standing => ({ ...stranger, received: 2, firstSeen, ...extra })
|
|
9
|
+
|
|
10
|
+
const promo = { from: 'Swift <customercare@swiftng.net>', spf: 'fail', subject: 'We Want You Back', receivedAt: '2026-09-23T20:36:05.281Z' }
|
|
11
|
+
assert.equal(judgeMessage(promo, known('2022-01-10T10:49:25.000Z')).risk, 'clean', 'a sender with years of history is not writing for the first time')
|
|
12
|
+
const firstTime = judgeMessage(promo, stranger)
|
|
13
|
+
assert.equal(firstTime.risk, 'suspicious')
|
|
14
|
+
assert.ok(firstTime.reasons.includes(FIRST_MESSAGE_REASON), 'a real first message still says so')
|
|
15
|
+
assert.ok(!judgeMessage(promo, known('2026-09-23T20:30:00.000Z')).reasons.includes(FIRST_MESSAGE_REASON), 'the second message is not the first')
|
|
16
|
+
assert.ok(judgeMessage(promo, known(promo.receivedAt)).reasons.includes(FIRST_MESSAGE_REASON), 'the earliest stored message is the first, when re-judged')
|
|
17
|
+
|
|
18
|
+
const enquiry = {
|
|
19
|
+
from: 'Example Website <no-reply@example.com>',
|
|
20
|
+
replyTo: ['someone@gmail.com'],
|
|
21
|
+
dmarc: 'pass',
|
|
22
|
+
spf: 'pass',
|
|
23
|
+
dkim: 'pass',
|
|
24
|
+
subject: 'New enquiry — Motor insurance',
|
|
25
|
+
receivedAt: '2026-09-14T07:24:40.046Z',
|
|
26
|
+
}
|
|
27
|
+
assert.equal(judgeMessage(enquiry, stranger).risk, 'clean', 'our own authenticated form mail is not a stranger misdirecting replies')
|
|
28
|
+
const forged = judgeMessage({ ...enquiry, dmarc: 'fail', spf: 'fail', dkim: 'fail' }, stranger)
|
|
29
|
+
assert.notEqual(forged.risk, 'clean', 'mail that only claims our domain earns nothing')
|
|
30
|
+
assert.equal(judgeMessage({ ...enquiry, from: 'Web <no-reply@example.org>' }, stranger).risk, 'suspicious', 'another domain doing the same is still flagged')
|
|
31
|
+
|
|
32
|
+
assert.equal(judgeMessage(promo, { ...stranger, trusted: true }).risk, 'clean', 'a trusted sender is forgiven the small stuff')
|
|
33
|
+
assert.notEqual(judgeMessage(promo, { ...stranger, trusted: true, markedSpam: 2 }).risk, 'clean', 'marking spam outranks trust')
|
|
34
|
+
|
|
35
|
+
console.log('risk: ok')
|
package/lib/risk.ts
ADDED
|
@@ -0,0 +1,203 @@
|
|
|
1
|
+
import { ADDRESS_DOMAINS } from './brand.ts'
|
|
2
|
+
|
|
3
|
+
export type SenderStanding = {
|
|
4
|
+
received: number
|
|
5
|
+
trashed: number
|
|
6
|
+
markedSpam: number
|
|
7
|
+
replied: number
|
|
8
|
+
/** Someone in this mailbox said so. */
|
|
9
|
+
trusted: boolean
|
|
10
|
+
/** The earliest mail this mailbox holds from the sender's domain, whoever it went to. */
|
|
11
|
+
firstSeen: string | null
|
|
12
|
+
}
|
|
13
|
+
|
|
14
|
+
export const FIRST_MESSAGE_REASON = 'This is the first message from this sender'
|
|
15
|
+
|
|
16
|
+
/** What the scanners and the sender's own domain said. */
|
|
17
|
+
export type Risk = 'clean' | 'suspicious' | 'spam' | 'virus'
|
|
18
|
+
|
|
19
|
+
export type RiskSignals = {
|
|
20
|
+
spam?: string | null
|
|
21
|
+
virus?: string | null
|
|
22
|
+
spf?: string | null
|
|
23
|
+
dkim?: string | null
|
|
24
|
+
dmarc?: string | null
|
|
25
|
+
/** The message itself, for the tells authentication cannot see. */
|
|
26
|
+
from?: string | null
|
|
27
|
+
replyTo?: string[] | null
|
|
28
|
+
subject?: string | null
|
|
29
|
+
text?: string | null
|
|
30
|
+
receivedAt?: string | null
|
|
31
|
+
}
|
|
32
|
+
|
|
33
|
+
/** Defaults only. Each is overridable per deployment, so a list can change without a
|
|
34
|
+
* release — metroperil can drop a word its own trade uses every day. */
|
|
35
|
+
const FREE_MAIL_DEFAULT = new Set([
|
|
36
|
+
'gmail.com', 'googlemail.com', 'yahoo.com', 'ymail.com', 'hotmail.com', 'outlook.com',
|
|
37
|
+
'live.com', 'aol.com', 'protonmail.com', 'proton.me', 'mail.com', 'gmx.com', 'yandex.com',
|
|
38
|
+
'icloud.com', 'zoho.com', 'inbox.lv', 'consultant.com', 'qq.com', '163.com',
|
|
39
|
+
])
|
|
40
|
+
|
|
41
|
+
const THROWAWAY_TLDS_DEFAULT = new Set([
|
|
42
|
+
'xyz', 'top', 'buzz', 'click', 'link', 'work', 'gq', 'cf', 'ml', 'tk', 'ga',
|
|
43
|
+
'loan', 'men', 'date', 'racing', 'win', 'stream', 'download', 'review', 'country', 'kim',
|
|
44
|
+
])
|
|
45
|
+
|
|
46
|
+
/** The shape of an advance-fee approach. Counted, never single-word: one alone is innocent. */
|
|
47
|
+
// Only wording that is odd in ordinary business correspondence belongs here. A single
|
|
48
|
+
// generic term is not evidence of anything: an insurance broker writes "beneficiary" and
|
|
49
|
+
// "bank draft" all day, a logistics firm writes "consignment", and every sales team sends
|
|
50
|
+
// a "business proposal". Add them per deployment through MAIL_SCAM_PHRASES if a mailbox
|
|
51
|
+
// genuinely never sees them.
|
|
52
|
+
const SCAM_PHRASES_DEFAULT = [
|
|
53
|
+
'next of kin', 'sole beneficiary', 'late client', 'deceased client',
|
|
54
|
+
'inheritance', 'died without', 'without a will', 'unclaimed inheritance',
|
|
55
|
+
'winning notification', 'lottery winner', 'western union', 'atm card',
|
|
56
|
+
'transfer to your account immediately', 'strictly confidential and urgent',
|
|
57
|
+
]
|
|
58
|
+
|
|
59
|
+
/** The registrable domain behind an address, for reputation to be keyed on. */
|
|
60
|
+
export const senderDomainOf = (address: string): string => registrable(domainOf(address))
|
|
61
|
+
|
|
62
|
+
const domainOf = (address: string): string => {
|
|
63
|
+
const angled = address.match(/<([^>]+)>/)
|
|
64
|
+
const bare = (angled ? angled[1] : address).trim().toLowerCase()
|
|
65
|
+
return bare.split('@').pop() ?? ''
|
|
66
|
+
}
|
|
67
|
+
|
|
68
|
+
/** example.co.uk and example.com both reduce to the name somebody actually registered. */
|
|
69
|
+
const registrable = (host: string): string => {
|
|
70
|
+
const parts = host.split('.').filter(Boolean)
|
|
71
|
+
if (parts.length <= 2) return parts.join('.')
|
|
72
|
+
const twoLevel = /^(co|com|org|net|gov|ac|edu|ltd|plc)\.[a-z]{2}$/.test(parts.slice(-2).join('.'))
|
|
73
|
+
return parts.slice(twoLevel ? -3 : -2).join('.')
|
|
74
|
+
}
|
|
75
|
+
|
|
76
|
+
const failed = (verdict: string | null | undefined): boolean =>
|
|
77
|
+
typeof verdict === 'string' && /^(fail|softfail|permerror)$/i.test(verdict.trim())
|
|
78
|
+
|
|
79
|
+
const listFrom = (raw: string | undefined, fallback: Iterable<string>): Set<string> => {
|
|
80
|
+
const parsed = (raw ?? '').split(',').map(entry => entry.trim().toLowerCase()).filter(Boolean)
|
|
81
|
+
return parsed.length ? new Set(parsed) : new Set(fallback)
|
|
82
|
+
}
|
|
83
|
+
|
|
84
|
+
// Read per call, so a deployment can change any of them without a release.
|
|
85
|
+
const freeProviders = () => listFrom(process.env.MAIL_FREE_PROVIDERS, FREE_MAIL_DEFAULT)
|
|
86
|
+
const throwawayTlds = () => listFrom(process.env.MAIL_THROWAWAY_TLDS, THROWAWAY_TLDS_DEFAULT)
|
|
87
|
+
|
|
88
|
+
// Bulk senders put their own bounce domain in From and the real correspondent in Reply-To.
|
|
89
|
+
// That is how the campaign gets replies, not an attempt to redirect them somewhere unexpected.
|
|
90
|
+
const BULK_SENDERS_DEFAULT = new Set([
|
|
91
|
+
'mailchimpapp.com', 'mcsv.net', 'rsgsv.net', 'mailchimp.com',
|
|
92
|
+
'sendgrid.net', 'sendgrid.com', 'sparkpostmail.com', 'amazonses.com',
|
|
93
|
+
'mailgun.org', 'mandrillapp.com', 'postmarkapp.com', 'sendinblue.com',
|
|
94
|
+
'brevo.com', 'constantcontact.com', 'cmail19.com', 'createsend.com',
|
|
95
|
+
'hubspotemail.net', 'mailerlite.com', 'klaviyomail.com', 'salesforce.com',
|
|
96
|
+
])
|
|
97
|
+
const bulkSenders = () => listFrom(process.env.MAIL_BULK_SENDERS, BULK_SENDERS_DEFAULT)
|
|
98
|
+
const scamPhrases = () => [...listFrom(process.env.MAIL_SCAM_PHRASES, SCAM_PHRASES_DEFAULT)]
|
|
99
|
+
|
|
100
|
+
/** Weight at which a message stops being labelled and is held out of the inbox instead. */
|
|
101
|
+
const quarantineAt = () => Number(process.env.MAIL_SPAM_THRESHOLD ?? 6)
|
|
102
|
+
|
|
103
|
+
/** Below this nothing is said at all. One small oddity is not a case. */
|
|
104
|
+
const flagAt = () => Number(process.env.MAIL_SUSPICION_THRESHOLD ?? 3)
|
|
105
|
+
|
|
106
|
+
export type RiskJudgement = { risk: Risk; reasons: string[]; score: number; quarantine: boolean }
|
|
107
|
+
|
|
108
|
+
/**
|
|
109
|
+
* What this mailbox knows, then what is true of the message. The standing a sender has
|
|
110
|
+
* built here leads: somebody you have written back to is not spam because their subject
|
|
111
|
+
* shouts, and somebody whose mail you have binned repeatedly does not get the benefit of
|
|
112
|
+
* the doubt again. The fixed rules only decide the cases with no history to go on, and
|
|
113
|
+
* every one of their lists can be changed per deployment without a release.
|
|
114
|
+
*/
|
|
115
|
+
export function judgeMessage(signals: RiskSignals, standing: SenderStanding): RiskJudgement {
|
|
116
|
+
const reasons: string[] = []
|
|
117
|
+
let score = 0
|
|
118
|
+
// A finding is "telling" when it is hard to trip by accident. Failing an authentication
|
|
119
|
+
// check or shouting in the subject line is neither: ordinary mail does both. Holding a
|
|
120
|
+
// message back takes at least one finding of the first kind, however the weights add up.
|
|
121
|
+
let telling = 0
|
|
122
|
+
const add = (weight: number, why: string, isTelling = false) => {
|
|
123
|
+
score += weight
|
|
124
|
+
if (isTelling) telling += 1
|
|
125
|
+
reasons.push(why)
|
|
126
|
+
}
|
|
127
|
+
|
|
128
|
+
if (/^fail$/i.test((signals.virus ?? '').trim())) {
|
|
129
|
+
return { risk: 'virus', reasons: ['A virus scan failed on this message'], score: 100, quarantine: true }
|
|
130
|
+
}
|
|
131
|
+
|
|
132
|
+
// Trust is earned by being written back to, never by volume alone: a sender whose mail
|
|
133
|
+
// arrives forty times and is binned every time has not earned anything.
|
|
134
|
+
const fromDomain = registrable(domainOf(signals.from ?? ''))
|
|
135
|
+
const authenticated = /^pass$/i.test((signals.dmarc ?? '').trim())
|
|
136
|
+
// Our own domain, proven by DMARC: the website's forms and the platform's own notices.
|
|
137
|
+
// They set Reply-To to the customer on purpose and arrive from an address nobody writes back to.
|
|
138
|
+
const ownDomain = authenticated && ADDRESS_DOMAINS.some(domain => registrable(domain) === fromDomain)
|
|
139
|
+
const trusted = (standing.replied > 0 || standing.trusted || ownDomain) && standing.markedSpam === 0
|
|
140
|
+
if (standing.markedSpam > 0) {
|
|
141
|
+
add(4 + Math.min(standing.markedSpam, 4),
|
|
142
|
+
`You marked ${standing.markedSpam} earlier message${standing.markedSpam === 1 ? '' : 's'} from this sender as spam`, true)
|
|
143
|
+
} else if (standing.trashed >= 3 && standing.replied === 0) {
|
|
144
|
+
add(3, `You have deleted ${standing.trashed} messages from this sender without ever replying`, true)
|
|
145
|
+
}
|
|
146
|
+
|
|
147
|
+
if (/^fail$/i.test((signals.spam ?? '').trim())) add(4, 'The provider\u2019s spam filter flagged this message', true)
|
|
148
|
+
|
|
149
|
+
// Heavy, but not enough on its own to hide a message: mail forwarded through a list
|
|
150
|
+
// breaks alignment and fails DMARC while being perfectly legitimate. It warns loudly;
|
|
151
|
+
// it takes a second finding to put a message out of sight.
|
|
152
|
+
if (failed(signals.dmarc)) add(4, 'The sending domain says this message is not from them (DMARC failed)')
|
|
153
|
+
else if (!authenticated) {
|
|
154
|
+
if (failed(signals.spf)) add(2, 'The sending server is not authorised by that domain (SPF failed)')
|
|
155
|
+
if (failed(signals.dkim)) add(2, 'The signature does not match the sending domain (DKIM failed)')
|
|
156
|
+
}
|
|
157
|
+
|
|
158
|
+
const replyDomains = (signals.replyTo ?? [])
|
|
159
|
+
.map(entry => registrable(domainOf(entry)))
|
|
160
|
+
.filter(entry => entry && entry !== fromDomain)
|
|
161
|
+
const free = freeProviders()
|
|
162
|
+
const freeReply = replyDomains.find(entry => free.has(entry))
|
|
163
|
+
const bulk = bulkSenders().has(fromDomain)
|
|
164
|
+
if (bulk) {
|
|
165
|
+
// Nothing to say: a campaign's replies are meant to land somewhere other than the
|
|
166
|
+
// sending platform, and treating that as misdirection buries ordinary bulk mail.
|
|
167
|
+
} else if (freeReply && fromDomain && !free.has(fromDomain)) {
|
|
168
|
+
add(4, `Replies to this message go to ${freeReply}, not to ${fromDomain}`, true)
|
|
169
|
+
} else if (replyDomains.length) {
|
|
170
|
+
add(1, `Replies go to ${replyDomains[0]} rather than ${fromDomain || 'the sender'}`)
|
|
171
|
+
}
|
|
172
|
+
|
|
173
|
+
const tld = fromDomain.split('.').pop() ?? ''
|
|
174
|
+
if (throwawayTlds().has(tld)) add(2, `The sender\u2019s domain ends in .${tld}, which is cheap to register and often disposable`, true)
|
|
175
|
+
if (/^\d{4,}$/.test(fromDomain.split('.')[0] ?? '')) add(2, 'The sender\u2019s domain name is just a string of digits', true)
|
|
176
|
+
|
|
177
|
+
const subject = (signals.subject ?? '').trim()
|
|
178
|
+
const letters = subject.replace(/[^A-Za-z]/g, '')
|
|
179
|
+
if (letters.length >= 12 && letters === letters.toUpperCase()) add(1, 'The subject is written entirely in capitals')
|
|
180
|
+
|
|
181
|
+
const body = (signals.text ?? '').toLowerCase()
|
|
182
|
+
const hits = scamPhrases().filter(phrase => body.includes(phrase))
|
|
183
|
+
if (hits.length >= 2) add(3, `The wording follows a known advance-fee approach (${hits.slice(0, 3).join(', ')})`, true)
|
|
184
|
+
else if (hits.length === 1) add(1, `Wording associated with advance-fee mail (${hits[0]})`)
|
|
185
|
+
|
|
186
|
+
// Never heard from before is not suspicious by itself — everyone writes once for the
|
|
187
|
+
// first time — but it is what turns a couple of small oddities into a pattern.
|
|
188
|
+
const seenBefore = Boolean(standing.firstSeen) && (!signals.receivedAt || standing.firstSeen! < signals.receivedAt)
|
|
189
|
+
if (!trusted && !seenBefore && score > 0) add(1, FIRST_MESSAGE_REASON)
|
|
190
|
+
|
|
191
|
+
// Someone this mailbox corresponds with is forgiven the small stuff; only findings heavy
|
|
192
|
+
// enough to stand on their own still count against them.
|
|
193
|
+
const limit = quarantineAt()
|
|
194
|
+
if (trusted && score < limit) return { risk: 'clean', reasons: [], score: 0, quarantine: false }
|
|
195
|
+
|
|
196
|
+
// One small oddity is not a case to answer. A subject in capitals from somebody writing
|
|
197
|
+
// for the first time is a stranger in a hurry, not a scam, and saying otherwise every
|
|
198
|
+
// time teaches the reader to ignore the warning.
|
|
199
|
+
if (score < flagAt()) return { risk: 'clean', reasons: [], score, quarantine: false }
|
|
200
|
+
|
|
201
|
+
const quarantine = score >= limit && telling > 0
|
|
202
|
+
return { risk: quarantine ? 'spam' : 'suspicious', reasons, score, quarantine }
|
|
203
|
+
}
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@novacraft-engineering/mailbox",
|
|
3
|
-
"version": "0.4.
|
|
3
|
+
"version": "0.4.30",
|
|
4
4
|
"description": "A shared webmail app: one codebase, one deployment per mailbox. Threads, a rich composer with signatures, attachments on S3-compatible storage, sharing links, full-text search, web push and PWA install.",
|
|
5
5
|
"keywords": [
|
|
6
6
|
"webmail",
|