imapkit 0.0.0-stage → 4.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +16 -0
- package/README.md +608 -2
- package/bin/help.txt +98 -0
- package/bin/imapkit.js +108 -0
- package/cert/server.crt +20 -0
- package/cert/server.key +28 -0
- package/lib/addressparser.js +283 -0
- package/lib/arguments.js +112 -0
- package/lib/bodystructure.js +149 -0
- package/lib/command-states.js +109 -0
- package/lib/commands/append.js +313 -0
- package/lib/commands/capability.js +47 -0
- package/lib/commands/check.js +21 -0
- package/lib/commands/close.js +30 -0
- package/lib/commands/copy.js +115 -0
- package/lib/commands/create.js +52 -0
- package/lib/commands/delete.js +64 -0
- package/lib/commands/examine.js +7 -0
- package/lib/commands/expunge.js +27 -0
- package/lib/commands/fetch.js +229 -0
- package/lib/commands/handlers/fetch.js +209 -0
- package/lib/commands/handlers/flags.js +42 -0
- package/lib/commands/handlers/search.js +519 -0
- package/lib/commands/handlers/status.js +85 -0
- package/lib/commands/handlers/store.js +127 -0
- package/lib/commands/list.js +100 -0
- package/lib/commands/login.js +67 -0
- package/lib/commands/logout.js +41 -0
- package/lib/commands/lsub.js +87 -0
- package/lib/commands/noop.js +21 -0
- package/lib/commands/rename.js +102 -0
- package/lib/commands/search.js +76 -0
- package/lib/commands/select.js +289 -0
- package/lib/commands/status.js +63 -0
- package/lib/commands/store.js +151 -0
- package/lib/commands/subscribe.js +53 -0
- package/lib/commands/uid copy.js +7 -0
- package/lib/commands/uid fetch.js +5 -0
- package/lib/commands/uid search.js +5 -0
- package/lib/commands/uid store.js +5 -0
- package/lib/commands/unsubscribe.js +50 -0
- package/lib/dates.js +123 -0
- package/lib/deflate-layer.js +232 -0
- package/lib/envelope.js +82 -0
- package/lib/esearch.js +208 -0
- package/lib/framing.js +102 -0
- package/lib/list-extensions.js +36 -0
- package/lib/load-plugins.js +109 -0
- package/lib/mailbox-name.js +133 -0
- package/lib/mimeparser.js +778 -0
- package/lib/mock-client.js +233 -0
- package/lib/numbers.js +52 -0
- package/lib/plugins/acl.js +964 -0
- package/lib/plugins/appendlimit.js +83 -0
- package/lib/plugins/auth-plain.js +94 -0
- package/lib/plugins/binary.js +256 -0
- package/lib/plugins/catenate.js +253 -0
- package/lib/plugins/compress.js +76 -0
- package/lib/plugins/condstore.js +563 -0
- package/lib/plugins/context-search.js +321 -0
- package/lib/plugins/context-sort.js +19 -0
- package/lib/plugins/create-special-use.js +108 -0
- package/lib/plugins/enable.js +155 -0
- package/lib/plugins/esearch.js +156 -0
- package/lib/plugins/esort.js +60 -0
- package/lib/plugins/id.js +138 -0
- package/lib/plugins/idle.js +105 -0
- package/lib/plugins/imap4rev2.js +202 -0
- package/lib/plugins/list-extended.js +258 -0
- package/lib/plugins/list-status.js +31 -0
- package/lib/plugins/literalminus.js +20 -0
- package/lib/plugins/literalplus.js +18 -0
- package/lib/plugins/logindisabled.js +50 -0
- package/lib/plugins/messagelimit.js +234 -0
- package/lib/plugins/metadata-server.js +13 -0
- package/lib/plugins/metadata.js +475 -0
- package/lib/plugins/move.js +110 -0
- package/lib/plugins/multiappend.js +26 -0
- package/lib/plugins/multisearch.js +269 -0
- package/lib/plugins/namespace.js +67 -0
- package/lib/plugins/notify.js +654 -0
- package/lib/plugins/oauthbearer.js +217 -0
- package/lib/plugins/objectid.js +243 -0
- package/lib/plugins/partial.js +68 -0
- package/lib/plugins/preview.js +400 -0
- package/lib/plugins/qresync.js +525 -0
- package/lib/plugins/quota.js +285 -0
- package/lib/plugins/replace.js +145 -0
- package/lib/plugins/sasl-ir.js +12 -0
- package/lib/plugins/savedate.js +59 -0
- package/lib/plugins/savelimit.js +18 -0
- package/lib/plugins/searchres.js +82 -0
- package/lib/plugins/sort-display.js +23 -0
- package/lib/plugins/sort.js +132 -0
- package/lib/plugins/special-use.js +95 -0
- package/lib/plugins/starttls.js +57 -0
- package/lib/plugins/status-size.js +19 -0
- package/lib/plugins/thread-orderedsubject.js +16 -0
- package/lib/plugins/thread-references.js +16 -0
- package/lib/plugins/uidonly.js +135 -0
- package/lib/plugins/uidplus.js +124 -0
- package/lib/plugins/unauthenticate.js +28 -0
- package/lib/plugins/unselect.js +36 -0
- package/lib/plugins/utf8-accept.js +68 -0
- package/lib/plugins/x-gm-ext-1.js +456 -0
- package/lib/plugins/xoauth2.js +188 -0
- package/lib/plugins/xtoybird.js +282 -0
- package/lib/server.js +2880 -0
- package/lib/smtp-listener.js +51 -0
- package/lib/sorting.js +373 -0
- package/lib/threading.js +357 -0
- package/lib/utf8-session.js +123 -0
- package/lib/vanished.js +57 -0
- package/package.json +61 -5
|
@@ -0,0 +1,400 @@
|
|
|
1
|
+
'use strict';
|
|
2
|
+
|
|
3
|
+
const { getMessageData, partsOf } = require('../mimeparser');
|
|
4
|
+
const { isAtom } = require('../arguments');
|
|
5
|
+
|
|
6
|
+
/**
|
|
7
|
+
* @help Adds PREVIEW [RFC8970] capability
|
|
8
|
+
* @help Previews come from the first text/plain or text/html body part,
|
|
9
|
+
* @help or from the "preview" property of a message in storage.
|
|
10
|
+
* @help PREVIEW (LAZY) returns NIL until a preview has been generated
|
|
11
|
+
*
|
|
12
|
+
* PREVIEW: https://www.rfc-editor.org/rfc/rfc8970
|
|
13
|
+
*
|
|
14
|
+
* Message storage property:
|
|
15
|
+
* - preview: preview text to return instead of a generated one
|
|
16
|
+
*/
|
|
17
|
+
module.exports = function (server) {
|
|
18
|
+
server.registerCapability('PREVIEW');
|
|
19
|
+
|
|
20
|
+
// Generated previews, so PREVIEW (LAZY) knows which ones are available without "undue delay"
|
|
21
|
+
const generated = new WeakMap();
|
|
22
|
+
|
|
23
|
+
// The preview that is available without generating it: the storage value or an earlier generated one
|
|
24
|
+
const readyPreview = message => {
|
|
25
|
+
if (typeof message.preview === 'string') {
|
|
26
|
+
// RFC 8970 3.3: the MUST NOT limits apply to the value from storage as well
|
|
27
|
+
return finishPreview(message.preview, MAX_PREVIEW_LENGTH);
|
|
28
|
+
}
|
|
29
|
+
const cached = generated.get(message);
|
|
30
|
+
return cached && cached.source === message.raw ? cached.preview : undefined;
|
|
31
|
+
};
|
|
32
|
+
|
|
33
|
+
server.fetchHandlers.PREVIEW = function (connection, message, query) {
|
|
34
|
+
let preview = readyPreview(message);
|
|
35
|
+
if (preview === undefined) {
|
|
36
|
+
// RFC 8970 4.1: with LAZY the server returns NIL when the preview is not readily available. Here that
|
|
37
|
+
// means a preview that has not been generated yet by a FETCH without LAZY (Dovecot behaves the same way)
|
|
38
|
+
if (query.previewLazy) {
|
|
39
|
+
return null;
|
|
40
|
+
}
|
|
41
|
+
preview = generatePreview(getMessageData(message).tree);
|
|
42
|
+
generated.set(message, { source: message.raw, preview });
|
|
43
|
+
}
|
|
44
|
+
// RFC 8970 3.3: UTF-8, sent as a literal by the compiler when it holds 8-bit characters
|
|
45
|
+
return Buffer.from(preview, 'utf8').toString('binary');
|
|
46
|
+
};
|
|
47
|
+
|
|
48
|
+
// RFC 8970 6: fetch-att =/ "PREVIEW" [SP "(" preview-mod *(SP preview-mod) ")"]. The parser returns the
|
|
49
|
+
// modifiers as a list that follows the PREVIEW atom, so fold them into the atom before FETCH sees them
|
|
50
|
+
['FETCH', 'UID FETCH'].forEach(command => {
|
|
51
|
+
const prevHandler = server.getCommandHandler(command);
|
|
52
|
+
server.setCommandHandler(command, (connection, parsed, data, callback) => {
|
|
53
|
+
try {
|
|
54
|
+
foldModifiers(parsed);
|
|
55
|
+
} catch (E) {
|
|
56
|
+
connection.sendStatus(parsed, data, 'BAD', E.message, false, 'PREVIEW FAILED');
|
|
57
|
+
return callback();
|
|
58
|
+
}
|
|
59
|
+
prevHandler(connection, parsed, data, callback);
|
|
60
|
+
});
|
|
61
|
+
});
|
|
62
|
+
};
|
|
63
|
+
|
|
64
|
+
// RFC 8970 3.3: the server SHOULD limit previews to 200 characters and MUST NOT exceed 256
|
|
65
|
+
const PREVIEW_LENGTH = 200;
|
|
66
|
+
const MAX_PREVIEW_LENGTH = 256;
|
|
67
|
+
|
|
68
|
+
// Input of preview generation is limited to the start of the part, enough for a 200 character preview
|
|
69
|
+
const MAX_INPUT = 64 * 1024;
|
|
70
|
+
|
|
71
|
+
const isModifierList = list => Array.isArray(list) && list.length > 0 && list.every(item => isAtom(item, 'LAZY'));
|
|
72
|
+
|
|
73
|
+
/**
|
|
74
|
+
* Returns the PREVIEW item with its modifiers. RFC 8970 6: preview-mod = "LAZY", at least one is required
|
|
75
|
+
*
|
|
76
|
+
* @param {Object} item PREVIEW atom
|
|
77
|
+
* @param {Array} list Parsed list that follows PREVIEW
|
|
78
|
+
* @return {Object} PREVIEW atom with the previewLazy flag
|
|
79
|
+
*/
|
|
80
|
+
function withModifiers(item, list) {
|
|
81
|
+
if (!list.length) {
|
|
82
|
+
throw new Error('PREVIEW modifier list can not be empty');
|
|
83
|
+
}
|
|
84
|
+
if (!isModifierList(list)) {
|
|
85
|
+
throw new Error('Unknown PREVIEW modifier');
|
|
86
|
+
}
|
|
87
|
+
return Object.assign({}, item, { previewLazy: true });
|
|
88
|
+
}
|
|
89
|
+
|
|
90
|
+
/**
|
|
91
|
+
* Turns `PREVIEW (LAZY)` in the FETCH arguments into a single PREVIEW item with the previewLazy flag
|
|
92
|
+
*
|
|
93
|
+
* @param {Object} parsed Parsed command
|
|
94
|
+
*/
|
|
95
|
+
function foldModifiers(parsed) {
|
|
96
|
+
const attributes = parsed.attributes;
|
|
97
|
+
if (!Array.isArray(attributes)) {
|
|
98
|
+
return;
|
|
99
|
+
}
|
|
100
|
+
|
|
101
|
+
const items = attributes[1];
|
|
102
|
+
if (Array.isArray(items)) {
|
|
103
|
+
// inside the item list every list that follows PREVIEW is a modifier list
|
|
104
|
+
for (let i = items.length - 2; i >= 0; i--) {
|
|
105
|
+
if (isAtom(items[i], 'PREVIEW') && Array.isArray(items[i + 1])) {
|
|
106
|
+
items[i] = withModifiers(items[i], items[i + 1]);
|
|
107
|
+
items.splice(i + 1, 1);
|
|
108
|
+
}
|
|
109
|
+
}
|
|
110
|
+
return;
|
|
111
|
+
}
|
|
112
|
+
|
|
113
|
+
// A single PREVIEW item may be followed by both its modifiers and the FETCH modifiers of other extensions
|
|
114
|
+
// (RFC 7162 CHANGEDSINCE), so "FETCH 1 PREVIEW (X)" is a preview modifier only if it looks like one or if
|
|
115
|
+
// another list follows
|
|
116
|
+
if (isAtom(items, 'PREVIEW') && Array.isArray(attributes[2]) && (attributes.length > 3 || isModifierList(attributes[2]))) {
|
|
117
|
+
attributes[1] = withModifiers(items, attributes[2]);
|
|
118
|
+
attributes.splice(2, 1);
|
|
119
|
+
}
|
|
120
|
+
}
|
|
121
|
+
|
|
122
|
+
/**
|
|
123
|
+
* Generates the preview text of a message (RFC 8970 3.3) from the first text/plain or text/html part
|
|
124
|
+
*
|
|
125
|
+
* @param {Object} tree Root node of the MIME tree
|
|
126
|
+
* @return {String} Preview, an empty string if there is no text to show
|
|
127
|
+
*/
|
|
128
|
+
function generatePreview(tree) {
|
|
129
|
+
const part = findTextPart(tree);
|
|
130
|
+
if (!part) {
|
|
131
|
+
// RFC 8970 3.2: the server MUST return an empty string when there is no meaningful preview
|
|
132
|
+
return '';
|
|
133
|
+
}
|
|
134
|
+
|
|
135
|
+
const text = decodeBody(part);
|
|
136
|
+
if (text === false) {
|
|
137
|
+
return '';
|
|
138
|
+
}
|
|
139
|
+
|
|
140
|
+
return finishPreview(part.parsedHeader['content-type'].subtype === 'html' ? htmlToText(text) : plainToText(text), PREVIEW_LENGTH);
|
|
141
|
+
}
|
|
142
|
+
|
|
143
|
+
/**
|
|
144
|
+
* Finds the part to build the preview from. Attachments, attached messages and encrypted content are skipped.
|
|
145
|
+
* In a multipart/alternative a text/plain part is preferred over text/html
|
|
146
|
+
*
|
|
147
|
+
* @param {Object} node Tree node
|
|
148
|
+
* @return {Object|Boolean} Part node or false
|
|
149
|
+
*/
|
|
150
|
+
function findTextPart(node) {
|
|
151
|
+
// the parser fills in the RFC 2045 5.2 and RFC 2046 5.1.5 default content types
|
|
152
|
+
const { type, subtype } = node.parsedHeader['content-type'];
|
|
153
|
+
const disposition = node.parsedHeader['content-disposition'];
|
|
154
|
+
|
|
155
|
+
if (disposition && disposition.type === 'attachment') {
|
|
156
|
+
return false;
|
|
157
|
+
}
|
|
158
|
+
|
|
159
|
+
if (type === 'multipart') {
|
|
160
|
+
if (subtype === 'encrypted') {
|
|
161
|
+
return false;
|
|
162
|
+
}
|
|
163
|
+
const parts = partsOf(node) || [];
|
|
164
|
+
if (subtype === 'alternative') {
|
|
165
|
+
const candidates = parts.map(findTextPart).filter(Boolean);
|
|
166
|
+
return candidates.find(part => part.parsedHeader['content-type'].subtype === 'plain') || candidates[0] || false;
|
|
167
|
+
}
|
|
168
|
+
for (const part of parts) {
|
|
169
|
+
const found = findTextPart(part);
|
|
170
|
+
if (found) {
|
|
171
|
+
return found;
|
|
172
|
+
}
|
|
173
|
+
}
|
|
174
|
+
return false;
|
|
175
|
+
}
|
|
176
|
+
|
|
177
|
+
return type === 'text' && (subtype === 'plain' || subtype === 'html') ? node : false;
|
|
178
|
+
}
|
|
179
|
+
|
|
180
|
+
/**
|
|
181
|
+
* Decodes the content transfer encoding and the charset of the start of a part
|
|
182
|
+
*
|
|
183
|
+
* @param {Object} node Part node
|
|
184
|
+
* @return {String|Boolean} Unicode text, or false for an unknown transfer encoding
|
|
185
|
+
*/
|
|
186
|
+
function decodeBody(node) {
|
|
187
|
+
const encoding = String(node.parsedHeader['content-transfer-encoding'] || '7bit').toLowerCase();
|
|
188
|
+
const body = node.body;
|
|
189
|
+
let octets;
|
|
190
|
+
|
|
191
|
+
switch (encoding) {
|
|
192
|
+
case '7bit':
|
|
193
|
+
case '8bit':
|
|
194
|
+
case 'binary':
|
|
195
|
+
octets = Buffer.from(body.slice(0, MAX_INPUT), 'binary');
|
|
196
|
+
break;
|
|
197
|
+
case 'base64':
|
|
198
|
+
// Buffer skips characters outside the base64 alphabet, the cut keeps whole 4 character groups
|
|
199
|
+
octets = Buffer.from(body.slice(0, (MAX_INPUT * 4) / 3), 'base64');
|
|
200
|
+
break;
|
|
201
|
+
case 'quoted-printable':
|
|
202
|
+
octets = decodeQuotedPrintable(body.slice(0, MAX_INPUT));
|
|
203
|
+
break;
|
|
204
|
+
default:
|
|
205
|
+
return false;
|
|
206
|
+
}
|
|
207
|
+
|
|
208
|
+
const charset = String(node.parsedHeader['content-type'].params.charset || '')
|
|
209
|
+
.trim()
|
|
210
|
+
.toLowerCase();
|
|
211
|
+
|
|
212
|
+
// Like Dovecot, a missing, US-ASCII or unsupported charset is decoded as UTF-8, invalid octets become U+FFFD
|
|
213
|
+
let decoder;
|
|
214
|
+
if (charset && !['us-ascii', 'ascii'].includes(charset)) {
|
|
215
|
+
try {
|
|
216
|
+
decoder = new TextDecoder(charset);
|
|
217
|
+
} catch {
|
|
218
|
+
// unknown charset label
|
|
219
|
+
}
|
|
220
|
+
}
|
|
221
|
+
return (decoder || new TextDecoder('utf-8')).decode(octets);
|
|
222
|
+
}
|
|
223
|
+
|
|
224
|
+
/**
|
|
225
|
+
* Decodes quoted-printable content (RFC 2045 6.7). Malformed escapes are kept as they are
|
|
226
|
+
*
|
|
227
|
+
* @param {String} body Binary string
|
|
228
|
+
* @return {Buffer} Decoded octets
|
|
229
|
+
*/
|
|
230
|
+
function decodeQuotedPrintable(body) {
|
|
231
|
+
// trailing whitespace of encoded lines is padding (RFC 2045 6.7 rule 3). Trimmed with a loop, a regex would
|
|
232
|
+
// backtrack over long runs of whitespace. Line breaks are CRLF after getMessageData()
|
|
233
|
+
const decoded = body
|
|
234
|
+
.split('\r\n')
|
|
235
|
+
.map(line => {
|
|
236
|
+
let end = line.length;
|
|
237
|
+
while (end > 0 && (line[end - 1] === ' ' || line[end - 1] === '\t')) {
|
|
238
|
+
end--;
|
|
239
|
+
}
|
|
240
|
+
return line.slice(0, end);
|
|
241
|
+
})
|
|
242
|
+
.join('\r\n')
|
|
243
|
+
.replace(/=\r\n/g, '')
|
|
244
|
+
.replace(/=([0-9A-Fa-f]{2})/g, (match, hex) => String.fromCharCode(parseInt(hex, 16)));
|
|
245
|
+
return Buffer.from(decoded, 'binary');
|
|
246
|
+
}
|
|
247
|
+
|
|
248
|
+
/**
|
|
249
|
+
* Plain text without quoted lines. Like Dovecot, lines that start with ">" are left out unless nothing else is left
|
|
250
|
+
*
|
|
251
|
+
* @param {String} text Decoded text
|
|
252
|
+
* @return {String} Text
|
|
253
|
+
*/
|
|
254
|
+
function plainToText(text) {
|
|
255
|
+
const unquoted = text
|
|
256
|
+
.split(/\r?\n/)
|
|
257
|
+
.filter(line => !/^>/.test(line))
|
|
258
|
+
.join('\n');
|
|
259
|
+
return unquoted.trim() ? unquoted : text;
|
|
260
|
+
}
|
|
261
|
+
|
|
262
|
+
// Inline elements are removed without a trace, every other tag separates words like a space
|
|
263
|
+
const INLINE_ELEMENTS = new Set([
|
|
264
|
+
'a',
|
|
265
|
+
'abbr',
|
|
266
|
+
'b',
|
|
267
|
+
'bdi',
|
|
268
|
+
'bdo',
|
|
269
|
+
'big',
|
|
270
|
+
'cite',
|
|
271
|
+
'code',
|
|
272
|
+
'del',
|
|
273
|
+
'dfn',
|
|
274
|
+
'em',
|
|
275
|
+
'font',
|
|
276
|
+
'i',
|
|
277
|
+
'ins',
|
|
278
|
+
'kbd',
|
|
279
|
+
'mark',
|
|
280
|
+
'q',
|
|
281
|
+
's',
|
|
282
|
+
'samp',
|
|
283
|
+
'small',
|
|
284
|
+
'span',
|
|
285
|
+
'strike',
|
|
286
|
+
'strong',
|
|
287
|
+
'sub',
|
|
288
|
+
'sup',
|
|
289
|
+
'time',
|
|
290
|
+
'tt',
|
|
291
|
+
'u',
|
|
292
|
+
'var',
|
|
293
|
+
'wbr'
|
|
294
|
+
]);
|
|
295
|
+
|
|
296
|
+
// HTML 4 character entity references for U+00A0 to U+00FF, in code point order
|
|
297
|
+
const LATIN1_ENTITIES =
|
|
298
|
+
'nbsp iexcl cent pound curren yen brvbar sect uml copy ordf laquo not shy reg macr deg plusmn sup2 sup3 acute micro para ' +
|
|
299
|
+
'middot cedil sup1 ordm raquo frac14 frac12 frac34 iquest Agrave Aacute Acirc Atilde Auml Aring AElig Ccedil Egrave Eacute ' +
|
|
300
|
+
'Ecirc Euml Igrave Iacute Icirc Iuml ETH Ntilde Ograve Oacute Ocirc Otilde Ouml times Oslash Ugrave Uacute Ucirc Uuml Yacute ' +
|
|
301
|
+
'THORN szlig agrave aacute acirc atilde auml aring aelig ccedil egrave eacute ecirc euml igrave iacute icirc iuml eth ntilde ' +
|
|
302
|
+
'ograve oacute ocirc otilde ouml divide oslash ugrave uacute ucirc uuml yacute thorn yuml';
|
|
303
|
+
|
|
304
|
+
const NAMED_ENTITIES = Object.assign(Object.fromEntries(LATIN1_ENTITIES.split(' ').map((name, i) => [name, String.fromCharCode(0xa0 + i)])), {
|
|
305
|
+
amp: '&',
|
|
306
|
+
lt: '<',
|
|
307
|
+
gt: '>',
|
|
308
|
+
quot: '"',
|
|
309
|
+
apos: "'",
|
|
310
|
+
trade: '\u2122',
|
|
311
|
+
hellip: '\u2026',
|
|
312
|
+
mdash: '\u2014',
|
|
313
|
+
ndash: '\u2013',
|
|
314
|
+
lsquo: '\u2018',
|
|
315
|
+
rsquo: '\u2019',
|
|
316
|
+
ldquo: '\u201c',
|
|
317
|
+
rdquo: '\u201d',
|
|
318
|
+
bull: '\u2022',
|
|
319
|
+
euro: '\u20ac',
|
|
320
|
+
zwnj: '\u200c',
|
|
321
|
+
zwj: '\u200d'
|
|
322
|
+
});
|
|
323
|
+
|
|
324
|
+
/**
|
|
325
|
+
* Converts HTML into plain text: markup, comments and non-rendered elements (head, script, style) are removed,
|
|
326
|
+
* character references are decoded. Like quoted lines of plain text, blockquote elements are left out unless
|
|
327
|
+
* nothing else is left
|
|
328
|
+
*
|
|
329
|
+
* @param {String} html Decoded HTML
|
|
330
|
+
* @return {String} Text
|
|
331
|
+
*/
|
|
332
|
+
function htmlToText(html) {
|
|
333
|
+
html = html.replace(/<!--[\s\S]*?(?:-->|$)/g, ' ').replace(/<(head|script|style|title|template)\b[^<>]*>[\s\S]*?(?:<\/\1\s*>|$)/gi, ' ');
|
|
334
|
+
|
|
335
|
+
const text = markupToText(removeBlockquotes(html));
|
|
336
|
+
return text.trim() ? text : markupToText(html);
|
|
337
|
+
}
|
|
338
|
+
|
|
339
|
+
/**
|
|
340
|
+
* Removes blockquote elements, nested ones included, in a single pass
|
|
341
|
+
*
|
|
342
|
+
* @param {String} html HTML
|
|
343
|
+
* @return {String} HTML without blockquotes
|
|
344
|
+
*/
|
|
345
|
+
function removeBlockquotes(html) {
|
|
346
|
+
const re = /<(\/?)blockquote\b[^<>]*>?/gi;
|
|
347
|
+
let output = '';
|
|
348
|
+
let depth = 0;
|
|
349
|
+
let pos = 0;
|
|
350
|
+
let match;
|
|
351
|
+
while ((match = re.exec(html))) {
|
|
352
|
+
if (!depth) {
|
|
353
|
+
output += html.slice(pos, match.index) + ' ';
|
|
354
|
+
}
|
|
355
|
+
depth = match[1] ? Math.max(depth - 1, 0) : depth + 1;
|
|
356
|
+
pos = re.lastIndex;
|
|
357
|
+
}
|
|
358
|
+
return depth ? output : output + html.slice(pos);
|
|
359
|
+
}
|
|
360
|
+
|
|
361
|
+
/**
|
|
362
|
+
* Removes tags and decodes character references
|
|
363
|
+
*
|
|
364
|
+
* @param {String} html HTML without comments and non-rendered elements
|
|
365
|
+
* @return {String} Text
|
|
366
|
+
*/
|
|
367
|
+
function markupToText(html) {
|
|
368
|
+
return html
|
|
369
|
+
.replace(/<!\[CDATA\[([\s\S]*?)(?:\]\]>|$)/g, '$1')
|
|
370
|
+
.replace(/<\/?([A-Za-z][A-Za-z0-9-]*)(?![A-Za-z0-9-])[^<>]*>|<[!?][^<>]*>/g, (match, name) =>
|
|
371
|
+
name && INLINE_ELEMENTS.has(name.toLowerCase()) ? '' : ' '
|
|
372
|
+
)
|
|
373
|
+
.replace(/&(#[0-9]+|#[xX][0-9A-Fa-f]+|[A-Za-z][A-Za-z0-9]*);/g, (match, ref) => {
|
|
374
|
+
if (ref[0] === '#') {
|
|
375
|
+
const code = /^#x/i.test(ref) ? parseInt(ref.slice(2), 16) : parseInt(ref.slice(1), 10);
|
|
376
|
+
// NUL, surrogates and values outside Unicode are not characters
|
|
377
|
+
return code > 0 && code <= 0x10ffff && (code < 0xd800 || code > 0xdfff) ? String.fromCodePoint(code) : '\ufffd';
|
|
378
|
+
}
|
|
379
|
+
return Object.hasOwn(NAMED_ENTITIES, ref) ? NAMED_ENTITIES[ref] : match;
|
|
380
|
+
});
|
|
381
|
+
}
|
|
382
|
+
|
|
383
|
+
/**
|
|
384
|
+
* Normalizes preview text: control characters are removed, whitespace runs become a single space (so there are no
|
|
385
|
+
* CR or LF characters) and the text is cut to at most `length` characters (code points, RFC 8970 3.3)
|
|
386
|
+
*
|
|
387
|
+
* @param {String} text Text
|
|
388
|
+
* @param {Number} length Maximum number of characters
|
|
389
|
+
* @return {String} Preview
|
|
390
|
+
*/
|
|
391
|
+
function finishPreview(text, length) {
|
|
392
|
+
const normalized = text
|
|
393
|
+
.replace(/[^\P{Cc}\t\n\v\f\r]|\u00ad/gu, '')
|
|
394
|
+
.replace(/[\s\u200b]+/g, ' ')
|
|
395
|
+
.trim();
|
|
396
|
+
return Array.from(normalized).slice(0, length).join('').trimEnd();
|
|
397
|
+
}
|
|
398
|
+
|
|
399
|
+
module.exports.generatePreview = generatePreview;
|
|
400
|
+
module.exports.finishPreview = finishPreview;
|