imapkit 4.0.1 → 4.0.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -4,6 +4,7 @@ const { isUtf8 } = require('buffer');
4
4
  const { getMessageData, render } = require('../../mimeparser');
5
5
  const { monthIndex, dateKey, parseDateTime, parseHeaderDate } = require('../../dates');
6
6
  const { MAX_NUMBER, MAX_NUMBER64, isNumber } = require('../../numbers');
7
+ const { decodeHeader, decodeUtf8 } = require('../../encoded-words');
7
8
 
8
9
  // RFC 3501 6.4.4 search keys and their arguments
9
10
  const searchKeys = {
@@ -142,7 +143,8 @@ function sendSearchError(connection, parsed, data, err, description) {
142
143
  }
143
144
 
144
145
  /**
145
- * Lower cases ASCII letters only, so that 8-bit octets in binary strings stay intact
146
+ * Lower cases ASCII letters only, so that 8-bit octets in binary strings stay intact. RFC 9051
147
+ * section 6.4.4 asks for case insensitive matching only within the ASCII range
146
148
  */
147
149
  function asciiLowerCase(str) {
148
150
  return str.replace(/[A-Z]+/g, chars => chars.toLowerCase());
@@ -371,11 +373,13 @@ module.exports = function (connection, messageSource, params, getMessageRange) {
371
373
  if (node.key === '_SEQ' || node.key === 'UID') {
372
374
  node.set = toMessageSet(node.args[0]);
373
375
  }
374
- // lower case the string to look for once, not for every message
375
- if (['BCC', 'BODY', 'CC', 'FROM', 'SUBJECT', 'TEXT', 'TO'].indexOf(node.key) >= 0) {
376
+ // lower case the string to look for once, not for every message. Header values are compared
377
+ // as Unicode text after decoding their encoded words, so the UTF-8 octets of the string are
378
+ // decoded too (RFC 9051 section 6.4.4)
379
+ if (['BODY', 'TEXT'].indexOf(node.key) >= 0) {
376
380
  node.needle = asciiLowerCase(node.args[0]);
377
- } else if (node.key === 'HEADER') {
378
- node.needle = asciiLowerCase(node.args[1]);
381
+ } else if (['BCC', 'CC', 'FROM', 'HEADER', 'SUBJECT', 'TO'].indexOf(node.key) >= 0) {
382
+ node.needle = asciiLowerCase(decodeUtf8(node.args[node.key === 'HEADER' ? 1 : 0]));
379
383
  }
380
384
  node.args.forEach(arg => {
381
385
  if (arg && typeof arg === 'object' && arg.key) {
@@ -387,16 +391,16 @@ module.exports = function (connection, messageSource, params, getMessageRange) {
387
391
 
388
392
  const hasFlag = (message, flag) => message.flags.indexOf(flag) >= 0;
389
393
 
390
- // header lines as [lowercase name, unfolded value]
391
- const getHeaders = message =>
392
- (getMessageData(message).tree.header || []).map(line => {
393
- const parts = line.split(':');
394
- return [(parts.shift() || '').trim().toLowerCase(), parts.join(':').replace(/\r?\n(?=[ \t])/g, '')];
395
- });
396
-
394
+ // compares the unfolded values of the header lines with that name, after decoding their encoded
395
+ // words. RFC 3501 and RFC 9051 section 6.4.4: [MIME-HDRS] strings in headers MUST be decoded before comparing text
397
396
  const matchHeader = (message, name, needle) => {
398
397
  name = name.toLowerCase();
399
- return getHeaders(message).some(header => header[0] === name && contains(header[1], needle));
398
+ return (getMessageData(message).tree.header || []).some(line => {
399
+ const colon = line.indexOf(':');
400
+ const lineName = (colon < 0 ? line : line.substr(0, colon)).trim().toLowerCase();
401
+ const value = colon < 0 ? '' : line.substr(colon + 1).replace(/\r?\n(?=[ \t])/g, '');
402
+ return lineName === name && contains(decodeHeader(value), needle);
403
+ });
400
404
  };
401
405
 
402
406
  const matches = (node, message, index) => {
@@ -0,0 +1,98 @@
1
+ 'use strict';
2
+
3
+ // RFC 2047 encoded words in header values, decoded for SEARCH, SORT and THREAD
4
+
5
+ // RFC 2047 section 2: encoded-word = "=?" charset "?" encoding "?" encoded-text "?=", RFC 2231 section 5
6
+ // adds an optional "*" language suffix to the charset
7
+ const ENCODED_WORD = /=\?([^?\s*]+)(?:\*[^?\s]*)?\?([BbQq])\?([^?\s]*)\?=/g;
8
+
9
+ const utf8Decoder = new TextDecoder('utf-8');
10
+
11
+ /**
12
+ * Reads a binary string (one char per octet) as UTF-8, invalid sequences become U+FFFD
13
+ */
14
+ function decodeUtf8(value) {
15
+ return utf8Decoder.decode(Buffer.from(value, 'binary'));
16
+ }
17
+
18
+ // decoders by lower case charset name, false for a charset that TextDecoder does not know
19
+ const decoders = new Map();
20
+
21
+ /**
22
+ * Decodes the octets of an encoded word, or returns false if the charset is unknown
23
+ */
24
+ function decodeCharset(charset, octets) {
25
+ charset = charset.toLowerCase();
26
+ if (!decoders.has(charset)) {
27
+ let decoder = false;
28
+ try {
29
+ decoder = new TextDecoder(charset);
30
+ } catch {
31
+ // unknown charset
32
+ }
33
+ decoders.set(charset, decoder);
34
+ }
35
+ const decoder = decoders.get(charset);
36
+ return decoder && decoder.decode(octets);
37
+ }
38
+
39
+ /**
40
+ * Decodes an RFC 2047 header value to a Unicode string. Text outside encoded words is read as UTF-8
41
+ * (invalid sequences become U+FFFD). Adjacent encoded words in the same charset are decoded together,
42
+ * so a multi-octet character may span them, and the white space between them is dropped (RFC 2047
43
+ * section 6.2). Encoded words in an unknown charset are kept as they are.
44
+ *
45
+ * @param {String} value Header value as a binary string
46
+ * @return {String} Decoded value
47
+ */
48
+ function decodeHeader(value) {
49
+ value = (value || '').toString();
50
+ if (value.indexOf('=?') < 0 && !/[\u0080-\u00ff]/.test(value)) {
51
+ // nothing to decode
52
+ return value;
53
+ }
54
+ let result = '';
55
+ let pending = null;
56
+ let lastIndex = 0;
57
+
58
+ const flush = () => {
59
+ if (pending) {
60
+ const decoded = decodeCharset(pending.charset, Buffer.concat(pending.octets));
61
+ result += decoded === false ? pending.source : decoded;
62
+ pending = null;
63
+ }
64
+ };
65
+
66
+ ENCODED_WORD.lastIndex = 0;
67
+ let match;
68
+ while ((match = ENCODED_WORD.exec(value))) {
69
+ const between = value.substring(lastIndex, match.index);
70
+ const adjacent = pending && /^\s*$/.test(between);
71
+ if (!adjacent) {
72
+ flush();
73
+ result += decodeUtf8(between);
74
+ }
75
+ lastIndex = ENCODED_WORD.lastIndex;
76
+
77
+ const octets =
78
+ match[2].toUpperCase() === 'B'
79
+ ? Buffer.from(match[3], 'base64')
80
+ : Buffer.from(
81
+ match[3].replace(/_/g, ' ').replace(/=([0-9a-fA-F]{2})/g, (m, hex) => String.fromCharCode(parseInt(hex, 16))),
82
+ 'binary'
83
+ );
84
+
85
+ if (pending && pending.charset.toLowerCase() !== match[1].toLowerCase()) {
86
+ flush();
87
+ }
88
+ if (!pending) {
89
+ pending = { charset: match[1], octets: [], source: '' };
90
+ }
91
+ pending.octets.push(octets);
92
+ pending.source += (pending.source ? between : '') + match[0];
93
+ }
94
+ flush();
95
+ return result + decodeUtf8(value.substr(lastIndex));
96
+ }
97
+
98
+ module.exports = { decodeHeader, decodeUtf8 };
package/lib/sorting.js CHANGED
@@ -8,88 +8,7 @@ const { processAddress } = require('./envelope');
8
8
  const { parseDateTime, parseHeaderDate, toTimestamp } = require('./dates');
9
9
  const makeSearch = require('./commands/handlers/search');
10
10
  const { badError, criteriaValues, sendSearchError } = makeSearch;
11
-
12
- // RFC 2047 section 2: encoded-word = "=?" charset "?" encoding "?" encoded-text "?=", RFC 2231 section 5
13
- // adds an optional "*" language suffix to the charset
14
- const ENCODED_WORD = /=\?([^?\s*]+)(?:\*[^?\s]*)?\?([BbQq])\?([^?\s]*)\?=/g;
15
-
16
- const utf8Decoder = new TextDecoder('utf-8');
17
-
18
- // decoders by lower case charset name, false for a charset that TextDecoder does not know
19
- const decoders = new Map();
20
-
21
- /**
22
- * Decodes the octets of an encoded word, or returns false if the charset is unknown
23
- */
24
- function decodeCharset(charset, octets) {
25
- charset = charset.toLowerCase();
26
- if (!decoders.has(charset)) {
27
- let decoder = false;
28
- try {
29
- decoder = new TextDecoder(charset);
30
- } catch {
31
- // unknown charset
32
- }
33
- decoders.set(charset, decoder);
34
- }
35
- const decoder = decoders.get(charset);
36
- return decoder && decoder.decode(octets);
37
- }
38
-
39
- /**
40
- * Decodes an RFC 2047 header value to a Unicode string. Text outside encoded words is read as UTF-8
41
- * (invalid sequences become U+FFFD). Adjacent encoded words in the same charset are decoded together,
42
- * so a multi-octet character may span them, and the white space between them is dropped (RFC 2047
43
- * section 6.2). Encoded words in an unknown charset are kept as they are.
44
- *
45
- * @param {String} value Header value as a binary string
46
- * @return {String} Decoded value
47
- */
48
- function decodeHeader(value) {
49
- value = (value || '').toString();
50
- let result = '';
51
- let pending = null;
52
- let lastIndex = 0;
53
-
54
- const flush = () => {
55
- if (pending) {
56
- const decoded = decodeCharset(pending.charset, Buffer.concat(pending.octets));
57
- result += decoded === false ? pending.source : decoded;
58
- pending = null;
59
- }
60
- };
61
-
62
- ENCODED_WORD.lastIndex = 0;
63
- let match;
64
- while ((match = ENCODED_WORD.exec(value))) {
65
- const between = value.substring(lastIndex, match.index);
66
- const adjacent = pending && /^\s*$/.test(between);
67
- if (!adjacent) {
68
- flush();
69
- result += utf8Decoder.decode(Buffer.from(between, 'binary'));
70
- }
71
- lastIndex = ENCODED_WORD.lastIndex;
72
-
73
- const octets =
74
- match[2].toUpperCase() === 'B'
75
- ? Buffer.from(match[3], 'base64')
76
- : Buffer.from(
77
- match[3].replace(/_/g, ' ').replace(/=([0-9a-fA-F]{2})/g, (m, hex) => String.fromCharCode(parseInt(hex, 16))),
78
- 'binary'
79
- );
80
-
81
- if (pending && pending.charset.toLowerCase() !== match[1].toLowerCase()) {
82
- flush();
83
- }
84
- if (!pending) {
85
- pending = { charset: match[1], octets: [], source: '' };
86
- }
87
- pending.octets.push(octets);
88
- pending.source += (pending.source ? between : '') + match[0];
89
- }
90
- flush();
91
- return result + utf8Decoder.decode(Buffer.from(value.substr(lastIndex), 'binary'));
92
- }
11
+ const { decodeHeader } = require('./encoded-words');
93
12
 
94
13
  // UnicodeData.txt titlecase mappings that differ from the single code point uppercase mapping
95
14
  // that String#toUpperCase gives: the Latin digraphs and the Greek letters with ypogegrammeni
@@ -361,7 +280,6 @@ function searchMessages(connection, parsed, data, attributes) {
361
280
  }
362
281
 
363
282
  module.exports = {
364
- decodeHeader,
365
283
  collationKey,
366
284
  baseSubject,
367
285
  arrivalTime,
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "imapkit",
3
- "version": "4.0.1",
3
+ "version": "4.0.2",
4
4
  "description": "Scriptable, strictly RFC compliant in-memory IMAP server for testing IMAP clients",
5
5
  "main": "lib/server.js",
6
6
  "bin": {
@@ -46,7 +46,7 @@
46
46
  "author": "Postal Systems OÜ",
47
47
  "license": "MIT",
48
48
  "dependencies": {
49
- "imap-handler": "1.3.1",
49
+ "imap-handler": "1.3.2",
50
50
  "smtp-server": "3.19.17"
51
51
  },
52
52
  "devDependencies": {