@email-utils/validator-syntax 1.0.0-rc.1 → 1.0.0-rc.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/fixtures.cjs +125 -8
- package/dist/fixtures.cjs.map +1 -1
- package/dist/fixtures.d.cts +133 -7
- package/dist/fixtures.d.mts +133 -7
- package/dist/fixtures.mjs +125 -8
- package/dist/fixtures.mjs.map +1 -1
- package/dist/index.cjs +237 -68
- package/dist/index.cjs.map +1 -1
- package/dist/index.d.cts +24 -15
- package/dist/index.d.mts +24 -15
- package/dist/index.mjs +237 -68
- package/dist/index.mjs.map +1 -1
- package/dist/{result-BZjmlevS.d.cts → result-CNbJ54oT.d.cts} +17 -1
- package/dist/{result-BZjmlevS.d.mts → result-CNbJ54oT.d.mts} +17 -1
- package/package.json +7 -3
package/dist/index.mjs
CHANGED
|
@@ -1,6 +1,18 @@
|
|
|
1
|
+
//#region src/chars.ts
|
|
2
|
+
const charCodeAt = String.prototype.charCodeAt;
|
|
3
|
+
/**
|
|
4
|
+
* `text.charCodeAt(i)`, through the builtin itself. Called as a method,
|
|
5
|
+
* `charCodeAt` is looked up on the string, and a call site that has seen
|
|
6
|
+
* enough kinds of string (one-byte and two-byte, flat, sliced, and
|
|
7
|
+
* concatenated) goes megamorphic: V8 stops inlining it, and after arbitrary
|
|
8
|
+
* Unicode input every scan ran 2–3× slower (validator-syntax#15).
|
|
9
|
+
*/
|
|
10
|
+
function codeAt(text, i) {
|
|
11
|
+
return charCodeAt.call(text, i);
|
|
12
|
+
}
|
|
1
13
|
const table = /* @__PURE__ */ new Uint8Array(128);
|
|
2
14
|
function mark(chars, flag) {
|
|
3
|
-
for (let i = 0; i < chars.length; i++) table[chars
|
|
15
|
+
for (let i = 0; i < chars.length; i++) table[codeAt(chars, i)] |= flag;
|
|
4
16
|
}
|
|
5
17
|
const alnum = "abcdefghijklmnopqrstuvwxyzABCDEFGHIJKLMNOPQRSTUVWXYZ0123456789";
|
|
6
18
|
mark(alnum + "!#$%&'*+-/=?^_`{|}~", 1);
|
|
@@ -21,22 +33,35 @@ function isWhitespace(code) {
|
|
|
21
33
|
* lone surrogate, which no UTF-8 text can hold.
|
|
22
34
|
*/
|
|
23
35
|
function nonAsciiAt(text, i, end) {
|
|
24
|
-
const code = text
|
|
36
|
+
const code = codeAt(text, i);
|
|
25
37
|
if (code < 128 || code >= 56320 && code <= 57343) return 0;
|
|
26
38
|
if (code < 55296 || code > 56319) return 1;
|
|
27
|
-
const low = i + 1 < end ? text
|
|
39
|
+
const low = i + 1 < end ? codeAt(text, i + 1) : 0;
|
|
28
40
|
return low >= 56320 && low <= 57343 ? 2 : 0;
|
|
29
41
|
}
|
|
30
42
|
/** The length of `text` in UTF-8 octets, the unit RFC 6531 caps a local part in. */
|
|
31
43
|
function utf8Length(text) {
|
|
32
|
-
|
|
33
|
-
|
|
34
|
-
|
|
44
|
+
const end = text.length;
|
|
45
|
+
let octets = end;
|
|
46
|
+
for (let i = 0; i < end; i++) {
|
|
47
|
+
const code = codeAt(text, i);
|
|
35
48
|
if (code >= 2048 && (code < 55296 || code > 57343)) octets += 2;
|
|
36
49
|
else if (code >= 128) octets += 1;
|
|
37
50
|
}
|
|
38
51
|
return octets;
|
|
39
52
|
}
|
|
53
|
+
/**
|
|
54
|
+
* The code points in `text` from `start` to `end`, which must not split a
|
|
55
|
+
* surrogate pair: its length with each pair counted once.
|
|
56
|
+
*/
|
|
57
|
+
function codePoints(text, start, end) {
|
|
58
|
+
let count = end - start;
|
|
59
|
+
for (let i = start; i < end; i++) {
|
|
60
|
+
const code = codeAt(text, i);
|
|
61
|
+
if (code >= 56320 && code <= 57343) count--;
|
|
62
|
+
}
|
|
63
|
+
return count;
|
|
64
|
+
}
|
|
40
65
|
//#endregion
|
|
41
66
|
//#region src/options.ts
|
|
42
67
|
const presets = {
|
|
@@ -52,7 +77,8 @@ const presets = {
|
|
|
52
77
|
checkTld: true,
|
|
53
78
|
allowNoTld: false,
|
|
54
79
|
unicode: false,
|
|
55
|
-
idn: false
|
|
80
|
+
idn: false,
|
|
81
|
+
maxLength: 512
|
|
56
82
|
},
|
|
57
83
|
rfc5321: {
|
|
58
84
|
local: 1,
|
|
@@ -66,7 +92,8 @@ const presets = {
|
|
|
66
92
|
checkTld: false,
|
|
67
93
|
allowNoTld: false,
|
|
68
94
|
unicode: false,
|
|
69
|
-
idn: false
|
|
95
|
+
idn: false,
|
|
96
|
+
maxLength: 512
|
|
70
97
|
},
|
|
71
98
|
rfc5322: {
|
|
72
99
|
local: 1,
|
|
@@ -80,7 +107,8 @@ const presets = {
|
|
|
80
107
|
checkTld: false,
|
|
81
108
|
allowNoTld: false,
|
|
82
109
|
unicode: false,
|
|
83
|
-
idn: false
|
|
110
|
+
idn: false,
|
|
111
|
+
maxLength: 512
|
|
84
112
|
},
|
|
85
113
|
html5: {
|
|
86
114
|
local: 4,
|
|
@@ -94,7 +122,8 @@ const presets = {
|
|
|
94
122
|
checkTld: false,
|
|
95
123
|
allowNoTld: true,
|
|
96
124
|
unicode: false,
|
|
97
|
-
idn: false
|
|
125
|
+
idn: false,
|
|
126
|
+
maxLength: 512
|
|
98
127
|
}
|
|
99
128
|
};
|
|
100
129
|
const overrides = [
|
|
@@ -120,13 +149,14 @@ const unsupported = {
|
|
|
120
149
|
* Resolves `options` into the rules they select.
|
|
121
150
|
*
|
|
122
151
|
* @throws TypeError when `options` isn't an object, names an unknown option
|
|
123
|
-
* or preset, gives a non-boolean override
|
|
124
|
-
* preset's grammar
|
|
152
|
+
* or preset, gives a non-boolean override or a `maxLength` that isn't a
|
|
153
|
+
* positive integer or `Infinity`, or turns on something the preset's grammar
|
|
154
|
+
* has no room for.
|
|
125
155
|
*/
|
|
126
156
|
function resolve(options) {
|
|
127
157
|
if (options === void 0) return presets.practical;
|
|
128
158
|
if (typeof options !== "object" || options === null) throw new TypeError("Expected the options to be an object");
|
|
129
|
-
for (const key of Object.keys(options)) if (key !== "preset" && !overrides.includes(key)) throw new TypeError(`Unknown option: ${key}`);
|
|
159
|
+
for (const key of Object.keys(options)) if (key !== "preset" && key !== "maxLength" && !overrides.includes(key)) throw new TypeError(`Unknown option: ${key}`);
|
|
130
160
|
const preset = options.preset ?? "practical";
|
|
131
161
|
if (!Object.hasOwn(presets, preset)) throw new TypeError(`Unknown preset: ${preset}`);
|
|
132
162
|
const base = presets[preset];
|
|
@@ -134,8 +164,10 @@ function resolve(options) {
|
|
|
134
164
|
const value = options[key];
|
|
135
165
|
if (value !== void 0 && typeof value !== "boolean") throw new TypeError(`Expected ${key} to be a boolean`);
|
|
136
166
|
}
|
|
167
|
+
const { maxLength } = options;
|
|
168
|
+
if (maxLength !== void 0 && maxLength !== Infinity && !(Number.isInteger(maxLength) && maxLength > 0)) throw new TypeError("Expected maxLength to be a positive integer or Infinity");
|
|
137
169
|
for (const key of unsupported[preset]) if (options[key] === true) throw new TypeError(`${key} can’t be true with the ${preset} preset`);
|
|
138
|
-
if (overrides.every((key) => options[key] === void 0)) return base;
|
|
170
|
+
if (maxLength === void 0 && overrides.every((key) => options[key] === void 0)) return base;
|
|
139
171
|
return {
|
|
140
172
|
...base,
|
|
141
173
|
checkTld: options.checkTld ?? base.checkTld,
|
|
@@ -143,13 +175,27 @@ function resolve(options) {
|
|
|
143
175
|
comments: options.allowComments ?? base.comments,
|
|
144
176
|
unicode: options.allowUnicode ?? base.unicode,
|
|
145
177
|
idn: options.allowIdn ?? base.idn,
|
|
146
|
-
literals: options.allowIpLiteral ?? base.literals
|
|
178
|
+
literals: options.allowIpLiteral ?? base.literals,
|
|
179
|
+
maxLength: maxLength ?? base.maxLength
|
|
147
180
|
};
|
|
148
181
|
}
|
|
149
182
|
//#endregion
|
|
150
183
|
//#region src/idn.ts
|
|
151
184
|
const ldh = /^[\da-z-]+$/;
|
|
152
185
|
/**
|
|
186
|
+
* The hostname of `http://x.` and `host`, or `undefined` when the WHATWG URL
|
|
187
|
+
* parser rejects it. `URL.parse` says so without the cost of throwing, a few
|
|
188
|
+
* µs a time, so an address built to fail many conversions stays fast;
|
|
189
|
+
* runtimes that predate it (Node before 22.1) throw instead.
|
|
190
|
+
*/
|
|
191
|
+
const hostnameOf = typeof URL.parse === "function" ? (host) => URL.parse(`http://x.${host}`)?.hostname : (host) => {
|
|
192
|
+
try {
|
|
193
|
+
return new URL(`http://x.${host}`).hostname;
|
|
194
|
+
} catch {
|
|
195
|
+
return;
|
|
196
|
+
}
|
|
197
|
+
};
|
|
198
|
+
/**
|
|
153
199
|
* The A-label for a U-label, or `undefined` when UTS #46 rejects it or maps
|
|
154
200
|
* it to something other than one IDN label (plain ASCII, or two labels, as
|
|
155
201
|
* `。` becomes a dot).
|
|
@@ -160,14 +206,65 @@ const ldh = /^[\da-z-]+$/;
|
|
|
160
206
|
* checks. It leaves hyphens and lengths alone, so the caller checks those.
|
|
161
207
|
*/
|
|
162
208
|
function toALabel(label) {
|
|
163
|
-
|
|
164
|
-
|
|
165
|
-
|
|
166
|
-
|
|
167
|
-
|
|
209
|
+
const aLabel = hostnameOf(label)?.slice(2);
|
|
210
|
+
return aLabel?.startsWith("xn--") === true && ldh.test(aLabel) ? aLabel : void 0;
|
|
211
|
+
}
|
|
212
|
+
const rtl = /[\u0590-\u08ff\u2135-\u2138\ufb1d-\ufdff\ufe70-\ufeff\u{10800}-\u{10fff}\u{1e800}-\u{1efff}]/u;
|
|
213
|
+
/**
|
|
214
|
+
* The A-label for each U-label in `labels`, as {@link toALabel} gives it,
|
|
215
|
+
* converted as few domains as it can: one URL costs about as much as one
|
|
216
|
+
* label does.
|
|
217
|
+
*
|
|
218
|
+
* @remarks
|
|
219
|
+
* Every entry up to the first `undefined` is exact; entries after it may be
|
|
220
|
+
* left `undefined` unconverted. Converted together, a right-to-left label
|
|
221
|
+
* would hold the others to RFC 5893's rules too, which it doesn't alone, so
|
|
222
|
+
* right-to-left labels go in a domain of their own. Converting more labels
|
|
223
|
+
* at once only adds rules, so a domain that converts means every label
|
|
224
|
+
* would alone; one that doesn't is narrowed down to its first label that
|
|
225
|
+
* fails alone.
|
|
226
|
+
*/
|
|
227
|
+
function toALabels(labels) {
|
|
228
|
+
if (!labels.some((label) => rtl.test(label))) return convertGroup(labels);
|
|
229
|
+
const aLabels = labels.map(() => void 0);
|
|
230
|
+
for (const bidi of [false, true]) {
|
|
231
|
+
const members = [];
|
|
232
|
+
labels.forEach((label, k) => {
|
|
233
|
+
if (rtl.test(label) === bidi) members.push(k);
|
|
234
|
+
});
|
|
235
|
+
if (members.length > 0) convertGroup(members.map((k) => labels[k])).forEach((aLabel, j) => {
|
|
236
|
+
aLabels[members[j]] = aLabel;
|
|
237
|
+
});
|
|
168
238
|
}
|
|
169
|
-
|
|
170
|
-
|
|
239
|
+
return aLabels;
|
|
240
|
+
}
|
|
241
|
+
/** {@link toALabels} for labels converted together. */
|
|
242
|
+
function convertGroup(labels) {
|
|
243
|
+
const all = convert(labels);
|
|
244
|
+
if (all !== void 0) return all;
|
|
245
|
+
let pass = [];
|
|
246
|
+
let failing = labels.length;
|
|
247
|
+
while (failing - pass.length > 1) {
|
|
248
|
+
const some = convert(labels.slice(0, pass.length + failing >> 1));
|
|
249
|
+
if (some === void 0) failing = pass.length + failing >> 1;
|
|
250
|
+
else pass = some;
|
|
251
|
+
}
|
|
252
|
+
const aLabels = pass;
|
|
253
|
+
for (let k = pass.length; k < labels.length; k++) {
|
|
254
|
+
const aLabel = toALabel(labels[k]);
|
|
255
|
+
aLabels.push(aLabel);
|
|
256
|
+
if (aLabel === void 0) break;
|
|
257
|
+
}
|
|
258
|
+
return aLabels;
|
|
259
|
+
}
|
|
260
|
+
const aLabelHost = /^x(?:\.xn--[\da-z-]+)+$/;
|
|
261
|
+
/** `labels` converted as one domain, or `undefined`. */
|
|
262
|
+
function convert(labels) {
|
|
263
|
+
const hostname = hostnameOf(labels.join("."));
|
|
264
|
+
if (hostname === void 0 || !aLabelHost.test(hostname)) return;
|
|
265
|
+
const aLabels = hostname.split(".");
|
|
266
|
+
aLabels.shift();
|
|
267
|
+
return aLabels.length === labels.length ? aLabels : void 0;
|
|
171
268
|
}
|
|
172
269
|
//#endregion
|
|
173
270
|
//#region src/ip.ts
|
|
@@ -175,6 +272,7 @@ const octet = /^\d{1,3}$/;
|
|
|
175
272
|
const hexGroup = /^[\dA-Fa-f]{1,4}$/;
|
|
176
273
|
/** An IPv4 address: four dot-separated decimal octets, each 0–255. */
|
|
177
274
|
function isIPv4(text) {
|
|
275
|
+
if (text.length > 15) return false;
|
|
178
276
|
const parts = text.split(".");
|
|
179
277
|
return parts.length === 4 && parts.every((part) => octet.test(part) && Number(part) <= 255);
|
|
180
278
|
}
|
|
@@ -184,6 +282,7 @@ function isIPv4(text) {
|
|
|
184
282
|
* place of the last two groups.
|
|
185
283
|
*/
|
|
186
284
|
function isIPv6(text) {
|
|
285
|
+
if (text.length > 45) return false;
|
|
187
286
|
const colon = text.lastIndexOf(":");
|
|
188
287
|
const tail = text.slice(colon + 1);
|
|
189
288
|
if (tail.includes(".")) {
|
|
@@ -1667,7 +1766,7 @@ const messages = {
|
|
|
1667
1766
|
"syntax.local.too_long": "The local part is longer than 64 characters",
|
|
1668
1767
|
"syntax.local.invalid_char": "The local part has a character it can’t hold",
|
|
1669
1768
|
"syntax.local.consecutive_dots": "The local part has two dots in a row",
|
|
1670
|
-
"syntax.local.unquoted_space": "The local part has
|
|
1769
|
+
"syntax.local.unquoted_space": "The local part has whitespace outside quotes",
|
|
1671
1770
|
"syntax.domain.empty": "Nothing comes after the @",
|
|
1672
1771
|
"syntax.domain.no_dot": "The domain has no dot",
|
|
1673
1772
|
"syntax.domain.label_invalid": "A domain label is empty, too long, or starts or ends with a hyphen",
|
|
@@ -1710,8 +1809,9 @@ function findAt(email, rules) {
|
|
|
1710
1809
|
let domainStart = false;
|
|
1711
1810
|
let depth = 0;
|
|
1712
1811
|
let at = -1;
|
|
1713
|
-
|
|
1714
|
-
|
|
1812
|
+
const end = email.length;
|
|
1813
|
+
for (let i = 0; i < end; i++) {
|
|
1814
|
+
const code = codeAt(email, i);
|
|
1715
1815
|
if (state === NORMAL) {
|
|
1716
1816
|
if (code === 64) {
|
|
1717
1817
|
at = i;
|
|
@@ -1742,6 +1842,7 @@ function findAt(email, rules) {
|
|
|
1742
1842
|
function parse(email, rules) {
|
|
1743
1843
|
if (typeof email !== "string") throw new TypeError(`Expected a string, got ${typeof email}`);
|
|
1744
1844
|
if (email === "") return fail("syntax.address.empty");
|
|
1845
|
+
if (email.length > rules.maxLength) return fail("syntax.address.too_long");
|
|
1745
1846
|
const at = findAt(email, rules);
|
|
1746
1847
|
if (at < 0) return fail("syntax.address.no_at");
|
|
1747
1848
|
const comments = [];
|
|
@@ -1785,7 +1886,7 @@ function scanLocal(email, at, rules, comments) {
|
|
|
1785
1886
|
let trailing = comments.length;
|
|
1786
1887
|
let i = 0;
|
|
1787
1888
|
while (i < at) {
|
|
1788
|
-
const code = email
|
|
1889
|
+
const code = codeAt(email, i);
|
|
1789
1890
|
const quote = code === 34 && rules.quotes && (rules.obs || i === 0);
|
|
1790
1891
|
const atom = inClass(code, rules.local);
|
|
1791
1892
|
if (quote || atom || rules.unicode && nonAsciiAt(email, i, at) > 0) {
|
|
@@ -1798,10 +1899,10 @@ function scanLocal(email, at, rules, comments) {
|
|
|
1798
1899
|
closed = !rules.obs;
|
|
1799
1900
|
} else {
|
|
1800
1901
|
next = atom ? i + 1 : i;
|
|
1801
|
-
while (next < at && inClass(email
|
|
1902
|
+
while (next < at && inClass(codeAt(email, next), rules.local)) next++;
|
|
1802
1903
|
if (rules.unicode) next = skipUnicode(email, next, at, rules.local);
|
|
1803
1904
|
}
|
|
1804
|
-
local += unfold(email
|
|
1905
|
+
local += unfold(email, i, next);
|
|
1805
1906
|
words++;
|
|
1806
1907
|
afterWord = true;
|
|
1807
1908
|
dot = -1;
|
|
@@ -1846,6 +1947,63 @@ function scanLocal(email, at, rules, comments) {
|
|
|
1846
1947
|
* folding whitespace.
|
|
1847
1948
|
*/
|
|
1848
1949
|
function scanDomain(email, at, rules, comments) {
|
|
1950
|
+
const uLabels = [];
|
|
1951
|
+
const domain = scanLabels(email, at, rules, comments, uLabels);
|
|
1952
|
+
if (uLabels.length === 0) return domain;
|
|
1953
|
+
const growth = checkULabels(email, uLabels, rules.domainCap);
|
|
1954
|
+
if (typeof growth !== "number") return growth;
|
|
1955
|
+
if (!("reason" in domain)) domain.size += growth;
|
|
1956
|
+
return domain;
|
|
1957
|
+
}
|
|
1958
|
+
/**
|
|
1959
|
+
* Checks the U-labels at `spans`, triples of start, end, and the length of
|
|
1960
|
+
* the domain before it, as A-labels: each must convert, and pass the
|
|
1961
|
+
* hostname rules as its A-label. Returns the first failure, or what the
|
|
1962
|
+
* A-labels add to the domain's length: `Infinity` once it's past `cap`.
|
|
1963
|
+
*
|
|
1964
|
+
* @remarks
|
|
1965
|
+
* The U-labels are converted a few at a time, about 32 characters of them
|
|
1966
|
+
* in each URL: one URL costs about as much as one label alone, and a small
|
|
1967
|
+
* batch wastes little past the first failure or past `cap`. Once the
|
|
1968
|
+
* A-labels take the domain past `cap`, the rest aren't checked: the domain
|
|
1969
|
+
* fails as too long, unless the scan found something earlier. Punycode's
|
|
1970
|
+
* cost grows with the square of a label's length, so converting labels that
|
|
1971
|
+
* can't change the outcome would cost the most on input built to be slow
|
|
1972
|
+
* (validator-syntax#15).
|
|
1973
|
+
*/
|
|
1974
|
+
function checkULabels(email, spans, cap) {
|
|
1975
|
+
let growth = 0;
|
|
1976
|
+
for (let first = 0; first < spans.length;) {
|
|
1977
|
+
const chunk = [];
|
|
1978
|
+
let written = 0;
|
|
1979
|
+
let next = first;
|
|
1980
|
+
while (next < spans.length && written < 32) {
|
|
1981
|
+
const label = email.slice(spans[next], spans[next + 1]);
|
|
1982
|
+
chunk.push(label);
|
|
1983
|
+
written += label.length;
|
|
1984
|
+
next += 3;
|
|
1985
|
+
}
|
|
1986
|
+
const aLabels = toALabels(chunk);
|
|
1987
|
+
for (let k = 0; k < chunk.length; k++) {
|
|
1988
|
+
const start = spans[first + 3 * k];
|
|
1989
|
+
const end = spans[first + 3 * k + 1];
|
|
1990
|
+
const aLabel = aLabels[k];
|
|
1991
|
+
if (aLabel === void 0) return fail("syntax.domain.label_invalid", start);
|
|
1992
|
+
const bad = checkLabel(email, start, end, aLabel.length);
|
|
1993
|
+
if (bad >= 0) return fail("syntax.domain.label_invalid", bad);
|
|
1994
|
+
growth += aLabel.length - (end - start);
|
|
1995
|
+
if (spans[first + 3 * k + 2] + (end - start) + growth > cap) return Infinity;
|
|
1996
|
+
}
|
|
1997
|
+
first = next;
|
|
1998
|
+
}
|
|
1999
|
+
return growth;
|
|
2000
|
+
}
|
|
2001
|
+
/**
|
|
2002
|
+
* The scan behind {@link scanDomain}, which leaves the U-labels it finds in
|
|
2003
|
+
* `uLabels`, as triples of start, end, and the domain's length before it,
|
|
2004
|
+
* unchecked but for their length.
|
|
2005
|
+
*/
|
|
2006
|
+
function scanLabels(email, at, rules, comments, uLabels) {
|
|
1849
2007
|
const end = email.length;
|
|
1850
2008
|
let domain = "";
|
|
1851
2009
|
let labels = 0;
|
|
@@ -1856,10 +2014,9 @@ function scanDomain(email, at, rules, comments) {
|
|
|
1856
2014
|
let edge = -1;
|
|
1857
2015
|
let trailing = comments.length;
|
|
1858
2016
|
let tld = "";
|
|
1859
|
-
let growth = 0;
|
|
1860
2017
|
let i = at + 1;
|
|
1861
2018
|
while (i < end) {
|
|
1862
|
-
const code = email
|
|
2019
|
+
const code = codeAt(email, i);
|
|
1863
2020
|
const opensLiteral = code === 91 && rules.literals && labels === 0;
|
|
1864
2021
|
const ldh = inClass(code, rules.domain);
|
|
1865
2022
|
if (opensLiteral || ldh || rules.idn && nonAsciiAt(email, i, end) > 0) {
|
|
@@ -1872,20 +2029,18 @@ function scanDomain(email, at, rules, comments) {
|
|
|
1872
2029
|
literal = true;
|
|
1873
2030
|
} else {
|
|
1874
2031
|
next = ldh ? i + 1 : i;
|
|
1875
|
-
while (next < end && inClass(email
|
|
2032
|
+
while (next < end && inClass(codeAt(email, next), rules.domain)) next++;
|
|
1876
2033
|
const ascii = next;
|
|
1877
2034
|
if (rules.idn) next = skipUnicode(email, next, end, rules.domain);
|
|
1878
|
-
let size = next - i;
|
|
1879
2035
|
if (next > ascii) {
|
|
1880
|
-
|
|
1881
|
-
|
|
1882
|
-
|
|
1883
|
-
|
|
2036
|
+
if (codePoints(email, i, next) > 63) return fail("syntax.domain.label_invalid", i);
|
|
2037
|
+
uLabels.push(i, next, domain.length);
|
|
2038
|
+
} else {
|
|
2039
|
+
const bad = checkLabel(email, i, next, next - i);
|
|
2040
|
+
if (bad >= 0) return fail("syntax.domain.label_invalid", bad);
|
|
1884
2041
|
}
|
|
1885
|
-
const bad = checkLabel(email, i, next, size);
|
|
1886
|
-
if (bad >= 0) return fail("syntax.domain.label_invalid", bad);
|
|
1887
2042
|
}
|
|
1888
|
-
tld = unfold(email
|
|
2043
|
+
tld = unfold(email, i, next);
|
|
1889
2044
|
domain += tld;
|
|
1890
2045
|
labels++;
|
|
1891
2046
|
afterLabel = true;
|
|
@@ -1924,7 +2079,7 @@ function scanDomain(email, at, rules, comments) {
|
|
|
1924
2079
|
if (labels === 0) return fail("syntax.domain.empty");
|
|
1925
2080
|
if (dot >= 0) return fail("syntax.domain.label_invalid", dot);
|
|
1926
2081
|
settle(comments, trailing, "after-domain");
|
|
1927
|
-
const size = domain.length
|
|
2082
|
+
const size = domain.length;
|
|
1928
2083
|
return literal || labels === 1 ? {
|
|
1929
2084
|
domain,
|
|
1930
2085
|
literal,
|
|
@@ -1944,7 +2099,7 @@ function skipUnicode(email, i, end, flag) {
|
|
|
1944
2099
|
let width = i < end ? nonAsciiAt(email, i, end) : 0;
|
|
1945
2100
|
while (width > 0) {
|
|
1946
2101
|
i += width;
|
|
1947
|
-
while (i < end && inClass(email
|
|
2102
|
+
while (i < end && inClass(codeAt(email, i), flag)) i++;
|
|
1948
2103
|
width = i < end ? nonAsciiAt(email, i, end) : 0;
|
|
1949
2104
|
}
|
|
1950
2105
|
return i;
|
|
@@ -1955,9 +2110,9 @@ function skipUnicode(email, i, end, flag) {
|
|
|
1955
2110
|
* a label over 63 characters, or -1.
|
|
1956
2111
|
*/
|
|
1957
2112
|
function checkLabel(email, start, end, size) {
|
|
1958
|
-
if (email
|
|
2113
|
+
if (codeAt(email, start) === 45) return start;
|
|
1959
2114
|
if (size > 63) return start;
|
|
1960
|
-
return email
|
|
2115
|
+
return codeAt(email, end - 1) === 45 ? end - 1 : -1;
|
|
1961
2116
|
}
|
|
1962
2117
|
/** Marks the comments from `from` on, which no word follows, as trailing. */
|
|
1963
2118
|
function settle(comments, from, position) {
|
|
@@ -1966,18 +2121,23 @@ function settle(comments, from, position) {
|
|
|
1966
2121
|
if (comment.position.startsWith("inside")) comment.position = position;
|
|
1967
2122
|
}
|
|
1968
2123
|
}
|
|
1969
|
-
|
|
1970
|
-
|
|
1971
|
-
|
|
2124
|
+
const { indexOf, slice } = String.prototype;
|
|
2125
|
+
/**
|
|
2126
|
+
* `email` from `start` to `end`, without the CRLFs of folding whitespace,
|
|
2127
|
+
* keeping the space or tab after each.
|
|
2128
|
+
*/
|
|
2129
|
+
function unfold(email, start, end) {
|
|
2130
|
+
const text = slice.call(email, start, end);
|
|
2131
|
+
return indexOf.call(text, "\r") < 0 ? text : text.replaceAll("\r\n", "");
|
|
1972
2132
|
}
|
|
1973
2133
|
/**
|
|
1974
2134
|
* Skips one character of folding whitespace: a space, a tab, or a CRLF with
|
|
1975
2135
|
* a space or tab after it. A lone CR or LF breaks it.
|
|
1976
2136
|
*/
|
|
1977
2137
|
function skipSpace(email, i, end) {
|
|
1978
|
-
const code = email
|
|
2138
|
+
const code = codeAt(email, i);
|
|
1979
2139
|
if (code === 32 || code === 9) return i + 1;
|
|
1980
|
-
return code === 13 && i + 2 < end && email
|
|
2140
|
+
return code === 13 && i + 2 < end && codeAt(email, i + 1) === 10 && (codeAt(email, i + 2) === 32 || codeAt(email, i + 2) === 9) ? i + 2 : -1 - i;
|
|
1981
2141
|
}
|
|
1982
2142
|
/** Whether a backslash may escape `code`: any ASCII with `obs`, else VCHAR and space. */
|
|
1983
2143
|
function escapable(code, obs) {
|
|
@@ -1991,10 +2151,10 @@ function escapable(code, obs) {
|
|
|
1991
2151
|
function skipQuoted(email, open, end, obs, unicode) {
|
|
1992
2152
|
let i = open + 1;
|
|
1993
2153
|
while (i < end) {
|
|
1994
|
-
const code = email
|
|
2154
|
+
const code = codeAt(email, i);
|
|
1995
2155
|
if (code === 34) return i + 1;
|
|
1996
2156
|
if (code === 92) {
|
|
1997
|
-
if (!escapable(email
|
|
2157
|
+
if (!escapable(codeAt(email, i + 1), obs)) return -2 - i;
|
|
1998
2158
|
i += 2;
|
|
1999
2159
|
} else if (obs && isWhitespace(code)) {
|
|
2000
2160
|
i = skipSpace(email, i, end);
|
|
@@ -2016,14 +2176,14 @@ function skipComment(email, open, end, unicode) {
|
|
|
2016
2176
|
let depth = 1;
|
|
2017
2177
|
let i = open + 1;
|
|
2018
2178
|
while (i < end) {
|
|
2019
|
-
const code = email
|
|
2179
|
+
const code = codeAt(email, i);
|
|
2020
2180
|
if (code === 40 || code === 41) {
|
|
2021
2181
|
depth += code === 40 ? 1 : -1;
|
|
2022
2182
|
i++;
|
|
2023
2183
|
if (depth === 0) return i;
|
|
2024
2184
|
} else if (code === 92) {
|
|
2025
2185
|
if (i + 1 >= end) break;
|
|
2026
|
-
if (!escapable(email
|
|
2186
|
+
if (!escapable(codeAt(email, i + 1), true)) return -2 - i;
|
|
2027
2187
|
i += 2;
|
|
2028
2188
|
} else if (isWhitespace(code)) {
|
|
2029
2189
|
i = skipSpace(email, i, end);
|
|
@@ -2044,10 +2204,10 @@ function skipComment(email, open, end, unicode) {
|
|
|
2044
2204
|
function skipLiteral(email, open, end) {
|
|
2045
2205
|
let i = open + 1;
|
|
2046
2206
|
while (i < end) {
|
|
2047
|
-
const code = email
|
|
2207
|
+
const code = codeAt(email, i);
|
|
2048
2208
|
if (code === 93) return i + 1;
|
|
2049
2209
|
if (code === 92) {
|
|
2050
|
-
if (i + 1 >= end || !escapable(email
|
|
2210
|
+
if (i + 1 >= end || !escapable(codeAt(email, i + 1), true)) return -1;
|
|
2051
2211
|
i += 2;
|
|
2052
2212
|
} else if (isWhitespace(code)) {
|
|
2053
2213
|
i = skipSpace(email, i, end);
|
|
@@ -2078,18 +2238,20 @@ function skipAddressLiteral(email, open, end) {
|
|
|
2078
2238
|
* the first check it fails.
|
|
2079
2239
|
*
|
|
2080
2240
|
* @remarks
|
|
2081
|
-
*
|
|
2082
|
-
*
|
|
2083
|
-
*
|
|
2241
|
+
* Input longer than `maxLength` (512 UTF-16 code units by default) fails as
|
|
2242
|
+
* `syntax.address.too_long` before it's scanned. Otherwise the local part is
|
|
2243
|
+
* checked before the domain, each left to right, then the lengths, then a
|
|
2244
|
+
* dotless domain, then the TLD. `index`, where a failure has one, points at
|
|
2245
|
+
* the offending character.
|
|
2084
2246
|
*
|
|
2085
2247
|
* @example
|
|
2086
2248
|
* ```ts
|
|
2087
|
-
*
|
|
2088
|
-
*
|
|
2089
|
-
*
|
|
2090
|
-
*
|
|
2091
|
-
*
|
|
2092
|
-
* }
|
|
2249
|
+
* import { parseAddress } from '@email-utils/validator-syntax';
|
|
2250
|
+
*
|
|
2251
|
+
* parseAddress('ada.lovelace@example.co.uk');
|
|
2252
|
+
* // => { ok: true, value: { local: 'ada.lovelace', tld: 'uk' } }
|
|
2253
|
+
* parseAddress('ada..lovelace@example.com');
|
|
2254
|
+
* // => { ok: false, reason: 'syntax.local.consecutive_dots', index: 4 }
|
|
2093
2255
|
* ```
|
|
2094
2256
|
*
|
|
2095
2257
|
* @throws TypeError when `email` isn't a string, or `options` are malformed.
|
|
@@ -2102,9 +2264,11 @@ function parseAddress(email, options) {
|
|
|
2102
2264
|
*
|
|
2103
2265
|
* @example
|
|
2104
2266
|
* ```ts
|
|
2105
|
-
* isValidSyntax
|
|
2106
|
-
*
|
|
2107
|
-
* isValidSyntax('
|
|
2267
|
+
* import { isValidSyntax } from '@email-utils/validator-syntax';
|
|
2268
|
+
*
|
|
2269
|
+
* isValidSyntax('ada@example.com'); // => true
|
|
2270
|
+
* isValidSyntax('"ada"@example.com'); // => false
|
|
2271
|
+
* isValidSyntax('"ada"@example.com', { preset: 'rfc5321' }); // => true
|
|
2108
2272
|
* ```
|
|
2109
2273
|
*
|
|
2110
2274
|
* @throws TypeError when `email` isn't a string, or `options` are malformed.
|
|
@@ -2118,9 +2282,14 @@ function isValidSyntax(email, options) {
|
|
|
2118
2282
|
*
|
|
2119
2283
|
* @example
|
|
2120
2284
|
* ```ts
|
|
2285
|
+
* import { createSyntaxValidator } from '@email-utils/validator-syntax';
|
|
2286
|
+
*
|
|
2121
2287
|
* const strict = createSyntaxValidator({ preset: 'rfc5321' });
|
|
2122
|
-
* strict.parse('a b@example.com');
|
|
2123
|
-
*
|
|
2288
|
+
* strict.parse('a b@example.com');
|
|
2289
|
+
* // => { ok: false, reason: 'syntax.local.unquoted_space', index: 1 }
|
|
2290
|
+
* strict.isValid('"a b"@example.com'); // => true
|
|
2291
|
+
* createSyntaxValidator({ preset: 'html5', allowComments: true });
|
|
2292
|
+
* // => throws TypeError
|
|
2124
2293
|
* ```
|
|
2125
2294
|
*
|
|
2126
2295
|
* @throws TypeError when `options` are malformed.
|