@email-utils/validator-syntax 1.0.0-rc.1 → 1.0.0-rc.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.mjs CHANGED
@@ -1,6 +1,18 @@
1
+ //#region src/chars.ts
2
+ const charCodeAt = String.prototype.charCodeAt;
3
+ /**
4
+ * `text.charCodeAt(i)`, through the builtin itself. Called as a method,
5
+ * `charCodeAt` is looked up on the string, and a call site that has seen
6
+ * enough kinds of string (one-byte and two-byte, flat, sliced, and
7
+ * concatenated) goes megamorphic: V8 stops inlining it, and after arbitrary
8
+ * Unicode input every scan ran 2–3× slower (validator-syntax#15).
9
+ */
10
+ function codeAt(text, i) {
11
+ return charCodeAt.call(text, i);
12
+ }
1
13
  const table = /* @__PURE__ */ new Uint8Array(128);
2
14
  function mark(chars, flag) {
3
- for (let i = 0; i < chars.length; i++) table[chars.charCodeAt(i)] |= flag;
15
+ for (let i = 0; i < chars.length; i++) table[codeAt(chars, i)] |= flag;
4
16
  }
5
17
  const alnum = "abcdefghijklmnopqrstuvwxyzABCDEFGHIJKLMNOPQRSTUVWXYZ0123456789";
6
18
  mark(alnum + "!#$%&'*+-/=?^_`{|}~", 1);
@@ -21,22 +33,35 @@ function isWhitespace(code) {
21
33
  * lone surrogate, which no UTF-8 text can hold.
22
34
  */
23
35
  function nonAsciiAt(text, i, end) {
24
- const code = text.charCodeAt(i);
36
+ const code = codeAt(text, i);
25
37
  if (code < 128 || code >= 56320 && code <= 57343) return 0;
26
38
  if (code < 55296 || code > 56319) return 1;
27
- const low = i + 1 < end ? text.charCodeAt(i + 1) : 0;
39
+ const low = i + 1 < end ? codeAt(text, i + 1) : 0;
28
40
  return low >= 56320 && low <= 57343 ? 2 : 0;
29
41
  }
30
42
  /** The length of `text` in UTF-8 octets, the unit RFC 6531 caps a local part in. */
31
43
  function utf8Length(text) {
32
- let octets = text.length;
33
- for (let i = 0; i < text.length; i++) {
34
- const code = text.charCodeAt(i);
44
+ const end = text.length;
45
+ let octets = end;
46
+ for (let i = 0; i < end; i++) {
47
+ const code = codeAt(text, i);
35
48
  if (code >= 2048 && (code < 55296 || code > 57343)) octets += 2;
36
49
  else if (code >= 128) octets += 1;
37
50
  }
38
51
  return octets;
39
52
  }
53
+ /**
54
+ * The code points in `text` from `start` to `end`, which must not split a
55
+ * surrogate pair: its length with each pair counted once.
56
+ */
57
+ function codePoints(text, start, end) {
58
+ let count = end - start;
59
+ for (let i = start; i < end; i++) {
60
+ const code = codeAt(text, i);
61
+ if (code >= 56320 && code <= 57343) count--;
62
+ }
63
+ return count;
64
+ }
40
65
  //#endregion
41
66
  //#region src/options.ts
42
67
  const presets = {
@@ -52,7 +77,8 @@ const presets = {
52
77
  checkTld: true,
53
78
  allowNoTld: false,
54
79
  unicode: false,
55
- idn: false
80
+ idn: false,
81
+ maxLength: 512
56
82
  },
57
83
  rfc5321: {
58
84
  local: 1,
@@ -66,7 +92,8 @@ const presets = {
66
92
  checkTld: false,
67
93
  allowNoTld: false,
68
94
  unicode: false,
69
- idn: false
95
+ idn: false,
96
+ maxLength: 512
70
97
  },
71
98
  rfc5322: {
72
99
  local: 1,
@@ -80,7 +107,8 @@ const presets = {
80
107
  checkTld: false,
81
108
  allowNoTld: false,
82
109
  unicode: false,
83
- idn: false
110
+ idn: false,
111
+ maxLength: 512
84
112
  },
85
113
  html5: {
86
114
  local: 4,
@@ -94,7 +122,8 @@ const presets = {
94
122
  checkTld: false,
95
123
  allowNoTld: true,
96
124
  unicode: false,
97
- idn: false
125
+ idn: false,
126
+ maxLength: 512
98
127
  }
99
128
  };
100
129
  const overrides = [
@@ -120,13 +149,14 @@ const unsupported = {
120
149
  * Resolves `options` into the rules they select.
121
150
  *
122
151
  * @throws TypeError when `options` isn't an object, names an unknown option
123
- * or preset, gives a non-boolean override, or turns on something the
124
- * preset's grammar has no room for.
152
+ * or preset, gives a non-boolean override or a `maxLength` that isn't a
153
+ * positive integer or `Infinity`, or turns on something the preset's grammar
154
+ * has no room for.
125
155
  */
126
156
  function resolve(options) {
127
157
  if (options === void 0) return presets.practical;
128
158
  if (typeof options !== "object" || options === null) throw new TypeError("Expected the options to be an object");
129
- for (const key of Object.keys(options)) if (key !== "preset" && !overrides.includes(key)) throw new TypeError(`Unknown option: ${key}`);
159
+ for (const key of Object.keys(options)) if (key !== "preset" && key !== "maxLength" && !overrides.includes(key)) throw new TypeError(`Unknown option: ${key}`);
130
160
  const preset = options.preset ?? "practical";
131
161
  if (!Object.hasOwn(presets, preset)) throw new TypeError(`Unknown preset: ${preset}`);
132
162
  const base = presets[preset];
@@ -134,8 +164,10 @@ function resolve(options) {
134
164
  const value = options[key];
135
165
  if (value !== void 0 && typeof value !== "boolean") throw new TypeError(`Expected ${key} to be a boolean`);
136
166
  }
167
+ const { maxLength } = options;
168
+ if (maxLength !== void 0 && maxLength !== Infinity && !(Number.isInteger(maxLength) && maxLength > 0)) throw new TypeError("Expected maxLength to be a positive integer or Infinity");
137
169
  for (const key of unsupported[preset]) if (options[key] === true) throw new TypeError(`${key} can’t be true with the ${preset} preset`);
138
- if (overrides.every((key) => options[key] === void 0)) return base;
170
+ if (maxLength === void 0 && overrides.every((key) => options[key] === void 0)) return base;
139
171
  return {
140
172
  ...base,
141
173
  checkTld: options.checkTld ?? base.checkTld,
@@ -143,13 +175,27 @@ function resolve(options) {
143
175
  comments: options.allowComments ?? base.comments,
144
176
  unicode: options.allowUnicode ?? base.unicode,
145
177
  idn: options.allowIdn ?? base.idn,
146
- literals: options.allowIpLiteral ?? base.literals
178
+ literals: options.allowIpLiteral ?? base.literals,
179
+ maxLength: maxLength ?? base.maxLength
147
180
  };
148
181
  }
149
182
  //#endregion
150
183
  //#region src/idn.ts
151
184
  const ldh = /^[\da-z-]+$/;
152
185
  /**
186
+ * The hostname of `http://x.` and `host`, or `undefined` when the WHATWG URL
187
+ * parser rejects it. `URL.parse` says so without the cost of throwing, a few
188
+ * µs a time, so an address built to fail many conversions stays fast;
189
+ * runtimes that predate it (Node before 22.1) throw instead.
190
+ */
191
+ const hostnameOf = typeof URL.parse === "function" ? (host) => URL.parse(`http://x.${host}`)?.hostname : (host) => {
192
+ try {
193
+ return new URL(`http://x.${host}`).hostname;
194
+ } catch {
195
+ return;
196
+ }
197
+ };
198
+ /**
153
199
  * The A-label for a U-label, or `undefined` when UTS #46 rejects it or maps
154
200
  * it to something other than one IDN label (plain ASCII, or two labels, as
155
201
  * `。` becomes a dot).
@@ -160,14 +206,65 @@ const ldh = /^[\da-z-]+$/;
160
206
  * checks. It leaves hyphens and lengths alone, so the caller checks those.
161
207
  */
162
208
  function toALabel(label) {
163
- let host;
164
- try {
165
- host = new URL(`http://x.${label}`).hostname;
166
- } catch {
167
- return;
209
+ const aLabel = hostnameOf(label)?.slice(2);
210
+ return aLabel?.startsWith("xn--") === true && ldh.test(aLabel) ? aLabel : void 0;
211
+ }
212
+ const rtl = /[\u0590-\u08ff\u2135-\u2138\ufb1d-\ufdff\ufe70-\ufeff\u{10800}-\u{10fff}\u{1e800}-\u{1efff}]/u;
213
+ /**
214
+ * The A-label for each U-label in `labels`, as {@link toALabel} gives it,
215
+ * converted as few domains as it can: one URL costs about as much as one
216
+ * label does.
217
+ *
218
+ * @remarks
219
+ * Every entry up to the first `undefined` is exact; entries after it may be
220
+ * left `undefined` unconverted. Converted together, a right-to-left label
221
+ * would hold the others to RFC 5893's rules too, which it doesn't alone, so
222
+ * right-to-left labels go in a domain of their own. Converting more labels
223
+ * at once only adds rules, so a domain that converts means every label
224
+ * would alone; one that doesn't is narrowed down to its first label that
225
+ * fails alone.
226
+ */
227
+ function toALabels(labels) {
228
+ if (!labels.some((label) => rtl.test(label))) return convertGroup(labels);
229
+ const aLabels = labels.map(() => void 0);
230
+ for (const bidi of [false, true]) {
231
+ const members = [];
232
+ labels.forEach((label, k) => {
233
+ if (rtl.test(label) === bidi) members.push(k);
234
+ });
235
+ if (members.length > 0) convertGroup(members.map((k) => labels[k])).forEach((aLabel, j) => {
236
+ aLabels[members[j]] = aLabel;
237
+ });
168
238
  }
169
- const aLabel = host.slice(2);
170
- return aLabel.startsWith("xn--") && ldh.test(aLabel) ? aLabel : void 0;
239
+ return aLabels;
240
+ }
241
+ /** {@link toALabels} for labels converted together. */
242
+ function convertGroup(labels) {
243
+ const all = convert(labels);
244
+ if (all !== void 0) return all;
245
+ let pass = [];
246
+ let failing = labels.length;
247
+ while (failing - pass.length > 1) {
248
+ const some = convert(labels.slice(0, pass.length + failing >> 1));
249
+ if (some === void 0) failing = pass.length + failing >> 1;
250
+ else pass = some;
251
+ }
252
+ const aLabels = pass;
253
+ for (let k = pass.length; k < labels.length; k++) {
254
+ const aLabel = toALabel(labels[k]);
255
+ aLabels.push(aLabel);
256
+ if (aLabel === void 0) break;
257
+ }
258
+ return aLabels;
259
+ }
260
+ const aLabelHost = /^x(?:\.xn--[\da-z-]+)+$/;
261
+ /** `labels` converted as one domain, or `undefined`. */
262
+ function convert(labels) {
263
+ const hostname = hostnameOf(labels.join("."));
264
+ if (hostname === void 0 || !aLabelHost.test(hostname)) return;
265
+ const aLabels = hostname.split(".");
266
+ aLabels.shift();
267
+ return aLabels.length === labels.length ? aLabels : void 0;
171
268
  }
172
269
  //#endregion
173
270
  //#region src/ip.ts
@@ -175,6 +272,7 @@ const octet = /^\d{1,3}$/;
175
272
  const hexGroup = /^[\dA-Fa-f]{1,4}$/;
176
273
  /** An IPv4 address: four dot-separated decimal octets, each 0–255. */
177
274
  function isIPv4(text) {
275
+ if (text.length > 15) return false;
178
276
  const parts = text.split(".");
179
277
  return parts.length === 4 && parts.every((part) => octet.test(part) && Number(part) <= 255);
180
278
  }
@@ -184,6 +282,7 @@ function isIPv4(text) {
184
282
  * place of the last two groups.
185
283
  */
186
284
  function isIPv6(text) {
285
+ if (text.length > 45) return false;
187
286
  const colon = text.lastIndexOf(":");
188
287
  const tail = text.slice(colon + 1);
189
288
  if (tail.includes(".")) {
@@ -1667,7 +1766,7 @@ const messages = {
1667
1766
  "syntax.local.too_long": "The local part is longer than 64 characters",
1668
1767
  "syntax.local.invalid_char": "The local part has a character it can’t hold",
1669
1768
  "syntax.local.consecutive_dots": "The local part has two dots in a row",
1670
- "syntax.local.unquoted_space": "The local part has a space outside quotes",
1769
+ "syntax.local.unquoted_space": "The local part has whitespace outside quotes",
1671
1770
  "syntax.domain.empty": "Nothing comes after the @",
1672
1771
  "syntax.domain.no_dot": "The domain has no dot",
1673
1772
  "syntax.domain.label_invalid": "A domain label is empty, too long, or starts or ends with a hyphen",
@@ -1710,8 +1809,9 @@ function findAt(email, rules) {
1710
1809
  let domainStart = false;
1711
1810
  let depth = 0;
1712
1811
  let at = -1;
1713
- for (let i = 0; i < email.length; i++) {
1714
- const code = email.charCodeAt(i);
1812
+ const end = email.length;
1813
+ for (let i = 0; i < end; i++) {
1814
+ const code = codeAt(email, i);
1715
1815
  if (state === NORMAL) {
1716
1816
  if (code === 64) {
1717
1817
  at = i;
@@ -1742,6 +1842,7 @@ function findAt(email, rules) {
1742
1842
  function parse(email, rules) {
1743
1843
  if (typeof email !== "string") throw new TypeError(`Expected a string, got ${typeof email}`);
1744
1844
  if (email === "") return fail("syntax.address.empty");
1845
+ if (email.length > rules.maxLength) return fail("syntax.address.too_long");
1745
1846
  const at = findAt(email, rules);
1746
1847
  if (at < 0) return fail("syntax.address.no_at");
1747
1848
  const comments = [];
@@ -1785,7 +1886,7 @@ function scanLocal(email, at, rules, comments) {
1785
1886
  let trailing = comments.length;
1786
1887
  let i = 0;
1787
1888
  while (i < at) {
1788
- const code = email.charCodeAt(i);
1889
+ const code = codeAt(email, i);
1789
1890
  const quote = code === 34 && rules.quotes && (rules.obs || i === 0);
1790
1891
  const atom = inClass(code, rules.local);
1791
1892
  if (quote || atom || rules.unicode && nonAsciiAt(email, i, at) > 0) {
@@ -1798,10 +1899,10 @@ function scanLocal(email, at, rules, comments) {
1798
1899
  closed = !rules.obs;
1799
1900
  } else {
1800
1901
  next = atom ? i + 1 : i;
1801
- while (next < at && inClass(email.charCodeAt(next), rules.local)) next++;
1902
+ while (next < at && inClass(codeAt(email, next), rules.local)) next++;
1802
1903
  if (rules.unicode) next = skipUnicode(email, next, at, rules.local);
1803
1904
  }
1804
- local += unfold(email.slice(i, next));
1905
+ local += unfold(email, i, next);
1805
1906
  words++;
1806
1907
  afterWord = true;
1807
1908
  dot = -1;
@@ -1846,6 +1947,63 @@ function scanLocal(email, at, rules, comments) {
1846
1947
  * folding whitespace.
1847
1948
  */
1848
1949
  function scanDomain(email, at, rules, comments) {
1950
+ const uLabels = [];
1951
+ const domain = scanLabels(email, at, rules, comments, uLabels);
1952
+ if (uLabels.length === 0) return domain;
1953
+ const growth = checkULabels(email, uLabels, rules.domainCap);
1954
+ if (typeof growth !== "number") return growth;
1955
+ if (!("reason" in domain)) domain.size += growth;
1956
+ return domain;
1957
+ }
1958
+ /**
1959
+ * Checks the U-labels at `spans`, triples of start, end, and the length of
1960
+ * the domain before it, as A-labels: each must convert, and pass the
1961
+ * hostname rules as its A-label. Returns the first failure, or what the
1962
+ * A-labels add to the domain's length: `Infinity` once it's past `cap`.
1963
+ *
1964
+ * @remarks
1965
+ * The U-labels are converted a few at a time, about 32 characters of them
1966
+ * in each URL: one URL costs about as much as one label alone, and a small
1967
+ * batch wastes little past the first failure or past `cap`. Once the
1968
+ * A-labels take the domain past `cap`, the rest aren't checked: the domain
1969
+ * fails as too long, unless the scan found something earlier. Punycode's
1970
+ * cost grows with the square of a label's length, so converting labels that
1971
+ * can't change the outcome would cost the most on input built to be slow
1972
+ * (validator-syntax#15).
1973
+ */
1974
+ function checkULabels(email, spans, cap) {
1975
+ let growth = 0;
1976
+ for (let first = 0; first < spans.length;) {
1977
+ const chunk = [];
1978
+ let written = 0;
1979
+ let next = first;
1980
+ while (next < spans.length && written < 32) {
1981
+ const label = email.slice(spans[next], spans[next + 1]);
1982
+ chunk.push(label);
1983
+ written += label.length;
1984
+ next += 3;
1985
+ }
1986
+ const aLabels = toALabels(chunk);
1987
+ for (let k = 0; k < chunk.length; k++) {
1988
+ const start = spans[first + 3 * k];
1989
+ const end = spans[first + 3 * k + 1];
1990
+ const aLabel = aLabels[k];
1991
+ if (aLabel === void 0) return fail("syntax.domain.label_invalid", start);
1992
+ const bad = checkLabel(email, start, end, aLabel.length);
1993
+ if (bad >= 0) return fail("syntax.domain.label_invalid", bad);
1994
+ growth += aLabel.length - (end - start);
1995
+ if (spans[first + 3 * k + 2] + (end - start) + growth > cap) return Infinity;
1996
+ }
1997
+ first = next;
1998
+ }
1999
+ return growth;
2000
+ }
2001
+ /**
2002
+ * The scan behind {@link scanDomain}, which leaves the U-labels it finds in
2003
+ * `uLabels`, as triples of start, end, and the domain's length before it,
2004
+ * unchecked but for their length.
2005
+ */
2006
+ function scanLabels(email, at, rules, comments, uLabels) {
1849
2007
  const end = email.length;
1850
2008
  let domain = "";
1851
2009
  let labels = 0;
@@ -1856,10 +2014,9 @@ function scanDomain(email, at, rules, comments) {
1856
2014
  let edge = -1;
1857
2015
  let trailing = comments.length;
1858
2016
  let tld = "";
1859
- let growth = 0;
1860
2017
  let i = at + 1;
1861
2018
  while (i < end) {
1862
- const code = email.charCodeAt(i);
2019
+ const code = codeAt(email, i);
1863
2020
  const opensLiteral = code === 91 && rules.literals && labels === 0;
1864
2021
  const ldh = inClass(code, rules.domain);
1865
2022
  if (opensLiteral || ldh || rules.idn && nonAsciiAt(email, i, end) > 0) {
@@ -1872,20 +2029,18 @@ function scanDomain(email, at, rules, comments) {
1872
2029
  literal = true;
1873
2030
  } else {
1874
2031
  next = ldh ? i + 1 : i;
1875
- while (next < end && inClass(email.charCodeAt(next), rules.domain)) next++;
2032
+ while (next < end && inClass(codeAt(email, next), rules.domain)) next++;
1876
2033
  const ascii = next;
1877
2034
  if (rules.idn) next = skipUnicode(email, next, end, rules.domain);
1878
- let size = next - i;
1879
2035
  if (next > ascii) {
1880
- const aLabel = toALabel(email.slice(i, next));
1881
- if (aLabel === void 0) return fail("syntax.domain.label_invalid", i);
1882
- size = aLabel.length;
1883
- growth += size - (next - i);
2036
+ if (codePoints(email, i, next) > 63) return fail("syntax.domain.label_invalid", i);
2037
+ uLabels.push(i, next, domain.length);
2038
+ } else {
2039
+ const bad = checkLabel(email, i, next, next - i);
2040
+ if (bad >= 0) return fail("syntax.domain.label_invalid", bad);
1884
2041
  }
1885
- const bad = checkLabel(email, i, next, size);
1886
- if (bad >= 0) return fail("syntax.domain.label_invalid", bad);
1887
2042
  }
1888
- tld = unfold(email.slice(i, next));
2043
+ tld = unfold(email, i, next);
1889
2044
  domain += tld;
1890
2045
  labels++;
1891
2046
  afterLabel = true;
@@ -1924,7 +2079,7 @@ function scanDomain(email, at, rules, comments) {
1924
2079
  if (labels === 0) return fail("syntax.domain.empty");
1925
2080
  if (dot >= 0) return fail("syntax.domain.label_invalid", dot);
1926
2081
  settle(comments, trailing, "after-domain");
1927
- const size = domain.length + growth;
2082
+ const size = domain.length;
1928
2083
  return literal || labels === 1 ? {
1929
2084
  domain,
1930
2085
  literal,
@@ -1944,7 +2099,7 @@ function skipUnicode(email, i, end, flag) {
1944
2099
  let width = i < end ? nonAsciiAt(email, i, end) : 0;
1945
2100
  while (width > 0) {
1946
2101
  i += width;
1947
- while (i < end && inClass(email.charCodeAt(i), flag)) i++;
2102
+ while (i < end && inClass(codeAt(email, i), flag)) i++;
1948
2103
  width = i < end ? nonAsciiAt(email, i, end) : 0;
1949
2104
  }
1950
2105
  return i;
@@ -1955,9 +2110,9 @@ function skipUnicode(email, i, end, flag) {
1955
2110
  * a label over 63 characters, or -1.
1956
2111
  */
1957
2112
  function checkLabel(email, start, end, size) {
1958
- if (email.charCodeAt(start) === 45) return start;
2113
+ if (codeAt(email, start) === 45) return start;
1959
2114
  if (size > 63) return start;
1960
- return email.charCodeAt(end - 1) === 45 ? end - 1 : -1;
2115
+ return codeAt(email, end - 1) === 45 ? end - 1 : -1;
1961
2116
  }
1962
2117
  /** Marks the comments from `from` on, which no word follows, as trailing. */
1963
2118
  function settle(comments, from, position) {
@@ -1966,18 +2121,23 @@ function settle(comments, from, position) {
1966
2121
  if (comment.position.startsWith("inside")) comment.position = position;
1967
2122
  }
1968
2123
  }
1969
- /** Removes the CRLFs of folding whitespace, keeping the space or tab after. */
1970
- function unfold(text) {
1971
- return text.includes("\r") ? text.replaceAll("\r\n", "") : text;
2124
+ const { indexOf, slice } = String.prototype;
2125
+ /**
2126
+ * `email` from `start` to `end`, without the CRLFs of folding whitespace,
2127
+ * keeping the space or tab after each.
2128
+ */
2129
+ function unfold(email, start, end) {
2130
+ const text = slice.call(email, start, end);
2131
+ return indexOf.call(text, "\r") < 0 ? text : text.replaceAll("\r\n", "");
1972
2132
  }
1973
2133
  /**
1974
2134
  * Skips one character of folding whitespace: a space, a tab, or a CRLF with
1975
2135
  * a space or tab after it. A lone CR or LF breaks it.
1976
2136
  */
1977
2137
  function skipSpace(email, i, end) {
1978
- const code = email.charCodeAt(i);
2138
+ const code = codeAt(email, i);
1979
2139
  if (code === 32 || code === 9) return i + 1;
1980
- return code === 13 && i + 2 < end && email.charCodeAt(i + 1) === 10 && (email.charCodeAt(i + 2) === 32 || email.charCodeAt(i + 2) === 9) ? i + 2 : -1 - i;
2140
+ return code === 13 && i + 2 < end && codeAt(email, i + 1) === 10 && (codeAt(email, i + 2) === 32 || codeAt(email, i + 2) === 9) ? i + 2 : -1 - i;
1981
2141
  }
1982
2142
  /** Whether a backslash may escape `code`: any ASCII with `obs`, else VCHAR and space. */
1983
2143
  function escapable(code, obs) {
@@ -1991,10 +2151,10 @@ function escapable(code, obs) {
1991
2151
  function skipQuoted(email, open, end, obs, unicode) {
1992
2152
  let i = open + 1;
1993
2153
  while (i < end) {
1994
- const code = email.charCodeAt(i);
2154
+ const code = codeAt(email, i);
1995
2155
  if (code === 34) return i + 1;
1996
2156
  if (code === 92) {
1997
- if (!escapable(email.charCodeAt(i + 1), obs)) return -2 - i;
2157
+ if (!escapable(codeAt(email, i + 1), obs)) return -2 - i;
1998
2158
  i += 2;
1999
2159
  } else if (obs && isWhitespace(code)) {
2000
2160
  i = skipSpace(email, i, end);
@@ -2016,14 +2176,14 @@ function skipComment(email, open, end, unicode) {
2016
2176
  let depth = 1;
2017
2177
  let i = open + 1;
2018
2178
  while (i < end) {
2019
- const code = email.charCodeAt(i);
2179
+ const code = codeAt(email, i);
2020
2180
  if (code === 40 || code === 41) {
2021
2181
  depth += code === 40 ? 1 : -1;
2022
2182
  i++;
2023
2183
  if (depth === 0) return i;
2024
2184
  } else if (code === 92) {
2025
2185
  if (i + 1 >= end) break;
2026
- if (!escapable(email.charCodeAt(i + 1), true)) return -2 - i;
2186
+ if (!escapable(codeAt(email, i + 1), true)) return -2 - i;
2027
2187
  i += 2;
2028
2188
  } else if (isWhitespace(code)) {
2029
2189
  i = skipSpace(email, i, end);
@@ -2044,10 +2204,10 @@ function skipComment(email, open, end, unicode) {
2044
2204
  function skipLiteral(email, open, end) {
2045
2205
  let i = open + 1;
2046
2206
  while (i < end) {
2047
- const code = email.charCodeAt(i);
2207
+ const code = codeAt(email, i);
2048
2208
  if (code === 93) return i + 1;
2049
2209
  if (code === 92) {
2050
- if (i + 1 >= end || !escapable(email.charCodeAt(i + 1), true)) return -1;
2210
+ if (i + 1 >= end || !escapable(codeAt(email, i + 1), true)) return -1;
2051
2211
  i += 2;
2052
2212
  } else if (isWhitespace(code)) {
2053
2213
  i = skipSpace(email, i, end);
@@ -2078,18 +2238,20 @@ function skipAddressLiteral(email, open, end) {
2078
2238
  * the first check it fails.
2079
2239
  *
2080
2240
  * @remarks
2081
- * The local part is checked before the domain, each left to right, then the
2082
- * lengths, then a dotless domain, then the TLD. `index`, where a failure has
2083
- * one, points at the offending character.
2241
+ * Input longer than `maxLength` (512 UTF-16 code units by default) fails as
2242
+ * `syntax.address.too_long` before it's scanned. Otherwise the local part is
2243
+ * checked before the domain, each left to right, then the lengths, then a
2244
+ * dotless domain, then the TLD. `index`, where a failure has one, points at
2245
+ * the offending character.
2084
2246
  *
2085
2247
  * @example
2086
2248
  * ```ts
2087
- * const result = parseAddress('ada.lovelace@example.co.uk');
2088
- * if (result.ok) {
2089
- * result.value.tld; // 'uk'
2090
- * } else {
2091
- * result.reason; // e.g. 'syntax.local.invalid_char'
2092
- * }
2249
+ * import { parseAddress } from '@email-utils/validator-syntax';
2250
+ *
2251
+ * parseAddress('ada.lovelace@example.co.uk');
2252
+ * // => { ok: true, value: { local: 'ada.lovelace', tld: 'uk' } }
2253
+ * parseAddress('ada..lovelace@example.com');
2254
+ * // => { ok: false, reason: 'syntax.local.consecutive_dots', index: 4 }
2093
2255
  * ```
2094
2256
  *
2095
2257
  * @throws TypeError when `email` isn't a string, or `options` are malformed.
@@ -2102,9 +2264,11 @@ function parseAddress(email, options) {
2102
2264
  *
2103
2265
  * @example
2104
2266
  * ```ts
2105
- * isValidSyntax('ada@example.com'); // true
2106
- * isValidSyntax('"ada"@example.com'); // false
2107
- * isValidSyntax('"ada"@example.com', { preset: 'rfc5321' }); // true
2267
+ * import { isValidSyntax } from '@email-utils/validator-syntax';
2268
+ *
2269
+ * isValidSyntax('ada@example.com'); // => true
2270
+ * isValidSyntax('"ada"@example.com'); // => false
2271
+ * isValidSyntax('"ada"@example.com', { preset: 'rfc5321' }); // => true
2108
2272
  * ```
2109
2273
  *
2110
2274
  * @throws TypeError when `email` isn't a string, or `options` are malformed.
@@ -2118,9 +2282,14 @@ function isValidSyntax(email, options) {
2118
2282
  *
2119
2283
  * @example
2120
2284
  * ```ts
2285
+ * import { createSyntaxValidator } from '@email-utils/validator-syntax';
2286
+ *
2121
2287
  * const strict = createSyntaxValidator({ preset: 'rfc5321' });
2122
- * strict.parse('a b@example.com'); // { ok: false, reason: 'syntax.local.unquoted_space', … }
2123
- * strict.isValid('"a b"@example.com'); // true
2288
+ * strict.parse('a b@example.com');
2289
+ * // => { ok: false, reason: 'syntax.local.unquoted_space', index: 1 }
2290
+ * strict.isValid('"a b"@example.com'); // => true
2291
+ * createSyntaxValidator({ preset: 'html5', allowComments: true });
2292
+ * // => throws TypeError
2124
2293
  * ```
2125
2294
  *
2126
2295
  * @throws TypeError when `options` are malformed.