@email-utils/validator-syntax 1.0.0-rc.1 → 1.0.0-rc.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.cjs CHANGED
@@ -1,7 +1,19 @@
1
1
  Object.defineProperty(exports, Symbol.toStringTag, { value: "Module" });
2
+ //#region src/chars.ts
3
+ const charCodeAt = String.prototype.charCodeAt;
4
+ /**
5
+ * `text.charCodeAt(i)`, through the builtin itself. Called as a method,
6
+ * `charCodeAt` is looked up on the string, and a call site that has seen
7
+ * enough kinds of string (one-byte and two-byte, flat, sliced, and
8
+ * concatenated) goes megamorphic: V8 stops inlining it, and after arbitrary
9
+ * Unicode input every scan ran 2–3× slower (validator-syntax#15).
10
+ */
11
+ function codeAt(text, i) {
12
+ return charCodeAt.call(text, i);
13
+ }
2
14
  const table = /* @__PURE__ */ new Uint8Array(128);
3
15
  function mark(chars, flag) {
4
- for (let i = 0; i < chars.length; i++) table[chars.charCodeAt(i)] |= flag;
16
+ for (let i = 0; i < chars.length; i++) table[codeAt(chars, i)] |= flag;
5
17
  }
6
18
  const alnum = "abcdefghijklmnopqrstuvwxyzABCDEFGHIJKLMNOPQRSTUVWXYZ0123456789";
7
19
  mark(alnum + "!#$%&'*+-/=?^_`{|}~", 1);
@@ -22,22 +34,35 @@ function isWhitespace(code) {
22
34
  * lone surrogate, which no UTF-8 text can hold.
23
35
  */
24
36
  function nonAsciiAt(text, i, end) {
25
- const code = text.charCodeAt(i);
37
+ const code = codeAt(text, i);
26
38
  if (code < 128 || code >= 56320 && code <= 57343) return 0;
27
39
  if (code < 55296 || code > 56319) return 1;
28
- const low = i + 1 < end ? text.charCodeAt(i + 1) : 0;
40
+ const low = i + 1 < end ? codeAt(text, i + 1) : 0;
29
41
  return low >= 56320 && low <= 57343 ? 2 : 0;
30
42
  }
31
43
  /** The length of `text` in UTF-8 octets, the unit RFC 6531 caps a local part in. */
32
44
  function utf8Length(text) {
33
- let octets = text.length;
34
- for (let i = 0; i < text.length; i++) {
35
- const code = text.charCodeAt(i);
45
+ const end = text.length;
46
+ let octets = end;
47
+ for (let i = 0; i < end; i++) {
48
+ const code = codeAt(text, i);
36
49
  if (code >= 2048 && (code < 55296 || code > 57343)) octets += 2;
37
50
  else if (code >= 128) octets += 1;
38
51
  }
39
52
  return octets;
40
53
  }
54
+ /**
55
+ * The code points in `text` from `start` to `end`, which must not split a
56
+ * surrogate pair: its length with each pair counted once.
57
+ */
58
+ function codePoints(text, start, end) {
59
+ let count = end - start;
60
+ for (let i = start; i < end; i++) {
61
+ const code = codeAt(text, i);
62
+ if (code >= 56320 && code <= 57343) count--;
63
+ }
64
+ return count;
65
+ }
41
66
  //#endregion
42
67
  //#region src/options.ts
43
68
  const presets = {
@@ -53,7 +78,8 @@ const presets = {
53
78
  checkTld: true,
54
79
  allowNoTld: false,
55
80
  unicode: false,
56
- idn: false
81
+ idn: false,
82
+ maxLength: 512
57
83
  },
58
84
  rfc5321: {
59
85
  local: 1,
@@ -67,7 +93,8 @@ const presets = {
67
93
  checkTld: false,
68
94
  allowNoTld: false,
69
95
  unicode: false,
70
- idn: false
96
+ idn: false,
97
+ maxLength: 512
71
98
  },
72
99
  rfc5322: {
73
100
  local: 1,
@@ -81,7 +108,8 @@ const presets = {
81
108
  checkTld: false,
82
109
  allowNoTld: false,
83
110
  unicode: false,
84
- idn: false
111
+ idn: false,
112
+ maxLength: 512
85
113
  },
86
114
  html5: {
87
115
  local: 4,
@@ -95,7 +123,8 @@ const presets = {
95
123
  checkTld: false,
96
124
  allowNoTld: true,
97
125
  unicode: false,
98
- idn: false
126
+ idn: false,
127
+ maxLength: 512
99
128
  }
100
129
  };
101
130
  const overrides = [
@@ -121,13 +150,14 @@ const unsupported = {
121
150
  * Resolves `options` into the rules they select.
122
151
  *
123
152
  * @throws TypeError when `options` isn't an object, names an unknown option
124
- * or preset, gives a non-boolean override, or turns on something the
125
- * preset's grammar has no room for.
153
+ * or preset, gives a non-boolean override or a `maxLength` that isn't a
154
+ * positive integer or `Infinity`, or turns on something the preset's grammar
155
+ * has no room for.
126
156
  */
127
157
  function resolve(options) {
128
158
  if (options === void 0) return presets.practical;
129
159
  if (typeof options !== "object" || options === null) throw new TypeError("Expected the options to be an object");
130
- for (const key of Object.keys(options)) if (key !== "preset" && !overrides.includes(key)) throw new TypeError(`Unknown option: ${key}`);
160
+ for (const key of Object.keys(options)) if (key !== "preset" && key !== "maxLength" && !overrides.includes(key)) throw new TypeError(`Unknown option: ${key}`);
131
161
  const preset = options.preset ?? "practical";
132
162
  if (!Object.hasOwn(presets, preset)) throw new TypeError(`Unknown preset: ${preset}`);
133
163
  const base = presets[preset];
@@ -135,8 +165,10 @@ function resolve(options) {
135
165
  const value = options[key];
136
166
  if (value !== void 0 && typeof value !== "boolean") throw new TypeError(`Expected ${key} to be a boolean`);
137
167
  }
168
+ const { maxLength } = options;
169
+ if (maxLength !== void 0 && maxLength !== Infinity && !(Number.isInteger(maxLength) && maxLength > 0)) throw new TypeError("Expected maxLength to be a positive integer or Infinity");
138
170
  for (const key of unsupported[preset]) if (options[key] === true) throw new TypeError(`${key} can’t be true with the ${preset} preset`);
139
- if (overrides.every((key) => options[key] === void 0)) return base;
171
+ if (maxLength === void 0 && overrides.every((key) => options[key] === void 0)) return base;
140
172
  return {
141
173
  ...base,
142
174
  checkTld: options.checkTld ?? base.checkTld,
@@ -144,13 +176,27 @@ function resolve(options) {
144
176
  comments: options.allowComments ?? base.comments,
145
177
  unicode: options.allowUnicode ?? base.unicode,
146
178
  idn: options.allowIdn ?? base.idn,
147
- literals: options.allowIpLiteral ?? base.literals
179
+ literals: options.allowIpLiteral ?? base.literals,
180
+ maxLength: maxLength ?? base.maxLength
148
181
  };
149
182
  }
150
183
  //#endregion
151
184
  //#region src/idn.ts
152
185
  const ldh = /^[\da-z-]+$/;
153
186
  /**
187
+ * The hostname of `http://x.` and `host`, or `undefined` when the WHATWG URL
188
+ * parser rejects it. `URL.parse` says so without the cost of throwing, a few
189
+ * µs a time, so an address built to fail many conversions stays fast;
190
+ * runtimes that predate it (Node before 22.1) throw instead.
191
+ */
192
+ const hostnameOf = typeof URL.parse === "function" ? (host) => URL.parse(`http://x.${host}`)?.hostname : (host) => {
193
+ try {
194
+ return new URL(`http://x.${host}`).hostname;
195
+ } catch {
196
+ return;
197
+ }
198
+ };
199
+ /**
154
200
  * The A-label for a U-label, or `undefined` when UTS #46 rejects it or maps
155
201
  * it to something other than one IDN label (plain ASCII, or two labels, as
156
202
  * `。` becomes a dot).
@@ -161,14 +207,65 @@ const ldh = /^[\da-z-]+$/;
161
207
  * checks. It leaves hyphens and lengths alone, so the caller checks those.
162
208
  */
163
209
  function toALabel(label) {
164
- let host;
165
- try {
166
- host = new URL(`http://x.${label}`).hostname;
167
- } catch {
168
- return;
210
+ const aLabel = hostnameOf(label)?.slice(2);
211
+ return aLabel?.startsWith("xn--") === true && ldh.test(aLabel) ? aLabel : void 0;
212
+ }
213
+ const rtl = /[\u0590-\u08ff\u2135-\u2138\ufb1d-\ufdff\ufe70-\ufeff\u{10800}-\u{10fff}\u{1e800}-\u{1efff}]/u;
214
+ /**
215
+ * The A-label for each U-label in `labels`, as {@link toALabel} gives it,
216
+ * converted as few domains as it can: one URL costs about as much as one
217
+ * label does.
218
+ *
219
+ * @remarks
220
+ * Every entry up to the first `undefined` is exact; entries after it may be
221
+ * left `undefined` unconverted. Converted together, a right-to-left label
222
+ * would hold the others to RFC 5893's rules too, which it doesn't alone, so
223
+ * right-to-left labels go in a domain of their own. Converting more labels
224
+ * at once only adds rules, so a domain that converts means every label
225
+ * would alone; one that doesn't is narrowed down to its first label that
226
+ * fails alone.
227
+ */
228
+ function toALabels(labels) {
229
+ if (!labels.some((label) => rtl.test(label))) return convertGroup(labels);
230
+ const aLabels = labels.map(() => void 0);
231
+ for (const bidi of [false, true]) {
232
+ const members = [];
233
+ labels.forEach((label, k) => {
234
+ if (rtl.test(label) === bidi) members.push(k);
235
+ });
236
+ if (members.length > 0) convertGroup(members.map((k) => labels[k])).forEach((aLabel, j) => {
237
+ aLabels[members[j]] = aLabel;
238
+ });
169
239
  }
170
- const aLabel = host.slice(2);
171
- return aLabel.startsWith("xn--") && ldh.test(aLabel) ? aLabel : void 0;
240
+ return aLabels;
241
+ }
242
+ /** {@link toALabels} for labels converted together. */
243
+ function convertGroup(labels) {
244
+ const all = convert(labels);
245
+ if (all !== void 0) return all;
246
+ let pass = [];
247
+ let failing = labels.length;
248
+ while (failing - pass.length > 1) {
249
+ const some = convert(labels.slice(0, pass.length + failing >> 1));
250
+ if (some === void 0) failing = pass.length + failing >> 1;
251
+ else pass = some;
252
+ }
253
+ const aLabels = pass;
254
+ for (let k = pass.length; k < labels.length; k++) {
255
+ const aLabel = toALabel(labels[k]);
256
+ aLabels.push(aLabel);
257
+ if (aLabel === void 0) break;
258
+ }
259
+ return aLabels;
260
+ }
261
+ const aLabelHost = /^x(?:\.xn--[\da-z-]+)+$/;
262
+ /** `labels` converted as one domain, or `undefined`. */
263
+ function convert(labels) {
264
+ const hostname = hostnameOf(labels.join("."));
265
+ if (hostname === void 0 || !aLabelHost.test(hostname)) return;
266
+ const aLabels = hostname.split(".");
267
+ aLabels.shift();
268
+ return aLabels.length === labels.length ? aLabels : void 0;
172
269
  }
173
270
  //#endregion
174
271
  //#region src/ip.ts
@@ -176,6 +273,7 @@ const octet = /^\d{1,3}$/;
176
273
  const hexGroup = /^[\dA-Fa-f]{1,4}$/;
177
274
  /** An IPv4 address: four dot-separated decimal octets, each 0–255. */
178
275
  function isIPv4(text) {
276
+ if (text.length > 15) return false;
179
277
  const parts = text.split(".");
180
278
  return parts.length === 4 && parts.every((part) => octet.test(part) && Number(part) <= 255);
181
279
  }
@@ -185,6 +283,7 @@ function isIPv4(text) {
185
283
  * place of the last two groups.
186
284
  */
187
285
  function isIPv6(text) {
286
+ if (text.length > 45) return false;
188
287
  const colon = text.lastIndexOf(":");
189
288
  const tail = text.slice(colon + 1);
190
289
  if (tail.includes(".")) {
@@ -1668,7 +1767,7 @@ const messages = {
1668
1767
  "syntax.local.too_long": "The local part is longer than 64 characters",
1669
1768
  "syntax.local.invalid_char": "The local part has a character it can’t hold",
1670
1769
  "syntax.local.consecutive_dots": "The local part has two dots in a row",
1671
- "syntax.local.unquoted_space": "The local part has a space outside quotes",
1770
+ "syntax.local.unquoted_space": "The local part has whitespace outside quotes",
1672
1771
  "syntax.domain.empty": "Nothing comes after the @",
1673
1772
  "syntax.domain.no_dot": "The domain has no dot",
1674
1773
  "syntax.domain.label_invalid": "A domain label is empty, too long, or starts or ends with a hyphen",
@@ -1711,8 +1810,9 @@ function findAt(email, rules) {
1711
1810
  let domainStart = false;
1712
1811
  let depth = 0;
1713
1812
  let at = -1;
1714
- for (let i = 0; i < email.length; i++) {
1715
- const code = email.charCodeAt(i);
1813
+ const end = email.length;
1814
+ for (let i = 0; i < end; i++) {
1815
+ const code = codeAt(email, i);
1716
1816
  if (state === NORMAL) {
1717
1817
  if (code === 64) {
1718
1818
  at = i;
@@ -1743,6 +1843,7 @@ function findAt(email, rules) {
1743
1843
  function parse(email, rules) {
1744
1844
  if (typeof email !== "string") throw new TypeError(`Expected a string, got ${typeof email}`);
1745
1845
  if (email === "") return fail("syntax.address.empty");
1846
+ if (email.length > rules.maxLength) return fail("syntax.address.too_long");
1746
1847
  const at = findAt(email, rules);
1747
1848
  if (at < 0) return fail("syntax.address.no_at");
1748
1849
  const comments = [];
@@ -1786,7 +1887,7 @@ function scanLocal(email, at, rules, comments) {
1786
1887
  let trailing = comments.length;
1787
1888
  let i = 0;
1788
1889
  while (i < at) {
1789
- const code = email.charCodeAt(i);
1890
+ const code = codeAt(email, i);
1790
1891
  const quote = code === 34 && rules.quotes && (rules.obs || i === 0);
1791
1892
  const atom = inClass(code, rules.local);
1792
1893
  if (quote || atom || rules.unicode && nonAsciiAt(email, i, at) > 0) {
@@ -1799,10 +1900,10 @@ function scanLocal(email, at, rules, comments) {
1799
1900
  closed = !rules.obs;
1800
1901
  } else {
1801
1902
  next = atom ? i + 1 : i;
1802
- while (next < at && inClass(email.charCodeAt(next), rules.local)) next++;
1903
+ while (next < at && inClass(codeAt(email, next), rules.local)) next++;
1803
1904
  if (rules.unicode) next = skipUnicode(email, next, at, rules.local);
1804
1905
  }
1805
- local += unfold(email.slice(i, next));
1906
+ local += unfold(email, i, next);
1806
1907
  words++;
1807
1908
  afterWord = true;
1808
1909
  dot = -1;
@@ -1847,6 +1948,63 @@ function scanLocal(email, at, rules, comments) {
1847
1948
  * folding whitespace.
1848
1949
  */
1849
1950
  function scanDomain(email, at, rules, comments) {
1951
+ const uLabels = [];
1952
+ const domain = scanLabels(email, at, rules, comments, uLabels);
1953
+ if (uLabels.length === 0) return domain;
1954
+ const growth = checkULabels(email, uLabels, rules.domainCap);
1955
+ if (typeof growth !== "number") return growth;
1956
+ if (!("reason" in domain)) domain.size += growth;
1957
+ return domain;
1958
+ }
1959
+ /**
1960
+ * Checks the U-labels at `spans`, triples of start, end, and the length of
1961
+ * the domain before it, as A-labels: each must convert, and pass the
1962
+ * hostname rules as its A-label. Returns the first failure, or what the
1963
+ * A-labels add to the domain's length: `Infinity` once it's past `cap`.
1964
+ *
1965
+ * @remarks
1966
+ * The U-labels are converted a few at a time, about 32 characters of them
1967
+ * in each URL: one URL costs about as much as one label alone, and a small
1968
+ * batch wastes little past the first failure or past `cap`. Once the
1969
+ * A-labels take the domain past `cap`, the rest aren't checked: the domain
1970
+ * fails as too long, unless the scan found something earlier. Punycode's
1971
+ * cost grows with the square of a label's length, so converting labels that
1972
+ * can't change the outcome would cost the most on input built to be slow
1973
+ * (validator-syntax#15).
1974
+ */
1975
+ function checkULabels(email, spans, cap) {
1976
+ let growth = 0;
1977
+ for (let first = 0; first < spans.length;) {
1978
+ const chunk = [];
1979
+ let written = 0;
1980
+ let next = first;
1981
+ while (next < spans.length && written < 32) {
1982
+ const label = email.slice(spans[next], spans[next + 1]);
1983
+ chunk.push(label);
1984
+ written += label.length;
1985
+ next += 3;
1986
+ }
1987
+ const aLabels = toALabels(chunk);
1988
+ for (let k = 0; k < chunk.length; k++) {
1989
+ const start = spans[first + 3 * k];
1990
+ const end = spans[first + 3 * k + 1];
1991
+ const aLabel = aLabels[k];
1992
+ if (aLabel === void 0) return fail("syntax.domain.label_invalid", start);
1993
+ const bad = checkLabel(email, start, end, aLabel.length);
1994
+ if (bad >= 0) return fail("syntax.domain.label_invalid", bad);
1995
+ growth += aLabel.length - (end - start);
1996
+ if (spans[first + 3 * k + 2] + (end - start) + growth > cap) return Infinity;
1997
+ }
1998
+ first = next;
1999
+ }
2000
+ return growth;
2001
+ }
2002
+ /**
2003
+ * The scan behind {@link scanDomain}, which leaves the U-labels it finds in
2004
+ * `uLabels`, as triples of start, end, and the domain's length before it,
2005
+ * unchecked but for their length.
2006
+ */
2007
+ function scanLabels(email, at, rules, comments, uLabels) {
1850
2008
  const end = email.length;
1851
2009
  let domain = "";
1852
2010
  let labels = 0;
@@ -1857,10 +2015,9 @@ function scanDomain(email, at, rules, comments) {
1857
2015
  let edge = -1;
1858
2016
  let trailing = comments.length;
1859
2017
  let tld = "";
1860
- let growth = 0;
1861
2018
  let i = at + 1;
1862
2019
  while (i < end) {
1863
- const code = email.charCodeAt(i);
2020
+ const code = codeAt(email, i);
1864
2021
  const opensLiteral = code === 91 && rules.literals && labels === 0;
1865
2022
  const ldh = inClass(code, rules.domain);
1866
2023
  if (opensLiteral || ldh || rules.idn && nonAsciiAt(email, i, end) > 0) {
@@ -1873,20 +2030,18 @@ function scanDomain(email, at, rules, comments) {
1873
2030
  literal = true;
1874
2031
  } else {
1875
2032
  next = ldh ? i + 1 : i;
1876
- while (next < end && inClass(email.charCodeAt(next), rules.domain)) next++;
2033
+ while (next < end && inClass(codeAt(email, next), rules.domain)) next++;
1877
2034
  const ascii = next;
1878
2035
  if (rules.idn) next = skipUnicode(email, next, end, rules.domain);
1879
- let size = next - i;
1880
2036
  if (next > ascii) {
1881
- const aLabel = toALabel(email.slice(i, next));
1882
- if (aLabel === void 0) return fail("syntax.domain.label_invalid", i);
1883
- size = aLabel.length;
1884
- growth += size - (next - i);
2037
+ if (codePoints(email, i, next) > 63) return fail("syntax.domain.label_invalid", i);
2038
+ uLabels.push(i, next, domain.length);
2039
+ } else {
2040
+ const bad = checkLabel(email, i, next, next - i);
2041
+ if (bad >= 0) return fail("syntax.domain.label_invalid", bad);
1885
2042
  }
1886
- const bad = checkLabel(email, i, next, size);
1887
- if (bad >= 0) return fail("syntax.domain.label_invalid", bad);
1888
2043
  }
1889
- tld = unfold(email.slice(i, next));
2044
+ tld = unfold(email, i, next);
1890
2045
  domain += tld;
1891
2046
  labels++;
1892
2047
  afterLabel = true;
@@ -1925,7 +2080,7 @@ function scanDomain(email, at, rules, comments) {
1925
2080
  if (labels === 0) return fail("syntax.domain.empty");
1926
2081
  if (dot >= 0) return fail("syntax.domain.label_invalid", dot);
1927
2082
  settle(comments, trailing, "after-domain");
1928
- const size = domain.length + growth;
2083
+ const size = domain.length;
1929
2084
  return literal || labels === 1 ? {
1930
2085
  domain,
1931
2086
  literal,
@@ -1945,7 +2100,7 @@ function skipUnicode(email, i, end, flag) {
1945
2100
  let width = i < end ? nonAsciiAt(email, i, end) : 0;
1946
2101
  while (width > 0) {
1947
2102
  i += width;
1948
- while (i < end && inClass(email.charCodeAt(i), flag)) i++;
2103
+ while (i < end && inClass(codeAt(email, i), flag)) i++;
1949
2104
  width = i < end ? nonAsciiAt(email, i, end) : 0;
1950
2105
  }
1951
2106
  return i;
@@ -1956,9 +2111,9 @@ function skipUnicode(email, i, end, flag) {
1956
2111
  * a label over 63 characters, or -1.
1957
2112
  */
1958
2113
  function checkLabel(email, start, end, size) {
1959
- if (email.charCodeAt(start) === 45) return start;
2114
+ if (codeAt(email, start) === 45) return start;
1960
2115
  if (size > 63) return start;
1961
- return email.charCodeAt(end - 1) === 45 ? end - 1 : -1;
2116
+ return codeAt(email, end - 1) === 45 ? end - 1 : -1;
1962
2117
  }
1963
2118
  /** Marks the comments from `from` on, which no word follows, as trailing. */
1964
2119
  function settle(comments, from, position) {
@@ -1967,18 +2122,23 @@ function settle(comments, from, position) {
1967
2122
  if (comment.position.startsWith("inside")) comment.position = position;
1968
2123
  }
1969
2124
  }
1970
- /** Removes the CRLFs of folding whitespace, keeping the space or tab after. */
1971
- function unfold(text) {
1972
- return text.includes("\r") ? text.replaceAll("\r\n", "") : text;
2125
+ const { indexOf, slice } = String.prototype;
2126
+ /**
2127
+ * `email` from `start` to `end`, without the CRLFs of folding whitespace,
2128
+ * keeping the space or tab after each.
2129
+ */
2130
+ function unfold(email, start, end) {
2131
+ const text = slice.call(email, start, end);
2132
+ return indexOf.call(text, "\r") < 0 ? text : text.replaceAll("\r\n", "");
1973
2133
  }
1974
2134
  /**
1975
2135
  * Skips one character of folding whitespace: a space, a tab, or a CRLF with
1976
2136
  * a space or tab after it. A lone CR or LF breaks it.
1977
2137
  */
1978
2138
  function skipSpace(email, i, end) {
1979
- const code = email.charCodeAt(i);
2139
+ const code = codeAt(email, i);
1980
2140
  if (code === 32 || code === 9) return i + 1;
1981
- return code === 13 && i + 2 < end && email.charCodeAt(i + 1) === 10 && (email.charCodeAt(i + 2) === 32 || email.charCodeAt(i + 2) === 9) ? i + 2 : -1 - i;
2141
+ return code === 13 && i + 2 < end && codeAt(email, i + 1) === 10 && (codeAt(email, i + 2) === 32 || codeAt(email, i + 2) === 9) ? i + 2 : -1 - i;
1982
2142
  }
1983
2143
  /** Whether a backslash may escape `code`: any ASCII with `obs`, else VCHAR and space. */
1984
2144
  function escapable(code, obs) {
@@ -1992,10 +2152,10 @@ function escapable(code, obs) {
1992
2152
  function skipQuoted(email, open, end, obs, unicode) {
1993
2153
  let i = open + 1;
1994
2154
  while (i < end) {
1995
- const code = email.charCodeAt(i);
2155
+ const code = codeAt(email, i);
1996
2156
  if (code === 34) return i + 1;
1997
2157
  if (code === 92) {
1998
- if (!escapable(email.charCodeAt(i + 1), obs)) return -2 - i;
2158
+ if (!escapable(codeAt(email, i + 1), obs)) return -2 - i;
1999
2159
  i += 2;
2000
2160
  } else if (obs && isWhitespace(code)) {
2001
2161
  i = skipSpace(email, i, end);
@@ -2017,14 +2177,14 @@ function skipComment(email, open, end, unicode) {
2017
2177
  let depth = 1;
2018
2178
  let i = open + 1;
2019
2179
  while (i < end) {
2020
- const code = email.charCodeAt(i);
2180
+ const code = codeAt(email, i);
2021
2181
  if (code === 40 || code === 41) {
2022
2182
  depth += code === 40 ? 1 : -1;
2023
2183
  i++;
2024
2184
  if (depth === 0) return i;
2025
2185
  } else if (code === 92) {
2026
2186
  if (i + 1 >= end) break;
2027
- if (!escapable(email.charCodeAt(i + 1), true)) return -2 - i;
2187
+ if (!escapable(codeAt(email, i + 1), true)) return -2 - i;
2028
2188
  i += 2;
2029
2189
  } else if (isWhitespace(code)) {
2030
2190
  i = skipSpace(email, i, end);
@@ -2045,10 +2205,10 @@ function skipComment(email, open, end, unicode) {
2045
2205
  function skipLiteral(email, open, end) {
2046
2206
  let i = open + 1;
2047
2207
  while (i < end) {
2048
- const code = email.charCodeAt(i);
2208
+ const code = codeAt(email, i);
2049
2209
  if (code === 93) return i + 1;
2050
2210
  if (code === 92) {
2051
- if (i + 1 >= end || !escapable(email.charCodeAt(i + 1), true)) return -1;
2211
+ if (i + 1 >= end || !escapable(codeAt(email, i + 1), true)) return -1;
2052
2212
  i += 2;
2053
2213
  } else if (isWhitespace(code)) {
2054
2214
  i = skipSpace(email, i, end);
@@ -2079,18 +2239,20 @@ function skipAddressLiteral(email, open, end) {
2079
2239
  * the first check it fails.
2080
2240
  *
2081
2241
  * @remarks
2082
- * The local part is checked before the domain, each left to right, then the
2083
- * lengths, then a dotless domain, then the TLD. `index`, where a failure has
2084
- * one, points at the offending character.
2242
+ * Input longer than `maxLength` (512 UTF-16 code units by default) fails as
2243
+ * `syntax.address.too_long` before it's scanned. Otherwise the local part is
2244
+ * checked before the domain, each left to right, then the lengths, then a
2245
+ * dotless domain, then the TLD. `index`, where a failure has one, points at
2246
+ * the offending character.
2085
2247
  *
2086
2248
  * @example
2087
2249
  * ```ts
2088
- * const result = parseAddress('ada.lovelace@example.co.uk');
2089
- * if (result.ok) {
2090
- * result.value.tld; // 'uk'
2091
- * } else {
2092
- * result.reason; // e.g. 'syntax.local.invalid_char'
2093
- * }
2250
+ * import { parseAddress } from '@email-utils/validator-syntax';
2251
+ *
2252
+ * parseAddress('ada.lovelace@example.co.uk');
2253
+ * // => { ok: true, value: { local: 'ada.lovelace', tld: 'uk' } }
2254
+ * parseAddress('ada..lovelace@example.com');
2255
+ * // => { ok: false, reason: 'syntax.local.consecutive_dots', index: 4 }
2094
2256
  * ```
2095
2257
  *
2096
2258
  * @throws TypeError when `email` isn't a string, or `options` are malformed.
@@ -2103,9 +2265,11 @@ function parseAddress(email, options) {
2103
2265
  *
2104
2266
  * @example
2105
2267
  * ```ts
2106
- * isValidSyntax('ada@example.com'); // true
2107
- * isValidSyntax('"ada"@example.com'); // false
2108
- * isValidSyntax('"ada"@example.com', { preset: 'rfc5321' }); // true
2268
+ * import { isValidSyntax } from '@email-utils/validator-syntax';
2269
+ *
2270
+ * isValidSyntax('ada@example.com'); // => true
2271
+ * isValidSyntax('"ada"@example.com'); // => false
2272
+ * isValidSyntax('"ada"@example.com', { preset: 'rfc5321' }); // => true
2109
2273
  * ```
2110
2274
  *
2111
2275
  * @throws TypeError when `email` isn't a string, or `options` are malformed.
@@ -2119,9 +2283,14 @@ function isValidSyntax(email, options) {
2119
2283
  *
2120
2284
  * @example
2121
2285
  * ```ts
2286
+ * import { createSyntaxValidator } from '@email-utils/validator-syntax';
2287
+ *
2122
2288
  * const strict = createSyntaxValidator({ preset: 'rfc5321' });
2123
- * strict.parse('a b@example.com'); // { ok: false, reason: 'syntax.local.unquoted_space', … }
2124
- * strict.isValid('"a b"@example.com'); // true
2289
+ * strict.parse('a b@example.com');
2290
+ * // => { ok: false, reason: 'syntax.local.unquoted_space', index: 1 }
2291
+ * strict.isValid('"a b"@example.com'); // => true
2292
+ * createSyntaxValidator({ preset: 'html5', allowComments: true });
2293
+ * // => throws TypeError
2125
2294
  * ```
2126
2295
  *
2127
2296
  * @throws TypeError when `options` are malformed.