n3 3.0.0-alpha.4 → 3.0.0-alpha.6

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/lib/N3Lexer.js CHANGED
@@ -4,7 +4,6 @@ Object.defineProperty(exports, "__esModule", {
4
4
  value: true
5
5
  });
6
6
  exports.default = void 0;
7
- var _buffer = require("buffer");
8
7
  var _IRIs = _interopRequireDefault(require("./IRIs"));
9
8
  function _interopRequireDefault(e) { return e && e.__esModule ? e : { default: e }; }
10
9
  // **N3Lexer** tokenizes N3 documents.
@@ -54,7 +53,8 @@ const localNameEscapeReplacements = {
54
53
  };
55
54
  const illegalIriChars = /[\x00-\x20<>\\"\{\}\|\^\`]/;
56
55
  // Characters that cannot occur in a prefixed name, not even escaped
57
- const nonPrefixedNameChar = /[\s<>"{}|^`]/;
56
+ // (global, so that testAt searches the rest of the input from a position)
57
+ const nonPrefixedNameChar = /[\s<>"{}|^`]/g;
58
58
 
59
59
  // A valid code point is a Unicode scalar value: at most U+10FFFF and not a surrogate
60
60
  function isValidCodePoint(charCode) {
@@ -70,30 +70,59 @@ const lineModeRegExps = {
70
70
  _whitespace: true
71
71
  };
72
72
  const invalidRegExp = /$0^/;
73
+ const nonWhitespace = /\S*/y;
74
+
75
+ // Matches a sticky regular expression at the given position of the input
76
+ function execAt(regExp, input, pos) {
77
+ regExp.lastIndex = pos;
78
+ return regExp.exec(input);
79
+ }
80
+ function testAt(regExp, input, pos) {
81
+ regExp.lastIndex = pos;
82
+ return regExp.test(input);
83
+ }
84
+ // Matches the rest of the input followed by a space, as at the end of the input,
85
+ // a token that can contain (but not end with) a dot needs a non-dot character after it
86
+ function execAtEnd(regExp, input, pos) {
87
+ regExp.lastIndex = 0;
88
+ return regExp.exec(`${input.slice(pos)} `);
89
+ }
90
+
91
+ // Whitespace or the start of a comment
92
+ function isSeparatorCode(code) {
93
+ return code === SPACE || code === TAB || code === LF || code === CR || code === HASH;
94
+ }
95
+
96
+ // Words with a fixed meaning in the grammar, which cannot name an additional directive
97
+ const reservedWords = /^(?:prefix|base|version|graph|forsome|forall|iri|a|true|false|has|is|of|id)$/i;
98
+
99
+ // Unfinished input in a stream up to this length is tokenized again with every chunk
100
+ const MIN_RESCAN_LENGTH = 1024;
73
101
 
74
102
  // ## Constructor
75
103
  class N3Lexer {
76
104
  constructor(options) {
77
105
  // ## Regular expressions
78
- // It's slightly faster to have these as properties than as in-scope variables
79
- this._iri = /^<((?:[^ <>{}\\]|\\[uU])+)>[ \t]*/; // IRI with escape sequences; needs sanity check after unescaping
80
- this._unescapedIri = /^<([^\x00-\x20<>\\"\{\}\|\^\`]*)>[ \t]*/; // IRI without escape sequences; no unescaping
81
- this._simpleQuotedString = /^"([^"\\\r\n]*)"(?=[^"])/; // string without escape sequences
82
- this._simpleApostropheString = /^'([^'\\\r\n]*)'(?=[^'])/;
83
- this._langcode = /^@([a-z]+(?:-[a-z0-9]+)*)(?=[^a-z0-9])/i;
84
- this._prefix = /^((?:[A-Za-z\xc0-\xd6\xd8-\xf6\xf8-\u02ff\u0370-\u037d\u037f-\u1fff\u200c\u200d\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff])(?:\.?[\-0-9A-Z_a-z\xb7\xc0-\xd6\xd8-\xf6\xf8-\u037d\u037f-\u1fff\u200c\u200d\u203f\u2040\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff])*)?:(?=[#\s<])/;
85
- this._prefixed = /^((?:[A-Za-z\xc0-\xd6\xd8-\xf6\xf8-\u02ff\u0370-\u037d\u037f-\u1fff\u200c\u200d\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff])(?:\.?[\-0-9A-Z_a-z\xb7\xc0-\xd6\xd8-\xf6\xf8-\u037d\u037f-\u1fff\u200c\u200d\u203f\u2040\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff])*)?:((?:(?:[0-9:A-Z_a-z\xc0-\xd6\xd8-\xf6\xf8-\u02ff\u0370-\u037d\u037f-\u1fff\u200c\u200d\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff]|%[0-9a-fA-F]{2}|\\[!#-\/;=?\-@_~])(?:(?:[\.\-0-9:A-Z_a-z\xb7\xc0-\xd6\xd8-\xf6\xf8-\u037d\u037f-\u1fff\u200c\u200d\u203f\u2040\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff]|%[0-9a-fA-F]{2}|\\[!#-\/;=?\-@_~])*(?:[\-0-9:A-Z_a-z\xb7\xc0-\xd6\xd8-\xf6\xf8-\u037d\u037f-\u1fff\u200c\u200d\u203f\u2040\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff]|%[0-9a-fA-F]{2}|\\[!#-\/;=?\-@_~]))?)?)(?:[ \t]+|(?=\.?[,;!\^\s#()\[\]\{\}"'<>]))/;
86
- this._variable = /^\?(?:(?:[A-Z_a-z\xc0-\xd6\xd8-\xf6\xf8-\u02ff\u0370-\u037d\u037f-\u1fff\u200c\u200d\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff])(?:[\-0-9:A-Z_a-z\xb7\xc0-\xd6\xd8-\xf6\xf8-\u037d\u037f-\u1fff\u200c\u200d\u203f\u2040\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff])*)(?=[.,;!\^\s#()\[\]\{\}"'<>])/;
87
- this._blank = /^_:((?:[0-9A-Z_a-z\xc0-\xd6\xd8-\xf6\xf8-\u02ff\u0370-\u037d\u037f-\u1fff\u200c\u200d\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff])(?:\.?[\-0-9A-Z_a-z\xb7\xc0-\xd6\xd8-\xf6\xf8-\u037d\u037f-\u1fff\u200c\u200d\u203f\u2040\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff])*)(?:[ \t]+|(?=\.?[,;:!\^\s#()\[\]\{\}"'<>]))/;
88
- this._number = /^[\-+]?(?:(\d+\.\d*|\.?\d+)[eE][\-+]?\d+|(?=\.?\d)\d*(?:(\.)\d+)?)(?=\.?[,;:!\^\s#()\[\]\{\}"'<>])/;
89
- this._boolean = /^(?:true|false)(?=[.,;!\^\s#()\[\]\{\}"'<>])/;
90
- this._atKeyword = /^@[a-z]+(?=[\s#<:])/i;
91
- this._keyword = /^(?:PREFIX|BASE|VERSION|GRAPH)(?=[\s#<])/i;
92
- this._n3Verb = /^(?:has|is|of)(?=[\s#()\[\]\{\}"'<>?_+\-0-9])/;
93
- this._n3Id = /^id(?=[\s#<])/;
94
- this._shortPredicates = /^a(?=[\s#()\[\]\{\}"'<>])/;
95
- this._commentLine = /^[ \t]*#([^\n\r]*)(?:\r\n|\n|\r)([ \t]*)/;
96
- this._whitespace = /^[ \t]+/;
106
+ // It's slightly faster to have these as properties than as in-scope variables.
107
+ // They are sticky, so they only match at the `lastIndex` set by `execAt`.
108
+ this._iri = /<((?:[^ <>{}\\]|\\[uU])+)>[ \t]*/y; // IRI with escape sequences; needs sanity check after unescaping
109
+ this._unescapedIri = /<([^\x00-\x20<>\\"\{\}\|\^\`]*)>[ \t]*/y; // IRI without escape sequences; no unescaping
110
+ this._simpleQuotedString = /"([^"\\\r\n]*)"(?=[^"])/y; // string without escape sequences
111
+ this._simpleApostropheString = /'([^'\\\r\n]*)'(?=[^'])/y;
112
+ this._langcode = /@([a-z]+(?:-[a-z0-9]+)*)(?=[^a-z0-9])/iy;
113
+ this._prefix = /((?:[A-Za-z\xc0-\xd6\xd8-\xf6\xf8-\u02ff\u0370-\u037d\u037f-\u1fff\u200c\u200d\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff])(?:\.?[\-0-9A-Z_a-z\xb7\xc0-\xd6\xd8-\xf6\xf8-\u037d\u037f-\u1fff\u200c\u200d\u203f\u2040\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff])*)?:(?=[#\s<])/y;
114
+ this._prefixed = /((?:[A-Za-z\xc0-\xd6\xd8-\xf6\xf8-\u02ff\u0370-\u037d\u037f-\u1fff\u200c\u200d\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff])(?:\.?[\-0-9A-Z_a-z\xb7\xc0-\xd6\xd8-\xf6\xf8-\u037d\u037f-\u1fff\u200c\u200d\u203f\u2040\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff])*)?:((?:(?:[0-9:A-Z_a-z\xc0-\xd6\xd8-\xf6\xf8-\u02ff\u0370-\u037d\u037f-\u1fff\u200c\u200d\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff]|%[0-9a-fA-F]{2}|\\[!#-\/;=?\-@_~])(?:(?:[\.\-0-9:A-Z_a-z\xb7\xc0-\xd6\xd8-\xf6\xf8-\u037d\u037f-\u1fff\u200c\u200d\u203f\u2040\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff]|%[0-9a-fA-F]{2}|\\[!#-\/;=?\-@_~])*(?:[\-0-9:A-Z_a-z\xb7\xc0-\xd6\xd8-\xf6\xf8-\u037d\u037f-\u1fff\u200c\u200d\u203f\u2040\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff]|%[0-9a-fA-F]{2}|\\[!#-\/;=?\-@_~]))?)?)(?:[ \t]+|(?=\.?[,;!\^\s#()\[\]\{\}"'<>]))/y;
115
+ this._variable = /\?(?:(?:[A-Z_a-z\xc0-\xd6\xd8-\xf6\xf8-\u02ff\u0370-\u037d\u037f-\u1fff\u200c\u200d\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff])(?:[\-0-9:A-Z_a-z\xb7\xc0-\xd6\xd8-\xf6\xf8-\u037d\u037f-\u1fff\u200c\u200d\u203f\u2040\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff])*)(?=[.,;!\^\s#()\[\]\{\}"'<>])/y;
116
+ this._blank = /_:((?:[0-9A-Z_a-z\xc0-\xd6\xd8-\xf6\xf8-\u02ff\u0370-\u037d\u037f-\u1fff\u200c\u200d\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff])(?:\.?[\-0-9A-Z_a-z\xb7\xc0-\xd6\xd8-\xf6\xf8-\u037d\u037f-\u1fff\u200c\u200d\u203f\u2040\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff])*)(?:[ \t]+|(?=\.?[,;:!\^\s#()\[\]\{\}"'<>]))/y;
117
+ this._number = /[\-+]?(?:(\d+\.\d*|\.?\d+)[eE][\-+]?\d+|(?=\.?\d)\d*(?:(\.)\d+)?)(?=\.?[,;:!\^\s#()\[\]\{\}"'<>])/y;
118
+ this._boolean = /(?:true|false)(?=[.,;!\^\s#()\[\]\{\}"'<>])/y;
119
+ this._atKeyword = /@[a-z]+(?=[\s#<:"'])/iy;
120
+ this._keyword = /(?:PREFIX|BASE|VERSION|GRAPH)(?=[\s#<"'])/iy;
121
+ this._n3Verb = /(?:has|is|of)(?=[\s#()\[\]\{\}"'<>?_+\-0-9])/y;
122
+ this._n3Id = /id(?=[\s#<])/y;
123
+ this._shortPredicates = /a(?=[\s#()\[\]\{\}"'<>])/y;
124
+ this._commentLine = /[ \t]*#([^\n\r]*)(?:\r\n|\n|\r)([ \t]*)/y;
125
+ this._whitespace = /[ \t]+/y;
97
126
  options = options || {};
98
127
 
99
128
  // Whether the log:isImpliedBy predicate is supported
@@ -106,11 +135,25 @@ class N3Lexer {
106
135
  for (const key in this) {
107
136
  if (!(key in lineModeRegExps) && this[key] instanceof RegExp) this[key] = invalidRegExp;
108
137
  }
138
+ // The only keyword in N-Triples and N-Quads is VERSION, which is case-sensitive
139
+ this._keyword = /VERSION(?=[\s#<"])/y;
109
140
  }
110
141
  // When not in line mode, enable N3 functionality by default
111
142
  else {
112
143
  this._n3Mode = options.n3 !== false;
113
144
  }
145
+ // Recognize additional directive keywords, such as MESSAGE
146
+ // (the @-form of a directive is always tokenized as an @-keyword)
147
+ this._directive = null;
148
+ if (options.directives && options.directives.length !== 0) {
149
+ for (const name of options.directives) {
150
+ if (!/^[a-z]+$/i.test(name) || reservedWords.test(name)) throw new Error(`Invalid directive name: "${name}"`);
151
+ }
152
+ this._directive = new RegExp(`(?:${options.directives.join('|')})(?=[\\s#<])`, 'iy');
153
+ this._directiveMaxLength = Math.max(...options.directives.map(name => name.length));
154
+ // The first characters of directive names, so other words skip the regular expression
155
+ this._directiveStarts = options.directives.map(name => name[0].toLowerCase() + name[0].toUpperCase()).join('');
156
+ }
114
157
  // Don't output comment tokens by default
115
158
  this.comments = !!options.comments;
116
159
  // Cache the last tested closing position of long literals
@@ -121,77 +164,73 @@ class N3Lexer {
121
164
 
122
165
  // ### `_tokenizeToEnd` tokenizes as for as possible, emitting tokens through the callback
123
166
  _tokenizeToEnd(callback, inputFinished) {
124
- // Continue parsing as far as possible; the loop will return eventually
125
- let input = this._input;
167
+ // Continue parsing as far as possible; the loop will return eventually.
168
+ // Rather than slicing off every token, track the position of the remaining input;
169
+ // the regular expressions are sticky, so they match at that position.
170
+ const input = this._input;
171
+ let pos = 0;
126
172
  let currentLineLength = this._linePosition + input.length;
127
173
  while (true) {
128
174
  // Consume one separator line at a time, including its following indentation.
129
175
  while (true) {
130
- let charCode = input.charCodeAt(0),
176
+ let charCode = input.charCodeAt(pos),
131
177
  separatorLength = 0;
132
178
  if (charCode === SPACE || charCode === TAB) {
133
- const next = input.charCodeAt(1);
134
- separatorLength = next === SPACE || next === TAB ? this._whitespace.exec(input)[0].length : 1;
135
- charCode = input.charCodeAt(separatorLength);
179
+ const next = input.charCodeAt(pos + 1);
180
+ separatorLength = next === SPACE || next === TAB ? execAt(this._whitespace, input, pos)[0].length : 1;
181
+ charCode = input.charCodeAt(pos + separatorLength);
136
182
  }
137
183
  if (charCode === HASH) {
138
- const comment = this._commentLine.exec(input);
184
+ const comment = execAt(this._commentLine, input, pos);
139
185
  if (comment) {
140
186
  const commentLength = comment[0].length;
141
187
  // Keep a trailing CR buffered in case the next chunk starts with LF.
142
- if (!inputFinished && commentLength === input.length && input.charCodeAt(commentLength - 1) === CR) {
143
- this._linePosition = currentLineLength - input.length;
144
- return this._input = input;
145
- }
146
- if (this.comments) emitComment(comment[1], this._line, separatorLength);
147
- input = input.slice(commentLength);
148
- currentLineLength = input.length + comment[2].length;
188
+ if (!inputFinished && pos + commentLength === input.length && input.charCodeAt(input.length - 1) === CR) return this._suspend(input, pos, currentLineLength);
189
+ if (this.comments) emitComment(comment[1], this._line, currentLineLength - (input.length - pos) + separatorLength);
190
+ pos += commentLength;
191
+ currentLineLength = input.length - pos + comment[2].length;
149
192
  this._line++;
150
193
  } else {
151
194
  // A comment without a line ending stays buffered until EOF.
152
- input = input.slice(separatorLength);
153
- if (!inputFinished) {
154
- this._linePosition = currentLineLength - input.length;
155
- return this._input = input;
156
- }
157
- if (this.comments) emitComment(input.slice(1), this._line, 0);
158
- input = '';
195
+ pos += separatorLength;
196
+ if (!inputFinished) return this._suspend(input, pos, currentLineLength);
197
+ if (this.comments) emitComment(input.slice(pos + 1), this._line, currentLineLength - (input.length - pos));
198
+ pos = input.length;
159
199
  break;
160
200
  }
161
201
  } else if (charCode === LF || charCode === CR) {
162
202
  // A CR at the end of a chunk may still be followed by LF.
163
- if (!inputFinished && charCode === CR && separatorLength + 1 === input.length) {
164
- this._linePosition = currentLineLength - input.length;
165
- return this._input = input;
166
- }
167
- separatorLength += charCode === CR && input.charCodeAt(separatorLength + 1) === LF ? 2 : 1;
203
+ if (!inputFinished && charCode === CR && pos + separatorLength + 1 === input.length) return this._suspend(input, pos, currentLineLength);
204
+ separatorLength += charCode === CR && input.charCodeAt(pos + separatorLength + 1) === LF ? 2 : 1;
168
205
  // Indentation is consumed with the newline, but belongs to the next line's columns.
169
206
  let indentationLength = 0;
170
- const next = input.charCodeAt(separatorLength);
207
+ const next = input.charCodeAt(pos + separatorLength);
171
208
  if (next === SPACE || next === TAB) {
172
- const following = input.charCodeAt(separatorLength + 1);
173
- indentationLength = following === SPACE || following === TAB ? this._whitespace.exec(input.slice(separatorLength))[0].length : 1;
209
+ const following = input.charCodeAt(pos + separatorLength + 1);
210
+ indentationLength = following === SPACE || following === TAB ? execAt(this._whitespace, input, pos + separatorLength)[0].length : 1;
174
211
  }
175
- input = input.slice(separatorLength + indentationLength);
176
- currentLineLength = input.length + indentationLength;
212
+ pos += separatorLength + indentationLength;
213
+ currentLineLength = input.length - pos + indentationLength;
177
214
  this._line++;
178
215
  } else {
179
- if (separatorLength !== 0) input = input.slice(separatorLength);
216
+ pos += separatorLength;
180
217
  break;
181
218
  }
182
219
  }
183
- if (input.length === 0) {
220
+ if (pos >= input.length) {
221
+ // A datatype marker needs a type
222
+ if (inputFinished && this._previousMarker === '^^') return reportSyntaxError(this, input, pos);
223
+ this._linePosition = currentLineLength;
184
224
  if (inputFinished) {
185
- input = null;
186
- emitToken('eof', '', '', this._line, 0);
225
+ emitToken('eof', '', '', this._line, currentLineLength, 0);
226
+ return this._input = null;
187
227
  }
188
- this._linePosition = currentLineLength;
189
- return this._input = input;
228
+ return this._input = '';
190
229
  }
191
230
 
192
231
  // Look for specific token types based on the first character
193
232
  const line = this._line,
194
- firstChar = input[0];
233
+ firstChar = input[pos];
195
234
  let type = '',
196
235
  value = '',
197
236
  prefix = '',
@@ -199,17 +238,22 @@ class N3Lexer {
199
238
  matchLength = 0,
200
239
  lexicalLength = 0,
201
240
  finalLineLength = 0,
202
- inconclusive = false;
241
+ inconclusive = false,
242
+ tripleQuoted = false;
203
243
  switch (firstChar) {
204
244
  case '^':
245
+ // A datatype marker separated from its type cannot be followed by another marker
246
+ if (this._previousMarker === '^^') return reportSyntaxError(this, input, pos);
205
247
  // We need at least 3 tokens lookahead to distinguish ^^<IRI> and ^^pre:fixed
206
- if (input.length < 3) break;
248
+ if (input.length - pos < 3) break;
207
249
  // Try to match a type
208
- else if (input[1] === '^') {
250
+ else if (input[pos + 1] === '^') {
209
251
  this._previousMarker = '^^';
210
252
  // Move to type IRI or prefixed name
211
- input = input.slice(2);
212
- if (input[0] !== '<') {
253
+ pos += 2;
254
+ if (input[pos] !== '<') {
255
+ // Whitespace and comments may separate the marker from the type
256
+ if (isSeparatorCode(input.charCodeAt(pos))) continue; // eslint-disable-line no-continue
213
257
  inconclusive = true;
214
258
  break;
215
259
  }
@@ -225,53 +269,54 @@ class N3Lexer {
225
269
  // Fall through in case the type is an IRI
226
270
  case '<':
227
271
  // Try to find a full IRI without escape sequences
228
- if (match = this._unescapedIri.exec(input)) {
272
+ if (match = execAt(this._unescapedIri, input, pos)) {
229
273
  type = 'IRI', value = match[1];
230
274
  lexicalLength = match[1].length + 2;
231
275
  }
232
276
  // Try to find a full IRI with escape sequences
233
- else if (match = this._iri.exec(input)) {
277
+ else if (match = execAt(this._iri, input, pos)) {
234
278
  value = this._unescape(match[1], stringEscapeReplacements);
235
- if (value === null || illegalIriChars.test(value)) return reportSyntaxError(this);
279
+ if (value === null || illegalIriChars.test(value)) return reportSyntaxError(this, input, pos);
236
280
  type = 'IRI';
237
281
  lexicalLength = match[1].length + 2;
238
282
  }
239
283
  // Try to find a triple term
240
- else if (input.length > 2 && input[1] === '<' && input[2] === '(') type = '<<(', matchLength = 3;
284
+ else if (input.length - pos > 2 && input[pos + 1] === '<' && input[pos + 2] === '(') type = '<<(', matchLength = 3;
241
285
  // Try to find a reified triple
242
- else if (!this._lineMode && input.length > (inputFinished ? 1 : 2) && input[1] === '<') type = '<<', matchLength = 2;
286
+ else if (!this._lineMode && input.length - pos > (inputFinished ? 1 : 2) && input[pos + 1] === '<') type = '<<', matchLength = 2;
243
287
  // Try to find a backwards implication arrow
244
- else if (this._n3Mode && input.length > 1 && input[1] === '=') {
288
+ else if (this._n3Mode && input.length - pos > 1 && input[pos + 1] === '=') {
245
289
  matchLength = 2;
246
290
  if (this._isImpliedBy) type = 'abbreviation', value = '<';else type = 'inverse', value = '>';
247
291
  }
248
292
  // Try to find an inverted predicate marker
249
- else if (this._n3Mode && input.length > 1 && input[1] === '-') type = 'inversePredicate', matchLength = 2;
293
+ else if (this._n3Mode && input.length - pos > 1 && input[pos + 1] === '-') type = 'inversePredicate', matchLength = 2;
250
294
  break;
251
295
  case '>':
252
296
  // Try to find a reified triple
253
- if (input.length > 1 && input[1] === '>') type = '>>', matchLength = 2;
297
+ if (input.length - pos > 1 && input[pos + 1] === '>') type = '>>', matchLength = 2;
254
298
  break;
255
299
  case '_':
256
300
  // Try to find a blank node. Since it can contain (but not end with) a dot,
257
301
  // we always need a non-dot character before deciding it is a blank node.
258
302
  // Therefore, try inserting a space if we're at the end of the input.
259
- if ((match = this._blank.exec(input)) || inputFinished && (match = this._blank.exec(`${input} `))) {
303
+ if ((match = execAt(this._blank, input, pos)) || inputFinished && (match = execAtEnd(this._blank, input, pos))) {
260
304
  type = 'blank', prefix = '_', value = match[1];
261
305
  lexicalLength = match[1].length + 2;
262
306
  }
263
307
  break;
264
308
  case '"':
265
309
  // Try to find a literal without escape sequences
266
- if (match = this._simpleQuotedString.exec(input)) value = match[1];
310
+ if (match = execAt(this._simpleQuotedString, input, pos)) value = match[1];
267
311
  // Try to find a literal wrapped in three pairs of quotes
268
312
  else {
269
313
  ({
270
314
  value,
271
315
  matchLength,
272
- finalLineLength
273
- } = this._parseLiteral(input));
274
- if (value === null) return reportSyntaxError(this);
316
+ finalLineLength,
317
+ tripleQuoted
318
+ } = this._parseLiteral(input, pos));
319
+ if (value === null) return reportSyntaxError(this, input, pos);
275
320
  }
276
321
  if (match !== null || matchLength !== 0) {
277
322
  type = 'literal';
@@ -281,15 +326,16 @@ class N3Lexer {
281
326
  case "'":
282
327
  if (!this._lineMode) {
283
328
  // Try to find a literal without escape sequences
284
- if (match = this._simpleApostropheString.exec(input)) value = match[1];
329
+ if (match = execAt(this._simpleApostropheString, input, pos)) value = match[1];
285
330
  // Try to find a literal wrapped in three pairs of quotes
286
331
  else {
287
332
  ({
288
333
  value,
289
334
  matchLength,
290
- finalLineLength
291
- } = this._parseLiteral(input));
292
- if (value === null) return reportSyntaxError(this);
335
+ finalLineLength,
336
+ tripleQuoted
337
+ } = this._parseLiteral(input, pos));
338
+ if (value === null) return reportSyntaxError(this, input, pos);
293
339
  }
294
340
  if (match !== null || matchLength !== 0) {
295
341
  type = 'literal';
@@ -299,7 +345,7 @@ class N3Lexer {
299
345
  break;
300
346
  case '?':
301
347
  // Try to find a variable
302
- if (this._n3Mode && (match = this._variable.exec(input))) type = 'var', value = match[0];
348
+ if (this._n3Mode && (match = execAt(this._variable, input, pos))) type = 'var', value = match[0];
303
349
  break;
304
350
  case '@':
305
351
  // Try to find a language code. A language code can contain dash-separated
@@ -307,15 +353,16 @@ class N3Lexer {
307
353
  // input is not finished, another subtag may still arrive in a later chunk and
308
354
  // the match would be premature; wait for more input in that case.
309
355
  // A double dash starts a direction code, which cannot extend the language code.
310
- if (this._previousMarker === 'literal' && (match = this._langcode.exec(input)) && match[1] !== 'version') {
311
- if (!inputFinished && input[match[0].length] === '-' && input[match[0].length + 1] !== '-') match = null;else type = 'langcode', value = match[1];
356
+ if (this._previousMarker === 'literal' && (match = execAt(this._langcode, input, pos)) && match[1] !== 'version') {
357
+ const end = pos + match[0].length;
358
+ if (!inputFinished && input[end] === '-' && input[end + 1] !== '-') match = null;else type = 'langcode', value = match[1];
312
359
  }
313
360
  // Try to find a keyword
314
- else if (match = this._atKeyword.exec(input)) type = match[0];
361
+ else if (match = execAt(this._atKeyword, input, pos)) type = match[0];
315
362
  break;
316
363
  case '.':
317
364
  // Try to find a dot as punctuation
318
- if (input.length === 1 ? inputFinished : input[1] < '0' || input[1] > '9') {
365
+ if (input.length - pos === 1 ? inputFinished : input[pos + 1] < '0' || input[pos + 1] > '9') {
319
366
  type = '.';
320
367
  matchLength = 1;
321
368
  break;
@@ -334,10 +381,10 @@ class N3Lexer {
334
381
  case '9':
335
382
  case '+':
336
383
  case '-':
337
- if (input[1] === '-') {
384
+ if (input[pos + 1] === '-') {
338
385
  // Try to find a direction code
339
386
  if (this._previousMarker === 'langcode') {
340
- if (input.startsWith('--ltr')) type = 'dircode', value = 'ltr', matchLength = 5;else if (input.startsWith('--rtl')) type = 'dircode', value = 'rtl', matchLength = 5;
387
+ if (input.startsWith('--ltr', pos)) type = 'dircode', value = 'ltr', matchLength = 5;else if (input.startsWith('--rtl', pos)) type = 'dircode', value = 'rtl', matchLength = 5;
341
388
  }
342
389
  break;
343
390
  }
@@ -345,7 +392,7 @@ class N3Lexer {
345
392
  // Try to find a number. Since it can contain (but not end with) a dot,
346
393
  // we always need a non-dot character before deciding it is a number.
347
394
  // Therefore, try inserting a space if we're at the end of the input.
348
- if (match = this._number.exec(input) || inputFinished && (match = this._number.exec(`${input} `))) {
395
+ if (match = execAt(this._number, input, pos) || inputFinished && (match = execAtEnd(this._number, input, pos))) {
349
396
  type = 'literal', value = match[0];
350
397
  prefix = typeof match[1] === 'string' ? xsd.double : typeof match[2] === 'string' ? xsd.decimal : xsd.integer;
351
398
  }
@@ -359,42 +406,42 @@ class N3Lexer {
359
406
  case 'V':
360
407
  case 'v':
361
408
  // Try to find a SPARQL-style keyword
362
- if (match = this._keyword.exec(input)) type = match[0].toUpperCase();else inconclusive = true;
409
+ if (match = execAt(this._keyword, input, pos)) type = match[0].toUpperCase();else inconclusive = true;
363
410
  break;
364
411
  case 'f':
365
412
  case 't':
366
413
  // Try to match a boolean
367
- if (this._boolean.test(input)) type = 'literal', value = firstChar === 't' ? 'true' : 'false', prefix = xsd.boolean, matchLength = value.length;else inconclusive = true;
414
+ if (testAt(this._boolean, input, pos)) type = 'literal', value = firstChar === 't' ? 'true' : 'false', prefix = xsd.boolean, matchLength = value.length;else inconclusive = true;
368
415
  break;
369
416
  case 'a':
370
417
  // Try to find an abbreviated predicate
371
- if (this._shortPredicates.test(input)) type = 'abbreviation', value = 'a', matchLength = 1;else inconclusive = true;
418
+ if (testAt(this._shortPredicates, input, pos)) type = 'abbreviation', value = 'a', matchLength = 1;else inconclusive = true;
372
419
  break;
373
420
  case 'h':
374
421
  case 'o':
375
422
  // Try to find an N3 verb keyword
376
- if (this._n3Mode && (match = this._matchN3Verb(input, inputFinished))) type = match[0];else inconclusive = true;
423
+ if (this._n3Mode && (match = this._matchN3Verb(input, pos, inputFinished))) type = match[0];else inconclusive = true;
377
424
  break;
378
425
  case 'i':
379
426
  // Try to find an IRI property list identifier or N3 verb keyword
380
- if (this._n3Mode && this._n3Id.test(input)) type = 'id', matchLength = 2;else if (this._n3Mode && (match = this._matchN3Verb(input, inputFinished))) type = match[0];else inconclusive = true;
427
+ if (this._n3Mode && testAt(this._n3Id, input, pos)) type = 'id', matchLength = 2;else if (this._n3Mode && (match = this._matchN3Verb(input, pos, inputFinished))) type = match[0];else inconclusive = true;
381
428
  break;
382
429
  case '=':
383
430
  // Try to find an implication arrow or equals sign
384
- if (this._n3Mode && input.length > 1) {
431
+ if (this._n3Mode && input.length - pos > 1) {
385
432
  type = 'abbreviation';
386
- if (input[1] !== '>') matchLength = 1, value = '=';else matchLength = 2, value = '>';
433
+ if (input[pos + 1] !== '>') matchLength = 1, value = '=';else matchLength = 2, value = '>';
387
434
  }
388
435
  break;
389
436
  case '!':
390
437
  if (!this._n3Mode) break;
391
438
  case ')':
392
- if (!inputFinished && (input.length === 1 || input.length === 2 && input[1] === '>')) {
439
+ if (!inputFinished && (input.length - pos === 1 || input.length - pos === 2 && input[pos + 1] === '>')) {
393
440
  // Don't consume yet, as it *could* become a triple term end.
394
441
  break;
395
442
  }
396
443
  // Try to find a triple term
397
- if (input.length > 2 && input[1] === '>' && input[2] === '>') {
444
+ if (input.length - pos > 2 && input[pos + 1] === '>' && input[pos + 2] === '>') {
398
445
  type = ')>>', matchLength = 3;
399
446
  break;
400
447
  }
@@ -412,15 +459,15 @@ class N3Lexer {
412
459
  break;
413
460
  case '{':
414
461
  // We need at least 2 tokens lookahead to distinguish "{|" and "{ "
415
- if (!this._lineMode && input.length >= 2) {
462
+ if (!this._lineMode && input.length - pos >= 2) {
416
463
  // Try to find a quoted triple annotation start
417
- if (input[1] === '|') type = '{|', matchLength = 2;else type = firstChar, matchLength = 1;
464
+ if (input[pos + 1] === '|') type = '{|', matchLength = 2;else type = firstChar, matchLength = 1;
418
465
  }
419
466
  break;
420
467
  case '|':
421
468
  // We need 2 tokens lookahead to parse "|}"
422
469
  // Try to find a quoted triple annotation end
423
- if (input.length >= 2 && input[1] === '}') type = '|}', matchLength = 2;
470
+ if (input.length - pos >= 2 && input[pos + 1] === '}') type = '|}', matchLength = 2;
424
471
  break;
425
472
  default:
426
473
  inconclusive = true;
@@ -429,11 +476,14 @@ class N3Lexer {
429
476
  // Some first characters do not allow an immediate decision, so inspect more
430
477
  if (inconclusive) {
431
478
  // Try to find a prefix
432
- if ((this._previousMarker === '@prefix' || this._previousMarker === 'PREFIX') && (match = this._prefix.exec(input))) type = 'prefix', value = match[1] || '';
479
+ if ((this._previousMarker === '@prefix' || this._previousMarker === 'PREFIX') && (match = execAt(this._prefix, input, pos))) type = 'prefix', value = match[1] || '';
480
+ // Try to find an additional directive keyword
481
+ // (at the end of the input, only a short final word can be one)
482
+ else if (this._directive !== null && this._directiveStarts.includes(firstChar) && ((match = execAt(this._directive, input, pos)) || inputFinished && input.length - pos <= this._directiveMaxLength && (match = execAtEnd(this._directive, input, pos)))) type = match[0].toUpperCase();
433
483
  // Try to find a prefixed name. Since it can contain (but not end with) a dot,
434
484
  // we always need a non-dot character before deciding it is a prefixed name.
435
485
  // Therefore, try inserting a space if we're at the end of the input.
436
- else if ((match = this._prefixed.exec(input)) || inputFinished && (match = this._prefixed.exec(`${input} `))) {
486
+ else if ((match = execAt(this._prefixed, input, pos)) || inputFinished && (match = execAtEnd(this._prefixed, input, pos))) {
437
487
  type = 'prefixed', prefix = match[1] || '';
438
488
  value = this._unescape(match[2], localNameEscapeReplacements);
439
489
  lexicalLength = prefix.length + match[2].length + 1;
@@ -459,16 +509,14 @@ class N3Lexer {
459
509
  // We could be in streaming mode, and then we just wait for more input to arrive.
460
510
  // Otherwise, a syntax error has occurred in the input.
461
511
  // One exception: error on an unaccounted linebreak (= not inside a triple-quoted literal).
462
- if (inputFinished || !/^'''|^"""/.test(input) && /\n|\r/.test(input)) return reportSyntaxError(this);else {
463
- this._linePosition = currentLineLength - input.length;
464
- return this._input = input;
465
- }
512
+ if (inputFinished || !input.startsWith("'''", pos) && !input.startsWith('"""', pos) && /\n|\r/.test(input.slice(pos))) return reportSyntaxError(this, input, pos);else return this._suspend(input, pos, currentLineLength);
466
513
  }
467
514
 
468
515
  // Emit the parsed token
469
516
  // Consumption includes separator whitespace; lexicalLength excludes it
470
- // and any synthetic EOF space. slice below clamps consumption to the input.
517
+ // and any synthetic EOF space. Consumption is clamped to the input below.
471
518
  const length = matchLength || match[0].length;
519
+ const start = currentLineLength - (input.length - pos);
472
520
  let token;
473
521
  if (finalLineLength) {
474
522
  token = {
@@ -476,23 +524,37 @@ class N3Lexer {
476
524
  value,
477
525
  prefix,
478
526
  line,
479
- start: currentLineLength - input.length,
527
+ start,
480
528
  end: finalLineLength,
481
- endLine: this._line
529
+ endLine: this._line,
530
+ tripleQuoted
482
531
  };
483
532
  callback(null, token);
484
- } else token = emitToken(type, value, prefix, line, lexicalLength || length);
533
+ }
534
+ // Triple-quoted strings are marked, since version declarations do not allow them
535
+ else if (tripleQuoted) {
536
+ token = {
537
+ type,
538
+ value,
539
+ prefix,
540
+ line,
541
+ start,
542
+ end: start + length,
543
+ tripleQuoted
544
+ };
545
+ callback(null, token);
546
+ } else token = emitToken(type, value, prefix, line, start, lexicalLength || length);
485
547
  this.previousToken = token;
486
- this._previousMarker = type;
548
+ // The string of a version declaration cannot take a language tag, so a following @keyword is a keyword
549
+ this._previousMarker = type === 'literal' && (this._previousMarker === 'VERSION' || this._previousMarker === '@version') ? 'version' : type;
487
550
 
488
551
  // Advance to next part to tokenize
489
- input = input.slice(length);
490
- if (finalLineLength) currentLineLength = input.length + finalLineLength;
552
+ pos = Math.min(pos + length, input.length);
553
+ if (finalLineLength) currentLineLength = input.length - pos + finalLineLength;
491
554
  }
492
555
 
493
556
  // Emits a comment at its exact position within matched whitespace.
494
- function emitComment(value, line, offset) {
495
- const start = currentLineLength - input.length + offset;
557
+ function emitComment(value, line, start) {
496
558
  callback(null, {
497
559
  type: 'comment',
498
560
  value,
@@ -503,49 +565,53 @@ class N3Lexer {
503
565
  });
504
566
  }
505
567
  // Emits the token through the callback
506
- function emitToken(type, value, prefix, line, length) {
507
- const start = input ? currentLineLength - input.length : currentLineLength;
508
- const end = start + length;
568
+ function emitToken(type, value, prefix, line, start, length) {
509
569
  const token = {
510
570
  type,
511
571
  value,
512
572
  prefix,
513
573
  line,
514
574
  start,
515
- end
575
+ end: start + length
516
576
  };
517
577
  callback(null, token);
518
578
  return token;
519
579
  }
520
580
  // Signals the syntax error through the callback
521
- function reportSyntaxError(self) {
522
- callback(self._syntaxError(/^\S*/.exec(input)[0]));
581
+ function reportSyntaxError(self, input, pos) {
582
+ callback(self._syntaxError(execAt(nonWhitespace, input, pos)[0]));
523
583
  }
524
584
  }
525
585
 
586
+ // ### `_suspend` keeps the unconsumed input until more input arrives
587
+ _suspend(input, pos, currentLineLength) {
588
+ this._linePosition = currentLineLength - (input.length - pos);
589
+ return this._input = input.slice(pos);
590
+ }
591
+
526
592
  // ### `_matchN3Verb` matches an N3 verb unless the input is a longer prefixed name
527
- _matchN3Verb(input, inputFinished) {
528
- const verb = this._n3Verb.exec(input);
593
+ _matchN3Verb(input, pos, inputFinished) {
594
+ const verb = execAt(this._n3Verb, input, pos);
529
595
  if (!verb) return null;
530
596
 
531
597
  // Most verb boundaries cannot be part of a prefix, so keep the common path fast.
532
- const next = input[verb[0].length];
598
+ const next = input[pos + verb[0].length];
533
599
  if (next !== '-' && next !== '_' && (next < '0' || next > '9')) return verb;
534
600
 
535
601
  // A prefix can start with a verb and continue with characters that are also
536
602
  // valid verb boundaries. Prefer the longer prefixed name when it is complete.
537
- if (this._prefixed.exec(input)) return null;
603
+ if (execAt(this._prefixed, input, pos)) return null;
538
604
  // Appending to the input only matters when a prefixed name could run up to
539
605
  // its end, which a character that cannot occur in prefixed names rules out.
540
606
  // This avoids copying the rest of the document for every such verb.
541
- if (nonPrefixedNameChar.test(input)) return verb;
542
- if (this._prefixed.exec(`${input} `)) return null;
607
+ if (testAt(nonPrefixedNameChar, input, pos)) return verb;
608
+ if (execAtEnd(this._prefixed, input, pos)) return null;
543
609
 
544
610
  // If a stream chunk ends partway through such a prefix, wait for the colon
545
611
  // instead of prematurely emitting the verb. Appending ": " lets the prefix
546
612
  // grammar determine whether all input seen so far can be a complete prefix.
547
613
  if (!inputFinished) {
548
- const prefix = this._prefix.exec(`${input}: `);
614
+ const prefix = execAt(this._prefix, `${input.slice(pos)}: `, 0);
549
615
  if (prefix) return null;
550
616
  }
551
617
  return verb;
@@ -592,19 +658,19 @@ class N3Lexer {
592
658
  return result + item.slice(start);
593
659
  }
594
660
 
595
- // ### `_parseLiteral` parses a literal into an unescaped value
596
- _parseLiteral(input) {
661
+ // ### `_parseLiteral` parses a literal at the given position into an unescaped value
662
+ _parseLiteral(input, pos) {
597
663
  // Ensure we have enough lookahead to identify triple-quoted strings
598
- if (input.length >= 3) {
664
+ if (input.length - pos >= 3) {
599
665
  // The caller has already identified a single or double quote.
600
- const quote = input[0];
601
- const openingLength = input[1] === quote && input[2] === quote ? 3 : 1;
666
+ const quote = input[pos];
667
+ const openingLength = input[pos + 1] === quote && input[pos + 2] === quote ? 3 : 1;
602
668
  let opening = quote;
603
669
  if (openingLength === 3) opening = quote === '"' ? '"""' : "'''";
604
670
 
605
671
  // Find the next candidate closing quotes
606
- let closingPos = Math.max(this._literalClosingPos, openingLength);
607
- while ((closingPos = input.indexOf(opening, closingPos)) > 0) {
672
+ let closingPos = pos + Math.max(this._literalClosingPos, openingLength);
673
+ while ((closingPos = input.indexOf(opening, closingPos)) > pos) {
608
674
  // Count backslashes right before the closing quotes
609
675
  let backslashCount = 0;
610
676
  while (input[closingPos - backslashCount - 1] === '\\') backslashCount++;
@@ -613,10 +679,10 @@ class N3Lexer {
613
679
  // means these are actual, non-escaped closing quotes
614
680
  if (backslashCount % 2 === 0) {
615
681
  // Extract and unescape the value
616
- const raw = input.substring(openingLength, closingPos),
682
+ const raw = input.substring(pos + openingLength, closingPos),
617
683
  lines = raw.split(/\r\n|\r|\n/),
618
684
  lineCount = lines.length - 1;
619
- const matchLength = closingPos + openingLength;
685
+ const matchLength = closingPos - pos + openingLength;
620
686
  // Only triple-quoted strings can be multi-line
621
687
  if (openingLength === 1 && lineCount !== 0 || openingLength === 3 && this._lineMode) break;
622
688
  this._line += lineCount;
@@ -624,24 +690,45 @@ class N3Lexer {
624
690
  return {
625
691
  value: this._unescape(raw, stringEscapeReplacements),
626
692
  matchLength,
627
- finalLineLength
693
+ finalLineLength,
694
+ tripleQuoted: openingLength === 3
628
695
  };
629
696
  }
630
697
  closingPos++;
631
698
  }
632
- this._literalClosingPos = input.length - openingLength + 1;
699
+ this._literalClosingPos = input.length - pos - openingLength + 1;
633
700
  }
634
701
  return {
635
702
  value: '',
636
703
  matchLength: 0,
637
- finalLineLength: 0
704
+ finalLineLength: 0,
705
+ tripleQuoted: false
638
706
  };
639
707
  }
640
708
 
709
+ // ### `_tryTokenizeToEnd` tokenizes as far as possible, reporting failures through the callback
710
+ _tryTokenizeToEnd(callback, inputFinished) {
711
+ // Keep track of errors thrown by the callback, which must reach the caller unchanged
712
+ let callbackError;
713
+ try {
714
+ this._tokenizeToEnd((error, token) => {
715
+ try {
716
+ return callback(error, token);
717
+ } catch (thrown) {
718
+ throw callbackError = thrown;
719
+ }
720
+ }, inputFinished);
721
+ } catch (error) {
722
+ // Matching an extremely long token can exhaust the regular expression stack
723
+ if (error === callbackError || !(error instanceof RangeError)) throw error;
724
+ callback(this._syntaxError(null, `Token too long on line ${this._line}.`));
725
+ }
726
+ }
727
+
641
728
  // ### `_syntaxError` creates a syntax error for the given issue
642
- _syntaxError(issue) {
729
+ _syntaxError(issue, message = `Unexpected "${issue}" on line ${this._line}.`) {
643
730
  this._input = null;
644
- const err = new Error(`Unexpected "${issue}" on line ${this._line}.`);
731
+ const err = new Error(message);
645
732
  err.context = {
646
733
  token: undefined,
647
734
  line: this._line,
@@ -682,44 +769,52 @@ class N3Lexer {
682
769
  this._input = this._readStartingBom(input);
683
770
  // If a callback was passed, asynchronously call it
684
771
  if (typeof callback === 'function') queueMicrotask(() => {
685
- if (this._tokenization === tokenization) this._tokenizeToEnd(callback, true);
772
+ if (this._tokenization === tokenization) this._tryTokenizeToEnd(callback, true);
686
773
  });
687
774
  // If no callback was passed, tokenize synchronously and return
688
775
  else {
689
776
  const tokens = [];
690
777
  let error;
691
- this._tokenizeToEnd((e, t) => e ? error = e : tokens.push(t), true);
778
+ this._tryTokenizeToEnd((e, t) => e ? error = e : tokens.push(t), true);
692
779
  if (error) throw error;
693
780
  return tokens;
694
781
  }
695
782
  }
696
783
  // Otherwise, the input must be a stream
697
784
  else {
698
- this._pendingBuffer = null;
785
+ let decoder,
786
+ retryLength = 0;
699
787
  if (typeof input.setEncoding === 'function') input.setEncoding('utf8');
700
788
  // Adds the data chunk to the buffer and parses as far as possible
701
789
  input.on('data', data => {
702
790
  if (this._tokenization === tokenization && this._input !== null && data.length !== 0) {
703
- // Prepend any previous pending writes
704
- if (this._pendingBuffer) {
705
- data = _buffer.Buffer.concat([this._pendingBuffer, data]);
706
- this._pendingBuffer = null;
707
- }
708
- // Hold if the buffer ends in an incomplete unicode sequence
709
- if (data[data.length - 1] & 0x80) {
710
- this._pendingBuffer = data;
791
+ // Decode bytes, keeping an incomplete trailing character for the next chunk
792
+ if (typeof data !== 'string') {
793
+ decoder = decoder || new TextDecoder('utf-8', {
794
+ ignoreBOM: true
795
+ });
796
+ if (!(data = decoder.decode(data, {
797
+ stream: true
798
+ }))) return;
711
799
  }
712
- // Otherwise, tokenize as far as possible
713
- else {
714
- // Only read a BOM at the start
715
- if (typeof this._input === 'undefined') this._input = this._readStartingBom(typeof data === 'string' ? data : data.toString());else this._input += data;
716
- this._tokenizeToEnd(callback, false);
800
+ // Only read a BOM at the start
801
+ if (typeof this._input === 'undefined') this._input = this._readStartingBom(data);else this._input += data;
802
+ // Tokenize as far as possible. When a previous attempt left a long unfinished token,
803
+ // wait until the buffered input has doubled, so the token is not rescanned for every chunk.
804
+ if (this._input.length >= retryLength) {
805
+ this._tryTokenizeToEnd(callback, false);
806
+ retryLength = this._input !== null && this._input.length > MIN_RESCAN_LENGTH ? 2 * this._input.length : 0;
717
807
  }
718
808
  }
719
809
  });
720
810
  // Parses until the end
721
811
  input.on('end', () => {
722
- if (this._tokenization === tokenization && typeof this._input === 'string') this._tokenizeToEnd(callback, true);
812
+ if (this._tokenization === tokenization && this._input !== null) {
813
+ // Decode any incomplete character left at the end
814
+ const rest = decoder ? decoder.decode() : '';
815
+ if (rest) this._input = typeof this._input === 'string' ? this._input + rest : rest;
816
+ if (typeof this._input === 'string') this._tryTokenizeToEnd(callback, true);
817
+ }
723
818
  });
724
819
  input.on('error', error => {
725
820
  if (this._tokenization === tokenization) callback(error);