n3 3.0.0-alpha.4 → 3.0.0-alpha.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/lib/N3Lexer.js CHANGED
@@ -4,7 +4,6 @@ Object.defineProperty(exports, "__esModule", {
4
4
  value: true
5
5
  });
6
6
  exports.default = void 0;
7
- var _buffer = require("buffer");
8
7
  var _IRIs = _interopRequireDefault(require("./IRIs"));
9
8
  function _interopRequireDefault(e) { return e && e.__esModule ? e : { default: e }; }
10
9
  // **N3Lexer** tokenizes N3 documents.
@@ -54,7 +53,8 @@ const localNameEscapeReplacements = {
54
53
  };
55
54
  const illegalIriChars = /[\x00-\x20<>\\"\{\}\|\^\`]/;
56
55
  // Characters that cannot occur in a prefixed name, not even escaped
57
- const nonPrefixedNameChar = /[\s<>"{}|^`]/;
56
+ // (global, so that testAt searches the rest of the input from a position)
57
+ const nonPrefixedNameChar = /[\s<>"{}|^`]/g;
58
58
 
59
59
  // A valid code point is a Unicode scalar value: at most U+10FFFF and not a surrogate
60
60
  function isValidCodePoint(charCode) {
@@ -70,30 +70,59 @@ const lineModeRegExps = {
70
70
  _whitespace: true
71
71
  };
72
72
  const invalidRegExp = /$0^/;
73
+ const nonWhitespace = /\S*/y;
74
+
75
+ // Matches a sticky regular expression at the given position of the input
76
+ function execAt(regExp, input, pos) {
77
+ regExp.lastIndex = pos;
78
+ return regExp.exec(input);
79
+ }
80
+ function testAt(regExp, input, pos) {
81
+ regExp.lastIndex = pos;
82
+ return regExp.test(input);
83
+ }
84
+ // Matches the rest of the input followed by a space, as at the end of the input,
85
+ // a token that can contain (but not end with) a dot needs a non-dot character after it
86
+ function execAtEnd(regExp, input, pos) {
87
+ regExp.lastIndex = 0;
88
+ return regExp.exec(`${input.slice(pos)} `);
89
+ }
90
+
91
+ // Whitespace or the start of a comment
92
+ function isSeparatorCode(code) {
93
+ return code === SPACE || code === TAB || code === LF || code === CR || code === HASH;
94
+ }
95
+
96
+ // Words with a fixed meaning in the grammar, which cannot name an additional directive
97
+ const reservedWords = /^(?:prefix|base|version|graph|forsome|forall|iri|a|true|false|has|is|of|id)$/i;
98
+
99
+ // Unfinished input in a stream up to this length is tokenized again with every chunk
100
+ const MIN_RESCAN_LENGTH = 1024;
73
101
 
74
102
  // ## Constructor
75
103
  class N3Lexer {
76
104
  constructor(options) {
77
105
  // ## Regular expressions
78
- // It's slightly faster to have these as properties than as in-scope variables
79
- this._iri = /^<((?:[^ <>{}\\]|\\[uU])+)>[ \t]*/; // IRI with escape sequences; needs sanity check after unescaping
80
- this._unescapedIri = /^<([^\x00-\x20<>\\"\{\}\|\^\`]*)>[ \t]*/; // IRI without escape sequences; no unescaping
81
- this._simpleQuotedString = /^"([^"\\\r\n]*)"(?=[^"])/; // string without escape sequences
82
- this._simpleApostropheString = /^'([^'\\\r\n]*)'(?=[^'])/;
83
- this._langcode = /^@([a-z]+(?:-[a-z0-9]+)*)(?=[^a-z0-9])/i;
84
- this._prefix = /^((?:[A-Za-z\xc0-\xd6\xd8-\xf6\xf8-\u02ff\u0370-\u037d\u037f-\u1fff\u200c\u200d\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff])(?:\.?[\-0-9A-Z_a-z\xb7\xc0-\xd6\xd8-\xf6\xf8-\u037d\u037f-\u1fff\u200c\u200d\u203f\u2040\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff])*)?:(?=[#\s<])/;
85
- this._prefixed = /^((?:[A-Za-z\xc0-\xd6\xd8-\xf6\xf8-\u02ff\u0370-\u037d\u037f-\u1fff\u200c\u200d\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff])(?:\.?[\-0-9A-Z_a-z\xb7\xc0-\xd6\xd8-\xf6\xf8-\u037d\u037f-\u1fff\u200c\u200d\u203f\u2040\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff])*)?:((?:(?:[0-9:A-Z_a-z\xc0-\xd6\xd8-\xf6\xf8-\u02ff\u0370-\u037d\u037f-\u1fff\u200c\u200d\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff]|%[0-9a-fA-F]{2}|\\[!#-\/;=?\-@_~])(?:(?:[\.\-0-9:A-Z_a-z\xb7\xc0-\xd6\xd8-\xf6\xf8-\u037d\u037f-\u1fff\u200c\u200d\u203f\u2040\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff]|%[0-9a-fA-F]{2}|\\[!#-\/;=?\-@_~])*(?:[\-0-9:A-Z_a-z\xb7\xc0-\xd6\xd8-\xf6\xf8-\u037d\u037f-\u1fff\u200c\u200d\u203f\u2040\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff]|%[0-9a-fA-F]{2}|\\[!#-\/;=?\-@_~]))?)?)(?:[ \t]+|(?=\.?[,;!\^\s#()\[\]\{\}"'<>]))/;
86
- this._variable = /^\?(?:(?:[A-Z_a-z\xc0-\xd6\xd8-\xf6\xf8-\u02ff\u0370-\u037d\u037f-\u1fff\u200c\u200d\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff])(?:[\-0-9:A-Z_a-z\xb7\xc0-\xd6\xd8-\xf6\xf8-\u037d\u037f-\u1fff\u200c\u200d\u203f\u2040\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff])*)(?=[.,;!\^\s#()\[\]\{\}"'<>])/;
87
- this._blank = /^_:((?:[0-9A-Z_a-z\xc0-\xd6\xd8-\xf6\xf8-\u02ff\u0370-\u037d\u037f-\u1fff\u200c\u200d\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff])(?:\.?[\-0-9A-Z_a-z\xb7\xc0-\xd6\xd8-\xf6\xf8-\u037d\u037f-\u1fff\u200c\u200d\u203f\u2040\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff])*)(?:[ \t]+|(?=\.?[,;:!\^\s#()\[\]\{\}"'<>]))/;
88
- this._number = /^[\-+]?(?:(\d+\.\d*|\.?\d+)[eE][\-+]?\d+|(?=\.?\d)\d*(?:(\.)\d+)?)(?=\.?[,;:!\^\s#()\[\]\{\}"'<>])/;
89
- this._boolean = /^(?:true|false)(?=[.,;!\^\s#()\[\]\{\}"'<>])/;
90
- this._atKeyword = /^@[a-z]+(?=[\s#<:])/i;
91
- this._keyword = /^(?:PREFIX|BASE|VERSION|GRAPH)(?=[\s#<])/i;
92
- this._n3Verb = /^(?:has|is|of)(?=[\s#()\[\]\{\}"'<>?_+\-0-9])/;
93
- this._n3Id = /^id(?=[\s#<])/;
94
- this._shortPredicates = /^a(?=[\s#()\[\]\{\}"'<>])/;
95
- this._commentLine = /^[ \t]*#([^\n\r]*)(?:\r\n|\n|\r)([ \t]*)/;
96
- this._whitespace = /^[ \t]+/;
106
+ // It's slightly faster to have these as properties than as in-scope variables.
107
+ // They are sticky, so they only match at the `lastIndex` set by `execAt`.
108
+ this._iri = /<((?:[^ <>{}\\]|\\[uU])+)>[ \t]*/y; // IRI with escape sequences; needs sanity check after unescaping
109
+ this._unescapedIri = /<([^\x00-\x20<>\\"\{\}\|\^\`]*)>[ \t]*/y; // IRI without escape sequences; no unescaping
110
+ this._simpleQuotedString = /"([^"\\\r\n]*)"(?=[^"])/y; // string without escape sequences
111
+ this._simpleApostropheString = /'([^'\\\r\n]*)'(?=[^'])/y;
112
+ this._langcode = /@([a-z]+(?:-[a-z0-9]+)*)(?=[^a-z0-9])/iy;
113
+ this._prefix = /((?:[A-Za-z\xc0-\xd6\xd8-\xf6\xf8-\u02ff\u0370-\u037d\u037f-\u1fff\u200c\u200d\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff])(?:\.?[\-0-9A-Z_a-z\xb7\xc0-\xd6\xd8-\xf6\xf8-\u037d\u037f-\u1fff\u200c\u200d\u203f\u2040\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff])*)?:(?=[#\s<])/y;
114
+ this._prefixed = /((?:[A-Za-z\xc0-\xd6\xd8-\xf6\xf8-\u02ff\u0370-\u037d\u037f-\u1fff\u200c\u200d\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff])(?:\.?[\-0-9A-Z_a-z\xb7\xc0-\xd6\xd8-\xf6\xf8-\u037d\u037f-\u1fff\u200c\u200d\u203f\u2040\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff])*)?:((?:(?:[0-9:A-Z_a-z\xc0-\xd6\xd8-\xf6\xf8-\u02ff\u0370-\u037d\u037f-\u1fff\u200c\u200d\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff]|%[0-9a-fA-F]{2}|\\[!#-\/;=?\-@_~])(?:(?:[\.\-0-9:A-Z_a-z\xb7\xc0-\xd6\xd8-\xf6\xf8-\u037d\u037f-\u1fff\u200c\u200d\u203f\u2040\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff]|%[0-9a-fA-F]{2}|\\[!#-\/;=?\-@_~])*(?:[\-0-9:A-Z_a-z\xb7\xc0-\xd6\xd8-\xf6\xf8-\u037d\u037f-\u1fff\u200c\u200d\u203f\u2040\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff]|%[0-9a-fA-F]{2}|\\[!#-\/;=?\-@_~]))?)?)(?:[ \t]+|(?=\.?[,;!\^\s#()\[\]\{\}"'<>]))/y;
115
+ this._variable = /\?(?:(?:[A-Z_a-z\xc0-\xd6\xd8-\xf6\xf8-\u02ff\u0370-\u037d\u037f-\u1fff\u200c\u200d\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff])(?:[\-0-9:A-Z_a-z\xb7\xc0-\xd6\xd8-\xf6\xf8-\u037d\u037f-\u1fff\u200c\u200d\u203f\u2040\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff])*)(?=[.,;!\^\s#()\[\]\{\}"'<>])/y;
116
+ this._blank = /_:((?:[0-9A-Z_a-z\xc0-\xd6\xd8-\xf6\xf8-\u02ff\u0370-\u037d\u037f-\u1fff\u200c\u200d\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff])(?:\.?[\-0-9A-Z_a-z\xb7\xc0-\xd6\xd8-\xf6\xf8-\u037d\u037f-\u1fff\u200c\u200d\u203f\u2040\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff])*)(?:[ \t]+|(?=\.?[,;:!\^\s#()\[\]\{\}"'<>]))/y;
117
+ this._number = /[\-+]?(?:(\d+\.\d*|\.?\d+)[eE][\-+]?\d+|(?=\.?\d)\d*(?:(\.)\d+)?)(?=\.?[,;:!\^\s#()\[\]\{\}"'<>])/y;
118
+ this._boolean = /(?:true|false)(?=[.,;!\^\s#()\[\]\{\}"'<>])/y;
119
+ this._atKeyword = /@[a-z]+(?=[\s#<:])/iy;
120
+ this._keyword = /(?:PREFIX|BASE|VERSION|GRAPH)(?=[\s#<])/iy;
121
+ this._n3Verb = /(?:has|is|of)(?=[\s#()\[\]\{\}"'<>?_+\-0-9])/y;
122
+ this._n3Id = /id(?=[\s#<])/y;
123
+ this._shortPredicates = /a(?=[\s#()\[\]\{\}"'<>])/y;
124
+ this._commentLine = /[ \t]*#([^\n\r]*)(?:\r\n|\n|\r)([ \t]*)/y;
125
+ this._whitespace = /[ \t]+/y;
97
126
  options = options || {};
98
127
 
99
128
  // Whether the log:isImpliedBy predicate is supported
@@ -111,6 +140,18 @@ class N3Lexer {
111
140
  else {
112
141
  this._n3Mode = options.n3 !== false;
113
142
  }
143
+ // Recognize additional directive keywords, such as MESSAGE
144
+ // (the @-form of a directive is always tokenized as an @-keyword)
145
+ this._directive = null;
146
+ if (options.directives && options.directives.length !== 0) {
147
+ for (const name of options.directives) {
148
+ if (!/^[a-z]+$/i.test(name) || reservedWords.test(name)) throw new Error(`Invalid directive name: "${name}"`);
149
+ }
150
+ this._directive = new RegExp(`(?:${options.directives.join('|')})(?=[\\s#<])`, 'iy');
151
+ this._directiveMaxLength = Math.max(...options.directives.map(name => name.length));
152
+ // The first characters of directive names, so other words skip the regular expression
153
+ this._directiveStarts = options.directives.map(name => name[0].toLowerCase() + name[0].toUpperCase()).join('');
154
+ }
114
155
  // Don't output comment tokens by default
115
156
  this.comments = !!options.comments;
116
157
  // Cache the last tested closing position of long literals
@@ -121,77 +162,73 @@ class N3Lexer {
121
162
 
122
163
  // ### `_tokenizeToEnd` tokenizes as for as possible, emitting tokens through the callback
123
164
  _tokenizeToEnd(callback, inputFinished) {
124
- // Continue parsing as far as possible; the loop will return eventually
125
- let input = this._input;
165
+ // Continue parsing as far as possible; the loop will return eventually.
166
+ // Rather than slicing off every token, track the position of the remaining input;
167
+ // the regular expressions are sticky, so they match at that position.
168
+ const input = this._input;
169
+ let pos = 0;
126
170
  let currentLineLength = this._linePosition + input.length;
127
171
  while (true) {
128
172
  // Consume one separator line at a time, including its following indentation.
129
173
  while (true) {
130
- let charCode = input.charCodeAt(0),
174
+ let charCode = input.charCodeAt(pos),
131
175
  separatorLength = 0;
132
176
  if (charCode === SPACE || charCode === TAB) {
133
- const next = input.charCodeAt(1);
134
- separatorLength = next === SPACE || next === TAB ? this._whitespace.exec(input)[0].length : 1;
135
- charCode = input.charCodeAt(separatorLength);
177
+ const next = input.charCodeAt(pos + 1);
178
+ separatorLength = next === SPACE || next === TAB ? execAt(this._whitespace, input, pos)[0].length : 1;
179
+ charCode = input.charCodeAt(pos + separatorLength);
136
180
  }
137
181
  if (charCode === HASH) {
138
- const comment = this._commentLine.exec(input);
182
+ const comment = execAt(this._commentLine, input, pos);
139
183
  if (comment) {
140
184
  const commentLength = comment[0].length;
141
185
  // Keep a trailing CR buffered in case the next chunk starts with LF.
142
- if (!inputFinished && commentLength === input.length && input.charCodeAt(commentLength - 1) === CR) {
143
- this._linePosition = currentLineLength - input.length;
144
- return this._input = input;
145
- }
146
- if (this.comments) emitComment(comment[1], this._line, separatorLength);
147
- input = input.slice(commentLength);
148
- currentLineLength = input.length + comment[2].length;
186
+ if (!inputFinished && pos + commentLength === input.length && input.charCodeAt(input.length - 1) === CR) return this._suspend(input, pos, currentLineLength);
187
+ if (this.comments) emitComment(comment[1], this._line, currentLineLength - (input.length - pos) + separatorLength);
188
+ pos += commentLength;
189
+ currentLineLength = input.length - pos + comment[2].length;
149
190
  this._line++;
150
191
  } else {
151
192
  // A comment without a line ending stays buffered until EOF.
152
- input = input.slice(separatorLength);
153
- if (!inputFinished) {
154
- this._linePosition = currentLineLength - input.length;
155
- return this._input = input;
156
- }
157
- if (this.comments) emitComment(input.slice(1), this._line, 0);
158
- input = '';
193
+ pos += separatorLength;
194
+ if (!inputFinished) return this._suspend(input, pos, currentLineLength);
195
+ if (this.comments) emitComment(input.slice(pos + 1), this._line, currentLineLength - (input.length - pos));
196
+ pos = input.length;
159
197
  break;
160
198
  }
161
199
  } else if (charCode === LF || charCode === CR) {
162
200
  // A CR at the end of a chunk may still be followed by LF.
163
- if (!inputFinished && charCode === CR && separatorLength + 1 === input.length) {
164
- this._linePosition = currentLineLength - input.length;
165
- return this._input = input;
166
- }
167
- separatorLength += charCode === CR && input.charCodeAt(separatorLength + 1) === LF ? 2 : 1;
201
+ if (!inputFinished && charCode === CR && pos + separatorLength + 1 === input.length) return this._suspend(input, pos, currentLineLength);
202
+ separatorLength += charCode === CR && input.charCodeAt(pos + separatorLength + 1) === LF ? 2 : 1;
168
203
  // Indentation is consumed with the newline, but belongs to the next line's columns.
169
204
  let indentationLength = 0;
170
- const next = input.charCodeAt(separatorLength);
205
+ const next = input.charCodeAt(pos + separatorLength);
171
206
  if (next === SPACE || next === TAB) {
172
- const following = input.charCodeAt(separatorLength + 1);
173
- indentationLength = following === SPACE || following === TAB ? this._whitespace.exec(input.slice(separatorLength))[0].length : 1;
207
+ const following = input.charCodeAt(pos + separatorLength + 1);
208
+ indentationLength = following === SPACE || following === TAB ? execAt(this._whitespace, input, pos + separatorLength)[0].length : 1;
174
209
  }
175
- input = input.slice(separatorLength + indentationLength);
176
- currentLineLength = input.length + indentationLength;
210
+ pos += separatorLength + indentationLength;
211
+ currentLineLength = input.length - pos + indentationLength;
177
212
  this._line++;
178
213
  } else {
179
- if (separatorLength !== 0) input = input.slice(separatorLength);
214
+ pos += separatorLength;
180
215
  break;
181
216
  }
182
217
  }
183
- if (input.length === 0) {
218
+ if (pos >= input.length) {
219
+ // A datatype marker needs a type
220
+ if (inputFinished && this._previousMarker === '^^') return reportSyntaxError(this, input, pos);
221
+ this._linePosition = currentLineLength;
184
222
  if (inputFinished) {
185
- input = null;
186
- emitToken('eof', '', '', this._line, 0);
223
+ emitToken('eof', '', '', this._line, currentLineLength, 0);
224
+ return this._input = null;
187
225
  }
188
- this._linePosition = currentLineLength;
189
- return this._input = input;
226
+ return this._input = '';
190
227
  }
191
228
 
192
229
  // Look for specific token types based on the first character
193
230
  const line = this._line,
194
- firstChar = input[0];
231
+ firstChar = input[pos];
195
232
  let type = '',
196
233
  value = '',
197
234
  prefix = '',
@@ -202,14 +239,18 @@ class N3Lexer {
202
239
  inconclusive = false;
203
240
  switch (firstChar) {
204
241
  case '^':
242
+ // A datatype marker separated from its type cannot be followed by another marker
243
+ if (this._previousMarker === '^^') return reportSyntaxError(this, input, pos);
205
244
  // We need at least 3 tokens lookahead to distinguish ^^<IRI> and ^^pre:fixed
206
- if (input.length < 3) break;
245
+ if (input.length - pos < 3) break;
207
246
  // Try to match a type
208
- else if (input[1] === '^') {
247
+ else if (input[pos + 1] === '^') {
209
248
  this._previousMarker = '^^';
210
249
  // Move to type IRI or prefixed name
211
- input = input.slice(2);
212
- if (input[0] !== '<') {
250
+ pos += 2;
251
+ if (input[pos] !== '<') {
252
+ // Whitespace and comments may separate the marker from the type
253
+ if (isSeparatorCode(input.charCodeAt(pos))) continue; // eslint-disable-line no-continue
213
254
  inconclusive = true;
214
255
  break;
215
256
  }
@@ -225,53 +266,53 @@ class N3Lexer {
225
266
  // Fall through in case the type is an IRI
226
267
  case '<':
227
268
  // Try to find a full IRI without escape sequences
228
- if (match = this._unescapedIri.exec(input)) {
269
+ if (match = execAt(this._unescapedIri, input, pos)) {
229
270
  type = 'IRI', value = match[1];
230
271
  lexicalLength = match[1].length + 2;
231
272
  }
232
273
  // Try to find a full IRI with escape sequences
233
- else if (match = this._iri.exec(input)) {
274
+ else if (match = execAt(this._iri, input, pos)) {
234
275
  value = this._unescape(match[1], stringEscapeReplacements);
235
- if (value === null || illegalIriChars.test(value)) return reportSyntaxError(this);
276
+ if (value === null || illegalIriChars.test(value)) return reportSyntaxError(this, input, pos);
236
277
  type = 'IRI';
237
278
  lexicalLength = match[1].length + 2;
238
279
  }
239
280
  // Try to find a triple term
240
- else if (input.length > 2 && input[1] === '<' && input[2] === '(') type = '<<(', matchLength = 3;
281
+ else if (input.length - pos > 2 && input[pos + 1] === '<' && input[pos + 2] === '(') type = '<<(', matchLength = 3;
241
282
  // Try to find a reified triple
242
- else if (!this._lineMode && input.length > (inputFinished ? 1 : 2) && input[1] === '<') type = '<<', matchLength = 2;
283
+ else if (!this._lineMode && input.length - pos > (inputFinished ? 1 : 2) && input[pos + 1] === '<') type = '<<', matchLength = 2;
243
284
  // Try to find a backwards implication arrow
244
- else if (this._n3Mode && input.length > 1 && input[1] === '=') {
285
+ else if (this._n3Mode && input.length - pos > 1 && input[pos + 1] === '=') {
245
286
  matchLength = 2;
246
287
  if (this._isImpliedBy) type = 'abbreviation', value = '<';else type = 'inverse', value = '>';
247
288
  }
248
289
  // Try to find an inverted predicate marker
249
- else if (this._n3Mode && input.length > 1 && input[1] === '-') type = 'inversePredicate', matchLength = 2;
290
+ else if (this._n3Mode && input.length - pos > 1 && input[pos + 1] === '-') type = 'inversePredicate', matchLength = 2;
250
291
  break;
251
292
  case '>':
252
293
  // Try to find a reified triple
253
- if (input.length > 1 && input[1] === '>') type = '>>', matchLength = 2;
294
+ if (input.length - pos > 1 && input[pos + 1] === '>') type = '>>', matchLength = 2;
254
295
  break;
255
296
  case '_':
256
297
  // Try to find a blank node. Since it can contain (but not end with) a dot,
257
298
  // we always need a non-dot character before deciding it is a blank node.
258
299
  // Therefore, try inserting a space if we're at the end of the input.
259
- if ((match = this._blank.exec(input)) || inputFinished && (match = this._blank.exec(`${input} `))) {
300
+ if ((match = execAt(this._blank, input, pos)) || inputFinished && (match = execAtEnd(this._blank, input, pos))) {
260
301
  type = 'blank', prefix = '_', value = match[1];
261
302
  lexicalLength = match[1].length + 2;
262
303
  }
263
304
  break;
264
305
  case '"':
265
306
  // Try to find a literal without escape sequences
266
- if (match = this._simpleQuotedString.exec(input)) value = match[1];
307
+ if (match = execAt(this._simpleQuotedString, input, pos)) value = match[1];
267
308
  // Try to find a literal wrapped in three pairs of quotes
268
309
  else {
269
310
  ({
270
311
  value,
271
312
  matchLength,
272
313
  finalLineLength
273
- } = this._parseLiteral(input));
274
- if (value === null) return reportSyntaxError(this);
314
+ } = this._parseLiteral(input, pos));
315
+ if (value === null) return reportSyntaxError(this, input, pos);
275
316
  }
276
317
  if (match !== null || matchLength !== 0) {
277
318
  type = 'literal';
@@ -281,15 +322,15 @@ class N3Lexer {
281
322
  case "'":
282
323
  if (!this._lineMode) {
283
324
  // Try to find a literal without escape sequences
284
- if (match = this._simpleApostropheString.exec(input)) value = match[1];
325
+ if (match = execAt(this._simpleApostropheString, input, pos)) value = match[1];
285
326
  // Try to find a literal wrapped in three pairs of quotes
286
327
  else {
287
328
  ({
288
329
  value,
289
330
  matchLength,
290
331
  finalLineLength
291
- } = this._parseLiteral(input));
292
- if (value === null) return reportSyntaxError(this);
332
+ } = this._parseLiteral(input, pos));
333
+ if (value === null) return reportSyntaxError(this, input, pos);
293
334
  }
294
335
  if (match !== null || matchLength !== 0) {
295
336
  type = 'literal';
@@ -299,7 +340,7 @@ class N3Lexer {
299
340
  break;
300
341
  case '?':
301
342
  // Try to find a variable
302
- if (this._n3Mode && (match = this._variable.exec(input))) type = 'var', value = match[0];
343
+ if (this._n3Mode && (match = execAt(this._variable, input, pos))) type = 'var', value = match[0];
303
344
  break;
304
345
  case '@':
305
346
  // Try to find a language code. A language code can contain dash-separated
@@ -307,15 +348,16 @@ class N3Lexer {
307
348
  // input is not finished, another subtag may still arrive in a later chunk and
308
349
  // the match would be premature; wait for more input in that case.
309
350
  // A double dash starts a direction code, which cannot extend the language code.
310
- if (this._previousMarker === 'literal' && (match = this._langcode.exec(input)) && match[1] !== 'version') {
311
- if (!inputFinished && input[match[0].length] === '-' && input[match[0].length + 1] !== '-') match = null;else type = 'langcode', value = match[1];
351
+ if (this._previousMarker === 'literal' && (match = execAt(this._langcode, input, pos)) && match[1] !== 'version') {
352
+ const end = pos + match[0].length;
353
+ if (!inputFinished && input[end] === '-' && input[end + 1] !== '-') match = null;else type = 'langcode', value = match[1];
312
354
  }
313
355
  // Try to find a keyword
314
- else if (match = this._atKeyword.exec(input)) type = match[0];
356
+ else if (match = execAt(this._atKeyword, input, pos)) type = match[0];
315
357
  break;
316
358
  case '.':
317
359
  // Try to find a dot as punctuation
318
- if (input.length === 1 ? inputFinished : input[1] < '0' || input[1] > '9') {
360
+ if (input.length - pos === 1 ? inputFinished : input[pos + 1] < '0' || input[pos + 1] > '9') {
319
361
  type = '.';
320
362
  matchLength = 1;
321
363
  break;
@@ -334,10 +376,10 @@ class N3Lexer {
334
376
  case '9':
335
377
  case '+':
336
378
  case '-':
337
- if (input[1] === '-') {
379
+ if (input[pos + 1] === '-') {
338
380
  // Try to find a direction code
339
381
  if (this._previousMarker === 'langcode') {
340
- if (input.startsWith('--ltr')) type = 'dircode', value = 'ltr', matchLength = 5;else if (input.startsWith('--rtl')) type = 'dircode', value = 'rtl', matchLength = 5;
382
+ if (input.startsWith('--ltr', pos)) type = 'dircode', value = 'ltr', matchLength = 5;else if (input.startsWith('--rtl', pos)) type = 'dircode', value = 'rtl', matchLength = 5;
341
383
  }
342
384
  break;
343
385
  }
@@ -345,7 +387,7 @@ class N3Lexer {
345
387
  // Try to find a number. Since it can contain (but not end with) a dot,
346
388
  // we always need a non-dot character before deciding it is a number.
347
389
  // Therefore, try inserting a space if we're at the end of the input.
348
- if (match = this._number.exec(input) || inputFinished && (match = this._number.exec(`${input} `))) {
390
+ if (match = execAt(this._number, input, pos) || inputFinished && (match = execAtEnd(this._number, input, pos))) {
349
391
  type = 'literal', value = match[0];
350
392
  prefix = typeof match[1] === 'string' ? xsd.double : typeof match[2] === 'string' ? xsd.decimal : xsd.integer;
351
393
  }
@@ -359,42 +401,42 @@ class N3Lexer {
359
401
  case 'V':
360
402
  case 'v':
361
403
  // Try to find a SPARQL-style keyword
362
- if (match = this._keyword.exec(input)) type = match[0].toUpperCase();else inconclusive = true;
404
+ if (match = execAt(this._keyword, input, pos)) type = match[0].toUpperCase();else inconclusive = true;
363
405
  break;
364
406
  case 'f':
365
407
  case 't':
366
408
  // Try to match a boolean
367
- if (this._boolean.test(input)) type = 'literal', value = firstChar === 't' ? 'true' : 'false', prefix = xsd.boolean, matchLength = value.length;else inconclusive = true;
409
+ if (testAt(this._boolean, input, pos)) type = 'literal', value = firstChar === 't' ? 'true' : 'false', prefix = xsd.boolean, matchLength = value.length;else inconclusive = true;
368
410
  break;
369
411
  case 'a':
370
412
  // Try to find an abbreviated predicate
371
- if (this._shortPredicates.test(input)) type = 'abbreviation', value = 'a', matchLength = 1;else inconclusive = true;
413
+ if (testAt(this._shortPredicates, input, pos)) type = 'abbreviation', value = 'a', matchLength = 1;else inconclusive = true;
372
414
  break;
373
415
  case 'h':
374
416
  case 'o':
375
417
  // Try to find an N3 verb keyword
376
- if (this._n3Mode && (match = this._matchN3Verb(input, inputFinished))) type = match[0];else inconclusive = true;
418
+ if (this._n3Mode && (match = this._matchN3Verb(input, pos, inputFinished))) type = match[0];else inconclusive = true;
377
419
  break;
378
420
  case 'i':
379
421
  // Try to find an IRI property list identifier or N3 verb keyword
380
- if (this._n3Mode && this._n3Id.test(input)) type = 'id', matchLength = 2;else if (this._n3Mode && (match = this._matchN3Verb(input, inputFinished))) type = match[0];else inconclusive = true;
422
+ if (this._n3Mode && testAt(this._n3Id, input, pos)) type = 'id', matchLength = 2;else if (this._n3Mode && (match = this._matchN3Verb(input, pos, inputFinished))) type = match[0];else inconclusive = true;
381
423
  break;
382
424
  case '=':
383
425
  // Try to find an implication arrow or equals sign
384
- if (this._n3Mode && input.length > 1) {
426
+ if (this._n3Mode && input.length - pos > 1) {
385
427
  type = 'abbreviation';
386
- if (input[1] !== '>') matchLength = 1, value = '=';else matchLength = 2, value = '>';
428
+ if (input[pos + 1] !== '>') matchLength = 1, value = '=';else matchLength = 2, value = '>';
387
429
  }
388
430
  break;
389
431
  case '!':
390
432
  if (!this._n3Mode) break;
391
433
  case ')':
392
- if (!inputFinished && (input.length === 1 || input.length === 2 && input[1] === '>')) {
434
+ if (!inputFinished && (input.length - pos === 1 || input.length - pos === 2 && input[pos + 1] === '>')) {
393
435
  // Don't consume yet, as it *could* become a triple term end.
394
436
  break;
395
437
  }
396
438
  // Try to find a triple term
397
- if (input.length > 2 && input[1] === '>' && input[2] === '>') {
439
+ if (input.length - pos > 2 && input[pos + 1] === '>' && input[pos + 2] === '>') {
398
440
  type = ')>>', matchLength = 3;
399
441
  break;
400
442
  }
@@ -412,15 +454,15 @@ class N3Lexer {
412
454
  break;
413
455
  case '{':
414
456
  // We need at least 2 tokens lookahead to distinguish "{|" and "{ "
415
- if (!this._lineMode && input.length >= 2) {
457
+ if (!this._lineMode && input.length - pos >= 2) {
416
458
  // Try to find a quoted triple annotation start
417
- if (input[1] === '|') type = '{|', matchLength = 2;else type = firstChar, matchLength = 1;
459
+ if (input[pos + 1] === '|') type = '{|', matchLength = 2;else type = firstChar, matchLength = 1;
418
460
  }
419
461
  break;
420
462
  case '|':
421
463
  // We need 2 tokens lookahead to parse "|}"
422
464
  // Try to find a quoted triple annotation end
423
- if (input.length >= 2 && input[1] === '}') type = '|}', matchLength = 2;
465
+ if (input.length - pos >= 2 && input[pos + 1] === '}') type = '|}', matchLength = 2;
424
466
  break;
425
467
  default:
426
468
  inconclusive = true;
@@ -429,11 +471,14 @@ class N3Lexer {
429
471
  // Some first characters do not allow an immediate decision, so inspect more
430
472
  if (inconclusive) {
431
473
  // Try to find a prefix
432
- if ((this._previousMarker === '@prefix' || this._previousMarker === 'PREFIX') && (match = this._prefix.exec(input))) type = 'prefix', value = match[1] || '';
474
+ if ((this._previousMarker === '@prefix' || this._previousMarker === 'PREFIX') && (match = execAt(this._prefix, input, pos))) type = 'prefix', value = match[1] || '';
475
+ // Try to find an additional directive keyword
476
+ // (at the end of the input, only a short final word can be one)
477
+ else if (this._directive !== null && this._directiveStarts.includes(firstChar) && ((match = execAt(this._directive, input, pos)) || inputFinished && input.length - pos <= this._directiveMaxLength && (match = execAtEnd(this._directive, input, pos)))) type = match[0].toUpperCase();
433
478
  // Try to find a prefixed name. Since it can contain (but not end with) a dot,
434
479
  // we always need a non-dot character before deciding it is a prefixed name.
435
480
  // Therefore, try inserting a space if we're at the end of the input.
436
- else if ((match = this._prefixed.exec(input)) || inputFinished && (match = this._prefixed.exec(`${input} `))) {
481
+ else if ((match = execAt(this._prefixed, input, pos)) || inputFinished && (match = execAtEnd(this._prefixed, input, pos))) {
437
482
  type = 'prefixed', prefix = match[1] || '';
438
483
  value = this._unescape(match[2], localNameEscapeReplacements);
439
484
  lexicalLength = prefix.length + match[2].length + 1;
@@ -459,16 +504,14 @@ class N3Lexer {
459
504
  // We could be in streaming mode, and then we just wait for more input to arrive.
460
505
  // Otherwise, a syntax error has occurred in the input.
461
506
  // One exception: error on an unaccounted linebreak (= not inside a triple-quoted literal).
462
- if (inputFinished || !/^'''|^"""/.test(input) && /\n|\r/.test(input)) return reportSyntaxError(this);else {
463
- this._linePosition = currentLineLength - input.length;
464
- return this._input = input;
465
- }
507
+ if (inputFinished || !input.startsWith("'''", pos) && !input.startsWith('"""', pos) && /\n|\r/.test(input.slice(pos))) return reportSyntaxError(this, input, pos);else return this._suspend(input, pos, currentLineLength);
466
508
  }
467
509
 
468
510
  // Emit the parsed token
469
511
  // Consumption includes separator whitespace; lexicalLength excludes it
470
- // and any synthetic EOF space. slice below clamps consumption to the input.
512
+ // and any synthetic EOF space. Consumption is clamped to the input below.
471
513
  const length = matchLength || match[0].length;
514
+ const start = currentLineLength - (input.length - pos);
472
515
  let token;
473
516
  if (finalLineLength) {
474
517
  token = {
@@ -476,23 +519,23 @@ class N3Lexer {
476
519
  value,
477
520
  prefix,
478
521
  line,
479
- start: currentLineLength - input.length,
522
+ start,
480
523
  end: finalLineLength,
481
524
  endLine: this._line
482
525
  };
483
526
  callback(null, token);
484
- } else token = emitToken(type, value, prefix, line, lexicalLength || length);
527
+ } else token = emitToken(type, value, prefix, line, start, lexicalLength || length);
485
528
  this.previousToken = token;
486
- this._previousMarker = type;
529
+ // The string of a version declaration cannot take a language tag, so a following @keyword is a keyword
530
+ this._previousMarker = type === 'literal' && (this._previousMarker === 'VERSION' || this._previousMarker === '@version') ? 'version' : type;
487
531
 
488
532
  // Advance to next part to tokenize
489
- input = input.slice(length);
490
- if (finalLineLength) currentLineLength = input.length + finalLineLength;
533
+ pos = Math.min(pos + length, input.length);
534
+ if (finalLineLength) currentLineLength = input.length - pos + finalLineLength;
491
535
  }
492
536
 
493
537
  // Emits a comment at its exact position within matched whitespace.
494
- function emitComment(value, line, offset) {
495
- const start = currentLineLength - input.length + offset;
538
+ function emitComment(value, line, start) {
496
539
  callback(null, {
497
540
  type: 'comment',
498
541
  value,
@@ -503,49 +546,53 @@ class N3Lexer {
503
546
  });
504
547
  }
505
548
  // Emits the token through the callback
506
- function emitToken(type, value, prefix, line, length) {
507
- const start = input ? currentLineLength - input.length : currentLineLength;
508
- const end = start + length;
549
+ function emitToken(type, value, prefix, line, start, length) {
509
550
  const token = {
510
551
  type,
511
552
  value,
512
553
  prefix,
513
554
  line,
514
555
  start,
515
- end
556
+ end: start + length
516
557
  };
517
558
  callback(null, token);
518
559
  return token;
519
560
  }
520
561
  // Signals the syntax error through the callback
521
- function reportSyntaxError(self) {
522
- callback(self._syntaxError(/^\S*/.exec(input)[0]));
562
+ function reportSyntaxError(self, input, pos) {
563
+ callback(self._syntaxError(execAt(nonWhitespace, input, pos)[0]));
523
564
  }
524
565
  }
525
566
 
567
+ // ### `_suspend` keeps the unconsumed input until more input arrives
568
+ _suspend(input, pos, currentLineLength) {
569
+ this._linePosition = currentLineLength - (input.length - pos);
570
+ return this._input = input.slice(pos);
571
+ }
572
+
526
573
  // ### `_matchN3Verb` matches an N3 verb unless the input is a longer prefixed name
527
- _matchN3Verb(input, inputFinished) {
528
- const verb = this._n3Verb.exec(input);
574
+ _matchN3Verb(input, pos, inputFinished) {
575
+ const verb = execAt(this._n3Verb, input, pos);
529
576
  if (!verb) return null;
530
577
 
531
578
  // Most verb boundaries cannot be part of a prefix, so keep the common path fast.
532
- const next = input[verb[0].length];
579
+ const next = input[pos + verb[0].length];
533
580
  if (next !== '-' && next !== '_' && (next < '0' || next > '9')) return verb;
534
581
 
535
582
  // A prefix can start with a verb and continue with characters that are also
536
583
  // valid verb boundaries. Prefer the longer prefixed name when it is complete.
537
- if (this._prefixed.exec(input)) return null;
584
+ if (execAt(this._prefixed, input, pos)) return null;
538
585
  // Appending to the input only matters when a prefixed name could run up to
539
586
  // its end, which a character that cannot occur in prefixed names rules out.
540
587
  // This avoids copying the rest of the document for every such verb.
541
- if (nonPrefixedNameChar.test(input)) return verb;
542
- if (this._prefixed.exec(`${input} `)) return null;
588
+ if (testAt(nonPrefixedNameChar, input, pos)) return verb;
589
+ if (execAtEnd(this._prefixed, input, pos)) return null;
543
590
 
544
591
  // If a stream chunk ends partway through such a prefix, wait for the colon
545
592
  // instead of prematurely emitting the verb. Appending ": " lets the prefix
546
593
  // grammar determine whether all input seen so far can be a complete prefix.
547
594
  if (!inputFinished) {
548
- const prefix = this._prefix.exec(`${input}: `);
595
+ const prefix = execAt(this._prefix, `${input.slice(pos)}: `, 0);
549
596
  if (prefix) return null;
550
597
  }
551
598
  return verb;
@@ -592,19 +639,19 @@ class N3Lexer {
592
639
  return result + item.slice(start);
593
640
  }
594
641
 
595
- // ### `_parseLiteral` parses a literal into an unescaped value
596
- _parseLiteral(input) {
642
+ // ### `_parseLiteral` parses a literal at the given position into an unescaped value
643
+ _parseLiteral(input, pos) {
597
644
  // Ensure we have enough lookahead to identify triple-quoted strings
598
- if (input.length >= 3) {
645
+ if (input.length - pos >= 3) {
599
646
  // The caller has already identified a single or double quote.
600
- const quote = input[0];
601
- const openingLength = input[1] === quote && input[2] === quote ? 3 : 1;
647
+ const quote = input[pos];
648
+ const openingLength = input[pos + 1] === quote && input[pos + 2] === quote ? 3 : 1;
602
649
  let opening = quote;
603
650
  if (openingLength === 3) opening = quote === '"' ? '"""' : "'''";
604
651
 
605
652
  // Find the next candidate closing quotes
606
- let closingPos = Math.max(this._literalClosingPos, openingLength);
607
- while ((closingPos = input.indexOf(opening, closingPos)) > 0) {
653
+ let closingPos = pos + Math.max(this._literalClosingPos, openingLength);
654
+ while ((closingPos = input.indexOf(opening, closingPos)) > pos) {
608
655
  // Count backslashes right before the closing quotes
609
656
  let backslashCount = 0;
610
657
  while (input[closingPos - backslashCount - 1] === '\\') backslashCount++;
@@ -613,10 +660,10 @@ class N3Lexer {
613
660
  // means these are actual, non-escaped closing quotes
614
661
  if (backslashCount % 2 === 0) {
615
662
  // Extract and unescape the value
616
- const raw = input.substring(openingLength, closingPos),
663
+ const raw = input.substring(pos + openingLength, closingPos),
617
664
  lines = raw.split(/\r\n|\r|\n/),
618
665
  lineCount = lines.length - 1;
619
- const matchLength = closingPos + openingLength;
666
+ const matchLength = closingPos - pos + openingLength;
620
667
  // Only triple-quoted strings can be multi-line
621
668
  if (openingLength === 1 && lineCount !== 0 || openingLength === 3 && this._lineMode) break;
622
669
  this._line += lineCount;
@@ -629,7 +676,7 @@ class N3Lexer {
629
676
  }
630
677
  closingPos++;
631
678
  }
632
- this._literalClosingPos = input.length - openingLength + 1;
679
+ this._literalClosingPos = input.length - pos - openingLength + 1;
633
680
  }
634
681
  return {
635
682
  value: '',
@@ -695,31 +742,39 @@ class N3Lexer {
695
742
  }
696
743
  // Otherwise, the input must be a stream
697
744
  else {
698
- this._pendingBuffer = null;
745
+ let decoder,
746
+ retryLength = 0;
699
747
  if (typeof input.setEncoding === 'function') input.setEncoding('utf8');
700
748
  // Adds the data chunk to the buffer and parses as far as possible
701
749
  input.on('data', data => {
702
750
  if (this._tokenization === tokenization && this._input !== null && data.length !== 0) {
703
- // Prepend any previous pending writes
704
- if (this._pendingBuffer) {
705
- data = _buffer.Buffer.concat([this._pendingBuffer, data]);
706
- this._pendingBuffer = null;
707
- }
708
- // Hold if the buffer ends in an incomplete unicode sequence
709
- if (data[data.length - 1] & 0x80) {
710
- this._pendingBuffer = data;
751
+ // Decode bytes, keeping an incomplete trailing character for the next chunk
752
+ if (typeof data !== 'string') {
753
+ decoder = decoder || new TextDecoder('utf-8', {
754
+ ignoreBOM: true
755
+ });
756
+ if (!(data = decoder.decode(data, {
757
+ stream: true
758
+ }))) return;
711
759
  }
712
- // Otherwise, tokenize as far as possible
713
- else {
714
- // Only read a BOM at the start
715
- if (typeof this._input === 'undefined') this._input = this._readStartingBom(typeof data === 'string' ? data : data.toString());else this._input += data;
760
+ // Only read a BOM at the start
761
+ if (typeof this._input === 'undefined') this._input = this._readStartingBom(data);else this._input += data;
762
+ // Tokenize as far as possible. When a previous attempt left a long unfinished token,
763
+ // wait until the buffered input has doubled, so the token is not rescanned for every chunk.
764
+ if (this._input.length >= retryLength) {
716
765
  this._tokenizeToEnd(callback, false);
766
+ retryLength = this._input !== null && this._input.length > MIN_RESCAN_LENGTH ? 2 * this._input.length : 0;
717
767
  }
718
768
  }
719
769
  });
720
770
  // Parses until the end
721
771
  input.on('end', () => {
722
- if (this._tokenization === tokenization && typeof this._input === 'string') this._tokenizeToEnd(callback, true);
772
+ if (this._tokenization === tokenization && this._input !== null) {
773
+ // Decode any incomplete character left at the end
774
+ const rest = decoder ? decoder.decode() : '';
775
+ if (rest) this._input = typeof this._input === 'string' ? this._input + rest : rest;
776
+ if (typeof this._input === 'string') this._tokenizeToEnd(callback, true);
777
+ }
723
778
  });
724
779
  input.on('error', error => {
725
780
  if (this._tokenization === tokenization) callback(error);