n3 2.11.2 → 3.0.0-alpha.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/lib/N3Lexer.js CHANGED
@@ -54,8 +54,7 @@ const localNameEscapeReplacements = {
54
54
  };
55
55
  const illegalIriChars = /[\x00-\x20<>\\"\{\}\|\^\`]/;
56
56
  // Characters that cannot occur in a prefixed name, not even escaped
57
- // (global, so that testAt searches the rest of the input from a position)
58
- const nonPrefixedNameChar = /[\s<>"{}|^`]/g;
57
+ const nonPrefixedNameChar = /[\s<>"{}|^`]/;
59
58
 
60
59
  // A valid code point is a Unicode scalar value: at most U+10FFFF and not a surrogate
61
60
  function isValidCodePoint(charCode) {
@@ -71,48 +70,30 @@ const lineModeRegExps = {
71
70
  _whitespace: true
72
71
  };
73
72
  const invalidRegExp = /$0^/;
74
- const nonWhitespace = /\S*/y;
75
-
76
- // Matches a sticky regular expression at the given position of the input
77
- function execAt(regExp, input, pos) {
78
- regExp.lastIndex = pos;
79
- return regExp.exec(input);
80
- }
81
- function testAt(regExp, input, pos) {
82
- regExp.lastIndex = pos;
83
- return regExp.test(input);
84
- }
85
- // Matches the rest of the input followed by a space, as at the end of the input,
86
- // a token that can contain (but not end with) a dot needs a non-dot character after it
87
- function execAtEnd(regExp, input, pos) {
88
- regExp.lastIndex = 0;
89
- return regExp.exec(`${input.slice(pos)} `);
90
- }
91
73
 
92
74
  // ## Constructor
93
75
  class N3Lexer {
94
76
  constructor(options) {
95
77
  // ## Regular expressions
96
- // It's slightly faster to have these as properties than as in-scope variables.
97
- // They are sticky, so they only match at the `lastIndex` set by `execAt`.
98
- this._iri = /<((?:[^ <>{}\\]|\\[uU])+)>[ \t]*/y; // IRI with escape sequences; needs sanity check after unescaping
99
- this._unescapedIri = /<([^\x00-\x20<>\\"\{\}\|\^\`]*)>[ \t]*/y; // IRI without escape sequences; no unescaping
100
- this._simpleQuotedString = /"([^"\\\r\n]*)"(?=[^"])/y; // string without escape sequences
101
- this._simpleApostropheString = /'([^'\\\r\n]*)'(?=[^'])/y;
102
- this._langcode = /@([a-z]+(?:-[a-z0-9]+)*)(?=[^a-z0-9])/iy;
103
- this._prefix = /((?:[A-Za-z\xc0-\xd6\xd8-\xf6\xf8-\u02ff\u0370-\u037d\u037f-\u1fff\u200c\u200d\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff])(?:\.?[\-0-9A-Z_a-z\xb7\xc0-\xd6\xd8-\xf6\xf8-\u037d\u037f-\u1fff\u200c\u200d\u203f\u2040\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff])*)?:(?=[#\s<])/y;
104
- this._prefixed = /((?:[A-Za-z\xc0-\xd6\xd8-\xf6\xf8-\u02ff\u0370-\u037d\u037f-\u1fff\u200c\u200d\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff])(?:\.?[\-0-9A-Z_a-z\xb7\xc0-\xd6\xd8-\xf6\xf8-\u037d\u037f-\u1fff\u200c\u200d\u203f\u2040\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff])*)?:((?:(?:[0-9:A-Z_a-z\xc0-\xd6\xd8-\xf6\xf8-\u02ff\u0370-\u037d\u037f-\u1fff\u200c\u200d\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff]|%[0-9a-fA-F]{2}|\\[!#-\/;=?\-@_~])(?:(?:[\.\-0-9:A-Z_a-z\xb7\xc0-\xd6\xd8-\xf6\xf8-\u037d\u037f-\u1fff\u200c\u200d\u203f\u2040\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff]|%[0-9a-fA-F]{2}|\\[!#-\/;=?\-@_~])*(?:[\-0-9:A-Z_a-z\xb7\xc0-\xd6\xd8-\xf6\xf8-\u037d\u037f-\u1fff\u200c\u200d\u203f\u2040\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff]|%[0-9a-fA-F]{2}|\\[!#-\/;=?\-@_~]))?)?)(?:[ \t]+|(?=\.?[,;!\^\s#()\[\]\{\}"'<>]))/y;
105
- this._variable = /\?(?:(?:[A-Z_a-z\xc0-\xd6\xd8-\xf6\xf8-\u02ff\u0370-\u037d\u037f-\u1fff\u200c\u200d\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff])(?:[\-0-9:A-Z_a-z\xb7\xc0-\xd6\xd8-\xf6\xf8-\u037d\u037f-\u1fff\u200c\u200d\u203f\u2040\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff])*)(?=[.,;!\^\s#()\[\]\{\}"'<>])/y;
106
- this._blank = /_:((?:[0-9A-Z_a-z\xc0-\xd6\xd8-\xf6\xf8-\u02ff\u0370-\u037d\u037f-\u1fff\u200c\u200d\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff])(?:\.?[\-0-9A-Z_a-z\xb7\xc0-\xd6\xd8-\xf6\xf8-\u037d\u037f-\u1fff\u200c\u200d\u203f\u2040\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff])*)(?:[ \t]+|(?=\.?[,;:!\^\s#()\[\]\{\}"'<>]))/y;
107
- this._number = /[\-+]?(?:(\d+\.\d*|\.?\d+)[eE][\-+]?\d+|(?=\.?\d)\d*(?:(\.)\d+)?)(?=\.?[,;:!\^\s#()\[\]\{\}"'<>])/y;
108
- this._boolean = /(?:true|false)(?=[.,;!\^\s#()\[\]\{\}"'<>])/y;
109
- this._atKeyword = /@[a-z]+(?=[\s#<:])/iy;
110
- this._keyword = /(?:PREFIX|BASE|VERSION|GRAPH)(?=[\s#<])/iy;
111
- this._n3Verb = /(?:has|is|of)(?=[\s#()\[\]\{\}"'<>?_+\-0-9])/y;
112
- this._n3Id = /id(?=[\s#<])/y;
113
- this._shortPredicates = /a(?=[\s#()\[\]\{\}"'<>])/y;
114
- this._commentLine = /[ \t]*#([^\n\r]*)(?:\r\n|\n|\r)([ \t]*)/y;
115
- this._whitespace = /[ \t]+/y;
78
+ // It's slightly faster to have these as properties than as in-scope variables
79
+ this._iri = /^<((?:[^ <>{}\\]|\\[uU])+)>[ \t]*/; // IRI with escape sequences; needs sanity check after unescaping
80
+ this._unescapedIri = /^<([^\x00-\x20<>\\"\{\}\|\^\`]*)>[ \t]*/; // IRI without escape sequences; no unescaping
81
+ this._simpleQuotedString = /^"([^"\\\r\n]*)"(?=[^"])/; // string without escape sequences
82
+ this._simpleApostropheString = /^'([^'\\\r\n]*)'(?=[^'])/;
83
+ this._langcode = /^@([a-z]+(?:-[a-z0-9]+)*)(?=[^a-z0-9])/i;
84
+ this._prefix = /^((?:[A-Za-z\xc0-\xd6\xd8-\xf6\xf8-\u02ff\u0370-\u037d\u037f-\u1fff\u200c\u200d\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff])(?:\.?[\-0-9A-Z_a-z\xb7\xc0-\xd6\xd8-\xf6\xf8-\u037d\u037f-\u1fff\u200c\u200d\u203f\u2040\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff])*)?:(?=[#\s<])/;
85
+ this._prefixed = /^((?:[A-Za-z\xc0-\xd6\xd8-\xf6\xf8-\u02ff\u0370-\u037d\u037f-\u1fff\u200c\u200d\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff])(?:\.?[\-0-9A-Z_a-z\xb7\xc0-\xd6\xd8-\xf6\xf8-\u037d\u037f-\u1fff\u200c\u200d\u203f\u2040\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff])*)?:((?:(?:[0-9:A-Z_a-z\xc0-\xd6\xd8-\xf6\xf8-\u02ff\u0370-\u037d\u037f-\u1fff\u200c\u200d\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff]|%[0-9a-fA-F]{2}|\\[!#-\/;=?\-@_~])(?:(?:[\.\-0-9:A-Z_a-z\xb7\xc0-\xd6\xd8-\xf6\xf8-\u037d\u037f-\u1fff\u200c\u200d\u203f\u2040\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff]|%[0-9a-fA-F]{2}|\\[!#-\/;=?\-@_~])*(?:[\-0-9:A-Z_a-z\xb7\xc0-\xd6\xd8-\xf6\xf8-\u037d\u037f-\u1fff\u200c\u200d\u203f\u2040\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff]|%[0-9a-fA-F]{2}|\\[!#-\/;=?\-@_~]))?)?)(?:[ \t]+|(?=\.?[,;!\^\s#()\[\]\{\}"'<>]))/;
86
+ this._variable = /^\?(?:(?:[A-Z_a-z\xc0-\xd6\xd8-\xf6\xf8-\u02ff\u0370-\u037d\u037f-\u1fff\u200c\u200d\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff])(?:[\-0-9:A-Z_a-z\xb7\xc0-\xd6\xd8-\xf6\xf8-\u037d\u037f-\u1fff\u200c\u200d\u203f\u2040\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff])*)(?=[.,;!\^\s#()\[\]\{\}"'<>])/;
87
+ this._blank = /^_:((?:[0-9A-Z_a-z\xc0-\xd6\xd8-\xf6\xf8-\u02ff\u0370-\u037d\u037f-\u1fff\u200c\u200d\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff])(?:\.?[\-0-9A-Z_a-z\xb7\xc0-\xd6\xd8-\xf6\xf8-\u037d\u037f-\u1fff\u200c\u200d\u203f\u2040\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff])*)(?:[ \t]+|(?=\.?[,;:!\^\s#()\[\]\{\}"'<>]))/;
88
+ this._number = /^[\-+]?(?:(\d+\.\d*|\.?\d+)[eE][\-+]?\d+|(?=\.?\d)\d*(?:(\.)\d+)?)(?=\.?[,;:!\^\s#()\[\]\{\}"'<>])/;
89
+ this._boolean = /^(?:true|false)(?=[.,;!\^\s#()\[\]\{\}"'<>])/;
90
+ this._atKeyword = /^@[a-z]+(?=[\s#<:])/i;
91
+ this._keyword = /^(?:PREFIX|BASE|VERSION|GRAPH)(?=[\s#<])/i;
92
+ this._n3Verb = /^(?:has|is|of)(?=[\s#()\[\]\{\}"'<>?_+\-0-9])/;
93
+ this._n3Id = /^id(?=[\s#<])/;
94
+ this._shortPredicates = /^a(?=[\s#()\[\]\{\}"'<>])/;
95
+ this._commentLine = /^[ \t]*#([^\n\r]*)(?:\r\n|\n|\r)([ \t]*)/;
96
+ this._whitespace = /^[ \t]+/;
116
97
  options = options || {};
117
98
 
118
99
  // Whether the log:isImpliedBy predicate is supported
@@ -140,71 +121,77 @@ class N3Lexer {
140
121
 
141
122
  // ### `_tokenizeToEnd` tokenizes as for as possible, emitting tokens through the callback
142
123
  _tokenizeToEnd(callback, inputFinished) {
143
- // Continue parsing as far as possible; the loop will return eventually.
144
- // Rather than slicing off every token, track the position of the remaining input;
145
- // the regular expressions are sticky, so they match at that position.
146
- const input = this._input;
147
- let pos = 0;
124
+ // Continue parsing as far as possible; the loop will return eventually
125
+ let input = this._input;
148
126
  let currentLineLength = this._linePosition + input.length;
149
127
  while (true) {
150
128
  // Consume one separator line at a time, including its following indentation.
151
129
  while (true) {
152
- let charCode = input.charCodeAt(pos),
130
+ let charCode = input.charCodeAt(0),
153
131
  separatorLength = 0;
154
132
  if (charCode === SPACE || charCode === TAB) {
155
- const next = input.charCodeAt(pos + 1);
156
- separatorLength = next === SPACE || next === TAB ? execAt(this._whitespace, input, pos)[0].length : 1;
157
- charCode = input.charCodeAt(pos + separatorLength);
133
+ const next = input.charCodeAt(1);
134
+ separatorLength = next === SPACE || next === TAB ? this._whitespace.exec(input)[0].length : 1;
135
+ charCode = input.charCodeAt(separatorLength);
158
136
  }
159
137
  if (charCode === HASH) {
160
- const comment = execAt(this._commentLine, input, pos);
138
+ const comment = this._commentLine.exec(input);
161
139
  if (comment) {
162
140
  const commentLength = comment[0].length;
163
141
  // Keep a trailing CR buffered in case the next chunk starts with LF.
164
- if (!inputFinished && pos + commentLength === input.length && input.charCodeAt(input.length - 1) === CR) return this._suspend(input, pos, currentLineLength);
165
- if (this.comments) emitComment(comment[1], this._line, currentLineLength - (input.length - pos) + separatorLength);
166
- pos += commentLength;
167
- currentLineLength = input.length - pos + comment[2].length;
142
+ if (!inputFinished && commentLength === input.length && input.charCodeAt(commentLength - 1) === CR) {
143
+ this._linePosition = currentLineLength - input.length;
144
+ return this._input = input;
145
+ }
146
+ if (this.comments) emitComment(comment[1], this._line, separatorLength);
147
+ input = input.slice(commentLength);
148
+ currentLineLength = input.length + comment[2].length;
168
149
  this._line++;
169
150
  } else {
170
151
  // A comment without a line ending stays buffered until EOF.
171
- pos += separatorLength;
172
- if (!inputFinished) return this._suspend(input, pos, currentLineLength);
173
- if (this.comments) emitComment(input.slice(pos + 1), this._line, currentLineLength - (input.length - pos));
174
- pos = input.length;
152
+ input = input.slice(separatorLength);
153
+ if (!inputFinished) {
154
+ this._linePosition = currentLineLength - input.length;
155
+ return this._input = input;
156
+ }
157
+ if (this.comments) emitComment(input.slice(1), this._line, 0);
158
+ input = '';
175
159
  break;
176
160
  }
177
161
  } else if (charCode === LF || charCode === CR) {
178
162
  // A CR at the end of a chunk may still be followed by LF.
179
- if (!inputFinished && charCode === CR && pos + separatorLength + 1 === input.length) return this._suspend(input, pos, currentLineLength);
180
- separatorLength += charCode === CR && input.charCodeAt(pos + separatorLength + 1) === LF ? 2 : 1;
163
+ if (!inputFinished && charCode === CR && separatorLength + 1 === input.length) {
164
+ this._linePosition = currentLineLength - input.length;
165
+ return this._input = input;
166
+ }
167
+ separatorLength += charCode === CR && input.charCodeAt(separatorLength + 1) === LF ? 2 : 1;
181
168
  // Indentation is consumed with the newline, but belongs to the next line's columns.
182
169
  let indentationLength = 0;
183
- const next = input.charCodeAt(pos + separatorLength);
170
+ const next = input.charCodeAt(separatorLength);
184
171
  if (next === SPACE || next === TAB) {
185
- const following = input.charCodeAt(pos + separatorLength + 1);
186
- indentationLength = following === SPACE || following === TAB ? execAt(this._whitespace, input, pos + separatorLength)[0].length : 1;
172
+ const following = input.charCodeAt(separatorLength + 1);
173
+ indentationLength = following === SPACE || following === TAB ? this._whitespace.exec(input.slice(separatorLength))[0].length : 1;
187
174
  }
188
- pos += separatorLength + indentationLength;
189
- currentLineLength = input.length - pos + indentationLength;
175
+ input = input.slice(separatorLength + indentationLength);
176
+ currentLineLength = input.length + indentationLength;
190
177
  this._line++;
191
178
  } else {
192
- pos += separatorLength;
179
+ if (separatorLength !== 0) input = input.slice(separatorLength);
193
180
  break;
194
181
  }
195
182
  }
196
- if (pos >= input.length) {
197
- this._linePosition = currentLineLength;
183
+ if (input.length === 0) {
198
184
  if (inputFinished) {
199
- emitToken('eof', '', '', this._line, currentLineLength, 0);
200
- return this._input = null;
185
+ input = null;
186
+ emitToken('eof', '', '', this._line, 0);
201
187
  }
202
- return this._input = '';
188
+ this._linePosition = currentLineLength;
189
+ return this._input = input;
203
190
  }
204
191
 
205
192
  // Look for specific token types based on the first character
206
193
  const line = this._line,
207
- firstChar = input[pos];
194
+ firstChar = input[0];
208
195
  let type = '',
209
196
  value = '',
210
197
  prefix = '',
@@ -216,13 +203,13 @@ class N3Lexer {
216
203
  switch (firstChar) {
217
204
  case '^':
218
205
  // We need at least 3 tokens lookahead to distinguish ^^<IRI> and ^^pre:fixed
219
- if (input.length - pos < 3) break;
206
+ if (input.length < 3) break;
220
207
  // Try to match a type
221
- else if (input[pos + 1] === '^') {
208
+ else if (input[1] === '^') {
222
209
  this._previousMarker = '^^';
223
210
  // Move to type IRI or prefixed name
224
- pos += 2;
225
- if (input[pos] !== '<') {
211
+ input = input.slice(2);
212
+ if (input[0] !== '<') {
226
213
  inconclusive = true;
227
214
  break;
228
215
  }
@@ -238,53 +225,53 @@ class N3Lexer {
238
225
  // Fall through in case the type is an IRI
239
226
  case '<':
240
227
  // Try to find a full IRI without escape sequences
241
- if (match = execAt(this._unescapedIri, input, pos)) {
228
+ if (match = this._unescapedIri.exec(input)) {
242
229
  type = 'IRI', value = match[1];
243
230
  lexicalLength = match[1].length + 2;
244
231
  }
245
232
  // Try to find a full IRI with escape sequences
246
- else if (match = execAt(this._iri, input, pos)) {
233
+ else if (match = this._iri.exec(input)) {
247
234
  value = this._unescape(match[1], stringEscapeReplacements);
248
- if (value === null || illegalIriChars.test(value)) return reportSyntaxError(this, input, pos);
235
+ if (value === null || illegalIriChars.test(value)) return reportSyntaxError(this);
249
236
  type = 'IRI';
250
237
  lexicalLength = match[1].length + 2;
251
238
  }
252
239
  // Try to find a triple term
253
- else if (input.length - pos > 2 && input[pos + 1] === '<' && input[pos + 2] === '(') type = '<<(', matchLength = 3;
240
+ else if (input.length > 2 && input[1] === '<' && input[2] === '(') type = '<<(', matchLength = 3;
254
241
  // Try to find a reified triple
255
- else if (!this._lineMode && input.length - pos > (inputFinished ? 1 : 2) && input[pos + 1] === '<') type = '<<', matchLength = 2;
242
+ else if (!this._lineMode && input.length > (inputFinished ? 1 : 2) && input[1] === '<') type = '<<', matchLength = 2;
256
243
  // Try to find a backwards implication arrow
257
- else if (this._n3Mode && input.length - pos > 1 && input[pos + 1] === '=') {
244
+ else if (this._n3Mode && input.length > 1 && input[1] === '=') {
258
245
  matchLength = 2;
259
246
  if (this._isImpliedBy) type = 'abbreviation', value = '<';else type = 'inverse', value = '>';
260
247
  }
261
248
  // Try to find an inverted predicate marker
262
- else if (this._n3Mode && input.length - pos > 1 && input[pos + 1] === '-') type = 'inversePredicate', matchLength = 2;
249
+ else if (this._n3Mode && input.length > 1 && input[1] === '-') type = 'inversePredicate', matchLength = 2;
263
250
  break;
264
251
  case '>':
265
252
  // Try to find a reified triple
266
- if (input.length - pos > 1 && input[pos + 1] === '>') type = '>>', matchLength = 2;
253
+ if (input.length > 1 && input[1] === '>') type = '>>', matchLength = 2;
267
254
  break;
268
255
  case '_':
269
256
  // Try to find a blank node. Since it can contain (but not end with) a dot,
270
257
  // we always need a non-dot character before deciding it is a blank node.
271
258
  // Therefore, try inserting a space if we're at the end of the input.
272
- if ((match = execAt(this._blank, input, pos)) || inputFinished && (match = execAtEnd(this._blank, input, pos))) {
259
+ if ((match = this._blank.exec(input)) || inputFinished && (match = this._blank.exec(`${input} `))) {
273
260
  type = 'blank', prefix = '_', value = match[1];
274
261
  lexicalLength = match[1].length + 2;
275
262
  }
276
263
  break;
277
264
  case '"':
278
265
  // Try to find a literal without escape sequences
279
- if (match = execAt(this._simpleQuotedString, input, pos)) value = match[1];
266
+ if (match = this._simpleQuotedString.exec(input)) value = match[1];
280
267
  // Try to find a literal wrapped in three pairs of quotes
281
268
  else {
282
269
  ({
283
270
  value,
284
271
  matchLength,
285
272
  finalLineLength
286
- } = this._parseLiteral(input, pos));
287
- if (value === null) return reportSyntaxError(this, input, pos);
273
+ } = this._parseLiteral(input));
274
+ if (value === null) return reportSyntaxError(this);
288
275
  }
289
276
  if (match !== null || matchLength !== 0) {
290
277
  type = 'literal';
@@ -294,15 +281,15 @@ class N3Lexer {
294
281
  case "'":
295
282
  if (!this._lineMode) {
296
283
  // Try to find a literal without escape sequences
297
- if (match = execAt(this._simpleApostropheString, input, pos)) value = match[1];
284
+ if (match = this._simpleApostropheString.exec(input)) value = match[1];
298
285
  // Try to find a literal wrapped in three pairs of quotes
299
286
  else {
300
287
  ({
301
288
  value,
302
289
  matchLength,
303
290
  finalLineLength
304
- } = this._parseLiteral(input, pos));
305
- if (value === null) return reportSyntaxError(this, input, pos);
291
+ } = this._parseLiteral(input));
292
+ if (value === null) return reportSyntaxError(this);
306
293
  }
307
294
  if (match !== null || matchLength !== 0) {
308
295
  type = 'literal';
@@ -312,7 +299,7 @@ class N3Lexer {
312
299
  break;
313
300
  case '?':
314
301
  // Try to find a variable
315
- if (this._n3Mode && (match = execAt(this._variable, input, pos))) type = 'var', value = match[0];
302
+ if (this._n3Mode && (match = this._variable.exec(input))) type = 'var', value = match[0];
316
303
  break;
317
304
  case '@':
318
305
  // Try to find a language code. A language code can contain dash-separated
@@ -320,16 +307,15 @@ class N3Lexer {
320
307
  // input is not finished, another subtag may still arrive in a later chunk and
321
308
  // the match would be premature; wait for more input in that case.
322
309
  // A double dash starts a direction code, which cannot extend the language code.
323
- if (this._previousMarker === 'literal' && (match = execAt(this._langcode, input, pos)) && match[1] !== 'version') {
324
- const end = pos + match[0].length;
325
- if (!inputFinished && input[end] === '-' && input[end + 1] !== '-') match = null;else type = 'langcode', value = match[1];
310
+ if (this._previousMarker === 'literal' && (match = this._langcode.exec(input)) && match[1] !== 'version') {
311
+ if (!inputFinished && input[match[0].length] === '-' && input[match[0].length + 1] !== '-') match = null;else type = 'langcode', value = match[1];
326
312
  }
327
313
  // Try to find a keyword
328
- else if (match = execAt(this._atKeyword, input, pos)) type = match[0];
314
+ else if (match = this._atKeyword.exec(input)) type = match[0];
329
315
  break;
330
316
  case '.':
331
317
  // Try to find a dot as punctuation
332
- if (input.length - pos === 1 ? inputFinished : input[pos + 1] < '0' || input[pos + 1] > '9') {
318
+ if (input.length === 1 ? inputFinished : input[1] < '0' || input[1] > '9') {
333
319
  type = '.';
334
320
  matchLength = 1;
335
321
  break;
@@ -348,10 +334,10 @@ class N3Lexer {
348
334
  case '9':
349
335
  case '+':
350
336
  case '-':
351
- if (input[pos + 1] === '-') {
337
+ if (input[1] === '-') {
352
338
  // Try to find a direction code
353
339
  if (this._previousMarker === 'langcode') {
354
- if (input.startsWith('--ltr', pos)) type = 'dircode', value = 'ltr', matchLength = 5;else if (input.startsWith('--rtl', pos)) type = 'dircode', value = 'rtl', matchLength = 5;
340
+ if (input.startsWith('--ltr')) type = 'dircode', value = 'ltr', matchLength = 5;else if (input.startsWith('--rtl')) type = 'dircode', value = 'rtl', matchLength = 5;
355
341
  }
356
342
  break;
357
343
  }
@@ -359,7 +345,7 @@ class N3Lexer {
359
345
  // Try to find a number. Since it can contain (but not end with) a dot,
360
346
  // we always need a non-dot character before deciding it is a number.
361
347
  // Therefore, try inserting a space if we're at the end of the input.
362
- if (match = execAt(this._number, input, pos) || inputFinished && (match = execAtEnd(this._number, input, pos))) {
348
+ if (match = this._number.exec(input) || inputFinished && (match = this._number.exec(`${input} `))) {
363
349
  type = 'literal', value = match[0];
364
350
  prefix = typeof match[1] === 'string' ? xsd.double : typeof match[2] === 'string' ? xsd.decimal : xsd.integer;
365
351
  }
@@ -373,42 +359,42 @@ class N3Lexer {
373
359
  case 'V':
374
360
  case 'v':
375
361
  // Try to find a SPARQL-style keyword
376
- if (match = execAt(this._keyword, input, pos)) type = match[0].toUpperCase();else inconclusive = true;
362
+ if (match = this._keyword.exec(input)) type = match[0].toUpperCase();else inconclusive = true;
377
363
  break;
378
364
  case 'f':
379
365
  case 't':
380
366
  // Try to match a boolean
381
- if (testAt(this._boolean, input, pos)) type = 'literal', value = firstChar === 't' ? 'true' : 'false', prefix = xsd.boolean, matchLength = value.length;else inconclusive = true;
367
+ if (this._boolean.test(input)) type = 'literal', value = firstChar === 't' ? 'true' : 'false', prefix = xsd.boolean, matchLength = value.length;else inconclusive = true;
382
368
  break;
383
369
  case 'a':
384
370
  // Try to find an abbreviated predicate
385
- if (testAt(this._shortPredicates, input, pos)) type = 'abbreviation', value = 'a', matchLength = 1;else inconclusive = true;
371
+ if (this._shortPredicates.test(input)) type = 'abbreviation', value = 'a', matchLength = 1;else inconclusive = true;
386
372
  break;
387
373
  case 'h':
388
374
  case 'o':
389
375
  // Try to find an N3 verb keyword
390
- if (this._n3Mode && (match = this._matchN3Verb(input, pos, inputFinished))) type = match[0];else inconclusive = true;
376
+ if (this._n3Mode && (match = this._matchN3Verb(input, inputFinished))) type = match[0];else inconclusive = true;
391
377
  break;
392
378
  case 'i':
393
379
  // Try to find an IRI property list identifier or N3 verb keyword
394
- if (this._n3Mode && testAt(this._n3Id, input, pos)) type = 'id', matchLength = 2;else if (this._n3Mode && (match = this._matchN3Verb(input, pos, inputFinished))) type = match[0];else inconclusive = true;
380
+ if (this._n3Mode && this._n3Id.test(input)) type = 'id', matchLength = 2;else if (this._n3Mode && (match = this._matchN3Verb(input, inputFinished))) type = match[0];else inconclusive = true;
395
381
  break;
396
382
  case '=':
397
383
  // Try to find an implication arrow or equals sign
398
- if (this._n3Mode && input.length - pos > 1) {
384
+ if (this._n3Mode && input.length > 1) {
399
385
  type = 'abbreviation';
400
- if (input[pos + 1] !== '>') matchLength = 1, value = '=';else matchLength = 2, value = '>';
386
+ if (input[1] !== '>') matchLength = 1, value = '=';else matchLength = 2, value = '>';
401
387
  }
402
388
  break;
403
389
  case '!':
404
390
  if (!this._n3Mode) break;
405
391
  case ')':
406
- if (!inputFinished && (input.length - pos === 1 || input.length - pos === 2 && input[pos + 1] === '>')) {
392
+ if (!inputFinished && (input.length === 1 || input.length === 2 && input[1] === '>')) {
407
393
  // Don't consume yet, as it *could* become a triple term end.
408
394
  break;
409
395
  }
410
396
  // Try to find a triple term
411
- if (input.length - pos > 2 && input[pos + 1] === '>' && input[pos + 2] === '>') {
397
+ if (input.length > 2 && input[1] === '>' && input[2] === '>') {
412
398
  type = ')>>', matchLength = 3;
413
399
  break;
414
400
  }
@@ -426,15 +412,15 @@ class N3Lexer {
426
412
  break;
427
413
  case '{':
428
414
  // We need at least 2 tokens lookahead to distinguish "{|" and "{ "
429
- if (!this._lineMode && input.length - pos >= 2) {
415
+ if (!this._lineMode && input.length >= 2) {
430
416
  // Try to find a quoted triple annotation start
431
- if (input[pos + 1] === '|') type = '{|', matchLength = 2;else type = firstChar, matchLength = 1;
417
+ if (input[1] === '|') type = '{|', matchLength = 2;else type = firstChar, matchLength = 1;
432
418
  }
433
419
  break;
434
420
  case '|':
435
421
  // We need 2 tokens lookahead to parse "|}"
436
422
  // Try to find a quoted triple annotation end
437
- if (input.length - pos >= 2 && input[pos + 1] === '}') type = '|}', matchLength = 2;
423
+ if (input.length >= 2 && input[1] === '}') type = '|}', matchLength = 2;
438
424
  break;
439
425
  default:
440
426
  inconclusive = true;
@@ -443,11 +429,11 @@ class N3Lexer {
443
429
  // Some first characters do not allow an immediate decision, so inspect more
444
430
  if (inconclusive) {
445
431
  // Try to find a prefix
446
- if ((this._previousMarker === '@prefix' || this._previousMarker === 'PREFIX') && (match = execAt(this._prefix, input, pos))) type = 'prefix', value = match[1] || '';
432
+ if ((this._previousMarker === '@prefix' || this._previousMarker === 'PREFIX') && (match = this._prefix.exec(input))) type = 'prefix', value = match[1] || '';
447
433
  // Try to find a prefixed name. Since it can contain (but not end with) a dot,
448
434
  // we always need a non-dot character before deciding it is a prefixed name.
449
435
  // Therefore, try inserting a space if we're at the end of the input.
450
- else if ((match = execAt(this._prefixed, input, pos)) || inputFinished && (match = execAtEnd(this._prefixed, input, pos))) {
436
+ else if ((match = this._prefixed.exec(input)) || inputFinished && (match = this._prefixed.exec(`${input} `))) {
451
437
  type = 'prefixed', prefix = match[1] || '';
452
438
  value = this._unescape(match[2], localNameEscapeReplacements);
453
439
  lexicalLength = prefix.length + match[2].length + 1;
@@ -473,14 +459,16 @@ class N3Lexer {
473
459
  // We could be in streaming mode, and then we just wait for more input to arrive.
474
460
  // Otherwise, a syntax error has occurred in the input.
475
461
  // One exception: error on an unaccounted linebreak (= not inside a triple-quoted literal).
476
- if (inputFinished || !input.startsWith("'''", pos) && !input.startsWith('"""', pos) && /\n|\r/.test(input.slice(pos))) return reportSyntaxError(this, input, pos);else return this._suspend(input, pos, currentLineLength);
462
+ if (inputFinished || !/^'''|^"""/.test(input) && /\n|\r/.test(input)) return reportSyntaxError(this);else {
463
+ this._linePosition = currentLineLength - input.length;
464
+ return this._input = input;
465
+ }
477
466
  }
478
467
 
479
468
  // Emit the parsed token
480
469
  // Consumption includes separator whitespace; lexicalLength excludes it
481
- // and any synthetic EOF space. Consumption is clamped to the input below.
470
+ // and any synthetic EOF space. slice below clamps consumption to the input.
482
471
  const length = matchLength || match[0].length;
483
- const start = currentLineLength - (input.length - pos);
484
472
  let token;
485
473
  if (finalLineLength) {
486
474
  token = {
@@ -488,22 +476,23 @@ class N3Lexer {
488
476
  value,
489
477
  prefix,
490
478
  line,
491
- start,
479
+ start: currentLineLength - input.length,
492
480
  end: finalLineLength,
493
481
  endLine: this._line
494
482
  };
495
483
  callback(null, token);
496
- } else token = emitToken(type, value, prefix, line, start, lexicalLength || length);
484
+ } else token = emitToken(type, value, prefix, line, lexicalLength || length);
497
485
  this.previousToken = token;
498
486
  this._previousMarker = type;
499
487
 
500
488
  // Advance to next part to tokenize
501
- pos = Math.min(pos + length, input.length);
502
- if (finalLineLength) currentLineLength = input.length - pos + finalLineLength;
489
+ input = input.slice(length);
490
+ if (finalLineLength) currentLineLength = input.length + finalLineLength;
503
491
  }
504
492
 
505
493
  // Emits a comment at its exact position within matched whitespace.
506
- function emitComment(value, line, start) {
494
+ function emitComment(value, line, offset) {
495
+ const start = currentLineLength - input.length + offset;
507
496
  callback(null, {
508
497
  type: 'comment',
509
498
  value,
@@ -514,53 +503,49 @@ class N3Lexer {
514
503
  });
515
504
  }
516
505
  // Emits the token through the callback
517
- function emitToken(type, value, prefix, line, start, length) {
506
+ function emitToken(type, value, prefix, line, length) {
507
+ const start = input ? currentLineLength - input.length : currentLineLength;
508
+ const end = start + length;
518
509
  const token = {
519
510
  type,
520
511
  value,
521
512
  prefix,
522
513
  line,
523
514
  start,
524
- end: start + length
515
+ end
525
516
  };
526
517
  callback(null, token);
527
518
  return token;
528
519
  }
529
520
  // Signals the syntax error through the callback
530
- function reportSyntaxError(self, input, pos) {
531
- callback(self._syntaxError(execAt(nonWhitespace, input, pos)[0]));
521
+ function reportSyntaxError(self) {
522
+ callback(self._syntaxError(/^\S*/.exec(input)[0]));
532
523
  }
533
524
  }
534
525
 
535
- // ### `_suspend` keeps the unconsumed input until more input arrives
536
- _suspend(input, pos, currentLineLength) {
537
- this._linePosition = currentLineLength - (input.length - pos);
538
- return this._input = input.slice(pos);
539
- }
540
-
541
526
  // ### `_matchN3Verb` matches an N3 verb unless the input is a longer prefixed name
542
- _matchN3Verb(input, pos, inputFinished) {
543
- const verb = execAt(this._n3Verb, input, pos);
527
+ _matchN3Verb(input, inputFinished) {
528
+ const verb = this._n3Verb.exec(input);
544
529
  if (!verb) return null;
545
530
 
546
531
  // Most verb boundaries cannot be part of a prefix, so keep the common path fast.
547
- const next = input[pos + verb[0].length];
532
+ const next = input[verb[0].length];
548
533
  if (next !== '-' && next !== '_' && (next < '0' || next > '9')) return verb;
549
534
 
550
535
  // A prefix can start with a verb and continue with characters that are also
551
536
  // valid verb boundaries. Prefer the longer prefixed name when it is complete.
552
- if (execAt(this._prefixed, input, pos)) return null;
537
+ if (this._prefixed.exec(input)) return null;
553
538
  // Appending to the input only matters when a prefixed name could run up to
554
539
  // its end, which a character that cannot occur in prefixed names rules out.
555
540
  // This avoids copying the rest of the document for every such verb.
556
- if (testAt(nonPrefixedNameChar, input, pos)) return verb;
557
- if (execAtEnd(this._prefixed, input, pos)) return null;
541
+ if (nonPrefixedNameChar.test(input)) return verb;
542
+ if (this._prefixed.exec(`${input} `)) return null;
558
543
 
559
544
  // If a stream chunk ends partway through such a prefix, wait for the colon
560
545
  // instead of prematurely emitting the verb. Appending ": " lets the prefix
561
546
  // grammar determine whether all input seen so far can be a complete prefix.
562
547
  if (!inputFinished) {
563
- const prefix = execAt(this._prefix, `${input.slice(pos)}: `, 0);
548
+ const prefix = this._prefix.exec(`${input}: `);
564
549
  if (prefix) return null;
565
550
  }
566
551
  return verb;
@@ -607,19 +592,19 @@ class N3Lexer {
607
592
  return result + item.slice(start);
608
593
  }
609
594
 
610
- // ### `_parseLiteral` parses a literal at the given position into an unescaped value
611
- _parseLiteral(input, pos) {
595
+ // ### `_parseLiteral` parses a literal into an unescaped value
596
+ _parseLiteral(input) {
612
597
  // Ensure we have enough lookahead to identify triple-quoted strings
613
- if (input.length - pos >= 3) {
598
+ if (input.length >= 3) {
614
599
  // The caller has already identified a single or double quote.
615
- const quote = input[pos];
616
- const openingLength = input[pos + 1] === quote && input[pos + 2] === quote ? 3 : 1;
600
+ const quote = input[0];
601
+ const openingLength = input[1] === quote && input[2] === quote ? 3 : 1;
617
602
  let opening = quote;
618
603
  if (openingLength === 3) opening = quote === '"' ? '"""' : "'''";
619
604
 
620
605
  // Find the next candidate closing quotes
621
- let closingPos = pos + Math.max(this._literalClosingPos, openingLength);
622
- while ((closingPos = input.indexOf(opening, closingPos)) > pos) {
606
+ let closingPos = Math.max(this._literalClosingPos, openingLength);
607
+ while ((closingPos = input.indexOf(opening, closingPos)) > 0) {
623
608
  // Count backslashes right before the closing quotes
624
609
  let backslashCount = 0;
625
610
  while (input[closingPos - backslashCount - 1] === '\\') backslashCount++;
@@ -628,10 +613,10 @@ class N3Lexer {
628
613
  // means these are actual, non-escaped closing quotes
629
614
  if (backslashCount % 2 === 0) {
630
615
  // Extract and unescape the value
631
- const raw = input.substring(pos + openingLength, closingPos),
616
+ const raw = input.substring(openingLength, closingPos),
632
617
  lines = raw.split(/\r\n|\r|\n/),
633
618
  lineCount = lines.length - 1;
634
- const matchLength = closingPos - pos + openingLength;
619
+ const matchLength = closingPos + openingLength;
635
620
  // Only triple-quoted strings can be multi-line
636
621
  if (openingLength === 1 && lineCount !== 0 || openingLength === 3 && this._lineMode) break;
637
622
  this._line += lineCount;
@@ -644,7 +629,7 @@ class N3Lexer {
644
629
  }
645
630
  closingPos++;
646
631
  }
647
- this._literalClosingPos = input.length - pos - openingLength + 1;
632
+ this._literalClosingPos = input.length - openingLength + 1;
648
633
  }
649
634
  return {
650
635
  value: '',
package/lib/N3Parser.js CHANGED
@@ -44,7 +44,8 @@ class N3Parser {
44
44
  // Whether the log:isImpliedBy predicate is supported
45
45
  this._isImpliedBy = options.isImpliedBy;
46
46
  // Whether an undeclared empty prefix resolves against the document IRI
47
- this._implicitEmptyPrefix = !!options.implicitEmptyPrefix;
47
+ // (enabled unless explicitly disabled)
48
+ this._implicitEmptyPrefix = options.implicitEmptyPrefix !== false;
48
49
  // Whether an empty formula is read as the boolean literal true,
49
50
  // as in the N3 spec tests (opt-in until the next major version)
50
51
  this._emptyFormulaAsTrue = !!options.emptyFormulaAsTrue;