n3 2.11.2 → 3.0.0-alpha.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/N3Lexer.js CHANGED
@@ -18,8 +18,7 @@ const localNameEscapeReplacements = {
18
18
  };
19
19
  const illegalIriChars = /[\x00-\x20<>\\"\{\}\|\^\`]/;
20
20
  // Characters that cannot occur in a prefixed name, not even escaped
21
- // (global, so that testAt searches the rest of the input from a position)
22
- const nonPrefixedNameChar = /[\s<>"{}|^`]/g;
21
+ const nonPrefixedNameChar = /[\s<>"{}|^`]/;
23
22
 
24
23
  // A valid code point is a Unicode scalar value: at most U+10FFFF and not a surrogate
25
24
  function isValidCodePoint(charCode) {
@@ -36,48 +35,30 @@ const lineModeRegExps = {
36
35
  _whitespace: true,
37
36
  };
38
37
  const invalidRegExp = /$0^/;
39
- const nonWhitespace = /\S*/y;
40
-
41
- // Matches a sticky regular expression at the given position of the input
42
- function execAt(regExp, input, pos) {
43
- regExp.lastIndex = pos;
44
- return regExp.exec(input);
45
- }
46
- function testAt(regExp, input, pos) {
47
- regExp.lastIndex = pos;
48
- return regExp.test(input);
49
- }
50
- // Matches the rest of the input followed by a space, as at the end of the input,
51
- // a token that can contain (but not end with) a dot needs a non-dot character after it
52
- function execAtEnd(regExp, input, pos) {
53
- regExp.lastIndex = 0;
54
- return regExp.exec(`${input.slice(pos)} `);
55
- }
56
38
 
57
39
  // ## Constructor
58
40
  export default class N3Lexer {
59
41
  constructor(options) {
60
42
  // ## Regular expressions
61
- // It's slightly faster to have these as properties than as in-scope variables.
62
- // They are sticky, so they only match at the `lastIndex` set by `execAt`.
63
- this._iri = /<((?:[^ <>{}\\]|\\[uU])+)>[ \t]*/y; // IRI with escape sequences; needs sanity check after unescaping
64
- this._unescapedIri = /<([^\x00-\x20<>\\"\{\}\|\^\`]*)>[ \t]*/y; // IRI without escape sequences; no unescaping
65
- this._simpleQuotedString = /"([^"\\\r\n]*)"(?=[^"])/y; // string without escape sequences
66
- this._simpleApostropheString = /'([^'\\\r\n]*)'(?=[^'])/y;
67
- this._langcode = /@([a-z]+(?:-[a-z0-9]+)*)(?=[^a-z0-9])/iy;
68
- this._prefix = /((?:[A-Za-z\xc0-\xd6\xd8-\xf6\xf8-\u02ff\u0370-\u037d\u037f-\u1fff\u200c\u200d\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff])(?:\.?[\-0-9A-Z_a-z\xb7\xc0-\xd6\xd8-\xf6\xf8-\u037d\u037f-\u1fff\u200c\u200d\u203f\u2040\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff])*)?:(?=[#\s<])/y;
69
- this._prefixed = /((?:[A-Za-z\xc0-\xd6\xd8-\xf6\xf8-\u02ff\u0370-\u037d\u037f-\u1fff\u200c\u200d\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff])(?:\.?[\-0-9A-Z_a-z\xb7\xc0-\xd6\xd8-\xf6\xf8-\u037d\u037f-\u1fff\u200c\u200d\u203f\u2040\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff])*)?:((?:(?:[0-9:A-Z_a-z\xc0-\xd6\xd8-\xf6\xf8-\u02ff\u0370-\u037d\u037f-\u1fff\u200c\u200d\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff]|%[0-9a-fA-F]{2}|\\[!#-\/;=?\-@_~])(?:(?:[\.\-0-9:A-Z_a-z\xb7\xc0-\xd6\xd8-\xf6\xf8-\u037d\u037f-\u1fff\u200c\u200d\u203f\u2040\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff]|%[0-9a-fA-F]{2}|\\[!#-\/;=?\-@_~])*(?:[\-0-9:A-Z_a-z\xb7\xc0-\xd6\xd8-\xf6\xf8-\u037d\u037f-\u1fff\u200c\u200d\u203f\u2040\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff]|%[0-9a-fA-F]{2}|\\[!#-\/;=?\-@_~]))?)?)(?:[ \t]+|(?=\.?[,;!\^\s#()\[\]\{\}"'<>]))/y;
70
- this._variable = /\?(?:(?:[A-Z_a-z\xc0-\xd6\xd8-\xf6\xf8-\u02ff\u0370-\u037d\u037f-\u1fff\u200c\u200d\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff])(?:[\-0-9:A-Z_a-z\xb7\xc0-\xd6\xd8-\xf6\xf8-\u037d\u037f-\u1fff\u200c\u200d\u203f\u2040\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff])*)(?=[.,;!\^\s#()\[\]\{\}"'<>])/y;
71
- this._blank = /_:((?:[0-9A-Z_a-z\xc0-\xd6\xd8-\xf6\xf8-\u02ff\u0370-\u037d\u037f-\u1fff\u200c\u200d\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff])(?:\.?[\-0-9A-Z_a-z\xb7\xc0-\xd6\xd8-\xf6\xf8-\u037d\u037f-\u1fff\u200c\u200d\u203f\u2040\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff])*)(?:[ \t]+|(?=\.?[,;:!\^\s#()\[\]\{\}"'<>]))/y;
72
- this._number = /[\-+]?(?:(\d+\.\d*|\.?\d+)[eE][\-+]?\d+|(?=\.?\d)\d*(?:(\.)\d+)?)(?=\.?[,;:!\^\s#()\[\]\{\}"'<>])/y;
73
- this._boolean = /(?:true|false)(?=[.,;!\^\s#()\[\]\{\}"'<>])/y;
74
- this._atKeyword = /@[a-z]+(?=[\s#<:])/iy;
75
- this._keyword = /(?:PREFIX|BASE|VERSION|GRAPH)(?=[\s#<])/iy;
76
- this._n3Verb = /(?:has|is|of)(?=[\s#()\[\]\{\}"'<>?_+\-0-9])/y;
77
- this._n3Id = /id(?=[\s#<])/y;
78
- this._shortPredicates = /a(?=[\s#()\[\]\{\}"'<>])/y;
79
- this._commentLine = /[ \t]*#([^\n\r]*)(?:\r\n|\n|\r)([ \t]*)/y;
80
- this._whitespace = /[ \t]+/y;
43
+ // It's slightly faster to have these as properties than as in-scope variables
44
+ this._iri = /^<((?:[^ <>{}\\]|\\[uU])+)>[ \t]*/; // IRI with escape sequences; needs sanity check after unescaping
45
+ this._unescapedIri = /^<([^\x00-\x20<>\\"\{\}\|\^\`]*)>[ \t]*/; // IRI without escape sequences; no unescaping
46
+ this._simpleQuotedString = /^"([^"\\\r\n]*)"(?=[^"])/; // string without escape sequences
47
+ this._simpleApostropheString = /^'([^'\\\r\n]*)'(?=[^'])/;
48
+ this._langcode = /^@([a-z]+(?:-[a-z0-9]+)*)(?=[^a-z0-9])/i;
49
+ this._prefix = /^((?:[A-Za-z\xc0-\xd6\xd8-\xf6\xf8-\u02ff\u0370-\u037d\u037f-\u1fff\u200c\u200d\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff])(?:\.?[\-0-9A-Z_a-z\xb7\xc0-\xd6\xd8-\xf6\xf8-\u037d\u037f-\u1fff\u200c\u200d\u203f\u2040\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff])*)?:(?=[#\s<])/;
50
+ this._prefixed = /^((?:[A-Za-z\xc0-\xd6\xd8-\xf6\xf8-\u02ff\u0370-\u037d\u037f-\u1fff\u200c\u200d\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff])(?:\.?[\-0-9A-Z_a-z\xb7\xc0-\xd6\xd8-\xf6\xf8-\u037d\u037f-\u1fff\u200c\u200d\u203f\u2040\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff])*)?:((?:(?:[0-9:A-Z_a-z\xc0-\xd6\xd8-\xf6\xf8-\u02ff\u0370-\u037d\u037f-\u1fff\u200c\u200d\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff]|%[0-9a-fA-F]{2}|\\[!#-\/;=?\-@_~])(?:(?:[\.\-0-9:A-Z_a-z\xb7\xc0-\xd6\xd8-\xf6\xf8-\u037d\u037f-\u1fff\u200c\u200d\u203f\u2040\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff]|%[0-9a-fA-F]{2}|\\[!#-\/;=?\-@_~])*(?:[\-0-9:A-Z_a-z\xb7\xc0-\xd6\xd8-\xf6\xf8-\u037d\u037f-\u1fff\u200c\u200d\u203f\u2040\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff]|%[0-9a-fA-F]{2}|\\[!#-\/;=?\-@_~]))?)?)(?:[ \t]+|(?=\.?[,;!\^\s#()\[\]\{\}"'<>]))/;
51
+ this._variable = /^\?(?:(?:[A-Z_a-z\xc0-\xd6\xd8-\xf6\xf8-\u02ff\u0370-\u037d\u037f-\u1fff\u200c\u200d\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff])(?:[\-0-9:A-Z_a-z\xb7\xc0-\xd6\xd8-\xf6\xf8-\u037d\u037f-\u1fff\u200c\u200d\u203f\u2040\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff])*)(?=[.,;!\^\s#()\[\]\{\}"'<>])/;
52
+ this._blank = /^_:((?:[0-9A-Z_a-z\xc0-\xd6\xd8-\xf6\xf8-\u02ff\u0370-\u037d\u037f-\u1fff\u200c\u200d\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff])(?:\.?[\-0-9A-Z_a-z\xb7\xc0-\xd6\xd8-\xf6\xf8-\u037d\u037f-\u1fff\u200c\u200d\u203f\u2040\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff])*)(?:[ \t]+|(?=\.?[,;:!\^\s#()\[\]\{\}"'<>]))/;
53
+ this._number = /^[\-+]?(?:(\d+\.\d*|\.?\d+)[eE][\-+]?\d+|(?=\.?\d)\d*(?:(\.)\d+)?)(?=\.?[,;:!\^\s#()\[\]\{\}"'<>])/;
54
+ this._boolean = /^(?:true|false)(?=[.,;!\^\s#()\[\]\{\}"'<>])/;
55
+ this._atKeyword = /^@[a-z]+(?=[\s#<:])/i;
56
+ this._keyword = /^(?:PREFIX|BASE|VERSION|GRAPH)(?=[\s#<])/i;
57
+ this._n3Verb = /^(?:has|is|of)(?=[\s#()\[\]\{\}"'<>?_+\-0-9])/;
58
+ this._n3Id = /^id(?=[\s#<])/;
59
+ this._shortPredicates = /^a(?=[\s#()\[\]\{\}"'<>])/;
60
+ this._commentLine = /^[ \t]*#([^\n\r]*)(?:\r\n|\n|\r)([ \t]*)/;
61
+ this._whitespace = /^[ \t]+/;
81
62
  options = options || {};
82
63
 
83
64
  // Whether the log:isImpliedBy predicate is supported
@@ -106,94 +87,98 @@ export default class N3Lexer {
106
87
 
107
88
  // ### `_tokenizeToEnd` tokenizes as for as possible, emitting tokens through the callback
108
89
  _tokenizeToEnd(callback, inputFinished) {
109
- // Continue parsing as far as possible; the loop will return eventually.
110
- // Rather than slicing off every token, track the position of the remaining input;
111
- // the regular expressions are sticky, so they match at that position.
112
- const input = this._input;
113
- let pos = 0;
90
+ // Continue parsing as far as possible; the loop will return eventually
91
+ let input = this._input;
114
92
  let currentLineLength = this._linePosition + input.length;
115
93
  while (true) {
116
94
  // Consume one separator line at a time, including its following indentation.
117
95
  while (true) {
118
- let charCode = input.charCodeAt(pos), separatorLength = 0;
96
+ let charCode = input.charCodeAt(0), separatorLength = 0;
119
97
  if (charCode === SPACE || charCode === TAB) {
120
- const next = input.charCodeAt(pos + 1);
98
+ const next = input.charCodeAt(1);
121
99
  separatorLength = next === SPACE || next === TAB ?
122
- execAt(this._whitespace, input, pos)[0].length : 1;
123
- charCode = input.charCodeAt(pos + separatorLength);
100
+ this._whitespace.exec(input)[0].length : 1;
101
+ charCode = input.charCodeAt(separatorLength);
124
102
  }
125
103
  if (charCode === HASH) {
126
- const comment = execAt(this._commentLine, input, pos);
104
+ const comment = this._commentLine.exec(input);
127
105
  if (comment) {
128
106
  const commentLength = comment[0].length;
129
107
  // Keep a trailing CR buffered in case the next chunk starts with LF.
130
- if (!inputFinished && pos + commentLength === input.length &&
131
- input.charCodeAt(input.length - 1) === CR)
132
- return this._suspend(input, pos, currentLineLength);
108
+ if (!inputFinished && commentLength === input.length &&
109
+ input.charCodeAt(commentLength - 1) === CR) {
110
+ this._linePosition = currentLineLength - input.length;
111
+ return this._input = input;
112
+ }
133
113
  if (this.comments)
134
- emitComment(comment[1], this._line, currentLineLength - (input.length - pos) + separatorLength);
135
- pos += commentLength;
136
- currentLineLength = input.length - pos + comment[2].length;
114
+ emitComment(comment[1], this._line, separatorLength);
115
+ input = input.slice(commentLength);
116
+ currentLineLength = input.length + comment[2].length;
137
117
  this._line++;
138
118
  }
139
119
  else {
140
120
  // A comment without a line ending stays buffered until EOF.
141
- pos += separatorLength;
142
- if (!inputFinished)
143
- return this._suspend(input, pos, currentLineLength);
121
+ input = input.slice(separatorLength);
122
+ if (!inputFinished) {
123
+ this._linePosition = currentLineLength - input.length;
124
+ return this._input = input;
125
+ }
144
126
  if (this.comments)
145
- emitComment(input.slice(pos + 1), this._line, currentLineLength - (input.length - pos));
146
- pos = input.length;
127
+ emitComment(input.slice(1), this._line, 0);
128
+ input = '';
147
129
  break;
148
130
  }
149
131
  }
150
132
  else if (charCode === LF || charCode === CR) {
151
133
  // A CR at the end of a chunk may still be followed by LF.
152
- if (!inputFinished && charCode === CR && pos + separatorLength + 1 === input.length)
153
- return this._suspend(input, pos, currentLineLength);
154
- separatorLength += charCode === CR && input.charCodeAt(pos + separatorLength + 1) === LF ? 2 : 1;
134
+ if (!inputFinished && charCode === CR && separatorLength + 1 === input.length) {
135
+ this._linePosition = currentLineLength - input.length;
136
+ return this._input = input;
137
+ }
138
+ separatorLength += charCode === CR && input.charCodeAt(separatorLength + 1) === LF ? 2 : 1;
155
139
  // Indentation is consumed with the newline, but belongs to the next line's columns.
156
140
  let indentationLength = 0;
157
- const next = input.charCodeAt(pos + separatorLength);
141
+ const next = input.charCodeAt(separatorLength);
158
142
  if (next === SPACE || next === TAB) {
159
- const following = input.charCodeAt(pos + separatorLength + 1);
143
+ const following = input.charCodeAt(separatorLength + 1);
160
144
  indentationLength = following === SPACE || following === TAB ?
161
- execAt(this._whitespace, input, pos + separatorLength)[0].length : 1;
145
+ this._whitespace.exec(input.slice(separatorLength))[0].length : 1;
162
146
  }
163
- pos += separatorLength + indentationLength;
164
- currentLineLength = input.length - pos + indentationLength;
147
+ input = input.slice(separatorLength + indentationLength);
148
+ currentLineLength = input.length + indentationLength;
165
149
  this._line++;
166
150
  }
167
151
  else {
168
- pos += separatorLength;
152
+ if (separatorLength !== 0)
153
+ input = input.slice(separatorLength);
169
154
  break;
170
155
  }
171
156
  }
172
- if (pos >= input.length) {
173
- this._linePosition = currentLineLength;
157
+ if (input.length === 0) {
174
158
  if (inputFinished) {
175
- emitToken('eof', '', '', this._line, currentLineLength, 0);
176
- return this._input = null;
159
+ input = null;
160
+ emitToken('eof', '', '', this._line, 0);
177
161
  }
178
- return this._input = '';
162
+ this._linePosition = currentLineLength;
163
+ return this._input = input;
179
164
  }
180
165
 
181
166
  // Look for specific token types based on the first character
182
- const line = this._line, firstChar = input[pos];
167
+ const line = this._line, firstChar = input[0];
183
168
  let type = '', value = '', prefix = '',
184
169
  match = null, matchLength = 0, lexicalLength = 0,
185
170
  finalLineLength = 0, inconclusive = false;
186
171
  switch (firstChar) {
187
172
  case '^':
188
173
  // We need at least 3 tokens lookahead to distinguish ^^<IRI> and ^^pre:fixed
189
- if (input.length - pos < 3)
174
+ if (input.length < 3)
190
175
  break;
191
176
  // Try to match a type
192
- else if (input[pos + 1] === '^') {
177
+ else if (input[1] === '^') {
193
178
  this._previousMarker = '^^';
194
179
  // Move to type IRI or prefixed name
195
- pos += 2;
196
- if (input[pos] !== '<') {
180
+ input = input.slice(2);
181
+ if (input[0] !== '<') {
197
182
  inconclusive = true;
198
183
  break;
199
184
  }
@@ -209,38 +194,38 @@ export default class N3Lexer {
209
194
  // Fall through in case the type is an IRI
210
195
  case '<':
211
196
  // Try to find a full IRI without escape sequences
212
- if (match = execAt(this._unescapedIri, input, pos)) {
197
+ if (match = this._unescapedIri.exec(input)) {
213
198
  type = 'IRI', value = match[1];
214
199
  lexicalLength = match[1].length + 2;
215
200
  }
216
201
  // Try to find a full IRI with escape sequences
217
- else if (match = execAt(this._iri, input, pos)) {
202
+ else if (match = this._iri.exec(input)) {
218
203
  value = this._unescape(match[1], stringEscapeReplacements);
219
204
  if (value === null || illegalIriChars.test(value))
220
- return reportSyntaxError(this, input, pos);
205
+ return reportSyntaxError(this);
221
206
  type = 'IRI';
222
207
  lexicalLength = match[1].length + 2;
223
208
  }
224
209
  // Try to find a triple term
225
- else if (input.length - pos > 2 && input[pos + 1] === '<' && input[pos + 2] === '(')
210
+ else if (input.length > 2 && input[1] === '<' && input[2] === '(')
226
211
  type = '<<(', matchLength = 3;
227
212
  // Try to find a reified triple
228
- else if (!this._lineMode && input.length - pos > (inputFinished ? 1 : 2) && input[pos + 1] === '<')
213
+ else if (!this._lineMode && input.length > (inputFinished ? 1 : 2) && input[1] === '<')
229
214
  type = '<<', matchLength = 2;
230
215
  // Try to find a backwards implication arrow
231
- else if (this._n3Mode && input.length - pos > 1 && input[pos + 1] === '=') {
216
+ else if (this._n3Mode && input.length > 1 && input[1] === '=') {
232
217
  matchLength = 2;
233
218
  if (this._isImpliedBy) type = 'abbreviation', value = '<';
234
219
  else type = 'inverse', value = '>';
235
220
  }
236
221
  // Try to find an inverted predicate marker
237
- else if (this._n3Mode && input.length - pos > 1 && input[pos + 1] === '-')
222
+ else if (this._n3Mode && input.length > 1 && input[1] === '-')
238
223
  type = 'inversePredicate', matchLength = 2;
239
224
  break;
240
225
 
241
226
  case '>':
242
227
  // Try to find a reified triple
243
- if (input.length - pos > 1 && input[pos + 1] === '>')
228
+ if (input.length > 1 && input[1] === '>')
244
229
  type = '>>', matchLength = 2;
245
230
  break;
246
231
 
@@ -248,8 +233,8 @@ export default class N3Lexer {
248
233
  // Try to find a blank node. Since it can contain (but not end with) a dot,
249
234
  // we always need a non-dot character before deciding it is a blank node.
250
235
  // Therefore, try inserting a space if we're at the end of the input.
251
- if ((match = execAt(this._blank, input, pos)) ||
252
- inputFinished && (match = execAtEnd(this._blank, input, pos))) {
236
+ if ((match = this._blank.exec(input)) ||
237
+ inputFinished && (match = this._blank.exec(`${input} `))) {
253
238
  type = 'blank', prefix = '_', value = match[1];
254
239
  lexicalLength = match[1].length + 2;
255
240
  }
@@ -257,13 +242,13 @@ export default class N3Lexer {
257
242
 
258
243
  case '"':
259
244
  // Try to find a literal without escape sequences
260
- if (match = execAt(this._simpleQuotedString, input, pos))
245
+ if (match = this._simpleQuotedString.exec(input))
261
246
  value = match[1];
262
247
  // Try to find a literal wrapped in three pairs of quotes
263
248
  else {
264
- ({ value, matchLength, finalLineLength } = this._parseLiteral(input, pos));
249
+ ({ value, matchLength, finalLineLength } = this._parseLiteral(input));
265
250
  if (value === null)
266
- return reportSyntaxError(this, input, pos);
251
+ return reportSyntaxError(this);
267
252
  }
268
253
  if (match !== null || matchLength !== 0) {
269
254
  type = 'literal';
@@ -274,13 +259,13 @@ export default class N3Lexer {
274
259
  case "'":
275
260
  if (!this._lineMode) {
276
261
  // Try to find a literal without escape sequences
277
- if (match = execAt(this._simpleApostropheString, input, pos))
262
+ if (match = this._simpleApostropheString.exec(input))
278
263
  value = match[1];
279
264
  // Try to find a literal wrapped in three pairs of quotes
280
265
  else {
281
- ({ value, matchLength, finalLineLength } = this._parseLiteral(input, pos));
266
+ ({ value, matchLength, finalLineLength } = this._parseLiteral(input));
282
267
  if (value === null)
283
- return reportSyntaxError(this, input, pos);
268
+ return reportSyntaxError(this);
284
269
  }
285
270
  if (match !== null || matchLength !== 0) {
286
271
  type = 'literal';
@@ -291,7 +276,7 @@ export default class N3Lexer {
291
276
 
292
277
  case '?':
293
278
  // Try to find a variable
294
- if (this._n3Mode && (match = execAt(this._variable, input, pos)))
279
+ if (this._n3Mode && (match = this._variable.exec(input)))
295
280
  type = 'var', value = match[0];
296
281
  break;
297
282
 
@@ -301,21 +286,20 @@ export default class N3Lexer {
301
286
  // input is not finished, another subtag may still arrive in a later chunk and
302
287
  // the match would be premature; wait for more input in that case.
303
288
  // A double dash starts a direction code, which cannot extend the language code.
304
- if (this._previousMarker === 'literal' && (match = execAt(this._langcode, input, pos)) && match[1] !== 'version') {
305
- const end = pos + match[0].length;
306
- if (!inputFinished && input[end] === '-' && input[end + 1] !== '-')
289
+ if (this._previousMarker === 'literal' && (match = this._langcode.exec(input)) && match[1] !== 'version') {
290
+ if (!inputFinished && input[match[0].length] === '-' && input[match[0].length + 1] !== '-')
307
291
  match = null;
308
292
  else
309
293
  type = 'langcode', value = match[1];
310
294
  }
311
295
  // Try to find a keyword
312
- else if (match = execAt(this._atKeyword, input, pos))
296
+ else if (match = this._atKeyword.exec(input))
313
297
  type = match[0];
314
298
  break;
315
299
 
316
300
  case '.':
317
301
  // Try to find a dot as punctuation
318
- if (input.length - pos === 1 ? inputFinished : (input[pos + 1] < '0' || input[pos + 1] > '9')) {
302
+ if (input.length === 1 ? inputFinished : (input[1] < '0' || input[1] > '9')) {
319
303
  type = '.';
320
304
  matchLength = 1;
321
305
  break;
@@ -334,12 +318,12 @@ export default class N3Lexer {
334
318
  case '9':
335
319
  case '+':
336
320
  case '-':
337
- if (input[pos + 1] === '-') {
321
+ if (input[1] === '-') {
338
322
  // Try to find a direction code
339
323
  if (this._previousMarker === 'langcode') {
340
- if (input.startsWith('--ltr', pos))
324
+ if (input.startsWith('--ltr'))
341
325
  type = 'dircode', value = 'ltr', matchLength = 5;
342
- else if (input.startsWith('--rtl', pos))
326
+ else if (input.startsWith('--rtl'))
343
327
  type = 'dircode', value = 'rtl', matchLength = 5;
344
328
  }
345
329
  break;
@@ -348,8 +332,8 @@ export default class N3Lexer {
348
332
  // Try to find a number. Since it can contain (but not end with) a dot,
349
333
  // we always need a non-dot character before deciding it is a number.
350
334
  // Therefore, try inserting a space if we're at the end of the input.
351
- if (match = execAt(this._number, input, pos) ||
352
- inputFinished && (match = execAtEnd(this._number, input, pos))) {
335
+ if (match = this._number.exec(input) ||
336
+ inputFinished && (match = this._number.exec(`${input} `))) {
353
337
  type = 'literal', value = match[0];
354
338
  prefix = (typeof match[1] === 'string' ? xsd.double :
355
339
  (typeof match[2] === 'string' ? xsd.decimal : xsd.integer));
@@ -365,7 +349,7 @@ export default class N3Lexer {
365
349
  case 'V':
366
350
  case 'v':
367
351
  // Try to find a SPARQL-style keyword
368
- if (match = execAt(this._keyword, input, pos))
352
+ if (match = this._keyword.exec(input))
369
353
  type = match[0].toUpperCase();
370
354
  else
371
355
  inconclusive = true;
@@ -374,7 +358,7 @@ export default class N3Lexer {
374
358
  case 'f':
375
359
  case 't':
376
360
  // Try to match a boolean
377
- if (testAt(this._boolean, input, pos))
361
+ if (this._boolean.test(input))
378
362
  type = 'literal', value = firstChar === 't' ? 'true' : 'false', prefix = xsd.boolean, matchLength = value.length;
379
363
  else
380
364
  inconclusive = true;
@@ -382,7 +366,7 @@ export default class N3Lexer {
382
366
 
383
367
  case 'a':
384
368
  // Try to find an abbreviated predicate
385
- if (testAt(this._shortPredicates, input, pos))
369
+ if (this._shortPredicates.test(input))
386
370
  type = 'abbreviation', value = 'a', matchLength = 1;
387
371
  else
388
372
  inconclusive = true;
@@ -391,7 +375,7 @@ export default class N3Lexer {
391
375
  case 'h':
392
376
  case 'o':
393
377
  // Try to find an N3 verb keyword
394
- if (this._n3Mode && (match = this._matchN3Verb(input, pos, inputFinished)))
378
+ if (this._n3Mode && (match = this._matchN3Verb(input, inputFinished)))
395
379
  type = match[0];
396
380
  else
397
381
  inconclusive = true;
@@ -399,9 +383,9 @@ export default class N3Lexer {
399
383
 
400
384
  case 'i':
401
385
  // Try to find an IRI property list identifier or N3 verb keyword
402
- if (this._n3Mode && testAt(this._n3Id, input, pos))
386
+ if (this._n3Mode && this._n3Id.test(input))
403
387
  type = 'id', matchLength = 2;
404
- else if (this._n3Mode && (match = this._matchN3Verb(input, pos, inputFinished)))
388
+ else if (this._n3Mode && (match = this._matchN3Verb(input, inputFinished)))
405
389
  type = match[0];
406
390
  else
407
391
  inconclusive = true;
@@ -409,9 +393,9 @@ export default class N3Lexer {
409
393
 
410
394
  case '=':
411
395
  // Try to find an implication arrow or equals sign
412
- if (this._n3Mode && input.length - pos > 1) {
396
+ if (this._n3Mode && input.length > 1) {
413
397
  type = 'abbreviation';
414
- if (input[pos + 1] !== '>')
398
+ if (input[1] !== '>')
415
399
  matchLength = 1, value = '=';
416
400
  else
417
401
  matchLength = 2, value = '>';
@@ -422,12 +406,12 @@ export default class N3Lexer {
422
406
  if (!this._n3Mode)
423
407
  break;
424
408
  case ')':
425
- if (!inputFinished && (input.length - pos === 1 || (input.length - pos === 2 && input[pos + 1] === '>'))) {
409
+ if (!inputFinished && (input.length === 1 || (input.length === 2 && input[1] === '>'))) {
426
410
  // Don't consume yet, as it *could* become a triple term end.
427
411
  break;
428
412
  }
429
413
  // Try to find a triple term
430
- if (input.length - pos > 2 && input[pos + 1] === '>' && input[pos + 2] === '>') {
414
+ if (input.length > 2 && input[1] === '>' && input[2] === '>') {
431
415
  type = ')>>', matchLength = 3;
432
416
  break;
433
417
  }
@@ -445,9 +429,9 @@ export default class N3Lexer {
445
429
  break;
446
430
  case '{':
447
431
  // We need at least 2 tokens lookahead to distinguish "{|" and "{ "
448
- if (!this._lineMode && input.length - pos >= 2) {
432
+ if (!this._lineMode && input.length >= 2) {
449
433
  // Try to find a quoted triple annotation start
450
- if (input[pos + 1] === '|')
434
+ if (input[1] === '|')
451
435
  type = '{|', matchLength = 2;
452
436
  else
453
437
  type = firstChar, matchLength = 1;
@@ -456,7 +440,7 @@ export default class N3Lexer {
456
440
  case '|':
457
441
  // We need 2 tokens lookahead to parse "|}"
458
442
  // Try to find a quoted triple annotation end
459
- if (input.length - pos >= 2 && input[pos + 1] === '}')
443
+ if (input.length >= 2 && input[1] === '}')
460
444
  type = '|}', matchLength = 2;
461
445
  break;
462
446
 
@@ -468,13 +452,13 @@ export default class N3Lexer {
468
452
  if (inconclusive) {
469
453
  // Try to find a prefix
470
454
  if ((this._previousMarker === '@prefix' || this._previousMarker === 'PREFIX') &&
471
- (match = execAt(this._prefix, input, pos)))
455
+ (match = this._prefix.exec(input)))
472
456
  type = 'prefix', value = match[1] || '';
473
457
  // Try to find a prefixed name. Since it can contain (but not end with) a dot,
474
458
  // we always need a non-dot character before deciding it is a prefixed name.
475
459
  // Therefore, try inserting a space if we're at the end of the input.
476
- else if ((match = execAt(this._prefixed, input, pos)) ||
477
- inputFinished && (match = execAtEnd(this._prefixed, input, pos))) {
460
+ else if ((match = this._prefixed.exec(input)) ||
461
+ inputFinished && (match = this._prefixed.exec(`${input} `))) {
478
462
  type = 'prefixed', prefix = match[1] || '';
479
463
  value = this._unescape(match[2], localNameEscapeReplacements);
480
464
  lexicalLength = prefix.length + match[2].length + 1;
@@ -495,90 +479,86 @@ export default class N3Lexer {
495
479
  // We could be in streaming mode, and then we just wait for more input to arrive.
496
480
  // Otherwise, a syntax error has occurred in the input.
497
481
  // One exception: error on an unaccounted linebreak (= not inside a triple-quoted literal).
498
- if (inputFinished || (!input.startsWith("'''", pos) && !input.startsWith('"""', pos) &&
499
- /\n|\r/.test(input.slice(pos))))
500
- return reportSyntaxError(this, input, pos);
501
- else
502
- return this._suspend(input, pos, currentLineLength);
482
+ if (inputFinished || (!/^'''|^"""/.test(input) && /\n|\r/.test(input)))
483
+ return reportSyntaxError(this);
484
+ else {
485
+ this._linePosition = currentLineLength - input.length;
486
+ return this._input = input;
487
+ }
503
488
  }
504
489
 
505
490
  // Emit the parsed token
506
491
  // Consumption includes separator whitespace; lexicalLength excludes it
507
- // and any synthetic EOF space. Consumption is clamped to the input below.
492
+ // and any synthetic EOF space. slice below clamps consumption to the input.
508
493
  const length = matchLength || match[0].length;
509
- const start = currentLineLength - (input.length - pos);
510
494
  let token;
511
495
  if (finalLineLength) {
512
496
  token = {
513
- type, value, prefix, line, start,
497
+ type, value, prefix, line,
498
+ start: currentLineLength - input.length,
514
499
  end: finalLineLength, endLine: this._line,
515
500
  };
516
501
  callback(null, token);
517
502
  }
518
503
  else
519
- token = emitToken(type, value, prefix, line, start, lexicalLength || length);
504
+ token = emitToken(type, value, prefix, line, lexicalLength || length);
520
505
  this.previousToken = token;
521
506
  this._previousMarker = type;
522
507
 
523
508
  // Advance to next part to tokenize
524
- pos = Math.min(pos + length, input.length);
509
+ input = input.slice(length);
525
510
  if (finalLineLength)
526
- currentLineLength = input.length - pos + finalLineLength;
511
+ currentLineLength = input.length + finalLineLength;
527
512
  }
528
513
 
529
514
  // Emits a comment at its exact position within matched whitespace.
530
- function emitComment(value, line, start) {
515
+ function emitComment(value, line, offset) {
516
+ const start = currentLineLength - input.length + offset;
531
517
  callback(null, {
532
518
  type: 'comment', value, prefix: '', line,
533
519
  start, end: start + value.length + 1,
534
520
  });
535
521
  }
536
522
  // Emits the token through the callback
537
- function emitToken(type, value, prefix, line, start, length) {
538
- const token = { type, value, prefix, line, start, end: start + length };
523
+ function emitToken(type, value, prefix, line, length) {
524
+ const start = input ? currentLineLength - input.length : currentLineLength;
525
+ const end = start + length;
526
+ const token = { type, value, prefix, line, start, end };
539
527
  callback(null, token);
540
528
  return token;
541
529
  }
542
530
  // Signals the syntax error through the callback
543
- function reportSyntaxError(self, input, pos) {
544
- callback(self._syntaxError(execAt(nonWhitespace, input, pos)[0]));
545
- }
546
- }
547
-
548
- // ### `_suspend` keeps the unconsumed input until more input arrives
549
- _suspend(input, pos, currentLineLength) {
550
- this._linePosition = currentLineLength - (input.length - pos);
551
- return this._input = input.slice(pos);
531
+ function reportSyntaxError(self) { callback(self._syntaxError(/^\S*/.exec(input)[0])); }
552
532
  }
553
533
 
554
534
  // ### `_matchN3Verb` matches an N3 verb unless the input is a longer prefixed name
555
- _matchN3Verb(input, pos, inputFinished) {
556
- const verb = execAt(this._n3Verb, input, pos);
535
+ _matchN3Verb(input, inputFinished) {
536
+ const verb = this._n3Verb.exec(input);
557
537
  if (!verb)
558
538
  return null;
559
539
 
560
540
  // Most verb boundaries cannot be part of a prefix, so keep the common path fast.
561
- const next = input[pos + verb[0].length];
541
+ const next = input[verb[0].length];
562
542
  if (next !== '-' && next !== '_' && (next < '0' || next > '9'))
563
543
  return verb;
564
544
 
565
545
  // A prefix can start with a verb and continue with characters that are also
566
546
  // valid verb boundaries. Prefer the longer prefixed name when it is complete.
567
- if (execAt(this._prefixed, input, pos))
547
+ if (this._prefixed.exec(input))
568
548
  return null;
569
549
  // Appending to the input only matters when a prefixed name could run up to
570
550
  // its end, which a character that cannot occur in prefixed names rules out.
571
551
  // This avoids copying the rest of the document for every such verb.
572
- if (testAt(nonPrefixedNameChar, input, pos))
552
+ if (nonPrefixedNameChar.test(input))
573
553
  return verb;
574
- if (execAtEnd(this._prefixed, input, pos))
554
+ if (this._prefixed.exec(`${input} `))
575
555
  return null;
576
556
 
577
557
  // If a stream chunk ends partway through such a prefix, wait for the colon
578
558
  // instead of prematurely emitting the verb. Appending ": " lets the prefix
579
559
  // grammar determine whether all input seen so far can be a complete prefix.
580
560
  if (!inputFinished) {
581
- const prefix = execAt(this._prefix, `${input.slice(pos)}: `, 0);
561
+ const prefix = this._prefix.exec(`${input}: `);
582
562
  if (prefix)
583
563
  return null;
584
564
  }
@@ -632,20 +612,20 @@ export default class N3Lexer {
632
612
  return result + item.slice(start);
633
613
  }
634
614
 
635
- // ### `_parseLiteral` parses a literal at the given position into an unescaped value
636
- _parseLiteral(input, pos) {
615
+ // ### `_parseLiteral` parses a literal into an unescaped value
616
+ _parseLiteral(input) {
637
617
  // Ensure we have enough lookahead to identify triple-quoted strings
638
- if (input.length - pos >= 3) {
618
+ if (input.length >= 3) {
639
619
  // The caller has already identified a single or double quote.
640
- const quote = input[pos];
641
- const openingLength = input[pos + 1] === quote && input[pos + 2] === quote ? 3 : 1;
620
+ const quote = input[0];
621
+ const openingLength = input[1] === quote && input[2] === quote ? 3 : 1;
642
622
  let opening = quote;
643
623
  if (openingLength === 3)
644
624
  opening = quote === '"' ? '"""' : "'''";
645
625
 
646
626
  // Find the next candidate closing quotes
647
- let closingPos = pos + Math.max(this._literalClosingPos, openingLength);
648
- while ((closingPos = input.indexOf(opening, closingPos)) > pos) {
627
+ let closingPos = Math.max(this._literalClosingPos, openingLength);
628
+ while ((closingPos = input.indexOf(opening, closingPos)) > 0) {
649
629
  // Count backslashes right before the closing quotes
650
630
  let backslashCount = 0;
651
631
  while (input[closingPos - backslashCount - 1] === '\\')
@@ -655,10 +635,10 @@ export default class N3Lexer {
655
635
  // means these are actual, non-escaped closing quotes
656
636
  if (backslashCount % 2 === 0) {
657
637
  // Extract and unescape the value
658
- const raw = input.substring(pos + openingLength, closingPos),
638
+ const raw = input.substring(openingLength, closingPos),
659
639
  lines = raw.split(/\r\n|\r|\n/),
660
640
  lineCount = lines.length - 1;
661
- const matchLength = closingPos - pos + openingLength;
641
+ const matchLength = closingPos + openingLength;
662
642
  // Only triple-quoted strings can be multi-line
663
643
  if (openingLength === 1 && lineCount !== 0 ||
664
644
  openingLength === 3 && this._lineMode)
@@ -669,7 +649,7 @@ export default class N3Lexer {
669
649
  }
670
650
  closingPos++;
671
651
  }
672
- this._literalClosingPos = input.length - pos - openingLength + 1;
652
+ this._literalClosingPos = input.length - openingLength + 1;
673
653
  }
674
654
  return { value: '', matchLength: 0, finalLineLength: 0 };
675
655
  }