n3 3.0.0-alpha.4 → 3.0.0-alpha.6
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +56 -0
- package/browser/n3.esm.min.js +15 -16
- package/browser/n3.min.js +15 -16
- package/lib/N3Lexer.js +255 -160
- package/lib/N3Parser.js +70 -32
- package/lib/N3Reasoner.js +1 -1
- package/lib/N3Store.js +50 -6
- package/lib/N3StreamParser.js +4 -0
- package/lib/N3Writer.js +65 -19
- package/lib/Util.js +1 -1
- package/package.json +2 -29
- package/src/N3Lexer.js +269 -171
- package/src/N3Parser.js +81 -32
- package/src/N3Reasoner.js +1 -1
- package/src/N3Store.js +61 -9
- package/src/N3StreamParser.js +2 -0
- package/src/N3Writer.js +70 -23
- package/src/Util.js +1 -1
package/lib/N3Lexer.js
CHANGED
|
@@ -4,7 +4,6 @@ Object.defineProperty(exports, "__esModule", {
|
|
|
4
4
|
value: true
|
|
5
5
|
});
|
|
6
6
|
exports.default = void 0;
|
|
7
|
-
var _buffer = require("buffer");
|
|
8
7
|
var _IRIs = _interopRequireDefault(require("./IRIs"));
|
|
9
8
|
function _interopRequireDefault(e) { return e && e.__esModule ? e : { default: e }; }
|
|
10
9
|
// **N3Lexer** tokenizes N3 documents.
|
|
@@ -54,7 +53,8 @@ const localNameEscapeReplacements = {
|
|
|
54
53
|
};
|
|
55
54
|
const illegalIriChars = /[\x00-\x20<>\\"\{\}\|\^\`]/;
|
|
56
55
|
// Characters that cannot occur in a prefixed name, not even escaped
|
|
57
|
-
|
|
56
|
+
// (global, so that testAt searches the rest of the input from a position)
|
|
57
|
+
const nonPrefixedNameChar = /[\s<>"{}|^`]/g;
|
|
58
58
|
|
|
59
59
|
// A valid code point is a Unicode scalar value: at most U+10FFFF and not a surrogate
|
|
60
60
|
function isValidCodePoint(charCode) {
|
|
@@ -70,30 +70,59 @@ const lineModeRegExps = {
|
|
|
70
70
|
_whitespace: true
|
|
71
71
|
};
|
|
72
72
|
const invalidRegExp = /$0^/;
|
|
73
|
+
const nonWhitespace = /\S*/y;
|
|
74
|
+
|
|
75
|
+
// Matches a sticky regular expression at the given position of the input
|
|
76
|
+
function execAt(regExp, input, pos) {
|
|
77
|
+
regExp.lastIndex = pos;
|
|
78
|
+
return regExp.exec(input);
|
|
79
|
+
}
|
|
80
|
+
function testAt(regExp, input, pos) {
|
|
81
|
+
regExp.lastIndex = pos;
|
|
82
|
+
return regExp.test(input);
|
|
83
|
+
}
|
|
84
|
+
// Matches the rest of the input followed by a space, as at the end of the input,
|
|
85
|
+
// a token that can contain (but not end with) a dot needs a non-dot character after it
|
|
86
|
+
function execAtEnd(regExp, input, pos) {
|
|
87
|
+
regExp.lastIndex = 0;
|
|
88
|
+
return regExp.exec(`${input.slice(pos)} `);
|
|
89
|
+
}
|
|
90
|
+
|
|
91
|
+
// Whitespace or the start of a comment
|
|
92
|
+
function isSeparatorCode(code) {
|
|
93
|
+
return code === SPACE || code === TAB || code === LF || code === CR || code === HASH;
|
|
94
|
+
}
|
|
95
|
+
|
|
96
|
+
// Words with a fixed meaning in the grammar, which cannot name an additional directive
|
|
97
|
+
const reservedWords = /^(?:prefix|base|version|graph|forsome|forall|iri|a|true|false|has|is|of|id)$/i;
|
|
98
|
+
|
|
99
|
+
// Unfinished input in a stream up to this length is tokenized again with every chunk
|
|
100
|
+
const MIN_RESCAN_LENGTH = 1024;
|
|
73
101
|
|
|
74
102
|
// ## Constructor
|
|
75
103
|
class N3Lexer {
|
|
76
104
|
constructor(options) {
|
|
77
105
|
// ## Regular expressions
|
|
78
|
-
// It's slightly faster to have these as properties than as in-scope variables
|
|
79
|
-
|
|
80
|
-
this.
|
|
81
|
-
this.
|
|
82
|
-
this.
|
|
83
|
-
this.
|
|
84
|
-
this.
|
|
85
|
-
this.
|
|
86
|
-
this.
|
|
87
|
-
this.
|
|
88
|
-
this.
|
|
89
|
-
this.
|
|
90
|
-
this.
|
|
91
|
-
this.
|
|
92
|
-
this.
|
|
93
|
-
this.
|
|
94
|
-
this.
|
|
95
|
-
this.
|
|
96
|
-
this.
|
|
106
|
+
// It's slightly faster to have these as properties than as in-scope variables.
|
|
107
|
+
// They are sticky, so they only match at the `lastIndex` set by `execAt`.
|
|
108
|
+
this._iri = /<((?:[^ <>{}\\]|\\[uU])+)>[ \t]*/y; // IRI with escape sequences; needs sanity check after unescaping
|
|
109
|
+
this._unescapedIri = /<([^\x00-\x20<>\\"\{\}\|\^\`]*)>[ \t]*/y; // IRI without escape sequences; no unescaping
|
|
110
|
+
this._simpleQuotedString = /"([^"\\\r\n]*)"(?=[^"])/y; // string without escape sequences
|
|
111
|
+
this._simpleApostropheString = /'([^'\\\r\n]*)'(?=[^'])/y;
|
|
112
|
+
this._langcode = /@([a-z]+(?:-[a-z0-9]+)*)(?=[^a-z0-9])/iy;
|
|
113
|
+
this._prefix = /((?:[A-Za-z\xc0-\xd6\xd8-\xf6\xf8-\u02ff\u0370-\u037d\u037f-\u1fff\u200c\u200d\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff])(?:\.?[\-0-9A-Z_a-z\xb7\xc0-\xd6\xd8-\xf6\xf8-\u037d\u037f-\u1fff\u200c\u200d\u203f\u2040\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff])*)?:(?=[#\s<])/y;
|
|
114
|
+
this._prefixed = /((?:[A-Za-z\xc0-\xd6\xd8-\xf6\xf8-\u02ff\u0370-\u037d\u037f-\u1fff\u200c\u200d\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff])(?:\.?[\-0-9A-Z_a-z\xb7\xc0-\xd6\xd8-\xf6\xf8-\u037d\u037f-\u1fff\u200c\u200d\u203f\u2040\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff])*)?:((?:(?:[0-9:A-Z_a-z\xc0-\xd6\xd8-\xf6\xf8-\u02ff\u0370-\u037d\u037f-\u1fff\u200c\u200d\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff]|%[0-9a-fA-F]{2}|\\[!#-\/;=?\-@_~])(?:(?:[\.\-0-9:A-Z_a-z\xb7\xc0-\xd6\xd8-\xf6\xf8-\u037d\u037f-\u1fff\u200c\u200d\u203f\u2040\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff]|%[0-9a-fA-F]{2}|\\[!#-\/;=?\-@_~])*(?:[\-0-9:A-Z_a-z\xb7\xc0-\xd6\xd8-\xf6\xf8-\u037d\u037f-\u1fff\u200c\u200d\u203f\u2040\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff]|%[0-9a-fA-F]{2}|\\[!#-\/;=?\-@_~]))?)?)(?:[ \t]+|(?=\.?[,;!\^\s#()\[\]\{\}"'<>]))/y;
|
|
115
|
+
this._variable = /\?(?:(?:[A-Z_a-z\xc0-\xd6\xd8-\xf6\xf8-\u02ff\u0370-\u037d\u037f-\u1fff\u200c\u200d\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff])(?:[\-0-9:A-Z_a-z\xb7\xc0-\xd6\xd8-\xf6\xf8-\u037d\u037f-\u1fff\u200c\u200d\u203f\u2040\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff])*)(?=[.,;!\^\s#()\[\]\{\}"'<>])/y;
|
|
116
|
+
this._blank = /_:((?:[0-9A-Z_a-z\xc0-\xd6\xd8-\xf6\xf8-\u02ff\u0370-\u037d\u037f-\u1fff\u200c\u200d\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff])(?:\.?[\-0-9A-Z_a-z\xb7\xc0-\xd6\xd8-\xf6\xf8-\u037d\u037f-\u1fff\u200c\u200d\u203f\u2040\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff])*)(?:[ \t]+|(?=\.?[,;:!\^\s#()\[\]\{\}"'<>]))/y;
|
|
117
|
+
this._number = /[\-+]?(?:(\d+\.\d*|\.?\d+)[eE][\-+]?\d+|(?=\.?\d)\d*(?:(\.)\d+)?)(?=\.?[,;:!\^\s#()\[\]\{\}"'<>])/y;
|
|
118
|
+
this._boolean = /(?:true|false)(?=[.,;!\^\s#()\[\]\{\}"'<>])/y;
|
|
119
|
+
this._atKeyword = /@[a-z]+(?=[\s#<:"'])/iy;
|
|
120
|
+
this._keyword = /(?:PREFIX|BASE|VERSION|GRAPH)(?=[\s#<"'])/iy;
|
|
121
|
+
this._n3Verb = /(?:has|is|of)(?=[\s#()\[\]\{\}"'<>?_+\-0-9])/y;
|
|
122
|
+
this._n3Id = /id(?=[\s#<])/y;
|
|
123
|
+
this._shortPredicates = /a(?=[\s#()\[\]\{\}"'<>])/y;
|
|
124
|
+
this._commentLine = /[ \t]*#([^\n\r]*)(?:\r\n|\n|\r)([ \t]*)/y;
|
|
125
|
+
this._whitespace = /[ \t]+/y;
|
|
97
126
|
options = options || {};
|
|
98
127
|
|
|
99
128
|
// Whether the log:isImpliedBy predicate is supported
|
|
@@ -106,11 +135,25 @@ class N3Lexer {
|
|
|
106
135
|
for (const key in this) {
|
|
107
136
|
if (!(key in lineModeRegExps) && this[key] instanceof RegExp) this[key] = invalidRegExp;
|
|
108
137
|
}
|
|
138
|
+
// The only keyword in N-Triples and N-Quads is VERSION, which is case-sensitive
|
|
139
|
+
this._keyword = /VERSION(?=[\s#<"])/y;
|
|
109
140
|
}
|
|
110
141
|
// When not in line mode, enable N3 functionality by default
|
|
111
142
|
else {
|
|
112
143
|
this._n3Mode = options.n3 !== false;
|
|
113
144
|
}
|
|
145
|
+
// Recognize additional directive keywords, such as MESSAGE
|
|
146
|
+
// (the @-form of a directive is always tokenized as an @-keyword)
|
|
147
|
+
this._directive = null;
|
|
148
|
+
if (options.directives && options.directives.length !== 0) {
|
|
149
|
+
for (const name of options.directives) {
|
|
150
|
+
if (!/^[a-z]+$/i.test(name) || reservedWords.test(name)) throw new Error(`Invalid directive name: "${name}"`);
|
|
151
|
+
}
|
|
152
|
+
this._directive = new RegExp(`(?:${options.directives.join('|')})(?=[\\s#<])`, 'iy');
|
|
153
|
+
this._directiveMaxLength = Math.max(...options.directives.map(name => name.length));
|
|
154
|
+
// The first characters of directive names, so other words skip the regular expression
|
|
155
|
+
this._directiveStarts = options.directives.map(name => name[0].toLowerCase() + name[0].toUpperCase()).join('');
|
|
156
|
+
}
|
|
114
157
|
// Don't output comment tokens by default
|
|
115
158
|
this.comments = !!options.comments;
|
|
116
159
|
// Cache the last tested closing position of long literals
|
|
@@ -121,77 +164,73 @@ class N3Lexer {
|
|
|
121
164
|
|
|
122
165
|
// ### `_tokenizeToEnd` tokenizes as for as possible, emitting tokens through the callback
|
|
123
166
|
_tokenizeToEnd(callback, inputFinished) {
|
|
124
|
-
// Continue parsing as far as possible; the loop will return eventually
|
|
125
|
-
|
|
167
|
+
// Continue parsing as far as possible; the loop will return eventually.
|
|
168
|
+
// Rather than slicing off every token, track the position of the remaining input;
|
|
169
|
+
// the regular expressions are sticky, so they match at that position.
|
|
170
|
+
const input = this._input;
|
|
171
|
+
let pos = 0;
|
|
126
172
|
let currentLineLength = this._linePosition + input.length;
|
|
127
173
|
while (true) {
|
|
128
174
|
// Consume one separator line at a time, including its following indentation.
|
|
129
175
|
while (true) {
|
|
130
|
-
let charCode = input.charCodeAt(
|
|
176
|
+
let charCode = input.charCodeAt(pos),
|
|
131
177
|
separatorLength = 0;
|
|
132
178
|
if (charCode === SPACE || charCode === TAB) {
|
|
133
|
-
const next = input.charCodeAt(1);
|
|
134
|
-
separatorLength = next === SPACE || next === TAB ? this._whitespace
|
|
135
|
-
charCode = input.charCodeAt(separatorLength);
|
|
179
|
+
const next = input.charCodeAt(pos + 1);
|
|
180
|
+
separatorLength = next === SPACE || next === TAB ? execAt(this._whitespace, input, pos)[0].length : 1;
|
|
181
|
+
charCode = input.charCodeAt(pos + separatorLength);
|
|
136
182
|
}
|
|
137
183
|
if (charCode === HASH) {
|
|
138
|
-
const comment = this._commentLine
|
|
184
|
+
const comment = execAt(this._commentLine, input, pos);
|
|
139
185
|
if (comment) {
|
|
140
186
|
const commentLength = comment[0].length;
|
|
141
187
|
// Keep a trailing CR buffered in case the next chunk starts with LF.
|
|
142
|
-
if (!inputFinished && commentLength === input.length && input.charCodeAt(
|
|
143
|
-
|
|
144
|
-
|
|
145
|
-
|
|
146
|
-
if (this.comments) emitComment(comment[1], this._line, separatorLength);
|
|
147
|
-
input = input.slice(commentLength);
|
|
148
|
-
currentLineLength = input.length + comment[2].length;
|
|
188
|
+
if (!inputFinished && pos + commentLength === input.length && input.charCodeAt(input.length - 1) === CR) return this._suspend(input, pos, currentLineLength);
|
|
189
|
+
if (this.comments) emitComment(comment[1], this._line, currentLineLength - (input.length - pos) + separatorLength);
|
|
190
|
+
pos += commentLength;
|
|
191
|
+
currentLineLength = input.length - pos + comment[2].length;
|
|
149
192
|
this._line++;
|
|
150
193
|
} else {
|
|
151
194
|
// A comment without a line ending stays buffered until EOF.
|
|
152
|
-
|
|
153
|
-
if (!inputFinished)
|
|
154
|
-
|
|
155
|
-
|
|
156
|
-
}
|
|
157
|
-
if (this.comments) emitComment(input.slice(1), this._line, 0);
|
|
158
|
-
input = '';
|
|
195
|
+
pos += separatorLength;
|
|
196
|
+
if (!inputFinished) return this._suspend(input, pos, currentLineLength);
|
|
197
|
+
if (this.comments) emitComment(input.slice(pos + 1), this._line, currentLineLength - (input.length - pos));
|
|
198
|
+
pos = input.length;
|
|
159
199
|
break;
|
|
160
200
|
}
|
|
161
201
|
} else if (charCode === LF || charCode === CR) {
|
|
162
202
|
// A CR at the end of a chunk may still be followed by LF.
|
|
163
|
-
if (!inputFinished && charCode === CR && separatorLength + 1 === input.length)
|
|
164
|
-
|
|
165
|
-
return this._input = input;
|
|
166
|
-
}
|
|
167
|
-
separatorLength += charCode === CR && input.charCodeAt(separatorLength + 1) === LF ? 2 : 1;
|
|
203
|
+
if (!inputFinished && charCode === CR && pos + separatorLength + 1 === input.length) return this._suspend(input, pos, currentLineLength);
|
|
204
|
+
separatorLength += charCode === CR && input.charCodeAt(pos + separatorLength + 1) === LF ? 2 : 1;
|
|
168
205
|
// Indentation is consumed with the newline, but belongs to the next line's columns.
|
|
169
206
|
let indentationLength = 0;
|
|
170
|
-
const next = input.charCodeAt(separatorLength);
|
|
207
|
+
const next = input.charCodeAt(pos + separatorLength);
|
|
171
208
|
if (next === SPACE || next === TAB) {
|
|
172
|
-
const following = input.charCodeAt(separatorLength + 1);
|
|
173
|
-
indentationLength = following === SPACE || following === TAB ? this._whitespace
|
|
209
|
+
const following = input.charCodeAt(pos + separatorLength + 1);
|
|
210
|
+
indentationLength = following === SPACE || following === TAB ? execAt(this._whitespace, input, pos + separatorLength)[0].length : 1;
|
|
174
211
|
}
|
|
175
|
-
|
|
176
|
-
currentLineLength = input.length + indentationLength;
|
|
212
|
+
pos += separatorLength + indentationLength;
|
|
213
|
+
currentLineLength = input.length - pos + indentationLength;
|
|
177
214
|
this._line++;
|
|
178
215
|
} else {
|
|
179
|
-
|
|
216
|
+
pos += separatorLength;
|
|
180
217
|
break;
|
|
181
218
|
}
|
|
182
219
|
}
|
|
183
|
-
if (input.length
|
|
220
|
+
if (pos >= input.length) {
|
|
221
|
+
// A datatype marker needs a type
|
|
222
|
+
if (inputFinished && this._previousMarker === '^^') return reportSyntaxError(this, input, pos);
|
|
223
|
+
this._linePosition = currentLineLength;
|
|
184
224
|
if (inputFinished) {
|
|
185
|
-
|
|
186
|
-
|
|
225
|
+
emitToken('eof', '', '', this._line, currentLineLength, 0);
|
|
226
|
+
return this._input = null;
|
|
187
227
|
}
|
|
188
|
-
this.
|
|
189
|
-
return this._input = input;
|
|
228
|
+
return this._input = '';
|
|
190
229
|
}
|
|
191
230
|
|
|
192
231
|
// Look for specific token types based on the first character
|
|
193
232
|
const line = this._line,
|
|
194
|
-
firstChar = input[
|
|
233
|
+
firstChar = input[pos];
|
|
195
234
|
let type = '',
|
|
196
235
|
value = '',
|
|
197
236
|
prefix = '',
|
|
@@ -199,17 +238,22 @@ class N3Lexer {
|
|
|
199
238
|
matchLength = 0,
|
|
200
239
|
lexicalLength = 0,
|
|
201
240
|
finalLineLength = 0,
|
|
202
|
-
inconclusive = false
|
|
241
|
+
inconclusive = false,
|
|
242
|
+
tripleQuoted = false;
|
|
203
243
|
switch (firstChar) {
|
|
204
244
|
case '^':
|
|
245
|
+
// A datatype marker separated from its type cannot be followed by another marker
|
|
246
|
+
if (this._previousMarker === '^^') return reportSyntaxError(this, input, pos);
|
|
205
247
|
// We need at least 3 tokens lookahead to distinguish ^^<IRI> and ^^pre:fixed
|
|
206
|
-
if (input.length < 3) break;
|
|
248
|
+
if (input.length - pos < 3) break;
|
|
207
249
|
// Try to match a type
|
|
208
|
-
else if (input[1] === '^') {
|
|
250
|
+
else if (input[pos + 1] === '^') {
|
|
209
251
|
this._previousMarker = '^^';
|
|
210
252
|
// Move to type IRI or prefixed name
|
|
211
|
-
|
|
212
|
-
if (input[
|
|
253
|
+
pos += 2;
|
|
254
|
+
if (input[pos] !== '<') {
|
|
255
|
+
// Whitespace and comments may separate the marker from the type
|
|
256
|
+
if (isSeparatorCode(input.charCodeAt(pos))) continue; // eslint-disable-line no-continue
|
|
213
257
|
inconclusive = true;
|
|
214
258
|
break;
|
|
215
259
|
}
|
|
@@ -225,53 +269,54 @@ class N3Lexer {
|
|
|
225
269
|
// Fall through in case the type is an IRI
|
|
226
270
|
case '<':
|
|
227
271
|
// Try to find a full IRI without escape sequences
|
|
228
|
-
if (match = this._unescapedIri
|
|
272
|
+
if (match = execAt(this._unescapedIri, input, pos)) {
|
|
229
273
|
type = 'IRI', value = match[1];
|
|
230
274
|
lexicalLength = match[1].length + 2;
|
|
231
275
|
}
|
|
232
276
|
// Try to find a full IRI with escape sequences
|
|
233
|
-
else if (match = this._iri
|
|
277
|
+
else if (match = execAt(this._iri, input, pos)) {
|
|
234
278
|
value = this._unescape(match[1], stringEscapeReplacements);
|
|
235
|
-
if (value === null || illegalIriChars.test(value)) return reportSyntaxError(this);
|
|
279
|
+
if (value === null || illegalIriChars.test(value)) return reportSyntaxError(this, input, pos);
|
|
236
280
|
type = 'IRI';
|
|
237
281
|
lexicalLength = match[1].length + 2;
|
|
238
282
|
}
|
|
239
283
|
// Try to find a triple term
|
|
240
|
-
else if (input.length > 2 && input[1] === '<' && input[2] === '(') type = '<<(', matchLength = 3;
|
|
284
|
+
else if (input.length - pos > 2 && input[pos + 1] === '<' && input[pos + 2] === '(') type = '<<(', matchLength = 3;
|
|
241
285
|
// Try to find a reified triple
|
|
242
|
-
else if (!this._lineMode && input.length > (inputFinished ? 1 : 2) && input[1] === '<') type = '<<', matchLength = 2;
|
|
286
|
+
else if (!this._lineMode && input.length - pos > (inputFinished ? 1 : 2) && input[pos + 1] === '<') type = '<<', matchLength = 2;
|
|
243
287
|
// Try to find a backwards implication arrow
|
|
244
|
-
else if (this._n3Mode && input.length > 1 && input[1] === '=') {
|
|
288
|
+
else if (this._n3Mode && input.length - pos > 1 && input[pos + 1] === '=') {
|
|
245
289
|
matchLength = 2;
|
|
246
290
|
if (this._isImpliedBy) type = 'abbreviation', value = '<';else type = 'inverse', value = '>';
|
|
247
291
|
}
|
|
248
292
|
// Try to find an inverted predicate marker
|
|
249
|
-
else if (this._n3Mode && input.length > 1 && input[1] === '-') type = 'inversePredicate', matchLength = 2;
|
|
293
|
+
else if (this._n3Mode && input.length - pos > 1 && input[pos + 1] === '-') type = 'inversePredicate', matchLength = 2;
|
|
250
294
|
break;
|
|
251
295
|
case '>':
|
|
252
296
|
// Try to find a reified triple
|
|
253
|
-
if (input.length > 1 && input[1] === '>') type = '>>', matchLength = 2;
|
|
297
|
+
if (input.length - pos > 1 && input[pos + 1] === '>') type = '>>', matchLength = 2;
|
|
254
298
|
break;
|
|
255
299
|
case '_':
|
|
256
300
|
// Try to find a blank node. Since it can contain (but not end with) a dot,
|
|
257
301
|
// we always need a non-dot character before deciding it is a blank node.
|
|
258
302
|
// Therefore, try inserting a space if we're at the end of the input.
|
|
259
|
-
if ((match = this._blank
|
|
303
|
+
if ((match = execAt(this._blank, input, pos)) || inputFinished && (match = execAtEnd(this._blank, input, pos))) {
|
|
260
304
|
type = 'blank', prefix = '_', value = match[1];
|
|
261
305
|
lexicalLength = match[1].length + 2;
|
|
262
306
|
}
|
|
263
307
|
break;
|
|
264
308
|
case '"':
|
|
265
309
|
// Try to find a literal without escape sequences
|
|
266
|
-
if (match = this._simpleQuotedString
|
|
310
|
+
if (match = execAt(this._simpleQuotedString, input, pos)) value = match[1];
|
|
267
311
|
// Try to find a literal wrapped in three pairs of quotes
|
|
268
312
|
else {
|
|
269
313
|
({
|
|
270
314
|
value,
|
|
271
315
|
matchLength,
|
|
272
|
-
finalLineLength
|
|
273
|
-
|
|
274
|
-
|
|
316
|
+
finalLineLength,
|
|
317
|
+
tripleQuoted
|
|
318
|
+
} = this._parseLiteral(input, pos));
|
|
319
|
+
if (value === null) return reportSyntaxError(this, input, pos);
|
|
275
320
|
}
|
|
276
321
|
if (match !== null || matchLength !== 0) {
|
|
277
322
|
type = 'literal';
|
|
@@ -281,15 +326,16 @@ class N3Lexer {
|
|
|
281
326
|
case "'":
|
|
282
327
|
if (!this._lineMode) {
|
|
283
328
|
// Try to find a literal without escape sequences
|
|
284
|
-
if (match = this._simpleApostropheString
|
|
329
|
+
if (match = execAt(this._simpleApostropheString, input, pos)) value = match[1];
|
|
285
330
|
// Try to find a literal wrapped in three pairs of quotes
|
|
286
331
|
else {
|
|
287
332
|
({
|
|
288
333
|
value,
|
|
289
334
|
matchLength,
|
|
290
|
-
finalLineLength
|
|
291
|
-
|
|
292
|
-
|
|
335
|
+
finalLineLength,
|
|
336
|
+
tripleQuoted
|
|
337
|
+
} = this._parseLiteral(input, pos));
|
|
338
|
+
if (value === null) return reportSyntaxError(this, input, pos);
|
|
293
339
|
}
|
|
294
340
|
if (match !== null || matchLength !== 0) {
|
|
295
341
|
type = 'literal';
|
|
@@ -299,7 +345,7 @@ class N3Lexer {
|
|
|
299
345
|
break;
|
|
300
346
|
case '?':
|
|
301
347
|
// Try to find a variable
|
|
302
|
-
if (this._n3Mode && (match = this._variable
|
|
348
|
+
if (this._n3Mode && (match = execAt(this._variable, input, pos))) type = 'var', value = match[0];
|
|
303
349
|
break;
|
|
304
350
|
case '@':
|
|
305
351
|
// Try to find a language code. A language code can contain dash-separated
|
|
@@ -307,15 +353,16 @@ class N3Lexer {
|
|
|
307
353
|
// input is not finished, another subtag may still arrive in a later chunk and
|
|
308
354
|
// the match would be premature; wait for more input in that case.
|
|
309
355
|
// A double dash starts a direction code, which cannot extend the language code.
|
|
310
|
-
if (this._previousMarker === 'literal' && (match = this._langcode
|
|
311
|
-
|
|
356
|
+
if (this._previousMarker === 'literal' && (match = execAt(this._langcode, input, pos)) && match[1] !== 'version') {
|
|
357
|
+
const end = pos + match[0].length;
|
|
358
|
+
if (!inputFinished && input[end] === '-' && input[end + 1] !== '-') match = null;else type = 'langcode', value = match[1];
|
|
312
359
|
}
|
|
313
360
|
// Try to find a keyword
|
|
314
|
-
else if (match = this._atKeyword
|
|
361
|
+
else if (match = execAt(this._atKeyword, input, pos)) type = match[0];
|
|
315
362
|
break;
|
|
316
363
|
case '.':
|
|
317
364
|
// Try to find a dot as punctuation
|
|
318
|
-
if (input.length === 1 ? inputFinished : input[1] < '0' || input[1] > '9') {
|
|
365
|
+
if (input.length - pos === 1 ? inputFinished : input[pos + 1] < '0' || input[pos + 1] > '9') {
|
|
319
366
|
type = '.';
|
|
320
367
|
matchLength = 1;
|
|
321
368
|
break;
|
|
@@ -334,10 +381,10 @@ class N3Lexer {
|
|
|
334
381
|
case '9':
|
|
335
382
|
case '+':
|
|
336
383
|
case '-':
|
|
337
|
-
if (input[1] === '-') {
|
|
384
|
+
if (input[pos + 1] === '-') {
|
|
338
385
|
// Try to find a direction code
|
|
339
386
|
if (this._previousMarker === 'langcode') {
|
|
340
|
-
if (input.startsWith('--ltr')) type = 'dircode', value = 'ltr', matchLength = 5;else if (input.startsWith('--rtl')) type = 'dircode', value = 'rtl', matchLength = 5;
|
|
387
|
+
if (input.startsWith('--ltr', pos)) type = 'dircode', value = 'ltr', matchLength = 5;else if (input.startsWith('--rtl', pos)) type = 'dircode', value = 'rtl', matchLength = 5;
|
|
341
388
|
}
|
|
342
389
|
break;
|
|
343
390
|
}
|
|
@@ -345,7 +392,7 @@ class N3Lexer {
|
|
|
345
392
|
// Try to find a number. Since it can contain (but not end with) a dot,
|
|
346
393
|
// we always need a non-dot character before deciding it is a number.
|
|
347
394
|
// Therefore, try inserting a space if we're at the end of the input.
|
|
348
|
-
if (match = this._number
|
|
395
|
+
if (match = execAt(this._number, input, pos) || inputFinished && (match = execAtEnd(this._number, input, pos))) {
|
|
349
396
|
type = 'literal', value = match[0];
|
|
350
397
|
prefix = typeof match[1] === 'string' ? xsd.double : typeof match[2] === 'string' ? xsd.decimal : xsd.integer;
|
|
351
398
|
}
|
|
@@ -359,42 +406,42 @@ class N3Lexer {
|
|
|
359
406
|
case 'V':
|
|
360
407
|
case 'v':
|
|
361
408
|
// Try to find a SPARQL-style keyword
|
|
362
|
-
if (match = this._keyword
|
|
409
|
+
if (match = execAt(this._keyword, input, pos)) type = match[0].toUpperCase();else inconclusive = true;
|
|
363
410
|
break;
|
|
364
411
|
case 'f':
|
|
365
412
|
case 't':
|
|
366
413
|
// Try to match a boolean
|
|
367
|
-
if (this._boolean
|
|
414
|
+
if (testAt(this._boolean, input, pos)) type = 'literal', value = firstChar === 't' ? 'true' : 'false', prefix = xsd.boolean, matchLength = value.length;else inconclusive = true;
|
|
368
415
|
break;
|
|
369
416
|
case 'a':
|
|
370
417
|
// Try to find an abbreviated predicate
|
|
371
|
-
if (this._shortPredicates
|
|
418
|
+
if (testAt(this._shortPredicates, input, pos)) type = 'abbreviation', value = 'a', matchLength = 1;else inconclusive = true;
|
|
372
419
|
break;
|
|
373
420
|
case 'h':
|
|
374
421
|
case 'o':
|
|
375
422
|
// Try to find an N3 verb keyword
|
|
376
|
-
if (this._n3Mode && (match = this._matchN3Verb(input, inputFinished))) type = match[0];else inconclusive = true;
|
|
423
|
+
if (this._n3Mode && (match = this._matchN3Verb(input, pos, inputFinished))) type = match[0];else inconclusive = true;
|
|
377
424
|
break;
|
|
378
425
|
case 'i':
|
|
379
426
|
// Try to find an IRI property list identifier or N3 verb keyword
|
|
380
|
-
if (this._n3Mode && this._n3Id
|
|
427
|
+
if (this._n3Mode && testAt(this._n3Id, input, pos)) type = 'id', matchLength = 2;else if (this._n3Mode && (match = this._matchN3Verb(input, pos, inputFinished))) type = match[0];else inconclusive = true;
|
|
381
428
|
break;
|
|
382
429
|
case '=':
|
|
383
430
|
// Try to find an implication arrow or equals sign
|
|
384
|
-
if (this._n3Mode && input.length > 1) {
|
|
431
|
+
if (this._n3Mode && input.length - pos > 1) {
|
|
385
432
|
type = 'abbreviation';
|
|
386
|
-
if (input[1] !== '>') matchLength = 1, value = '=';else matchLength = 2, value = '>';
|
|
433
|
+
if (input[pos + 1] !== '>') matchLength = 1, value = '=';else matchLength = 2, value = '>';
|
|
387
434
|
}
|
|
388
435
|
break;
|
|
389
436
|
case '!':
|
|
390
437
|
if (!this._n3Mode) break;
|
|
391
438
|
case ')':
|
|
392
|
-
if (!inputFinished && (input.length === 1 || input.length === 2 && input[1] === '>')) {
|
|
439
|
+
if (!inputFinished && (input.length - pos === 1 || input.length - pos === 2 && input[pos + 1] === '>')) {
|
|
393
440
|
// Don't consume yet, as it *could* become a triple term end.
|
|
394
441
|
break;
|
|
395
442
|
}
|
|
396
443
|
// Try to find a triple term
|
|
397
|
-
if (input.length > 2 && input[1] === '>' && input[2] === '>') {
|
|
444
|
+
if (input.length - pos > 2 && input[pos + 1] === '>' && input[pos + 2] === '>') {
|
|
398
445
|
type = ')>>', matchLength = 3;
|
|
399
446
|
break;
|
|
400
447
|
}
|
|
@@ -412,15 +459,15 @@ class N3Lexer {
|
|
|
412
459
|
break;
|
|
413
460
|
case '{':
|
|
414
461
|
// We need at least 2 tokens lookahead to distinguish "{|" and "{ "
|
|
415
|
-
if (!this._lineMode && input.length >= 2) {
|
|
462
|
+
if (!this._lineMode && input.length - pos >= 2) {
|
|
416
463
|
// Try to find a quoted triple annotation start
|
|
417
|
-
if (input[1] === '|') type = '{|', matchLength = 2;else type = firstChar, matchLength = 1;
|
|
464
|
+
if (input[pos + 1] === '|') type = '{|', matchLength = 2;else type = firstChar, matchLength = 1;
|
|
418
465
|
}
|
|
419
466
|
break;
|
|
420
467
|
case '|':
|
|
421
468
|
// We need 2 tokens lookahead to parse "|}"
|
|
422
469
|
// Try to find a quoted triple annotation end
|
|
423
|
-
if (input.length >= 2 && input[1] === '}') type = '|}', matchLength = 2;
|
|
470
|
+
if (input.length - pos >= 2 && input[pos + 1] === '}') type = '|}', matchLength = 2;
|
|
424
471
|
break;
|
|
425
472
|
default:
|
|
426
473
|
inconclusive = true;
|
|
@@ -429,11 +476,14 @@ class N3Lexer {
|
|
|
429
476
|
// Some first characters do not allow an immediate decision, so inspect more
|
|
430
477
|
if (inconclusive) {
|
|
431
478
|
// Try to find a prefix
|
|
432
|
-
if ((this._previousMarker === '@prefix' || this._previousMarker === 'PREFIX') && (match = this._prefix
|
|
479
|
+
if ((this._previousMarker === '@prefix' || this._previousMarker === 'PREFIX') && (match = execAt(this._prefix, input, pos))) type = 'prefix', value = match[1] || '';
|
|
480
|
+
// Try to find an additional directive keyword
|
|
481
|
+
// (at the end of the input, only a short final word can be one)
|
|
482
|
+
else if (this._directive !== null && this._directiveStarts.includes(firstChar) && ((match = execAt(this._directive, input, pos)) || inputFinished && input.length - pos <= this._directiveMaxLength && (match = execAtEnd(this._directive, input, pos)))) type = match[0].toUpperCase();
|
|
433
483
|
// Try to find a prefixed name. Since it can contain (but not end with) a dot,
|
|
434
484
|
// we always need a non-dot character before deciding it is a prefixed name.
|
|
435
485
|
// Therefore, try inserting a space if we're at the end of the input.
|
|
436
|
-
else if ((match = this._prefixed
|
|
486
|
+
else if ((match = execAt(this._prefixed, input, pos)) || inputFinished && (match = execAtEnd(this._prefixed, input, pos))) {
|
|
437
487
|
type = 'prefixed', prefix = match[1] || '';
|
|
438
488
|
value = this._unescape(match[2], localNameEscapeReplacements);
|
|
439
489
|
lexicalLength = prefix.length + match[2].length + 1;
|
|
@@ -459,16 +509,14 @@ class N3Lexer {
|
|
|
459
509
|
// We could be in streaming mode, and then we just wait for more input to arrive.
|
|
460
510
|
// Otherwise, a syntax error has occurred in the input.
|
|
461
511
|
// One exception: error on an unaccounted linebreak (= not inside a triple-quoted literal).
|
|
462
|
-
if (inputFinished ||
|
|
463
|
-
this._linePosition = currentLineLength - input.length;
|
|
464
|
-
return this._input = input;
|
|
465
|
-
}
|
|
512
|
+
if (inputFinished || !input.startsWith("'''", pos) && !input.startsWith('"""', pos) && /\n|\r/.test(input.slice(pos))) return reportSyntaxError(this, input, pos);else return this._suspend(input, pos, currentLineLength);
|
|
466
513
|
}
|
|
467
514
|
|
|
468
515
|
// Emit the parsed token
|
|
469
516
|
// Consumption includes separator whitespace; lexicalLength excludes it
|
|
470
|
-
// and any synthetic EOF space.
|
|
517
|
+
// and any synthetic EOF space. Consumption is clamped to the input below.
|
|
471
518
|
const length = matchLength || match[0].length;
|
|
519
|
+
const start = currentLineLength - (input.length - pos);
|
|
472
520
|
let token;
|
|
473
521
|
if (finalLineLength) {
|
|
474
522
|
token = {
|
|
@@ -476,23 +524,37 @@ class N3Lexer {
|
|
|
476
524
|
value,
|
|
477
525
|
prefix,
|
|
478
526
|
line,
|
|
479
|
-
start
|
|
527
|
+
start,
|
|
480
528
|
end: finalLineLength,
|
|
481
|
-
endLine: this._line
|
|
529
|
+
endLine: this._line,
|
|
530
|
+
tripleQuoted
|
|
482
531
|
};
|
|
483
532
|
callback(null, token);
|
|
484
|
-
}
|
|
533
|
+
}
|
|
534
|
+
// Triple-quoted strings are marked, since version declarations do not allow them
|
|
535
|
+
else if (tripleQuoted) {
|
|
536
|
+
token = {
|
|
537
|
+
type,
|
|
538
|
+
value,
|
|
539
|
+
prefix,
|
|
540
|
+
line,
|
|
541
|
+
start,
|
|
542
|
+
end: start + length,
|
|
543
|
+
tripleQuoted
|
|
544
|
+
};
|
|
545
|
+
callback(null, token);
|
|
546
|
+
} else token = emitToken(type, value, prefix, line, start, lexicalLength || length);
|
|
485
547
|
this.previousToken = token;
|
|
486
|
-
|
|
548
|
+
// The string of a version declaration cannot take a language tag, so a following @keyword is a keyword
|
|
549
|
+
this._previousMarker = type === 'literal' && (this._previousMarker === 'VERSION' || this._previousMarker === '@version') ? 'version' : type;
|
|
487
550
|
|
|
488
551
|
// Advance to next part to tokenize
|
|
489
|
-
|
|
490
|
-
if (finalLineLength) currentLineLength = input.length + finalLineLength;
|
|
552
|
+
pos = Math.min(pos + length, input.length);
|
|
553
|
+
if (finalLineLength) currentLineLength = input.length - pos + finalLineLength;
|
|
491
554
|
}
|
|
492
555
|
|
|
493
556
|
// Emits a comment at its exact position within matched whitespace.
|
|
494
|
-
function emitComment(value, line,
|
|
495
|
-
const start = currentLineLength - input.length + offset;
|
|
557
|
+
function emitComment(value, line, start) {
|
|
496
558
|
callback(null, {
|
|
497
559
|
type: 'comment',
|
|
498
560
|
value,
|
|
@@ -503,49 +565,53 @@ class N3Lexer {
|
|
|
503
565
|
});
|
|
504
566
|
}
|
|
505
567
|
// Emits the token through the callback
|
|
506
|
-
function emitToken(type, value, prefix, line, length) {
|
|
507
|
-
const start = input ? currentLineLength - input.length : currentLineLength;
|
|
508
|
-
const end = start + length;
|
|
568
|
+
function emitToken(type, value, prefix, line, start, length) {
|
|
509
569
|
const token = {
|
|
510
570
|
type,
|
|
511
571
|
value,
|
|
512
572
|
prefix,
|
|
513
573
|
line,
|
|
514
574
|
start,
|
|
515
|
-
end
|
|
575
|
+
end: start + length
|
|
516
576
|
};
|
|
517
577
|
callback(null, token);
|
|
518
578
|
return token;
|
|
519
579
|
}
|
|
520
580
|
// Signals the syntax error through the callback
|
|
521
|
-
function reportSyntaxError(self) {
|
|
522
|
-
callback(self._syntaxError(
|
|
581
|
+
function reportSyntaxError(self, input, pos) {
|
|
582
|
+
callback(self._syntaxError(execAt(nonWhitespace, input, pos)[0]));
|
|
523
583
|
}
|
|
524
584
|
}
|
|
525
585
|
|
|
586
|
+
// ### `_suspend` keeps the unconsumed input until more input arrives
|
|
587
|
+
_suspend(input, pos, currentLineLength) {
|
|
588
|
+
this._linePosition = currentLineLength - (input.length - pos);
|
|
589
|
+
return this._input = input.slice(pos);
|
|
590
|
+
}
|
|
591
|
+
|
|
526
592
|
// ### `_matchN3Verb` matches an N3 verb unless the input is a longer prefixed name
|
|
527
|
-
_matchN3Verb(input, inputFinished) {
|
|
528
|
-
const verb = this._n3Verb
|
|
593
|
+
_matchN3Verb(input, pos, inputFinished) {
|
|
594
|
+
const verb = execAt(this._n3Verb, input, pos);
|
|
529
595
|
if (!verb) return null;
|
|
530
596
|
|
|
531
597
|
// Most verb boundaries cannot be part of a prefix, so keep the common path fast.
|
|
532
|
-
const next = input[verb[0].length];
|
|
598
|
+
const next = input[pos + verb[0].length];
|
|
533
599
|
if (next !== '-' && next !== '_' && (next < '0' || next > '9')) return verb;
|
|
534
600
|
|
|
535
601
|
// A prefix can start with a verb and continue with characters that are also
|
|
536
602
|
// valid verb boundaries. Prefer the longer prefixed name when it is complete.
|
|
537
|
-
if (this._prefixed
|
|
603
|
+
if (execAt(this._prefixed, input, pos)) return null;
|
|
538
604
|
// Appending to the input only matters when a prefixed name could run up to
|
|
539
605
|
// its end, which a character that cannot occur in prefixed names rules out.
|
|
540
606
|
// This avoids copying the rest of the document for every such verb.
|
|
541
|
-
if (nonPrefixedNameChar
|
|
542
|
-
if (this._prefixed
|
|
607
|
+
if (testAt(nonPrefixedNameChar, input, pos)) return verb;
|
|
608
|
+
if (execAtEnd(this._prefixed, input, pos)) return null;
|
|
543
609
|
|
|
544
610
|
// If a stream chunk ends partway through such a prefix, wait for the colon
|
|
545
611
|
// instead of prematurely emitting the verb. Appending ": " lets the prefix
|
|
546
612
|
// grammar determine whether all input seen so far can be a complete prefix.
|
|
547
613
|
if (!inputFinished) {
|
|
548
|
-
const prefix = this._prefix
|
|
614
|
+
const prefix = execAt(this._prefix, `${input.slice(pos)}: `, 0);
|
|
549
615
|
if (prefix) return null;
|
|
550
616
|
}
|
|
551
617
|
return verb;
|
|
@@ -592,19 +658,19 @@ class N3Lexer {
|
|
|
592
658
|
return result + item.slice(start);
|
|
593
659
|
}
|
|
594
660
|
|
|
595
|
-
// ### `_parseLiteral` parses a literal into an unescaped value
|
|
596
|
-
_parseLiteral(input) {
|
|
661
|
+
// ### `_parseLiteral` parses a literal at the given position into an unescaped value
|
|
662
|
+
_parseLiteral(input, pos) {
|
|
597
663
|
// Ensure we have enough lookahead to identify triple-quoted strings
|
|
598
|
-
if (input.length >= 3) {
|
|
664
|
+
if (input.length - pos >= 3) {
|
|
599
665
|
// The caller has already identified a single or double quote.
|
|
600
|
-
const quote = input[
|
|
601
|
-
const openingLength = input[1] === quote && input[2] === quote ? 3 : 1;
|
|
666
|
+
const quote = input[pos];
|
|
667
|
+
const openingLength = input[pos + 1] === quote && input[pos + 2] === quote ? 3 : 1;
|
|
602
668
|
let opening = quote;
|
|
603
669
|
if (openingLength === 3) opening = quote === '"' ? '"""' : "'''";
|
|
604
670
|
|
|
605
671
|
// Find the next candidate closing quotes
|
|
606
|
-
let closingPos = Math.max(this._literalClosingPos, openingLength);
|
|
607
|
-
while ((closingPos = input.indexOf(opening, closingPos)) >
|
|
672
|
+
let closingPos = pos + Math.max(this._literalClosingPos, openingLength);
|
|
673
|
+
while ((closingPos = input.indexOf(opening, closingPos)) > pos) {
|
|
608
674
|
// Count backslashes right before the closing quotes
|
|
609
675
|
let backslashCount = 0;
|
|
610
676
|
while (input[closingPos - backslashCount - 1] === '\\') backslashCount++;
|
|
@@ -613,10 +679,10 @@ class N3Lexer {
|
|
|
613
679
|
// means these are actual, non-escaped closing quotes
|
|
614
680
|
if (backslashCount % 2 === 0) {
|
|
615
681
|
// Extract and unescape the value
|
|
616
|
-
const raw = input.substring(openingLength, closingPos),
|
|
682
|
+
const raw = input.substring(pos + openingLength, closingPos),
|
|
617
683
|
lines = raw.split(/\r\n|\r|\n/),
|
|
618
684
|
lineCount = lines.length - 1;
|
|
619
|
-
const matchLength = closingPos + openingLength;
|
|
685
|
+
const matchLength = closingPos - pos + openingLength;
|
|
620
686
|
// Only triple-quoted strings can be multi-line
|
|
621
687
|
if (openingLength === 1 && lineCount !== 0 || openingLength === 3 && this._lineMode) break;
|
|
622
688
|
this._line += lineCount;
|
|
@@ -624,24 +690,45 @@ class N3Lexer {
|
|
|
624
690
|
return {
|
|
625
691
|
value: this._unescape(raw, stringEscapeReplacements),
|
|
626
692
|
matchLength,
|
|
627
|
-
finalLineLength
|
|
693
|
+
finalLineLength,
|
|
694
|
+
tripleQuoted: openingLength === 3
|
|
628
695
|
};
|
|
629
696
|
}
|
|
630
697
|
closingPos++;
|
|
631
698
|
}
|
|
632
|
-
this._literalClosingPos = input.length - openingLength + 1;
|
|
699
|
+
this._literalClosingPos = input.length - pos - openingLength + 1;
|
|
633
700
|
}
|
|
634
701
|
return {
|
|
635
702
|
value: '',
|
|
636
703
|
matchLength: 0,
|
|
637
|
-
finalLineLength: 0
|
|
704
|
+
finalLineLength: 0,
|
|
705
|
+
tripleQuoted: false
|
|
638
706
|
};
|
|
639
707
|
}
|
|
640
708
|
|
|
709
|
+
// ### `_tryTokenizeToEnd` tokenizes as far as possible, reporting failures through the callback
|
|
710
|
+
_tryTokenizeToEnd(callback, inputFinished) {
|
|
711
|
+
// Keep track of errors thrown by the callback, which must reach the caller unchanged
|
|
712
|
+
let callbackError;
|
|
713
|
+
try {
|
|
714
|
+
this._tokenizeToEnd((error, token) => {
|
|
715
|
+
try {
|
|
716
|
+
return callback(error, token);
|
|
717
|
+
} catch (thrown) {
|
|
718
|
+
throw callbackError = thrown;
|
|
719
|
+
}
|
|
720
|
+
}, inputFinished);
|
|
721
|
+
} catch (error) {
|
|
722
|
+
// Matching an extremely long token can exhaust the regular expression stack
|
|
723
|
+
if (error === callbackError || !(error instanceof RangeError)) throw error;
|
|
724
|
+
callback(this._syntaxError(null, `Token too long on line ${this._line}.`));
|
|
725
|
+
}
|
|
726
|
+
}
|
|
727
|
+
|
|
641
728
|
// ### `_syntaxError` creates a syntax error for the given issue
|
|
642
|
-
_syntaxError(issue) {
|
|
729
|
+
_syntaxError(issue, message = `Unexpected "${issue}" on line ${this._line}.`) {
|
|
643
730
|
this._input = null;
|
|
644
|
-
const err = new Error(
|
|
731
|
+
const err = new Error(message);
|
|
645
732
|
err.context = {
|
|
646
733
|
token: undefined,
|
|
647
734
|
line: this._line,
|
|
@@ -682,44 +769,52 @@ class N3Lexer {
|
|
|
682
769
|
this._input = this._readStartingBom(input);
|
|
683
770
|
// If a callback was passed, asynchronously call it
|
|
684
771
|
if (typeof callback === 'function') queueMicrotask(() => {
|
|
685
|
-
if (this._tokenization === tokenization) this.
|
|
772
|
+
if (this._tokenization === tokenization) this._tryTokenizeToEnd(callback, true);
|
|
686
773
|
});
|
|
687
774
|
// If no callback was passed, tokenize synchronously and return
|
|
688
775
|
else {
|
|
689
776
|
const tokens = [];
|
|
690
777
|
let error;
|
|
691
|
-
this.
|
|
778
|
+
this._tryTokenizeToEnd((e, t) => e ? error = e : tokens.push(t), true);
|
|
692
779
|
if (error) throw error;
|
|
693
780
|
return tokens;
|
|
694
781
|
}
|
|
695
782
|
}
|
|
696
783
|
// Otherwise, the input must be a stream
|
|
697
784
|
else {
|
|
698
|
-
|
|
785
|
+
let decoder,
|
|
786
|
+
retryLength = 0;
|
|
699
787
|
if (typeof input.setEncoding === 'function') input.setEncoding('utf8');
|
|
700
788
|
// Adds the data chunk to the buffer and parses as far as possible
|
|
701
789
|
input.on('data', data => {
|
|
702
790
|
if (this._tokenization === tokenization && this._input !== null && data.length !== 0) {
|
|
703
|
-
//
|
|
704
|
-
if (
|
|
705
|
-
|
|
706
|
-
|
|
707
|
-
|
|
708
|
-
|
|
709
|
-
|
|
710
|
-
|
|
791
|
+
// Decode bytes, keeping an incomplete trailing character for the next chunk
|
|
792
|
+
if (typeof data !== 'string') {
|
|
793
|
+
decoder = decoder || new TextDecoder('utf-8', {
|
|
794
|
+
ignoreBOM: true
|
|
795
|
+
});
|
|
796
|
+
if (!(data = decoder.decode(data, {
|
|
797
|
+
stream: true
|
|
798
|
+
}))) return;
|
|
711
799
|
}
|
|
712
|
-
//
|
|
713
|
-
else
|
|
714
|
-
|
|
715
|
-
|
|
716
|
-
|
|
800
|
+
// Only read a BOM at the start
|
|
801
|
+
if (typeof this._input === 'undefined') this._input = this._readStartingBom(data);else this._input += data;
|
|
802
|
+
// Tokenize as far as possible. When a previous attempt left a long unfinished token,
|
|
803
|
+
// wait until the buffered input has doubled, so the token is not rescanned for every chunk.
|
|
804
|
+
if (this._input.length >= retryLength) {
|
|
805
|
+
this._tryTokenizeToEnd(callback, false);
|
|
806
|
+
retryLength = this._input !== null && this._input.length > MIN_RESCAN_LENGTH ? 2 * this._input.length : 0;
|
|
717
807
|
}
|
|
718
808
|
}
|
|
719
809
|
});
|
|
720
810
|
// Parses until the end
|
|
721
811
|
input.on('end', () => {
|
|
722
|
-
if (this._tokenization === tokenization &&
|
|
812
|
+
if (this._tokenization === tokenization && this._input !== null) {
|
|
813
|
+
// Decode any incomplete character left at the end
|
|
814
|
+
const rest = decoder ? decoder.decode() : '';
|
|
815
|
+
if (rest) this._input = typeof this._input === 'string' ? this._input + rest : rest;
|
|
816
|
+
if (typeof this._input === 'string') this._tryTokenizeToEnd(callback, true);
|
|
817
|
+
}
|
|
723
818
|
});
|
|
724
819
|
input.on('error', error => {
|
|
725
820
|
if (this._tokenization === tokenization) callback(error);
|