n3 2.11.2 → 3.0.0-alpha.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +10 -9
- package/browser/n3.esm.min.js +12 -12
- package/browser/n3.min.js +12 -12
- package/lib/N3Lexer.js +133 -148
- package/lib/N3Parser.js +5 -4
- package/package.json +29 -2
- package/src/N3Lexer.js +142 -162
- package/src/N3Parser.js +5 -4
package/lib/N3Lexer.js
CHANGED
|
@@ -54,8 +54,7 @@ const localNameEscapeReplacements = {
|
|
|
54
54
|
};
|
|
55
55
|
const illegalIriChars = /[\x00-\x20<>\\"\{\}\|\^\`]/;
|
|
56
56
|
// Characters that cannot occur in a prefixed name, not even escaped
|
|
57
|
-
|
|
58
|
-
const nonPrefixedNameChar = /[\s<>"{}|^`]/g;
|
|
57
|
+
const nonPrefixedNameChar = /[\s<>"{}|^`]/;
|
|
59
58
|
|
|
60
59
|
// A valid code point is a Unicode scalar value: at most U+10FFFF and not a surrogate
|
|
61
60
|
function isValidCodePoint(charCode) {
|
|
@@ -71,48 +70,30 @@ const lineModeRegExps = {
|
|
|
71
70
|
_whitespace: true
|
|
72
71
|
};
|
|
73
72
|
const invalidRegExp = /$0^/;
|
|
74
|
-
const nonWhitespace = /\S*/y;
|
|
75
|
-
|
|
76
|
-
// Matches a sticky regular expression at the given position of the input
|
|
77
|
-
function execAt(regExp, input, pos) {
|
|
78
|
-
regExp.lastIndex = pos;
|
|
79
|
-
return regExp.exec(input);
|
|
80
|
-
}
|
|
81
|
-
function testAt(regExp, input, pos) {
|
|
82
|
-
regExp.lastIndex = pos;
|
|
83
|
-
return regExp.test(input);
|
|
84
|
-
}
|
|
85
|
-
// Matches the rest of the input followed by a space, as at the end of the input,
|
|
86
|
-
// a token that can contain (but not end with) a dot needs a non-dot character after it
|
|
87
|
-
function execAtEnd(regExp, input, pos) {
|
|
88
|
-
regExp.lastIndex = 0;
|
|
89
|
-
return regExp.exec(`${input.slice(pos)} `);
|
|
90
|
-
}
|
|
91
73
|
|
|
92
74
|
// ## Constructor
|
|
93
75
|
class N3Lexer {
|
|
94
76
|
constructor(options) {
|
|
95
77
|
// ## Regular expressions
|
|
96
|
-
// It's slightly faster to have these as properties than as in-scope variables
|
|
97
|
-
|
|
98
|
-
this.
|
|
99
|
-
this.
|
|
100
|
-
this.
|
|
101
|
-
this.
|
|
102
|
-
this.
|
|
103
|
-
this.
|
|
104
|
-
this.
|
|
105
|
-
this.
|
|
106
|
-
this.
|
|
107
|
-
this.
|
|
108
|
-
this.
|
|
109
|
-
this.
|
|
110
|
-
this.
|
|
111
|
-
this.
|
|
112
|
-
this.
|
|
113
|
-
this.
|
|
114
|
-
this.
|
|
115
|
-
this._whitespace = /[ \t]+/y;
|
|
78
|
+
// It's slightly faster to have these as properties than as in-scope variables
|
|
79
|
+
this._iri = /^<((?:[^ <>{}\\]|\\[uU])+)>[ \t]*/; // IRI with escape sequences; needs sanity check after unescaping
|
|
80
|
+
this._unescapedIri = /^<([^\x00-\x20<>\\"\{\}\|\^\`]*)>[ \t]*/; // IRI without escape sequences; no unescaping
|
|
81
|
+
this._simpleQuotedString = /^"([^"\\\r\n]*)"(?=[^"])/; // string without escape sequences
|
|
82
|
+
this._simpleApostropheString = /^'([^'\\\r\n]*)'(?=[^'])/;
|
|
83
|
+
this._langcode = /^@([a-z]+(?:-[a-z0-9]+)*)(?=[^a-z0-9])/i;
|
|
84
|
+
this._prefix = /^((?:[A-Za-z\xc0-\xd6\xd8-\xf6\xf8-\u02ff\u0370-\u037d\u037f-\u1fff\u200c\u200d\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff])(?:\.?[\-0-9A-Z_a-z\xb7\xc0-\xd6\xd8-\xf6\xf8-\u037d\u037f-\u1fff\u200c\u200d\u203f\u2040\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff])*)?:(?=[#\s<])/;
|
|
85
|
+
this._prefixed = /^((?:[A-Za-z\xc0-\xd6\xd8-\xf6\xf8-\u02ff\u0370-\u037d\u037f-\u1fff\u200c\u200d\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff])(?:\.?[\-0-9A-Z_a-z\xb7\xc0-\xd6\xd8-\xf6\xf8-\u037d\u037f-\u1fff\u200c\u200d\u203f\u2040\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff])*)?:((?:(?:[0-9:A-Z_a-z\xc0-\xd6\xd8-\xf6\xf8-\u02ff\u0370-\u037d\u037f-\u1fff\u200c\u200d\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff]|%[0-9a-fA-F]{2}|\\[!#-\/;=?\-@_~])(?:(?:[\.\-0-9:A-Z_a-z\xb7\xc0-\xd6\xd8-\xf6\xf8-\u037d\u037f-\u1fff\u200c\u200d\u203f\u2040\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff]|%[0-9a-fA-F]{2}|\\[!#-\/;=?\-@_~])*(?:[\-0-9:A-Z_a-z\xb7\xc0-\xd6\xd8-\xf6\xf8-\u037d\u037f-\u1fff\u200c\u200d\u203f\u2040\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff]|%[0-9a-fA-F]{2}|\\[!#-\/;=?\-@_~]))?)?)(?:[ \t]+|(?=\.?[,;!\^\s#()\[\]\{\}"'<>]))/;
|
|
86
|
+
this._variable = /^\?(?:(?:[A-Z_a-z\xc0-\xd6\xd8-\xf6\xf8-\u02ff\u0370-\u037d\u037f-\u1fff\u200c\u200d\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff])(?:[\-0-9:A-Z_a-z\xb7\xc0-\xd6\xd8-\xf6\xf8-\u037d\u037f-\u1fff\u200c\u200d\u203f\u2040\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff])*)(?=[.,;!\^\s#()\[\]\{\}"'<>])/;
|
|
87
|
+
this._blank = /^_:((?:[0-9A-Z_a-z\xc0-\xd6\xd8-\xf6\xf8-\u02ff\u0370-\u037d\u037f-\u1fff\u200c\u200d\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff])(?:\.?[\-0-9A-Z_a-z\xb7\xc0-\xd6\xd8-\xf6\xf8-\u037d\u037f-\u1fff\u200c\u200d\u203f\u2040\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff])*)(?:[ \t]+|(?=\.?[,;:!\^\s#()\[\]\{\}"'<>]))/;
|
|
88
|
+
this._number = /^[\-+]?(?:(\d+\.\d*|\.?\d+)[eE][\-+]?\d+|(?=\.?\d)\d*(?:(\.)\d+)?)(?=\.?[,;:!\^\s#()\[\]\{\}"'<>])/;
|
|
89
|
+
this._boolean = /^(?:true|false)(?=[.,;!\^\s#()\[\]\{\}"'<>])/;
|
|
90
|
+
this._atKeyword = /^@[a-z]+(?=[\s#<:])/i;
|
|
91
|
+
this._keyword = /^(?:PREFIX|BASE|VERSION|GRAPH)(?=[\s#<])/i;
|
|
92
|
+
this._n3Verb = /^(?:has|is|of)(?=[\s#()\[\]\{\}"'<>?_+\-0-9])/;
|
|
93
|
+
this._n3Id = /^id(?=[\s#<])/;
|
|
94
|
+
this._shortPredicates = /^a(?=[\s#()\[\]\{\}"'<>])/;
|
|
95
|
+
this._commentLine = /^[ \t]*#([^\n\r]*)(?:\r\n|\n|\r)([ \t]*)/;
|
|
96
|
+
this._whitespace = /^[ \t]+/;
|
|
116
97
|
options = options || {};
|
|
117
98
|
|
|
118
99
|
// Whether the log:isImpliedBy predicate is supported
|
|
@@ -140,71 +121,77 @@ class N3Lexer {
|
|
|
140
121
|
|
|
141
122
|
// ### `_tokenizeToEnd` tokenizes as for as possible, emitting tokens through the callback
|
|
142
123
|
_tokenizeToEnd(callback, inputFinished) {
|
|
143
|
-
// Continue parsing as far as possible; the loop will return eventually
|
|
144
|
-
|
|
145
|
-
// the regular expressions are sticky, so they match at that position.
|
|
146
|
-
const input = this._input;
|
|
147
|
-
let pos = 0;
|
|
124
|
+
// Continue parsing as far as possible; the loop will return eventually
|
|
125
|
+
let input = this._input;
|
|
148
126
|
let currentLineLength = this._linePosition + input.length;
|
|
149
127
|
while (true) {
|
|
150
128
|
// Consume one separator line at a time, including its following indentation.
|
|
151
129
|
while (true) {
|
|
152
|
-
let charCode = input.charCodeAt(
|
|
130
|
+
let charCode = input.charCodeAt(0),
|
|
153
131
|
separatorLength = 0;
|
|
154
132
|
if (charCode === SPACE || charCode === TAB) {
|
|
155
|
-
const next = input.charCodeAt(
|
|
156
|
-
separatorLength = next === SPACE || next === TAB ?
|
|
157
|
-
charCode = input.charCodeAt(
|
|
133
|
+
const next = input.charCodeAt(1);
|
|
134
|
+
separatorLength = next === SPACE || next === TAB ? this._whitespace.exec(input)[0].length : 1;
|
|
135
|
+
charCode = input.charCodeAt(separatorLength);
|
|
158
136
|
}
|
|
159
137
|
if (charCode === HASH) {
|
|
160
|
-
const comment =
|
|
138
|
+
const comment = this._commentLine.exec(input);
|
|
161
139
|
if (comment) {
|
|
162
140
|
const commentLength = comment[0].length;
|
|
163
141
|
// Keep a trailing CR buffered in case the next chunk starts with LF.
|
|
164
|
-
if (!inputFinished &&
|
|
165
|
-
|
|
166
|
-
|
|
167
|
-
|
|
142
|
+
if (!inputFinished && commentLength === input.length && input.charCodeAt(commentLength - 1) === CR) {
|
|
143
|
+
this._linePosition = currentLineLength - input.length;
|
|
144
|
+
return this._input = input;
|
|
145
|
+
}
|
|
146
|
+
if (this.comments) emitComment(comment[1], this._line, separatorLength);
|
|
147
|
+
input = input.slice(commentLength);
|
|
148
|
+
currentLineLength = input.length + comment[2].length;
|
|
168
149
|
this._line++;
|
|
169
150
|
} else {
|
|
170
151
|
// A comment without a line ending stays buffered until EOF.
|
|
171
|
-
|
|
172
|
-
if (!inputFinished)
|
|
173
|
-
|
|
174
|
-
|
|
152
|
+
input = input.slice(separatorLength);
|
|
153
|
+
if (!inputFinished) {
|
|
154
|
+
this._linePosition = currentLineLength - input.length;
|
|
155
|
+
return this._input = input;
|
|
156
|
+
}
|
|
157
|
+
if (this.comments) emitComment(input.slice(1), this._line, 0);
|
|
158
|
+
input = '';
|
|
175
159
|
break;
|
|
176
160
|
}
|
|
177
161
|
} else if (charCode === LF || charCode === CR) {
|
|
178
162
|
// A CR at the end of a chunk may still be followed by LF.
|
|
179
|
-
if (!inputFinished && charCode === CR &&
|
|
180
|
-
|
|
163
|
+
if (!inputFinished && charCode === CR && separatorLength + 1 === input.length) {
|
|
164
|
+
this._linePosition = currentLineLength - input.length;
|
|
165
|
+
return this._input = input;
|
|
166
|
+
}
|
|
167
|
+
separatorLength += charCode === CR && input.charCodeAt(separatorLength + 1) === LF ? 2 : 1;
|
|
181
168
|
// Indentation is consumed with the newline, but belongs to the next line's columns.
|
|
182
169
|
let indentationLength = 0;
|
|
183
|
-
const next = input.charCodeAt(
|
|
170
|
+
const next = input.charCodeAt(separatorLength);
|
|
184
171
|
if (next === SPACE || next === TAB) {
|
|
185
|
-
const following = input.charCodeAt(
|
|
186
|
-
indentationLength = following === SPACE || following === TAB ?
|
|
172
|
+
const following = input.charCodeAt(separatorLength + 1);
|
|
173
|
+
indentationLength = following === SPACE || following === TAB ? this._whitespace.exec(input.slice(separatorLength))[0].length : 1;
|
|
187
174
|
}
|
|
188
|
-
|
|
189
|
-
currentLineLength = input.length
|
|
175
|
+
input = input.slice(separatorLength + indentationLength);
|
|
176
|
+
currentLineLength = input.length + indentationLength;
|
|
190
177
|
this._line++;
|
|
191
178
|
} else {
|
|
192
|
-
|
|
179
|
+
if (separatorLength !== 0) input = input.slice(separatorLength);
|
|
193
180
|
break;
|
|
194
181
|
}
|
|
195
182
|
}
|
|
196
|
-
if (
|
|
197
|
-
this._linePosition = currentLineLength;
|
|
183
|
+
if (input.length === 0) {
|
|
198
184
|
if (inputFinished) {
|
|
199
|
-
|
|
200
|
-
|
|
185
|
+
input = null;
|
|
186
|
+
emitToken('eof', '', '', this._line, 0);
|
|
201
187
|
}
|
|
202
|
-
|
|
188
|
+
this._linePosition = currentLineLength;
|
|
189
|
+
return this._input = input;
|
|
203
190
|
}
|
|
204
191
|
|
|
205
192
|
// Look for specific token types based on the first character
|
|
206
193
|
const line = this._line,
|
|
207
|
-
firstChar = input[
|
|
194
|
+
firstChar = input[0];
|
|
208
195
|
let type = '',
|
|
209
196
|
value = '',
|
|
210
197
|
prefix = '',
|
|
@@ -216,13 +203,13 @@ class N3Lexer {
|
|
|
216
203
|
switch (firstChar) {
|
|
217
204
|
case '^':
|
|
218
205
|
// We need at least 3 tokens lookahead to distinguish ^^<IRI> and ^^pre:fixed
|
|
219
|
-
if (input.length
|
|
206
|
+
if (input.length < 3) break;
|
|
220
207
|
// Try to match a type
|
|
221
|
-
else if (input[
|
|
208
|
+
else if (input[1] === '^') {
|
|
222
209
|
this._previousMarker = '^^';
|
|
223
210
|
// Move to type IRI or prefixed name
|
|
224
|
-
|
|
225
|
-
if (input[
|
|
211
|
+
input = input.slice(2);
|
|
212
|
+
if (input[0] !== '<') {
|
|
226
213
|
inconclusive = true;
|
|
227
214
|
break;
|
|
228
215
|
}
|
|
@@ -238,53 +225,53 @@ class N3Lexer {
|
|
|
238
225
|
// Fall through in case the type is an IRI
|
|
239
226
|
case '<':
|
|
240
227
|
// Try to find a full IRI without escape sequences
|
|
241
|
-
if (match =
|
|
228
|
+
if (match = this._unescapedIri.exec(input)) {
|
|
242
229
|
type = 'IRI', value = match[1];
|
|
243
230
|
lexicalLength = match[1].length + 2;
|
|
244
231
|
}
|
|
245
232
|
// Try to find a full IRI with escape sequences
|
|
246
|
-
else if (match =
|
|
233
|
+
else if (match = this._iri.exec(input)) {
|
|
247
234
|
value = this._unescape(match[1], stringEscapeReplacements);
|
|
248
|
-
if (value === null || illegalIriChars.test(value)) return reportSyntaxError(this
|
|
235
|
+
if (value === null || illegalIriChars.test(value)) return reportSyntaxError(this);
|
|
249
236
|
type = 'IRI';
|
|
250
237
|
lexicalLength = match[1].length + 2;
|
|
251
238
|
}
|
|
252
239
|
// Try to find a triple term
|
|
253
|
-
else if (input.length
|
|
240
|
+
else if (input.length > 2 && input[1] === '<' && input[2] === '(') type = '<<(', matchLength = 3;
|
|
254
241
|
// Try to find a reified triple
|
|
255
|
-
else if (!this._lineMode && input.length
|
|
242
|
+
else if (!this._lineMode && input.length > (inputFinished ? 1 : 2) && input[1] === '<') type = '<<', matchLength = 2;
|
|
256
243
|
// Try to find a backwards implication arrow
|
|
257
|
-
else if (this._n3Mode && input.length
|
|
244
|
+
else if (this._n3Mode && input.length > 1 && input[1] === '=') {
|
|
258
245
|
matchLength = 2;
|
|
259
246
|
if (this._isImpliedBy) type = 'abbreviation', value = '<';else type = 'inverse', value = '>';
|
|
260
247
|
}
|
|
261
248
|
// Try to find an inverted predicate marker
|
|
262
|
-
else if (this._n3Mode && input.length
|
|
249
|
+
else if (this._n3Mode && input.length > 1 && input[1] === '-') type = 'inversePredicate', matchLength = 2;
|
|
263
250
|
break;
|
|
264
251
|
case '>':
|
|
265
252
|
// Try to find a reified triple
|
|
266
|
-
if (input.length
|
|
253
|
+
if (input.length > 1 && input[1] === '>') type = '>>', matchLength = 2;
|
|
267
254
|
break;
|
|
268
255
|
case '_':
|
|
269
256
|
// Try to find a blank node. Since it can contain (but not end with) a dot,
|
|
270
257
|
// we always need a non-dot character before deciding it is a blank node.
|
|
271
258
|
// Therefore, try inserting a space if we're at the end of the input.
|
|
272
|
-
if ((match =
|
|
259
|
+
if ((match = this._blank.exec(input)) || inputFinished && (match = this._blank.exec(`${input} `))) {
|
|
273
260
|
type = 'blank', prefix = '_', value = match[1];
|
|
274
261
|
lexicalLength = match[1].length + 2;
|
|
275
262
|
}
|
|
276
263
|
break;
|
|
277
264
|
case '"':
|
|
278
265
|
// Try to find a literal without escape sequences
|
|
279
|
-
if (match =
|
|
266
|
+
if (match = this._simpleQuotedString.exec(input)) value = match[1];
|
|
280
267
|
// Try to find a literal wrapped in three pairs of quotes
|
|
281
268
|
else {
|
|
282
269
|
({
|
|
283
270
|
value,
|
|
284
271
|
matchLength,
|
|
285
272
|
finalLineLength
|
|
286
|
-
} = this._parseLiteral(input
|
|
287
|
-
if (value === null) return reportSyntaxError(this
|
|
273
|
+
} = this._parseLiteral(input));
|
|
274
|
+
if (value === null) return reportSyntaxError(this);
|
|
288
275
|
}
|
|
289
276
|
if (match !== null || matchLength !== 0) {
|
|
290
277
|
type = 'literal';
|
|
@@ -294,15 +281,15 @@ class N3Lexer {
|
|
|
294
281
|
case "'":
|
|
295
282
|
if (!this._lineMode) {
|
|
296
283
|
// Try to find a literal without escape sequences
|
|
297
|
-
if (match =
|
|
284
|
+
if (match = this._simpleApostropheString.exec(input)) value = match[1];
|
|
298
285
|
// Try to find a literal wrapped in three pairs of quotes
|
|
299
286
|
else {
|
|
300
287
|
({
|
|
301
288
|
value,
|
|
302
289
|
matchLength,
|
|
303
290
|
finalLineLength
|
|
304
|
-
} = this._parseLiteral(input
|
|
305
|
-
if (value === null) return reportSyntaxError(this
|
|
291
|
+
} = this._parseLiteral(input));
|
|
292
|
+
if (value === null) return reportSyntaxError(this);
|
|
306
293
|
}
|
|
307
294
|
if (match !== null || matchLength !== 0) {
|
|
308
295
|
type = 'literal';
|
|
@@ -312,7 +299,7 @@ class N3Lexer {
|
|
|
312
299
|
break;
|
|
313
300
|
case '?':
|
|
314
301
|
// Try to find a variable
|
|
315
|
-
if (this._n3Mode && (match =
|
|
302
|
+
if (this._n3Mode && (match = this._variable.exec(input))) type = 'var', value = match[0];
|
|
316
303
|
break;
|
|
317
304
|
case '@':
|
|
318
305
|
// Try to find a language code. A language code can contain dash-separated
|
|
@@ -320,16 +307,15 @@ class N3Lexer {
|
|
|
320
307
|
// input is not finished, another subtag may still arrive in a later chunk and
|
|
321
308
|
// the match would be premature; wait for more input in that case.
|
|
322
309
|
// A double dash starts a direction code, which cannot extend the language code.
|
|
323
|
-
if (this._previousMarker === 'literal' && (match =
|
|
324
|
-
|
|
325
|
-
if (!inputFinished && input[end] === '-' && input[end + 1] !== '-') match = null;else type = 'langcode', value = match[1];
|
|
310
|
+
if (this._previousMarker === 'literal' && (match = this._langcode.exec(input)) && match[1] !== 'version') {
|
|
311
|
+
if (!inputFinished && input[match[0].length] === '-' && input[match[0].length + 1] !== '-') match = null;else type = 'langcode', value = match[1];
|
|
326
312
|
}
|
|
327
313
|
// Try to find a keyword
|
|
328
|
-
else if (match =
|
|
314
|
+
else if (match = this._atKeyword.exec(input)) type = match[0];
|
|
329
315
|
break;
|
|
330
316
|
case '.':
|
|
331
317
|
// Try to find a dot as punctuation
|
|
332
|
-
if (input.length
|
|
318
|
+
if (input.length === 1 ? inputFinished : input[1] < '0' || input[1] > '9') {
|
|
333
319
|
type = '.';
|
|
334
320
|
matchLength = 1;
|
|
335
321
|
break;
|
|
@@ -348,10 +334,10 @@ class N3Lexer {
|
|
|
348
334
|
case '9':
|
|
349
335
|
case '+':
|
|
350
336
|
case '-':
|
|
351
|
-
if (input[
|
|
337
|
+
if (input[1] === '-') {
|
|
352
338
|
// Try to find a direction code
|
|
353
339
|
if (this._previousMarker === 'langcode') {
|
|
354
|
-
if (input.startsWith('--ltr'
|
|
340
|
+
if (input.startsWith('--ltr')) type = 'dircode', value = 'ltr', matchLength = 5;else if (input.startsWith('--rtl')) type = 'dircode', value = 'rtl', matchLength = 5;
|
|
355
341
|
}
|
|
356
342
|
break;
|
|
357
343
|
}
|
|
@@ -359,7 +345,7 @@ class N3Lexer {
|
|
|
359
345
|
// Try to find a number. Since it can contain (but not end with) a dot,
|
|
360
346
|
// we always need a non-dot character before deciding it is a number.
|
|
361
347
|
// Therefore, try inserting a space if we're at the end of the input.
|
|
362
|
-
if (match =
|
|
348
|
+
if (match = this._number.exec(input) || inputFinished && (match = this._number.exec(`${input} `))) {
|
|
363
349
|
type = 'literal', value = match[0];
|
|
364
350
|
prefix = typeof match[1] === 'string' ? xsd.double : typeof match[2] === 'string' ? xsd.decimal : xsd.integer;
|
|
365
351
|
}
|
|
@@ -373,42 +359,42 @@ class N3Lexer {
|
|
|
373
359
|
case 'V':
|
|
374
360
|
case 'v':
|
|
375
361
|
// Try to find a SPARQL-style keyword
|
|
376
|
-
if (match =
|
|
362
|
+
if (match = this._keyword.exec(input)) type = match[0].toUpperCase();else inconclusive = true;
|
|
377
363
|
break;
|
|
378
364
|
case 'f':
|
|
379
365
|
case 't':
|
|
380
366
|
// Try to match a boolean
|
|
381
|
-
if (
|
|
367
|
+
if (this._boolean.test(input)) type = 'literal', value = firstChar === 't' ? 'true' : 'false', prefix = xsd.boolean, matchLength = value.length;else inconclusive = true;
|
|
382
368
|
break;
|
|
383
369
|
case 'a':
|
|
384
370
|
// Try to find an abbreviated predicate
|
|
385
|
-
if (
|
|
371
|
+
if (this._shortPredicates.test(input)) type = 'abbreviation', value = 'a', matchLength = 1;else inconclusive = true;
|
|
386
372
|
break;
|
|
387
373
|
case 'h':
|
|
388
374
|
case 'o':
|
|
389
375
|
// Try to find an N3 verb keyword
|
|
390
|
-
if (this._n3Mode && (match = this._matchN3Verb(input,
|
|
376
|
+
if (this._n3Mode && (match = this._matchN3Verb(input, inputFinished))) type = match[0];else inconclusive = true;
|
|
391
377
|
break;
|
|
392
378
|
case 'i':
|
|
393
379
|
// Try to find an IRI property list identifier or N3 verb keyword
|
|
394
|
-
if (this._n3Mode &&
|
|
380
|
+
if (this._n3Mode && this._n3Id.test(input)) type = 'id', matchLength = 2;else if (this._n3Mode && (match = this._matchN3Verb(input, inputFinished))) type = match[0];else inconclusive = true;
|
|
395
381
|
break;
|
|
396
382
|
case '=':
|
|
397
383
|
// Try to find an implication arrow or equals sign
|
|
398
|
-
if (this._n3Mode && input.length
|
|
384
|
+
if (this._n3Mode && input.length > 1) {
|
|
399
385
|
type = 'abbreviation';
|
|
400
|
-
if (input[
|
|
386
|
+
if (input[1] !== '>') matchLength = 1, value = '=';else matchLength = 2, value = '>';
|
|
401
387
|
}
|
|
402
388
|
break;
|
|
403
389
|
case '!':
|
|
404
390
|
if (!this._n3Mode) break;
|
|
405
391
|
case ')':
|
|
406
|
-
if (!inputFinished && (input.length
|
|
392
|
+
if (!inputFinished && (input.length === 1 || input.length === 2 && input[1] === '>')) {
|
|
407
393
|
// Don't consume yet, as it *could* become a triple term end.
|
|
408
394
|
break;
|
|
409
395
|
}
|
|
410
396
|
// Try to find a triple term
|
|
411
|
-
if (input.length
|
|
397
|
+
if (input.length > 2 && input[1] === '>' && input[2] === '>') {
|
|
412
398
|
type = ')>>', matchLength = 3;
|
|
413
399
|
break;
|
|
414
400
|
}
|
|
@@ -426,15 +412,15 @@ class N3Lexer {
|
|
|
426
412
|
break;
|
|
427
413
|
case '{':
|
|
428
414
|
// We need at least 2 tokens lookahead to distinguish "{|" and "{ "
|
|
429
|
-
if (!this._lineMode && input.length
|
|
415
|
+
if (!this._lineMode && input.length >= 2) {
|
|
430
416
|
// Try to find a quoted triple annotation start
|
|
431
|
-
if (input[
|
|
417
|
+
if (input[1] === '|') type = '{|', matchLength = 2;else type = firstChar, matchLength = 1;
|
|
432
418
|
}
|
|
433
419
|
break;
|
|
434
420
|
case '|':
|
|
435
421
|
// We need 2 tokens lookahead to parse "|}"
|
|
436
422
|
// Try to find a quoted triple annotation end
|
|
437
|
-
if (input.length
|
|
423
|
+
if (input.length >= 2 && input[1] === '}') type = '|}', matchLength = 2;
|
|
438
424
|
break;
|
|
439
425
|
default:
|
|
440
426
|
inconclusive = true;
|
|
@@ -443,11 +429,11 @@ class N3Lexer {
|
|
|
443
429
|
// Some first characters do not allow an immediate decision, so inspect more
|
|
444
430
|
if (inconclusive) {
|
|
445
431
|
// Try to find a prefix
|
|
446
|
-
if ((this._previousMarker === '@prefix' || this._previousMarker === 'PREFIX') && (match =
|
|
432
|
+
if ((this._previousMarker === '@prefix' || this._previousMarker === 'PREFIX') && (match = this._prefix.exec(input))) type = 'prefix', value = match[1] || '';
|
|
447
433
|
// Try to find a prefixed name. Since it can contain (but not end with) a dot,
|
|
448
434
|
// we always need a non-dot character before deciding it is a prefixed name.
|
|
449
435
|
// Therefore, try inserting a space if we're at the end of the input.
|
|
450
|
-
else if ((match =
|
|
436
|
+
else if ((match = this._prefixed.exec(input)) || inputFinished && (match = this._prefixed.exec(`${input} `))) {
|
|
451
437
|
type = 'prefixed', prefix = match[1] || '';
|
|
452
438
|
value = this._unescape(match[2], localNameEscapeReplacements);
|
|
453
439
|
lexicalLength = prefix.length + match[2].length + 1;
|
|
@@ -473,14 +459,16 @@ class N3Lexer {
|
|
|
473
459
|
// We could be in streaming mode, and then we just wait for more input to arrive.
|
|
474
460
|
// Otherwise, a syntax error has occurred in the input.
|
|
475
461
|
// One exception: error on an unaccounted linebreak (= not inside a triple-quoted literal).
|
|
476
|
-
if (inputFinished ||
|
|
462
|
+
if (inputFinished || !/^'''|^"""/.test(input) && /\n|\r/.test(input)) return reportSyntaxError(this);else {
|
|
463
|
+
this._linePosition = currentLineLength - input.length;
|
|
464
|
+
return this._input = input;
|
|
465
|
+
}
|
|
477
466
|
}
|
|
478
467
|
|
|
479
468
|
// Emit the parsed token
|
|
480
469
|
// Consumption includes separator whitespace; lexicalLength excludes it
|
|
481
|
-
// and any synthetic EOF space.
|
|
470
|
+
// and any synthetic EOF space. slice below clamps consumption to the input.
|
|
482
471
|
const length = matchLength || match[0].length;
|
|
483
|
-
const start = currentLineLength - (input.length - pos);
|
|
484
472
|
let token;
|
|
485
473
|
if (finalLineLength) {
|
|
486
474
|
token = {
|
|
@@ -488,22 +476,23 @@ class N3Lexer {
|
|
|
488
476
|
value,
|
|
489
477
|
prefix,
|
|
490
478
|
line,
|
|
491
|
-
start,
|
|
479
|
+
start: currentLineLength - input.length,
|
|
492
480
|
end: finalLineLength,
|
|
493
481
|
endLine: this._line
|
|
494
482
|
};
|
|
495
483
|
callback(null, token);
|
|
496
|
-
} else token = emitToken(type, value, prefix, line,
|
|
484
|
+
} else token = emitToken(type, value, prefix, line, lexicalLength || length);
|
|
497
485
|
this.previousToken = token;
|
|
498
486
|
this._previousMarker = type;
|
|
499
487
|
|
|
500
488
|
// Advance to next part to tokenize
|
|
501
|
-
|
|
502
|
-
if (finalLineLength) currentLineLength = input.length
|
|
489
|
+
input = input.slice(length);
|
|
490
|
+
if (finalLineLength) currentLineLength = input.length + finalLineLength;
|
|
503
491
|
}
|
|
504
492
|
|
|
505
493
|
// Emits a comment at its exact position within matched whitespace.
|
|
506
|
-
function emitComment(value, line,
|
|
494
|
+
function emitComment(value, line, offset) {
|
|
495
|
+
const start = currentLineLength - input.length + offset;
|
|
507
496
|
callback(null, {
|
|
508
497
|
type: 'comment',
|
|
509
498
|
value,
|
|
@@ -514,53 +503,49 @@ class N3Lexer {
|
|
|
514
503
|
});
|
|
515
504
|
}
|
|
516
505
|
// Emits the token through the callback
|
|
517
|
-
function emitToken(type, value, prefix, line,
|
|
506
|
+
function emitToken(type, value, prefix, line, length) {
|
|
507
|
+
const start = input ? currentLineLength - input.length : currentLineLength;
|
|
508
|
+
const end = start + length;
|
|
518
509
|
const token = {
|
|
519
510
|
type,
|
|
520
511
|
value,
|
|
521
512
|
prefix,
|
|
522
513
|
line,
|
|
523
514
|
start,
|
|
524
|
-
end
|
|
515
|
+
end
|
|
525
516
|
};
|
|
526
517
|
callback(null, token);
|
|
527
518
|
return token;
|
|
528
519
|
}
|
|
529
520
|
// Signals the syntax error through the callback
|
|
530
|
-
function reportSyntaxError(self
|
|
531
|
-
callback(self._syntaxError(
|
|
521
|
+
function reportSyntaxError(self) {
|
|
522
|
+
callback(self._syntaxError(/^\S*/.exec(input)[0]));
|
|
532
523
|
}
|
|
533
524
|
}
|
|
534
525
|
|
|
535
|
-
// ### `_suspend` keeps the unconsumed input until more input arrives
|
|
536
|
-
_suspend(input, pos, currentLineLength) {
|
|
537
|
-
this._linePosition = currentLineLength - (input.length - pos);
|
|
538
|
-
return this._input = input.slice(pos);
|
|
539
|
-
}
|
|
540
|
-
|
|
541
526
|
// ### `_matchN3Verb` matches an N3 verb unless the input is a longer prefixed name
|
|
542
|
-
_matchN3Verb(input,
|
|
543
|
-
const verb =
|
|
527
|
+
_matchN3Verb(input, inputFinished) {
|
|
528
|
+
const verb = this._n3Verb.exec(input);
|
|
544
529
|
if (!verb) return null;
|
|
545
530
|
|
|
546
531
|
// Most verb boundaries cannot be part of a prefix, so keep the common path fast.
|
|
547
|
-
const next = input[
|
|
532
|
+
const next = input[verb[0].length];
|
|
548
533
|
if (next !== '-' && next !== '_' && (next < '0' || next > '9')) return verb;
|
|
549
534
|
|
|
550
535
|
// A prefix can start with a verb and continue with characters that are also
|
|
551
536
|
// valid verb boundaries. Prefer the longer prefixed name when it is complete.
|
|
552
|
-
if (
|
|
537
|
+
if (this._prefixed.exec(input)) return null;
|
|
553
538
|
// Appending to the input only matters when a prefixed name could run up to
|
|
554
539
|
// its end, which a character that cannot occur in prefixed names rules out.
|
|
555
540
|
// This avoids copying the rest of the document for every such verb.
|
|
556
|
-
if (
|
|
557
|
-
if (
|
|
541
|
+
if (nonPrefixedNameChar.test(input)) return verb;
|
|
542
|
+
if (this._prefixed.exec(`${input} `)) return null;
|
|
558
543
|
|
|
559
544
|
// If a stream chunk ends partway through such a prefix, wait for the colon
|
|
560
545
|
// instead of prematurely emitting the verb. Appending ": " lets the prefix
|
|
561
546
|
// grammar determine whether all input seen so far can be a complete prefix.
|
|
562
547
|
if (!inputFinished) {
|
|
563
|
-
const prefix =
|
|
548
|
+
const prefix = this._prefix.exec(`${input}: `);
|
|
564
549
|
if (prefix) return null;
|
|
565
550
|
}
|
|
566
551
|
return verb;
|
|
@@ -607,19 +592,19 @@ class N3Lexer {
|
|
|
607
592
|
return result + item.slice(start);
|
|
608
593
|
}
|
|
609
594
|
|
|
610
|
-
// ### `_parseLiteral` parses a literal
|
|
611
|
-
_parseLiteral(input
|
|
595
|
+
// ### `_parseLiteral` parses a literal into an unescaped value
|
|
596
|
+
_parseLiteral(input) {
|
|
612
597
|
// Ensure we have enough lookahead to identify triple-quoted strings
|
|
613
|
-
if (input.length
|
|
598
|
+
if (input.length >= 3) {
|
|
614
599
|
// The caller has already identified a single or double quote.
|
|
615
|
-
const quote = input[
|
|
616
|
-
const openingLength = input[
|
|
600
|
+
const quote = input[0];
|
|
601
|
+
const openingLength = input[1] === quote && input[2] === quote ? 3 : 1;
|
|
617
602
|
let opening = quote;
|
|
618
603
|
if (openingLength === 3) opening = quote === '"' ? '"""' : "'''";
|
|
619
604
|
|
|
620
605
|
// Find the next candidate closing quotes
|
|
621
|
-
let closingPos =
|
|
622
|
-
while ((closingPos = input.indexOf(opening, closingPos)) >
|
|
606
|
+
let closingPos = Math.max(this._literalClosingPos, openingLength);
|
|
607
|
+
while ((closingPos = input.indexOf(opening, closingPos)) > 0) {
|
|
623
608
|
// Count backslashes right before the closing quotes
|
|
624
609
|
let backslashCount = 0;
|
|
625
610
|
while (input[closingPos - backslashCount - 1] === '\\') backslashCount++;
|
|
@@ -628,10 +613,10 @@ class N3Lexer {
|
|
|
628
613
|
// means these are actual, non-escaped closing quotes
|
|
629
614
|
if (backslashCount % 2 === 0) {
|
|
630
615
|
// Extract and unescape the value
|
|
631
|
-
const raw = input.substring(
|
|
616
|
+
const raw = input.substring(openingLength, closingPos),
|
|
632
617
|
lines = raw.split(/\r\n|\r|\n/),
|
|
633
618
|
lineCount = lines.length - 1;
|
|
634
|
-
const matchLength = closingPos
|
|
619
|
+
const matchLength = closingPos + openingLength;
|
|
635
620
|
// Only triple-quoted strings can be multi-line
|
|
636
621
|
if (openingLength === 1 && lineCount !== 0 || openingLength === 3 && this._lineMode) break;
|
|
637
622
|
this._line += lineCount;
|
|
@@ -644,7 +629,7 @@ class N3Lexer {
|
|
|
644
629
|
}
|
|
645
630
|
closingPos++;
|
|
646
631
|
}
|
|
647
|
-
this._literalClosingPos = input.length -
|
|
632
|
+
this._literalClosingPos = input.length - openingLength + 1;
|
|
648
633
|
}
|
|
649
634
|
return {
|
|
650
635
|
value: '',
|
package/lib/N3Parser.js
CHANGED
|
@@ -44,10 +44,11 @@ class N3Parser {
|
|
|
44
44
|
// Whether the log:isImpliedBy predicate is supported
|
|
45
45
|
this._isImpliedBy = options.isImpliedBy;
|
|
46
46
|
// Whether an undeclared empty prefix resolves against the document IRI
|
|
47
|
-
|
|
47
|
+
// (enabled unless explicitly disabled)
|
|
48
|
+
this._implicitEmptyPrefix = options.implicitEmptyPrefix !== false;
|
|
48
49
|
// Whether an empty formula is read as the boolean literal true,
|
|
49
|
-
// as in the N3 spec tests (
|
|
50
|
-
this._emptyFormulaAsTrue =
|
|
50
|
+
// as in the N3 spec tests (enabled unless explicitly disabled)
|
|
51
|
+
this._emptyFormulaAsTrue = options.emptyFormulaAsTrue !== false;
|
|
51
52
|
// Disable relative IRIs in N-Triples or N-Quads mode
|
|
52
53
|
if (isLineMode) this._resolveRelativeIRI = iri => {
|
|
53
54
|
return null;
|
|
@@ -872,7 +873,7 @@ class N3Parser {
|
|
|
872
873
|
// Restore the parent context containing this formula
|
|
873
874
|
this._restoreContext('formula', token);
|
|
874
875
|
|
|
875
|
-
//
|
|
876
|
+
// Unless the emptyFormulaAsTrue option is false, an empty formula
|
|
876
877
|
// is read as the boolean literal true, following the N3 spec tests
|
|
877
878
|
// and the direction discussed in https://github.com/w3c-cg/N3/issues/185
|
|
878
879
|
if (empty && this._emptyFormulaAsTrue) {
|