n3 2.11.2 → 3.0.0-alpha.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +10 -9
- package/browser/n3.esm.min.js +12 -12
- package/browser/n3.min.js +12 -12
- package/lib/N3Lexer.js +133 -148
- package/lib/N3Parser.js +5 -4
- package/package.json +29 -2
- package/src/N3Lexer.js +142 -162
- package/src/N3Parser.js +5 -4
package/src/N3Lexer.js
CHANGED
|
@@ -18,8 +18,7 @@ const localNameEscapeReplacements = {
|
|
|
18
18
|
};
|
|
19
19
|
const illegalIriChars = /[\x00-\x20<>\\"\{\}\|\^\`]/;
|
|
20
20
|
// Characters that cannot occur in a prefixed name, not even escaped
|
|
21
|
-
|
|
22
|
-
const nonPrefixedNameChar = /[\s<>"{}|^`]/g;
|
|
21
|
+
const nonPrefixedNameChar = /[\s<>"{}|^`]/;
|
|
23
22
|
|
|
24
23
|
// A valid code point is a Unicode scalar value: at most U+10FFFF and not a surrogate
|
|
25
24
|
function isValidCodePoint(charCode) {
|
|
@@ -36,48 +35,30 @@ const lineModeRegExps = {
|
|
|
36
35
|
_whitespace: true,
|
|
37
36
|
};
|
|
38
37
|
const invalidRegExp = /$0^/;
|
|
39
|
-
const nonWhitespace = /\S*/y;
|
|
40
|
-
|
|
41
|
-
// Matches a sticky regular expression at the given position of the input
|
|
42
|
-
function execAt(regExp, input, pos) {
|
|
43
|
-
regExp.lastIndex = pos;
|
|
44
|
-
return regExp.exec(input);
|
|
45
|
-
}
|
|
46
|
-
function testAt(regExp, input, pos) {
|
|
47
|
-
regExp.lastIndex = pos;
|
|
48
|
-
return regExp.test(input);
|
|
49
|
-
}
|
|
50
|
-
// Matches the rest of the input followed by a space, as at the end of the input,
|
|
51
|
-
// a token that can contain (but not end with) a dot needs a non-dot character after it
|
|
52
|
-
function execAtEnd(regExp, input, pos) {
|
|
53
|
-
regExp.lastIndex = 0;
|
|
54
|
-
return regExp.exec(`${input.slice(pos)} `);
|
|
55
|
-
}
|
|
56
38
|
|
|
57
39
|
// ## Constructor
|
|
58
40
|
export default class N3Lexer {
|
|
59
41
|
constructor(options) {
|
|
60
42
|
// ## Regular expressions
|
|
61
|
-
// It's slightly faster to have these as properties than as in-scope variables
|
|
62
|
-
|
|
63
|
-
this.
|
|
64
|
-
this.
|
|
65
|
-
this.
|
|
66
|
-
this.
|
|
67
|
-
this.
|
|
68
|
-
this.
|
|
69
|
-
this.
|
|
70
|
-
this.
|
|
71
|
-
this.
|
|
72
|
-
this.
|
|
73
|
-
this.
|
|
74
|
-
this.
|
|
75
|
-
this.
|
|
76
|
-
this.
|
|
77
|
-
this.
|
|
78
|
-
this.
|
|
79
|
-
this.
|
|
80
|
-
this._whitespace = /[ \t]+/y;
|
|
43
|
+
// It's slightly faster to have these as properties than as in-scope variables
|
|
44
|
+
this._iri = /^<((?:[^ <>{}\\]|\\[uU])+)>[ \t]*/; // IRI with escape sequences; needs sanity check after unescaping
|
|
45
|
+
this._unescapedIri = /^<([^\x00-\x20<>\\"\{\}\|\^\`]*)>[ \t]*/; // IRI without escape sequences; no unescaping
|
|
46
|
+
this._simpleQuotedString = /^"([^"\\\r\n]*)"(?=[^"])/; // string without escape sequences
|
|
47
|
+
this._simpleApostropheString = /^'([^'\\\r\n]*)'(?=[^'])/;
|
|
48
|
+
this._langcode = /^@([a-z]+(?:-[a-z0-9]+)*)(?=[^a-z0-9])/i;
|
|
49
|
+
this._prefix = /^((?:[A-Za-z\xc0-\xd6\xd8-\xf6\xf8-\u02ff\u0370-\u037d\u037f-\u1fff\u200c\u200d\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff])(?:\.?[\-0-9A-Z_a-z\xb7\xc0-\xd6\xd8-\xf6\xf8-\u037d\u037f-\u1fff\u200c\u200d\u203f\u2040\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff])*)?:(?=[#\s<])/;
|
|
50
|
+
this._prefixed = /^((?:[A-Za-z\xc0-\xd6\xd8-\xf6\xf8-\u02ff\u0370-\u037d\u037f-\u1fff\u200c\u200d\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff])(?:\.?[\-0-9A-Z_a-z\xb7\xc0-\xd6\xd8-\xf6\xf8-\u037d\u037f-\u1fff\u200c\u200d\u203f\u2040\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff])*)?:((?:(?:[0-9:A-Z_a-z\xc0-\xd6\xd8-\xf6\xf8-\u02ff\u0370-\u037d\u037f-\u1fff\u200c\u200d\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff]|%[0-9a-fA-F]{2}|\\[!#-\/;=?\-@_~])(?:(?:[\.\-0-9:A-Z_a-z\xb7\xc0-\xd6\xd8-\xf6\xf8-\u037d\u037f-\u1fff\u200c\u200d\u203f\u2040\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff]|%[0-9a-fA-F]{2}|\\[!#-\/;=?\-@_~])*(?:[\-0-9:A-Z_a-z\xb7\xc0-\xd6\xd8-\xf6\xf8-\u037d\u037f-\u1fff\u200c\u200d\u203f\u2040\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff]|%[0-9a-fA-F]{2}|\\[!#-\/;=?\-@_~]))?)?)(?:[ \t]+|(?=\.?[,;!\^\s#()\[\]\{\}"'<>]))/;
|
|
51
|
+
this._variable = /^\?(?:(?:[A-Z_a-z\xc0-\xd6\xd8-\xf6\xf8-\u02ff\u0370-\u037d\u037f-\u1fff\u200c\u200d\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff])(?:[\-0-9:A-Z_a-z\xb7\xc0-\xd6\xd8-\xf6\xf8-\u037d\u037f-\u1fff\u200c\u200d\u203f\u2040\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff])*)(?=[.,;!\^\s#()\[\]\{\}"'<>])/;
|
|
52
|
+
this._blank = /^_:((?:[0-9A-Z_a-z\xc0-\xd6\xd8-\xf6\xf8-\u02ff\u0370-\u037d\u037f-\u1fff\u200c\u200d\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff])(?:\.?[\-0-9A-Z_a-z\xb7\xc0-\xd6\xd8-\xf6\xf8-\u037d\u037f-\u1fff\u200c\u200d\u203f\u2040\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff])*)(?:[ \t]+|(?=\.?[,;:!\^\s#()\[\]\{\}"'<>]))/;
|
|
53
|
+
this._number = /^[\-+]?(?:(\d+\.\d*|\.?\d+)[eE][\-+]?\d+|(?=\.?\d)\d*(?:(\.)\d+)?)(?=\.?[,;:!\^\s#()\[\]\{\}"'<>])/;
|
|
54
|
+
this._boolean = /^(?:true|false)(?=[.,;!\^\s#()\[\]\{\}"'<>])/;
|
|
55
|
+
this._atKeyword = /^@[a-z]+(?=[\s#<:])/i;
|
|
56
|
+
this._keyword = /^(?:PREFIX|BASE|VERSION|GRAPH)(?=[\s#<])/i;
|
|
57
|
+
this._n3Verb = /^(?:has|is|of)(?=[\s#()\[\]\{\}"'<>?_+\-0-9])/;
|
|
58
|
+
this._n3Id = /^id(?=[\s#<])/;
|
|
59
|
+
this._shortPredicates = /^a(?=[\s#()\[\]\{\}"'<>])/;
|
|
60
|
+
this._commentLine = /^[ \t]*#([^\n\r]*)(?:\r\n|\n|\r)([ \t]*)/;
|
|
61
|
+
this._whitespace = /^[ \t]+/;
|
|
81
62
|
options = options || {};
|
|
82
63
|
|
|
83
64
|
// Whether the log:isImpliedBy predicate is supported
|
|
@@ -106,94 +87,98 @@ export default class N3Lexer {
|
|
|
106
87
|
|
|
107
88
|
// ### `_tokenizeToEnd` tokenizes as for as possible, emitting tokens through the callback
|
|
108
89
|
_tokenizeToEnd(callback, inputFinished) {
|
|
109
|
-
// Continue parsing as far as possible; the loop will return eventually
|
|
110
|
-
|
|
111
|
-
// the regular expressions are sticky, so they match at that position.
|
|
112
|
-
const input = this._input;
|
|
113
|
-
let pos = 0;
|
|
90
|
+
// Continue parsing as far as possible; the loop will return eventually
|
|
91
|
+
let input = this._input;
|
|
114
92
|
let currentLineLength = this._linePosition + input.length;
|
|
115
93
|
while (true) {
|
|
116
94
|
// Consume one separator line at a time, including its following indentation.
|
|
117
95
|
while (true) {
|
|
118
|
-
let charCode = input.charCodeAt(
|
|
96
|
+
let charCode = input.charCodeAt(0), separatorLength = 0;
|
|
119
97
|
if (charCode === SPACE || charCode === TAB) {
|
|
120
|
-
const next = input.charCodeAt(
|
|
98
|
+
const next = input.charCodeAt(1);
|
|
121
99
|
separatorLength = next === SPACE || next === TAB ?
|
|
122
|
-
|
|
123
|
-
charCode = input.charCodeAt(
|
|
100
|
+
this._whitespace.exec(input)[0].length : 1;
|
|
101
|
+
charCode = input.charCodeAt(separatorLength);
|
|
124
102
|
}
|
|
125
103
|
if (charCode === HASH) {
|
|
126
|
-
const comment =
|
|
104
|
+
const comment = this._commentLine.exec(input);
|
|
127
105
|
if (comment) {
|
|
128
106
|
const commentLength = comment[0].length;
|
|
129
107
|
// Keep a trailing CR buffered in case the next chunk starts with LF.
|
|
130
|
-
if (!inputFinished &&
|
|
131
|
-
input.charCodeAt(
|
|
132
|
-
|
|
108
|
+
if (!inputFinished && commentLength === input.length &&
|
|
109
|
+
input.charCodeAt(commentLength - 1) === CR) {
|
|
110
|
+
this._linePosition = currentLineLength - input.length;
|
|
111
|
+
return this._input = input;
|
|
112
|
+
}
|
|
133
113
|
if (this.comments)
|
|
134
|
-
emitComment(comment[1], this._line,
|
|
135
|
-
|
|
136
|
-
currentLineLength = input.length
|
|
114
|
+
emitComment(comment[1], this._line, separatorLength);
|
|
115
|
+
input = input.slice(commentLength);
|
|
116
|
+
currentLineLength = input.length + comment[2].length;
|
|
137
117
|
this._line++;
|
|
138
118
|
}
|
|
139
119
|
else {
|
|
140
120
|
// A comment without a line ending stays buffered until EOF.
|
|
141
|
-
|
|
142
|
-
if (!inputFinished)
|
|
143
|
-
|
|
121
|
+
input = input.slice(separatorLength);
|
|
122
|
+
if (!inputFinished) {
|
|
123
|
+
this._linePosition = currentLineLength - input.length;
|
|
124
|
+
return this._input = input;
|
|
125
|
+
}
|
|
144
126
|
if (this.comments)
|
|
145
|
-
emitComment(input.slice(
|
|
146
|
-
|
|
127
|
+
emitComment(input.slice(1), this._line, 0);
|
|
128
|
+
input = '';
|
|
147
129
|
break;
|
|
148
130
|
}
|
|
149
131
|
}
|
|
150
132
|
else if (charCode === LF || charCode === CR) {
|
|
151
133
|
// A CR at the end of a chunk may still be followed by LF.
|
|
152
|
-
if (!inputFinished && charCode === CR &&
|
|
153
|
-
|
|
154
|
-
|
|
134
|
+
if (!inputFinished && charCode === CR && separatorLength + 1 === input.length) {
|
|
135
|
+
this._linePosition = currentLineLength - input.length;
|
|
136
|
+
return this._input = input;
|
|
137
|
+
}
|
|
138
|
+
separatorLength += charCode === CR && input.charCodeAt(separatorLength + 1) === LF ? 2 : 1;
|
|
155
139
|
// Indentation is consumed with the newline, but belongs to the next line's columns.
|
|
156
140
|
let indentationLength = 0;
|
|
157
|
-
const next = input.charCodeAt(
|
|
141
|
+
const next = input.charCodeAt(separatorLength);
|
|
158
142
|
if (next === SPACE || next === TAB) {
|
|
159
|
-
const following = input.charCodeAt(
|
|
143
|
+
const following = input.charCodeAt(separatorLength + 1);
|
|
160
144
|
indentationLength = following === SPACE || following === TAB ?
|
|
161
|
-
|
|
145
|
+
this._whitespace.exec(input.slice(separatorLength))[0].length : 1;
|
|
162
146
|
}
|
|
163
|
-
|
|
164
|
-
currentLineLength = input.length
|
|
147
|
+
input = input.slice(separatorLength + indentationLength);
|
|
148
|
+
currentLineLength = input.length + indentationLength;
|
|
165
149
|
this._line++;
|
|
166
150
|
}
|
|
167
151
|
else {
|
|
168
|
-
|
|
152
|
+
if (separatorLength !== 0)
|
|
153
|
+
input = input.slice(separatorLength);
|
|
169
154
|
break;
|
|
170
155
|
}
|
|
171
156
|
}
|
|
172
|
-
if (
|
|
173
|
-
this._linePosition = currentLineLength;
|
|
157
|
+
if (input.length === 0) {
|
|
174
158
|
if (inputFinished) {
|
|
175
|
-
|
|
176
|
-
|
|
159
|
+
input = null;
|
|
160
|
+
emitToken('eof', '', '', this._line, 0);
|
|
177
161
|
}
|
|
178
|
-
|
|
162
|
+
this._linePosition = currentLineLength;
|
|
163
|
+
return this._input = input;
|
|
179
164
|
}
|
|
180
165
|
|
|
181
166
|
// Look for specific token types based on the first character
|
|
182
|
-
const line = this._line, firstChar = input[
|
|
167
|
+
const line = this._line, firstChar = input[0];
|
|
183
168
|
let type = '', value = '', prefix = '',
|
|
184
169
|
match = null, matchLength = 0, lexicalLength = 0,
|
|
185
170
|
finalLineLength = 0, inconclusive = false;
|
|
186
171
|
switch (firstChar) {
|
|
187
172
|
case '^':
|
|
188
173
|
// We need at least 3 tokens lookahead to distinguish ^^<IRI> and ^^pre:fixed
|
|
189
|
-
if (input.length
|
|
174
|
+
if (input.length < 3)
|
|
190
175
|
break;
|
|
191
176
|
// Try to match a type
|
|
192
|
-
else if (input[
|
|
177
|
+
else if (input[1] === '^') {
|
|
193
178
|
this._previousMarker = '^^';
|
|
194
179
|
// Move to type IRI or prefixed name
|
|
195
|
-
|
|
196
|
-
if (input[
|
|
180
|
+
input = input.slice(2);
|
|
181
|
+
if (input[0] !== '<') {
|
|
197
182
|
inconclusive = true;
|
|
198
183
|
break;
|
|
199
184
|
}
|
|
@@ -209,38 +194,38 @@ export default class N3Lexer {
|
|
|
209
194
|
// Fall through in case the type is an IRI
|
|
210
195
|
case '<':
|
|
211
196
|
// Try to find a full IRI without escape sequences
|
|
212
|
-
if (match =
|
|
197
|
+
if (match = this._unescapedIri.exec(input)) {
|
|
213
198
|
type = 'IRI', value = match[1];
|
|
214
199
|
lexicalLength = match[1].length + 2;
|
|
215
200
|
}
|
|
216
201
|
// Try to find a full IRI with escape sequences
|
|
217
|
-
else if (match =
|
|
202
|
+
else if (match = this._iri.exec(input)) {
|
|
218
203
|
value = this._unescape(match[1], stringEscapeReplacements);
|
|
219
204
|
if (value === null || illegalIriChars.test(value))
|
|
220
|
-
return reportSyntaxError(this
|
|
205
|
+
return reportSyntaxError(this);
|
|
221
206
|
type = 'IRI';
|
|
222
207
|
lexicalLength = match[1].length + 2;
|
|
223
208
|
}
|
|
224
209
|
// Try to find a triple term
|
|
225
|
-
else if (input.length
|
|
210
|
+
else if (input.length > 2 && input[1] === '<' && input[2] === '(')
|
|
226
211
|
type = '<<(', matchLength = 3;
|
|
227
212
|
// Try to find a reified triple
|
|
228
|
-
else if (!this._lineMode && input.length
|
|
213
|
+
else if (!this._lineMode && input.length > (inputFinished ? 1 : 2) && input[1] === '<')
|
|
229
214
|
type = '<<', matchLength = 2;
|
|
230
215
|
// Try to find a backwards implication arrow
|
|
231
|
-
else if (this._n3Mode && input.length
|
|
216
|
+
else if (this._n3Mode && input.length > 1 && input[1] === '=') {
|
|
232
217
|
matchLength = 2;
|
|
233
218
|
if (this._isImpliedBy) type = 'abbreviation', value = '<';
|
|
234
219
|
else type = 'inverse', value = '>';
|
|
235
220
|
}
|
|
236
221
|
// Try to find an inverted predicate marker
|
|
237
|
-
else if (this._n3Mode && input.length
|
|
222
|
+
else if (this._n3Mode && input.length > 1 && input[1] === '-')
|
|
238
223
|
type = 'inversePredicate', matchLength = 2;
|
|
239
224
|
break;
|
|
240
225
|
|
|
241
226
|
case '>':
|
|
242
227
|
// Try to find a reified triple
|
|
243
|
-
if (input.length
|
|
228
|
+
if (input.length > 1 && input[1] === '>')
|
|
244
229
|
type = '>>', matchLength = 2;
|
|
245
230
|
break;
|
|
246
231
|
|
|
@@ -248,8 +233,8 @@ export default class N3Lexer {
|
|
|
248
233
|
// Try to find a blank node. Since it can contain (but not end with) a dot,
|
|
249
234
|
// we always need a non-dot character before deciding it is a blank node.
|
|
250
235
|
// Therefore, try inserting a space if we're at the end of the input.
|
|
251
|
-
if ((match =
|
|
252
|
-
inputFinished && (match =
|
|
236
|
+
if ((match = this._blank.exec(input)) ||
|
|
237
|
+
inputFinished && (match = this._blank.exec(`${input} `))) {
|
|
253
238
|
type = 'blank', prefix = '_', value = match[1];
|
|
254
239
|
lexicalLength = match[1].length + 2;
|
|
255
240
|
}
|
|
@@ -257,13 +242,13 @@ export default class N3Lexer {
|
|
|
257
242
|
|
|
258
243
|
case '"':
|
|
259
244
|
// Try to find a literal without escape sequences
|
|
260
|
-
if (match =
|
|
245
|
+
if (match = this._simpleQuotedString.exec(input))
|
|
261
246
|
value = match[1];
|
|
262
247
|
// Try to find a literal wrapped in three pairs of quotes
|
|
263
248
|
else {
|
|
264
|
-
({ value, matchLength, finalLineLength } = this._parseLiteral(input
|
|
249
|
+
({ value, matchLength, finalLineLength } = this._parseLiteral(input));
|
|
265
250
|
if (value === null)
|
|
266
|
-
return reportSyntaxError(this
|
|
251
|
+
return reportSyntaxError(this);
|
|
267
252
|
}
|
|
268
253
|
if (match !== null || matchLength !== 0) {
|
|
269
254
|
type = 'literal';
|
|
@@ -274,13 +259,13 @@ export default class N3Lexer {
|
|
|
274
259
|
case "'":
|
|
275
260
|
if (!this._lineMode) {
|
|
276
261
|
// Try to find a literal without escape sequences
|
|
277
|
-
if (match =
|
|
262
|
+
if (match = this._simpleApostropheString.exec(input))
|
|
278
263
|
value = match[1];
|
|
279
264
|
// Try to find a literal wrapped in three pairs of quotes
|
|
280
265
|
else {
|
|
281
|
-
({ value, matchLength, finalLineLength } = this._parseLiteral(input
|
|
266
|
+
({ value, matchLength, finalLineLength } = this._parseLiteral(input));
|
|
282
267
|
if (value === null)
|
|
283
|
-
return reportSyntaxError(this
|
|
268
|
+
return reportSyntaxError(this);
|
|
284
269
|
}
|
|
285
270
|
if (match !== null || matchLength !== 0) {
|
|
286
271
|
type = 'literal';
|
|
@@ -291,7 +276,7 @@ export default class N3Lexer {
|
|
|
291
276
|
|
|
292
277
|
case '?':
|
|
293
278
|
// Try to find a variable
|
|
294
|
-
if (this._n3Mode && (match =
|
|
279
|
+
if (this._n3Mode && (match = this._variable.exec(input)))
|
|
295
280
|
type = 'var', value = match[0];
|
|
296
281
|
break;
|
|
297
282
|
|
|
@@ -301,21 +286,20 @@ export default class N3Lexer {
|
|
|
301
286
|
// input is not finished, another subtag may still arrive in a later chunk and
|
|
302
287
|
// the match would be premature; wait for more input in that case.
|
|
303
288
|
// A double dash starts a direction code, which cannot extend the language code.
|
|
304
|
-
if (this._previousMarker === 'literal' && (match =
|
|
305
|
-
|
|
306
|
-
if (!inputFinished && input[end] === '-' && input[end + 1] !== '-')
|
|
289
|
+
if (this._previousMarker === 'literal' && (match = this._langcode.exec(input)) && match[1] !== 'version') {
|
|
290
|
+
if (!inputFinished && input[match[0].length] === '-' && input[match[0].length + 1] !== '-')
|
|
307
291
|
match = null;
|
|
308
292
|
else
|
|
309
293
|
type = 'langcode', value = match[1];
|
|
310
294
|
}
|
|
311
295
|
// Try to find a keyword
|
|
312
|
-
else if (match =
|
|
296
|
+
else if (match = this._atKeyword.exec(input))
|
|
313
297
|
type = match[0];
|
|
314
298
|
break;
|
|
315
299
|
|
|
316
300
|
case '.':
|
|
317
301
|
// Try to find a dot as punctuation
|
|
318
|
-
if (input.length
|
|
302
|
+
if (input.length === 1 ? inputFinished : (input[1] < '0' || input[1] > '9')) {
|
|
319
303
|
type = '.';
|
|
320
304
|
matchLength = 1;
|
|
321
305
|
break;
|
|
@@ -334,12 +318,12 @@ export default class N3Lexer {
|
|
|
334
318
|
case '9':
|
|
335
319
|
case '+':
|
|
336
320
|
case '-':
|
|
337
|
-
if (input[
|
|
321
|
+
if (input[1] === '-') {
|
|
338
322
|
// Try to find a direction code
|
|
339
323
|
if (this._previousMarker === 'langcode') {
|
|
340
|
-
if (input.startsWith('--ltr'
|
|
324
|
+
if (input.startsWith('--ltr'))
|
|
341
325
|
type = 'dircode', value = 'ltr', matchLength = 5;
|
|
342
|
-
else if (input.startsWith('--rtl'
|
|
326
|
+
else if (input.startsWith('--rtl'))
|
|
343
327
|
type = 'dircode', value = 'rtl', matchLength = 5;
|
|
344
328
|
}
|
|
345
329
|
break;
|
|
@@ -348,8 +332,8 @@ export default class N3Lexer {
|
|
|
348
332
|
// Try to find a number. Since it can contain (but not end with) a dot,
|
|
349
333
|
// we always need a non-dot character before deciding it is a number.
|
|
350
334
|
// Therefore, try inserting a space if we're at the end of the input.
|
|
351
|
-
if (match =
|
|
352
|
-
inputFinished && (match =
|
|
335
|
+
if (match = this._number.exec(input) ||
|
|
336
|
+
inputFinished && (match = this._number.exec(`${input} `))) {
|
|
353
337
|
type = 'literal', value = match[0];
|
|
354
338
|
prefix = (typeof match[1] === 'string' ? xsd.double :
|
|
355
339
|
(typeof match[2] === 'string' ? xsd.decimal : xsd.integer));
|
|
@@ -365,7 +349,7 @@ export default class N3Lexer {
|
|
|
365
349
|
case 'V':
|
|
366
350
|
case 'v':
|
|
367
351
|
// Try to find a SPARQL-style keyword
|
|
368
|
-
if (match =
|
|
352
|
+
if (match = this._keyword.exec(input))
|
|
369
353
|
type = match[0].toUpperCase();
|
|
370
354
|
else
|
|
371
355
|
inconclusive = true;
|
|
@@ -374,7 +358,7 @@ export default class N3Lexer {
|
|
|
374
358
|
case 'f':
|
|
375
359
|
case 't':
|
|
376
360
|
// Try to match a boolean
|
|
377
|
-
if (
|
|
361
|
+
if (this._boolean.test(input))
|
|
378
362
|
type = 'literal', value = firstChar === 't' ? 'true' : 'false', prefix = xsd.boolean, matchLength = value.length;
|
|
379
363
|
else
|
|
380
364
|
inconclusive = true;
|
|
@@ -382,7 +366,7 @@ export default class N3Lexer {
|
|
|
382
366
|
|
|
383
367
|
case 'a':
|
|
384
368
|
// Try to find an abbreviated predicate
|
|
385
|
-
if (
|
|
369
|
+
if (this._shortPredicates.test(input))
|
|
386
370
|
type = 'abbreviation', value = 'a', matchLength = 1;
|
|
387
371
|
else
|
|
388
372
|
inconclusive = true;
|
|
@@ -391,7 +375,7 @@ export default class N3Lexer {
|
|
|
391
375
|
case 'h':
|
|
392
376
|
case 'o':
|
|
393
377
|
// Try to find an N3 verb keyword
|
|
394
|
-
if (this._n3Mode && (match = this._matchN3Verb(input,
|
|
378
|
+
if (this._n3Mode && (match = this._matchN3Verb(input, inputFinished)))
|
|
395
379
|
type = match[0];
|
|
396
380
|
else
|
|
397
381
|
inconclusive = true;
|
|
@@ -399,9 +383,9 @@ export default class N3Lexer {
|
|
|
399
383
|
|
|
400
384
|
case 'i':
|
|
401
385
|
// Try to find an IRI property list identifier or N3 verb keyword
|
|
402
|
-
if (this._n3Mode &&
|
|
386
|
+
if (this._n3Mode && this._n3Id.test(input))
|
|
403
387
|
type = 'id', matchLength = 2;
|
|
404
|
-
else if (this._n3Mode && (match = this._matchN3Verb(input,
|
|
388
|
+
else if (this._n3Mode && (match = this._matchN3Verb(input, inputFinished)))
|
|
405
389
|
type = match[0];
|
|
406
390
|
else
|
|
407
391
|
inconclusive = true;
|
|
@@ -409,9 +393,9 @@ export default class N3Lexer {
|
|
|
409
393
|
|
|
410
394
|
case '=':
|
|
411
395
|
// Try to find an implication arrow or equals sign
|
|
412
|
-
if (this._n3Mode && input.length
|
|
396
|
+
if (this._n3Mode && input.length > 1) {
|
|
413
397
|
type = 'abbreviation';
|
|
414
|
-
if (input[
|
|
398
|
+
if (input[1] !== '>')
|
|
415
399
|
matchLength = 1, value = '=';
|
|
416
400
|
else
|
|
417
401
|
matchLength = 2, value = '>';
|
|
@@ -422,12 +406,12 @@ export default class N3Lexer {
|
|
|
422
406
|
if (!this._n3Mode)
|
|
423
407
|
break;
|
|
424
408
|
case ')':
|
|
425
|
-
if (!inputFinished && (input.length
|
|
409
|
+
if (!inputFinished && (input.length === 1 || (input.length === 2 && input[1] === '>'))) {
|
|
426
410
|
// Don't consume yet, as it *could* become a triple term end.
|
|
427
411
|
break;
|
|
428
412
|
}
|
|
429
413
|
// Try to find a triple term
|
|
430
|
-
if (input.length
|
|
414
|
+
if (input.length > 2 && input[1] === '>' && input[2] === '>') {
|
|
431
415
|
type = ')>>', matchLength = 3;
|
|
432
416
|
break;
|
|
433
417
|
}
|
|
@@ -445,9 +429,9 @@ export default class N3Lexer {
|
|
|
445
429
|
break;
|
|
446
430
|
case '{':
|
|
447
431
|
// We need at least 2 tokens lookahead to distinguish "{|" and "{ "
|
|
448
|
-
if (!this._lineMode && input.length
|
|
432
|
+
if (!this._lineMode && input.length >= 2) {
|
|
449
433
|
// Try to find a quoted triple annotation start
|
|
450
|
-
if (input[
|
|
434
|
+
if (input[1] === '|')
|
|
451
435
|
type = '{|', matchLength = 2;
|
|
452
436
|
else
|
|
453
437
|
type = firstChar, matchLength = 1;
|
|
@@ -456,7 +440,7 @@ export default class N3Lexer {
|
|
|
456
440
|
case '|':
|
|
457
441
|
// We need 2 tokens lookahead to parse "|}"
|
|
458
442
|
// Try to find a quoted triple annotation end
|
|
459
|
-
if (input.length
|
|
443
|
+
if (input.length >= 2 && input[1] === '}')
|
|
460
444
|
type = '|}', matchLength = 2;
|
|
461
445
|
break;
|
|
462
446
|
|
|
@@ -468,13 +452,13 @@ export default class N3Lexer {
|
|
|
468
452
|
if (inconclusive) {
|
|
469
453
|
// Try to find a prefix
|
|
470
454
|
if ((this._previousMarker === '@prefix' || this._previousMarker === 'PREFIX') &&
|
|
471
|
-
(match =
|
|
455
|
+
(match = this._prefix.exec(input)))
|
|
472
456
|
type = 'prefix', value = match[1] || '';
|
|
473
457
|
// Try to find a prefixed name. Since it can contain (but not end with) a dot,
|
|
474
458
|
// we always need a non-dot character before deciding it is a prefixed name.
|
|
475
459
|
// Therefore, try inserting a space if we're at the end of the input.
|
|
476
|
-
else if ((match =
|
|
477
|
-
inputFinished && (match =
|
|
460
|
+
else if ((match = this._prefixed.exec(input)) ||
|
|
461
|
+
inputFinished && (match = this._prefixed.exec(`${input} `))) {
|
|
478
462
|
type = 'prefixed', prefix = match[1] || '';
|
|
479
463
|
value = this._unescape(match[2], localNameEscapeReplacements);
|
|
480
464
|
lexicalLength = prefix.length + match[2].length + 1;
|
|
@@ -495,90 +479,86 @@ export default class N3Lexer {
|
|
|
495
479
|
// We could be in streaming mode, and then we just wait for more input to arrive.
|
|
496
480
|
// Otherwise, a syntax error has occurred in the input.
|
|
497
481
|
// One exception: error on an unaccounted linebreak (= not inside a triple-quoted literal).
|
|
498
|
-
if (inputFinished || (
|
|
499
|
-
|
|
500
|
-
|
|
501
|
-
|
|
502
|
-
return this.
|
|
482
|
+
if (inputFinished || (!/^'''|^"""/.test(input) && /\n|\r/.test(input)))
|
|
483
|
+
return reportSyntaxError(this);
|
|
484
|
+
else {
|
|
485
|
+
this._linePosition = currentLineLength - input.length;
|
|
486
|
+
return this._input = input;
|
|
487
|
+
}
|
|
503
488
|
}
|
|
504
489
|
|
|
505
490
|
// Emit the parsed token
|
|
506
491
|
// Consumption includes separator whitespace; lexicalLength excludes it
|
|
507
|
-
// and any synthetic EOF space.
|
|
492
|
+
// and any synthetic EOF space. slice below clamps consumption to the input.
|
|
508
493
|
const length = matchLength || match[0].length;
|
|
509
|
-
const start = currentLineLength - (input.length - pos);
|
|
510
494
|
let token;
|
|
511
495
|
if (finalLineLength) {
|
|
512
496
|
token = {
|
|
513
|
-
type, value, prefix, line,
|
|
497
|
+
type, value, prefix, line,
|
|
498
|
+
start: currentLineLength - input.length,
|
|
514
499
|
end: finalLineLength, endLine: this._line,
|
|
515
500
|
};
|
|
516
501
|
callback(null, token);
|
|
517
502
|
}
|
|
518
503
|
else
|
|
519
|
-
token = emitToken(type, value, prefix, line,
|
|
504
|
+
token = emitToken(type, value, prefix, line, lexicalLength || length);
|
|
520
505
|
this.previousToken = token;
|
|
521
506
|
this._previousMarker = type;
|
|
522
507
|
|
|
523
508
|
// Advance to next part to tokenize
|
|
524
|
-
|
|
509
|
+
input = input.slice(length);
|
|
525
510
|
if (finalLineLength)
|
|
526
|
-
currentLineLength = input.length
|
|
511
|
+
currentLineLength = input.length + finalLineLength;
|
|
527
512
|
}
|
|
528
513
|
|
|
529
514
|
// Emits a comment at its exact position within matched whitespace.
|
|
530
|
-
function emitComment(value, line,
|
|
515
|
+
function emitComment(value, line, offset) {
|
|
516
|
+
const start = currentLineLength - input.length + offset;
|
|
531
517
|
callback(null, {
|
|
532
518
|
type: 'comment', value, prefix: '', line,
|
|
533
519
|
start, end: start + value.length + 1,
|
|
534
520
|
});
|
|
535
521
|
}
|
|
536
522
|
// Emits the token through the callback
|
|
537
|
-
function emitToken(type, value, prefix, line,
|
|
538
|
-
const
|
|
523
|
+
function emitToken(type, value, prefix, line, length) {
|
|
524
|
+
const start = input ? currentLineLength - input.length : currentLineLength;
|
|
525
|
+
const end = start + length;
|
|
526
|
+
const token = { type, value, prefix, line, start, end };
|
|
539
527
|
callback(null, token);
|
|
540
528
|
return token;
|
|
541
529
|
}
|
|
542
530
|
// Signals the syntax error through the callback
|
|
543
|
-
function reportSyntaxError(self
|
|
544
|
-
callback(self._syntaxError(execAt(nonWhitespace, input, pos)[0]));
|
|
545
|
-
}
|
|
546
|
-
}
|
|
547
|
-
|
|
548
|
-
// ### `_suspend` keeps the unconsumed input until more input arrives
|
|
549
|
-
_suspend(input, pos, currentLineLength) {
|
|
550
|
-
this._linePosition = currentLineLength - (input.length - pos);
|
|
551
|
-
return this._input = input.slice(pos);
|
|
531
|
+
function reportSyntaxError(self) { callback(self._syntaxError(/^\S*/.exec(input)[0])); }
|
|
552
532
|
}
|
|
553
533
|
|
|
554
534
|
// ### `_matchN3Verb` matches an N3 verb unless the input is a longer prefixed name
|
|
555
|
-
_matchN3Verb(input,
|
|
556
|
-
const verb =
|
|
535
|
+
_matchN3Verb(input, inputFinished) {
|
|
536
|
+
const verb = this._n3Verb.exec(input);
|
|
557
537
|
if (!verb)
|
|
558
538
|
return null;
|
|
559
539
|
|
|
560
540
|
// Most verb boundaries cannot be part of a prefix, so keep the common path fast.
|
|
561
|
-
const next = input[
|
|
541
|
+
const next = input[verb[0].length];
|
|
562
542
|
if (next !== '-' && next !== '_' && (next < '0' || next > '9'))
|
|
563
543
|
return verb;
|
|
564
544
|
|
|
565
545
|
// A prefix can start with a verb and continue with characters that are also
|
|
566
546
|
// valid verb boundaries. Prefer the longer prefixed name when it is complete.
|
|
567
|
-
if (
|
|
547
|
+
if (this._prefixed.exec(input))
|
|
568
548
|
return null;
|
|
569
549
|
// Appending to the input only matters when a prefixed name could run up to
|
|
570
550
|
// its end, which a character that cannot occur in prefixed names rules out.
|
|
571
551
|
// This avoids copying the rest of the document for every such verb.
|
|
572
|
-
if (
|
|
552
|
+
if (nonPrefixedNameChar.test(input))
|
|
573
553
|
return verb;
|
|
574
|
-
if (
|
|
554
|
+
if (this._prefixed.exec(`${input} `))
|
|
575
555
|
return null;
|
|
576
556
|
|
|
577
557
|
// If a stream chunk ends partway through such a prefix, wait for the colon
|
|
578
558
|
// instead of prematurely emitting the verb. Appending ": " lets the prefix
|
|
579
559
|
// grammar determine whether all input seen so far can be a complete prefix.
|
|
580
560
|
if (!inputFinished) {
|
|
581
|
-
const prefix =
|
|
561
|
+
const prefix = this._prefix.exec(`${input}: `);
|
|
582
562
|
if (prefix)
|
|
583
563
|
return null;
|
|
584
564
|
}
|
|
@@ -632,20 +612,20 @@ export default class N3Lexer {
|
|
|
632
612
|
return result + item.slice(start);
|
|
633
613
|
}
|
|
634
614
|
|
|
635
|
-
// ### `_parseLiteral` parses a literal
|
|
636
|
-
_parseLiteral(input
|
|
615
|
+
// ### `_parseLiteral` parses a literal into an unescaped value
|
|
616
|
+
_parseLiteral(input) {
|
|
637
617
|
// Ensure we have enough lookahead to identify triple-quoted strings
|
|
638
|
-
if (input.length
|
|
618
|
+
if (input.length >= 3) {
|
|
639
619
|
// The caller has already identified a single or double quote.
|
|
640
|
-
const quote = input[
|
|
641
|
-
const openingLength = input[
|
|
620
|
+
const quote = input[0];
|
|
621
|
+
const openingLength = input[1] === quote && input[2] === quote ? 3 : 1;
|
|
642
622
|
let opening = quote;
|
|
643
623
|
if (openingLength === 3)
|
|
644
624
|
opening = quote === '"' ? '"""' : "'''";
|
|
645
625
|
|
|
646
626
|
// Find the next candidate closing quotes
|
|
647
|
-
let closingPos =
|
|
648
|
-
while ((closingPos = input.indexOf(opening, closingPos)) >
|
|
627
|
+
let closingPos = Math.max(this._literalClosingPos, openingLength);
|
|
628
|
+
while ((closingPos = input.indexOf(opening, closingPos)) > 0) {
|
|
649
629
|
// Count backslashes right before the closing quotes
|
|
650
630
|
let backslashCount = 0;
|
|
651
631
|
while (input[closingPos - backslashCount - 1] === '\\')
|
|
@@ -655,10 +635,10 @@ export default class N3Lexer {
|
|
|
655
635
|
// means these are actual, non-escaped closing quotes
|
|
656
636
|
if (backslashCount % 2 === 0) {
|
|
657
637
|
// Extract and unescape the value
|
|
658
|
-
const raw = input.substring(
|
|
638
|
+
const raw = input.substring(openingLength, closingPos),
|
|
659
639
|
lines = raw.split(/\r\n|\r|\n/),
|
|
660
640
|
lineCount = lines.length - 1;
|
|
661
|
-
const matchLength = closingPos
|
|
641
|
+
const matchLength = closingPos + openingLength;
|
|
662
642
|
// Only triple-quoted strings can be multi-line
|
|
663
643
|
if (openingLength === 1 && lineCount !== 0 ||
|
|
664
644
|
openingLength === 3 && this._lineMode)
|
|
@@ -669,7 +649,7 @@ export default class N3Lexer {
|
|
|
669
649
|
}
|
|
670
650
|
closingPos++;
|
|
671
651
|
}
|
|
672
|
-
this._literalClosingPos = input.length -
|
|
652
|
+
this._literalClosingPos = input.length - openingLength + 1;
|
|
673
653
|
}
|
|
674
654
|
return { value: '', matchLength: 0, finalLineLength: 0 };
|
|
675
655
|
}
|