n3 3.0.0-alpha.7 → 3.0.0-alpha.9
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +10 -1
- package/browser/n3.esm.min.js +10 -9
- package/browser/n3.min.js +10 -9
- package/lib/N3DataFactory.js +3 -2
- package/lib/N3Lexer.js +153 -39
- package/lib/N3Parser.js +22 -3
- package/lib/N3Reasoner.js +5 -4
- package/lib/N3Store.js +128 -133
- package/lib/N3Writer.js +9 -38
- package/package.json +7 -2
- package/src/N3DataFactory.js +3 -2
- package/src/N3Lexer.js +176 -43
- package/src/N3Parser.js +26 -3
- package/src/N3Reasoner.js +6 -4
- package/src/N3Store.js +122 -143
- package/src/N3Writer.js +12 -42
package/lib/N3DataFactory.js
CHANGED
|
@@ -53,8 +53,9 @@ class Term {
|
|
|
53
53
|
// ### Returns whether this object represents the same term as the other
|
|
54
54
|
equals(other) {
|
|
55
55
|
// If both terms were created by this library,
|
|
56
|
-
// equality can be computed through ids
|
|
57
|
-
|
|
56
|
+
// equality can be computed through ids of the same term type,
|
|
57
|
+
// since IRIs such as `?x` have the same id as other terms
|
|
58
|
+
if (other instanceof Term) return this.id === other.id && this.termType === other.termType;
|
|
58
59
|
// Otherwise, compare term type and value
|
|
59
60
|
return !!other && this.termType === other.termType && this.value === other.value;
|
|
60
61
|
}
|
package/lib/N3Lexer.js
CHANGED
|
@@ -15,7 +15,33 @@ const SPACE = 0x20,
|
|
|
15
15
|
TAB = 0x09,
|
|
16
16
|
LF = 0x0A,
|
|
17
17
|
CR = 0x0D,
|
|
18
|
-
HASH = 0x23
|
|
18
|
+
HASH = 0x23,
|
|
19
|
+
DOT = 0x2E,
|
|
20
|
+
COLON = 0x3A,
|
|
21
|
+
ZERO = 0x30,
|
|
22
|
+
NINE = 0x39,
|
|
23
|
+
PERCENT = 0x25,
|
|
24
|
+
BACKSLASH = 0x5C;
|
|
25
|
+
|
|
26
|
+
// Whitespace as matched by `\s`
|
|
27
|
+
function isWhitespace(charCode) {
|
|
28
|
+
return charCode === SPACE || charCode >= TAB && charCode <= CR || charCode >= 0xA0 && (charCode === 0xA0 || charCode === 0x1680 || charCode >= 0x2000 && charCode <= 0x200A || charCode === 0x2028 || charCode === 0x2029 || charCode === 0x202F || charCode === 0x205F || charCode === 0x3000 || charCode === 0xFEFF);
|
|
29
|
+
}
|
|
30
|
+
// Characters that can directly follow a name: whitespace and punctuation
|
|
31
|
+
// (the lookahead `[,;!\^\s#()\[\]\{\}"'<>]` of the former regular expressions)
|
|
32
|
+
const asciiDelimiters = new Uint8Array(0x80);
|
|
33
|
+
for (const char of ',;!^#()[]{}"\'<> \t\n\v\f\r') asciiDelimiters[char.charCodeAt(0)] = 1;
|
|
34
|
+
function isDelimiter(charCode) {
|
|
35
|
+
return charCode < 0x80 ? asciiDelimiters[charCode] === 1 : isWhitespace(charCode);
|
|
36
|
+
}
|
|
37
|
+
// Whether a name can end before the given position: it must be followed by
|
|
38
|
+
// a delimiter, optionally after a dot. At the end of finished input, it can always end.
|
|
39
|
+
function canEndName(input, pos, inputFinished) {
|
|
40
|
+
let charCode = input.charCodeAt(pos);
|
|
41
|
+
if (charCode === DOT) charCode = input.charCodeAt(++pos);
|
|
42
|
+
if (pos >= input.length) return inputFinished;
|
|
43
|
+
return isDelimiter(charCode);
|
|
44
|
+
}
|
|
19
45
|
|
|
20
46
|
// Fixed escape sequences allowed in string literals (ECHAR)
|
|
21
47
|
const stringEscapeReplacements = {
|
|
@@ -52,9 +78,87 @@ const localNameEscapeReplacements = {
|
|
|
52
78
|
'%': '%'
|
|
53
79
|
};
|
|
54
80
|
const illegalIriChars = /[\x00-\x20<>\\"\{\}\|\^\`]/;
|
|
55
|
-
|
|
56
|
-
//
|
|
57
|
-
const
|
|
81
|
+
|
|
82
|
+
// Character classes of names, as bit flags for ASCII characters
|
|
83
|
+
const PREFIX_START = 1,
|
|
84
|
+
// PN_CHARS_BASE
|
|
85
|
+
LOCAL_START = 2,
|
|
86
|
+
// PN_CHARS_U, digits, and colon
|
|
87
|
+
NAME_CHAR = 4,
|
|
88
|
+
// PN_CHARS
|
|
89
|
+
LOCAL_CHAR = 8,
|
|
90
|
+
// PN_CHARS and colon
|
|
91
|
+
LOCAL_ESCAPE = 16; // characters that can be escaped in local names (PN_LOCAL_ESC)
|
|
92
|
+
const asciiNameClasses = new Uint8Array(0x80);
|
|
93
|
+
for (let charCode = 0; charCode < 0x80; charCode++) {
|
|
94
|
+
const char = String.fromCharCode(charCode);
|
|
95
|
+
const letter = char >= 'A' && char <= 'Z' || char >= 'a' && char <= 'z';
|
|
96
|
+
const nameChar = letter || char === '_' || char === '-' || char >= '0' && char <= '9';
|
|
97
|
+
asciiNameClasses[charCode] = (letter ? PREFIX_START : 0) | (nameChar && char !== '-' || char === ':' ? LOCAL_START : 0) | (nameChar ? NAME_CHAR : 0) | (nameChar || char === ':' ? LOCAL_CHAR : 0) | (char in localNameEscapeReplacements ? LOCAL_ESCAPE : 0);
|
|
98
|
+
}
|
|
99
|
+
// Returns the length (0, 1, or 2 code units) of the character at the given position
|
|
100
|
+
// if it is in the given name character class
|
|
101
|
+
function nameCharLength(input, pos, charClass) {
|
|
102
|
+
const charCode = input.charCodeAt(pos);
|
|
103
|
+
if (charCode < 0x80) return (asciiNameClasses[charCode] & charClass) !== 0 ? 1 : 0;
|
|
104
|
+
// Characters from U+10000 to U+EFFFF consist of a surrogate pair
|
|
105
|
+
if (charCode >= 0xD800 && charCode <= 0xDB7F) {
|
|
106
|
+
const low = input.charCodeAt(pos + 1);
|
|
107
|
+
return low >= 0xDC00 && low <= 0xDFFF ? 2 : 0;
|
|
108
|
+
}
|
|
109
|
+
// PN_CHARS has some characters that PN_CHARS_BASE does not
|
|
110
|
+
if (charClass >= NAME_CHAR && (charCode === 0xB7 || charCode >= 0x300 && charCode <= 0x36F || charCode === 0x203F || charCode === 0x2040)) return 1;
|
|
111
|
+
// PN_CHARS_BASE
|
|
112
|
+
return charCode >= 0xC0 && charCode <= 0x1FFF && charCode !== 0xD7 && charCode !== 0xF7 && (charCode < 0x300 || charCode >= 0x370 && charCode !== 0x37E) || charCode >= 0x200C && charCode <= 0x200D || charCode >= 0x2070 && charCode <= 0x218F || charCode >= 0x2C00 && charCode <= 0x2FEF || charCode >= 0x3001 && charCode <= 0xD7FF || charCode >= 0xF900 && charCode <= 0xFDCF || charCode >= 0xFDF0 && charCode <= 0xFFFD ? 1 : 0;
|
|
113
|
+
}
|
|
114
|
+
function isLocalEscape(charCode) {
|
|
115
|
+
return charCode < 0x80 && (asciiNameClasses[charCode] & LOCAL_ESCAPE) !== 0;
|
|
116
|
+
}
|
|
117
|
+
function isHexDigit(charCode) {
|
|
118
|
+
return charCode >= ZERO && charCode <= NINE || charCode >= 0x41 && charCode <= 0x46 || charCode >= 0x61 && charCode <= 0x66;
|
|
119
|
+
}
|
|
120
|
+
// Returns the end of the prefix (PN_PREFIX) at the given position,
|
|
121
|
+
// which can contain single dots, but not start or end with one
|
|
122
|
+
function skipPrefix(input, pos) {
|
|
123
|
+
let length = nameCharLength(input, pos, PREFIX_START);
|
|
124
|
+
while (length !== 0) {
|
|
125
|
+
pos += length;
|
|
126
|
+
const next = input.charCodeAt(pos) === DOT ? pos + 1 : pos;
|
|
127
|
+
// Most names are ASCII, so look those up without a call
|
|
128
|
+
const charCode = input.charCodeAt(next);
|
|
129
|
+
length = charCode < 0x80 ? (asciiNameClasses[charCode] & NAME_CHAR) !== 0 ? 1 : 0 : nameCharLength(input, next, NAME_CHAR);
|
|
130
|
+
if (length !== 0) pos = next;
|
|
131
|
+
}
|
|
132
|
+
return pos;
|
|
133
|
+
}
|
|
134
|
+
// Returns the end of the local name (PN_LOCAL) at the given position,
|
|
135
|
+
// which can contain dots, but not start or end with one
|
|
136
|
+
function skipLocalName(input, pos) {
|
|
137
|
+
let end = pos,
|
|
138
|
+
charClass = LOCAL_START;
|
|
139
|
+
while (true) {
|
|
140
|
+
// Most names are ASCII, so look those up without a call
|
|
141
|
+
const charCode = input.charCodeAt(pos);
|
|
142
|
+
let length = charCode < 0x80 ? (asciiNameClasses[charCode] & charClass) !== 0 ? 1 : 0 : nameCharLength(input, pos, charClass);
|
|
143
|
+
// Percent-encoded character (PERCENT)
|
|
144
|
+
if (length === 0 && charCode === PERCENT && isHexDigit(input.charCodeAt(pos + 1)) && isHexDigit(input.charCodeAt(pos + 2))) length = 3;
|
|
145
|
+
// Escaped character (PN_LOCAL_ESC)
|
|
146
|
+
else if (length === 0 && charCode === BACKSLASH && isLocalEscape(input.charCodeAt(pos + 1))) length = 2;
|
|
147
|
+
if (length !== 0) {
|
|
148
|
+
end = pos += length;
|
|
149
|
+
charClass = LOCAL_CHAR;
|
|
150
|
+
}
|
|
151
|
+
// Dots are allowed after the first character, but not at the end
|
|
152
|
+
else if (charCode === DOT && charClass === LOCAL_CHAR) pos++;else return end;
|
|
153
|
+
}
|
|
154
|
+
}
|
|
155
|
+
// Returns the end of the prefixed name at the given position, or -1 if there is none
|
|
156
|
+
function skipPrefixedName(input, pos, inputFinished) {
|
|
157
|
+
const colon = skipPrefix(input, pos);
|
|
158
|
+
if (input.charCodeAt(colon) !== COLON) return -1;
|
|
159
|
+
const end = skipLocalName(input, colon + 1);
|
|
160
|
+
return canEndName(input, end, inputFinished) ? end : -1;
|
|
161
|
+
}
|
|
58
162
|
|
|
59
163
|
// A valid code point is a Unicode scalar value: at most U+10FFFF and not a surrogate
|
|
60
164
|
function isValidCodePoint(charCode) {
|
|
@@ -110,13 +214,11 @@ class N3Lexer {
|
|
|
110
214
|
this._simpleQuotedString = /"([^"\\\r\n]*)"(?=[^"])/y; // string without escape sequences
|
|
111
215
|
this._simpleApostropheString = /'([^'\\\r\n]*)'(?=[^'])/y;
|
|
112
216
|
this._langcode = /@([a-z]+(?:-[a-z0-9]+)*)(?=[^a-z0-9])/iy;
|
|
113
|
-
this._prefix = /((?:[A-Za-z\xc0-\xd6\xd8-\xf6\xf8-\u02ff\u0370-\u037d\u037f-\u1fff\u200c\u200d\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff])(?:\.?[\-0-9A-Z_a-z\xb7\xc0-\xd6\xd8-\xf6\xf8-\u037d\u037f-\u1fff\u200c\u200d\u203f\u2040\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff])*)?:(?=[#\s<])/y;
|
|
114
|
-
this._prefixed = /((?:[A-Za-z\xc0-\xd6\xd8-\xf6\xf8-\u02ff\u0370-\u037d\u037f-\u1fff\u200c\u200d\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff])(?:\.?[\-0-9A-Z_a-z\xb7\xc0-\xd6\xd8-\xf6\xf8-\u037d\u037f-\u1fff\u200c\u200d\u203f\u2040\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff])*)?:((?:(?:[0-9:A-Z_a-z\xc0-\xd6\xd8-\xf6\xf8-\u02ff\u0370-\u037d\u037f-\u1fff\u200c\u200d\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff]|%[0-9a-fA-F]{2}|\\[!#-\/;=?\-@_~])(?:(?:[\.\-0-9:A-Z_a-z\xb7\xc0-\xd6\xd8-\xf6\xf8-\u037d\u037f-\u1fff\u200c\u200d\u203f\u2040\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff]|%[0-9a-fA-F]{2}|\\[!#-\/;=?\-@_~])*(?:[\-0-9:A-Z_a-z\xb7\xc0-\xd6\xd8-\xf6\xf8-\u037d\u037f-\u1fff\u200c\u200d\u203f\u2040\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff]|%[0-9a-fA-F]{2}|\\[!#-\/;=?\-@_~]))?)?)(?:[ \t]+|(?=\.?[,;!\^\s#()\[\]\{\}"'<>]))/y;
|
|
115
217
|
this._variable = /\?(?:(?:[A-Z_a-z\xc0-\xd6\xd8-\xf6\xf8-\u02ff\u0370-\u037d\u037f-\u1fff\u200c\u200d\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff])(?:[\-0-9:A-Z_a-z\xb7\xc0-\xd6\xd8-\xf6\xf8-\u037d\u037f-\u1fff\u200c\u200d\u203f\u2040\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff])*)(?=[.,;!\^\s#()\[\]\{\}"'<>])/y;
|
|
116
218
|
this._blank = /_:((?:[0-9A-Z_a-z\xc0-\xd6\xd8-\xf6\xf8-\u02ff\u0370-\u037d\u037f-\u1fff\u200c\u200d\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff])(?:\.?[\-0-9A-Z_a-z\xb7\xc0-\xd6\xd8-\xf6\xf8-\u037d\u037f-\u1fff\u200c\u200d\u203f\u2040\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff])*)(?:[ \t]+|(?=\.?[,;:!\^\s#()\[\]\{\}"'<>]))/y;
|
|
117
219
|
this._number = /[\-+]?(?:(\d+\.\d*|\.?\d+)[eE][\-+]?\d+|(?=\.?\d)\d*(?:(\.)\d+)?)(?=\.?[,;:!\^\s#()\[\]\{\}"'<>])/y;
|
|
118
220
|
this._boolean = /(?:true|false)(?=[.,;!\^\s#()\[\]\{\}"'<>])/y;
|
|
119
|
-
this._atKeyword = /@[a-z]+(?=[\s#<:"'])/iy;
|
|
221
|
+
this._atKeyword = /@[a-z]+(?=[\s#<:"'.])/iy;
|
|
120
222
|
this._keyword = /(?:PREFIX|BASE|VERSION|GRAPH)(?=[\s#<"'])/iy;
|
|
121
223
|
this._n3Verb = /(?:has|is|of)(?=[\s#()\[\]\{\}"'<>?_+\-0-9])/y;
|
|
122
224
|
this._n3Id = /id(?=[\s#<])/y;
|
|
@@ -149,7 +251,9 @@ class N3Lexer {
|
|
|
149
251
|
for (const name of options.directives) {
|
|
150
252
|
if (!/^[a-z]+$/i.test(name) || reservedWords.test(name)) throw new Error(`Invalid directive name: "${name}"`);
|
|
151
253
|
}
|
|
152
|
-
|
|
254
|
+
// Like VERSION, directives are case-sensitive in N-Triples and N-Quads
|
|
255
|
+
const names = this._lineMode ? options.directives.map(name => name.toUpperCase()) : options.directives;
|
|
256
|
+
this._directive = new RegExp(`(?:${names.join('|')})(?=[\\s#<])`, this._lineMode ? 'y' : 'iy');
|
|
153
257
|
this._directiveMaxLength = Math.max(...options.directives.map(name => name.length));
|
|
154
258
|
// The first characters of directive names, so other words skip the regular expression
|
|
155
259
|
this._directiveStarts = options.directives.map(name => name[0].toLowerCase() + name[0].toUpperCase()).join('');
|
|
@@ -476,17 +580,22 @@ class N3Lexer {
|
|
|
476
580
|
// Some first characters do not allow an immediate decision, so inspect more
|
|
477
581
|
if (inconclusive) {
|
|
478
582
|
// Try to find a prefix
|
|
479
|
-
|
|
583
|
+
let end;
|
|
584
|
+
if ((this._previousMarker === '@prefix' || this._previousMarker === 'PREFIX') && input.charCodeAt(end = skipPrefix(input, pos)) === COLON && end + 1 < input.length && (input[end + 1] === '#' || input[end + 1] === '<' || isWhitespace(input.charCodeAt(end + 1)))) {
|
|
585
|
+
type = 'prefix', value = input.slice(pos, end);
|
|
586
|
+
matchLength = end + 1 - pos;
|
|
587
|
+
}
|
|
480
588
|
// Try to find an additional directive keyword
|
|
481
589
|
// (at the end of the input, only a short final word can be one)
|
|
482
590
|
else if (this._directive !== null && this._directiveStarts.includes(firstChar) && ((match = execAt(this._directive, input, pos)) || inputFinished && input.length - pos <= this._directiveMaxLength && (match = execAtEnd(this._directive, input, pos)))) type = match[0].toUpperCase();
|
|
483
591
|
// Try to find a prefixed name. Since it can contain (but not end with) a dot,
|
|
484
|
-
// we always need a non-dot character before deciding it is a prefixed name
|
|
485
|
-
//
|
|
486
|
-
else if (
|
|
487
|
-
|
|
488
|
-
|
|
489
|
-
|
|
592
|
+
// we always need a non-dot character before deciding it is a prefixed name,
|
|
593
|
+
// except at the end of the input.
|
|
594
|
+
else if (!this._lineMode && (end = skipPrefixedName(input, pos, inputFinished)) >= 0) {
|
|
595
|
+
const colon = input.indexOf(':', pos);
|
|
596
|
+
type = 'prefixed', prefix = input.slice(pos, colon);
|
|
597
|
+
value = this._unescape(input.slice(colon + 1, end), localNameEscapeReplacements);
|
|
598
|
+
matchLength = end - pos;
|
|
490
599
|
}
|
|
491
600
|
}
|
|
492
601
|
|
|
@@ -595,25 +704,17 @@ class N3Lexer {
|
|
|
595
704
|
if (!verb) return null;
|
|
596
705
|
|
|
597
706
|
// Most verb boundaries cannot be part of a prefix, so keep the common path fast.
|
|
707
|
+
// U+1680 and U+FEFF are whitespace to the regular expression but name characters.
|
|
598
708
|
const next = input[pos + verb[0].length];
|
|
599
|
-
if (next !== '-' && next !== '_' && (next < '0' || next > '9')) return verb;
|
|
709
|
+
if (next !== '-' && next !== '_' && (next < '0' || next > '9') && next !== '\u1680' && next !== '\ufeff') return verb;
|
|
600
710
|
|
|
601
711
|
// A prefix can start with a verb and continue with characters that are also
|
|
602
712
|
// valid verb boundaries. Prefer the longer prefixed name when it is complete.
|
|
603
|
-
if (
|
|
604
|
-
|
|
605
|
-
//
|
|
606
|
-
//
|
|
607
|
-
if (
|
|
608
|
-
if (execAtEnd(this._prefixed, input, pos)) return null;
|
|
609
|
-
|
|
610
|
-
// If a stream chunk ends partway through such a prefix, wait for the colon
|
|
611
|
-
// instead of prematurely emitting the verb. Appending ": " lets the prefix
|
|
612
|
-
// grammar determine whether all input seen so far can be a complete prefix.
|
|
613
|
-
if (!inputFinished) {
|
|
614
|
-
const prefix = execAt(this._prefix, `${input.slice(pos)}: `, 0);
|
|
615
|
-
if (prefix) return null;
|
|
616
|
-
}
|
|
713
|
+
if (skipPrefixedName(input, pos, true) >= 0) return null;
|
|
714
|
+
|
|
715
|
+
// If a stream chunk ends partway through such a prefix,
|
|
716
|
+
// wait for the colon instead of prematurely emitting the verb.
|
|
717
|
+
if (!inputFinished && skipPrefix(input, pos) === input.length) return null;
|
|
617
718
|
return verb;
|
|
618
719
|
}
|
|
619
720
|
|
|
@@ -737,6 +838,27 @@ class N3Lexer {
|
|
|
737
838
|
return err;
|
|
738
839
|
}
|
|
739
840
|
|
|
841
|
+
// ### `_startTokenization` resets the lexer state for a new input
|
|
842
|
+
_startTokenization() {
|
|
843
|
+
this._line = 1;
|
|
844
|
+
this._linePosition = 0;
|
|
845
|
+
this._previousMarker = undefined;
|
|
846
|
+
this.previousToken = undefined;
|
|
847
|
+
this._literalClosingPos = 0;
|
|
848
|
+
this._input = undefined;
|
|
849
|
+
// Deferred tokenization and stream events can outlive their invocation.
|
|
850
|
+
// Ignore them once a later call takes ownership of the lexer state.
|
|
851
|
+
return this._tokenization = {};
|
|
852
|
+
}
|
|
853
|
+
|
|
854
|
+
// ### `_tokenizeString` synchronously emits the tokens of a complete string through the callback,
|
|
855
|
+
// so that the caller can consume each token without the lexer collecting them all first
|
|
856
|
+
_tokenizeString(input, callback) {
|
|
857
|
+
this._startTokenization();
|
|
858
|
+
this._input = this._readStartingBom(input);
|
|
859
|
+
this._tryTokenizeToEnd(callback, true);
|
|
860
|
+
}
|
|
861
|
+
|
|
740
862
|
// ### Strips off any starting UTF BOM mark.
|
|
741
863
|
_readStartingBom(input) {
|
|
742
864
|
if (input.startsWith('\ufeff')) {
|
|
@@ -754,15 +876,7 @@ class N3Lexer {
|
|
|
754
876
|
// Separator whitespace counts towards the next token's start, outside either range.
|
|
755
877
|
// Multiline tokens also have endLine; their end column is relative to that line.
|
|
756
878
|
tokenize(input, callback) {
|
|
757
|
-
|
|
758
|
-
// Ignore them once a later call takes ownership of the lexer state.
|
|
759
|
-
const tokenization = this._tokenization = {};
|
|
760
|
-
this._line = 1;
|
|
761
|
-
this._linePosition = 0;
|
|
762
|
-
this._previousMarker = undefined;
|
|
763
|
-
this.previousToken = undefined;
|
|
764
|
-
this._literalClosingPos = 0;
|
|
765
|
-
this._input = undefined;
|
|
879
|
+
const tokenization = this._startTokenization();
|
|
766
880
|
|
|
767
881
|
// If the input is a string, continuously emit tokens through the callback until the end
|
|
768
882
|
if (typeof input === 'string') {
|
package/lib/N3Parser.js
CHANGED
|
@@ -1555,6 +1555,10 @@ class N3Parser {
|
|
|
1555
1555
|
// The read callback is the next function to be executed when a token arrives.
|
|
1556
1556
|
// We start reading in the top context.
|
|
1557
1557
|
this._readCallback = this._readBeforeTopContext;
|
|
1558
|
+
// A parse that failed part-way can have left scopes and a statement open
|
|
1559
|
+
this._contextStack = [];
|
|
1560
|
+
this._graph = this._subject = this._predicate = this._object = null;
|
|
1561
|
+
this._tripleTerm = this._reifier = null;
|
|
1558
1562
|
this._sparqlStyle = false;
|
|
1559
1563
|
this._prefixes = Object.create(null);
|
|
1560
1564
|
this._prefixes._ = this._blankNodePrefix ? this._blankNodePrefix.substr(2) : `b${blankNodePrefix++}_`;
|
|
@@ -1571,14 +1575,14 @@ class N3Parser {
|
|
|
1571
1575
|
this._quantifiedChanges = null;
|
|
1572
1576
|
this._emptyFormula = false;
|
|
1573
1577
|
this._tripleTermDepth = 0;
|
|
1574
|
-
|
|
1578
|
+
const readGrammarToken = token => {
|
|
1575
1579
|
return this._readCallback = this._readCallback(token);
|
|
1576
1580
|
};
|
|
1581
|
+
let readToken = readGrammarToken;
|
|
1577
1582
|
|
|
1578
1583
|
// Comments bypass the grammar, but participate in the token lifecycle.
|
|
1579
1584
|
if (onComment || this._lexer.comments) {
|
|
1580
1585
|
this._lexer.comments = true;
|
|
1581
|
-
const readGrammarToken = readToken;
|
|
1582
1586
|
readToken = token => {
|
|
1583
1587
|
if (token.type !== 'comment') return readGrammarToken(token);
|
|
1584
1588
|
if (onComment) onComment(token.value);
|
|
@@ -1612,7 +1616,22 @@ class N3Parser {
|
|
|
1612
1616
|
this._callback = (e, t) => {
|
|
1613
1617
|
e ? error = e : t && quads.push(t);
|
|
1614
1618
|
};
|
|
1615
|
-
this._lexer
|
|
1619
|
+
const lexer = this._lexer;
|
|
1620
|
+
// Without callbacks, nothing observes the parse before the whole document
|
|
1621
|
+
// has lexed, so tokens can be parsed as they are lexed instead of being
|
|
1622
|
+
// collected into an array first. A lexer that replaces the built-in
|
|
1623
|
+
// tokenize keeps going through its own implementation.
|
|
1624
|
+
if (!onPrefix && !onVersion && !onDirective && !onComment && !onToken && !onTokenEnd && lexer.tokenize === _N3Lexer.default.prototype.tokenize && typeof lexer._tokenizeString === 'function') {
|
|
1625
|
+
// A lexical error used to stop the parse before any token was read,
|
|
1626
|
+
// so undo base declarations that were read before it
|
|
1627
|
+
const base = [this._base, this._basePath, this._baseRoot, this._baseScheme];
|
|
1628
|
+
lexer._tokenizeString(input, (e, token) => {
|
|
1629
|
+
if (e) {
|
|
1630
|
+
[this._base, this._basePath, this._baseRoot, this._baseScheme] = base;
|
|
1631
|
+
this._callback(e), this._callback = noop;
|
|
1632
|
+
} else if (this._readCallback) readToken(token);
|
|
1633
|
+
});
|
|
1634
|
+
} else lexer.tokenize(input).every(readToken);
|
|
1616
1635
|
if (error) throw error;
|
|
1617
1636
|
return quads;
|
|
1618
1637
|
}
|
package/lib/N3Reasoner.js
CHANGED
|
@@ -44,6 +44,7 @@ class N3Reasoner {
|
|
|
44
44
|
if (!store._addToIndex(graphItem.subjects, subject, predicate, object)) return;
|
|
45
45
|
store._addToIndex(graphItem.predicates, predicate, object, subject);
|
|
46
46
|
store._addToIndex(graphItem.objects, object, subject, predicate);
|
|
47
|
+
this._indexed++;
|
|
47
48
|
}
|
|
48
49
|
// Count genuinely new derivations and fail past the budget. The check comes
|
|
49
50
|
// after all three indexes are updated, so a caught error leaves the store
|
|
@@ -176,7 +177,7 @@ class N3Reasoner {
|
|
|
176
177
|
};
|
|
177
178
|
}
|
|
178
179
|
reason(rules) {
|
|
179
|
-
this._derivations = 0;
|
|
180
|
+
this._derivations = this._indexed = 0;
|
|
180
181
|
if (!Array.isArray(rules)) {
|
|
181
182
|
rules = getRulesFromDataset(rules);
|
|
182
183
|
}
|
|
@@ -236,9 +237,9 @@ class N3Reasoner {
|
|
|
236
237
|
this._reasonGraphNaive(rules, graphs[graphId]);
|
|
237
238
|
}
|
|
238
239
|
} finally {
|
|
239
|
-
//
|
|
240
|
-
// so a caught budget error leaves the store
|
|
241
|
-
this._store._size
|
|
240
|
+
// Count the quads added directly to the indexes, even if a derivation
|
|
241
|
+
// budget was exceeded, so a caught budget error leaves the store consistent
|
|
242
|
+
if (this._store._size !== null) this._store._size += this._indexed;
|
|
242
243
|
}
|
|
243
244
|
}
|
|
244
245
|
}
|