n3 3.0.0-alpha.5 → 3.0.0-alpha.7
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +26 -0
- package/browser/n3.esm.min.js +12 -13
- package/browser/n3.min.js +12 -13
- package/lib/N3Lexer.js +54 -14
- package/lib/N3Parser.js +16 -5
- package/lib/N3Store.js +61 -3
- package/lib/N3Writer.js +20 -8
- package/package.json +1 -1
- package/src/N3Lexer.js +46 -14
- package/src/N3Parser.js +20 -5
- package/src/N3Store.js +52 -4
- package/src/N3Writer.js +23 -8
package/lib/N3Lexer.js
CHANGED
|
@@ -116,8 +116,8 @@ class N3Lexer {
|
|
|
116
116
|
this._blank = /_:((?:[0-9A-Z_a-z\xc0-\xd6\xd8-\xf6\xf8-\u02ff\u0370-\u037d\u037f-\u1fff\u200c\u200d\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff])(?:\.?[\-0-9A-Z_a-z\xb7\xc0-\xd6\xd8-\xf6\xf8-\u037d\u037f-\u1fff\u200c\u200d\u203f\u2040\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff])*)(?:[ \t]+|(?=\.?[,;:!\^\s#()\[\]\{\}"'<>]))/y;
|
|
117
117
|
this._number = /[\-+]?(?:(\d+\.\d*|\.?\d+)[eE][\-+]?\d+|(?=\.?\d)\d*(?:(\.)\d+)?)(?=\.?[,;:!\^\s#()\[\]\{\}"'<>])/y;
|
|
118
118
|
this._boolean = /(?:true|false)(?=[.,;!\^\s#()\[\]\{\}"'<>])/y;
|
|
119
|
-
this._atKeyword = /@[a-z]+(?=[\s#<:])/iy;
|
|
120
|
-
this._keyword = /(?:PREFIX|BASE|VERSION|GRAPH)(?=[\s#<])/iy;
|
|
119
|
+
this._atKeyword = /@[a-z]+(?=[\s#<:"'])/iy;
|
|
120
|
+
this._keyword = /(?:PREFIX|BASE|VERSION|GRAPH)(?=[\s#<"'])/iy;
|
|
121
121
|
this._n3Verb = /(?:has|is|of)(?=[\s#()\[\]\{\}"'<>?_+\-0-9])/y;
|
|
122
122
|
this._n3Id = /id(?=[\s#<])/y;
|
|
123
123
|
this._shortPredicates = /a(?=[\s#()\[\]\{\}"'<>])/y;
|
|
@@ -135,6 +135,8 @@ class N3Lexer {
|
|
|
135
135
|
for (const key in this) {
|
|
136
136
|
if (!(key in lineModeRegExps) && this[key] instanceof RegExp) this[key] = invalidRegExp;
|
|
137
137
|
}
|
|
138
|
+
// The only keyword in N-Triples and N-Quads is VERSION, which is case-sensitive
|
|
139
|
+
this._keyword = /VERSION(?=[\s#<"])/y;
|
|
138
140
|
}
|
|
139
141
|
// When not in line mode, enable N3 functionality by default
|
|
140
142
|
else {
|
|
@@ -236,7 +238,8 @@ class N3Lexer {
|
|
|
236
238
|
matchLength = 0,
|
|
237
239
|
lexicalLength = 0,
|
|
238
240
|
finalLineLength = 0,
|
|
239
|
-
inconclusive = false
|
|
241
|
+
inconclusive = false,
|
|
242
|
+
tripleQuoted = false;
|
|
240
243
|
switch (firstChar) {
|
|
241
244
|
case '^':
|
|
242
245
|
// A datatype marker separated from its type cannot be followed by another marker
|
|
@@ -310,7 +313,8 @@ class N3Lexer {
|
|
|
310
313
|
({
|
|
311
314
|
value,
|
|
312
315
|
matchLength,
|
|
313
|
-
finalLineLength
|
|
316
|
+
finalLineLength,
|
|
317
|
+
tripleQuoted
|
|
314
318
|
} = this._parseLiteral(input, pos));
|
|
315
319
|
if (value === null) return reportSyntaxError(this, input, pos);
|
|
316
320
|
}
|
|
@@ -328,7 +332,8 @@ class N3Lexer {
|
|
|
328
332
|
({
|
|
329
333
|
value,
|
|
330
334
|
matchLength,
|
|
331
|
-
finalLineLength
|
|
335
|
+
finalLineLength,
|
|
336
|
+
tripleQuoted
|
|
332
337
|
} = this._parseLiteral(input, pos));
|
|
333
338
|
if (value === null) return reportSyntaxError(this, input, pos);
|
|
334
339
|
}
|
|
@@ -521,7 +526,21 @@ class N3Lexer {
|
|
|
521
526
|
line,
|
|
522
527
|
start,
|
|
523
528
|
end: finalLineLength,
|
|
524
|
-
endLine: this._line
|
|
529
|
+
endLine: this._line,
|
|
530
|
+
tripleQuoted
|
|
531
|
+
};
|
|
532
|
+
callback(null, token);
|
|
533
|
+
}
|
|
534
|
+
// Triple-quoted strings are marked, since version declarations do not allow them
|
|
535
|
+
else if (tripleQuoted) {
|
|
536
|
+
token = {
|
|
537
|
+
type,
|
|
538
|
+
value,
|
|
539
|
+
prefix,
|
|
540
|
+
line,
|
|
541
|
+
start,
|
|
542
|
+
end: start + length,
|
|
543
|
+
tripleQuoted
|
|
525
544
|
};
|
|
526
545
|
callback(null, token);
|
|
527
546
|
} else token = emitToken(type, value, prefix, line, start, lexicalLength || length);
|
|
@@ -671,7 +690,8 @@ class N3Lexer {
|
|
|
671
690
|
return {
|
|
672
691
|
value: this._unescape(raw, stringEscapeReplacements),
|
|
673
692
|
matchLength,
|
|
674
|
-
finalLineLength
|
|
693
|
+
finalLineLength,
|
|
694
|
+
tripleQuoted: openingLength === 3
|
|
675
695
|
};
|
|
676
696
|
}
|
|
677
697
|
closingPos++;
|
|
@@ -681,14 +701,34 @@ class N3Lexer {
|
|
|
681
701
|
return {
|
|
682
702
|
value: '',
|
|
683
703
|
matchLength: 0,
|
|
684
|
-
finalLineLength: 0
|
|
704
|
+
finalLineLength: 0,
|
|
705
|
+
tripleQuoted: false
|
|
685
706
|
};
|
|
686
707
|
}
|
|
687
708
|
|
|
709
|
+
// ### `_tryTokenizeToEnd` tokenizes as far as possible, reporting failures through the callback
|
|
710
|
+
_tryTokenizeToEnd(callback, inputFinished) {
|
|
711
|
+
// Keep track of errors thrown by the callback, which must reach the caller unchanged
|
|
712
|
+
let callbackError;
|
|
713
|
+
try {
|
|
714
|
+
this._tokenizeToEnd((error, token) => {
|
|
715
|
+
try {
|
|
716
|
+
return callback(error, token);
|
|
717
|
+
} catch (thrown) {
|
|
718
|
+
throw callbackError = thrown;
|
|
719
|
+
}
|
|
720
|
+
}, inputFinished);
|
|
721
|
+
} catch (error) {
|
|
722
|
+
// Matching an extremely long token can exhaust the regular expression stack
|
|
723
|
+
if (error === callbackError || !(error instanceof RangeError)) throw error;
|
|
724
|
+
callback(this._syntaxError(null, `Token too long on line ${this._line}.`));
|
|
725
|
+
}
|
|
726
|
+
}
|
|
727
|
+
|
|
688
728
|
// ### `_syntaxError` creates a syntax error for the given issue
|
|
689
|
-
_syntaxError(issue) {
|
|
729
|
+
_syntaxError(issue, message = `Unexpected "${issue}" on line ${this._line}.`) {
|
|
690
730
|
this._input = null;
|
|
691
|
-
const err = new Error(
|
|
731
|
+
const err = new Error(message);
|
|
692
732
|
err.context = {
|
|
693
733
|
token: undefined,
|
|
694
734
|
line: this._line,
|
|
@@ -729,13 +769,13 @@ class N3Lexer {
|
|
|
729
769
|
this._input = this._readStartingBom(input);
|
|
730
770
|
// If a callback was passed, asynchronously call it
|
|
731
771
|
if (typeof callback === 'function') queueMicrotask(() => {
|
|
732
|
-
if (this._tokenization === tokenization) this.
|
|
772
|
+
if (this._tokenization === tokenization) this._tryTokenizeToEnd(callback, true);
|
|
733
773
|
});
|
|
734
774
|
// If no callback was passed, tokenize synchronously and return
|
|
735
775
|
else {
|
|
736
776
|
const tokens = [];
|
|
737
777
|
let error;
|
|
738
|
-
this.
|
|
778
|
+
this._tryTokenizeToEnd((e, t) => e ? error = e : tokens.push(t), true);
|
|
739
779
|
if (error) throw error;
|
|
740
780
|
return tokens;
|
|
741
781
|
}
|
|
@@ -762,7 +802,7 @@ class N3Lexer {
|
|
|
762
802
|
// Tokenize as far as possible. When a previous attempt left a long unfinished token,
|
|
763
803
|
// wait until the buffered input has doubled, so the token is not rescanned for every chunk.
|
|
764
804
|
if (this._input.length >= retryLength) {
|
|
765
|
-
this.
|
|
805
|
+
this._tryTokenizeToEnd(callback, false);
|
|
766
806
|
retryLength = this._input !== null && this._input.length > MIN_RESCAN_LENGTH ? 2 * this._input.length : 0;
|
|
767
807
|
}
|
|
768
808
|
}
|
|
@@ -773,7 +813,7 @@ class N3Lexer {
|
|
|
773
813
|
// Decode any incomplete character left at the end
|
|
774
814
|
const rest = decoder ? decoder.decode() : '';
|
|
775
815
|
if (rest) this._input = typeof this._input === 'string' ? this._input + rest : rest;
|
|
776
|
-
if (typeof this._input === 'string') this.
|
|
816
|
+
if (typeof this._input === 'string') this._tryTokenizeToEnd(callback, true);
|
|
777
817
|
}
|
|
778
818
|
});
|
|
779
819
|
input.on('error', error => {
|
package/lib/N3Parser.js
CHANGED
|
@@ -11,6 +11,10 @@ function _interopRequireDefault(e) { return e && e.__esModule ? e : { default: e
|
|
|
11
11
|
// **N3Parser** parses N3 documents.
|
|
12
12
|
|
|
13
13
|
let blankNodePrefix = 0;
|
|
14
|
+
// Detects `.` and `..` path segments in an IRI
|
|
15
|
+
const dotSegments = /(^|\/)\.\.?($|[/#?])/;
|
|
16
|
+
// Detects `.` and `..` segments in the path of an IRI, ignoring its query and fragment
|
|
17
|
+
const pathDotSegments = /^[^?#]*(?:^|\/)\.\.?(?:$|[/#?])/;
|
|
14
18
|
|
|
15
19
|
// ## Constructor
|
|
16
20
|
class N3Parser {
|
|
@@ -92,6 +96,7 @@ class N3Parser {
|
|
|
92
96
|
if (!baseIRI) {
|
|
93
97
|
this._base = '';
|
|
94
98
|
this._basePath = '';
|
|
99
|
+
this._basePathHasDotSegments = false;
|
|
95
100
|
} else {
|
|
96
101
|
// Remove fragment if present
|
|
97
102
|
const fragmentPos = baseIRI.indexOf('#');
|
|
@@ -108,6 +113,8 @@ class N3Parser {
|
|
|
108
113
|
// If the base has an authority but an empty path,
|
|
109
114
|
// relative IRIs merge under the path '/' (RFC 3986 §5.2.3)
|
|
110
115
|
if (baseIRI[2] !== undefined && (base.length === this._baseRoot.length || base[this._baseRoot.length] === '?')) this._basePath = `${this._baseRoot}/`;
|
|
116
|
+
// Check once whether resolving against the base path needs to remove dot segments
|
|
117
|
+
this._basePathHasDotSegments = dotSegments.test(this._basePath);
|
|
111
118
|
}
|
|
112
119
|
}
|
|
113
120
|
|
|
@@ -154,7 +161,7 @@ class N3Parser {
|
|
|
154
161
|
// Prefix and base declarations are scoped to their formula,
|
|
155
162
|
// so record prefix changes to undo them when the formula ends
|
|
156
163
|
if (type === 'formula') {
|
|
157
|
-
context.base = [this._base, this._basePath, this._baseRoot, this._baseScheme];
|
|
164
|
+
context.base = [this._base, this._basePath, this._baseRoot, this._baseScheme, this._basePathHasDotSegments];
|
|
158
165
|
this._prefixChanges = [];
|
|
159
166
|
}
|
|
160
167
|
this._contextStack.push(context);
|
|
@@ -194,7 +201,7 @@ class N3Parser {
|
|
|
194
201
|
if (this._n3Mode) {
|
|
195
202
|
this._inversePredicate = context.inverse;
|
|
196
203
|
this._expectOf = context.expectOf;
|
|
197
|
-
if (type === 'formula') [this._base, this._basePath, this._baseRoot, this._baseScheme] = context.base;
|
|
204
|
+
if (type === 'formula') [this._base, this._basePath, this._baseRoot, this._baseScheme, this._basePathHasDotSegments] = context.base;
|
|
198
205
|
if (this._prefixChanges !== context.prefixChanges) {
|
|
199
206
|
undoChanges(this._prefixes, this._prefixChanges);
|
|
200
207
|
this._prefixChanges = context.prefixChanges;
|
|
@@ -1026,7 +1033,8 @@ class N3Parser {
|
|
|
1026
1033
|
// ### `_readVersion` reads version string declaration
|
|
1027
1034
|
_readVersion(token) {
|
|
1028
1035
|
if (token.type !== 'literal') return this._error('Expected literal to follow version declaration', token);
|
|
1029
|
-
|
|
1036
|
+
// Only short strings are allowed, so no numbers or booleans (which have a datatype prefix)
|
|
1037
|
+
if (token.prefix !== '' || token.tripleQuoted) return this._error('Version declarations must use single quotes', token);
|
|
1030
1038
|
this._versionCallback(token.value);
|
|
1031
1039
|
if (!this._isValidVersion(token.value)) return this._error(`Detected unsupported version: "${token.value}"`, token);
|
|
1032
1040
|
return this._readDeclarationPunctuation;
|
|
@@ -1455,14 +1463,17 @@ class N3Parser {
|
|
|
1455
1463
|
// Resolve all other IRIs at the base IRI's path
|
|
1456
1464
|
default:
|
|
1457
1465
|
// Relative IRIs cannot contain a colon in the first path segment
|
|
1458
|
-
|
|
1466
|
+
if (/^[^/:]*:/.test(iri)) return null;
|
|
1467
|
+
// Only scan the joined IRI for dot segments if either part can contain them,
|
|
1468
|
+
// as the base path can be long and the joined IRI would need to be copied
|
|
1469
|
+
return this._basePathHasDotSegments || pathDotSegments.test(iri) ? this._removeDotSegments(this._basePath + iri) : this._basePath + iri;
|
|
1459
1470
|
}
|
|
1460
1471
|
}
|
|
1461
1472
|
|
|
1462
1473
|
// ### `_removeDotSegments` resolves './' and '../' path segments in an IRI as per RFC3986
|
|
1463
1474
|
_removeDotSegments(iri) {
|
|
1464
1475
|
// Don't modify the IRI if it does not contain any dot segments
|
|
1465
|
-
if (
|
|
1476
|
+
if (!dotSegments.test(iri)) return iri;
|
|
1466
1477
|
|
|
1467
1478
|
// Start with an imaginary slash before the IRI in order to resolve trailing './' and '../'
|
|
1468
1479
|
const length = iri.length;
|
package/lib/N3Store.js
CHANGED
|
@@ -572,12 +572,67 @@ class N3Store {
|
|
|
572
572
|
return !this.readQuads(subjectOrQuad, predicate, object, graph).next().done;
|
|
573
573
|
}
|
|
574
574
|
|
|
575
|
-
// ### `import` adds a stream of quads to the store
|
|
575
|
+
// ### `import` adds a stream of quads to the store.
|
|
576
|
+
// It returns the stream, wrapped such that it can also be awaited
|
|
577
|
+
// as a promise of the store (per the RDF/JS `Dataset.import` signature).
|
|
576
578
|
import(stream) {
|
|
577
579
|
stream.on('data', quad => {
|
|
578
580
|
this.addQuad(quad);
|
|
579
581
|
});
|
|
580
|
-
|
|
582
|
+
|
|
583
|
+
// Only track completion once awaited, so unawaited imports keep the stream's behavior
|
|
584
|
+
const store = this;
|
|
585
|
+
let promise = null;
|
|
586
|
+
function completion() {
|
|
587
|
+
return promise || (promise = new Promise((resolve, reject) => {
|
|
588
|
+
let stopListening = null;
|
|
589
|
+
function settle(error) {
|
|
590
|
+
stopListening();
|
|
591
|
+
error ? reject(error) : resolve(store);
|
|
592
|
+
}
|
|
593
|
+
try {
|
|
594
|
+
stopListening = (0, _readableStream.finished)(stream, {
|
|
595
|
+
readable: true,
|
|
596
|
+
writable: false
|
|
597
|
+
}, settle);
|
|
598
|
+
}
|
|
599
|
+
// RDF/JS streams that are not Node.js streams only signal their end and errors
|
|
600
|
+
// and must be awaited before they finish
|
|
601
|
+
catch (_unused) {
|
|
602
|
+
stopListening = () => {
|
|
603
|
+
if (stream.removeListener) {
|
|
604
|
+
stream.removeListener('end', onFinish);
|
|
605
|
+
stream.removeListener('error', settle);
|
|
606
|
+
}
|
|
607
|
+
};
|
|
608
|
+
function onFinish() {
|
|
609
|
+
settle();
|
|
610
|
+
}
|
|
611
|
+
stream.on('end', onFinish);
|
|
612
|
+
stream.on('error', settle);
|
|
613
|
+
}
|
|
614
|
+
}));
|
|
615
|
+
}
|
|
616
|
+
const thenable = {
|
|
617
|
+
then(onFulfilled, onRejected) {
|
|
618
|
+
return completion().then(onFulfilled, onRejected);
|
|
619
|
+
},
|
|
620
|
+
catch(onRejected) {
|
|
621
|
+
return completion().catch(onRejected);
|
|
622
|
+
},
|
|
623
|
+
finally(onFinally) {
|
|
624
|
+
return completion().finally(onFinally);
|
|
625
|
+
}
|
|
626
|
+
};
|
|
627
|
+
// Return a wrapper that behaves as the stream, but is also awaitable
|
|
628
|
+
return new Proxy(stream, {
|
|
629
|
+
get(target, property) {
|
|
630
|
+
if (property === 'then' || property === 'catch' || property === 'finally') return thenable[property];
|
|
631
|
+
const value = target[property];
|
|
632
|
+
// Methods run on the stream itself, which may rely on private fields
|
|
633
|
+
return typeof value === 'function' ? value.bind(target) : value;
|
|
634
|
+
}
|
|
635
|
+
});
|
|
581
636
|
}
|
|
582
637
|
|
|
583
638
|
// ### `removeQuad` removes a quad from the store if it exists
|
|
@@ -617,8 +672,11 @@ class N3Store {
|
|
|
617
672
|
}
|
|
618
673
|
|
|
619
674
|
// ### `removeQuads` removes multiple quads from the store
|
|
675
|
+
// returns `true` if all quads were removed, `false` if some were not found
|
|
620
676
|
removeQuads(quads) {
|
|
621
|
-
|
|
677
|
+
let removed = true;
|
|
678
|
+
for (let i = 0; i < quads.length; i++) removed = this.removeQuad(quads[i]) && removed;
|
|
679
|
+
return removed;
|
|
622
680
|
}
|
|
623
681
|
|
|
624
682
|
// ### `remove` removes a stream of quads from the store
|
package/lib/N3Writer.js
CHANGED
|
@@ -41,6 +41,9 @@ const escape = /["\\\t\n\r\b\f\u0000-\u0019\ud800-\udbff]/,
|
|
|
41
41
|
'\f': '\\f'
|
|
42
42
|
};
|
|
43
43
|
|
|
44
|
+
// Characters that a version label cannot contain
|
|
45
|
+
const invalidVersionLabel = /["\\\u0000-\u001f\u007f]|[\ud800-\udbff](?![\udc00-\udfff])|(?:^|[^\ud800-\udbff])[\udc00-\udfff]/;
|
|
46
|
+
|
|
44
47
|
// ## Placeholder class to represent already pretty-printed terms
|
|
45
48
|
class SerializedTerm extends _N3DataFactory.Term {
|
|
46
49
|
// Pretty-printed nodes are not equal to any other node
|
|
@@ -80,12 +83,17 @@ class N3Writer {
|
|
|
80
83
|
this._endStream = options.end === undefined ? true : !!options.end;
|
|
81
84
|
}
|
|
82
85
|
|
|
86
|
+
// A version label is written as-is, so it cannot contain characters that need escaping,
|
|
87
|
+
// nor unpaired surrogates (which cannot be encoded)
|
|
88
|
+
if (options.version && invalidVersionLabel.test(options.version)) throw new Error(`Invalid version label: ${JSON.stringify(options.version)}`);
|
|
89
|
+
|
|
83
90
|
// Initialize writer, depending on the format
|
|
84
91
|
this._subject = null;
|
|
85
92
|
if (!/triple|quad/i.test(options.format)) {
|
|
86
93
|
this._lineMode = false;
|
|
87
94
|
this._escape = escape, this._escapeAll = escapeAll, this._characterReplacer = characterReplacer;
|
|
88
95
|
this._graph = DEFAULTGRAPH;
|
|
96
|
+
if (options.version) this._write(`@version "${options.version}".\n`);
|
|
89
97
|
this._prefixIRIs = Object.create(null);
|
|
90
98
|
// Escaped prefix IRIs and names for the prefix matcher, computed once per prefix
|
|
91
99
|
this._prefixPatterns = Object.create(null);
|
|
@@ -100,6 +108,7 @@ class N3Writer {
|
|
|
100
108
|
// N-Triples and N-Quads are written in their canonical form
|
|
101
109
|
this._escape = canonicalEscape, this._escapeAll = canonicalEscapeAll;
|
|
102
110
|
this._characterReplacer = canonicalCharacterReplacer;
|
|
111
|
+
if (options.version) this._write(`VERSION "${options.version}"\n`);
|
|
103
112
|
}
|
|
104
113
|
}
|
|
105
114
|
|
|
@@ -115,6 +124,15 @@ class N3Writer {
|
|
|
115
124
|
this._outputStream.write(string, 'utf8', callback);
|
|
116
125
|
}
|
|
117
126
|
|
|
127
|
+
// ### `_endStatement` finishes a pending statement and closes an open graph block
|
|
128
|
+
_endStatement() {
|
|
129
|
+
if (this._subject !== null) {
|
|
130
|
+
this._write(this._inDefaultGraph ? '.\n' : '\n}\n');
|
|
131
|
+
this._subject = null;
|
|
132
|
+
this._graph = DEFAULTGRAPH;
|
|
133
|
+
}
|
|
134
|
+
}
|
|
135
|
+
|
|
118
136
|
// ### `_writeQuad` writes the quad to the output stream
|
|
119
137
|
_writeQuad(subject, predicate, object, graph, done) {
|
|
120
138
|
try {
|
|
@@ -313,10 +331,7 @@ class N3Writer {
|
|
|
313
331
|
if (typeof iri !== 'string') iri = iri.value;
|
|
314
332
|
hasPrefixes = true;
|
|
315
333
|
// Finish a possible pending quad
|
|
316
|
-
|
|
317
|
-
this._write(this._inDefaultGraph ? '.\n' : '\n}\n');
|
|
318
|
-
this._subject = null, this._graph = '';
|
|
319
|
-
}
|
|
334
|
+
this._endStatement();
|
|
320
335
|
// Store and write the prefix
|
|
321
336
|
this._prefixIRIs[iri] = prefix += ':';
|
|
322
337
|
this._prefixPatterns[iri] = (0, _Util.escapeRegex)(iri);
|
|
@@ -390,10 +405,7 @@ class N3Writer {
|
|
|
390
405
|
// ### `end` signals the end of the output stream
|
|
391
406
|
end(done) {
|
|
392
407
|
// Finish a possible pending quad
|
|
393
|
-
|
|
394
|
-
this._write(this._inDefaultGraph ? '.\n' : '\n}\n');
|
|
395
|
-
this._subject = null;
|
|
396
|
-
}
|
|
408
|
+
this._endStatement();
|
|
397
409
|
// Disallow further writing
|
|
398
410
|
this._write = this._blockedWrite;
|
|
399
411
|
|
package/package.json
CHANGED
package/src/N3Lexer.js
CHANGED
|
@@ -81,8 +81,8 @@ export default class N3Lexer {
|
|
|
81
81
|
this._blank = /_:((?:[0-9A-Z_a-z\xc0-\xd6\xd8-\xf6\xf8-\u02ff\u0370-\u037d\u037f-\u1fff\u200c\u200d\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff])(?:\.?[\-0-9A-Z_a-z\xb7\xc0-\xd6\xd8-\xf6\xf8-\u037d\u037f-\u1fff\u200c\u200d\u203f\u2040\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff])*)(?:[ \t]+|(?=\.?[,;:!\^\s#()\[\]\{\}"'<>]))/y;
|
|
82
82
|
this._number = /[\-+]?(?:(\d+\.\d*|\.?\d+)[eE][\-+]?\d+|(?=\.?\d)\d*(?:(\.)\d+)?)(?=\.?[,;:!\^\s#()\[\]\{\}"'<>])/y;
|
|
83
83
|
this._boolean = /(?:true|false)(?=[.,;!\^\s#()\[\]\{\}"'<>])/y;
|
|
84
|
-
this._atKeyword = /@[a-z]+(?=[\s#<:])/iy;
|
|
85
|
-
this._keyword = /(?:PREFIX|BASE|VERSION|GRAPH)(?=[\s#<])/iy;
|
|
84
|
+
this._atKeyword = /@[a-z]+(?=[\s#<:"'])/iy;
|
|
85
|
+
this._keyword = /(?:PREFIX|BASE|VERSION|GRAPH)(?=[\s#<"'])/iy;
|
|
86
86
|
this._n3Verb = /(?:has|is|of)(?=[\s#()\[\]\{\}"'<>?_+\-0-9])/y;
|
|
87
87
|
this._n3Id = /id(?=[\s#<])/y;
|
|
88
88
|
this._shortPredicates = /a(?=[\s#()\[\]\{\}"'<>])/y;
|
|
@@ -101,6 +101,8 @@ export default class N3Lexer {
|
|
|
101
101
|
if (!(key in lineModeRegExps) && this[key] instanceof RegExp)
|
|
102
102
|
this[key] = invalidRegExp;
|
|
103
103
|
}
|
|
104
|
+
// The only keyword in N-Triples and N-Quads is VERSION, which is case-sensitive
|
|
105
|
+
this._keyword = /VERSION(?=[\s#<"])/y;
|
|
104
106
|
}
|
|
105
107
|
// When not in line mode, enable N3 functionality by default
|
|
106
108
|
else {
|
|
@@ -208,7 +210,7 @@ export default class N3Lexer {
|
|
|
208
210
|
const line = this._line, firstChar = input[pos];
|
|
209
211
|
let type = '', value = '', prefix = '',
|
|
210
212
|
match = null, matchLength = 0, lexicalLength = 0,
|
|
211
|
-
finalLineLength = 0, inconclusive = false;
|
|
213
|
+
finalLineLength = 0, inconclusive = false, tripleQuoted = false;
|
|
212
214
|
switch (firstChar) {
|
|
213
215
|
case '^':
|
|
214
216
|
// A datatype marker separated from its type cannot be followed by another marker
|
|
@@ -293,7 +295,7 @@ export default class N3Lexer {
|
|
|
293
295
|
value = match[1];
|
|
294
296
|
// Try to find a literal wrapped in three pairs of quotes
|
|
295
297
|
else {
|
|
296
|
-
({ value, matchLength, finalLineLength } = this._parseLiteral(input, pos));
|
|
298
|
+
({ value, matchLength, finalLineLength, tripleQuoted } = this._parseLiteral(input, pos));
|
|
297
299
|
if (value === null)
|
|
298
300
|
return reportSyntaxError(this, input, pos);
|
|
299
301
|
}
|
|
@@ -310,7 +312,7 @@ export default class N3Lexer {
|
|
|
310
312
|
value = match[1];
|
|
311
313
|
// Try to find a literal wrapped in three pairs of quotes
|
|
312
314
|
else {
|
|
313
|
-
({ value, matchLength, finalLineLength } = this._parseLiteral(input, pos));
|
|
315
|
+
({ value, matchLength, finalLineLength, tripleQuoted } = this._parseLiteral(input, pos));
|
|
314
316
|
if (value === null)
|
|
315
317
|
return reportSyntaxError(this, input, pos);
|
|
316
318
|
}
|
|
@@ -550,10 +552,15 @@ export default class N3Lexer {
|
|
|
550
552
|
if (finalLineLength) {
|
|
551
553
|
token = {
|
|
552
554
|
type, value, prefix, line, start,
|
|
553
|
-
end: finalLineLength, endLine: this._line,
|
|
555
|
+
end: finalLineLength, endLine: this._line, tripleQuoted,
|
|
554
556
|
};
|
|
555
557
|
callback(null, token);
|
|
556
558
|
}
|
|
559
|
+
// Triple-quoted strings are marked, since version declarations do not allow them
|
|
560
|
+
else if (tripleQuoted) {
|
|
561
|
+
token = { type, value, prefix, line, start, end: start + length, tripleQuoted };
|
|
562
|
+
callback(null, token);
|
|
563
|
+
}
|
|
557
564
|
else
|
|
558
565
|
token = emitToken(type, value, prefix, line, start, lexicalLength || length);
|
|
559
566
|
this.previousToken = token;
|
|
@@ -705,19 +712,44 @@ export default class N3Lexer {
|
|
|
705
712
|
break;
|
|
706
713
|
this._line += lineCount;
|
|
707
714
|
const finalLineLength = lineCount === 0 ? 0 : lines[lines.length - 1].length + openingLength;
|
|
708
|
-
return {
|
|
715
|
+
return {
|
|
716
|
+
value: this._unescape(raw, stringEscapeReplacements), matchLength, finalLineLength,
|
|
717
|
+
tripleQuoted: openingLength === 3,
|
|
718
|
+
};
|
|
709
719
|
}
|
|
710
720
|
closingPos++;
|
|
711
721
|
}
|
|
712
722
|
this._literalClosingPos = input.length - pos - openingLength + 1;
|
|
713
723
|
}
|
|
714
|
-
return { value: '', matchLength: 0, finalLineLength: 0 };
|
|
724
|
+
return { value: '', matchLength: 0, finalLineLength: 0, tripleQuoted: false };
|
|
725
|
+
}
|
|
726
|
+
|
|
727
|
+
// ### `_tryTokenizeToEnd` tokenizes as far as possible, reporting failures through the callback
|
|
728
|
+
_tryTokenizeToEnd(callback, inputFinished) {
|
|
729
|
+
// Keep track of errors thrown by the callback, which must reach the caller unchanged
|
|
730
|
+
let callbackError;
|
|
731
|
+
try {
|
|
732
|
+
this._tokenizeToEnd((error, token) => {
|
|
733
|
+
try {
|
|
734
|
+
return callback(error, token);
|
|
735
|
+
}
|
|
736
|
+
catch (thrown) {
|
|
737
|
+
throw (callbackError = thrown);
|
|
738
|
+
}
|
|
739
|
+
}, inputFinished);
|
|
740
|
+
}
|
|
741
|
+
catch (error) {
|
|
742
|
+
// Matching an extremely long token can exhaust the regular expression stack
|
|
743
|
+
if (error === callbackError || !(error instanceof RangeError))
|
|
744
|
+
throw error;
|
|
745
|
+
callback(this._syntaxError(null, `Token too long on line ${this._line}.`));
|
|
746
|
+
}
|
|
715
747
|
}
|
|
716
748
|
|
|
717
749
|
// ### `_syntaxError` creates a syntax error for the given issue
|
|
718
|
-
_syntaxError(issue) {
|
|
750
|
+
_syntaxError(issue, message = `Unexpected "${issue}" on line ${this._line}.`) {
|
|
719
751
|
this._input = null;
|
|
720
|
-
const err = new Error(
|
|
752
|
+
const err = new Error(message);
|
|
721
753
|
err.context = {
|
|
722
754
|
token: undefined,
|
|
723
755
|
line: this._line,
|
|
@@ -760,13 +792,13 @@ export default class N3Lexer {
|
|
|
760
792
|
if (typeof callback === 'function')
|
|
761
793
|
queueMicrotask(() => {
|
|
762
794
|
if (this._tokenization === tokenization)
|
|
763
|
-
this.
|
|
795
|
+
this._tryTokenizeToEnd(callback, true);
|
|
764
796
|
});
|
|
765
797
|
// If no callback was passed, tokenize synchronously and return
|
|
766
798
|
else {
|
|
767
799
|
const tokens = [];
|
|
768
800
|
let error;
|
|
769
|
-
this.
|
|
801
|
+
this._tryTokenizeToEnd((e, t) => e ? (error = e) : tokens.push(t), true);
|
|
770
802
|
if (error) throw error;
|
|
771
803
|
return tokens;
|
|
772
804
|
}
|
|
@@ -793,7 +825,7 @@ export default class N3Lexer {
|
|
|
793
825
|
// Tokenize as far as possible. When a previous attempt left a long unfinished token,
|
|
794
826
|
// wait until the buffered input has doubled, so the token is not rescanned for every chunk.
|
|
795
827
|
if (this._input.length >= retryLength) {
|
|
796
|
-
this.
|
|
828
|
+
this._tryTokenizeToEnd(callback, false);
|
|
797
829
|
retryLength = this._input !== null && this._input.length > MIN_RESCAN_LENGTH ?
|
|
798
830
|
2 * this._input.length : 0;
|
|
799
831
|
}
|
|
@@ -807,7 +839,7 @@ export default class N3Lexer {
|
|
|
807
839
|
if (rest)
|
|
808
840
|
this._input = typeof this._input === 'string' ? this._input + rest : rest;
|
|
809
841
|
if (typeof this._input === 'string')
|
|
810
|
-
this.
|
|
842
|
+
this._tryTokenizeToEnd(callback, true);
|
|
811
843
|
}
|
|
812
844
|
});
|
|
813
845
|
input.on('error', error => {
|
package/src/N3Parser.js
CHANGED
|
@@ -4,6 +4,10 @@ import N3DataFactory from './N3DataFactory';
|
|
|
4
4
|
import namespaces from './IRIs';
|
|
5
5
|
|
|
6
6
|
let blankNodePrefix = 0;
|
|
7
|
+
// Detects `.` and `..` path segments in an IRI
|
|
8
|
+
const dotSegments = /(^|\/)\.\.?($|[/#?])/;
|
|
9
|
+
// Detects `.` and `..` segments in the path of an IRI, ignoring its query and fragment
|
|
10
|
+
const pathDotSegments = /^[^?#]*(?:^|\/)\.\.?(?:$|[/#?])/;
|
|
7
11
|
|
|
8
12
|
// ## Constructor
|
|
9
13
|
export default class N3Parser {
|
|
@@ -84,6 +88,7 @@ export default class N3Parser {
|
|
|
84
88
|
if (!baseIRI) {
|
|
85
89
|
this._base = '';
|
|
86
90
|
this._basePath = '';
|
|
91
|
+
this._basePathHasDotSegments = false;
|
|
87
92
|
}
|
|
88
93
|
else {
|
|
89
94
|
// Remove fragment if present
|
|
@@ -104,6 +109,8 @@ export default class N3Parser {
|
|
|
104
109
|
// relative IRIs merge under the path '/' (RFC 3986 §5.2.3)
|
|
105
110
|
if (baseIRI[2] !== undefined && (base.length === this._baseRoot.length || base[this._baseRoot.length] === '?'))
|
|
106
111
|
this._basePath = `${this._baseRoot}/`;
|
|
112
|
+
// Check once whether resolving against the base path needs to remove dot segments
|
|
113
|
+
this._basePathHasDotSegments = dotSegments.test(this._basePath);
|
|
107
114
|
}
|
|
108
115
|
}
|
|
109
116
|
|
|
@@ -141,7 +148,8 @@ export default class N3Parser {
|
|
|
141
148
|
// Prefix and base declarations are scoped to their formula,
|
|
142
149
|
// so record prefix changes to undo them when the formula ends
|
|
143
150
|
if (type === 'formula') {
|
|
144
|
-
context.base = [this._base, this._basePath, this._baseRoot, this._baseScheme
|
|
151
|
+
context.base = [this._base, this._basePath, this._baseRoot, this._baseScheme,
|
|
152
|
+
this._basePathHasDotSegments];
|
|
145
153
|
this._prefixChanges = [];
|
|
146
154
|
}
|
|
147
155
|
this._contextStack.push(context);
|
|
@@ -183,7 +191,8 @@ export default class N3Parser {
|
|
|
183
191
|
this._inversePredicate = context.inverse;
|
|
184
192
|
this._expectOf = context.expectOf;
|
|
185
193
|
if (type === 'formula')
|
|
186
|
-
[this._base, this._basePath, this._baseRoot, this._baseScheme
|
|
194
|
+
[this._base, this._basePath, this._baseRoot, this._baseScheme,
|
|
195
|
+
this._basePathHasDotSegments] = context.base;
|
|
187
196
|
if (this._prefixChanges !== context.prefixChanges) {
|
|
188
197
|
undoChanges(this._prefixes, this._prefixChanges);
|
|
189
198
|
this._prefixChanges = context.prefixChanges;
|
|
@@ -1112,7 +1121,8 @@ export default class N3Parser {
|
|
|
1112
1121
|
_readVersion(token) {
|
|
1113
1122
|
if (token.type !== 'literal')
|
|
1114
1123
|
return this._error('Expected literal to follow version declaration', token);
|
|
1115
|
-
|
|
1124
|
+
// Only short strings are allowed, so no numbers or booleans (which have a datatype prefix)
|
|
1125
|
+
if (token.prefix !== '' || token.tripleQuoted)
|
|
1116
1126
|
return this._error('Version declarations must use single quotes', token);
|
|
1117
1127
|
this._versionCallback(token.value);
|
|
1118
1128
|
if (!this._isValidVersion(token.value))
|
|
@@ -1576,14 +1586,19 @@ export default class N3Parser {
|
|
|
1576
1586
|
// Resolve all other IRIs at the base IRI's path
|
|
1577
1587
|
default:
|
|
1578
1588
|
// Relative IRIs cannot contain a colon in the first path segment
|
|
1579
|
-
|
|
1589
|
+
if (/^[^/:]*:/.test(iri))
|
|
1590
|
+
return null;
|
|
1591
|
+
// Only scan the joined IRI for dot segments if either part can contain them,
|
|
1592
|
+
// as the base path can be long and the joined IRI would need to be copied
|
|
1593
|
+
return this._basePathHasDotSegments || pathDotSegments.test(iri) ?
|
|
1594
|
+
this._removeDotSegments(this._basePath + iri) : this._basePath + iri;
|
|
1580
1595
|
}
|
|
1581
1596
|
}
|
|
1582
1597
|
|
|
1583
1598
|
// ### `_removeDotSegments` resolves './' and '../' path segments in an IRI as per RFC3986
|
|
1584
1599
|
_removeDotSegments(iri) {
|
|
1585
1600
|
// Don't modify the IRI if it does not contain any dot segments
|
|
1586
|
-
if (
|
|
1601
|
+
if (!dotSegments.test(iri))
|
|
1587
1602
|
return iri;
|
|
1588
1603
|
|
|
1589
1604
|
// Start with an imaginary slash before the IRI in order to resolve trailing './' and '../'
|