n3 3.0.0-alpha.5 → 3.0.0-alpha.7

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/lib/N3Lexer.js CHANGED
@@ -116,8 +116,8 @@ class N3Lexer {
116
116
  this._blank = /_:((?:[0-9A-Z_a-z\xc0-\xd6\xd8-\xf6\xf8-\u02ff\u0370-\u037d\u037f-\u1fff\u200c\u200d\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff])(?:\.?[\-0-9A-Z_a-z\xb7\xc0-\xd6\xd8-\xf6\xf8-\u037d\u037f-\u1fff\u200c\u200d\u203f\u2040\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff])*)(?:[ \t]+|(?=\.?[,;:!\^\s#()\[\]\{\}"'<>]))/y;
117
117
  this._number = /[\-+]?(?:(\d+\.\d*|\.?\d+)[eE][\-+]?\d+|(?=\.?\d)\d*(?:(\.)\d+)?)(?=\.?[,;:!\^\s#()\[\]\{\}"'<>])/y;
118
118
  this._boolean = /(?:true|false)(?=[.,;!\^\s#()\[\]\{\}"'<>])/y;
119
- this._atKeyword = /@[a-z]+(?=[\s#<:])/iy;
120
- this._keyword = /(?:PREFIX|BASE|VERSION|GRAPH)(?=[\s#<])/iy;
119
+ this._atKeyword = /@[a-z]+(?=[\s#<:"'])/iy;
120
+ this._keyword = /(?:PREFIX|BASE|VERSION|GRAPH)(?=[\s#<"'])/iy;
121
121
  this._n3Verb = /(?:has|is|of)(?=[\s#()\[\]\{\}"'<>?_+\-0-9])/y;
122
122
  this._n3Id = /id(?=[\s#<])/y;
123
123
  this._shortPredicates = /a(?=[\s#()\[\]\{\}"'<>])/y;
@@ -135,6 +135,8 @@ class N3Lexer {
135
135
  for (const key in this) {
136
136
  if (!(key in lineModeRegExps) && this[key] instanceof RegExp) this[key] = invalidRegExp;
137
137
  }
138
+ // The only keyword in N-Triples and N-Quads is VERSION, which is case-sensitive
139
+ this._keyword = /VERSION(?=[\s#<"])/y;
138
140
  }
139
141
  // When not in line mode, enable N3 functionality by default
140
142
  else {
@@ -236,7 +238,8 @@ class N3Lexer {
236
238
  matchLength = 0,
237
239
  lexicalLength = 0,
238
240
  finalLineLength = 0,
239
- inconclusive = false;
241
+ inconclusive = false,
242
+ tripleQuoted = false;
240
243
  switch (firstChar) {
241
244
  case '^':
242
245
  // A datatype marker separated from its type cannot be followed by another marker
@@ -310,7 +313,8 @@ class N3Lexer {
310
313
  ({
311
314
  value,
312
315
  matchLength,
313
- finalLineLength
316
+ finalLineLength,
317
+ tripleQuoted
314
318
  } = this._parseLiteral(input, pos));
315
319
  if (value === null) return reportSyntaxError(this, input, pos);
316
320
  }
@@ -328,7 +332,8 @@ class N3Lexer {
328
332
  ({
329
333
  value,
330
334
  matchLength,
331
- finalLineLength
335
+ finalLineLength,
336
+ tripleQuoted
332
337
  } = this._parseLiteral(input, pos));
333
338
  if (value === null) return reportSyntaxError(this, input, pos);
334
339
  }
@@ -521,7 +526,21 @@ class N3Lexer {
521
526
  line,
522
527
  start,
523
528
  end: finalLineLength,
524
- endLine: this._line
529
+ endLine: this._line,
530
+ tripleQuoted
531
+ };
532
+ callback(null, token);
533
+ }
534
+ // Triple-quoted strings are marked, since version declarations do not allow them
535
+ else if (tripleQuoted) {
536
+ token = {
537
+ type,
538
+ value,
539
+ prefix,
540
+ line,
541
+ start,
542
+ end: start + length,
543
+ tripleQuoted
525
544
  };
526
545
  callback(null, token);
527
546
  } else token = emitToken(type, value, prefix, line, start, lexicalLength || length);
@@ -671,7 +690,8 @@ class N3Lexer {
671
690
  return {
672
691
  value: this._unescape(raw, stringEscapeReplacements),
673
692
  matchLength,
674
- finalLineLength
693
+ finalLineLength,
694
+ tripleQuoted: openingLength === 3
675
695
  };
676
696
  }
677
697
  closingPos++;
@@ -681,14 +701,34 @@ class N3Lexer {
681
701
  return {
682
702
  value: '',
683
703
  matchLength: 0,
684
- finalLineLength: 0
704
+ finalLineLength: 0,
705
+ tripleQuoted: false
685
706
  };
686
707
  }
687
708
 
709
+ // ### `_tryTokenizeToEnd` tokenizes as far as possible, reporting failures through the callback
710
+ _tryTokenizeToEnd(callback, inputFinished) {
711
+ // Keep track of errors thrown by the callback, which must reach the caller unchanged
712
+ let callbackError;
713
+ try {
714
+ this._tokenizeToEnd((error, token) => {
715
+ try {
716
+ return callback(error, token);
717
+ } catch (thrown) {
718
+ throw callbackError = thrown;
719
+ }
720
+ }, inputFinished);
721
+ } catch (error) {
722
+ // Matching an extremely long token can exhaust the regular expression stack
723
+ if (error === callbackError || !(error instanceof RangeError)) throw error;
724
+ callback(this._syntaxError(null, `Token too long on line ${this._line}.`));
725
+ }
726
+ }
727
+
688
728
  // ### `_syntaxError` creates a syntax error for the given issue
689
- _syntaxError(issue) {
729
+ _syntaxError(issue, message = `Unexpected "${issue}" on line ${this._line}.`) {
690
730
  this._input = null;
691
- const err = new Error(`Unexpected "${issue}" on line ${this._line}.`);
731
+ const err = new Error(message);
692
732
  err.context = {
693
733
  token: undefined,
694
734
  line: this._line,
@@ -729,13 +769,13 @@ class N3Lexer {
729
769
  this._input = this._readStartingBom(input);
730
770
  // If a callback was passed, asynchronously call it
731
771
  if (typeof callback === 'function') queueMicrotask(() => {
732
- if (this._tokenization === tokenization) this._tokenizeToEnd(callback, true);
772
+ if (this._tokenization === tokenization) this._tryTokenizeToEnd(callback, true);
733
773
  });
734
774
  // If no callback was passed, tokenize synchronously and return
735
775
  else {
736
776
  const tokens = [];
737
777
  let error;
738
- this._tokenizeToEnd((e, t) => e ? error = e : tokens.push(t), true);
778
+ this._tryTokenizeToEnd((e, t) => e ? error = e : tokens.push(t), true);
739
779
  if (error) throw error;
740
780
  return tokens;
741
781
  }
@@ -762,7 +802,7 @@ class N3Lexer {
762
802
  // Tokenize as far as possible. When a previous attempt left a long unfinished token,
763
803
  // wait until the buffered input has doubled, so the token is not rescanned for every chunk.
764
804
  if (this._input.length >= retryLength) {
765
- this._tokenizeToEnd(callback, false);
805
+ this._tryTokenizeToEnd(callback, false);
766
806
  retryLength = this._input !== null && this._input.length > MIN_RESCAN_LENGTH ? 2 * this._input.length : 0;
767
807
  }
768
808
  }
@@ -773,7 +813,7 @@ class N3Lexer {
773
813
  // Decode any incomplete character left at the end
774
814
  const rest = decoder ? decoder.decode() : '';
775
815
  if (rest) this._input = typeof this._input === 'string' ? this._input + rest : rest;
776
- if (typeof this._input === 'string') this._tokenizeToEnd(callback, true);
816
+ if (typeof this._input === 'string') this._tryTokenizeToEnd(callback, true);
777
817
  }
778
818
  });
779
819
  input.on('error', error => {
package/lib/N3Parser.js CHANGED
@@ -11,6 +11,10 @@ function _interopRequireDefault(e) { return e && e.__esModule ? e : { default: e
11
11
  // **N3Parser** parses N3 documents.
12
12
 
13
13
  let blankNodePrefix = 0;
14
+ // Detects `.` and `..` path segments in an IRI
15
+ const dotSegments = /(^|\/)\.\.?($|[/#?])/;
16
+ // Detects `.` and `..` segments in the path of an IRI, ignoring its query and fragment
17
+ const pathDotSegments = /^[^?#]*(?:^|\/)\.\.?(?:$|[/#?])/;
14
18
 
15
19
  // ## Constructor
16
20
  class N3Parser {
@@ -92,6 +96,7 @@ class N3Parser {
92
96
  if (!baseIRI) {
93
97
  this._base = '';
94
98
  this._basePath = '';
99
+ this._basePathHasDotSegments = false;
95
100
  } else {
96
101
  // Remove fragment if present
97
102
  const fragmentPos = baseIRI.indexOf('#');
@@ -108,6 +113,8 @@ class N3Parser {
108
113
  // If the base has an authority but an empty path,
109
114
  // relative IRIs merge under the path '/' (RFC 3986 §5.2.3)
110
115
  if (baseIRI[2] !== undefined && (base.length === this._baseRoot.length || base[this._baseRoot.length] === '?')) this._basePath = `${this._baseRoot}/`;
116
+ // Check once whether resolving against the base path needs to remove dot segments
117
+ this._basePathHasDotSegments = dotSegments.test(this._basePath);
111
118
  }
112
119
  }
113
120
 
@@ -154,7 +161,7 @@ class N3Parser {
154
161
  // Prefix and base declarations are scoped to their formula,
155
162
  // so record prefix changes to undo them when the formula ends
156
163
  if (type === 'formula') {
157
- context.base = [this._base, this._basePath, this._baseRoot, this._baseScheme];
164
+ context.base = [this._base, this._basePath, this._baseRoot, this._baseScheme, this._basePathHasDotSegments];
158
165
  this._prefixChanges = [];
159
166
  }
160
167
  this._contextStack.push(context);
@@ -194,7 +201,7 @@ class N3Parser {
194
201
  if (this._n3Mode) {
195
202
  this._inversePredicate = context.inverse;
196
203
  this._expectOf = context.expectOf;
197
- if (type === 'formula') [this._base, this._basePath, this._baseRoot, this._baseScheme] = context.base;
204
+ if (type === 'formula') [this._base, this._basePath, this._baseRoot, this._baseScheme, this._basePathHasDotSegments] = context.base;
198
205
  if (this._prefixChanges !== context.prefixChanges) {
199
206
  undoChanges(this._prefixes, this._prefixChanges);
200
207
  this._prefixChanges = context.prefixChanges;
@@ -1026,7 +1033,8 @@ class N3Parser {
1026
1033
  // ### `_readVersion` reads version string declaration
1027
1034
  _readVersion(token) {
1028
1035
  if (token.type !== 'literal') return this._error('Expected literal to follow version declaration', token);
1029
- if (token.end - token.start !== token.value.length + 2) return this._error('Version declarations must use single quotes', token);
1036
+ // Only short strings are allowed, so no numbers or booleans (which have a datatype prefix)
1037
+ if (token.prefix !== '' || token.tripleQuoted) return this._error('Version declarations must use single quotes', token);
1030
1038
  this._versionCallback(token.value);
1031
1039
  if (!this._isValidVersion(token.value)) return this._error(`Detected unsupported version: "${token.value}"`, token);
1032
1040
  return this._readDeclarationPunctuation;
@@ -1455,14 +1463,17 @@ class N3Parser {
1455
1463
  // Resolve all other IRIs at the base IRI's path
1456
1464
  default:
1457
1465
  // Relative IRIs cannot contain a colon in the first path segment
1458
- return /^[^/:]*:/.test(iri) ? null : this._removeDotSegments(this._basePath + iri);
1466
+ if (/^[^/:]*:/.test(iri)) return null;
1467
+ // Only scan the joined IRI for dot segments if either part can contain them,
1468
+ // as the base path can be long and the joined IRI would need to be copied
1469
+ return this._basePathHasDotSegments || pathDotSegments.test(iri) ? this._removeDotSegments(this._basePath + iri) : this._basePath + iri;
1459
1470
  }
1460
1471
  }
1461
1472
 
1462
1473
  // ### `_removeDotSegments` resolves './' and '../' path segments in an IRI as per RFC3986
1463
1474
  _removeDotSegments(iri) {
1464
1475
  // Don't modify the IRI if it does not contain any dot segments
1465
- if (!/(^|\/)\.\.?($|[/#?])/.test(iri)) return iri;
1476
+ if (!dotSegments.test(iri)) return iri;
1466
1477
 
1467
1478
  // Start with an imaginary slash before the IRI in order to resolve trailing './' and '../'
1468
1479
  const length = iri.length;
package/lib/N3Store.js CHANGED
@@ -572,12 +572,67 @@ class N3Store {
572
572
  return !this.readQuads(subjectOrQuad, predicate, object, graph).next().done;
573
573
  }
574
574
 
575
- // ### `import` adds a stream of quads to the store
575
+ // ### `import` adds a stream of quads to the store.
576
+ // It returns the stream, wrapped such that it can also be awaited
577
+ // as a promise of the store (per the RDF/JS `Dataset.import` signature).
576
578
  import(stream) {
577
579
  stream.on('data', quad => {
578
580
  this.addQuad(quad);
579
581
  });
580
- return stream;
582
+
583
+ // Only track completion once awaited, so unawaited imports keep the stream's behavior
584
+ const store = this;
585
+ let promise = null;
586
+ function completion() {
587
+ return promise || (promise = new Promise((resolve, reject) => {
588
+ let stopListening = null;
589
+ function settle(error) {
590
+ stopListening();
591
+ error ? reject(error) : resolve(store);
592
+ }
593
+ try {
594
+ stopListening = (0, _readableStream.finished)(stream, {
595
+ readable: true,
596
+ writable: false
597
+ }, settle);
598
+ }
599
+ // RDF/JS streams that are not Node.js streams only signal their end and errors
600
+ // and must be awaited before they finish
601
+ catch (_unused) {
602
+ stopListening = () => {
603
+ if (stream.removeListener) {
604
+ stream.removeListener('end', onFinish);
605
+ stream.removeListener('error', settle);
606
+ }
607
+ };
608
+ function onFinish() {
609
+ settle();
610
+ }
611
+ stream.on('end', onFinish);
612
+ stream.on('error', settle);
613
+ }
614
+ }));
615
+ }
616
+ const thenable = {
617
+ then(onFulfilled, onRejected) {
618
+ return completion().then(onFulfilled, onRejected);
619
+ },
620
+ catch(onRejected) {
621
+ return completion().catch(onRejected);
622
+ },
623
+ finally(onFinally) {
624
+ return completion().finally(onFinally);
625
+ }
626
+ };
627
+ // Return a wrapper that behaves as the stream, but is also awaitable
628
+ return new Proxy(stream, {
629
+ get(target, property) {
630
+ if (property === 'then' || property === 'catch' || property === 'finally') return thenable[property];
631
+ const value = target[property];
632
+ // Methods run on the stream itself, which may rely on private fields
633
+ return typeof value === 'function' ? value.bind(target) : value;
634
+ }
635
+ });
581
636
  }
582
637
 
583
638
  // ### `removeQuad` removes a quad from the store if it exists
@@ -617,8 +672,11 @@ class N3Store {
617
672
  }
618
673
 
619
674
  // ### `removeQuads` removes multiple quads from the store
675
+ // returns `true` if all quads were removed, `false` if some were not found
620
676
  removeQuads(quads) {
621
- for (let i = 0; i < quads.length; i++) this.removeQuad(quads[i]);
677
+ let removed = true;
678
+ for (let i = 0; i < quads.length; i++) removed = this.removeQuad(quads[i]) && removed;
679
+ return removed;
622
680
  }
623
681
 
624
682
  // ### `remove` removes a stream of quads from the store
package/lib/N3Writer.js CHANGED
@@ -41,6 +41,9 @@ const escape = /["\\\t\n\r\b\f\u0000-\u0019\ud800-\udbff]/,
41
41
  '\f': '\\f'
42
42
  };
43
43
 
44
+ // Characters that a version label cannot contain
45
+ const invalidVersionLabel = /["\\\u0000-\u001f\u007f]|[\ud800-\udbff](?![\udc00-\udfff])|(?:^|[^\ud800-\udbff])[\udc00-\udfff]/;
46
+
44
47
  // ## Placeholder class to represent already pretty-printed terms
45
48
  class SerializedTerm extends _N3DataFactory.Term {
46
49
  // Pretty-printed nodes are not equal to any other node
@@ -80,12 +83,17 @@ class N3Writer {
80
83
  this._endStream = options.end === undefined ? true : !!options.end;
81
84
  }
82
85
 
86
+ // A version label is written as-is, so it cannot contain characters that need escaping,
87
+ // nor unpaired surrogates (which cannot be encoded)
88
+ if (options.version && invalidVersionLabel.test(options.version)) throw new Error(`Invalid version label: ${JSON.stringify(options.version)}`);
89
+
83
90
  // Initialize writer, depending on the format
84
91
  this._subject = null;
85
92
  if (!/triple|quad/i.test(options.format)) {
86
93
  this._lineMode = false;
87
94
  this._escape = escape, this._escapeAll = escapeAll, this._characterReplacer = characterReplacer;
88
95
  this._graph = DEFAULTGRAPH;
96
+ if (options.version) this._write(`@version "${options.version}".\n`);
89
97
  this._prefixIRIs = Object.create(null);
90
98
  // Escaped prefix IRIs and names for the prefix matcher, computed once per prefix
91
99
  this._prefixPatterns = Object.create(null);
@@ -100,6 +108,7 @@ class N3Writer {
100
108
  // N-Triples and N-Quads are written in their canonical form
101
109
  this._escape = canonicalEscape, this._escapeAll = canonicalEscapeAll;
102
110
  this._characterReplacer = canonicalCharacterReplacer;
111
+ if (options.version) this._write(`VERSION "${options.version}"\n`);
103
112
  }
104
113
  }
105
114
 
@@ -115,6 +124,15 @@ class N3Writer {
115
124
  this._outputStream.write(string, 'utf8', callback);
116
125
  }
117
126
 
127
+ // ### `_endStatement` finishes a pending statement and closes an open graph block
128
+ _endStatement() {
129
+ if (this._subject !== null) {
130
+ this._write(this._inDefaultGraph ? '.\n' : '\n}\n');
131
+ this._subject = null;
132
+ this._graph = DEFAULTGRAPH;
133
+ }
134
+ }
135
+
118
136
  // ### `_writeQuad` writes the quad to the output stream
119
137
  _writeQuad(subject, predicate, object, graph, done) {
120
138
  try {
@@ -313,10 +331,7 @@ class N3Writer {
313
331
  if (typeof iri !== 'string') iri = iri.value;
314
332
  hasPrefixes = true;
315
333
  // Finish a possible pending quad
316
- if (this._subject !== null) {
317
- this._write(this._inDefaultGraph ? '.\n' : '\n}\n');
318
- this._subject = null, this._graph = '';
319
- }
334
+ this._endStatement();
320
335
  // Store and write the prefix
321
336
  this._prefixIRIs[iri] = prefix += ':';
322
337
  this._prefixPatterns[iri] = (0, _Util.escapeRegex)(iri);
@@ -390,10 +405,7 @@ class N3Writer {
390
405
  // ### `end` signals the end of the output stream
391
406
  end(done) {
392
407
  // Finish a possible pending quad
393
- if (this._subject !== null) {
394
- this._write(this._inDefaultGraph ? '.\n' : '\n}\n');
395
- this._subject = null;
396
- }
408
+ this._endStatement();
397
409
  // Disallow further writing
398
410
  this._write = this._blockedWrite;
399
411
 
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "n3",
3
- "version": "3.0.0-alpha.5",
3
+ "version": "3.0.0-alpha.7",
4
4
  "description": "Lightning fast, asynchronous, streaming Turtle / N3 / RDF library.",
5
5
  "author": "Ruben Verborgh <ruben.verborgh@gmail.com>",
6
6
  "keywords": [
package/src/N3Lexer.js CHANGED
@@ -81,8 +81,8 @@ export default class N3Lexer {
81
81
  this._blank = /_:((?:[0-9A-Z_a-z\xc0-\xd6\xd8-\xf6\xf8-\u02ff\u0370-\u037d\u037f-\u1fff\u200c\u200d\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff])(?:\.?[\-0-9A-Z_a-z\xb7\xc0-\xd6\xd8-\xf6\xf8-\u037d\u037f-\u1fff\u200c\u200d\u203f\u2040\u2070-\u218f\u2c00-\u2fef\u3001-\ud7ff\uf900-\ufdcf\ufdf0-\ufffd]|[\ud800-\udb7f][\udc00-\udfff])*)(?:[ \t]+|(?=\.?[,;:!\^\s#()\[\]\{\}"'<>]))/y;
82
82
  this._number = /[\-+]?(?:(\d+\.\d*|\.?\d+)[eE][\-+]?\d+|(?=\.?\d)\d*(?:(\.)\d+)?)(?=\.?[,;:!\^\s#()\[\]\{\}"'<>])/y;
83
83
  this._boolean = /(?:true|false)(?=[.,;!\^\s#()\[\]\{\}"'<>])/y;
84
- this._atKeyword = /@[a-z]+(?=[\s#<:])/iy;
85
- this._keyword = /(?:PREFIX|BASE|VERSION|GRAPH)(?=[\s#<])/iy;
84
+ this._atKeyword = /@[a-z]+(?=[\s#<:"'])/iy;
85
+ this._keyword = /(?:PREFIX|BASE|VERSION|GRAPH)(?=[\s#<"'])/iy;
86
86
  this._n3Verb = /(?:has|is|of)(?=[\s#()\[\]\{\}"'<>?_+\-0-9])/y;
87
87
  this._n3Id = /id(?=[\s#<])/y;
88
88
  this._shortPredicates = /a(?=[\s#()\[\]\{\}"'<>])/y;
@@ -101,6 +101,8 @@ export default class N3Lexer {
101
101
  if (!(key in lineModeRegExps) && this[key] instanceof RegExp)
102
102
  this[key] = invalidRegExp;
103
103
  }
104
+ // The only keyword in N-Triples and N-Quads is VERSION, which is case-sensitive
105
+ this._keyword = /VERSION(?=[\s#<"])/y;
104
106
  }
105
107
  // When not in line mode, enable N3 functionality by default
106
108
  else {
@@ -208,7 +210,7 @@ export default class N3Lexer {
208
210
  const line = this._line, firstChar = input[pos];
209
211
  let type = '', value = '', prefix = '',
210
212
  match = null, matchLength = 0, lexicalLength = 0,
211
- finalLineLength = 0, inconclusive = false;
213
+ finalLineLength = 0, inconclusive = false, tripleQuoted = false;
212
214
  switch (firstChar) {
213
215
  case '^':
214
216
  // A datatype marker separated from its type cannot be followed by another marker
@@ -293,7 +295,7 @@ export default class N3Lexer {
293
295
  value = match[1];
294
296
  // Try to find a literal wrapped in three pairs of quotes
295
297
  else {
296
- ({ value, matchLength, finalLineLength } = this._parseLiteral(input, pos));
298
+ ({ value, matchLength, finalLineLength, tripleQuoted } = this._parseLiteral(input, pos));
297
299
  if (value === null)
298
300
  return reportSyntaxError(this, input, pos);
299
301
  }
@@ -310,7 +312,7 @@ export default class N3Lexer {
310
312
  value = match[1];
311
313
  // Try to find a literal wrapped in three pairs of quotes
312
314
  else {
313
- ({ value, matchLength, finalLineLength } = this._parseLiteral(input, pos));
315
+ ({ value, matchLength, finalLineLength, tripleQuoted } = this._parseLiteral(input, pos));
314
316
  if (value === null)
315
317
  return reportSyntaxError(this, input, pos);
316
318
  }
@@ -550,10 +552,15 @@ export default class N3Lexer {
550
552
  if (finalLineLength) {
551
553
  token = {
552
554
  type, value, prefix, line, start,
553
- end: finalLineLength, endLine: this._line,
555
+ end: finalLineLength, endLine: this._line, tripleQuoted,
554
556
  };
555
557
  callback(null, token);
556
558
  }
559
+ // Triple-quoted strings are marked, since version declarations do not allow them
560
+ else if (tripleQuoted) {
561
+ token = { type, value, prefix, line, start, end: start + length, tripleQuoted };
562
+ callback(null, token);
563
+ }
557
564
  else
558
565
  token = emitToken(type, value, prefix, line, start, lexicalLength || length);
559
566
  this.previousToken = token;
@@ -705,19 +712,44 @@ export default class N3Lexer {
705
712
  break;
706
713
  this._line += lineCount;
707
714
  const finalLineLength = lineCount === 0 ? 0 : lines[lines.length - 1].length + openingLength;
708
- return { value: this._unescape(raw, stringEscapeReplacements), matchLength, finalLineLength };
715
+ return {
716
+ value: this._unescape(raw, stringEscapeReplacements), matchLength, finalLineLength,
717
+ tripleQuoted: openingLength === 3,
718
+ };
709
719
  }
710
720
  closingPos++;
711
721
  }
712
722
  this._literalClosingPos = input.length - pos - openingLength + 1;
713
723
  }
714
- return { value: '', matchLength: 0, finalLineLength: 0 };
724
+ return { value: '', matchLength: 0, finalLineLength: 0, tripleQuoted: false };
725
+ }
726
+
727
+ // ### `_tryTokenizeToEnd` tokenizes as far as possible, reporting failures through the callback
728
+ _tryTokenizeToEnd(callback, inputFinished) {
729
+ // Keep track of errors thrown by the callback, which must reach the caller unchanged
730
+ let callbackError;
731
+ try {
732
+ this._tokenizeToEnd((error, token) => {
733
+ try {
734
+ return callback(error, token);
735
+ }
736
+ catch (thrown) {
737
+ throw (callbackError = thrown);
738
+ }
739
+ }, inputFinished);
740
+ }
741
+ catch (error) {
742
+ // Matching an extremely long token can exhaust the regular expression stack
743
+ if (error === callbackError || !(error instanceof RangeError))
744
+ throw error;
745
+ callback(this._syntaxError(null, `Token too long on line ${this._line}.`));
746
+ }
715
747
  }
716
748
 
717
749
  // ### `_syntaxError` creates a syntax error for the given issue
718
- _syntaxError(issue) {
750
+ _syntaxError(issue, message = `Unexpected "${issue}" on line ${this._line}.`) {
719
751
  this._input = null;
720
- const err = new Error(`Unexpected "${issue}" on line ${this._line}.`);
752
+ const err = new Error(message);
721
753
  err.context = {
722
754
  token: undefined,
723
755
  line: this._line,
@@ -760,13 +792,13 @@ export default class N3Lexer {
760
792
  if (typeof callback === 'function')
761
793
  queueMicrotask(() => {
762
794
  if (this._tokenization === tokenization)
763
- this._tokenizeToEnd(callback, true);
795
+ this._tryTokenizeToEnd(callback, true);
764
796
  });
765
797
  // If no callback was passed, tokenize synchronously and return
766
798
  else {
767
799
  const tokens = [];
768
800
  let error;
769
- this._tokenizeToEnd((e, t) => e ? (error = e) : tokens.push(t), true);
801
+ this._tryTokenizeToEnd((e, t) => e ? (error = e) : tokens.push(t), true);
770
802
  if (error) throw error;
771
803
  return tokens;
772
804
  }
@@ -793,7 +825,7 @@ export default class N3Lexer {
793
825
  // Tokenize as far as possible. When a previous attempt left a long unfinished token,
794
826
  // wait until the buffered input has doubled, so the token is not rescanned for every chunk.
795
827
  if (this._input.length >= retryLength) {
796
- this._tokenizeToEnd(callback, false);
828
+ this._tryTokenizeToEnd(callback, false);
797
829
  retryLength = this._input !== null && this._input.length > MIN_RESCAN_LENGTH ?
798
830
  2 * this._input.length : 0;
799
831
  }
@@ -807,7 +839,7 @@ export default class N3Lexer {
807
839
  if (rest)
808
840
  this._input = typeof this._input === 'string' ? this._input + rest : rest;
809
841
  if (typeof this._input === 'string')
810
- this._tokenizeToEnd(callback, true);
842
+ this._tryTokenizeToEnd(callback, true);
811
843
  }
812
844
  });
813
845
  input.on('error', error => {
package/src/N3Parser.js CHANGED
@@ -4,6 +4,10 @@ import N3DataFactory from './N3DataFactory';
4
4
  import namespaces from './IRIs';
5
5
 
6
6
  let blankNodePrefix = 0;
7
+ // Detects `.` and `..` path segments in an IRI
8
+ const dotSegments = /(^|\/)\.\.?($|[/#?])/;
9
+ // Detects `.` and `..` segments in the path of an IRI, ignoring its query and fragment
10
+ const pathDotSegments = /^[^?#]*(?:^|\/)\.\.?(?:$|[/#?])/;
7
11
 
8
12
  // ## Constructor
9
13
  export default class N3Parser {
@@ -84,6 +88,7 @@ export default class N3Parser {
84
88
  if (!baseIRI) {
85
89
  this._base = '';
86
90
  this._basePath = '';
91
+ this._basePathHasDotSegments = false;
87
92
  }
88
93
  else {
89
94
  // Remove fragment if present
@@ -104,6 +109,8 @@ export default class N3Parser {
104
109
  // relative IRIs merge under the path '/' (RFC 3986 §5.2.3)
105
110
  if (baseIRI[2] !== undefined && (base.length === this._baseRoot.length || base[this._baseRoot.length] === '?'))
106
111
  this._basePath = `${this._baseRoot}/`;
112
+ // Check once whether resolving against the base path needs to remove dot segments
113
+ this._basePathHasDotSegments = dotSegments.test(this._basePath);
107
114
  }
108
115
  }
109
116
 
@@ -141,7 +148,8 @@ export default class N3Parser {
141
148
  // Prefix and base declarations are scoped to their formula,
142
149
  // so record prefix changes to undo them when the formula ends
143
150
  if (type === 'formula') {
144
- context.base = [this._base, this._basePath, this._baseRoot, this._baseScheme];
151
+ context.base = [this._base, this._basePath, this._baseRoot, this._baseScheme,
152
+ this._basePathHasDotSegments];
145
153
  this._prefixChanges = [];
146
154
  }
147
155
  this._contextStack.push(context);
@@ -183,7 +191,8 @@ export default class N3Parser {
183
191
  this._inversePredicate = context.inverse;
184
192
  this._expectOf = context.expectOf;
185
193
  if (type === 'formula')
186
- [this._base, this._basePath, this._baseRoot, this._baseScheme] = context.base;
194
+ [this._base, this._basePath, this._baseRoot, this._baseScheme,
195
+ this._basePathHasDotSegments] = context.base;
187
196
  if (this._prefixChanges !== context.prefixChanges) {
188
197
  undoChanges(this._prefixes, this._prefixChanges);
189
198
  this._prefixChanges = context.prefixChanges;
@@ -1112,7 +1121,8 @@ export default class N3Parser {
1112
1121
  _readVersion(token) {
1113
1122
  if (token.type !== 'literal')
1114
1123
  return this._error('Expected literal to follow version declaration', token);
1115
- if ((token.end - token.start) !== token.value.length + 2)
1124
+ // Only short strings are allowed, so no numbers or booleans (which have a datatype prefix)
1125
+ if (token.prefix !== '' || token.tripleQuoted)
1116
1126
  return this._error('Version declarations must use single quotes', token);
1117
1127
  this._versionCallback(token.value);
1118
1128
  if (!this._isValidVersion(token.value))
@@ -1576,14 +1586,19 @@ export default class N3Parser {
1576
1586
  // Resolve all other IRIs at the base IRI's path
1577
1587
  default:
1578
1588
  // Relative IRIs cannot contain a colon in the first path segment
1579
- return (/^[^/:]*:/.test(iri)) ? null : this._removeDotSegments(this._basePath + iri);
1589
+ if (/^[^/:]*:/.test(iri))
1590
+ return null;
1591
+ // Only scan the joined IRI for dot segments if either part can contain them,
1592
+ // as the base path can be long and the joined IRI would need to be copied
1593
+ return this._basePathHasDotSegments || pathDotSegments.test(iri) ?
1594
+ this._removeDotSegments(this._basePath + iri) : this._basePath + iri;
1580
1595
  }
1581
1596
  }
1582
1597
 
1583
1598
  // ### `_removeDotSegments` resolves './' and '../' path segments in an IRI as per RFC3986
1584
1599
  _removeDotSegments(iri) {
1585
1600
  // Don't modify the IRI if it does not contain any dot segments
1586
- if (!/(^|\/)\.\.?($|[/#?])/.test(iri))
1601
+ if (!dotSegments.test(iri))
1587
1602
  return iri;
1588
1603
 
1589
1604
  // Start with an imaginary slash before the IRI in order to resolve trailing './' and '../'