eyeprolog 1.3.40 → 1.3.42
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +12 -4
- package/examples/book/README.md +2 -1
- package/examples/book/chapter-38/02-look_ahead.pl +1 -1
- package/examples/book/chapter-40/02-program.pl +4 -4
- package/examples/book/chapter-40/03-program-2.pl +5 -0
- package/package.json +1 -1
- package/src/io.js +1 -1
- package/src/iso.js +17 -4
- package/src/parser.js +126 -61
- package/src/program.js +2 -2
- package/src/quads.js +56 -16
- package/src/solver.js +2 -2
- package/src/standard-library.js +1 -1
- package/src/write.js +59 -6
- package/test/conformance/ISO-COMPLIANCE.md +24 -4
- package/test/conformance/ISO-IMPLEMENTATION-DEFINED.md +9 -8
- package/test/conformance/ISO-MATRIX.md +6 -4
- package/test/conformance/README.md +3 -4
- package/test/conformance/WG17-SYNTAX-STATUS.md +2 -2
- package/test/conformance/cases/iso/dcg_module_nonterminal_indicator.pl +1 -1
- package/test/conformance/expected/iso/exceptions_and_flags.pl +1 -1
- package/test/conformance/expected/iso/wg17_syntax_high_risk.pl +2 -2
- package/test/conformance/wg17-syntax-cases.json +99 -90
- package/test/fixtures/README.md +7 -0
- package/test/run-iso-strict.mjs +20 -0
- package/test/run-regression.mjs +76 -7
- package/the-art-of-eyeprolog.md +67 -34
- package/why-eyeprolog.md +14 -13
package/README.md
CHANGED
|
@@ -221,10 +221,18 @@ newly upgraded cases run directly against the upstream Codex expectation.
|
|
|
221
221
|
EyeProlog does not yet claim independent certification or closure of every
|
|
222
222
|
normative Part 1 requirement.
|
|
223
223
|
|
|
224
|
-
##
|
|
224
|
+
## Module and definite clause grammar compatibility profiles
|
|
225
225
|
|
|
226
|
-
EyeProlog
|
|
227
|
-
`
|
|
226
|
+
Normal EyeProlog supports the widely used `module/2`, `use_module/1-2`,
|
|
227
|
+
`meta_predicate/1`, and `Module:Goal` interface reflected in later WG17 module
|
|
228
|
+
amendment work. This is a practical module compatibility profile; EyeProlog
|
|
229
|
+
does **not** currently claim a clause-by-clause implementation or certification
|
|
230
|
+
of the complete ISO/IEC 13211-2:2000 module model.
|
|
231
|
+
|
|
232
|
+
Definite clause grammar support follows the ISO/IEC TS 13211-3 grammar-rule and
|
|
233
|
+
`phrase/2-3` model and is exercised by the conformance corpus. As with Part 1,
|
|
234
|
+
that implementation evidence is not an independent certification of every
|
|
235
|
+
Part 3 requirement. For example:
|
|
228
236
|
|
|
229
237
|
```prolog
|
|
230
238
|
sentence --> [hello], noun.
|
|
@@ -254,7 +262,7 @@ EyeProlog also adds 128 public library predicate indicators to its 129-entry ISO
|
|
|
254
262
|
profile. **88 are implemented entirely as ordinary Prolog clauses** in focused
|
|
255
263
|
modules under `src/lib/`; the remaining control predicates and finite-domain
|
|
256
264
|
`library(clpz)` kernel use backtrackable host support.
|
|
257
|
-
They are
|
|
265
|
+
They are ordinary Prolog modules using EyeProlog's documented module compatibility surface, loaded explicitly by purpose, such as
|
|
258
266
|
`library(lists)`, `library(lambda)`, `library(strings)`, `library(aggregate)`, or
|
|
259
267
|
`library(clpz)`.
|
|
260
268
|
Portable text
|
package/examples/book/README.md
CHANGED
|
@@ -234,7 +234,7 @@ npm run generate
|
|
|
234
234
|
## Chapter 38: Language and ISO profile
|
|
235
235
|
|
|
236
236
|
- [01-city.pl](chapter-38/01-city.pl)
|
|
237
|
-
- [02-look_ahead.pl](chapter-38/02-look_ahead.pl) —
|
|
237
|
+
- [02-look_ahead.pl](chapter-38/02-look_ahead.pl) — Part 3-oriented definite clause grammars
|
|
238
238
|
|
|
239
239
|
## Chapter 39: Built-in predicates by programming role
|
|
240
240
|
|
|
@@ -249,3 +249,4 @@ npm run generate
|
|
|
249
249
|
|
|
250
250
|
- [01-color.pl](chapter-40/01-color.pl) — Embedded quad tests
|
|
251
251
|
- [02-program.pl](chapter-40/02-program.pl)
|
|
252
|
+
- [03-program-2.pl](chapter-40/03-program-2.pl)
|
|
@@ -1,2 +1,2 @@
|
|
|
1
|
-
% From The Art of EyeProlog, Chapter 38 —
|
|
1
|
+
% From The Art of EyeProlog, Chapter 38 — Part 3-oriented definite clause grammars.
|
|
2
2
|
look_ahead(X), [X] --> [X].
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
% From The Art of EyeProlog, Chapter 40.
|
|
2
|
-
|
|
3
|
-
|
|
4
|
-
|
|
5
|
-
|
|
2
|
+
?- read(X).
|
|
3
|
+
inputs("1."), X = 1, unexpected.
|
|
4
|
+
inputs("1."), peeks(" "), X = 1.
|
|
5
|
+
inputs("1. "), peeks(" "), X = 1, unexpected.
|
package/package.json
CHANGED
package/src/io.js
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
// Synchronous ISO stream state shared by a solver and all of its inner solvers.
|
|
2
2
|
import { BufferCtor, fs } from './platform.js';
|
|
3
3
|
|
|
4
|
-
const INVALID_UTF8_SENTINEL = '\udc00';
|
|
4
|
+
export const INVALID_UTF8_SENTINEL = '\udc00';
|
|
5
5
|
|
|
6
6
|
export class InvalidCharacterEncodingError extends Error {
|
|
7
7
|
constructor() {
|
package/src/iso.js
CHANGED
|
@@ -12,6 +12,7 @@ import {
|
|
|
12
12
|
} from './parser.js';
|
|
13
13
|
import { formatTermForWrite } from './write.js';
|
|
14
14
|
import { emptyTerminalSequence, expandDcgBody, isListOrPartialList, validateDcgEmbeddedGoals } from './dcg.js';
|
|
15
|
+
import { INVALID_UTF8_SENTINEL } from './io.js';
|
|
15
16
|
import {
|
|
16
17
|
characterCodeConstantEnd, continuesGraphicToken, isTerminatingFullStop, quotedEscapeEnd,
|
|
17
18
|
} from './syntax-scan.js';
|
|
@@ -661,6 +662,11 @@ function* currentPrologFlagBuiltin({ solver, goal, env }) {
|
|
|
661
662
|
throw new PrologError('domain_error(prolog_flag)', flag);
|
|
662
663
|
}
|
|
663
664
|
for (const [name, definition] of solver.prologFlags) {
|
|
665
|
+
// ISO 7.11.1.1: when bounded=false, max_integer and min_integer have no
|
|
666
|
+
// current value and current_prolog_flag/2 must therefore not enumerate
|
|
667
|
+
// them. The definitions remain registered so attempts to change these
|
|
668
|
+
// non-changeable flags still receive the normal flag error handling.
|
|
669
|
+
if (definition.value == null) continue;
|
|
664
670
|
const next = env.clone();
|
|
665
671
|
if (unify(goal.args[0], atom(name), next) && unify(goal.args[1], definition.value, next)) yield next;
|
|
666
672
|
}
|
|
@@ -1089,6 +1095,13 @@ function* termTextCandidates(stream, solver) {
|
|
|
1089
1095
|
let quote = null, lineComment = false, blockComment = false;
|
|
1090
1096
|
for (let i = stream.position; i < source.length; i++) {
|
|
1091
1097
|
const ch = source[i], next = source[i + 1];
|
|
1098
|
+
// Text streams preserve invalid UTF-8 bytes as an impossible Unicode
|
|
1099
|
+
// sentinel. read/1-2 and read_term/2-3 must surface the same character
|
|
1100
|
+
// representation error as get_char/1-2 instead of misclassifying the
|
|
1101
|
+
// undecodable byte as malformed Prolog syntax (issue #64).
|
|
1102
|
+
if (stream.strictUtf8 && ch === INVALID_UTF8_SENTINEL) {
|
|
1103
|
+
throw new PrologError('representation_error(character)');
|
|
1104
|
+
}
|
|
1092
1105
|
if (lineComment) { if (ch === '\n') lineComment = false; continue; }
|
|
1093
1106
|
if (blockComment) { if (ch === '*' && next === '/') { blockComment = false; i++; } continue; }
|
|
1094
1107
|
if (quote) {
|
|
@@ -1305,10 +1318,10 @@ function* readTermBuiltin({ solver, goal, env }) {
|
|
|
1305
1318
|
yield next;
|
|
1306
1319
|
}
|
|
1307
1320
|
function defaultTermWriteOptions(mode) {
|
|
1308
|
-
if (mode === 'writeq') return { quoted: true, ignoreOps: false, numbervars: true, variableNames: new Map(), compact: true, operatorAtomsAsArgs: true, doubleQuotes: null };
|
|
1309
|
-
if (mode === 'canonical') return { quoted: true, ignoreOps: true, numbervars: false, variableNames: new Map(), compact: true, operatorAtomsAsArgs: true, doubleQuotes: null };
|
|
1310
|
-
if (mode === 'write_term') return { quoted: false, ignoreOps: false, numbervars: false, variableNames: new Map(), compact: true, operatorAtomsAsArgs: true, doubleQuotes: null };
|
|
1311
|
-
return { quoted: false, ignoreOps: false, numbervars: true, variableNames: new Map(), compact: true, operatorAtomsAsArgs: true, doubleQuotes: null };
|
|
1321
|
+
if (mode === 'writeq') return { quoted: true, ignoreOps: false, numbervars: true, variableNames: new Map(), compact: true, minimalOperatorSpacing: true, operatorAtomsAsArgs: true, doubleQuotes: null };
|
|
1322
|
+
if (mode === 'canonical') return { quoted: true, ignoreOps: true, numbervars: false, variableNames: new Map(), compact: true, minimalOperatorSpacing: true, operatorAtomsAsArgs: true, doubleQuotes: null };
|
|
1323
|
+
if (mode === 'write_term') return { quoted: false, ignoreOps: false, numbervars: false, variableNames: new Map(), compact: true, minimalOperatorSpacing: true, operatorAtomsAsArgs: true, doubleQuotes: null };
|
|
1324
|
+
return { quoted: false, ignoreOps: false, numbervars: true, variableNames: new Map(), compact: true, minimalOperatorSpacing: true, operatorAtomsAsArgs: true, doubleQuotes: null };
|
|
1312
1325
|
}
|
|
1313
1326
|
|
|
1314
1327
|
function writeOptionBoolean(value, env, option) {
|
package/src/parser.js
CHANGED
|
@@ -207,7 +207,11 @@ class Parser {
|
|
|
207
207
|
this.strictIso = options.isoStrict === true;
|
|
208
208
|
this.parserFlagState = options.parserFlagState ?? {
|
|
209
209
|
doubleQuotes: options.doubleQuotes ?? 'chars',
|
|
210
|
+
charConversion: 'on',
|
|
211
|
+
charConversions: new Map(),
|
|
210
212
|
};
|
|
213
|
+
this.parserFlagState.charConversion ??= 'on';
|
|
214
|
+
this.parserFlagState.charConversions ??= new Map();
|
|
211
215
|
if (!['chars', 'codes', 'atom'].includes(this.parserFlagState.doubleQuotes)) {
|
|
212
216
|
throw new Error(`invalid double_quotes parser flag: ${this.parserFlagState.doubleQuotes}`);
|
|
213
217
|
}
|
|
@@ -223,6 +227,26 @@ class Parser {
|
|
|
223
227
|
this.previousToken = null;
|
|
224
228
|
this.token = this.nextToken();
|
|
225
229
|
}
|
|
230
|
+
convertCharacter(character) {
|
|
231
|
+
if (this.parserFlagState.charConversion !== 'on' || !character) return character;
|
|
232
|
+
return this.parserFlagState.charConversions.get(character) ?? character;
|
|
233
|
+
}
|
|
234
|
+
rawPeek(offset = 0) {
|
|
235
|
+
return this.source[this.pos + offset] ?? '';
|
|
236
|
+
}
|
|
237
|
+
rawTake() {
|
|
238
|
+
const ch = this.rawPeek();
|
|
239
|
+
if (ch) {
|
|
240
|
+
this.pos++;
|
|
241
|
+
if (ch === '\n') this.line++;
|
|
242
|
+
}
|
|
243
|
+
return ch;
|
|
244
|
+
}
|
|
245
|
+
convertedSlice(start, end) {
|
|
246
|
+
let out = '';
|
|
247
|
+
for (const ch of this.source.slice(start, end)) out += this.convertCharacter(ch);
|
|
248
|
+
return out;
|
|
249
|
+
}
|
|
226
250
|
terminatingFullStop(index = this.pos) {
|
|
227
251
|
// read/1 tries each possible full stop in turn. While parsing a later
|
|
228
252
|
// candidate, a preceding dot after a graphic character belongs to that
|
|
@@ -231,12 +255,12 @@ class Parser {
|
|
|
231
255
|
// designated read-term candidate.
|
|
232
256
|
if (this.readTermEnd != null) {
|
|
233
257
|
if (index === this.readTermEnd) return true;
|
|
234
|
-
if (index < this.readTermEnd && continuesGraphicToken(this.source, index)) {
|
|
235
|
-
const next = this.source[index + 1] ?? '';
|
|
258
|
+
if (index < this.readTermEnd && continuesGraphicToken(this.source, index, (ch) => this.convertCharacter(ch))) {
|
|
259
|
+
const next = this.convertCharacter(this.source[index + 1] ?? '');
|
|
236
260
|
if (next === '%' || /^[\u0009-\u000d\u0020]$/.test(next)) return false;
|
|
237
261
|
}
|
|
238
262
|
}
|
|
239
|
-
return isTerminatingFullStop(this.source, index);
|
|
263
|
+
return isTerminatingFullStop(this.source, index, (ch) => this.convertCharacter(ch));
|
|
240
264
|
}
|
|
241
265
|
defineOperator(priority, specifier, name) {
|
|
242
266
|
defineParserOperator(this, priority, specifier, name);
|
|
@@ -273,12 +297,30 @@ class Parser {
|
|
|
273
297
|
for (const name of names) this.defineOperator(priority, specifierTerm.name, name);
|
|
274
298
|
return true;
|
|
275
299
|
}
|
|
276
|
-
applyParserFlagDirective(directive) {
|
|
277
|
-
if (directive.type !== 'compound' || directive.
|
|
278
|
-
|
|
279
|
-
|
|
280
|
-
|
|
281
|
-
|
|
300
|
+
applyParserFlagDirective(directive, line = this.line) {
|
|
301
|
+
if (directive.type !== 'compound' || directive.arity !== 2) return;
|
|
302
|
+
if (directive.name === 'set_prolog_flag') {
|
|
303
|
+
const [flag, value] = directive.args;
|
|
304
|
+
if (flag.type === 'atom' && flag.name === 'double_quotes' &&
|
|
305
|
+
value.type === 'atom' && ['chars', 'codes', 'atom'].includes(value.name)) {
|
|
306
|
+
this.parserFlagState.doubleQuotes = value.name;
|
|
307
|
+
} else if (flag.type === 'atom' && flag.name === 'char_conversion' &&
|
|
308
|
+
value.type === 'atom' && ['on', 'off'].includes(value.name)) {
|
|
309
|
+
this.parserFlagState.charConversion = value.name;
|
|
310
|
+
}
|
|
311
|
+
return;
|
|
312
|
+
}
|
|
313
|
+
if (directive.name === 'char_conversion') {
|
|
314
|
+
const [input, output] = directive.args;
|
|
315
|
+
const validCharacter = (term) => term.type === 'atom' && Array.from(term.name).length === 1;
|
|
316
|
+
if (input.type === 'var' || output.type === 'var') {
|
|
317
|
+
throw new Error(`parse line ${line}: char_conversion/2 arguments must be instantiated characters`);
|
|
318
|
+
}
|
|
319
|
+
if (!validCharacter(input) || !validCharacter(output)) {
|
|
320
|
+
throw new Error(`parse line ${line}: char_conversion/2 requires one-character atoms`);
|
|
321
|
+
}
|
|
322
|
+
if (input.name === output.name) this.parserFlagState.charConversions.delete(input.name);
|
|
323
|
+
else this.parserFlagState.charConversions.set(input.name, output.name);
|
|
282
324
|
}
|
|
283
325
|
}
|
|
284
326
|
applyImportedLibraryOperators(directive) {
|
|
@@ -301,46 +343,45 @@ class Parser {
|
|
|
301
343
|
return null;
|
|
302
344
|
}
|
|
303
345
|
peek(offset = 0) {
|
|
304
|
-
return this.
|
|
346
|
+
return this.convertCharacter(this.rawPeek(offset));
|
|
305
347
|
}
|
|
306
348
|
take() {
|
|
307
|
-
const
|
|
308
|
-
|
|
349
|
+
const raw = this.rawPeek();
|
|
350
|
+
const ch = this.convertCharacter(raw);
|
|
351
|
+
if (raw) {
|
|
309
352
|
this.pos++;
|
|
310
|
-
if (
|
|
353
|
+
if (raw === '\n') this.line++;
|
|
311
354
|
}
|
|
312
355
|
return ch;
|
|
313
356
|
}
|
|
314
357
|
skipWhitespaceAndComments() {
|
|
315
|
-
const source = this.source;
|
|
316
|
-
const len = source.length;
|
|
317
358
|
while (true) {
|
|
318
|
-
while (this.
|
|
319
|
-
const
|
|
320
|
-
if (!isWhitespaceCode(
|
|
321
|
-
|
|
322
|
-
this.pos++;
|
|
359
|
+
while (this.rawPeek()) {
|
|
360
|
+
const ch = this.peek();
|
|
361
|
+
if (!isWhitespaceCode(ch.charCodeAt(0))) break;
|
|
362
|
+
this.take();
|
|
323
363
|
}
|
|
324
|
-
if (
|
|
325
|
-
while (this.
|
|
364
|
+
if (this.peek() === '%') {
|
|
365
|
+
while (this.rawPeek() && this.peek() !== '\n') this.take();
|
|
326
366
|
continue;
|
|
327
367
|
}
|
|
328
|
-
if (
|
|
368
|
+
if (this.peek() === '/' && this.peek(1) === '*') {
|
|
329
369
|
const line = this.line;
|
|
330
|
-
this.
|
|
331
|
-
|
|
332
|
-
|
|
333
|
-
|
|
334
|
-
|
|
335
|
-
|
|
336
|
-
this.pos += 2;
|
|
370
|
+
this.take();
|
|
371
|
+
this.take();
|
|
372
|
+
while (this.rawPeek() && !(this.peek() === '*' && this.peek(1) === '/')) this.take();
|
|
373
|
+
if (!this.rawPeek()) throw new Error(`parse line ${line}: unterminated block comment`);
|
|
374
|
+
this.take();
|
|
375
|
+
this.take();
|
|
337
376
|
continue;
|
|
338
377
|
}
|
|
339
378
|
break;
|
|
340
379
|
}
|
|
341
380
|
}
|
|
342
381
|
readEscape(line, options = {}) {
|
|
343
|
-
const
|
|
382
|
+
const takeChar = () => options.raw ? this.rawTake() : this.take();
|
|
383
|
+
const peekChar = () => options.raw ? this.rawPeek() : this.peek();
|
|
384
|
+
const escaped = takeChar();
|
|
344
385
|
if (!escaped) throw new Error(`parse line ${line}: unterminated escape sequence`);
|
|
345
386
|
|
|
346
387
|
// ISO 6.4.2 permits a continuation escape only inside quoted tokens: a
|
|
@@ -350,9 +391,9 @@ class Parser {
|
|
|
350
391
|
if (options.allowContinuation !== false) return '';
|
|
351
392
|
throw new Error(`parse line ${line}: bad escape sequence`);
|
|
352
393
|
}
|
|
353
|
-
if (escaped === '\r' &&
|
|
394
|
+
if (escaped === '\r' && peekChar() === '\n') {
|
|
354
395
|
if (options.allowContinuation !== false) {
|
|
355
|
-
|
|
396
|
+
takeChar();
|
|
356
397
|
return '';
|
|
357
398
|
}
|
|
358
399
|
throw new Error(`parse line ${line}: bad escape sequence`);
|
|
@@ -363,8 +404,8 @@ class Parser {
|
|
|
363
404
|
|
|
364
405
|
if (escaped === 'x') {
|
|
365
406
|
let digits = '';
|
|
366
|
-
while (/^[0-9A-Fa-f]$/.test(
|
|
367
|
-
if (!digits ||
|
|
407
|
+
while (/^[0-9A-Fa-f]$/.test(peekChar())) digits += takeChar();
|
|
408
|
+
if (!digits || takeChar() !== '\\') throw new Error(`parse line ${line}: bad hexadecimal escape`);
|
|
368
409
|
const code = Number.parseInt(digits, 16);
|
|
369
410
|
if (code > 0x10ffff || (code >= 0xd800 && code <= 0xdfff)) {
|
|
370
411
|
throw new Error(`parse line ${line}: character escape out of range`);
|
|
@@ -373,8 +414,8 @@ class Parser {
|
|
|
373
414
|
}
|
|
374
415
|
if (/^[0-7]$/.test(escaped)) {
|
|
375
416
|
let digits = escaped;
|
|
376
|
-
while (/^[0-7]$/.test(
|
|
377
|
-
if (
|
|
417
|
+
while (/^[0-7]$/.test(peekChar())) digits += takeChar();
|
|
418
|
+
if (takeChar() !== '\\') throw new Error(`parse line ${line}: bad octal escape`);
|
|
378
419
|
const code = Number.parseInt(digits, 8);
|
|
379
420
|
if (code > 0x10ffff || (code >= 0xd800 && code <= 0xdfff)) {
|
|
380
421
|
throw new Error(`parse line ${line}: character escape out of range`);
|
|
@@ -411,7 +452,7 @@ class Parser {
|
|
|
411
452
|
this.take();
|
|
412
453
|
while (isGraphicAtomCode(this.peek().charCodeAt(0)) &&
|
|
413
454
|
!this.terminatingFullStop()) this.take();
|
|
414
|
-
return { type: TOK.ATOM, text: this.
|
|
455
|
+
return { type: TOK.ATOM, text: this.convertedSlice(start, this.pos), line };
|
|
415
456
|
}
|
|
416
457
|
if (ch === '!') {
|
|
417
458
|
this.take();
|
|
@@ -442,20 +483,29 @@ class Parser {
|
|
|
442
483
|
}
|
|
443
484
|
|
|
444
485
|
if (ch === '"' || ch === "'") {
|
|
486
|
+
// Whether an input character is quoted is determined from the target
|
|
487
|
+
// source before character conversion (ISO 8.14.1 note 1). A literal
|
|
488
|
+
// quote therefore protects its contents from Convc; if a conversion
|
|
489
|
+
// itself produces a quote token, the following raw characters remain
|
|
490
|
+
// subject to conversion while that resulting token is parsed.
|
|
491
|
+
const rawOpening = this.rawPeek();
|
|
445
492
|
const quote = this.take();
|
|
493
|
+
const literalQuote = rawOpening === quote;
|
|
494
|
+
const peekQuoted = () => literalQuote ? this.rawPeek() : this.peek();
|
|
495
|
+
const takeQuoted = () => literalQuote ? this.rawTake() : this.take();
|
|
446
496
|
let text = '';
|
|
447
497
|
while (true) {
|
|
448
|
-
if (!
|
|
449
|
-
let value =
|
|
498
|
+
if (!peekQuoted()) throw new Error(`parse line ${line}: unterminated quoted term`);
|
|
499
|
+
let value = takeQuoted();
|
|
450
500
|
if (value === quote) {
|
|
451
|
-
if (
|
|
452
|
-
|
|
501
|
+
if (peekQuoted() === quote) {
|
|
502
|
+
takeQuoted();
|
|
453
503
|
value = quote;
|
|
454
504
|
} else {
|
|
455
505
|
break;
|
|
456
506
|
}
|
|
457
|
-
} else if (value === '\\' &&
|
|
458
|
-
value = this.readEscape(line);
|
|
507
|
+
} else if (value === '\\' && peekQuoted()) {
|
|
508
|
+
value = this.readEscape(line, { raw: literalQuote });
|
|
459
509
|
} else if (value !== ' ' && isWhitespaceCode(value.charCodeAt(0))) {
|
|
460
510
|
// ISO 6.4.2.1 allows an ordinary space in a quoted character, but
|
|
461
511
|
// not literal layout characters such as tab or newline. Newlines
|
|
@@ -489,17 +539,20 @@ class Parser {
|
|
|
489
539
|
(this.peek(2) !== "'" || this.peek(3) === "'") &&
|
|
490
540
|
!(this.peek(2) === '\\' && this.peek(3) === '\n');
|
|
491
541
|
if (startsQuotedCharacter) {
|
|
542
|
+
const rawQuotedCharacter = this.rawPeek() === '0' && this.rawPeek(1) === "'";
|
|
492
543
|
this.take();
|
|
493
544
|
this.take();
|
|
494
|
-
|
|
545
|
+
const takeCharacter = () => rawQuotedCharacter ? this.rawTake() : this.take();
|
|
546
|
+
const peekCharacter = () => rawQuotedCharacter ? this.rawPeek() : this.peek();
|
|
547
|
+
let value = takeCharacter();
|
|
495
548
|
if (value) {
|
|
496
549
|
const firstCode = value.charCodeAt(0);
|
|
497
550
|
if (firstCode >= 0xd800 && firstCode <= 0xdbff) {
|
|
498
|
-
const secondCode =
|
|
551
|
+
const secondCode = peekCharacter().charCodeAt(0);
|
|
499
552
|
if (secondCode < 0xdc00 || secondCode > 0xdfff) {
|
|
500
553
|
throw new Error(`parse line ${line}: bad character code constant`);
|
|
501
554
|
}
|
|
502
|
-
value +=
|
|
555
|
+
value += takeCharacter();
|
|
503
556
|
} else if (firstCode >= 0xdc00 && firstCode <= 0xdfff) {
|
|
504
557
|
throw new Error(`parse line ${line}: bad character code constant`);
|
|
505
558
|
}
|
|
@@ -512,10 +565,10 @@ class Parser {
|
|
|
512
565
|
// apostrophe is doubled just as it is inside a quoted atom. Thus
|
|
513
566
|
// 0''' is one numeric token denoting character code 39, while the
|
|
514
567
|
// undoubled 0'' is not a complete single quoted character.
|
|
515
|
-
if (
|
|
516
|
-
|
|
568
|
+
if (peekCharacter() !== "'") throw new Error(`parse line ${line}: bad character code constant`);
|
|
569
|
+
takeCharacter();
|
|
517
570
|
} else if (value === '\\') {
|
|
518
|
-
value = this.readEscape(line, { allowContinuation: false });
|
|
571
|
+
value = this.readEscape(line, { allowContinuation: false, raw: rawQuotedCharacter });
|
|
519
572
|
}
|
|
520
573
|
const code = value.codePointAt(0);
|
|
521
574
|
return { type: TOK.NUMBER, text: String(negative ? -code : code), line };
|
|
@@ -550,14 +603,14 @@ class Parser {
|
|
|
550
603
|
// followed by the name E9 and is not a valid term without an operator.
|
|
551
604
|
if (hasFraction && (this.peek() === 'e' || this.peek() === 'E')) {
|
|
552
605
|
let idx = this.pos + 1;
|
|
553
|
-
if (
|
|
554
|
-
if (isDigitCode((this.source[idx] ?? '').charCodeAt(0))) {
|
|
606
|
+
if (['+', '-'].includes(this.convertCharacter(this.source[idx] ?? ''))) idx++;
|
|
607
|
+
if (isDigitCode(this.convertCharacter(this.source[idx] ?? '').charCodeAt(0))) {
|
|
555
608
|
this.take();
|
|
556
609
|
if (this.peek() === '+' || this.peek() === '-') this.take();
|
|
557
610
|
while (isDigitCode(this.peek().charCodeAt(0))) this.take();
|
|
558
611
|
}
|
|
559
612
|
}
|
|
560
|
-
let text = this.
|
|
613
|
+
let text = this.convertedSlice(start, this.pos);
|
|
561
614
|
if (!hasFraction) text = BigInt(text).toString();
|
|
562
615
|
else text = finiteFloatTokenText(text);
|
|
563
616
|
return { type: TOK.NUMBER, text, line };
|
|
@@ -567,7 +620,7 @@ class Parser {
|
|
|
567
620
|
const start = this.pos;
|
|
568
621
|
this.take();
|
|
569
622
|
while (isNameContinueCode(this.peek().charCodeAt(0))) this.take();
|
|
570
|
-
const text = this.
|
|
623
|
+
const text = this.convertedSlice(start, this.pos);
|
|
571
624
|
return { type: TOK.VAR, text, line };
|
|
572
625
|
}
|
|
573
626
|
|
|
@@ -575,7 +628,7 @@ class Parser {
|
|
|
575
628
|
const start = this.pos;
|
|
576
629
|
this.take();
|
|
577
630
|
while (isNameContinueCode(this.peek().charCodeAt(0))) this.take();
|
|
578
|
-
return { type: TOK.ATOM, text: this.
|
|
631
|
+
return { type: TOK.ATOM, text: this.convertedSlice(start, this.pos), line };
|
|
579
632
|
}
|
|
580
633
|
|
|
581
634
|
if (isGraphicAtomCode(ch.charCodeAt(0))) {
|
|
@@ -583,7 +636,7 @@ class Parser {
|
|
|
583
636
|
this.take();
|
|
584
637
|
while (isGraphicAtomCode(this.peek().charCodeAt(0)) &&
|
|
585
638
|
!this.terminatingFullStop()) this.take();
|
|
586
|
-
return { type: TOK.ATOM, text: this.
|
|
639
|
+
return { type: TOK.ATOM, text: this.convertedSlice(start, this.pos), line };
|
|
587
640
|
}
|
|
588
641
|
|
|
589
642
|
throw new Error(`parse line ${line}: bad character ${JSON.stringify(ch)}`);
|
|
@@ -934,7 +987,7 @@ class Parser {
|
|
|
934
987
|
throw new Error(`parse line ${line}: bad term`);
|
|
935
988
|
}
|
|
936
989
|
this.expect(TOK.DOT, '.');
|
|
937
|
-
this.applyParserFlagDirective(directive);
|
|
990
|
+
this.applyParserFlagDirective(directive, line);
|
|
938
991
|
this.applyImportedLibraryOperators(directive);
|
|
939
992
|
this.advance();
|
|
940
993
|
const clause = { head: compound(':-', [directive]), body: [] };
|
|
@@ -1080,12 +1133,16 @@ export function parseClauses(source, options = {}) {
|
|
|
1080
1133
|
const ownsParserFlagState = options.parserFlagState == null;
|
|
1081
1134
|
const initialDoubleQuotes = options.doubleQuotes ?? 'chars';
|
|
1082
1135
|
const parserOptions = ownsParserFlagState
|
|
1083
|
-
? { ...options, parserFlagState: { doubleQuotes: initialDoubleQuotes } }
|
|
1136
|
+
? { ...options, parserFlagState: { doubleQuotes: initialDoubleQuotes, charConversion: 'on', charConversions: new Map() } }
|
|
1084
1137
|
: options;
|
|
1085
1138
|
if (options.sourceMetadata === false && options.readTermEnd == null) {
|
|
1086
1139
|
const clauses = parseClausesFastNoSource(source, null, null, parserOptions);
|
|
1087
1140
|
if (clauses) return clauses;
|
|
1088
|
-
if (ownsParserFlagState)
|
|
1141
|
+
if (ownsParserFlagState) {
|
|
1142
|
+
parserOptions.parserFlagState.doubleQuotes = initialDoubleQuotes;
|
|
1143
|
+
parserOptions.parserFlagState.charConversion = 'on';
|
|
1144
|
+
parserOptions.parserFlagState.charConversions.clear();
|
|
1145
|
+
}
|
|
1089
1146
|
}
|
|
1090
1147
|
return new Parser(source, parserOptions).parseProgram();
|
|
1091
1148
|
}
|
|
@@ -1381,11 +1438,19 @@ function parseClausesFastNoSource(source, emit = null, emitBinary = null, option
|
|
|
1381
1438
|
return head && bodyGoal ? { head, body: [bodyGoal] } : null;
|
|
1382
1439
|
};
|
|
1383
1440
|
|
|
1441
|
+
const preparationConversionActive = () =>
|
|
1442
|
+
options.parserFlagState?.charConversion === 'on' &&
|
|
1443
|
+
(options.parserFlagState?.charConversions?.size ?? 0) > 0;
|
|
1444
|
+
|
|
1384
1445
|
const flush = () => {
|
|
1385
1446
|
const text = chunk.trim();
|
|
1386
1447
|
chunk = '';
|
|
1387
1448
|
if (!text) return true;
|
|
1388
|
-
|
|
1449
|
+
// Once a preparation-time char_conversion/2 mapping is active, every
|
|
1450
|
+
// subsequent source character must pass through the full tokenizer. The
|
|
1451
|
+
// compact parser deliberately operates on raw source ranges, so using it
|
|
1452
|
+
// here would silently bypass Convc for otherwise simple clauses.
|
|
1453
|
+
const simple = preparationConversionActive() ? null : parseSimple(text);
|
|
1389
1454
|
if (simple) {
|
|
1390
1455
|
accept(simple);
|
|
1391
1456
|
return true;
|
|
@@ -1408,7 +1473,7 @@ function parseClausesFastNoSource(source, emit = null, emitBinary = null, option
|
|
|
1408
1473
|
if (contentEnd > contentStart && source.charCodeAt(contentEnd - 1) === 13) contentEnd--;
|
|
1409
1474
|
[contentStart, contentEnd] = trimRange(source, contentStart, contentEnd);
|
|
1410
1475
|
if (contentStart < contentEnd && source.charCodeAt(contentStart) !== 37) {
|
|
1411
|
-
if (!chunk && source.charCodeAt(contentEnd - 1) === 46) {
|
|
1476
|
+
if (!chunk && source.charCodeAt(contentEnd - 1) === 46 && !preparationConversionActive()) {
|
|
1412
1477
|
if (emitFastBinaryRange(source, contentStart, contentEnd)) {
|
|
1413
1478
|
// The program builder accepted a compact binary clause directly.
|
|
1414
1479
|
} else {
|
package/src/program.js
CHANGED
|
@@ -650,7 +650,7 @@ export function autoloadProgramGoals(program, inputs, options = {}) {
|
|
|
650
650
|
false,
|
|
651
651
|
{ isoStrict: program.strictIso },
|
|
652
652
|
);
|
|
653
|
-
const parserFlagState = { doubleQuotes: program.doubleQuotes ?? options.doubleQuotes ?? 'chars' };
|
|
653
|
+
const parserFlagState = { doubleQuotes: program.doubleQuotes ?? options.doubleQuotes ?? 'chars', charConversion: 'on', charConversions: new Map() };
|
|
654
654
|
const goals = parseInteropGoalInputs(inputs, {
|
|
655
655
|
...options,
|
|
656
656
|
isoStrict: program.strictIso,
|
|
@@ -684,7 +684,7 @@ function loadSourcesIntoBuilder(builder, sources, options, fast) {
|
|
|
684
684
|
const ensured = new Set();
|
|
685
685
|
const loadedModules = new Set();
|
|
686
686
|
const operatorState = createParserOperatorState([], true, { isoStrict: options.isoStrict === true });
|
|
687
|
-
const parserFlagState = { doubleQuotes: options.doubleQuotes ?? 'chars' };
|
|
687
|
+
const parserFlagState = { doubleQuotes: options.doubleQuotes ?? 'chars', charConversion: 'on', charConversions: new Map() };
|
|
688
688
|
const prepared = sources.map((source) => ({
|
|
689
689
|
source,
|
|
690
690
|
options: { ...sourceOptionsFor(source, options), operatorState, parserFlagState },
|