eyeprolog 1.3.41 → 1.3.43

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/parser.js CHANGED
@@ -2,6 +2,7 @@
2
2
  // It preserves the compact Prolog-like syntax while producing Term objects for the solver.
3
3
  import { ATOM, COMPOUND, atom, compound, cons, emptyList, numberTerm, variable } from './term.js';
4
4
  import { continuesGraphicToken, isTerminatingFullStop } from './syntax-scan.js';
5
+ import { CharacterRepresentationError, isStrictIsoPcsCodePoint } from './iso-character.js';
5
6
 
6
7
 
7
8
  export class NumberRepresentationError extends Error {
@@ -33,7 +34,10 @@ const TOK = {
33
34
  };
34
35
 
35
36
  function isWhitespaceCode(code) {
36
- return code === 32 || code === 9 || code === 10 || code === 13 || code === 12 || code === 11;
37
+ // EyeProlog classifies ASCII C0 controls and DEL as layout characters. In
38
+ // strict mode these are the implementation-defined extended-layout members
39
+ // of the ASCII processor character set.
40
+ return (code >= 0 && code <= 32) || code === 127;
37
41
  }
38
42
 
39
43
  function isDigitCode(code) {
@@ -207,7 +211,11 @@ class Parser {
207
211
  this.strictIso = options.isoStrict === true;
208
212
  this.parserFlagState = options.parserFlagState ?? {
209
213
  doubleQuotes: options.doubleQuotes ?? 'chars',
214
+ charConversion: 'on',
215
+ charConversions: new Map(),
210
216
  };
217
+ this.parserFlagState.charConversion ??= 'on';
218
+ this.parserFlagState.charConversions ??= new Map();
211
219
  if (!['chars', 'codes', 'atom'].includes(this.parserFlagState.doubleQuotes)) {
212
220
  throw new Error(`invalid double_quotes parser flag: ${this.parserFlagState.doubleQuotes}`);
213
221
  }
@@ -223,6 +231,30 @@ class Parser {
223
231
  this.previousToken = null;
224
232
  this.token = this.nextToken();
225
233
  }
234
+ convertCharacter(character) {
235
+ if (this.parserFlagState.charConversion !== 'on' || !character) return character;
236
+ return this.parserFlagState.charConversions.get(character) ?? character;
237
+ }
238
+ rawPeek(offset = 0) {
239
+ const ch = this.source[this.pos + offset] ?? '';
240
+ if (this.strictIso && ch && !isStrictIsoPcsCodePoint(ch.charCodeAt(0))) {
241
+ throw new CharacterRepresentationError();
242
+ }
243
+ return ch;
244
+ }
245
+ rawTake() {
246
+ const ch = this.rawPeek();
247
+ if (ch) {
248
+ this.pos++;
249
+ if (ch === '\n') this.line++;
250
+ }
251
+ return ch;
252
+ }
253
+ convertedSlice(start, end) {
254
+ let out = '';
255
+ for (const ch of this.source.slice(start, end)) out += this.convertCharacter(ch);
256
+ return out;
257
+ }
226
258
  terminatingFullStop(index = this.pos) {
227
259
  // read/1 tries each possible full stop in turn. While parsing a later
228
260
  // candidate, a preceding dot after a graphic character belongs to that
@@ -231,12 +263,12 @@ class Parser {
231
263
  // designated read-term candidate.
232
264
  if (this.readTermEnd != null) {
233
265
  if (index === this.readTermEnd) return true;
234
- if (index < this.readTermEnd && continuesGraphicToken(this.source, index)) {
235
- const next = this.source[index + 1] ?? '';
266
+ if (index < this.readTermEnd && continuesGraphicToken(this.source, index, (ch) => this.convertCharacter(ch))) {
267
+ const next = this.convertCharacter(this.source[index + 1] ?? '');
236
268
  if (next === '%' || /^[\u0009-\u000d\u0020]$/.test(next)) return false;
237
269
  }
238
270
  }
239
- return isTerminatingFullStop(this.source, index);
271
+ return isTerminatingFullStop(this.source, index, (ch) => this.convertCharacter(ch));
240
272
  }
241
273
  defineOperator(priority, specifier, name) {
242
274
  defineParserOperator(this, priority, specifier, name);
@@ -273,12 +305,30 @@ class Parser {
273
305
  for (const name of names) this.defineOperator(priority, specifierTerm.name, name);
274
306
  return true;
275
307
  }
276
- applyParserFlagDirective(directive) {
277
- if (directive.type !== 'compound' || directive.name !== 'set_prolog_flag' || directive.arity !== 2) return;
278
- const [flag, value] = directive.args;
279
- if (flag.type === 'atom' && flag.name === 'double_quotes' &&
280
- value.type === 'atom' && ['chars', 'codes', 'atom'].includes(value.name)) {
281
- this.parserFlagState.doubleQuotes = value.name;
308
+ applyParserFlagDirective(directive, line = this.line) {
309
+ if (directive.type !== 'compound' || directive.arity !== 2) return;
310
+ if (directive.name === 'set_prolog_flag') {
311
+ const [flag, value] = directive.args;
312
+ if (flag.type === 'atom' && flag.name === 'double_quotes' &&
313
+ value.type === 'atom' && ['chars', 'codes', 'atom'].includes(value.name)) {
314
+ this.parserFlagState.doubleQuotes = value.name;
315
+ } else if (flag.type === 'atom' && flag.name === 'char_conversion' &&
316
+ value.type === 'atom' && ['on', 'off'].includes(value.name)) {
317
+ this.parserFlagState.charConversion = value.name;
318
+ }
319
+ return;
320
+ }
321
+ if (directive.name === 'char_conversion') {
322
+ const [input, output] = directive.args;
323
+ const validCharacter = (term) => term.type === 'atom' && Array.from(term.name).length === 1;
324
+ if (input.type === 'var' || output.type === 'var') {
325
+ throw new Error(`parse line ${line}: char_conversion/2 arguments must be instantiated characters`);
326
+ }
327
+ if (!validCharacter(input) || !validCharacter(output)) {
328
+ throw new Error(`parse line ${line}: char_conversion/2 requires one-character atoms`);
329
+ }
330
+ if (input.name === output.name) this.parserFlagState.charConversions.delete(input.name);
331
+ else this.parserFlagState.charConversions.set(input.name, output.name);
282
332
  }
283
333
  }
284
334
  applyImportedLibraryOperators(directive) {
@@ -301,46 +351,45 @@ class Parser {
301
351
  return null;
302
352
  }
303
353
  peek(offset = 0) {
304
- return this.source[this.pos + offset] ?? '';
354
+ return this.convertCharacter(this.rawPeek(offset));
305
355
  }
306
356
  take() {
307
- const ch = this.peek();
308
- if (ch) {
357
+ const raw = this.rawPeek();
358
+ const ch = this.convertCharacter(raw);
359
+ if (raw) {
309
360
  this.pos++;
310
- if (ch === '\n') this.line++;
361
+ if (raw === '\n') this.line++;
311
362
  }
312
363
  return ch;
313
364
  }
314
365
  skipWhitespaceAndComments() {
315
- const source = this.source;
316
- const len = source.length;
317
366
  while (true) {
318
- while (this.pos < len) {
319
- const code = source.charCodeAt(this.pos);
320
- if (!isWhitespaceCode(code)) break;
321
- if (code === 10) this.line++;
322
- this.pos++;
367
+ while (this.rawPeek()) {
368
+ const ch = this.peek();
369
+ if (!isWhitespaceCode(ch.charCodeAt(0))) break;
370
+ this.take();
323
371
  }
324
- if (source.charCodeAt(this.pos) === 37) { // % line comment
325
- while (this.pos < len && source.charCodeAt(this.pos) !== 10) this.pos++;
372
+ if (this.peek() === '%') {
373
+ while (this.rawPeek() && this.peek() !== '\n') this.take();
326
374
  continue;
327
375
  }
328
- if (source[this.pos] === '/' && source[this.pos + 1] === '*') {
376
+ if (this.peek() === '/' && this.peek(1) === '*') {
329
377
  const line = this.line;
330
- this.pos += 2;
331
- while (this.pos < len && !(source[this.pos] === '*' && source[this.pos + 1] === '/')) {
332
- if (source.charCodeAt(this.pos) === 10) this.line++;
333
- this.pos++;
334
- }
335
- if (this.pos >= len) throw new Error(`parse line ${line}: unterminated block comment`);
336
- this.pos += 2;
378
+ this.take();
379
+ this.take();
380
+ while (this.rawPeek() && !(this.peek() === '*' && this.peek(1) === '/')) this.take();
381
+ if (!this.rawPeek()) throw new Error(`parse line ${line}: unterminated block comment`);
382
+ this.take();
383
+ this.take();
337
384
  continue;
338
385
  }
339
386
  break;
340
387
  }
341
388
  }
342
389
  readEscape(line, options = {}) {
343
- const escaped = this.take();
390
+ const takeChar = () => options.raw ? this.rawTake() : this.take();
391
+ const peekChar = () => options.raw ? this.rawPeek() : this.peek();
392
+ const escaped = takeChar();
344
393
  if (!escaped) throw new Error(`parse line ${line}: unterminated escape sequence`);
345
394
 
346
395
  // ISO 6.4.2 permits a continuation escape only inside quoted tokens: a
@@ -350,9 +399,9 @@ class Parser {
350
399
  if (options.allowContinuation !== false) return '';
351
400
  throw new Error(`parse line ${line}: bad escape sequence`);
352
401
  }
353
- if (escaped === '\r' && this.peek() === '\n') {
402
+ if (escaped === '\r' && peekChar() === '\n') {
354
403
  if (options.allowContinuation !== false) {
355
- this.take();
404
+ takeChar();
356
405
  return '';
357
406
  }
358
407
  throw new Error(`parse line ${line}: bad escape sequence`);
@@ -363,22 +412,24 @@ class Parser {
363
412
 
364
413
  if (escaped === 'x') {
365
414
  let digits = '';
366
- while (/^[0-9A-Fa-f]$/.test(this.peek())) digits += this.take();
367
- if (!digits || this.take() !== '\\') throw new Error(`parse line ${line}: bad hexadecimal escape`);
415
+ while (/^[0-9A-Fa-f]$/.test(peekChar())) digits += takeChar();
416
+ if (!digits || takeChar() !== '\\') throw new Error(`parse line ${line}: bad hexadecimal escape`);
368
417
  const code = Number.parseInt(digits, 16);
369
418
  if (code > 0x10ffff || (code >= 0xd800 && code <= 0xdfff)) {
370
419
  throw new Error(`parse line ${line}: character escape out of range`);
371
420
  }
421
+ if (this.strictIso && !isStrictIsoPcsCodePoint(code)) throw new CharacterRepresentationError();
372
422
  return String.fromCodePoint(code);
373
423
  }
374
424
  if (/^[0-7]$/.test(escaped)) {
375
425
  let digits = escaped;
376
- while (/^[0-7]$/.test(this.peek())) digits += this.take();
377
- if (this.take() !== '\\') throw new Error(`parse line ${line}: bad octal escape`);
426
+ while (/^[0-7]$/.test(peekChar())) digits += takeChar();
427
+ if (takeChar() !== '\\') throw new Error(`parse line ${line}: bad octal escape`);
378
428
  const code = Number.parseInt(digits, 8);
379
429
  if (code > 0x10ffff || (code >= 0xd800 && code <= 0xdfff)) {
380
430
  throw new Error(`parse line ${line}: character escape out of range`);
381
431
  }
432
+ if (this.strictIso && !isStrictIsoPcsCodePoint(code)) throw new CharacterRepresentationError();
382
433
  return String.fromCodePoint(code);
383
434
  }
384
435
  // A backslash followed by a decimal digit is numeric-escape syntax, but
@@ -411,7 +462,7 @@ class Parser {
411
462
  this.take();
412
463
  while (isGraphicAtomCode(this.peek().charCodeAt(0)) &&
413
464
  !this.terminatingFullStop()) this.take();
414
- return { type: TOK.ATOM, text: this.source.slice(start, this.pos), line };
465
+ return { type: TOK.ATOM, text: this.convertedSlice(start, this.pos), line };
415
466
  }
416
467
  if (ch === '!') {
417
468
  this.take();
@@ -442,20 +493,29 @@ class Parser {
442
493
  }
443
494
 
444
495
  if (ch === '"' || ch === "'") {
496
+ // Whether an input character is quoted is determined from the target
497
+ // source before character conversion (ISO 8.14.1 note 1). A literal
498
+ // quote therefore protects its contents from Convc; if a conversion
499
+ // itself produces a quote token, the following raw characters remain
500
+ // subject to conversion while that resulting token is parsed.
501
+ const rawOpening = this.rawPeek();
445
502
  const quote = this.take();
503
+ const literalQuote = rawOpening === quote;
504
+ const peekQuoted = () => literalQuote ? this.rawPeek() : this.peek();
505
+ const takeQuoted = () => literalQuote ? this.rawTake() : this.take();
446
506
  let text = '';
447
507
  while (true) {
448
- if (!this.peek()) throw new Error(`parse line ${line}: unterminated quoted term`);
449
- let value = this.take();
508
+ if (!peekQuoted()) throw new Error(`parse line ${line}: unterminated quoted term`);
509
+ let value = takeQuoted();
450
510
  if (value === quote) {
451
- if (this.peek() === quote) {
452
- this.take();
511
+ if (peekQuoted() === quote) {
512
+ takeQuoted();
453
513
  value = quote;
454
514
  } else {
455
515
  break;
456
516
  }
457
- } else if (value === '\\' && this.peek()) {
458
- value = this.readEscape(line);
517
+ } else if (value === '\\' && peekQuoted()) {
518
+ value = this.readEscape(line, { raw: literalQuote });
459
519
  } else if (value !== ' ' && isWhitespaceCode(value.charCodeAt(0))) {
460
520
  // ISO 6.4.2.1 allows an ordinary space in a quoted character, but
461
521
  // not literal layout characters such as tab or newline. Newlines
@@ -489,17 +549,20 @@ class Parser {
489
549
  (this.peek(2) !== "'" || this.peek(3) === "'") &&
490
550
  !(this.peek(2) === '\\' && this.peek(3) === '\n');
491
551
  if (startsQuotedCharacter) {
552
+ const rawQuotedCharacter = this.rawPeek() === '0' && this.rawPeek(1) === "'";
492
553
  this.take();
493
554
  this.take();
494
- let value = this.take();
555
+ const takeCharacter = () => rawQuotedCharacter ? this.rawTake() : this.take();
556
+ const peekCharacter = () => rawQuotedCharacter ? this.rawPeek() : this.peek();
557
+ let value = takeCharacter();
495
558
  if (value) {
496
559
  const firstCode = value.charCodeAt(0);
497
560
  if (firstCode >= 0xd800 && firstCode <= 0xdbff) {
498
- const secondCode = this.peek().charCodeAt(0);
561
+ const secondCode = peekCharacter().charCodeAt(0);
499
562
  if (secondCode < 0xdc00 || secondCode > 0xdfff) {
500
563
  throw new Error(`parse line ${line}: bad character code constant`);
501
564
  }
502
- value += this.take();
565
+ value += takeCharacter();
503
566
  } else if (firstCode >= 0xdc00 && firstCode <= 0xdfff) {
504
567
  throw new Error(`parse line ${line}: bad character code constant`);
505
568
  }
@@ -512,10 +575,10 @@ class Parser {
512
575
  // apostrophe is doubled just as it is inside a quoted atom. Thus
513
576
  // 0''' is one numeric token denoting character code 39, while the
514
577
  // undoubled 0'' is not a complete single quoted character.
515
- if (this.peek() !== "'") throw new Error(`parse line ${line}: bad character code constant`);
516
- this.take();
578
+ if (peekCharacter() !== "'") throw new Error(`parse line ${line}: bad character code constant`);
579
+ takeCharacter();
517
580
  } else if (value === '\\') {
518
- value = this.readEscape(line, { allowContinuation: false });
581
+ value = this.readEscape(line, { allowContinuation: false, raw: rawQuotedCharacter });
519
582
  }
520
583
  const code = value.codePointAt(0);
521
584
  return { type: TOK.NUMBER, text: String(negative ? -code : code), line };
@@ -550,14 +613,14 @@ class Parser {
550
613
  // followed by the name E9 and is not a valid term without an operator.
551
614
  if (hasFraction && (this.peek() === 'e' || this.peek() === 'E')) {
552
615
  let idx = this.pos + 1;
553
- if (this.source[idx] === '+' || this.source[idx] === '-') idx++;
554
- if (isDigitCode((this.source[idx] ?? '').charCodeAt(0))) {
616
+ if (['+', '-'].includes(this.convertCharacter(this.source[idx] ?? ''))) idx++;
617
+ if (isDigitCode(this.convertCharacter(this.source[idx] ?? '').charCodeAt(0))) {
555
618
  this.take();
556
619
  if (this.peek() === '+' || this.peek() === '-') this.take();
557
620
  while (isDigitCode(this.peek().charCodeAt(0))) this.take();
558
621
  }
559
622
  }
560
- let text = this.source.slice(start, this.pos);
623
+ let text = this.convertedSlice(start, this.pos);
561
624
  if (!hasFraction) text = BigInt(text).toString();
562
625
  else text = finiteFloatTokenText(text);
563
626
  return { type: TOK.NUMBER, text, line };
@@ -567,7 +630,7 @@ class Parser {
567
630
  const start = this.pos;
568
631
  this.take();
569
632
  while (isNameContinueCode(this.peek().charCodeAt(0))) this.take();
570
- const text = this.source.slice(start, this.pos);
633
+ const text = this.convertedSlice(start, this.pos);
571
634
  return { type: TOK.VAR, text, line };
572
635
  }
573
636
 
@@ -575,7 +638,7 @@ class Parser {
575
638
  const start = this.pos;
576
639
  this.take();
577
640
  while (isNameContinueCode(this.peek().charCodeAt(0))) this.take();
578
- return { type: TOK.ATOM, text: this.source.slice(start, this.pos), line };
641
+ return { type: TOK.ATOM, text: this.convertedSlice(start, this.pos), line };
579
642
  }
580
643
 
581
644
  if (isGraphicAtomCode(ch.charCodeAt(0))) {
@@ -583,7 +646,7 @@ class Parser {
583
646
  this.take();
584
647
  while (isGraphicAtomCode(this.peek().charCodeAt(0)) &&
585
648
  !this.terminatingFullStop()) this.take();
586
- return { type: TOK.ATOM, text: this.source.slice(start, this.pos), line };
649
+ return { type: TOK.ATOM, text: this.convertedSlice(start, this.pos), line };
587
650
  }
588
651
 
589
652
  throw new Error(`parse line ${line}: bad character ${JSON.stringify(ch)}`);
@@ -934,7 +997,7 @@ class Parser {
934
997
  throw new Error(`parse line ${line}: bad term`);
935
998
  }
936
999
  this.expect(TOK.DOT, '.');
937
- this.applyParserFlagDirective(directive);
1000
+ this.applyParserFlagDirective(directive, line);
938
1001
  this.applyImportedLibraryOperators(directive);
939
1002
  this.advance();
940
1003
  const clause = { head: compound(':-', [directive]), body: [] };
@@ -1080,12 +1143,16 @@ export function parseClauses(source, options = {}) {
1080
1143
  const ownsParserFlagState = options.parserFlagState == null;
1081
1144
  const initialDoubleQuotes = options.doubleQuotes ?? 'chars';
1082
1145
  const parserOptions = ownsParserFlagState
1083
- ? { ...options, parserFlagState: { doubleQuotes: initialDoubleQuotes } }
1146
+ ? { ...options, parserFlagState: { doubleQuotes: initialDoubleQuotes, charConversion: 'on', charConversions: new Map() } }
1084
1147
  : options;
1085
1148
  if (options.sourceMetadata === false && options.readTermEnd == null) {
1086
1149
  const clauses = parseClausesFastNoSource(source, null, null, parserOptions);
1087
1150
  if (clauses) return clauses;
1088
- if (ownsParserFlagState) parserOptions.parserFlagState.doubleQuotes = initialDoubleQuotes;
1151
+ if (ownsParserFlagState) {
1152
+ parserOptions.parserFlagState.doubleQuotes = initialDoubleQuotes;
1153
+ parserOptions.parserFlagState.charConversion = 'on';
1154
+ parserOptions.parserFlagState.charConversions.clear();
1155
+ }
1089
1156
  }
1090
1157
  return new Parser(source, parserOptions).parseProgram();
1091
1158
  }
@@ -1381,11 +1448,19 @@ function parseClausesFastNoSource(source, emit = null, emitBinary = null, option
1381
1448
  return head && bodyGoal ? { head, body: [bodyGoal] } : null;
1382
1449
  };
1383
1450
 
1451
+ const preparationConversionActive = () =>
1452
+ options.parserFlagState?.charConversion === 'on' &&
1453
+ (options.parserFlagState?.charConversions?.size ?? 0) > 0;
1454
+
1384
1455
  const flush = () => {
1385
1456
  const text = chunk.trim();
1386
1457
  chunk = '';
1387
1458
  if (!text) return true;
1388
- const simple = parseSimple(text);
1459
+ // Once a preparation-time char_conversion/2 mapping is active, every
1460
+ // subsequent source character must pass through the full tokenizer. The
1461
+ // compact parser deliberately operates on raw source ranges, so using it
1462
+ // here would silently bypass Convc for otherwise simple clauses.
1463
+ const simple = preparationConversionActive() ? null : parseSimple(text);
1389
1464
  if (simple) {
1390
1465
  accept(simple);
1391
1466
  return true;
@@ -1408,7 +1483,7 @@ function parseClausesFastNoSource(source, emit = null, emitBinary = null, option
1408
1483
  if (contentEnd > contentStart && source.charCodeAt(contentEnd - 1) === 13) contentEnd--;
1409
1484
  [contentStart, contentEnd] = trimRange(source, contentStart, contentEnd);
1410
1485
  if (contentStart < contentEnd && source.charCodeAt(contentStart) !== 37) {
1411
- if (!chunk && source.charCodeAt(contentEnd - 1) === 46) {
1486
+ if (!chunk && source.charCodeAt(contentEnd - 1) === 46 && !preparationConversionActive()) {
1412
1487
  if (emitFastBinaryRange(source, contentStart, contentEnd)) {
1413
1488
  // The program builder accepted a compact binary clause directly.
1414
1489
  } else {
package/src/program.js CHANGED
@@ -650,7 +650,7 @@ export function autoloadProgramGoals(program, inputs, options = {}) {
650
650
  false,
651
651
  { isoStrict: program.strictIso },
652
652
  );
653
- const parserFlagState = { doubleQuotes: program.doubleQuotes ?? options.doubleQuotes ?? 'chars' };
653
+ const parserFlagState = { doubleQuotes: program.doubleQuotes ?? options.doubleQuotes ?? 'chars', charConversion: 'on', charConversions: new Map() };
654
654
  const goals = parseInteropGoalInputs(inputs, {
655
655
  ...options,
656
656
  isoStrict: program.strictIso,
@@ -684,7 +684,7 @@ function loadSourcesIntoBuilder(builder, sources, options, fast) {
684
684
  const ensured = new Set();
685
685
  const loadedModules = new Set();
686
686
  const operatorState = createParserOperatorState([], true, { isoStrict: options.isoStrict === true });
687
- const parserFlagState = { doubleQuotes: options.doubleQuotes ?? 'chars' };
687
+ const parserFlagState = { doubleQuotes: options.doubleQuotes ?? 'chars', charConversion: 'on', charConversions: new Map() };
688
688
  const prepared = sources.map((source) => ({
689
689
  source,
690
690
  options: { ...sourceOptionsFor(source, options), operatorState, parserFlagState },
package/src/solver.js CHANGED
@@ -955,8 +955,8 @@ function defaultPrologFlags(unknown = 'error', strictIso = false) {
955
955
  ['integer_rounding_function', { value: compound('toward_zero', []), allowed: ['toward_zero'], changeable: false }],
956
956
  ['char_conversion', { value: compound('on', []), allowed: ['on', 'off'], changeable: true }],
957
957
  ['debug', { value: compound('off', []), allowed: ['on', 'off'], changeable: true }],
958
- ['max_integer', { value: compound('unbounded', []), allowed: ['unbounded'], changeable: false }],
959
- ['min_integer', { value: compound('unbounded', []), allowed: ['unbounded'], changeable: false }],
958
+ ['max_integer', { value: null, allowed: ['unbounded'], changeable: false }],
959
+ ['min_integer', { value: null, allowed: ['unbounded'], changeable: false }],
960
960
  ['max_arity', { value: compound('unbounded', []), allowed: ['unbounded'], changeable: false }],
961
961
  ['unknown', { value: compound(unknown, []), allowed: ['error', 'fail', 'warning'], changeable: true }],
962
962
  ['double_quotes', { value: compound('chars', []), allowed: ['chars', 'codes', 'atom'], changeable: true }],
@@ -1,4 +1,4 @@
1
- // ISO/IEC 13211-2 module sources shipped with EyeProlog.
1
+ // Prolog library modules shipped with EyeProlog and loaded through its documented module compatibility surface.
2
2
  // The sources are registered here so library(Name) works in Node and browsers.
3
3
  // Modules are loaded on demand by explicit use_module/1-2 or by the conservative
4
4
  // source-level interop autoloader declared below.
@@ -46,7 +46,7 @@ export function isTerminatingFullStop(source, index, convert = null) {
46
46
  // token character accepted by continuesGraphicToken().
47
47
  if (continuesGraphicToken(source, index, convert)) return false;
48
48
  if (next === '' || next === '%' || next === '\n' || next === '\r') return true;
49
- if (/^[\u0009\u000b\u000c\u0020]$/.test(next)) return true;
49
+ if (/^[\u0000-\u0020\u007f]$/.test(next)) return true;
50
50
  return false;
51
51
  }
52
52
 
@@ -30,15 +30,15 @@ error-ordering alternative to an individual executable assertion.
30
30
 
31
31
  | Standard area | Status | Current evidence |
32
32
  | --- | --- | --- |
33
- | Clause 6 — tokens, terms, lists, operators, quoted text | audit | Complete vendored WG17 syntax matrix, `lexical_and_curly_terms`, `scryer_lexical_terms`, operator suites, syntax-error cases, quoted-layout/escape error cases, and writer/read-back regressions. |
34
- | 7.1-7.3 — term types, term order, unification | audit | Standard-order, identity, finite-tree and occurs-check suites, Corrigendum 2 term predicates. |
35
- | 7.4 — Prolog text and directives | audit | All Part 1 directive indicators are parsed; include/ensure-loaded/operator/flag/character-conversion behavior has executable coverage. Cross-text `multifile/1` and ordering constraints require explicit shall-by-shall audit. |
33
+ | Clause 6 — tokens, terms, lists, operators, quoted text | audit | Complete vendored WG17 syntax matrix, `lexical_and_curly_terms`, `scryer_lexical_terms`, operator suites, syntax-error cases, quoted-layout/escape error cases, writer/read-back regressions, and strict ASCII PCS/collation boundary tests. The implementation-defined 6.5/6.6 character-model decisions are now closed; wider shall-by-shall lexical mapping remains open. |
34
+ | 7.1-7.3 — term types, term order, unification | audit | Standard-order, identity, finite-tree and occurs-check suites, Corrigendum 2 term predicates, plus strict checks for the required `variable < float < integer < atom < compound` type order and PCS-based atom collation. |
35
+ | 7.4 — Prolog text and directives | audit | All Part 1 directive indicators are parsed; include/ensure-loaded/operator/flag/character-conversion behavior has executable coverage. Preparation-time `char_conversion/2` now affects later unquoted source text and respects `char_conversion=off`. Cross-text `multifile/1` and ordering constraints still require explicit shall-by-shall audit. |
36
36
  | 7.5-7.6 — database and term/clause conversion | audit | Dynamic database and logical-update-view suites. Strict mode restores Part 1 private-static/public-dynamic `clause/2` access. Public/private and multi-text requirements still need complete mapping. |
37
37
  | 7.7 — execution and backtracking | audit | Control/search suites. Strict mode disables EyeProlog automatic tabling, cycle guards, and recursive numeric shortcuts so core execution uses ordinary clause selection/backtracking. |
38
38
  | 7.8 — control constructs and exceptions | audit | call, cut, conjunction, disjunction (including failed branches after callee-local cuts), if-then, catch/throw, renamed-copy tests. |
39
39
  | 7.9 — expression evaluation | audit | Arithmetic/evaluation/error suites, including Corrigenda. Exceptional-value/error-precedence matrix remains to be exhaustively enumerated. |
40
40
  | 7.10 — input/output concepts | audit | Stream, character/byte I/O, read/write options, operator-sensitive write-back, Corrigendum 3 writer cases. Full option cross-product remains open. |
41
- | 7.11 — flags | audit | Required Part 1 flags implemented. Normal and strict modes use the ISO `unknown=error` default; strict mode excludes the EyeProlog `occurs_check` extension. |
41
+ | 7.11 — flags | audit | Required Part 1 flags implemented. Normal and strict modes use the ISO `unknown=error` default; strict mode excludes the EyeProlog `occurs_check` extension. With `bounded=false`, `max_integer` and `min_integer` correctly have no current value. |
42
42
  | 7.12 — errors | audit | ISO `error(Error, Context)` envelope, type/domain/permission/representation/evaluation/syntax families and focused error cases. Complete prescribed-error ordering remains open. |
43
43
  | 8.2-8.17 — built-in predicates | audit | Predicate-family coverage is mapped in ISO-MATRIX.md; Corrigendum 2 additions (`subsumes_term/2`, `term_variables/2`, `call/2..8`, `false/0`) are in the strict registry. Mode/error matrix is not yet one-row-per-standard-row. |
44
44
  | Clause 9 — evaluable functors | audit | Integer/float/rounding/transcendental/bitwise suites and corrigendum cases. Host floating-point representation choices remain documented implementation-defined behavior. |
@@ -46,6 +46,29 @@ error-ordering alternative to an individual executable assertion.
46
46
  | Corrigendum 2 | covered | Added predicates/functors, catch corrections, bar/operator and uninstantiation corrections have dedicated cases. |
47
47
  | Corrigendum 3 | covered | Writer options, `variable_names/1`, canonical list output and negative-power corrections have dedicated cases. |
48
48
 
49
+ ## Issue #65 conformance corrections
50
+
51
+ The issue #65 audit against the licensed Part 1 text and Corrigenda closed two
52
+ concrete mismatches without changing the remaining audit rows into blanket
53
+ conformance claims:
54
+
55
+ - `bounded=false` no longer exposes implementation-specific `unbounded` values
56
+ for `max_integer` or `min_integer`; the corresponding
57
+ `current_prolog_flag/2` queries fail as specified by 7.11.1.1;
58
+ - preparation-time `char_conversion/2` now converts later unquoted source text
59
+ when the `char_conversion` flag is `on`, leaves quoted characters unchanged,
60
+ and feeds the same mapping into execution-time term input.
61
+
62
+ A follow-on audit closes the processor-character-set/collation choices rather
63
+ than leaving them implicit. `--iso-strict` now selects the 128-character ASCII
64
+ PCS U+0000..U+007F, classifies C0 controls and DEL as extended layout
65
+ characters, and uses the code point itself as each collating-sequence integer.
66
+ Characters/codes outside that PCS raise representation errors in strict
67
+ parsing, term input, character conversion, and character-code predicates. The
68
+ normal profile retains Unicode scalar character data as an explicit extension.
69
+ The complete WG17 syntax matrix remains green under this narrower strict
70
+ boundary.
71
+
49
72
  ## Strict-core boundary
50
73
 
51
74
  `isoStrict: true` is intentionally a **Part 1 + Corrigenda 1-3** mode. It does
@@ -64,9 +87,11 @@ remain ordinary operator syntax in strict core mode. A conforming `op/3`
64
87
  directive may still add an infix `?-` definition; strict mode reads that as an
65
88
  ordinary term rather than as a quad.
66
89
 
67
- Part 2 modules and Part 3 grammar rules remain supported and tested in the
90
+ Module and DCG compatibility features remain supported and tested in the
68
91
  normal EyeProlog profile. They are tracked separately in ISO-MATRIX.md rather
69
- than being silently folded into the Part 1 strict-core claim.
92
+ than being silently folded into the Part 1 strict-core claim. The project does
93
+ not currently assert that this evidence closes every requirement of ISO/IEC
94
+ 13211-2:2000 or ISO/IEC TS 13211-3.
70
95
 
71
96
  ## Release gate
72
97
 
@@ -19,23 +19,23 @@ Status values are:
19
19
  - **defined** — the current behavior is implemented and stated here;
20
20
  - **not applicable** — the standard decision is conditional and the condition
21
21
  is false for EyeProlog's selected profile;
22
- - **audit gap** — the current code behavior is stated, but strict-mode
23
- conformance still needs a correction or a narrower profile before this row
24
- can support a full conformance claim.
22
+ - **audit gap** — retained for any future implementation-defined choice whose
23
+ code/documentation boundary is still unresolved. Open *normative* shall-by-
24
+ shall work is tracked separately in `ISO-COMPLIANCE.md`.
25
25
 
26
26
  ## Explicit implementation-defined decisions
27
27
 
28
28
  | Clause | Decision completed by ISO 5.4 documentation | EyeProlog choice | Status / implementation evidence |
29
29
  | --- | --- | --- | --- |
30
30
  | 5.5.11 | Reserved atoms and the effect of instantiating a variable to one | EyeProlog reserves no Prolog atom under 5.5.11. Atoms with implementation-looking names remain ordinary terms unless a particular predicate interprets them. | **defined** — term representation and built-ins in `src/term.js`, `src/iso.js`. |
31
- | 6.5 | Processor character set (PCS) | The unquoted ISO lexical classes are the Part 1 ASCII characters implemented by `src/parser.js`. Quoted character data additionally accepts Unicode scalar values. | **audit gap** — the Unicode quoted-character extension is currently also accepted by `--iso-strict`; strict extension rejection still needs a narrower PCS rule or a documented conforming classification. |
32
- | 6.5 | Classification of additional/extended PCS characters | Non-ASCII scalar values are not accepted as unquoted small-letter, capital-letter, graphic, solo, layout, or meta characters; they are accepted only inside quoted character data. | **audit gap** — same strict-mode boundary as the preceding row. |
33
- | 6.6 | Collating-sequence integers | Character codes are Unicode scalar values. Atom comparison uses ECMAScript string lexicographic order; on the ISO ASCII repertoire this is code-point order and satisfies the required monotonic ranges. | **defined** — `src/term.js` (`compareTerms`), `src/iso.js` character-code predicates. |
34
- | 6.6 | Collating values of control escapes and extended characters | Control escapes and character-code predicates use Unicode scalar values. Atom ordering is ECMAScript string order; for non-BMP one-char atoms that order is based on UTF-16 code units rather than scalar values. | **audit gap** — the ISO ASCII repertoire is conforming, but the extended-character collating rule still needs one coherent documented integer/order mapping in strict mode. |
31
+ | 6.5 | Processor character set (PCS) | In `--iso-strict`, PCS is the 128-character 7-bit ASCII set U+0000..U+007F. Normal mode additionally accepts Unicode scalar values in character data as an implementation-specific extension. | **defined** — `src/iso-character.js`, strict parser/character-I/O guards, and strict-core/WG17 coverage. |
32
+ | 6.5 | Classification of additional/extended PCS characters | Printable ASCII uses the lexical classes specified by Part 1. ASCII C0 controls U+0000..U+001F and DEL U+007F are EyeProlog's extended **layout** characters; they may therefore separate tokens, while quoted control values are written/read through the ISO escape forms. No non-ASCII character belongs to strict PCS. | **defined** — `src/parser.js`, `src/syntax-scan.js`, `src/iso-character.js`. |
33
+ | 6.6 | Collating-sequence integers | In strict mode each PCS character's collating-sequence integer is its ASCII/Unicode code point, 0..127. Atom comparison is lexicographic by the same code-unit values, which coincide with those integers throughout strict PCS and satisfy the required capital-letter, small-letter, and decimal-digit constraints. | **defined** — `src/iso-character.js`, `src/term.js` (`compareTerms`), `src/iso.js` character-code predicates. |
34
+ | 6.6 | Collating values of control escapes and extended characters | Strict control/extended-layout characters use their ASCII code point as collating integer, so octal/hexadecimal escapes and character-code constants map to the same 0..127 PCS. Normal-mode Unicode character codes use Unicode scalar values; that broader ordering is outside the strict Part 1 profile. | **defined** — `src/iso-character.js`, parser escape handling, `char_code/2`, `atom_codes/2`, and WG17 escape cases. |
35
35
  | 7.1.2.2 | Mapping between a character code and bytes | Text file streams decode and encode UTF-8. Binary streams expose bytes 0..255 directly. | **defined** — `src/io.js`. |
36
- | 7.1.4.1 | Set `C` of characters represented by one-char atoms | Character predicates accept Unicode scalar values U+0000..U+10FFFF excluding surrogate code points. | **defined** — `src/iso.js` character-code validation. |
36
+ | 7.1.4.1 | Set `C` of characters represented by one-char atoms | In strict mode `C` is exactly the ASCII PCS U+0000..U+007F. Normal mode extends character predicates to Unicode scalar values U+0000..U+10FFFF excluding surrogates. | **defined** — `src/iso-character.js`, `src/iso.js` character-code validation. |
37
37
  | 7.4.2.4 | Whether `op/3` directives affect other Prolog texts or execution | An `op/3` directive changes parsing of subsequent text loaded into the same `Program`; the resulting operator table is also used by execution-time term I/O. Separately created `Program` objects are independent. | **defined** — `src/parser.js`, `src/program.js`, `src/iso.js`. |
38
- | 7.4.2.5 | Whether directive-created `Convc` affects other text/execution | Directive mappings are copied into the solver's execution-time `charConversions` map. The source tokenizer itself does not currently apply `char_conversion/2` to later unquoted source characters. | **audit gap** — `src/program.js`, `src/solver.js`; source-preparation conversion remains to be aligned with 7.11.2.1. |
38
+ | 7.4.2.5 | Whether directive-created `Convc` affects other text/execution | Yes. A `char_conversion/2` directive updates preparation-time conversion for later unquoted source characters and the recorded mapping initializes execution-time term input. Quoted characters are not converted; `char_conversion=off` disables following preparation-time conversion. | **defined** — `src/parser.js`, `src/program.js`, `src/solver.js`; strict-core regression coverage. |
39
39
  | 7.4.2.6 | Order of `initialization/1` goals | Initialization goals run once, in source/inclusion order, before requested goals; each must obtain a first solution. | **defined** — `Program.initializations`, `Solver.runInitializations()`. |
40
40
  | 7.4.2.7 | Ground term designating a Prolog text for `include/1` | A source designation is an atom interpreted as a file path; relative paths resolve from the including file's directory (or the current working directory for unanchored input). | **defined** — `src/program.js` (`readIncludedSource`). |
41
41
  | 7.4.2.8 | Ground term designating a Prolog text for `ensure_loaded/1` | Same atom/file-path designation as `include/1`. | **defined** — `src/program.js`. |
@@ -59,17 +59,17 @@ Status values are:
59
59
  | 7.10.2.13 | `eof_action(Action)` property when the stream uses its default action | Reports the same selected action: `error` for ordinary opened files and `reset` for the predefined standard streams. | **defined** — `streamProperties()` in `src/iso.js`, defaults in `src/io.js`. |
60
60
  | 7.10.2.11 | Whether `reposition(false)` streams may nevertheless be repositioned | No. `set_stream_position/2` raises a permission error unless the stream was created with `reposition(true)`. | **defined** — `src/iso.js`. |
61
61
  | 7.11.1.1 | Default `bounded` flag | `false`: EyeProlog's Prolog integer model uses arbitrary-precision `BigInt`, subject to host resources. | **defined** — `defaultPrologFlags()` in `src/solver.js`. |
62
- | 7.11.1.2 | Default `max_integer` when `bounded=true` | The bounded-profile default is not applicable because EyeProlog selects `bounded=false`. However, EyeProlog currently exposes the sentinel atom `unbounded` through `current_prolog_flag/2` instead of making the `max_integer` query fail as 7.11.1.1 prescribes for an unbounded processor. | **audit gap** — `src/solver.js`, `current_prolog_flag/2`. |
63
- | 7.11.1.3 | Default `min_integer` when `bounded=true` | The bounded-profile default is not applicable because EyeProlog selects `bounded=false`. However, EyeProlog currently exposes the sentinel atom `unbounded` through `current_prolog_flag/2` instead of making the `min_integer` query fail as 7.11.1.1 prescribes for an unbounded processor. | **audit gap** — `src/solver.js`, `current_prolog_flag/2`. |
62
+ | 7.11.1.2 | Default `max_integer` when `bounded=true` | Not applicable because EyeProlog selects `bounded=false`; consequently `current_prolog_flag(max_integer, _)` fails. | **not applicable** — `src/solver.js`, `current_prolog_flag/2`; strict-core regression coverage. |
63
+ | 7.11.1.3 | Default `min_integer` when `bounded=true` | Not applicable because EyeProlog selects `bounded=false`; consequently `current_prolog_flag(min_integer, _)` fails. | **not applicable** — `src/solver.js`, `current_prolog_flag/2`; strict-core regression coverage. |
64
64
  | 7.11.1.4 | Default `integer_rounding_function` flag | `toward_zero`; `//` uses truncation toward zero. `div` is provided separately with downward/floor division semantics. | **defined** — `src/solver.js`, `src/iso-arithmetic.js`. |
65
65
  | 9.1.3.1 | Integer division rounding function `rndI` | Truncation toward zero, matching the `integer_rounding_function=toward_zero` flag. | **defined** — BigInt division for `//` in `src/iso-arithmetic.js`. |
66
- | 7.11.2.1 | Whether preparation-time `Convc` affects execution-time `Convc` | Yes: mappings recorded by `char_conversion/2` directives initialize the solver map. Source-token conversion itself has the audit gap noted at 7.4.2.5. | **defined / audit note** — `src/solver.js`. |
66
+ | 7.11.2.1 | Whether preparation-time `Convc` affects execution-time `Convc` | Yes: mappings created while Prolog text is prepared are retained and initialize the solver's execution-time conversion map. | **defined** — `src/parser.js`, `src/program.js`, `src/solver.js`. |
67
67
  | 7.11.2.2 | Effect when `debug=on` | The flag is accepted and stored; it does not change goal semantics or enable a debugger. | **defined** — `src/solver.js`; no semantic branch depends on `debug`. |
68
68
  | 7.11.2.3 | Default `max_arity` | `unbounded` in the Prolog model, subject to host memory and practical JavaScript array/index limits. | **defined** — `src/solver.js`; relevant guards report representation/resource errors. |
69
69
  | 7.11.2.5 | Default `double_quotes` | `chars`. | **defined** — `src/solver.js`, parser flag state. |
70
70
  | 7.12.1 | Second argument of `error/2` | The default context term is the atom `eyeprolog`. A few implementation-specific diagnostics may deliberately supply a more specific context term. | **defined** — `formalErrorTerm()` in `src/iso.js`. |
71
- | 7.12.2(f) | Implementation-defined representation limits | Character and character-code operations use Unicode scalar limits; arity/integer values are modeled as unbounded but may hit host/resource limits. Float input overflow uses the implementation-specific `max_float`/`min_float` representation names documented by the STC-oriented tests. | **defined** — parser/ISO numeric and character guards. |
72
- | 8.17.1 | Implementation-defined flag value ranges | Strict mode exposes only Part 1 core flags and their standard value sets. Normal mode additionally exposes EyeProlog's `occurs_check` flag; `max_integer`/`min_integer` use the `unbounded` sentinel described above. | **defined** — strict registry/flag filtering in `src/solver.js`. |
71
+ | 7.12.2(f) | Implementation-defined representation limits | Strict character and character-code operations are limited to the selected ASCII PCS/collating integers 0..127. Normal mode extends characters to Unicode scalar values. Arity/integer values are modeled as unbounded but may hit host/resource limits. Float input overflow uses the implementation-specific `max_float`/`min_float` representation names documented by the STC-oriented tests. | **defined** — parser/ISO numeric and character guards. |
72
+ | 8.17.1 | Implementation-defined flag value ranges | Strict mode exposes only Part 1 core flags and their standard value sets. Normal mode additionally exposes EyeProlog's `occurs_check` flag. With `bounded=false`, `max_integer` and `min_integer` have no current value and their `current_prolog_flag/2` queries fail. | **defined** — strict registry/flag filtering in `src/solver.js`. |
73
73
  | 8.17.3 | Other effects of `halt/0` | Terminates EyeProlog execution and returns host/process status `0`; it produces no Prolog solution. | **defined** — `HaltSignal`, `haltBuiltin()`, CLI/runner handling. |
74
74
  | 8.17.4 | Meaning/effects of `halt(Status)` | Integer `Status` is converted to the host process/runner halt code; it produces no Prolog solution. | **defined** — `haltBuiltin()`, `src/execute.js`, `src/cli.js`. |
75
75
  | 9.1.4.1 | Floating-point rounding function `rndF` | Floating values and operations use ECMAScript `Number` (IEEE-754 binary64) and the host's specified binary64 arithmetic/conversions. | **defined** — `src/iso-arithmetic.js`, `src/number-value.js`. |
@@ -105,7 +105,7 @@ families; `--iso-strict` is intended to remove their Part 1 interpretation.
105
105
 
106
106
  | Part 1 extension hook | EyeProlog normal-profile feature | Strict-core disposition |
107
107
  | --- | --- | --- |
108
- | 5.5.1 Syntax | Part 2 modules, Part 3 grammar-rule expansion, and embedded quad syntax | Module directives are rejected; grammar rules remain ordinary `-->/2` terms rather than being expanded; quad syntax is rejected. Unicode quoted-character handling is the open PCS audit item above. |
108
+ | 5.5.1 Syntax | Part 2 modules, Part 3 grammar-rule expansion, embedded quad syntax, and normal-mode Unicode character data | Module directives are rejected; grammar rules remain ordinary `-->/2` terms rather than being expanded; quad syntax is rejected; strict character syntax/data is limited to the documented ASCII PCS. |
109
109
  | 5.5.2 Predefined operators | Part 3 `|` and EyeProlog's labelable infix `(?-)/2`; CLP(Z) operators when that library is imported | Only the Part 1 operator table is predefined; a conforming `op/3` may still add permitted operators. |
110
110
  | 5.5.3 Character-conversion mapping | No non-identity initial `Convc` extension | Identity initial mapping. |
111
111
  | 5.5.4 Types | No additional runtime Prolog term type is exposed by the core solver | Only variable, integer, float, atom, and compound term ordering participates in strict mode. |
@@ -118,9 +118,10 @@ families; `--iso-strict` is intended to remove their Part 1 interpretation.
118
118
  | 5.5.11 Reserved atoms | None | None. |
119
119
  | Cor.3 5.5.12 Options | Extra library/host options may exist outside core option lists | Strict core accepts the standard option names implemented for Part 1; invalid options follow Corrigendum 3 validation. |
120
120
 
121
- Part 2 and Part 3 are separately standardized profiles when EyeProlog is run in
122
- its normal mode; the table above describes them as extensions only relative to
123
- the Part 1 strict-core boundary.
121
+ Normal mode provides documented module and DCG compatibility profiles whose
122
+ features overlap standardized Part 2 and Part 3 facilities. They are extensions
123
+ relative to the Part 1 strict-core boundary and are tested separately; this
124
+ ledger does not assert complete Part 2 or Part 3 conformance.
124
125
 
125
126
  ## Important implementation-dependent behavior (not the 5.4 mandatory table)
126
127