eyeprolog 1.3.42 → 1.3.43

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -210,6 +210,14 @@ expansion/`phrase/2-3`; it also removes the EyeProlog `occurs_check` flag,
210
210
  Normal mode is unchanged and continues to support modules, DCGs, quads,
211
211
  libraries, proofs, cleanup-aware control, and the other documented extensions.
212
212
 
213
+ Strict mode also makes the Part 1 processor-character-set choice explicit: its
214
+ PCS is 7-bit ASCII (U+0000..U+007F), C0 controls and DEL are classified as
215
+ extended layout characters, and the collating-sequence integer of each
216
+ character is its ASCII code. Characters or character codes outside that set
217
+ raise representation errors in strict parsing and character operations. Normal
218
+ mode retains Unicode scalar character data as an implementation-specific
219
+ extension.
220
+
213
221
  The auditable processor-requirement checklist lives in
214
222
  [`test/conformance/ISO-COMPLIANCE.md`](test/conformance/ISO-COMPLIANCE.md).
215
223
  The ISO 5.4 implementation-defined/implementation-specific decision index is
package/package.json CHANGED
@@ -3,7 +3,7 @@
3
3
  "publishConfig": {
4
4
  "access": "public"
5
5
  },
6
- "version": "1.3.42",
6
+ "version": "1.3.43",
7
7
  "description": "EyeProlog turns facts and rules into answers and proofs.",
8
8
  "type": "module",
9
9
  "main": "./index.js",
@@ -0,0 +1,37 @@
1
+ // ISO/IEC 13211-1 processor-character-set choices used by --iso-strict.
2
+ //
3
+ // EyeProlog's strict Part 1 profile deliberately chooses a finite PCS so every
4
+ // accepted character has one documented lexical classification and one stable
5
+ // collating-sequence integer. Normal mode keeps the broader Unicode surface as
6
+ // an implementation-specific extension.
7
+
8
+ export class CharacterRepresentationError extends Error {
9
+ constructor(formal = 'representation_error(character)') {
10
+ super(`error(${formal})`);
11
+ this.name = 'CharacterRepresentationError';
12
+ this.formal = formal;
13
+ }
14
+ }
15
+
16
+ // Strict PCS: the 128 ASCII characters U+0000..U+007F. Printable ASCII
17
+ // carries the Part 1 lexical classes; the remaining C0 controls plus DEL are
18
+ // implementation-defined extended layout characters. This lets every ISO
19
+ // octal/hexadecimal character escape in the ASCII range denote a PCS member.
20
+ // The collating-sequence integer is the Unicode/ASCII code point.
21
+ export function isStrictIsoPcsCodePoint(code) {
22
+ return Number.isInteger(code) && code >= 0 && code <= 0x7f;
23
+ }
24
+
25
+ export function isStrictIsoPcsCharacter(character) {
26
+ if (typeof character !== 'string' || Array.from(character).length !== 1) return false;
27
+ return isStrictIsoPcsCodePoint(character.codePointAt(0));
28
+ }
29
+
30
+ export function strictIsoCollatingInteger(character) {
31
+ return isStrictIsoPcsCharacter(character) ? character.codePointAt(0) : null;
32
+ }
33
+
34
+ export function assertStrictIsoPcsCharacter(character, formal = 'representation_error(character)') {
35
+ if (!isStrictIsoPcsCharacter(character)) throw new CharacterRepresentationError(formal);
36
+ return character;
37
+ }
package/src/iso.js CHANGED
@@ -13,6 +13,7 @@ import {
13
13
  import { formatTermForWrite } from './write.js';
14
14
  import { emptyTerminalSequence, expandDcgBody, isListOrPartialList, validateDcgEmbeddedGoals } from './dcg.js';
15
15
  import { INVALID_UTF8_SENTINEL } from './io.js';
16
+ import { CharacterRepresentationError, isStrictIsoPcsCharacter, isStrictIsoPcsCodePoint } from './iso-character.js';
16
17
  import {
17
18
  characterCodeConstantEnd, continuesGraphicToken, isTerminatingFullStop, quotedEscapeEnd,
18
19
  } from './syntax-scan.js';
@@ -777,7 +778,7 @@ function* currentOpBuiltin({ solver, goal, env }) {
777
778
  }
778
779
  }
779
780
 
780
- function conversionCharacter(term, env, current = false) {
781
+ function conversionCharacter(term, env, current = false, solver = null) {
781
782
  const value = deref(term, env);
782
783
  if (value.type === VAR) {
783
784
  if (current) return value;
@@ -787,18 +788,21 @@ function conversionCharacter(term, env, current = false) {
787
788
  if (current) throw new PrologError('type_error(character)', value);
788
789
  throw new PrologError('representation_error(character)');
789
790
  }
791
+ if (solver?.isoStrict && !isStrictIsoPcsCharacter(value.name)) {
792
+ throw new PrologError('representation_error(character)', value);
793
+ }
790
794
  return value;
791
795
  }
792
796
  function* charConversionBuiltin({ solver, goal, env }) {
793
- const input = conversionCharacter(goal.args[0], env);
794
- const output = conversionCharacter(goal.args[1], env);
797
+ const input = conversionCharacter(goal.args[0], env, false, solver);
798
+ const output = conversionCharacter(goal.args[1], env, false, solver);
795
799
  if (input.name === output.name) solver.charConversions.delete(input.name);
796
800
  else solver.charConversions.set(input.name, output.name);
797
801
  yield env;
798
802
  }
799
803
  function* currentCharConversionBuiltin({ solver, goal, env }) {
800
- const input = conversionCharacter(goal.args[0], env, true);
801
- const output = conversionCharacter(goal.args[1], env, true);
804
+ const input = conversionCharacter(goal.args[0], env, true, solver);
805
+ const output = conversionCharacter(goal.args[1], env, true, solver);
802
806
  for (const [from, to] of [...solver.charConversions]) {
803
807
  const next = env.clone();
804
808
  if (unify(input, atom(from), next) && unify(output, atom(to), next)) yield next;
@@ -1044,6 +1048,9 @@ function inputUnitBuiltin(name) {
1044
1048
  throw error;
1045
1049
  }
1046
1050
  if (unit == null && !peek) stream.pastEnd = true;
1051
+ if (!binary && unit != null && solver.isoStrict && !isStrictIsoPcsCharacter(unit)) {
1052
+ throw new PrologError('representation_error(character)');
1053
+ }
1047
1054
  const result = unit == null ? (binary ? numberTerm(-1) : name.endsWith('code') ? numberTerm(-1) : atom('end_of_file'))
1048
1055
  : binary ? numberTerm(unit) : name.endsWith('code') ? numberTerm(unit.codePointAt(0)) : atom(unit);
1049
1056
  const target = goal.args[goal.arity - 1];
@@ -1059,6 +1066,9 @@ function outputUnitBuiltin(name) {
1059
1066
  if (name === 'put_char') {
1060
1067
  if (stream.type !== 'text') throw new PrologError('permission_error(output, binary_stream)', streamHandle(stream.id));
1061
1068
  if (!oneChar(value)) throw new PrologError('type_error(character)', value);
1069
+ if (solver.isoStrict && !isStrictIsoPcsCharacter(value.name)) {
1070
+ throw new PrologError('representation_error(character)', value);
1071
+ }
1062
1072
  solver.io.writeUnit(stream, value.name);
1063
1073
  } else {
1064
1074
  if ((name === 'put_byte') !== (stream.type === 'binary')) {
@@ -1068,6 +1078,9 @@ function outputUnitBuiltin(name) {
1068
1078
  const code = BigInt(value.name);
1069
1079
  const max = name === 'put_byte' ? 255n : 0x10ffffn;
1070
1080
  if (code < 0n || code > max) throw new PrologError(name === 'put_byte' ? 'type_error(byte)' : 'representation_error(character_code)');
1081
+ if (name === 'put_code' && solver.isoStrict && !isStrictIsoPcsCodePoint(Number(code))) {
1082
+ throw new PrologError('representation_error(character_code)');
1083
+ }
1071
1084
  solver.io.writeUnit(stream, name === 'put_byte' ? Number(code) : String.fromCodePoint(Number(code)));
1072
1085
  }
1073
1086
  yield env;
@@ -1095,6 +1108,9 @@ function* termTextCandidates(stream, solver) {
1095
1108
  let quote = null, lineComment = false, blockComment = false;
1096
1109
  for (let i = stream.position; i < source.length; i++) {
1097
1110
  const ch = source[i], next = source[i + 1];
1111
+ if (solver.isoStrict && !isStrictIsoPcsCodePoint(ch.charCodeAt(0))) {
1112
+ throw new PrologError('representation_error(character)');
1113
+ }
1098
1114
  // Text streams preserve invalid UTF-8 bytes as an impossible Unicode
1099
1115
  // sentinel. read/1-2 and read_term/2-3 must surface the same character
1100
1116
  // representation error as get_char/1-2 instead of misclassifying the
@@ -1106,7 +1122,7 @@ function* termTextCandidates(stream, solver) {
1106
1122
  if (blockComment) { if (ch === '*' && next === '/') { blockComment = false; i++; } continue; }
1107
1123
  if (quote) {
1108
1124
  if (ch === '\\') i = quotedEscapeEnd(source, i);
1109
- else if (ch !== ' ' && /^[\u0009-\u000d]$/.test(ch)) {
1125
+ else if (ch !== ' ' && /^[\u0000-\u001f\u007f]$/.test(ch)) {
1110
1126
  // Literal layout characters are not quoted characters (6.4.2.1).
1111
1127
  // Surface the lexical error immediately even when there is no later
1112
1128
  // full stop; otherwise read/1 would misreport malformed input as EOF.
@@ -1137,7 +1153,7 @@ function hasNonLayoutRemainder(source, start) {
1137
1153
  return lastNonLayoutIndex(source, start) >= start;
1138
1154
  }
1139
1155
  function lastNonLayoutIndex(source, start = 0) {
1140
- const ignored = /[\u0009-\u000d\u0020]+|%[^\n]*(?:\n|$)|\/\*[\s\S]*?\*\//g;
1156
+ const ignored = /[\u0000-\u0020\u007f]+|%[^\n]*(?:\n|$)|\/\*[\s\S]*?\*\//g;
1141
1157
  ignored.lastIndex = start;
1142
1158
  let cursor = start;
1143
1159
  let last = -1;
@@ -1248,7 +1264,9 @@ function readTermFromStream(stream, solver) {
1248
1264
  stream.position = candidate.end;
1249
1265
  return scopeReadTerm(term);
1250
1266
  } catch (error) {
1251
- if (error instanceof NumberRepresentationError) throw new PrologError(error.formal);
1267
+ if (error instanceof NumberRepresentationError || error instanceof CharacterRepresentationError) {
1268
+ throw new PrologError(error.formal);
1269
+ }
1252
1270
  // A dot inside a graphic operator, such as =.., is only a possible
1253
1271
  // terminator. Keep scanning until a complete term parses.
1254
1272
  }
@@ -1510,30 +1528,35 @@ function oneChar(value) {
1510
1528
  return value.type === ATOM && characters(value.name).length === 1;
1511
1529
  }
1512
1530
 
1513
- function validCharacterCode(value) {
1531
+ function validCharacterCode(value, solver = null) {
1514
1532
  if (value.type !== NUMBER || !isDecimalInteger(value.name)) return false;
1515
1533
  const code = BigInt(value.name);
1516
- return code >= 0n && code <= 0x10ffffn && !(code >= 0xd800n && code <= 0xdfffn);
1534
+ if (code < 0n || code > 0x10ffffn || (code >= 0xd800n && code <= 0xdfffn)) return false;
1535
+ return !solver?.isoStrict || isStrictIsoPcsCodePoint(Number(code));
1517
1536
  }
1518
1537
 
1519
- function listToAtomInput(list, env, kind) {
1538
+ function listToAtomInput(list, env, kind, solver = null) {
1520
1539
  const { items, tail } = listElements(list, env);
1521
1540
  if (tail.type === VAR || items.some((item) => item.type === VAR)) throw new PrologError('instantiation_error');
1522
1541
  if (tail.type !== ATOM || tail.name !== '[]') throw new PrologError('type_error(list)', tail);
1523
1542
  if (kind === 'chars') {
1524
1543
  const invalid = items.find((item) => !oneChar(item));
1525
1544
  if (invalid) throw new PrologError('type_error(character)', invalid);
1545
+ if (solver?.isoStrict) {
1546
+ const outsidePcs = items.find((item) => !isStrictIsoPcsCharacter(item.name));
1547
+ if (outsidePcs) throw new PrologError('representation_error(character)', outsidePcs);
1548
+ }
1526
1549
  return items.map((item) => item.name).join('');
1527
1550
  }
1528
1551
  const nonInteger = items.find((item) => item.type !== NUMBER || !isDecimalInteger(item.name));
1529
1552
  if (nonInteger) throw new PrologError('type_error(integer)', nonInteger);
1530
- const invalid = items.find((item) => !validCharacterCode(item));
1553
+ const invalid = items.find((item) => !validCharacterCode(item, solver));
1531
1554
  if (invalid) throw new PrologError('representation_error(character_code)');
1532
1555
  return items.map((item) => String.fromCodePoint(Number(item.name))).join('');
1533
1556
  }
1534
1557
 
1535
1558
  function atomListBuiltin(kind) {
1536
- return function* ({ goal, env }) {
1559
+ return function* ({ solver, goal, env }) {
1537
1560
  const value = deref(goal.args[0], env);
1538
1561
  if (value.type !== VAR && value.type !== ATOM) throw new PrologError('type_error(atom)', value);
1539
1562
  const list = deref(goal.args[1], env);
@@ -1546,7 +1569,7 @@ function atomListBuiltin(kind) {
1546
1569
  }
1547
1570
  const invalid = supplied.find((item) => item.type !== VAR &&
1548
1571
  (kind === 'chars' ? !oneChar(item) :
1549
- item.type !== NUMBER || !isDecimalInteger(item.name) || !validCharacterCode(item)));
1572
+ item.type !== NUMBER || !isDecimalInteger(item.name) || !validCharacterCode(item, solver)));
1550
1573
  if (invalid) {
1551
1574
  if (kind === 'chars') throw new PrologError('type_error(character)', invalid);
1552
1575
  if (invalid.type !== NUMBER || !isDecimalInteger(invalid.name)) {
@@ -1554,26 +1577,32 @@ function atomListBuiltin(kind) {
1554
1577
  }
1555
1578
  throw new PrologError('representation_error(character_code)');
1556
1579
  }
1580
+ if (solver.isoStrict && characters(value.name).some((ch) => !isStrictIsoPcsCharacter(ch))) {
1581
+ throw new PrologError('representation_error(character)', value);
1582
+ }
1557
1583
  const items = characters(value.name).map((ch) =>
1558
1584
  kind === 'chars' ? atom(ch) : numberTerm(ch.codePointAt(0)));
1559
1585
  if (unify(goal.args[1], listFromItems(items), next)) yield next;
1560
1586
  return;
1561
1587
  }
1562
- if (unify(goal.args[0], atom(listToAtomInput(list, env, kind)), next)) yield next;
1588
+ if (unify(goal.args[0], atom(listToAtomInput(list, env, kind, solver)), next)) yield next;
1563
1589
  };
1564
1590
  }
1565
1591
  const atomCharsBuiltin = atomListBuiltin('chars');
1566
1592
  const atomCodesBuiltin = atomListBuiltin('codes');
1567
1593
 
1568
- function* charCodeBuiltin({ goal, env }) {
1594
+ function* charCodeBuiltin({ solver, goal, env }) {
1569
1595
  const char = deref(goal.args[0], env);
1570
1596
  const code = deref(goal.args[1], env);
1571
1597
  if (char.type === VAR && code.type === VAR) throw new PrologError('instantiation_error');
1572
1598
  if (char.type !== VAR && !oneChar(char)) throw new PrologError('type_error(character)', char);
1599
+ if (char.type === ATOM && solver.isoStrict && !isStrictIsoPcsCharacter(char.name)) {
1600
+ throw new PrologError('representation_error(character)', char);
1601
+ }
1573
1602
  if (code.type !== VAR && (code.type !== NUMBER || !isDecimalInteger(code.name))) {
1574
1603
  throw new PrologError('type_error(integer)', code);
1575
1604
  }
1576
- if (code.type !== VAR && !validCharacterCode(code)) throw new PrologError('representation_error(character_code)');
1605
+ if (code.type !== VAR && !validCharacterCode(code, solver)) throw new PrologError('representation_error(character_code)');
1577
1606
  const next = env.clone();
1578
1607
  if (char.type === ATOM) {
1579
1608
  if (unify(goal.args[1], numberTerm(char.name.codePointAt(0)), next)) yield next;
@@ -1583,7 +1612,7 @@ function* charCodeBuiltin({ goal, env }) {
1583
1612
  function skipNumberLayout(text, start) {
1584
1613
  let position = start;
1585
1614
  while (true) {
1586
- while (position < text.length && /[\u0009-\u000d\u0020]/.test(text[position])) {
1615
+ while (position < text.length && /[\u0000-\u0020\u007f]/.test(text[position])) {
1587
1616
  position++;
1588
1617
  }
1589
1618
  if (text[position] === '%') {
@@ -1647,7 +1676,7 @@ function parseIsoNumber(text) {
1647
1676
  // there without separating layout: `/` *can* continue the graphic token,
1648
1677
  // and the eager-consumer rule therefore keeps `-/**/1` ill-formed (the
1649
1678
  // number_chars continuation corpus case 24).
1650
- if (/[\u0009-\u000d\u0020]/.test(next) || next === '%') {
1679
+ if (/[\u0000-\u0020\u007f]/.test(next) || next === '%') {
1651
1680
  position = skipNumberLayout(text, position + 1);
1652
1681
  sign = '-';
1653
1682
  }
package/src/parser.js CHANGED
@@ -2,6 +2,7 @@
2
2
  // It preserves the compact Prolog-like syntax while producing Term objects for the solver.
3
3
  import { ATOM, COMPOUND, atom, compound, cons, emptyList, numberTerm, variable } from './term.js';
4
4
  import { continuesGraphicToken, isTerminatingFullStop } from './syntax-scan.js';
5
+ import { CharacterRepresentationError, isStrictIsoPcsCodePoint } from './iso-character.js';
5
6
 
6
7
 
7
8
  export class NumberRepresentationError extends Error {
@@ -33,7 +34,10 @@ const TOK = {
33
34
  };
34
35
 
35
36
  function isWhitespaceCode(code) {
36
- return code === 32 || code === 9 || code === 10 || code === 13 || code === 12 || code === 11;
37
+ // EyeProlog classifies ASCII C0 controls and DEL as layout characters. In
38
+ // strict mode these are the implementation-defined extended-layout members
39
+ // of the ASCII processor character set.
40
+ return (code >= 0 && code <= 32) || code === 127;
37
41
  }
38
42
 
39
43
  function isDigitCode(code) {
@@ -232,7 +236,11 @@ class Parser {
232
236
  return this.parserFlagState.charConversions.get(character) ?? character;
233
237
  }
234
238
  rawPeek(offset = 0) {
235
- return this.source[this.pos + offset] ?? '';
239
+ const ch = this.source[this.pos + offset] ?? '';
240
+ if (this.strictIso && ch && !isStrictIsoPcsCodePoint(ch.charCodeAt(0))) {
241
+ throw new CharacterRepresentationError();
242
+ }
243
+ return ch;
236
244
  }
237
245
  rawTake() {
238
246
  const ch = this.rawPeek();
@@ -410,6 +418,7 @@ class Parser {
410
418
  if (code > 0x10ffff || (code >= 0xd800 && code <= 0xdfff)) {
411
419
  throw new Error(`parse line ${line}: character escape out of range`);
412
420
  }
421
+ if (this.strictIso && !isStrictIsoPcsCodePoint(code)) throw new CharacterRepresentationError();
413
422
  return String.fromCodePoint(code);
414
423
  }
415
424
  if (/^[0-7]$/.test(escaped)) {
@@ -420,6 +429,7 @@ class Parser {
420
429
  if (code > 0x10ffff || (code >= 0xd800 && code <= 0xdfff)) {
421
430
  throw new Error(`parse line ${line}: character escape out of range`);
422
431
  }
432
+ if (this.strictIso && !isStrictIsoPcsCodePoint(code)) throw new CharacterRepresentationError();
423
433
  return String.fromCodePoint(code);
424
434
  }
425
435
  // A backslash followed by a decimal digit is numeric-escape syntax, but
@@ -46,7 +46,7 @@ export function isTerminatingFullStop(source, index, convert = null) {
46
46
  // token character accepted by continuesGraphicToken().
47
47
  if (continuesGraphicToken(source, index, convert)) return false;
48
48
  if (next === '' || next === '%' || next === '\n' || next === '\r') return true;
49
- if (/^[\u0009\u000b\u000c\u0020]$/.test(next)) return true;
49
+ if (/^[\u0000-\u0020\u007f]$/.test(next)) return true;
50
50
  return false;
51
51
  }
52
52
 
@@ -30,8 +30,8 @@ error-ordering alternative to an individual executable assertion.
30
30
 
31
31
  | Standard area | Status | Current evidence |
32
32
  | --- | --- | --- |
33
- | Clause 6 — tokens, terms, lists, operators, quoted text | audit | Complete vendored WG17 syntax matrix, `lexical_and_curly_terms`, `scryer_lexical_terms`, operator suites, syntax-error cases, quoted-layout/escape error cases, and writer/read-back regressions. |
34
- | 7.1-7.3 — term types, term order, unification | audit | Standard-order, identity, finite-tree and occurs-check suites, Corrigendum 2 term predicates. |
33
+ | Clause 6 — tokens, terms, lists, operators, quoted text | audit | Complete vendored WG17 syntax matrix, `lexical_and_curly_terms`, `scryer_lexical_terms`, operator suites, syntax-error cases, quoted-layout/escape error cases, writer/read-back regressions, and strict ASCII PCS/collation boundary tests. The implementation-defined 6.5/6.6 character-model decisions are now closed; wider shall-by-shall lexical mapping remains open. |
34
+ | 7.1-7.3 — term types, term order, unification | audit | Standard-order, identity, finite-tree and occurs-check suites, Corrigendum 2 term predicates, plus strict checks for the required `variable < float < integer < atom < compound` type order and PCS-based atom collation. |
35
35
  | 7.4 — Prolog text and directives | audit | All Part 1 directive indicators are parsed; include/ensure-loaded/operator/flag/character-conversion behavior has executable coverage. Preparation-time `char_conversion/2` now affects later unquoted source text and respects `char_conversion=off`. Cross-text `multifile/1` and ordering constraints still require explicit shall-by-shall audit. |
36
36
  | 7.5-7.6 — database and term/clause conversion | audit | Dynamic database and logical-update-view suites. Strict mode restores Part 1 private-static/public-dynamic `clause/2` access. Public/private and multi-text requirements still need complete mapping. |
37
37
  | 7.7 — execution and backtracking | audit | Control/search suites. Strict mode disables EyeProlog automatic tabling, cycle guards, and recursive numeric shortcuts so core execution uses ordinary clause selection/backtracking. |
@@ -59,10 +59,15 @@ conformance claims:
59
59
  when the `char_conversion` flag is `on`, leaves quoted characters unchanged,
60
60
  and feeds the same mapping into execution-time term input.
61
61
 
62
- The processor-character-set/collation rows in
63
- [ISO-IMPLEMENTATION-DEFINED.md](ISO-IMPLEMENTATION-DEFINED.md) remain explicit
64
- `audit gap`s. They were not papered over by narrowing strict mode without full
65
- WG17 and read/write/error validation.
62
+ A follow-on audit closes the processor-character-set/collation choices rather
63
+ than leaving them implicit. `--iso-strict` now selects the 128-character ASCII
64
+ PCS U+0000..U+007F, classifies C0 controls and DEL as extended layout
65
+ characters, and uses the code point itself as each collating-sequence integer.
66
+ Characters/codes outside that PCS raise representation errors in strict
67
+ parsing, term input, character conversion, and character-code predicates. The
68
+ normal profile retains Unicode scalar character data as an explicit extension.
69
+ The complete WG17 syntax matrix remains green under this narrower strict
70
+ boundary.
66
71
 
67
72
  ## Strict-core boundary
68
73
 
@@ -19,21 +19,21 @@ Status values are:
19
19
  - **defined** — the current behavior is implemented and stated here;
20
20
  - **not applicable** — the standard decision is conditional and the condition
21
21
  is false for EyeProlog's selected profile;
22
- - **audit gap** — the current code behavior is stated, but strict-mode
23
- conformance still needs a correction or a narrower profile before this row
24
- can support a full conformance claim.
22
+ - **audit gap** — retained for any future implementation-defined choice whose
23
+ code/documentation boundary is still unresolved. Open *normative* shall-by-
24
+ shall work is tracked separately in `ISO-COMPLIANCE.md`.
25
25
 
26
26
  ## Explicit implementation-defined decisions
27
27
 
28
28
  | Clause | Decision completed by ISO 5.4 documentation | EyeProlog choice | Status / implementation evidence |
29
29
  | --- | --- | --- | --- |
30
30
  | 5.5.11 | Reserved atoms and the effect of instantiating a variable to one | EyeProlog reserves no Prolog atom under 5.5.11. Atoms with implementation-looking names remain ordinary terms unless a particular predicate interprets them. | **defined** — term representation and built-ins in `src/term.js`, `src/iso.js`. |
31
- | 6.5 | Processor character set (PCS) | The unquoted ISO lexical classes are the Part 1 ASCII characters implemented by `src/parser.js`. Quoted character data additionally accepts Unicode scalar values. | **audit gap** — the Unicode quoted-character extension is currently also accepted by `--iso-strict`; strict extension rejection still needs a narrower PCS rule or a documented conforming classification. |
32
- | 6.5 | Classification of additional/extended PCS characters | Non-ASCII scalar values are not accepted as unquoted small-letter, capital-letter, graphic, solo, layout, or meta characters; they are accepted only inside quoted character data. | **audit gap** — same strict-mode boundary as the preceding row. |
33
- | 6.6 | Collating-sequence integers | Character codes are Unicode scalar values. Atom comparison uses ECMAScript string lexicographic order; on the ISO ASCII repertoire this is code-point order and satisfies the required monotonic ranges. | **defined** — `src/term.js` (`compareTerms`), `src/iso.js` character-code predicates. |
34
- | 6.6 | Collating values of control escapes and extended characters | Control escapes and character-code predicates use Unicode scalar values. Atom ordering is ECMAScript string order; for non-BMP one-char atoms that order is based on UTF-16 code units rather than scalar values. | **audit gap** — the ISO ASCII repertoire is conforming, but the extended-character collating rule still needs one coherent documented integer/order mapping in strict mode. |
31
+ | 6.5 | Processor character set (PCS) | In `--iso-strict`, PCS is the 128-character 7-bit ASCII set U+0000..U+007F. Normal mode additionally accepts Unicode scalar values in character data as an implementation-specific extension. | **defined** — `src/iso-character.js`, strict parser/character-I/O guards, and strict-core/WG17 coverage. |
32
+ | 6.5 | Classification of additional/extended PCS characters | Printable ASCII uses the lexical classes specified by Part 1. ASCII C0 controls U+0000..U+001F and DEL U+007F are EyeProlog's extended **layout** characters; they may therefore separate tokens, while quoted control values are written/read through the ISO escape forms. No non-ASCII character belongs to strict PCS. | **defined** — `src/parser.js`, `src/syntax-scan.js`, `src/iso-character.js`. |
33
+ | 6.6 | Collating-sequence integers | In strict mode each PCS character's collating-sequence integer is its ASCII/Unicode code point, 0..127. Atom comparison is lexicographic by the same code-unit values, which coincide with those integers throughout strict PCS and satisfy the required capital-letter, small-letter, and decimal-digit constraints. | **defined** — `src/iso-character.js`, `src/term.js` (`compareTerms`), `src/iso.js` character-code predicates. |
34
+ | 6.6 | Collating values of control escapes and extended characters | Strict control/extended-layout characters use their ASCII code point as collating integer, so octal/hexadecimal escapes and character-code constants map to the same 0..127 PCS. Normal-mode Unicode character codes use Unicode scalar values; that broader ordering is outside the strict Part 1 profile. | **defined** — `src/iso-character.js`, parser escape handling, `char_code/2`, `atom_codes/2`, and WG17 escape cases. |
35
35
  | 7.1.2.2 | Mapping between a character code and bytes | Text file streams decode and encode UTF-8. Binary streams expose bytes 0..255 directly. | **defined** — `src/io.js`. |
36
- | 7.1.4.1 | Set `C` of characters represented by one-char atoms | Character predicates accept Unicode scalar values U+0000..U+10FFFF excluding surrogate code points. | **defined** — `src/iso.js` character-code validation. |
36
+ | 7.1.4.1 | Set `C` of characters represented by one-char atoms | In strict mode `C` is exactly the ASCII PCS U+0000..U+007F. Normal mode extends character predicates to Unicode scalar values U+0000..U+10FFFF excluding surrogates. | **defined** — `src/iso-character.js`, `src/iso.js` character-code validation. |
37
37
  | 7.4.2.4 | Whether `op/3` directives affect other Prolog texts or execution | An `op/3` directive changes parsing of subsequent text loaded into the same `Program`; the resulting operator table is also used by execution-time term I/O. Separately created `Program` objects are independent. | **defined** — `src/parser.js`, `src/program.js`, `src/iso.js`. |
38
38
  | 7.4.2.5 | Whether directive-created `Convc` affects other text/execution | Yes. A `char_conversion/2` directive updates preparation-time conversion for later unquoted source characters and the recorded mapping initializes execution-time term input. Quoted characters are not converted; `char_conversion=off` disables following preparation-time conversion. | **defined** — `src/parser.js`, `src/program.js`, `src/solver.js`; strict-core regression coverage. |
39
39
  | 7.4.2.6 | Order of `initialization/1` goals | Initialization goals run once, in source/inclusion order, before requested goals; each must obtain a first solution. | **defined** — `Program.initializations`, `Solver.runInitializations()`. |
@@ -68,7 +68,7 @@ Status values are:
68
68
  | 7.11.2.3 | Default `max_arity` | `unbounded` in the Prolog model, subject to host memory and practical JavaScript array/index limits. | **defined** — `src/solver.js`; relevant guards report representation/resource errors. |
69
69
  | 7.11.2.5 | Default `double_quotes` | `chars`. | **defined** — `src/solver.js`, parser flag state. |
70
70
  | 7.12.1 | Second argument of `error/2` | The default context term is the atom `eyeprolog`. A few implementation-specific diagnostics may deliberately supply a more specific context term. | **defined** — `formalErrorTerm()` in `src/iso.js`. |
71
- | 7.12.2(f) | Implementation-defined representation limits | Character and character-code operations use Unicode scalar limits; arity/integer values are modeled as unbounded but may hit host/resource limits. Float input overflow uses the implementation-specific `max_float`/`min_float` representation names documented by the STC-oriented tests. | **defined** — parser/ISO numeric and character guards. |
71
+ | 7.12.2(f) | Implementation-defined representation limits | Strict character and character-code operations are limited to the selected ASCII PCS/collating integers 0..127. Normal mode extends characters to Unicode scalar values. Arity/integer values are modeled as unbounded but may hit host/resource limits. Float input overflow uses the implementation-specific `max_float`/`min_float` representation names documented by the STC-oriented tests. | **defined** — parser/ISO numeric and character guards. |
72
72
  | 8.17.1 | Implementation-defined flag value ranges | Strict mode exposes only Part 1 core flags and their standard value sets. Normal mode additionally exposes EyeProlog's `occurs_check` flag. With `bounded=false`, `max_integer` and `min_integer` have no current value and their `current_prolog_flag/2` queries fail. | **defined** — strict registry/flag filtering in `src/solver.js`. |
73
73
  | 8.17.3 | Other effects of `halt/0` | Terminates EyeProlog execution and returns host/process status `0`; it produces no Prolog solution. | **defined** — `HaltSignal`, `haltBuiltin()`, CLI/runner handling. |
74
74
  | 8.17.4 | Meaning/effects of `halt(Status)` | Integer `Status` is converted to the host process/runner halt code; it produces no Prolog solution. | **defined** — `haltBuiltin()`, `src/execute.js`, `src/cli.js`. |
@@ -105,7 +105,7 @@ families; `--iso-strict` is intended to remove their Part 1 interpretation.
105
105
 
106
106
  | Part 1 extension hook | EyeProlog normal-profile feature | Strict-core disposition |
107
107
  | --- | --- | --- |
108
- | 5.5.1 Syntax | Part 2 modules, Part 3 grammar-rule expansion, and embedded quad syntax | Module directives are rejected; grammar rules remain ordinary `-->/2` terms rather than being expanded; quad syntax is rejected. Unicode quoted-character handling is the open PCS audit item above. |
108
+ | 5.5.1 Syntax | Part 2 modules, Part 3 grammar-rule expansion, embedded quad syntax, and normal-mode Unicode character data | Module directives are rejected; grammar rules remain ordinary `-->/2` terms rather than being expanded; quad syntax is rejected; strict character syntax/data is limited to the documented ASCII PCS. |
109
109
  | 5.5.2 Predefined operators | Part 3 `|` and EyeProlog's labelable infix `(?-)/2`; CLP(Z) operators when that library is imported | Only the Part 1 operator table is predefined; a conforming `op/3` may still add permitted operators. |
110
110
  | 5.5.3 Character-conversion mapping | No non-identity initial `Convc` extension | Identity initial mapping. |
111
111
  | 5.5.4 Types | No additional runtime Prolog term type is exposed by the core solver | Only variable, integer, float, atom, and compound term ordering participates in strict mode. |
@@ -8,7 +8,7 @@ compliance audit and the remaining work before a full conformance claim.
8
8
 
9
9
  | Standard area | Implementation | Representative executable coverage |
10
10
  | --- | --- | --- |
11
- | Clause 6 lexical and term syntax | tokenizer, operator parser, lists, curly terms, quotes, numeric syntax, comments | `scryer_lexical_terms`, `lexical_and_curly_terms`, `double_quoted_lists`, `corrigendum1_double_quote_operator`, `wg17_syntax_high_risk`, `wg17_invalid_octal_escape`, `wg17_unterminated_quoted_token`, `wg17_literal_newline_in_quote`, `wg17_non_iso_escape`, syntax error cases |
11
+ | Clause 6 lexical and term syntax | tokenizer, operator parser, lists, curly terms, quotes, numeric syntax, comments, strict ASCII PCS/collation | `scryer_lexical_terms`, `lexical_and_curly_terms`, `double_quoted_lists`, `corrigendum1_double_quote_operator`, `wg17_syntax_high_risk`, `wg17_invalid_octal_escape`, `wg17_unterminated_quoted_token`, `wg17_literal_newline_in_quote`, `wg17_non_iso_escape`, strict PCS/collation tests in `run-iso-strict.mjs`, syntax error cases |
12
12
  | Clause 7 term order and unification | finite-tree unification, identity, standard order, errors | `unification_control_information`, `swipl_occurs_check`, `term_modes_and_ordering`, `logtalk_compare_standard_order` |
13
13
  | Clause 7 control and exceptions | call, cut, conjunction, disjunction, if-then-else, catch and throw | `cut_control`, `control_and_terms`, `exceptions_and_flags`, `corrigenda_catch_callability`, `throw_copies_ball` |
14
14
  | 8.2-8.5 term predicates | unification, Corrigendum 2 tests, comparison, sorting, creation and decomposition | `corrigenda_term_predicates`, `corrigenda_sort_keysort`, `logtalk_arg_unification`, `logtalk_univ`, associated error cases |
@@ -26,7 +26,10 @@ identify standards-derived behavior; other directories cover EyeProlog host
26
26
  contracts and extensions. EyeProlog-only execution features such as automatic
27
27
  tabling and `tnot/1` well-founded negation are outside the Part 1 strict-core
28
28
  claim. Their focused semantic coverage lives primarily in regression tests;
29
- `tnot/1` is absent from the strict ISO registry.
29
+ `tnot/1` is absent from the strict ISO registry. The strict processor character
30
+ set is the documented 7-bit ASCII PCS with ASCII-code collation; normal-mode
31
+ Unicode character data is tested as an extension rather than folded into the
32
+ Part 1 claim.
30
33
 
31
34
  All conformance files live under topic directories such as `arithmetic/`, `lists/`, `syntax/`, or `variables/`; new top-level numbered files should not be added. The report uses those directories as coverage categories.
32
35
 
@@ -56,6 +56,57 @@ export function runIsoStrict(reporter = new TestReporter()) {
56
56
  equal(Boolean(program.findGroup('raw', 1)?.clauses.some((clause) => clause.head.args[0]?.name === 'x')), true, 'conversion disabled');
57
57
  });
58
58
 
59
+
60
+ reporter.test('uses a documented 7-bit ASCII processor character set and collation', () => {
61
+ const program = Program.parse('', { isoStrict: true });
62
+ const solver = new Solver(program, { isoStrict: true });
63
+ const answers = (text) => [...solver.solve([parseGoalText(text, { isoStrict: true })], new Env(), 0)].length;
64
+ equal(answers("char_code('\\0\\',0)"), 1, 'NUL collating integer');
65
+ equal(answers("char_code('A',65)"), 1, 'A collating integer');
66
+ equal(answers("char_code('\\177\\',127)"), 1, 'DEL collating integer');
67
+ equal(answers("'\\0\\' @< 'A'"), 1, 'control before capital');
68
+ equal(answers("'A' @< 'a'"), 1, 'capital before small letter');
69
+ });
70
+
71
+
72
+ reporter.test('follows the Part 1 standard term-type and atom ordering', () => {
73
+ const program = Program.parse('', { isoStrict: true });
74
+ const solver = new Solver(program, { isoStrict: true });
75
+ const answers = (text) => [...solver.solve([parseGoalText(text, { isoStrict: true })], new Env(), 0)].length;
76
+ equal(answers("X @< 1.0"), 1, 'variable before float');
77
+ equal(answers("1.0 @< 1"), 1, 'float before integer');
78
+ equal(answers("1 @< a"), 1, 'integer before atom');
79
+ equal(answers("a @< f(a)"), 1, 'atom before compound');
80
+ equal(answers("'' @< 'A'"), 1, 'null atom first');
81
+ equal(answers("'A' @< 'B'"), 1, 'atom collation');
82
+ });
83
+
84
+ reporter.test('rejects characters outside the strict processor character set', () => {
85
+ const sourceError = capture(() => Program.parse("p('é').\n", { isoStrict: true }));
86
+ equal(sourceError.formal, 'representation_error(character)', 'source representation error');
87
+
88
+ const readError = capture(() => run('', {
89
+ isoStrict: true,
90
+ goal: 'read(X)',
91
+ ioOptions: { input: "'é'." },
92
+ }));
93
+ equal(readError.formal, 'representation_error(character)', 'read representation error');
94
+ });
95
+
96
+ reporter.test('restricts strict character codes to the processor character set', () => {
97
+ const charCodeError = capture(() => run('', { isoStrict: true, goal: 'char_code(_,128)' }));
98
+ equal(charCodeError.formal, 'representation_error(character_code)', 'char_code/2');
99
+ const atomCodesError = capture(() => run('', { isoStrict: true, goal: 'atom_codes(_, [128])' }));
100
+ equal(atomCodesError.formal, 'representation_error(character_code)', 'atom_codes/2');
101
+ const putCodeError = capture(() => run('', { isoStrict: true, goal: 'put_code(128)' }));
102
+ equal(putCodeError.formal, 'representation_error(character_code)', 'put_code/1');
103
+ });
104
+
105
+ reporter.test('keeps broader Unicode character handling as a normal-mode extension', () => {
106
+ const result = run('', { goal: "char_code('é',233)" });
107
+ equal(result.stdout, "char_code('é', 233).\n", 'normal Unicode char_code/2');
108
+ });
109
+
59
110
  reporter.test('uses the Part 1 predefined operator table', () => {
60
111
  const program = Program.parse('', { isoStrict: true });
61
112
  equal(program.operators.has('fx\u0000?-'), true, 'fx ?-');
@@ -5119,8 +5119,13 @@ this part makes control, reflection, state, operators, and streams explicit.
5119
5119
  For Part 1 portability work, EyeProlog also provides a strict core mode:
5120
5120
  `--iso-strict` on the CLI or `isoStrict: true` in the JavaScript API restricts
5121
5121
  the language/runtime surface to ISO/IEC 13211-1:1995 plus Technical Corrigenda
5122
- 1–3. Isolated mode and error cases live in `test/conformance/cases/iso/`. The
5123
- examples here compose those operations into programs worth changing and
5122
+ 1–3. Its processor character set is the 128-character ASCII set U+0000..U+007F;
5123
+ C0 controls and DEL are implementation-defined extended layout characters, and
5124
+ collating-sequence integers are the corresponding ASCII codes. Character data
5125
+ outside that PCS is rejected with a representation error in strict mode. Normal
5126
+ mode keeps EyeProlog's broader Unicode scalar character support as an explicit
5127
+ extension. Isolated mode and error cases live in `test/conformance/cases/iso/`.
5128
+ The examples here compose those operations into programs worth changing and
5124
5129
  rerunning.
5125
5130
 
5126
5131
  These facilities do not all have the same declarative character. Term
@@ -5495,15 +5500,20 @@ normal-mode profiles are documented and tested compatibility surfaces; they are
5495
5500
  not currently claimed as complete clause-by-clause certifications of Part 2 or
5496
5501
  Part 3.
5497
5502
 
5498
- Prolog source accepted by EyeProlog is UTF-8. `%` starts a line comment and
5499
- `/* ... */` delimits a block comment. Plain atoms begin with a
5503
+ Normal-mode Prolog source accepted by EyeProlog is UTF-8. `%` starts a line
5504
+ comment and `/* ... */` delimits a block comment. Plain atoms begin with a
5500
5505
  lowercase ASCII letter. Variables begin with uppercase or underscore. The bare
5501
5506
  `_` is fresh each time. Single quotes delimit quoted atoms; double quotes use
5502
- ISO double-quoted-list notation. Integers, decimals, scientific notation, binary/octal/
5503
- hexadecimal integers, and character-code constants are accepted.
5507
+ ISO double-quoted-list notation. Integers, decimals, scientific notation,
5508
+ binary/octal/hexadecimal integers, and character-code constants are accepted.
5504
5509
 
5505
- Unquoted names deliberately use ASCII spelling. Unicode belongs inside quoted
5506
- atoms and double-quoted lists:
5510
+ For `--iso-strict`, the processor character set is deliberately narrower and
5511
+ fully documented: U+0000..U+007F. Printable ASCII uses the Part 1 lexical
5512
+ classes; C0 controls and DEL are extended layout characters; character-code and
5513
+ collation values are the same ASCII integers 0..127. Non-ASCII input is a
5514
+ representation error. In normal mode, unquoted names still deliberately use
5515
+ ASCII spelling while Unicode scalar values may appear inside quoted atoms and
5516
+ double-quoted lists:
5507
5517
 
5508
5518
  ```eyeprolog
5509
5519
 
@@ -6036,8 +6046,8 @@ quoted_atom("ab"). % quoted_atom(ab)
6036
6046
  | `atom_length(+Atom,?Length)` | Counts Unicode code points, not UTF-16 code units. A supplied length must be a nonnegative integer. |
6037
6047
  | `atom_concat(?Prefix,?Suffix,?Whole)` | Concatenates two atoms, removes a supplied prefix or suffix, or enumerates every split when only `Whole` is bound. At least `Whole`, or both parts, must determine the operation. |
6038
6048
  | `sub_atom(+Atom,?Before,?Length,?After,?SubAtom)` | Enumerates substrings and their Unicode-code-point offsets. Supplied counts must be nonnegative integers. |
6039
- | `atom_chars(?Atom,?Chars)`, `atom_codes(?Atom,?Codes)` | Convert between an atom and a proper list of one-character atoms or Unicode scalar codes. At least one side must be instantiated. |
6040
- | `char_code(?Character,?Code)` | Converts one character atom and one Unicode scalar code. Surrogates and values outside `0..0x10ffff` raise a representation error. |
6049
+ | `atom_chars(?Atom,?Chars)`, `atom_codes(?Atom,?Codes)` | Convert between an atom and a proper list of one-character atoms or character codes. Strict mode uses its ASCII PCS/codes `0..127`; normal mode extends codes to Unicode scalar values. At least one side must be instantiated. |
6050
+ | `char_code(?Character,?Code)` | Converts one character atom and its collating/code value. Strict mode accepts only the ASCII PCS `0..127`; normal mode accepts Unicode scalar codes and rejects surrogates/out-of-range values. |
6041
6051
  | `number_chars(?Number,?Chars)`, `number_codes(?Number,?Codes)` | Convert finite numbers to canonical text or parse a proper character/code list using ISO number and negative-number syntax, including radix integers, character-code constants, and leading layout. The input is not parsed as a general term: grouping such as `(0)` is a syntax error. At least one side must be instantiated; malformed numeric input raises *syntax_error(number)*. |
6042
6052
 
6043
6053
  Conversions accept partial output lists when the atomic input is known, but
@@ -6078,12 +6088,12 @@ input or output.
6078
6088
  | `at_end_of_stream`, `at_end_of_stream(+Stream)` | Succeeds when the current or selected input position is at or beyond its content. |
6079
6089
  | `get_char(?Character)`, `get_char(+Stream,?Character)` | Reads one text character; end of input is `end_of_file`. |
6080
6090
  | `peek_char(?Character)`, `peek_char(+Stream,?Character)` | Observes the next text character without advancing. |
6081
- | `get_code(?Code)`, `get_code(+Stream,?Code)` | Reads a Unicode code point; end of input is `-1`. |
6082
- | `peek_code(?Code)`, `peek_code(+Stream,?Code)` | Observes the next Unicode code point without advancing. |
6091
+ | `get_code(?Code)`, `get_code(+Stream,?Code)` | Reads a character code; end of input is `-1`. Strict mode requires the ASCII PCS, while normal mode returns Unicode scalar codes. |
6092
+ | `peek_code(?Code)`, `peek_code(+Stream,?Code)` | Observes the next character code without advancing; strict mode requires the ASCII PCS. |
6083
6093
  | `get_byte(?Byte)`, `get_byte(+Stream,?Byte)` | Reads one unit from a binary stream; end of input is `-1`. |
6084
6094
  | `peek_byte(?Byte)`, `peek_byte(+Stream,?Byte)` | Observes the next binary unit without advancing. |
6085
6095
  | `put_char(+Character)`, `put_char(+Stream,+Character)` | Writes one character atom to a text stream. |
6086
- | `put_code(+Code)`, `put_code(+Stream,+Code)` | Writes one Unicode scalar code to a text stream. |
6096
+ | `put_code(+Code)`, `put_code(+Stream,+Code)` | Writes one character code to a text stream. Strict mode accepts `0..127`; normal mode accepts Unicode scalar codes. |
6087
6097
  | `put_byte(+Byte)`, `put_byte(+Stream,+Byte)` | Writes an integer in `0..255` to a binary stream. |
6088
6098
  | `nl`, `nl(+Stream)` | Writes a newline to a text stream. |
6089
6099
 
package/why-eyeprolog.md CHANGED
@@ -19,7 +19,10 @@ syntax.
19
19
 
20
20
  EyeProlog targets the Part 1 core together with Technical Corrigenda 1, 2,
21
21
  and 3, and provides documented module and definite-clause-grammar compatibility
22
- profiles for normal-mode programs. Its executable conformance matrix and tests
22
+ profiles for normal-mode programs. Strict mode makes its processor character
23
+ model explicit: 7-bit ASCII is the PCS, ASCII code points are the collating
24
+ integers, and normal-mode Unicode character data is an extension rather than an
25
+ implicit part of the Part 1 claim. Its executable conformance matrix and tests
23
26
  document the supported behavior, including an executable trace of the vendored
24
27
  active WG17 syntax cases. This is extensive implementation evidence, not a
25
28
  claim that every Part 1, Part 2, or Part 3 normative requirement has already