eyeprolog 1.5.23 → 1.5.24

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -3,7 +3,7 @@
3
3
  "publishConfig": {
4
4
  "access": "public"
5
5
  },
6
- "version": "1.5.23",
6
+ "version": "1.5.24",
7
7
  "description": "EyeProlog turns facts and rules into answers and proofs.",
8
8
  "type": "module",
9
9
  "main": "./index.js",
package/src/iso.js CHANGED
@@ -1762,7 +1762,7 @@ function parseReadTermText(text, solver) {
1762
1762
  const numericText = converted.slice(0, -1).replace(/[\u0009-\u000d\u0020]+$/, '');
1763
1763
  let numericTerm;
1764
1764
  try {
1765
- numericTerm = parseIsoNumber(numericText);
1765
+ numericTerm = parseIsoNumber(numericText, { isoStrict: solver.isoStrict });
1766
1766
  } catch (error) {
1767
1767
  // parseIsoNumber/1 exposes Prolog errors to number_chars/2. The stream
1768
1768
  // reader's candidate loop instead recognizes parser representation errors
@@ -2338,7 +2338,7 @@ function quotedNumberSign(text, start) {
2338
2338
  return null;
2339
2339
  }
2340
2340
 
2341
- function parseIsoNumber(text) {
2341
+ function parseIsoNumber(text, options = {}) {
2342
2342
  if (text.length === 0) return null;
2343
2343
  let position = skipNumberLayout(text, 0);
2344
2344
  let sign = '';
@@ -2375,7 +2375,7 @@ function parseIsoNumber(text) {
2375
2375
  // ISO floating-point syntax requires a decimal fraction before an exponent.
2376
2376
  if (/^-?\d+[eE][+-]?\d+$/.test(numericText)) return null;
2377
2377
  try {
2378
- const value = parseNumberTokenText(numericText);
2378
+ const value = parseNumberTokenText(numericText, options);
2379
2379
  if (isDecimalInteger(value.name)) return numberTerm(BigInt(value.name).toString());
2380
2380
  const finite = Number(value.name);
2381
2381
  if (!Number.isFinite(finite)) return null;
@@ -2447,7 +2447,7 @@ function numberListBuiltin(kind) {
2447
2447
  const text = numberListText(list, env, kind, value.type === NUMBER, solver);
2448
2448
  if (value.type === NUMBER) {
2449
2449
  if (text != null) {
2450
- const parsed = parseIsoNumber(text);
2450
+ const parsed = parseIsoNumber(text, { isoStrict: solver?.isoStrict === true });
2451
2451
  if (parsed == null) throw numberSyntaxError;
2452
2452
  if (sameNumber(value, parsed)) yield env.clone();
2453
2453
  return;
@@ -2458,7 +2458,7 @@ function numberListBuiltin(kind) {
2458
2458
  if (unify(goal.args[1], listFromItems(items), next)) yield next;
2459
2459
  return;
2460
2460
  }
2461
- const parsed = parseIsoNumber(text);
2461
+ const parsed = parseIsoNumber(text, { isoStrict: solver?.isoStrict === true });
2462
2462
  if (parsed == null) throw numberSyntaxError;
2463
2463
  const next = env.clone();
2464
2464
  if (unify(goal.args[0], parsed, next)) yield next;
package/src/parser.js CHANGED
@@ -439,6 +439,21 @@ class Parser {
439
439
  break;
440
440
  }
441
441
  }
442
+ integerDigits(digitPattern, line) {
443
+ let digits = '';
444
+ let separated = false;
445
+ while (digitPattern.test(this.peek())) {
446
+ digits += this.take();
447
+ if (this.strictIso || this.peek() !== '_') continue;
448
+ separated = true;
449
+ this.take();
450
+ this.skipWhitespaceAndComments();
451
+ if (!digitPattern.test(this.peek())) {
452
+ throw new Error(`parse line ${line}: bad digit separator`);
453
+ }
454
+ }
455
+ return { digits, separated };
456
+ }
442
457
  readEscape(line, options = {}) {
443
458
  const takeChar = () => options.raw ? this.rawTake() : this.take();
444
459
  const peekChar = () => options.raw ? this.rawPeek() : this.peek();
@@ -646,17 +661,16 @@ class Parser {
646
661
  const kind = this.take();
647
662
  const radix = kind === 'b' ? 2 : kind === 'o' ? 8 : 16;
648
663
  const digitPattern = radix === 2 ? /^[01]$/ : radix === 8 ? /^[0-7]$/ : /^[0-9A-Fa-f]$/;
649
- let digits = '';
650
- while (digitPattern.test(this.peek())) digits += this.take();
664
+ const { digits } = this.integerDigits(digitPattern, line);
651
665
  if (!digits) throw new Error(`parse line ${line}: bad radix integer`);
652
666
  let integer = 0n;
653
667
  for (const digit of digits) integer = integer * BigInt(radix) + BigInt(Number.parseInt(digit, radix));
654
668
  if (negative) integer = -integer;
655
669
  return { type: TOK.NUMBER, text: integer.toString(), line };
656
670
  }
657
- while (isDigitCode(this.peek().charCodeAt(0))) this.take();
671
+ const { digits, separated } = this.integerDigits(/^[0-9]$/, line);
658
672
  let hasFraction = false;
659
- if (this.peek() === '.' && isDigitCode(this.peek(1).charCodeAt(0))) {
673
+ if (!separated && this.peek() === '.' && isDigitCode(this.peek(1).charCodeAt(0))) {
660
674
  hasFraction = true;
661
675
  this.take();
662
676
  while (isDigitCode(this.peek().charCodeAt(0))) this.take();
@@ -674,7 +688,7 @@ class Parser {
674
688
  }
675
689
  }
676
690
  let text = this.convertedSlice(start, this.pos);
677
- if (!hasFraction) text = BigInt(text).toString();
691
+ if (!hasFraction) text = BigInt(`${negative ? '-' : ''}${digits}`).toString();
678
692
  else text = finiteFloatTokenText(text);
679
693
  return { type: TOK.NUMBER, text, line };
680
694
  }
@@ -1704,8 +1718,43 @@ export function parseProgramText(source, options = {}) {
1704
1718
 
1705
1719
  const invalidNumberTokenError = new Error('not exactly one number token');
1706
1720
 
1707
- export function parseNumberTokenText(text) {
1721
+ function skipDigitSeparatorLayout(source, start) {
1722
+ let position = start;
1723
+ while (true) {
1724
+ while (isWhitespaceCharacter(source[position] ?? '')) position++;
1725
+ if (source[position] === '%') {
1726
+ const newline = source.indexOf('\n', position + 1);
1727
+ if (newline < 0) return source.length;
1728
+ position = newline + 1;
1729
+ continue;
1730
+ }
1731
+ if (source.startsWith('/*', position)) {
1732
+ const end = source.indexOf('*/', position + 2);
1733
+ if (end < 0) return source.length;
1734
+ position = end + 2;
1735
+ continue;
1736
+ }
1737
+ return position;
1738
+ }
1739
+ }
1740
+
1741
+ function separatedIntegerDigits(source, start, digitPattern, enabled) {
1742
+ let position = start;
1743
+ let digits = '';
1744
+ let separated = false;
1745
+ while (digitPattern.test(source[position] ?? '')) {
1746
+ digits += source[position++];
1747
+ if (!enabled || source[position] !== '_') continue;
1748
+ separated = true;
1749
+ position = skipDigitSeparatorLayout(source, position + 1);
1750
+ if (!digitPattern.test(source[position] ?? '')) throw invalidNumberTokenError;
1751
+ }
1752
+ return { digits, position, separated };
1753
+ }
1754
+
1755
+ export function parseNumberTokenText(text, options = {}) {
1708
1756
  const source = String(text ?? '');
1757
+ const digitSeparators = options.isoStrict !== true;
1709
1758
  let position = 0;
1710
1759
  let negative = false;
1711
1760
  if (source[position] === '-') {
@@ -1771,8 +1820,9 @@ export function parseNumberTokenText(text) {
1771
1820
  const radix = kind === 'b' ? 2 : kind === 'o' ? 8 : 16;
1772
1821
  const digitPattern = radix === 2 ? /^[01]$/ : radix === 8 ? /^[0-7]$/ : /^[0-9A-Fa-f]$/;
1773
1822
  position += 2;
1774
- let digits = '';
1775
- while (digitPattern.test(source[position] ?? '')) digits += source[position++];
1823
+ const scanned = separatedIntegerDigits(source, position, digitPattern, digitSeparators);
1824
+ const { digits } = scanned;
1825
+ position = scanned.position;
1776
1826
  if (!digits || position !== source.length) throw invalidNumberTokenError;
1777
1827
  let integer = 0n;
1778
1828
  for (const digit of digits) integer = integer * BigInt(radix) + BigInt(Number.parseInt(digit, radix));
@@ -1780,11 +1830,12 @@ export function parseNumberTokenText(text) {
1780
1830
  return numberTerm(integer.toString());
1781
1831
  }
1782
1832
 
1783
- const digitsStart = position;
1784
- while (isDigitCode(source.charCodeAt(position))) position++;
1785
- if (position === digitsStart) throw invalidNumberTokenError;
1833
+ const scanned = separatedIntegerDigits(source, position, /^[0-9]$/, digitSeparators);
1834
+ const { digits, separated } = scanned;
1835
+ position = scanned.position;
1836
+ if (!digits) throw invalidNumberTokenError;
1786
1837
  let hasFraction = false;
1787
- if (source[position] === '.' && isDigitCode(source.charCodeAt(position + 1))) {
1838
+ if (!separated && source[position] === '.' && isDigitCode(source.charCodeAt(position + 1))) {
1788
1839
  hasFraction = true;
1789
1840
  position++;
1790
1841
  while (isDigitCode(source.charCodeAt(position))) position++;
@@ -1797,7 +1848,7 @@ export function parseNumberTokenText(text) {
1797
1848
  if (position === exponentStart) throw invalidNumberTokenError;
1798
1849
  }
1799
1850
  if (position !== source.length) throw invalidNumberTokenError;
1800
- if (/^-?\d+$/.test(source)) return numberTerm(BigInt(source).toString());
1851
+ if (!hasFraction) return numberTerm(BigInt(`${negative ? '-' : ''}${digits}`).toString());
1801
1852
  return numberTerm(finiteFloatTokenText(source));
1802
1853
  }
1803
1854
 
@@ -113,7 +113,7 @@ families; `--iso-strict` is intended to remove their Part 1 interpretation.
113
113
 
114
114
  | Part 1 extension hook | EyeProlog normal-profile feature | Strict-core disposition |
115
115
  | --- | --- | --- |
116
- | 5.5.1 Syntax | Part 2 modules, Part 3 grammar-rule expansion, embedded quad syntax, and Trealla-compatible `"text"||Tail` right-splicing for double-quoted `chars`/`codes` lists | Module directives are rejected; grammar rules remain ordinary `-->/2` terms rather than being expanded; quad syntax and double-bar right-splicing are rejected. Unicode PCS membership/classification is implementation defined and therefore shared with normal mode rather than treated as an extension. |
116
+ | 5.5.1 Syntax | Part 2 modules, Part 3 grammar-rule expansion, embedded quad syntax, digit-separated integer constants (`1_000`, `0xCA_FE`, with optional layout after `_`), and Trealla-compatible `"text"||Tail` right-splicing for double-quoted `chars`/`codes` lists | Module directives are rejected; grammar rules remain ordinary `-->/2` terms rather than being expanded; quad syntax, integer digit separators, and double-bar right-splicing are rejected. Unicode PCS membership/classification is implementation defined and therefore shared with normal mode rather than treated as an extension. |
117
117
  | 5.5.2 Predefined operators | Part 3 `|` and EyeProlog's labelable infix `(?-)/2`; CLP(Z) operators when that library is imported | Only the Part 1 operator table is predefined; a conforming `op/3` may still add permitted operators. |
118
118
  | 5.5.3 Character-conversion mapping | No non-identity initial `Convc` extension | Identity initial mapping. |
119
119
  | 5.5.4 Types | The normal JavaScript API exposes an implementation-specific string term `stringTerm(Text)`. It is disjoint from the five Part 1 term types; normal term order places it after atoms and before compound terms, and normal `atomic/1` treats it as atomic. It has no Prolog source token syntax (double-quoted source still follows `double_quotes`), is non-callable for term-to-clause conversion, is not evaluable as an arithmetic expression, and writes as a double-quoted host string. | Strict mode rejects a programmatic string term at program/goal entry with `representation_error(term)`, so the additional type cannot enter the Part 1 execution domain. **defined** — `src/term.js`, `src/program.js`, `src/solver.js`; strict API-boundary regression. |
@@ -34,7 +34,7 @@ certification claim.
34
34
  | Requirement | Status | EyeProlog decision / evidence |
35
35
  | --- | --- | --- |
36
36
  | 5.5 general extension rule | covered | normal mode may provide documented extensions; strict mode removes their Part 1 interpretation rather than changing implementation-defined choices |
37
- | 5.5.1 syntax extensions preserve standard token/text meaning | covered | WG17 syntax is a release gate and strict mode removes module/DCG/quad interpretation. Every vendored WG17 case that succeeds in the strict reader has the same observable outcome in normal mode; the focused Clause 6 gate separately covers each standard token/term family and malformed counterparts. |
37
+ | 5.5.1 syntax extensions preserve standard token/text meaning | covered | WG17 syntax is a release gate and strict mode removes module/DCG/quad, digit-separator, and double-bar interpretation. Every vendored WG17 case that succeeds in the strict reader has the same observable outcome in normal mode; the focused Clause 6 gate separately covers each standard token/term family and malformed counterparts. |
38
38
  | 5.5.2 additional predefined operators | covered | strict mode starts from the Part 1 predefined operator table; normal-profile extra operators are documented and filtered |
39
39
  | 5.5.3 initial character-conversion mapping | covered | identity initial mapping; user changes are exercised through preparation/execution `char_conversion/2` behavior |
40
40
  | 5.5.4 additional term types | covered | the normal JavaScript API's `stringTerm(Text)` is documented as an implementation-specific sixth term type, including disjointness, ordering, clause conversion, lack of source token syntax, expression behavior, and writing; strict program/goal entry rejects that type with `representation_error(term)` |
@@ -7252,6 +7252,54 @@ function whiteBoxCases() {
7252
7252
  }
7253
7253
  },
7254
7254
  },
7255
+ {
7256
+ name: 'normal integer syntax accepts WG17 digit separators (issue #89)',
7257
+ run: () => {
7258
+ const values = parseGoalText(`values(
7259
+ 1_000,
7260
+ 0b1010_0101,
7261
+ 0o7_ 7,
7262
+ 0xCA_/* digit group */FE,
7263
+ -9_% line group
7264
+ 223
7265
+ )`).args;
7266
+ assertEqual(values.map((value) => value.name).join(','), '1000,165,63,51966,-9223',
7267
+ 'decimal and radix separator values');
7268
+ const program = Program.parse('decimal(1_000).\nhexadecimal(0xCA_FE).\n');
7269
+ assertEqual(program.findGroup('decimal', 1)?.clauses[0].head.args[0].name, '1000',
7270
+ 'program decimal literal');
7271
+ assertEqual(program.findGroup('hexadecimal', 1)?.clauses[0].head.args[0].name, '51966',
7272
+ 'program radix literal');
7273
+ assertEqual(parseNumberTokenText('1_ /* group */ 000').name, '1000',
7274
+ 'number-token conversion with layout');
7275
+ assertEqual(
7276
+ run('', { goal: 'number_chars(N,"1_000")' }).stdout,
7277
+ 'number_chars(1000, "1_000").\n',
7278
+ 'normal number_chars input',
7279
+ );
7280
+ assertEqual(
7281
+ run('', { goal: 'read_term(N, [])', ioOptions: { input: '1_ /* group */ 000. ' } }).stdout,
7282
+ 'read_term(1000, []).\n',
7283
+ 'stream term input',
7284
+ );
7285
+
7286
+ for (const source of ['1__000', '1_', '1_a', '0b1_2', '1_1.25', '1.2_5', '1.0e1_0']) {
7287
+ let threw = false;
7288
+ try { parseGoalText(`p(${source})`); } catch (_) { threw = true; }
7289
+ assertEqual(threw, true, `malformed or non-integer separator rejected: ${source}`);
7290
+ }
7291
+
7292
+ for (const source of ['1_000', '0b1010_0101', '0o7_7', '0xCA_FE']) {
7293
+ let threw = false;
7294
+ try { parseGoalText(`p(${source})`, { isoStrict: true }); } catch (_) { threw = true; }
7295
+ assertEqual(threw, true, `strict syntax rejects ${source}`);
7296
+ }
7297
+ let strictConversion = null;
7298
+ try { run('', { isoStrict: true, goal: 'number_chars(N,"1_000")' }); }
7299
+ catch (error) { strictConversion = error; }
7300
+ assertEqual(strictConversion?.formal, 'syntax_error(number)', 'strict number_chars rejection');
7301
+ },
7302
+ },
7255
7303
  {
7256
7304
  name: 'double_quotes(true) remains effective with ignore_ops(true) in either option order (issue #88 follow-up)',
7257
7305
  run: () => {
@@ -465,6 +465,11 @@ for `[a,b|Tail]`; with `double_quotes(codes)`, it denotes `[97,98|Tail]`. The
465
465
  splice is not available when `double_quotes(atom)` is active, and
466
466
  `--iso-strict` rejects it as an implementation-specific syntax extension.
467
467
 
468
+ Normal-mode integer constants may use one underscore between adjacent digits,
469
+ as in `1_000` or `0xCA_FE`. Layout, including comments and newlines, may follow
470
+ the underscore before the next digit. Separators do not apply to floating-point
471
+ fractions or exponents, and `--iso-strict` rejects them.
472
+
468
473
  Plain atom constants begin with a lowercase ASCII letter. Variables begin with
469
474
  an uppercase letter or underscore. The bare `_` is anonymous and every
470
475
  occurrence is fresh. `_Name` is a named variable; repeated occurrences refer to
@@ -5648,8 +5653,10 @@ lowercase ASCII letter. Variables begin with uppercase or underscore. The bare
5648
5653
  `_` is fresh each time. Single quotes delimit quoted atoms; double quotes use
5649
5654
  ISO double-quoted-list notation. Integers, decimals, scientific notation,
5650
5655
  binary/octal/hexadecimal integers, and character-code constants are accepted.
5651
- Normal mode additionally accepts the Trealla-compatible `"text"||Tail`
5652
- right-splice for double-quoted `chars`/`codes` lists; strict ISO mode does not.
5656
+ Normal mode additionally accepts digit-separated integer constants such as
5657
+ `1_000` and `0xCA_FE`, with optional layout after the underscore, and the
5658
+ Trealla-compatible `"text"||Tail` right-splice for double-quoted `chars`/`codes`
5659
+ lists; strict ISO mode accepts neither syntax extension.
5653
5660
 
5654
5661
  The processor character set is shared by normal and `--iso-strict` modes because
5655
5662
  Part 1 makes it implementation defined rather than an extension boundary.
@@ -6190,7 +6197,7 @@ quoted_atom("ab"). % quoted_atom(ab)
6190
6197
  - **`sub_atom(+Atom,?Before,?Length,?After,?SubAtom)`** — Enumerates substrings and their Unicode-code-point offsets. Supplied counts must be nonnegative integers.
6191
6198
  - **`atom_chars(?Atom,?Chars)`, `atom_codes(?Atom,?Codes)`** — Convert between an atom and a proper list of one-character atoms or character codes. Both profiles use EyeProlog's Unicode scalar PCS/codes; surrogates and values above U+10FFFF are rejected. At least one side must be instantiated.
6192
6199
  - **`char_code(?Character,?Code)`** — Converts one character atom and its collating/code value. Both profiles accept Unicode scalar codes and reject surrogates/out-of-range values.
6193
- - **`number_chars(?Number,?Chars)`, `number_codes(?Number,?Codes)`** — Convert finite numbers to canonical text or parse a proper character/code list using ISO number and negative-number syntax, including radix integers, character-code constants, and leading layout. The input is not parsed as a general term: grouping such as `(0)` is a syntax error. At least one side must be instantiated; malformed numeric input raises *syntax_error(number)*.
6200
+ - **`number_chars(?Number,?Chars)`, `number_codes(?Number,?Codes)`** — Convert finite numbers to canonical text or parse a proper character/code list using number and negative-number syntax, including radix integers, character-code constants, and leading layout. Normal mode also accepts digit separators in integer input; strict mode retains the ISO syntax. The input is not parsed as a general term: grouping such as `(0)` is a syntax error. At least one side must be instantiated; malformed numeric input raises *syntax_error(number)*.
6194
6201
 
6195
6202
  Conversions accept partial output lists when the atomic input is known, but
6196
6203
  constructing an atom or number requires a complete proper list with no unbound