unicode-logic-kit 0.31.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (237) hide show
  1. unicode_logic_kit/__init__.py +385 -0
  2. unicode_logic_kit/__main__.py +520 -0
  3. unicode_logic_kit/_deadline.py +219 -0
  4. unicode_logic_kit/ace/__init__.py +126 -0
  5. unicode_logic_kit/ace/_align.py +135 -0
  6. unicode_logic_kit/ace/chem_lexicon.py +128 -0
  7. unicode_logic_kit/ace/drs_reader.py +570 -0
  8. unicode_logic_kit/ace/mapping.py +666 -0
  9. unicode_logic_kit/ace/reverse_modal.py +138 -0
  10. unicode_logic_kit/ace/runner.py +551 -0
  11. unicode_logic_kit/ace/translate.py +452 -0
  12. unicode_logic_kit/ace/verbalize.py +1070 -0
  13. unicode_logic_kit/api.py +1284 -0
  14. unicode_logic_kit/atp/__init__.py +177 -0
  15. unicode_logic_kit/atp/_ascii_names.py +113 -0
  16. unicode_logic_kit/atp/_html.py +72 -0
  17. unicode_logic_kit/atp/_substructural_input.py +228 -0
  18. unicode_logic_kit/atp/_tff_problem.py +715 -0
  19. unicode_logic_kit/atp/_tptp_problem.py +1111 -0
  20. unicode_logic_kit/atp/_writer_support.py +289 -0
  21. unicode_logic_kit/atp/clingo_backend.py +1180 -0
  22. unicode_logic_kit/atp/cvc5_backend.py +1385 -0
  23. unicode_logic_kit/atp/eprover_backend.py +732 -0
  24. unicode_logic_kit/atp/finite_domain.py +1055 -0
  25. unicode_logic_kit/atp/fitch.py +1547 -0
  26. unicode_logic_kit/atp/fitch_search.py +551 -0
  27. unicode_logic_kit/atp/hets_backend.py +339 -0
  28. unicode_logic_kit/atp/hybrid_down.py +120 -0
  29. unicode_logic_kit/atp/incremental.py +250 -0
  30. unicode_logic_kit/atp/kripke_enum.py +741 -0
  31. unicode_logic_kit/atp/lambek.py +436 -0
  32. unicode_logic_kit/atp/leo3_backend.py +332 -0
  33. unicode_logic_kit/atp/linear.py +738 -0
  34. unicode_logic_kit/atp/lj.py +705 -0
  35. unicode_logic_kit/atp/logic_backends.py +566 -0
  36. unicode_logic_kit/atp/ltl_tableau.py +1084 -0
  37. unicode_logic_kit/atp/minizinc_backend.py +1402 -0
  38. unicode_logic_kit/atp/modal_tableau.py +1382 -0
  39. unicode_logic_kit/atp/nanocop_backend.py +410 -0
  40. unicode_logic_kit/atp/portfolio.py +489 -0
  41. unicode_logic_kit/atp/protocol.py +1803 -0
  42. unicode_logic_kit/atp/prover9_entailment.py +1153 -0
  43. unicode_logic_kit/atp/resolution.py +1376 -0
  44. unicode_logic_kit/atp/resolution_check.py +1114 -0
  45. unicode_logic_kit/atp/sequent.py +1050 -0
  46. unicode_logic_kit/atp/tableau.py +921 -0
  47. unicode_logic_kit/atp/tableau_check.py +543 -0
  48. unicode_logic_kit/atp/tptp_ncl.py +811 -0
  49. unicode_logic_kit/atp/tptp_tff.py +1546 -0
  50. unicode_logic_kit/atp/tstp.py +1333 -0
  51. unicode_logic_kit/atp/tstp_check.py +1096 -0
  52. unicode_logic_kit/atp/twee_backend.py +236 -0
  53. unicode_logic_kit/atp/twee_check.py +711 -0
  54. unicode_logic_kit/atp/twee_entailment.py +953 -0
  55. unicode_logic_kit/atp/vampire_entailment.py +540 -0
  56. unicode_logic_kit/atp/z3_arith.py +470 -0
  57. unicode_logic_kit/atp/z3_equivalence.py +36 -0
  58. unicode_logic_kit/atp/z3_fuzzy.py +362 -0
  59. unicode_logic_kit/atp/z3_input.py +500 -0
  60. unicode_logic_kit/atp/z3_models.py +208 -0
  61. unicode_logic_kit/chem/__init__.py +88 -0
  62. unicode_logic_kit/chem/_naming.py +284 -0
  63. unicode_logic_kit/chem/cache.py +185 -0
  64. unicode_logic_kit/chem/interop.py +244 -0
  65. unicode_logic_kit/chem/mol.py +525 -0
  66. unicode_logic_kit/chem/signature.py +112 -0
  67. unicode_logic_kit/comorphism.py +497 -0
  68. unicode_logic_kit/dl/__init__.py +384 -0
  69. unicode_logic_kit/dl/classification.py +227 -0
  70. unicode_logic_kit/dl/concepts.py +632 -0
  71. unicode_logic_kit/dl/datatypes.py +818 -0
  72. unicode_logic_kit/dl/owl_functional.py +2433 -0
  73. unicode_logic_kit/dl/owl_manchester.py +1637 -0
  74. unicode_logic_kit/dl/owl_reasoner.py +790 -0
  75. unicode_logic_kit/dl/parser.py +391 -0
  76. unicode_logic_kit/dl/tableau.py +4048 -0
  77. unicode_logic_kit/dl/translate.py +2704 -0
  78. unicode_logic_kit/drt/__init__.py +94 -0
  79. unicode_logic_kit/drt/export.py +179 -0
  80. unicode_logic_kit/drt/nodes.py +506 -0
  81. unicode_logic_kit/drt/parser.py +965 -0
  82. unicode_logic_kit/drt/resolve.py +195 -0
  83. unicode_logic_kit/drt/reverse.py +175 -0
  84. unicode_logic_kit/eval/__init__.py +106 -0
  85. unicode_logic_kit/eval/batch.py +382 -0
  86. unicode_logic_kit/eval/canonical.py +663 -0
  87. unicode_logic_kit/eval/chem_batch.py +606 -0
  88. unicode_logic_kit/eval/converses.py +200 -0
  89. unicode_logic_kit/eval/datasets/__init__.py +136 -0
  90. unicode_logic_kit/eval/datasets/_base.py +263 -0
  91. unicode_logic_kit/eval/datasets/_proofwriter_proof.py +422 -0
  92. unicode_logic_kit/eval/datasets/c3po.py +678 -0
  93. unicode_logic_kit/eval/datasets/folio.py +158 -0
  94. unicode_logic_kit/eval/datasets/fracas.py +418 -0
  95. unicode_logic_kit/eval/datasets/groves.py +191 -0
  96. unicode_logic_kit/eval/datasets/logicbench.py +467 -0
  97. unicode_logic_kit/eval/datasets/logicnli.py +303 -0
  98. unicode_logic_kit/eval/datasets/malls.py +133 -0
  99. unicode_logic_kit/eval/datasets/pfolio.py +594 -0
  100. unicode_logic_kit/eval/datasets/pmb.py +242 -0
  101. unicode_logic_kit/eval/datasets/prontoqa.py +611 -0
  102. unicode_logic_kit/eval/datasets/proofwriter.py +1431 -0
  103. unicode_logic_kit/eval/datasets/proverqa.py +674 -0
  104. unicode_logic_kit/eval/datasets/willow.py +478 -0
  105. unicode_logic_kit/eval/equivalence.py +466 -0
  106. unicode_logic_kit/eval/exercise_gen.py +533 -0
  107. unicode_logic_kit/eval/explain.py +791 -0
  108. unicode_logic_kit/eval/generality.py +750 -0
  109. unicode_logic_kit/eval/metric_hf.py +458 -0
  110. unicode_logic_kit/eval/predicate_match.py +343 -0
  111. unicode_logic_kit/eval/theory_check.py +1170 -0
  112. unicode_logic_kit/eval/validate.py +306 -0
  113. unicode_logic_kit/fol/__init__.py +177 -0
  114. unicode_logic_kit/fol/_atom_keys.py +510 -0
  115. unicode_logic_kit/fol/_fol_nodes.py +3586 -0
  116. unicode_logic_kit/fol/_free_parameters.py +105 -0
  117. unicode_logic_kit/fol/_ho_nodes.py +448 -0
  118. unicode_logic_kit/fol/_hybrid_nodes.py +308 -0
  119. unicode_logic_kit/fol/_identifiers.py +1091 -0
  120. unicode_logic_kit/fol/_lambek_nodes.py +112 -0
  121. unicode_logic_kit/fol/_linear_nodes.py +352 -0
  122. unicode_logic_kit/fol/_modal_nodes.py +1467 -0
  123. unicode_logic_kit/fol/_msfl_nodes.py +2196 -0
  124. unicode_logic_kit/fol/_numeral_symbols.py +231 -0
  125. unicode_logic_kit/fol/_so_nodes.py +200 -0
  126. unicode_logic_kit/fol/_symbol_names.py +81 -0
  127. unicode_logic_kit/fol/_team_nodes.py +181 -0
  128. unicode_logic_kit/fol/_tptp_symbols.py +551 -0
  129. unicode_logic_kit/fol/_truth_constants.py +117 -0
  130. unicode_logic_kit/fol/casl_export.py +1135 -0
  131. unicode_logic_kit/fol/casl_import.py +929 -0
  132. unicode_logic_kit/fol/derivation.py +367 -0
  133. unicode_logic_kit/fol/dialect_detect.py +70 -0
  134. unicode_logic_kit/fol/dialect_repair.py +537 -0
  135. unicode_logic_kit/fol/frames.py +637 -0
  136. unicode_logic_kit/fol/grammars/terminals.lark +31 -0
  137. unicode_logic_kit/fol/lambda_tools.py +297 -0
  138. unicode_logic_kit/fol/latex_input.py +429 -0
  139. unicode_logic_kit/fol/modal_translation.py +944 -0
  140. unicode_logic_kit/fol/msflparser.py +1033 -0
  141. unicode_logic_kit/fol/naming.py +422 -0
  142. unicode_logic_kit/fol/nodes.py +241 -0
  143. unicode_logic_kit/fol/normalforms.py +492 -0
  144. unicode_logic_kit/fol/pal.py +287 -0
  145. unicode_logic_kit/fol/prolog_export.py +566 -0
  146. unicode_logic_kit/fol/prolog_input.py +505 -0
  147. unicode_logic_kit/fol/prover9_input.py +1325 -0
  148. unicode_logic_kit/fol/qml.py +1760 -0
  149. unicode_logic_kit/fol/qmltp_input.py +525 -0
  150. unicode_logic_kit/fol/sanitize.py +221 -0
  151. unicode_logic_kit/fol/serialize.py +79 -0
  152. unicode_logic_kit/fol/signature.py +1290 -0
  153. unicode_logic_kit/fol/simplify_check.py +544 -0
  154. unicode_logic_kit/fol/spans.py +594 -0
  155. unicode_logic_kit/fol/tptp_input.py +1503 -0
  156. unicode_logic_kit/fol/tptp_repair.py +941 -0
  157. unicode_logic_kit/fol/unification.py +157 -0
  158. unicode_logic_kit/fol/verbalize.py +263 -0
  159. unicode_logic_kit/hets/__init__.py +163 -0
  160. unicode_logic_kit/hets/bridge.py +142 -0
  161. unicode_logic_kit/hets/client.py +748 -0
  162. unicode_logic_kit/hets/docker.py +420 -0
  163. unicode_logic_kit/hets/dol.py +712 -0
  164. unicode_logic_kit/hets/haskell_json.py +355 -0
  165. unicode_logic_kit/hets/owl_backend.py +794 -0
  166. unicode_logic_kit/hets/owl_cli.py +598 -0
  167. unicode_logic_kit/hets/symbols.py +512 -0
  168. unicode_logic_kit/hol/__init__.py +140 -0
  169. unicode_logic_kit/hol/_ho_common.py +323 -0
  170. unicode_logic_kit/hol/_isabelle_binders.py +125 -0
  171. unicode_logic_kit/hol/classical.py +812 -0
  172. unicode_logic_kit/hol/deepshallow/__init__.py +45 -0
  173. unicode_logic_kit/hol/deepshallow/_common.py +177 -0
  174. unicode_logic_kit/hol/deepshallow/conditional.py +225 -0
  175. unicode_logic_kit/hol/deepshallow/intuitionistic.py +181 -0
  176. unicode_logic_kit/hol/deepshallow/modal.py +217 -0
  177. unicode_logic_kit/hol/deepshallow/qml.py +406 -0
  178. unicode_logic_kit/hol/deepshallow/relevant.py +206 -0
  179. unicode_logic_kit/hol/free.py +753 -0
  180. unicode_logic_kit/hol/goedel.py +336 -0
  181. unicode_logic_kit/hol/ho_modal.py +1743 -0
  182. unicode_logic_kit/hol/intuitionistic.py +403 -0
  183. unicode_logic_kit/hol/isabelle_conditional.py +593 -0
  184. unicode_logic_kit/hol/isabelle_modal.py +1908 -0
  185. unicode_logic_kit/hol/isabelle_relevant.py +412 -0
  186. unicode_logic_kit/hol/isabelle_runner.py +1147 -0
  187. unicode_logic_kit/hol/isabelle_substructural.py +884 -0
  188. unicode_logic_kit/hol/lean.py +1018 -0
  189. unicode_logic_kit/hol/manyvalued.py +921 -0
  190. unicode_logic_kit/hol/secondorder.py +687 -0
  191. unicode_logic_kit/hol/thf_modal.py +941 -0
  192. unicode_logic_kit/hol/thirdorder.py +397 -0
  193. unicode_logic_kit/ilp/__init__.py +89 -0
  194. unicode_logic_kit/ilp/readback.py +389 -0
  195. unicode_logic_kit/ilp/separation.py +153 -0
  196. unicode_logic_kit/ilp/task.py +730 -0
  197. unicode_logic_kit/logic.py +163 -0
  198. unicode_logic_kit/mcp/__init__.py +28 -0
  199. unicode_logic_kit/mcp/__main__.py +5 -0
  200. unicode_logic_kit/mcp/chem_tools.py +1031 -0
  201. unicode_logic_kit/mcp/server.py +2453 -0
  202. unicode_logic_kit/mcp/syntax_spec.py +681 -0
  203. unicode_logic_kit/prob/__init__.py +53 -0
  204. unicode_logic_kit/prob/_bdd.py +225 -0
  205. unicode_logic_kit/prob/_column_gen.py +668 -0
  206. unicode_logic_kit/prob/distribution.py +686 -0
  207. unicode_logic_kit/prob/nilsson.py +470 -0
  208. unicode_logic_kit/py.typed +0 -0
  209. unicode_logic_kit/semantics/__init__.py +137 -0
  210. unicode_logic_kit/semantics/_modal_reject.py +156 -0
  211. unicode_logic_kit/semantics/action_models.py +466 -0
  212. unicode_logic_kit/semantics/asp_models.py +1200 -0
  213. unicode_logic_kit/semantics/conditional.py +580 -0
  214. unicode_logic_kit/semantics/dynamic_epistemic.py +95 -0
  215. unicode_logic_kit/semantics/free_logic.py +913 -0
  216. unicode_logic_kit/semantics/fuzzy.py +384 -0
  217. unicode_logic_kit/semantics/fuzzy_kripke.py +442 -0
  218. unicode_logic_kit/semantics/intuitionistic.py +581 -0
  219. unicode_logic_kit/semantics/kripke.py +1139 -0
  220. unicode_logic_kit/semantics/manyvalued.py +580 -0
  221. unicode_logic_kit/semantics/matrix.py +342 -0
  222. unicode_logic_kit/semantics/model_eval.py +1135 -0
  223. unicode_logic_kit/semantics/modelfinder.py +1036 -0
  224. unicode_logic_kit/semantics/nonmonotonic.py +372 -0
  225. unicode_logic_kit/semantics/relevant.py +331 -0
  226. unicode_logic_kit/semantics/secondorder.py +657 -0
  227. unicode_logic_kit/semantics/structures.py +352 -0
  228. unicode_logic_kit/semantics/tarski.py +975 -0
  229. unicode_logic_kit/semantics/team.py +315 -0
  230. unicode_logic_kit/semantics/team_translation.py +416 -0
  231. unicode_logic_kit/semantics/thirdorder.py +358 -0
  232. unicode_logic_kit/semantics/tnorm.py +85 -0
  233. unicode_logic_kit/semantics/truthtable.py +201 -0
  234. unicode_logic_kit-0.31.0.dist-info/METADATA +333 -0
  235. unicode_logic_kit-0.31.0.dist-info/RECORD +237 -0
  236. unicode_logic_kit-0.31.0.dist-info/WHEEL +4 -0
  237. unicode_logic_kit-0.31.0.dist-info/licenses/LICENSE +21 -0
@@ -0,0 +1,1091 @@
1
+ """Runtime-generated identifier-terminal patterns for the FOL grammar.
2
+
3
+ WHY GENERATED, NOT FROZEN INTO A .lark FILE
4
+ ---------------------------------------------
5
+ The predicate/name/constant/variable terminals used to be one hand-written
6
+ ASCII regex each (``PREDICATE: /[A-Z][a-zA-Z0-9]*/`` and siblings), which is
7
+ exactly why they rejected a FOLIO gold formula's ``LostTo(x, świątek)`` or
8
+ ``Hosted(beijing, 2008SummerOlympics)`` outright: nothing outside `A-Za-z0-9`
9
+ was a letter or digit as far as the grammar was concerned. The fix is not a
10
+ bigger hand-written regex — pasting in, say, the Latin-1 Supplement and Latin
11
+ Extended-A block boundaries would still reject Cyrillic, CJK, Devanagari, and
12
+ everything else, and worse, would go stale the moment Unicode adds a new
13
+ script or a codepoint's General_Category changes, silently drifting from
14
+ whatever Python interpreter (3.10+) actually runs the kit. There is already a
15
+ correct, always-current answer to "is this codepoint a letter, and is it
16
+ uppercase" sitting in the interpreter: ``str.isalpha()`` / ``str.isupper()``.
17
+ So the character classes below are computed BY SCANNING those tables once,
18
+ in Python, at first use in a process (behind ``functools.lru_cache`` — the
19
+ scan costs well under a tenth of a second, which is a one-time-per-process
20
+ cost this module pays exactly once, not a per-``MSFLParser()`` or per-parse
21
+ cost), and handed to Lark as the ordinary compiled regex text it already
22
+ expects. A codepoint range is never typed into source; it is read off the
23
+ interpreter that will also be doing the matching.
24
+
25
+ WHY GREEK AND OHM ARE CARVED OUT OF EVERY LETTER CLASS
26
+ ---------------------------------------------------------
27
+ ``λ`` is the LAMBDA terminal, ``μ`` is the measure-term operator, and the
28
+ plain lowercase Greek letters (``αβγδεζηθικνξοπρστυφχψω``) are already the
29
+ CONSTANT terminal's second, non-``c_`` alternative — all three predate this
30
+ module and are matched as literal/fixed patterns elsewhere in the grammar.
31
+ A "just widen the letter classes to every Unicode letter" pass would swallow
32
+ all of that: ``λ`` and ``μ`` would stop being operators and start being
33
+ ordinary lowercase letters eligible to open a NAME or VARIABLE token, and the
34
+ Greek CONSTANT alternative would become redundant with (and shadowed by, or
35
+ racing against) a NAME/VARIABLE token now built from the exact same
36
+ characters. So Greek and Coptic (U+0370-U+03FF), Greek Extended
37
+ (U+1F00-U+1FFF, the block holding the accented/breathed forms), and U+2126
38
+ OHM SIGN (Unicode's own compatibility duplicate of Greek OMEGA, canonically
39
+ equivalent to U+03A9) are excluded from every letter this module recognises
40
+ — uppercase, lowercase, and the combining-mark continuation class alike.
41
+ Greek identifiers do not get a UPGRADE from this module; they keep exactly
42
+ the behaviour they had before it existed.
43
+
44
+ WHY THE FIRST CHARACTER DECIDES PREDICATE VS. TERM, AND WHAT THAT MEANS FOR
45
+ SCRIPTS WITHOUT CASE
46
+ -----------------------------------------------------------------------------
47
+ The kit's grammar has always used capitalisation to tell a predicate from a
48
+ term at the lexer level — no keyword, no sigil, just "the first letter is
49
+ uppercase" — so widening the alphabet has to widen that same signal rather
50
+ than replace it. Concretely: a token whose first character satisfies
51
+ ``str.isupper()`` classifies as PREDICATE; every other first LETTER
52
+ classifies as term-valued (NAME, CONSTANT, or VARIABLE, depending on length
53
+ and the ``c_``/Greek forms). Most scripts have no case distinction at all
54
+ (CJK, Arabic, Hebrew, Devanagari, Thai, ...) — for every one of them,
55
+ ``str.isupper()`` is simply always False, so an identifier in such a script
56
+ is always term-valued and can never head an atom by itself. That is not an
57
+ oversight this module works around; it is the literal reading of the
58
+ existing rule extended honestly to alphabets it was never tested against:
59
+ predicate-hood is signalled by a case distinction, and where a script draws
60
+ no such distinction, there is nothing to signal it with, so term position is
61
+ what a bare identifier in that script gets. (A caseless-script PREDICATE is
62
+ still reachable the way it always was: spell the atom with a Latin
63
+ uppercase-first name.)
64
+
65
+ WHY A DIGIT-LEADING IDENTIFIER IS ALWAYS TERM-VALUED, NEVER A PREDICATE
66
+ ---------------------------------------------------------------------------
67
+ ``2008SummerOlympics`` has to lex as SOMETHING for
68
+ ``Hosted(beijing, 2008SummerOlympics)`` to parse, but it cannot become a
69
+ second numeric terminal (that would make ``NUMBER`` and the new form
70
+ ambiguous over every plain integer) and it cannot become eligible for
71
+ PREDICATE position (nothing in the classical FOL literature, and nothing
72
+ elsewhere in this grammar, lets an atom's head start with a digit — and
73
+ doing so would need its own case-style signal for predicate-hood, which a
74
+ digit doesn't carry). So a digit-leading identifier is folded into NAME:
75
+ one-or-more ASCII digits, then a letter (of either case-class), then the
76
+ ordinary NAME continuation. It reuses NAME's own transform (a bare NAME
77
+ token becomes a ``Constant``; ``NAME "(" ... ")"`` becomes a ``Function``
78
+ head), so ``2008SummerOlympics`` used bare is a ``Constant`` and
79
+ ``2008SummerOlympics(x)`` would be a function call — never an atom. ASCII
80
+ digits only, deliberately: NUMBER is untouched by this whole module (``2008``
81
+ still lexes as NUMBER, ``2.5`` still lexes as NUMBER), so the digit run here
82
+ is exactly the digits NUMBER itself would have recognised — the difference
83
+ is solely the mandatory trailing letter that pulls the token out of NUMBER's
84
+ territory and into NAME's.
85
+
86
+ WHY CONSTANT KEEPS ITS PRIORITY, AND WHY THAT PRIORITY NOW MATTERS MORE
87
+ ---------------------------------------------------------------------------
88
+ Before this module existed, CONSTANT (``c_...`` or a Greek run) and NAME
89
+ (``[a-z]...``) could never both match the same span: NAME never contained an
90
+ underscore, so ``c_alpha`` was CONSTANT-only ground. Widening NAME to accept
91
+ underscores as a continuation character (needed for ``dani_Shapiro`` and
92
+ ``family_History``) removes that separation — ``c_alpha`` now also matches
93
+ NAME's alpha-leading form in full (``c`` is a lowercase letter, ``_alpha``
94
+ is legal NAME continuation with one more letter further on). Lark resolves
95
+ same-span multi-terminal ambiguity by priority, and CONSTANT was already
96
+ declared at priority 3 against NAME's 2 (``CONSTANT.3`` / ``NAME.2`` in the
97
+ grammar, both unchanged by this module), so ``c_alpha`` still lexes as
98
+ CONSTANT. The node is ``Constant("c_alpha")`` on either path: the ``const_``
99
+ transform keeps the mark as part of the name. No BARE text reads as a
100
+ ``Constant`` whose name is a variable token (``k2`` is a variable); such a
101
+ constant is written in quotes, ``'k2'`` (see the QUOTED_NAME section below).
102
+ The two terminals now genuinely overlap where they never used to, so that
103
+ priority ordering has gone from "never exercised" to "load-bearing", and is
104
+ exercised by an explicit regression test (see
105
+ ``tests/test_identifier_widening.py``) rather than left to be an accident of
106
+ how NAME happened to be spelled.
107
+
108
+ The lexer takes the FIRST terminal that matches, by priority, not the longest
109
+ one, so CONSTANT has to match whole words only. Its ``c_`` form ends in a
110
+ negative lookahead for a character that continues a NAME (a letter, digit,
111
+ underscore or combining mark): ``c_new_york`` is then declined by CONSTANT and
112
+ read whole as a NAME, where before CONSTANT cut it at ``c_new`` and the
113
+ remaining ``_york`` could not be read (nine dialects refused the word; only the
114
+ modal dialect's Earley fallback, which weighs every terminal, read it). The
115
+ lookahead names the whole continuation class and not just the underscore,
116
+ because the engine would otherwise back off to ``c_ne`` and match that.
117
+
118
+ WHY THE GENERATED PATTERNS ARE SMALL: LOOKAHEAD, NOT A SECOND EXPLICIT LIST
119
+ -----------------------------------------------------------------------------
120
+ The five terminal patterns below used to be built by literally splicing the
121
+ explicit ``upper``/``lower``/``combining`` class bodies (see ``_classes()``)
122
+ into each terminal's own pattern text — several times each, since a
123
+ terminal's continuation class ("more letters/digits/underscores/marks") had
124
+ to be written out in full at every position it appeared. NAME alone spelled
125
+ that continuation class out three times inside its own pattern (once for
126
+ each of the two ``[cont]*`` runs in its alpha-leading form, once more in its
127
+ digit-leading form) plus the combined upper+lower class a fourth time for
128
+ its "one more letter" requirement; at over 100KB, that ONE terminal was the
129
+ bulk of a combined ~190KB ``terminal_block(include_sort=True)``, and
130
+ ``re``/Lark compiling text that size is where the measured slowdown (a bare
131
+ ``MSFLParser()`` going from single-digit milliseconds to ~200ms once the
132
+ identifier terminals widened) actually goes — NOT the codepoint scan in
133
+ ``_classes()``, which stays well under a tenth of a second and was already
134
+ the one thing this module paid exactly once per process, cached, before and
135
+ after this change; and NOT parse throughput, which is unaffected either way
136
+ (the regex engine still walks the input once per character regardless of
137
+ how its pattern text is spelled).
138
+
139
+ The fix is not a tighter enumeration — ``lower`` is already a minimal
140
+ run-length encoding of a genuinely scattered set (every alphabetic script's
141
+ lowercase block is its own disjoint ``\\uXXXX-\\uYYYY`` span; there are a
142
+ lot of scripts) and cannot get meaningfully smaller as a literal list. The
143
+ fix is to stop writing that list out over and over, using something
144
+ Python's ``re`` module already tests cheaply instead of an enumerated class:
145
+ ``\\w``. CPython defines ``\\w`` (in Unicode mode, the only mode this module
146
+ or Lark's parser ever runs in) as matching a codepoint exactly when
147
+ ``str.isalnum(c)`` is true or ``c == '_'``, and ``str.isalnum()`` is in turn
148
+ ``isalpha() or isdecimal() or isdigit() or isnumeric()``. So
149
+ ``[^\\W\\d_]`` — a word character, with decimal digits and underscore
150
+ carved back out via the leading ``\\d``/``_`` in the negated class — is
151
+ ``isalpha()`` PLUS one small extra sliver: codepoints that are
152
+ ``isdigit()`` or ``isnumeric()`` without being ``isdecimal()`` (already
153
+ excluded by ``\\d``) or ``isalpha()`` — superscript/subscript digits
154
+ (``²``), Roman numerals (``Ⅻ``), vulgar fractions (``½``), circled/
155
+ parenthesized number forms, and similar. That sliver (below, DELTA) is what
156
+ has to be excluded for ``[^\\W\\d_]`` to mean exactly "is a letter"; scanned
157
+ at 0x0000-0x2FFFF it comes to 80 disjoint spans, a little over a kilobyte of
158
+ pattern text — two orders of magnitude smaller than the 12KB-20KB classes it
159
+ stands in for, because it only has to list the exceptions \\w tacks on
160
+ beside "is a letter", not the tens of thousands of letters themselves.
161
+
162
+ So LETTER — "any letter this module recognises", i.e. the exact union of
163
+ :func:`uppercase_class` and :func:`lowercase_class` a continuation position
164
+ used to get by splicing both bodies in directly — becomes
165
+ ``(?:[UPPER_NONALPHA]|(?!GREEK)(?!DELTA)(?!CEILING)[^\\W\\d_])``: small,
166
+ FIXED-size pieces, used however many times a terminal's shape needs a
167
+ letter, instead of the same multi-kilobyte enumeration copy-pasted that
168
+ many times. Two alternatives, not one, because ``\\w`` alone cannot stand
169
+ in for "is a letter, either case": ``upper`` (``str.isupper()``) contains a
170
+ 120-codepoint sliver — Roman numerals (U+2160-U+216F, category Nl) and
171
+ circled/squared Latin capitals (e.g. U+24B6, category So) — that carries
172
+ no case-STYLE distinction Python calls "alphabetic" at all
173
+ (``str.isalpha()`` is false for every one of them), so ``\\w``-based DELTA
174
+ exclusion or plain non-membership rules them out of the second alternative
175
+ regardless of any lookahead tweak; some are not even ``\\w`` members to
176
+ begin with (So-category symbols are not ``isalnum()``), so no negative
177
+ lookahead over ``\\w`` could ever admit them — a lookahead can only narrow
178
+ what a pattern already matches, never widen it. UPPER_NONALPHA
179
+ (:data:`_Classes.upper_nonalpha`, ``upper`` intersected with "not
180
+ isalpha()") explicitly lists that 120-codepoint sliver instead (five
181
+ contiguous spans, closer to DELTA's size than to UPPER's). Every other
182
+ letter — every codepoint the union actually shares with plain
183
+ ``isalpha()`` — still goes through the cheap ``\\w``-based path, with a
184
+ CEILING lookahead alongside GREEK/DELTA so the lookahead atom, which has
185
+ no length limit of its own the way an enumerated class does, cannot admit
186
+ anything :func:`_class_body` did not itself scan up to
187
+ :data:`_MAX_CODEPOINT`. It is a regex ATOM, not a character-class body: it
188
+ opens with a group and lookahead assertions, so — unlike
189
+ :func:`uppercase_class`, :func:`lowercase_class`, and :func:`combining_class`
190
+ below, which still return plain class-body text — it cannot be spliced
191
+ inside a caller's own ``[...]``. Nothing outside this module needs to
192
+ (``dialect_repair.py`` is the only outside consumer of the raw class
193
+ bodies, and it always brackets them itself, so those three functions are
194
+ untouched, same computation and same output as before this section
195
+ existed). UPPER stays fully enumerated regardless: ``re`` has no built-in
196
+ uppercase test the way ``\\w`` doubles as a letter test, so there is nothing
197
+ to invert it out of. And the term-valued ("lowerish") letter — everywhere
198
+ ``lowercase_class()``'s ~12KB body used to be spliced in directly — becomes
199
+ LETTER with UPPER's already-computed text reused once more as a negative
200
+ lookahead in front of it, rather than a second, separately-enumerated
201
+ "letter minus upper" class (UPPER's exclusion also correctly screens out
202
+ LETTER's own UPPER_NONALPHA alternative, since that alternative is a
203
+ subset of UPPER).
204
+
205
+ Only the functions that build a whole terminal's regex (``predicate_pattern``
206
+ and its siblings, and ``terminal_block``) were rewritten on these smaller
207
+ atoms. What a caller of those gets back is a differently-spelled pattern for
208
+ the exact same set of strings, never a different one.
209
+
210
+ WHAT SECURES THE EQUIVALENCE
211
+ --------------------------------
212
+ DELTA and UPPER_NONALPHA are computed the same way UPPER/LOWER/COMBINING
213
+ already were — one more call each to :func:`_class_body`, scanning
214
+ 0x0000-0x2FFFF once more inside the same cached, once-per-process
215
+ :func:`_classes` — so neither can ever drift from whatever Unicode version
216
+ the running interpreter actually implements, the same guarantee the rest
217
+ of this module already gives; neither is typed in by hand and neither can
218
+ silently go stale across a Python upgrade. But that construction is an
219
+ assertion, not a proof of the claim above (that
220
+ ``(?:[UPPER_NONALPHA]|(?!GREEK)(?!DELTA)(?!CEILING)[^\\W\\d_])`` matches
221
+ precisely "is a letter (either case, including the caseless-but-cased
222
+ Roman-numeral/circled-symbol sliver), is not excluded, and is not beyond
223
+ ``_MAX_CODEPOINT``"), so ``tests/test_identifiers_equivalence.py`` checks
224
+ it EXHAUSTIVELY rather than on a sample: for every one of the 0x30000
225
+ codepoints from 0x0000 to 0x2FFFF, it confirms the new letter atom matches
226
+ exactly where ``str.isalpha() or str.isupper()`` holds and the codepoint is
227
+ not one of :data:`_EXCLUDED_RANGES`, and that the new term-valued
228
+ ("lowerish") atom matches exactly where
229
+ ``str.isalpha() and not str.isupper()`` holds under the same exclusion —
230
+ i.e. exactly the sets the old, fully-enumerated ``lower``/``upper`` classes
231
+ contained (in ``lower``'s case exactly; in ``upper``'s case, the LETTER
232
+ atom matches the SAME set ``upper`` does, just split across the two
233
+ alternatives above), codepoint for codepoint. A second test class checks
234
+ codepoints ABOVE ``_MAX_CODEPOINT`` are rejected, since the exhaustive scan
235
+ by construction cannot exercise the CEILING lookahead itself.
236
+
237
+ WHY A CONSTANT MAY BE QUOTED, AND WHEN ITS NAME IS BARE
238
+ -----------------------------------------------------------
239
+ The shape of a bare word decides what it is: ``k2`` is a variable, ``Alice``
240
+ (third-order dialect) a predicate term, ``1`` a number, ``G-910`` no term at
241
+ all. So a constant with such a name had no text, and ``Constant("k2")``
242
+ printed as ``k2``, which reads back as another node. QUOTED_NAME is the way to
243
+ write any constant: ``'k2'``, ``'Alice'``, ``'G-910'``, ``'John Doe'``. Between
244
+ the quotes stand one or more characters, each either an ordinary character
245
+ (anything but a quote, a backslash, a control character, U+0085, U+2028,
246
+ U+2029 or a surrogate) or one of the two escapes ``\\'`` and ``\\\\``. No other
247
+ escape exists, and the empty name has no quoted form. The terminal begins with
248
+ a character no other terminal can begin with, so it competes with none of them
249
+ and needs no priority. It is accepted exactly where a constant stands as a
250
+ term: bare in the unsorted dialects, and only as ``'k2':Mountain`` in the
251
+ sorted ones (which have no bare constant either). Names of functions,
252
+ predicates, variables, sorts, the subscript of a modal operator (``K_a``) and
253
+ nominals have no quoted form.
254
+
255
+ :func:`is_bare_constant` says when the bare text of a name reads back as that
256
+ very constant, and it follows the LEXER, not "some dialect happens to accept
257
+ it": the text has to be one whole CONSTANT token or one whole NAME token (with
258
+ CONSTANT's whole-word rule above, this is a full match of either pattern).
259
+ :func:`constant_text` is the text that reads back as ``Constant(name)``: the
260
+ bare name when it can stand bare, else the name in quotes, with ``'`` written
261
+ ``\\'`` and ``\\`` written ``\\\\``. A name that cannot be written at all (not a
262
+ string, empty, or holding a character the quoted form excludes) is refused by
263
+ name rather than printed as text that reads as something else.
264
+
265
+ WHAT THIS MODULE DOES NOT TOUCH
266
+ -----------------------------------
267
+ NUMBER (``[0-9]+(\\.[0-9]+)?``), FORALL, EXISTS, and LAMBDA stay exactly as
268
+ declared in ``fol/grammars/terminals.lark`` — none of them classify a
269
+ letter, so none of them need widening, and none of them are generated here.
270
+ ``fol/sanitize.py``, the layer that maps AST names down to strictly-ASCII
271
+ tokens for TPTP/Prover9/SMT-LIB/Isabelle/CASL export, is untouched on
272
+ purpose: it exists to make names safe for ASCII-only target formats, not to
273
+ make them parseable by this (now Unicode-wide) grammar, and widening it
274
+ would leak raw non-ASCII text into export formats that cannot represent it.
275
+
276
+ PUBLIC SURFACE
277
+ ------------------
278
+ :func:`uppercase_class`, :func:`lowercase_class`, :func:`combining_class`
279
+ return the three raw regex character-CLASS BODIES (no enclosing ``[``/``]``)
280
+ this module is built from, for reuse by anything that needs to recognise the
281
+ same alphabet outside the grammar (``fol/dialect_repair.py``'s legality
282
+ check, and tests) by splicing them inside its own ``[...]``. They are
283
+ unaffected by the LOOKAHEAD section above: same computation, same returned
284
+ text, before and after. :func:`predicate_pattern`, :func:`name_pattern`,
285
+ :func:`constant_pattern`, :func:`variable_pattern`, :func:`sort_pattern` and
286
+ :func:`quoted_name_pattern`
287
+ return the full terminal regex (again without a Lark terminal name or
288
+ priority — just the pattern text between the ``/.../``) for each widened
289
+ terminal, now built from the small internal lookahead atoms rather than
290
+ repeated copies of the class bodies. :func:`is_variable_name`,
291
+ :func:`is_bare_constant` and :func:`constant_text` are the questions the
292
+ terminal shapes answer about one name (is it a variable; does its bare text
293
+ read back as that constant; what text reads back as it) and are exported from
294
+ the package. :func:`terminal_block` renders the
295
+ complete, ready-to-splice Lark terminal declarations (name, priority, and
296
+ pattern together) that ``_fol_nodes.build_grammar`` inserts into every
297
+ mode's grammar text. :data:`HUMAN_READABLE_PATTERNS` maps each widened
298
+ terminal's name to a short English description of its shape, for
299
+ ``fol/naming.py`` to show in a NamingError instead of the generated regex
300
+ text.
301
+
302
+ The same module is the one place a MINTED name gets its shape, because the
303
+ shapes are the terminals' business: :func:`fresh_variables` (a batch of
304
+ ``letter`` + digits), :func:`fresh_variable_like` (one alpha-renamed binder,
305
+ keeping the old letter), :func:`fresh_like` (a lambda parameter, which keeps its
306
+ kind: variable, NAME or PREDICATE) and :func:`variable_names` (what a minted
307
+ name has to avoid). Anything that prints a formula the kit cannot read back has
308
+ minted a name some other way.
309
+ """
310
+
311
+ import dataclasses
312
+ import re
313
+ import unicodedata
314
+ from functools import lru_cache
315
+ from typing import Callable, NamedTuple, Optional
316
+
317
+ __all__ = [
318
+ "uppercase_class", "lowercase_class", "combining_class",
319
+ "predicate_pattern", "name_pattern", "constant_pattern",
320
+ "variable_pattern", "sort_pattern", "quoted_name_pattern",
321
+ "terminal_block", "HUMAN_READABLE_PATTERNS",
322
+ "is_variable_name", "is_bare_constant", "constant_text",
323
+ "fresh_variables", "fresh_variable_like", "fresh_like", "variable_names",
324
+ "symbol_names",
325
+ ]
326
+
327
+
328
+ # ---------------------------------------------------------------------------
329
+ # Fresh names the kit's own parser reads back
330
+ #
331
+ # Every generator that mints a bound name (alpha-renaming, witnesses for a
332
+ # counting quantifier, a relativisation guard, a frame axiom, ...) has to mint a
333
+ # name of the SHAPE the position needs, or the text it prints is text the kit
334
+ # cannot read back: ``∀y_0 R(y, y_0)`` was printed by capture-avoiding
335
+ # substitution and rejected by ``api.parse_any``. These helpers are the one place
336
+ # such names are made.
337
+ #
338
+ # The shapes, from the terminals above:
339
+ # * an object variable: ONE term-valued letter, then ASCII digits only
340
+ # (``y``, ``y0``, ``y12``) -- no underscore, no prefix;
341
+ # * a NAME (function symbol, lambda parameter): two or more letters, and
342
+ # underscores and digits are fine after the first (``foo_0``);
343
+ # * a PREDICATE (second-order variable, lambda parameter): an uppercase-led
344
+ # run that likewise takes ``_`` and digits (``P_0``).
345
+ # ---------------------------------------------------------------------------
346
+
347
+ #: ``[a-z][0-9]*`` -- the ASCII core of VARIABLE. Every string it accepts the full
348
+ #: VARIABLE terminal accepts too, and it needs no scan of the Unicode tables, so
349
+ #: the common case (``x``, ``y1``) never pays the one-off start-up cost of
350
+ #: :func:`variable_pattern`.
351
+ _ASCII_VARIABLE = re.compile(r"[a-z][0-9]*")
352
+
353
+
354
+ #: ASCII core of a BARE constant: the strings of ASCII characters that the CONSTANT
355
+ #: terminal (its ``c_`` form) or the NAME terminal accepts as one whole token. Written
356
+ #: out by hand so that the common case never pays the one-off scan of the Unicode
357
+ #: tables; ``tests/test_identifiers_equivalence.py`` pins it against the generated patterns.
358
+ #:
359
+ #: * ``[a-z][0-9_]*[a-zA-Z][a-zA-Z0-9_]*`` -- NAME led by a letter: the first letter
360
+ #: (lower case), then, up to the SECOND letter, only digits and underscores, then the
361
+ #: rest. That is "at least two letters" without a backtracking run.
362
+ #: * ``[0-9]+[a-zA-Z][a-zA-Z0-9_]*`` -- NAME led by digits.
363
+ #: * ``c_[a-zA-Z0-9]+`` -- the ``c_`` form of CONSTANT (its tail has no underscore).
364
+ _ASCII_BARE_CONSTANT = re.compile(
365
+ r"[a-z][0-9_]*[a-zA-Z][a-zA-Z0-9_]*"
366
+ r"|[0-9]+[a-zA-Z][a-zA-Z0-9_]*"
367
+ r"|c_[a-zA-Z0-9]+")
368
+
369
+ #: The characters a constant's name cannot hold in ANY spelling, bare or quoted: the
370
+ #: control characters (a line break would end the statement of the text that holds the
371
+ #: formula), DEL, NEL, the two Unicode line and paragraph separators, and surrogates
372
+ #: (which no text encoding can carry). The body of a character class, shared by the
373
+ #: QUOTED_NAME terminal and by the printer's refusal, so the two cannot drift apart.
374
+ _UNSPELLABLE_CLASS = r"\x00-\x1f\x7f\x85\u2028\u2029\ud800-\udfff"
375
+ _UNSPELLABLE = re.compile(f"[{_UNSPELLABLE_CLASS}]")
376
+
377
+ #: A backslash in front of a quote or of a backslash: the only two escapes a quoted
378
+ #: name has. ``_CHARACTER_TO_ESCAPE`` finds the characters that need one.
379
+ _ESCAPED_CHARACTER = re.compile(r"\\(['\\])")
380
+ _CHARACTER_TO_ESCAPE = re.compile(r"(['\\])")
381
+
382
+
383
+ @lru_cache(maxsize=None)
384
+ def _compiled(which: str):
385
+ """The compiled terminal pattern called ``which`` (cached per process)."""
386
+ builder = {"variable": variable_pattern, "name": name_pattern,
387
+ "predicate": predicate_pattern, "constant": constant_pattern}[which]
388
+ return re.compile(builder())
389
+
390
+
391
+ def is_variable_name(text) -> bool:
392
+ """Whether the VARIABLE terminal accepts ``text`` as one whole token.
393
+
394
+ One term-valued letter followed by ASCII digits and nothing else: ``x``,
395
+ ``y12``, ``é``, ``北``. So ``x`` as a term is always the variable, and a
396
+ constant of that name has to be written in quotes (``'x'``). ``text`` that is
397
+ not a string is no variable name.
398
+ """
399
+ if not isinstance(text, str):
400
+ return False
401
+ return bool(_ASCII_VARIABLE.fullmatch(text)) or bool(
402
+ _compiled("variable").fullmatch(text))
403
+
404
+
405
+ #: How many names the two decisions below remember. The printer asks about every constant
406
+ #: of every formula it writes, and the constants of one problem are few and recur, so the
407
+ #: answer is looked up and not worked out again. A name is a string, so it is its own key.
408
+ _MEMORY = 8192
409
+
410
+
411
+ @lru_cache(maxsize=_MEMORY)
412
+ def _is_bare_constant(name: str) -> bool:
413
+ """:func:`is_bare_constant` for a string (no type check)."""
414
+ if name.isascii():
415
+ return _ASCII_BARE_CONSTANT.fullmatch(name) is not None
416
+ return (_compiled("constant").fullmatch(name) is not None
417
+ or _compiled("name").fullmatch(name) is not None)
418
+
419
+
420
+ def is_bare_constant(name) -> bool:
421
+ """Whether the bare text ``name`` reads back as ``Constant(name)`` as a term.
422
+
423
+ "Reads back" is what the LEXER does: it takes the first terminal that
424
+ matches, by priority, not the longest, so the text has to be ONE whole
425
+ CONSTANT token or ONE whole NAME token. CONSTANT matches whole words only (a
426
+ ``c_`` word that continues, ``c_new_york``, is not CONSTANT's but NAME's), so
427
+ this is a full match of either pattern. Hence ``socrates``, ``c_k2``,
428
+ ``c_new_york``, ``θ``, ``2008SummerOlympics`` and ``świątek`` are bare, and
429
+ ``a`` and ``k2`` (variables), ``Alice`` (a predicate), ``1`` and ``-3``
430
+ (numbers), ``G-910``, ``C++``, ``a b``, ``_sk0``, ``λ`` and ``x_1`` are not.
431
+
432
+ A name that is not a string, or is empty, is not bare.
433
+ """
434
+ return isinstance(name, str) and _is_bare_constant(name)
435
+
436
+
437
+ def constant_text(name) -> str:
438
+ """The text that reads back as ``Constant(name)`` in term position.
439
+
440
+ The bare ``name`` when :func:`is_bare_constant` holds, else the name in single
441
+ quotes, with a quote written ``\\'`` and a backslash written ``\\\\``:
442
+ ``socrates`` stays ``socrates``, ``k2`` becomes ``'k2'``, ``it's`` becomes
443
+ ``'it\\'s'``. A sorted constant takes the same text of its name
444
+ (``socrates:Human``, ``'k2':Mountain``).
445
+
446
+ Raises:
447
+ TypeError: ``name`` is not a string.
448
+ ValueError: ``name`` is empty, or holds a control character (U+0000 to
449
+ U+001F, U+007F), U+0085, U+2028, U+2029 or a surrogate, which no
450
+ spelling can carry.
451
+ """
452
+ if not isinstance(name, str):
453
+ raise TypeError(
454
+ f"constant_text: the name of a constant must be a string, got "
455
+ f"{name!r} (a {type(name).__name__}). A constant without a string name "
456
+ f"has no text; give it one.")
457
+ return _constant_text(name)
458
+
459
+
460
+ @lru_cache(maxsize=_MEMORY)
461
+ def _constant_text(name: str) -> str:
462
+ """:func:`constant_text` for a string (no type check). An exception is not remembered."""
463
+ if _is_bare_constant(name):
464
+ return name
465
+ if not name:
466
+ raise ValueError(
467
+ f"constant_text: the constant {name!r} has an empty name, so it has no text: "
468
+ f"no bare word is empty, and two quotes in a row are no constant. Give it a "
469
+ f"name of at least one character.")
470
+ excluded = _UNSPELLABLE.search(name)
471
+ if excluded is not None:
472
+ raise ValueError(
473
+ f"constant_text: the constant named {name!r} has no text: it holds "
474
+ f"{excluded.group()!r} (U+{ord(excluded.group()):04X}), a character that "
475
+ f"neither the bare nor the quoted form can carry (control characters, "
476
+ f"U+007F, U+0085, U+2028, U+2029 and surrogates are excluded). Rename "
477
+ f"the constant, or drop that character from its name.")
478
+ return "'" + _CHARACTER_TO_ESCAPE.sub(r"\\\1", name) + "'"
479
+
480
+
481
+ def _unquote_constant(token_text: str) -> str:
482
+ """The name that a QUOTED_NAME token spells: the inverse of :func:`constant_text`.
483
+
484
+ ``token_text`` is the whole token, quotes included. Inside, a backslash always
485
+ stands in front of a quote or a backslash (the terminal admits no other escape),
486
+ so one left-to-right pass undoes the escapes.
487
+ """
488
+ return _ESCAPED_CHARACTER.sub(r"\1", token_text[1:-1])
489
+
490
+
491
+ def variable_names(*nodes) -> frozenset:
492
+ """Every variable name occurring in ``nodes``, bound or free.
493
+
494
+ What a minted name has to avoid: see :func:`fresh_variables`. Three kinds of
495
+ occurrence count, because a fresh name that equals any of them changes what
496
+ the formula says:
497
+
498
+ * an object ``Variable`` and every quantifier / counting binder;
499
+ * a lambda-bound ``LambdaVar`` (the text ``λy. … y …`` reads an object
500
+ variable ``y`` under it as the lambda's own parameter, so the two kinds
501
+ cannot share a name);
502
+ * a name in an IF-logic slash set (``∃y/{z} …``): it is a plain string, not
503
+ a node, yet it refers to an enclosing variable.
504
+ """
505
+ names = set()
506
+ for node in nodes:
507
+ for inner in node.walk():
508
+ name = getattr(inner, "name", None)
509
+ if name is not None and type(inner).__name__ in ("Variable", "LambdaVar"):
510
+ names.add(name)
511
+ # a quantifier's own binder is a Variable in .variable
512
+ binder = getattr(inner, "variable", None)
513
+ if binder is not None and getattr(binder, "name", None):
514
+ names.add(binder.name)
515
+ names.update(getattr(inner, "slashed", ()) or ())
516
+ return frozenset(names)
517
+
518
+
519
+ def symbol_names(*nodes, fold: Optional[Callable[[str], str]] = None) -> frozenset:
520
+ """Every name that ``nodes`` carry, of every kind: what a MINTED name has to avoid.
521
+
522
+ :func:`variable_names` is the avoid set for a new bound variable inside one
523
+ formula of this kit, where a variable can only meet another variable. A name
524
+ that is minted for a TARGET -- a Skolem constant, a tableau parameter, a
525
+ tracking literal of a solver, a witness written into SMT-LIB or Prover9
526
+ text, a node of a description-logic tableau -- can meet any symbol there:
527
+ a constant ``_sk0``, a proposition ``goal``, a predicate ``x0``, a sort
528
+ ``x0``. So this returns every string a node of the formulas holds: the name
529
+ of a predicate, function, constant, sorted constant, variable, nominal or
530
+ agent, the sort of a sorted node, the names in a slash set. It is an
531
+ over-approximation on purpose (the glyph of a quantifier is a string a node
532
+ holds too): for an avoid set a name too many costs nothing, and a node class
533
+ added later is covered without being listed here.
534
+
535
+ Pass every formula of the problem -- the premises, the conclusion and the
536
+ background sentences that are added -- or the name is fresh for one formula
537
+ and taken in the next.
538
+
539
+ ``fold`` maps each name to the form the target compares names in:
540
+ ``str.casefold`` for a target that reads ``x0`` and ``X0`` as one word (TPTP
541
+ and Prover9 fold the case of a first letter), nothing for a target with
542
+ case-sensitive names. Mint in the same form: a candidate is fresh when
543
+ ``fold(candidate)`` is not in the result.
544
+ """
545
+ names = set()
546
+ for node in nodes:
547
+ for inner in node.walk():
548
+ values = (getattr(inner, field.name, None) for field in dataclasses.fields(inner)) \
549
+ if dataclasses.is_dataclass(inner) else vars(inner).values()
550
+ for value in values:
551
+ if isinstance(value, str):
552
+ names.add(value)
553
+ elif isinstance(value, (tuple, list, set, frozenset)):
554
+ names.update(item for item in value if isinstance(item, str))
555
+ return frozenset(fold(name) for name in names) if fold is not None else frozenset(names)
556
+
557
+
558
+ def fresh_variables(count: int, *, letter: str = "x", avoid=()) -> tuple:
559
+ """``count`` variable names that the VARIABLE terminal actually accepts.
560
+
561
+ A translation that mints its own bound variables has to mint names the
562
+ kit's own parser reads back, or its output is a formula the kit cannot
563
+ re-read: ``∃x_1 (…)``, ``∀_hw0 R(_hw0, _hw0)`` and
564
+ ``∃_msfol_Human_witness Human(…)`` were all printed by the kit and all
565
+ rejected by :func:`unicode_logic_kit.api.parse_any`, because VARIABLE is one
566
+ term-valued letter followed by ASCII DIGITS only — no underscore, no
567
+ prefix (see :func:`variable_pattern`).
568
+
569
+ So the shape here is ``letter`` + digits (``x0``, ``x1``, …), skipping
570
+ everything in ``avoid`` — pass :func:`variable_names` of whatever the
571
+ result will sit next to. Collisions are only a READABILITY matter for
572
+ bound variables (two quantifiers may bind the same name without either
573
+ capturing the other), but a collision with a FREE variable of the host
574
+ formula would change its meaning, which is what ``avoid`` is for.
575
+
576
+ Raises:
577
+ ValueError: ``letter`` is not a single character the terminal accepts.
578
+ """
579
+ if not is_variable_name(letter):
580
+ raise ValueError(
581
+ f"fresh_variables: {letter!r} is not a legal variable name on its "
582
+ f"own, so {letter!r} + digits is not one either")
583
+ avoid = set(avoid)
584
+ out: list = []
585
+ index = 0
586
+ while len(out) < count:
587
+ candidate = f"{letter}{index}"
588
+ index += 1
589
+ if candidate not in avoid:
590
+ out.append(candidate)
591
+ return tuple(out)
592
+
593
+
594
+ def _variable_letter(base: str) -> str:
595
+ """The letter an alpha-renamed variable called ``base`` keeps.
596
+
597
+ ``base``'s own first character when the VARIABLE terminal accepts it (so
598
+ ``y`` is renamed to ``y0``, not to an unrelated letter), else its lowercase
599
+ form (a Prolog-style ``Y`` becomes ``y0``), else ``x``. The last two cover a
600
+ binder whose name was not minted by this kit -- an imported or hand-built
601
+ ``Variable("_tmp")`` -- which still has to be renameable: renaming a binder to
602
+ a fresh LEGAL name is exactly alpha-equivalence, whatever it was called.
603
+ """
604
+ first = base[:1]
605
+ for candidate in (first, first.lower()):
606
+ if candidate and is_variable_name(candidate):
607
+ return candidate
608
+ return "x"
609
+
610
+
611
+ def fresh_variable_like(base: str, avoid=()) -> str:
612
+ """One fresh object-variable name for a binder that is called ``base``.
613
+
614
+ The alpha-renaming counterpart of :func:`fresh_variables`: ``letter`` is
615
+ taken from ``base`` (see ``_variable_letter``), the digits count up from 0,
616
+ and every name in ``avoid`` is skipped. The result is always a legal
617
+ VARIABLE, which is why it is the right name for any quantifier, counting or
618
+ cardinality binder -- whatever kind of name ``base`` was.
619
+
620
+ ``avoid`` must hold EVERY name that could be confused with the new binder:
621
+ the free variables of whatever gets substituted in, and every name -- bound
622
+ ones too -- inside the scope being renamed, because renaming ``y`` to a name
623
+ that an inner binder already uses captures the occurrences that moved.
624
+ :func:`variable_names` collects exactly that.
625
+ """
626
+ return fresh_variables(1, letter=_variable_letter(base), avoid=avoid)[0]
627
+
628
+
629
+ def fresh_like(base: str, avoid=()) -> str:
630
+ """A fresh name of the same terminal kind as ``base``, for a lambda parameter.
631
+
632
+ A lambda parameter may be a VARIABLE (``λx.``), a NAME (``λfoo.``) or a
633
+ PREDICATE (``λP.``), and the body uses it in the matching position -- as an
634
+ argument, or as the head of an application -- so renaming it must keep its
635
+ kind. A VARIABLE is renamed like any variable (``y`` → ``y0``); a NAME or a
636
+ PREDICATE takes a ``_N`` suffix (``foo`` → ``foo_0``, ``P`` → ``P_0``), which
637
+ is legal for both because underscore and digits are continuation characters
638
+ of either terminal. A ``base`` that is none of the three (it was not made by
639
+ the kit's parser) is renamed to a legal variable.
640
+ """
641
+ avoid = set(avoid)
642
+ if not is_variable_name(base) and (_compiled("name").fullmatch(base)
643
+ or _compiled("predicate").fullmatch(base)):
644
+ index = 0
645
+ while True:
646
+ candidate = f"{base}_{index}"
647
+ index += 1
648
+ if candidate not in avoid:
649
+ return candidate
650
+ return fresh_variable_like(base, avoid)
651
+
652
+ # Single-character operator glyphs that Unicode also classifies as uppercase
653
+ # letters. Each is a registered operator symbol (``_fol_nodes.OPERATORS``), and
654
+ # each satisfies ``str.isupper()`` — which is precisely the test this module
655
+ # uses to decide "this opens a PREDICATE". Before 0.23.2 that made ``ⓄP`` two
656
+ # things at once, ``Obligatory(P)`` and "the predicate named ⓄP", and only the
657
+ # Earley parser's willingness to pick one hid it: asked for every derivation,
658
+ # it reports the node as ambiguous, and a table-driven lexer takes the other
659
+ # reading. Same rule as Greek below, for the same reason — a glyph that is an
660
+ # operator ANYWHERE is not an identifier character ANYWHERE, even in a mode
661
+ # where that operator is not registered.
662
+ #
663
+ # Only SINGLE-character symbols belong here. The multi-character ones (``K_``,
664
+ # ``B_``, ``Say_``, ``Want_``) open with ordinary letters that obviously cannot
665
+ # be carved out, and do not need to be: their terminals are longer than the
666
+ # prefix they share with a name, so longest-match settles it — ``K_alice``
667
+ # lexes as one KNOWS token, with no ambiguous node. Both halves of that are
668
+ # checked in tests/test_operator_glyphs.py against the live registry, so a
669
+ # newly registered letter-like operator fails there rather than silently
670
+ # becoming a name.
671
+ _OPERATOR_GLYPHS = (
672
+ 0x24B8, # Ⓒ CIRCLED LATIN CAPITAL LETTER C — Contrast, every classical mode
673
+ 0x24BB, # Ⓕ CIRCLED LATIN CAPITAL LETTER F — Eventually (temporal)
674
+ 0x24BC, # Ⓖ CIRCLED LATIN CAPITAL LETTER G — Always (temporal)
675
+ 0x24C3, # Ⓝ CIRCLED LATIN CAPITAL LETTER N — Next (temporal)
676
+ 0x24C4, # Ⓞ CIRCLED LATIN CAPITAL LETTER O — Obligatory (deontic)
677
+ 0x24C5, # Ⓟ CIRCLED LATIN CAPITAL LETTER P — Permitted (deontic)
678
+ 0x24CA, # Ⓤ CIRCLED LATIN CAPITAL LETTER U — Until (temporal)
679
+ )
680
+
681
+ # Greek and Coptic, Greek Extended, and the OHM SIGN — see the module
682
+ # docstring's "WHY GREEK AND OHM ARE CARVED OUT" section — plus the operator
683
+ # glyphs above. Note what is NOT here: the other circled capitals (Ⓐ, Ⓑ, …)
684
+ # and the Roman numerals stay legal identifier characters, because they are
685
+ # not operators. The carve-out is a list of symbols the grammar already
686
+ # spends, not a swipe at a Unicode block.
687
+ _EXCLUDED_RANGES = (
688
+ ((0x0370, 0x03FF), (0x1F00, 0x1FFF), (0x2126, 0x2126))
689
+ + tuple((cp, cp) for cp in _OPERATOR_GLYPHS)
690
+ )
691
+
692
+ # Plane 0 (BMP) through Plane 2 (CJK Extension B/C/... territory). Every
693
+ # script this kit's test corpora (FOLIO included) actually exercise lives
694
+ # below this, and the scan is a fixed one-time-per-process cost regardless
695
+ # of where the ceiling sits, so there is no pressure to trim it further.
696
+ _MAX_CODEPOINT = 0x2FFFF
697
+
698
+
699
+ def _excluded(codepoint: int) -> bool:
700
+ return any(lo <= codepoint <= hi for lo, hi in _EXCLUDED_RANGES)
701
+
702
+
703
+ def _escape(codepoint: int) -> str:
704
+ return f"\\u{codepoint:04x}" if codepoint <= 0xFFFF else f"\\U{codepoint:08x}"
705
+
706
+
707
+ def _class_body(predicate: Callable[[str], bool]) -> str:
708
+ """A regex character-class body (no enclosing ``[``/``]``) matching every
709
+ codepoint up to :data:`_MAX_CODEPOINT` for which ``predicate(chr(cp))``
710
+ holds and the codepoint is not one of :data:`_EXCLUDED_RANGES`, run-
711
+ length-encoded into ``\\uXXXX-\\uYYYY`` spans (astral codepoints use the
712
+ 8-digit ``\\U........`` form) so the result is a few kilobytes of regex
713
+ text rather than tens of thousands of single-character alternatives."""
714
+ spans = []
715
+ start = None
716
+ for codepoint in range(_MAX_CODEPOINT + 1):
717
+ keep = predicate(chr(codepoint)) and not _excluded(codepoint)
718
+ if keep and start is None:
719
+ start = codepoint
720
+ elif not keep and start is not None:
721
+ spans.append((start, codepoint - 1))
722
+ start = None
723
+ if start is not None:
724
+ spans.append((start, _MAX_CODEPOINT))
725
+ return "".join(
726
+ _escape(lo) if lo == hi else f"{_escape(lo)}-{_escape(hi)}"
727
+ for lo, hi in spans
728
+ )
729
+
730
+
731
+ class _Classes(NamedTuple):
732
+ upper: str
733
+ lower: str
734
+ combining: str
735
+ delta: str
736
+ upper_nonalpha: str
737
+
738
+
739
+ @lru_cache(maxsize=1)
740
+ def _classes() -> _Classes:
741
+ """Compute the four character classes this module is built from, once
742
+ per process. Deferred behind ``lru_cache`` rather than run at import
743
+ time: importing this module (or anything that imports it, which given
744
+ ``_fol_nodes.py``'s import means most of the package) must stay cheap
745
+ even for a caller who never actually builds a parser; the codepoint
746
+ scan is paid only when a terminal pattern is first asked for — in
747
+ practice, the first time a grammar for some mode is actually built —
748
+ and never again in that process.
749
+
750
+ ``upper`` — codepoints with ``str.isupper() == True``: the PREDICATE-
751
+ signalling class (see the module docstring's "WHY THE FIRST
752
+ CHARACTER DECIDES" section).
753
+ ``lower`` — every other alphabetic codepoint (``str.isalpha()`` true,
754
+ ``str.isupper()`` false): both ordinary lowercase letters and every
755
+ letter from a script with no case distinction at all, which is
756
+ exactly the "term-valued" class that section describes. Returned
757
+ as-is by :func:`lowercase_class` for outside callers; the terminal
758
+ patterns below no longer splice this in directly (see the module
759
+ docstring's LOOKAHEAD section) but it is still computed and
760
+ returned unchanged, since ``dialect_repair.py`` still needs it.
761
+ ``combining`` — codepoints in Unicode general categories Mn (nonspacing
762
+ mark) and Mc (spacing combining mark): NFD-decomposed accents such
763
+ as U+0301 COMBINING ACUTE ACCENT, needed so that ``świątek``
764
+ survives NFD normalisation (``ś`` → ``s`` + U+0301) and still lexes
765
+ as one identifier rather than two tokens plus a stray mark. Not
766
+ part of ``\\w`` (marks are not alphanumeric), so — unlike ``lower``
767
+ — there is no cheaper way to express this one; it stays a literal
768
+ enumerated class either way.
769
+ ``delta`` — codepoints ``\\w`` matches that ``[^\\W\\d_]`` alone would
770
+ wrongly admit as letters: ``isdigit()`` or ``isnumeric()`` without
771
+ being ``isdecimal()`` (already excluded by ``\\d``) or
772
+ ``isalpha()``. Not returned by any public function — nothing
773
+ outside this module needs the raw sliver, only the negative
774
+ lookahead built from it (see :func:`_letter_atom`).
775
+ ``upper_nonalpha`` — codepoints with ``str.isupper() == True`` but
776
+ ``str.isalpha() == False``: e.g. the Roman numeral block
777
+ U+2160-U+216F (category Nl, ``isupper()`` true, ``isalpha()``
778
+ false) and circled/squared Latin capitals such as U+24B6 (category
779
+ So). ``upper`` already contains these (it is built from
780
+ ``str.isupper()`` alone, with no ``isalpha()`` condition), so a
781
+ continuation position that used to splice ``upper`` in directly
782
+ (PREDICATE/SORT/NAME's own continuation, before this module's
783
+ lookahead rewrite) accepted them; a LETTER atom built purely from
784
+ ``\\w`` cannot reach them the same way — some are not even ``\\w``
785
+ members (So-category symbols are not ``isalnum()``), and the rest
786
+ (the Nl-category Roman numerals) are exactly what ``delta`` above
787
+ excludes. So they need a small explicit class of their own rather
788
+ than a lookahead tweak (see :func:`_letter_atom`); it is tiny (five
789
+ contiguous spans, 120 codepoints total) because it is exactly
790
+ ``upper`` minus ``lower``'s complement restricted to ``isupper()``
791
+ already-non-alphabetic codepoints, not a general enumeration.
792
+ """
793
+ upper = _class_body(str.isupper)
794
+ lower = _class_body(lambda ch: ch.isalpha() and not ch.isupper())
795
+ combining = _class_body(lambda ch: unicodedata.category(ch) in ("Mn", "Mc"))
796
+ delta = _class_body(
797
+ lambda ch: ch.isalnum() and not ch.isalpha() and not ch.isdecimal())
798
+ upper_nonalpha = _class_body(lambda ch: ch.isupper() and not ch.isalpha())
799
+ return _Classes(
800
+ upper=upper, lower=lower, combining=combining, delta=delta,
801
+ upper_nonalpha=upper_nonalpha)
802
+
803
+
804
+ def uppercase_class() -> str:
805
+ """Regex character-class body for the PREDICATE-signalling letters."""
806
+ return _classes().upper
807
+
808
+
809
+ def lowercase_class() -> str:
810
+ """Regex character-class body for every term-valued letter (ordinary
811
+ lowercase, plus every letter of a script without case)."""
812
+ return _classes().lower
813
+
814
+
815
+ def combining_class() -> str:
816
+ """Regex character-class body for combining marks (categories Mn/Mc)."""
817
+ return _classes().combining
818
+
819
+
820
+ # ---------------------------------------------------------------------------
821
+ # Small, fixed-size lookahead atoms — see the module docstring's "WHY THE
822
+ # GENERATED PATTERNS ARE SMALL" section for what these stand in for and why
823
+ # they are correct. Private: none of these is a character-class body (each
824
+ # opens with a lookahead assertion), so none of them can be spliced inside a
825
+ # caller's own ``[...]`` the way uppercase_class()/lowercase_class()/
826
+ # combining_class() can — they are used as bare regex atoms/alternatives
827
+ # only, exclusively by the pattern-builders below.
828
+ # ---------------------------------------------------------------------------
829
+
830
+ def _excluded_lookahead() -> str:
831
+ """``(?!...)`` excluding :data:`_EXCLUDED_RANGES` — built from that same
832
+ tuple (never a second, hand-typed copy of the ranges), so it cannot
833
+ silently drift from what ``_classes()`` already excludes from
834
+ upper/lower/combining/delta.
835
+
836
+ Named for what it does rather than for Greek: since 0.23.2 the tuple also
837
+ carries the single-character operator glyphs (Ⓞ, Ⓖ, Ⓒ, …)."""
838
+ body = "".join(
839
+ _escape(lo) if lo == hi else f"{_escape(lo)}-{_escape(hi)}"
840
+ for lo, hi in _EXCLUDED_RANGES
841
+ )
842
+ return f"(?![{body}])"
843
+
844
+
845
+ def _delta_lookahead() -> str:
846
+ """``(?!...)`` excluding :data:`_Classes.delta` — the ``\\w``-but-not-a-
847
+ letter sliver described in the module docstring's LOOKAHEAD section."""
848
+ return f"(?![{_classes().delta}])"
849
+
850
+
851
+ def _ceiling_lookahead() -> str:
852
+ """``(?!...)`` excluding every codepoint above :data:`_MAX_CODEPOINT`.
853
+
854
+ ``[^\\W\\d_]`` (Python's ``\\w`` minus digits/underscore) has no ceiling
855
+ of its own — ``\\w`` matches any codepoint with ``str.isalnum()`` true,
856
+ all the way to U+10FFFF, including scripts like CJK Unified Ideograph
857
+ Extension G (U+30000-U+3134A, category Lo) that :func:`_class_body`
858
+ never scans past ``_MAX_CODEPOINT`` to include. Without this lookahead
859
+ the LETTER atom would silently accept letters the fully-enumerated
860
+ ``upper``/``lower`` classes it stands in for never could, breaking the
861
+ "differently-spelled pattern for the exact same set of strings" promise
862
+ this module's docstring makes for the lookahead rewrite. Built from
863
+ ``_MAX_CODEPOINT`` itself (never a second, hand-typed boundary), so it
864
+ cannot drift out of sync with the ceiling every other class in this
865
+ module already respects."""
866
+ return f"(?![\\U{_MAX_CODEPOINT + 1:08x}-\\U0010ffff])"
867
+
868
+
869
+ @lru_cache(maxsize=1)
870
+ def _letter_atom() -> str:
871
+ """LETTER: matches exactly one codepoint for which ``str.isalpha()`` OR
872
+ ``str.isupper()`` holds (not ``isalpha()`` alone — see below), the
873
+ codepoint is not one of :data:`_EXCLUDED_RANGES`, and the codepoint is
874
+ at most :data:`_MAX_CODEPOINT` — i.e. exactly the union of
875
+ :func:`uppercase_class` and :func:`lowercase_class` (which is what "any
876
+ letter, either case" used to be built from directly, by literally
877
+ splicing both class bodies into a continuation position). Two
878
+ alternatives:
879
+
880
+ * ``[UPPER_NONALPHA]`` — the small explicit class of codepoints with
881
+ ``isupper()`` true but ``isalpha()`` false (Roman numerals, circled/
882
+ squared Latin capitals; see :func:`_classes`'s ``upper_nonalpha``
883
+ docstring for why these need spelling out rather than a lookahead
884
+ tweak: some are not ``\\w`` members at all, so no negative lookahead
885
+ over ``\\w`` could ever admit them).
886
+ * ``(?!GREEK)(?!DELTA)(?!CEILING)[^\\W\\d_]`` — every ``isalpha()``
887
+ codepoint (upper or lower alike), which is everything ``uppercase_class()``
888
+ and ``lowercase_class()`` contain that is not already covered by the
889
+ first alternative.
890
+
891
+ Cached: it is pure string formatting over already-cached data, but it
892
+ gets referenced several times per terminal pattern across nine grammar
893
+ modes, so there is no reason to re-format it every time."""
894
+ upper_nonalpha = _classes().upper_nonalpha
895
+ isalpha_branch = (
896
+ f"{_excluded_lookahead()}{_delta_lookahead()}{_ceiling_lookahead()}"
897
+ f"[^\\W\\d_]"
898
+ )
899
+ return f"(?:[{upper_nonalpha}]|{isalpha_branch})"
900
+
901
+
902
+ @lru_cache(maxsize=1)
903
+ def _lowerish_atom() -> str:
904
+ """The term-valued ("lowerish") letter: LETTER, with one more negative
905
+ lookahead excluding :func:`uppercase_class` stacked in front — i.e.
906
+ exactly what :func:`lowercase_class`'s body used to be spliced in for
907
+ directly, at every position a term-valued first character was needed.
908
+ Reuses UPPER's already-computed text as a lookahead rather than
909
+ re-deriving a second, separately-enumerated "letter minus upper"
910
+ class — see the module docstring's LOOKAHEAD section."""
911
+ return f"(?![{uppercase_class()}]){_letter_atom()}"
912
+
913
+
914
+ @lru_cache(maxsize=1)
915
+ def _continuation_atom(*, underscore: bool) -> str:
916
+ """The shared "letter, digit, or combining mark" continuation atom,
917
+ optionally with ``_`` added. Every identifier terminal now passes
918
+ ``underscore=True``; the flag survives because CONSTANT's ``c_`` form must
919
+ NOT (its own leading ``c_`` is the marker, and letting the tail carry more
920
+ underscores would widen the span it competes with NAME over — see the
921
+ module docstring's CONSTANT-priority section). Matches exactly one
922
+ character; callers append ``*``/``+`` themselves, the same way they would
923
+ to a character class — this is a non-capturing group standing in for one,
924
+ not a class body."""
925
+ digits = "0-9_" if underscore else "0-9"
926
+ return f"(?:{_letter_atom()}|[{digits}{combining_class()}])"
927
+
928
+
929
+ def predicate_pattern() -> str:
930
+ """PREDICATE: an uppercase-signalling letter, then letters/digits/
931
+ underscores/combining marks — the SAME continuation class the term-valued
932
+ terminals get.
933
+
934
+ 0.23.0 shipped this asymmetric, on the reasoning that keeping ``_`` out of
935
+ predicate position left an IRI-shaped name such as
936
+ ``Http___www_w3_org_owl_Thing`` as illegal a predicate token as it had
937
+ always been, which ``fol/sanitize.py`` relies on. That reasoning does not
938
+ hold: ``sanitize.py`` carries its OWN deliberately ASCII-strict
939
+ ``_PRED_RE`` (``[A-Z][a-zA-Z0-9]*``) and reaches its verdict without
940
+ consulting this module at all, so it renames that IRI either way.
941
+
942
+ What the asymmetry did break is ``chem/interop.py``. Its kit spelling of a
943
+ ChemLog predicate capitalises the FIRST character and nothing else, so 17
944
+ of the chemical signature's 40 predicates spell as ``Has_bond_to``,
945
+ ``In_ring_of_size_6``, ``Net_charge_neutral`` and the like — names the kit's
946
+ own parser then refused, leaving the chemical vocabulary impossible to
947
+ write down in the kit's own surface syntax. A generating model handed the
948
+ signature and told to use it produced ``Has_bond_to(c, x)``, was refused,
949
+ and fell back to ``HasBondTo`` — which parses but is in no signature, so
950
+ every molecule came back as an uninterpreted-symbol error rather than a
951
+ verdict.
952
+ """
953
+ cont = _continuation_atom(underscore=True)
954
+ return f"[{uppercase_class()}]{cont}*"
955
+
956
+
957
+ def name_pattern() -> str:
958
+ """NAME: two alternatives, both term-valued.
959
+
960
+ The alpha-leading form starts with a term-valued letter, then any
961
+ continuation characters, then one more explicit letter (upper- or
962
+ lower-class — a predicate-signalling letter is legal HERE, mid-token;
963
+ only the first character carries the predicate/term distinction), then
964
+ more continuation — the same "at least two letters" shape the original
965
+ ``[a-z][a-zA-Z0-9]*[a-zA-Z][a-zA-Z0-9]*`` had, just with underscores and
966
+ the wider alphabet spliced into every continuation run, so a single bare
967
+ letter still falls through to VARIABLE instead of NAME.
968
+
969
+ The digit-leading form is one-or-more ASCII digits, then a letter, then
970
+ the same continuation — see the module docstring's "WHY A DIGIT-LEADING
971
+ IDENTIFIER" section.
972
+ """
973
+ cont = _continuation_atom(underscore=True)
974
+ letter = _letter_atom()
975
+ alpha_led = f"{_lowerish_atom()}{cont}*{letter}{cont}*"
976
+ digit_led = f"[0-9]+{letter}{cont}*"
977
+ return f"(?:{alpha_led})|(?:{digit_led})"
978
+
979
+
980
+ def variable_pattern() -> str:
981
+ """VARIABLE: one term-valued letter, then only ASCII digits — unchanged
982
+ in shape from before this module, only the letter class is widened."""
983
+ return f"{_lowerish_atom()}[0-9]*"
984
+
985
+
986
+ def constant_pattern() -> str:
987
+ """CONSTANT: the ``c_`` form (accepting Unicode letters/digits/combining
988
+ marks after the literal ``c_``, e.g. ``c_świątek``) or the plain lowercase
989
+ Greek run — untouched, since Greek is excluded from every generated class
990
+ (see the module docstring).
991
+
992
+ The ``c_`` form matches WHOLE WORDS only: it may not be followed by a
993
+ character that continues a NAME (a letter, digit, underscore or combining
994
+ mark). The lexer takes the first terminal that matches, not the longest, and
995
+ CONSTANT has the higher priority, so without the lookahead ``c_new_york``
996
+ was cut at ``c_new`` and the rest (``_york``) could not be read. With it,
997
+ CONSTANT declines the word and NAME reads all of it. The lookahead has to
998
+ name the whole continuation class and not just the underscore: a bare
999
+ ``(?!_)`` lets the engine back off to ``c_ne`` and match that instead.
1000
+ """
1001
+ cont = _continuation_atom(underscore=False)
1002
+ word_end = f"(?!{_continuation_atom(underscore=True)})"
1003
+ c_form = f"c_{cont}+{word_end}"
1004
+ greek_form = "[αβγδεζηθικνξοπρστυφχψω]+"
1005
+ return f"(?:{c_form})|(?:{greek_form})"
1006
+
1007
+
1008
+ def sort_pattern() -> str:
1009
+ """SORT: a literal ``:`` then the same shape as PREDICATE — including the
1010
+ underscore, so a sorted signature can name a sort after the same vocabulary
1011
+ its predicates come from (``:In_ring``) instead of being the one identifier
1012
+ position that still cannot."""
1013
+ cont = _continuation_atom(underscore=True)
1014
+ return f":[{uppercase_class()}]{cont}*"
1015
+
1016
+
1017
+ def quoted_name_pattern() -> str:
1018
+ """QUOTED_NAME: a constant written in single quotes.
1019
+
1020
+ A quote, then one or more of: any character that is not a quote, a backslash
1021
+ or one of the characters no spelling can carry (control characters, DEL,
1022
+ U+0085, U+2028, U+2029, surrogates: see :data:`_UNSPELLABLE_CLASS`), or the
1023
+ escape ``\\'`` (a quote) or ``\\\\`` (a backslash); then the closing quote. No
1024
+ other escape exists and the empty ``''`` is not a name.
1025
+
1026
+ The class is written with escapes (``\\x00``, ``\\u2028``) and not with the
1027
+ characters themselves: a literal line break inside a Lark ``/.../`` terminal
1028
+ is a grammar error, and a bare U+2028 in the pattern text would be invisible.
1029
+ No other terminal begins with a quote, so this one has no priority to win
1030
+ and none to lose.
1031
+ """
1032
+ return f"'(?:[^'\\\\{_UNSPELLABLE_CLASS}]|\\\\['\\\\])+'"
1033
+
1034
+
1035
+ def terminal_block(*, include_sort: bool) -> str:
1036
+ """The complete Lark terminal declarations for PREDICATE, CONSTANT, NAME,
1037
+ VARIABLE, QUOTED_NAME, and — when ``include_sort`` is true (the many-sorted
1038
+ modes) — SORT, in the priorities the grammar has always used
1039
+ (``CONSTANT.3`` over ``NAME.2`` over ``VARIABLE.1``; PREDICATE, QUOTED_NAME
1040
+ and SORT are unambiguous with everything else so carry no explicit
1041
+ priority). ``include_sort`` is a
1042
+ parameter rather than always-on because SORT never appears in a
1043
+ classical/modal/second-order grammar's rules, and declaring an unused
1044
+ terminal there is needless generated text for no behavioural gain."""
1045
+ lines = [
1046
+ f"PREDICATE: /{predicate_pattern()}/",
1047
+ "",
1048
+ f"CONSTANT.3: /{constant_pattern()}/",
1049
+ "",
1050
+ f"NAME.2: /{name_pattern()}/",
1051
+ "",
1052
+ f"VARIABLE.1: /{variable_pattern()}/",
1053
+ "",
1054
+ f"QUOTED_NAME: /{quoted_name_pattern()}/",
1055
+ ]
1056
+ if include_sort:
1057
+ lines += ["", f"SORT: /{sort_pattern()}/"]
1058
+ return "\n".join(lines) + "\n"
1059
+
1060
+
1061
+ #: Terminal name -> short English description of its shape, for
1062
+ #: ``fol/naming.py``'s NamingError to show in place of the generated regex
1063
+ #: text (a few kilobytes of ``\uXXXX-\uYYYY`` ranges is not a message a
1064
+ #: human — or a model reading the error to retry — can act on).
1065
+ HUMAN_READABLE_PATTERNS = {
1066
+ "PREDICATE": (
1067
+ "an uppercase letter (in any script that has letter case) followed "
1068
+ "by letters, digits, or combining marks"
1069
+ ),
1070
+ "NAME": (
1071
+ "a lowercase or caseless letter, then letters/digits/underscores/"
1072
+ "combining marks with at least one more letter among them; or one "
1073
+ "or more digits followed by a letter and more of the same"
1074
+ ),
1075
+ "CONSTANT": (
1076
+ "'c_' followed by letters, digits, or combining marks, or one or "
1077
+ "more lowercase Greek letters (α-ω)"
1078
+ ),
1079
+ "QUOTED_NAME": (
1080
+ "a single quote, then the name, then a single quote; inside, a "
1081
+ "quote is written \\' and a backslash \\\\"
1082
+ ),
1083
+ "VARIABLE": (
1084
+ "a single lowercase or caseless letter, optionally followed by "
1085
+ "digits"
1086
+ ),
1087
+ "SORT": (
1088
+ "':' followed by an uppercase letter and then letters, digits, or "
1089
+ "combining marks"
1090
+ ),
1091
+ }