unicode-logic-kit 0.31.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (237) hide show
  1. unicode_logic_kit/__init__.py +385 -0
  2. unicode_logic_kit/__main__.py +520 -0
  3. unicode_logic_kit/_deadline.py +219 -0
  4. unicode_logic_kit/ace/__init__.py +126 -0
  5. unicode_logic_kit/ace/_align.py +135 -0
  6. unicode_logic_kit/ace/chem_lexicon.py +128 -0
  7. unicode_logic_kit/ace/drs_reader.py +570 -0
  8. unicode_logic_kit/ace/mapping.py +666 -0
  9. unicode_logic_kit/ace/reverse_modal.py +138 -0
  10. unicode_logic_kit/ace/runner.py +551 -0
  11. unicode_logic_kit/ace/translate.py +452 -0
  12. unicode_logic_kit/ace/verbalize.py +1070 -0
  13. unicode_logic_kit/api.py +1284 -0
  14. unicode_logic_kit/atp/__init__.py +177 -0
  15. unicode_logic_kit/atp/_ascii_names.py +113 -0
  16. unicode_logic_kit/atp/_html.py +72 -0
  17. unicode_logic_kit/atp/_substructural_input.py +228 -0
  18. unicode_logic_kit/atp/_tff_problem.py +715 -0
  19. unicode_logic_kit/atp/_tptp_problem.py +1111 -0
  20. unicode_logic_kit/atp/_writer_support.py +289 -0
  21. unicode_logic_kit/atp/clingo_backend.py +1180 -0
  22. unicode_logic_kit/atp/cvc5_backend.py +1385 -0
  23. unicode_logic_kit/atp/eprover_backend.py +732 -0
  24. unicode_logic_kit/atp/finite_domain.py +1055 -0
  25. unicode_logic_kit/atp/fitch.py +1547 -0
  26. unicode_logic_kit/atp/fitch_search.py +551 -0
  27. unicode_logic_kit/atp/hets_backend.py +339 -0
  28. unicode_logic_kit/atp/hybrid_down.py +120 -0
  29. unicode_logic_kit/atp/incremental.py +250 -0
  30. unicode_logic_kit/atp/kripke_enum.py +741 -0
  31. unicode_logic_kit/atp/lambek.py +436 -0
  32. unicode_logic_kit/atp/leo3_backend.py +332 -0
  33. unicode_logic_kit/atp/linear.py +738 -0
  34. unicode_logic_kit/atp/lj.py +705 -0
  35. unicode_logic_kit/atp/logic_backends.py +566 -0
  36. unicode_logic_kit/atp/ltl_tableau.py +1084 -0
  37. unicode_logic_kit/atp/minizinc_backend.py +1402 -0
  38. unicode_logic_kit/atp/modal_tableau.py +1382 -0
  39. unicode_logic_kit/atp/nanocop_backend.py +410 -0
  40. unicode_logic_kit/atp/portfolio.py +489 -0
  41. unicode_logic_kit/atp/protocol.py +1803 -0
  42. unicode_logic_kit/atp/prover9_entailment.py +1153 -0
  43. unicode_logic_kit/atp/resolution.py +1376 -0
  44. unicode_logic_kit/atp/resolution_check.py +1114 -0
  45. unicode_logic_kit/atp/sequent.py +1050 -0
  46. unicode_logic_kit/atp/tableau.py +921 -0
  47. unicode_logic_kit/atp/tableau_check.py +543 -0
  48. unicode_logic_kit/atp/tptp_ncl.py +811 -0
  49. unicode_logic_kit/atp/tptp_tff.py +1546 -0
  50. unicode_logic_kit/atp/tstp.py +1333 -0
  51. unicode_logic_kit/atp/tstp_check.py +1096 -0
  52. unicode_logic_kit/atp/twee_backend.py +236 -0
  53. unicode_logic_kit/atp/twee_check.py +711 -0
  54. unicode_logic_kit/atp/twee_entailment.py +953 -0
  55. unicode_logic_kit/atp/vampire_entailment.py +540 -0
  56. unicode_logic_kit/atp/z3_arith.py +470 -0
  57. unicode_logic_kit/atp/z3_equivalence.py +36 -0
  58. unicode_logic_kit/atp/z3_fuzzy.py +362 -0
  59. unicode_logic_kit/atp/z3_input.py +500 -0
  60. unicode_logic_kit/atp/z3_models.py +208 -0
  61. unicode_logic_kit/chem/__init__.py +88 -0
  62. unicode_logic_kit/chem/_naming.py +284 -0
  63. unicode_logic_kit/chem/cache.py +185 -0
  64. unicode_logic_kit/chem/interop.py +244 -0
  65. unicode_logic_kit/chem/mol.py +525 -0
  66. unicode_logic_kit/chem/signature.py +112 -0
  67. unicode_logic_kit/comorphism.py +497 -0
  68. unicode_logic_kit/dl/__init__.py +384 -0
  69. unicode_logic_kit/dl/classification.py +227 -0
  70. unicode_logic_kit/dl/concepts.py +632 -0
  71. unicode_logic_kit/dl/datatypes.py +818 -0
  72. unicode_logic_kit/dl/owl_functional.py +2433 -0
  73. unicode_logic_kit/dl/owl_manchester.py +1637 -0
  74. unicode_logic_kit/dl/owl_reasoner.py +790 -0
  75. unicode_logic_kit/dl/parser.py +391 -0
  76. unicode_logic_kit/dl/tableau.py +4048 -0
  77. unicode_logic_kit/dl/translate.py +2704 -0
  78. unicode_logic_kit/drt/__init__.py +94 -0
  79. unicode_logic_kit/drt/export.py +179 -0
  80. unicode_logic_kit/drt/nodes.py +506 -0
  81. unicode_logic_kit/drt/parser.py +965 -0
  82. unicode_logic_kit/drt/resolve.py +195 -0
  83. unicode_logic_kit/drt/reverse.py +175 -0
  84. unicode_logic_kit/eval/__init__.py +106 -0
  85. unicode_logic_kit/eval/batch.py +382 -0
  86. unicode_logic_kit/eval/canonical.py +663 -0
  87. unicode_logic_kit/eval/chem_batch.py +606 -0
  88. unicode_logic_kit/eval/converses.py +200 -0
  89. unicode_logic_kit/eval/datasets/__init__.py +136 -0
  90. unicode_logic_kit/eval/datasets/_base.py +263 -0
  91. unicode_logic_kit/eval/datasets/_proofwriter_proof.py +422 -0
  92. unicode_logic_kit/eval/datasets/c3po.py +678 -0
  93. unicode_logic_kit/eval/datasets/folio.py +158 -0
  94. unicode_logic_kit/eval/datasets/fracas.py +418 -0
  95. unicode_logic_kit/eval/datasets/groves.py +191 -0
  96. unicode_logic_kit/eval/datasets/logicbench.py +467 -0
  97. unicode_logic_kit/eval/datasets/logicnli.py +303 -0
  98. unicode_logic_kit/eval/datasets/malls.py +133 -0
  99. unicode_logic_kit/eval/datasets/pfolio.py +594 -0
  100. unicode_logic_kit/eval/datasets/pmb.py +242 -0
  101. unicode_logic_kit/eval/datasets/prontoqa.py +611 -0
  102. unicode_logic_kit/eval/datasets/proofwriter.py +1431 -0
  103. unicode_logic_kit/eval/datasets/proverqa.py +674 -0
  104. unicode_logic_kit/eval/datasets/willow.py +478 -0
  105. unicode_logic_kit/eval/equivalence.py +466 -0
  106. unicode_logic_kit/eval/exercise_gen.py +533 -0
  107. unicode_logic_kit/eval/explain.py +791 -0
  108. unicode_logic_kit/eval/generality.py +750 -0
  109. unicode_logic_kit/eval/metric_hf.py +458 -0
  110. unicode_logic_kit/eval/predicate_match.py +343 -0
  111. unicode_logic_kit/eval/theory_check.py +1170 -0
  112. unicode_logic_kit/eval/validate.py +306 -0
  113. unicode_logic_kit/fol/__init__.py +177 -0
  114. unicode_logic_kit/fol/_atom_keys.py +510 -0
  115. unicode_logic_kit/fol/_fol_nodes.py +3586 -0
  116. unicode_logic_kit/fol/_free_parameters.py +105 -0
  117. unicode_logic_kit/fol/_ho_nodes.py +448 -0
  118. unicode_logic_kit/fol/_hybrid_nodes.py +308 -0
  119. unicode_logic_kit/fol/_identifiers.py +1091 -0
  120. unicode_logic_kit/fol/_lambek_nodes.py +112 -0
  121. unicode_logic_kit/fol/_linear_nodes.py +352 -0
  122. unicode_logic_kit/fol/_modal_nodes.py +1467 -0
  123. unicode_logic_kit/fol/_msfl_nodes.py +2196 -0
  124. unicode_logic_kit/fol/_numeral_symbols.py +231 -0
  125. unicode_logic_kit/fol/_so_nodes.py +200 -0
  126. unicode_logic_kit/fol/_symbol_names.py +81 -0
  127. unicode_logic_kit/fol/_team_nodes.py +181 -0
  128. unicode_logic_kit/fol/_tptp_symbols.py +551 -0
  129. unicode_logic_kit/fol/_truth_constants.py +117 -0
  130. unicode_logic_kit/fol/casl_export.py +1135 -0
  131. unicode_logic_kit/fol/casl_import.py +929 -0
  132. unicode_logic_kit/fol/derivation.py +367 -0
  133. unicode_logic_kit/fol/dialect_detect.py +70 -0
  134. unicode_logic_kit/fol/dialect_repair.py +537 -0
  135. unicode_logic_kit/fol/frames.py +637 -0
  136. unicode_logic_kit/fol/grammars/terminals.lark +31 -0
  137. unicode_logic_kit/fol/lambda_tools.py +297 -0
  138. unicode_logic_kit/fol/latex_input.py +429 -0
  139. unicode_logic_kit/fol/modal_translation.py +944 -0
  140. unicode_logic_kit/fol/msflparser.py +1033 -0
  141. unicode_logic_kit/fol/naming.py +422 -0
  142. unicode_logic_kit/fol/nodes.py +241 -0
  143. unicode_logic_kit/fol/normalforms.py +492 -0
  144. unicode_logic_kit/fol/pal.py +287 -0
  145. unicode_logic_kit/fol/prolog_export.py +566 -0
  146. unicode_logic_kit/fol/prolog_input.py +505 -0
  147. unicode_logic_kit/fol/prover9_input.py +1325 -0
  148. unicode_logic_kit/fol/qml.py +1760 -0
  149. unicode_logic_kit/fol/qmltp_input.py +525 -0
  150. unicode_logic_kit/fol/sanitize.py +221 -0
  151. unicode_logic_kit/fol/serialize.py +79 -0
  152. unicode_logic_kit/fol/signature.py +1290 -0
  153. unicode_logic_kit/fol/simplify_check.py +544 -0
  154. unicode_logic_kit/fol/spans.py +594 -0
  155. unicode_logic_kit/fol/tptp_input.py +1503 -0
  156. unicode_logic_kit/fol/tptp_repair.py +941 -0
  157. unicode_logic_kit/fol/unification.py +157 -0
  158. unicode_logic_kit/fol/verbalize.py +263 -0
  159. unicode_logic_kit/hets/__init__.py +163 -0
  160. unicode_logic_kit/hets/bridge.py +142 -0
  161. unicode_logic_kit/hets/client.py +748 -0
  162. unicode_logic_kit/hets/docker.py +420 -0
  163. unicode_logic_kit/hets/dol.py +712 -0
  164. unicode_logic_kit/hets/haskell_json.py +355 -0
  165. unicode_logic_kit/hets/owl_backend.py +794 -0
  166. unicode_logic_kit/hets/owl_cli.py +598 -0
  167. unicode_logic_kit/hets/symbols.py +512 -0
  168. unicode_logic_kit/hol/__init__.py +140 -0
  169. unicode_logic_kit/hol/_ho_common.py +323 -0
  170. unicode_logic_kit/hol/_isabelle_binders.py +125 -0
  171. unicode_logic_kit/hol/classical.py +812 -0
  172. unicode_logic_kit/hol/deepshallow/__init__.py +45 -0
  173. unicode_logic_kit/hol/deepshallow/_common.py +177 -0
  174. unicode_logic_kit/hol/deepshallow/conditional.py +225 -0
  175. unicode_logic_kit/hol/deepshallow/intuitionistic.py +181 -0
  176. unicode_logic_kit/hol/deepshallow/modal.py +217 -0
  177. unicode_logic_kit/hol/deepshallow/qml.py +406 -0
  178. unicode_logic_kit/hol/deepshallow/relevant.py +206 -0
  179. unicode_logic_kit/hol/free.py +753 -0
  180. unicode_logic_kit/hol/goedel.py +336 -0
  181. unicode_logic_kit/hol/ho_modal.py +1743 -0
  182. unicode_logic_kit/hol/intuitionistic.py +403 -0
  183. unicode_logic_kit/hol/isabelle_conditional.py +593 -0
  184. unicode_logic_kit/hol/isabelle_modal.py +1908 -0
  185. unicode_logic_kit/hol/isabelle_relevant.py +412 -0
  186. unicode_logic_kit/hol/isabelle_runner.py +1147 -0
  187. unicode_logic_kit/hol/isabelle_substructural.py +884 -0
  188. unicode_logic_kit/hol/lean.py +1018 -0
  189. unicode_logic_kit/hol/manyvalued.py +921 -0
  190. unicode_logic_kit/hol/secondorder.py +687 -0
  191. unicode_logic_kit/hol/thf_modal.py +941 -0
  192. unicode_logic_kit/hol/thirdorder.py +397 -0
  193. unicode_logic_kit/ilp/__init__.py +89 -0
  194. unicode_logic_kit/ilp/readback.py +389 -0
  195. unicode_logic_kit/ilp/separation.py +153 -0
  196. unicode_logic_kit/ilp/task.py +730 -0
  197. unicode_logic_kit/logic.py +163 -0
  198. unicode_logic_kit/mcp/__init__.py +28 -0
  199. unicode_logic_kit/mcp/__main__.py +5 -0
  200. unicode_logic_kit/mcp/chem_tools.py +1031 -0
  201. unicode_logic_kit/mcp/server.py +2453 -0
  202. unicode_logic_kit/mcp/syntax_spec.py +681 -0
  203. unicode_logic_kit/prob/__init__.py +53 -0
  204. unicode_logic_kit/prob/_bdd.py +225 -0
  205. unicode_logic_kit/prob/_column_gen.py +668 -0
  206. unicode_logic_kit/prob/distribution.py +686 -0
  207. unicode_logic_kit/prob/nilsson.py +470 -0
  208. unicode_logic_kit/py.typed +0 -0
  209. unicode_logic_kit/semantics/__init__.py +137 -0
  210. unicode_logic_kit/semantics/_modal_reject.py +156 -0
  211. unicode_logic_kit/semantics/action_models.py +466 -0
  212. unicode_logic_kit/semantics/asp_models.py +1200 -0
  213. unicode_logic_kit/semantics/conditional.py +580 -0
  214. unicode_logic_kit/semantics/dynamic_epistemic.py +95 -0
  215. unicode_logic_kit/semantics/free_logic.py +913 -0
  216. unicode_logic_kit/semantics/fuzzy.py +384 -0
  217. unicode_logic_kit/semantics/fuzzy_kripke.py +442 -0
  218. unicode_logic_kit/semantics/intuitionistic.py +581 -0
  219. unicode_logic_kit/semantics/kripke.py +1139 -0
  220. unicode_logic_kit/semantics/manyvalued.py +580 -0
  221. unicode_logic_kit/semantics/matrix.py +342 -0
  222. unicode_logic_kit/semantics/model_eval.py +1135 -0
  223. unicode_logic_kit/semantics/modelfinder.py +1036 -0
  224. unicode_logic_kit/semantics/nonmonotonic.py +372 -0
  225. unicode_logic_kit/semantics/relevant.py +331 -0
  226. unicode_logic_kit/semantics/secondorder.py +657 -0
  227. unicode_logic_kit/semantics/structures.py +352 -0
  228. unicode_logic_kit/semantics/tarski.py +975 -0
  229. unicode_logic_kit/semantics/team.py +315 -0
  230. unicode_logic_kit/semantics/team_translation.py +416 -0
  231. unicode_logic_kit/semantics/thirdorder.py +358 -0
  232. unicode_logic_kit/semantics/tnorm.py +85 -0
  233. unicode_logic_kit/semantics/truthtable.py +201 -0
  234. unicode_logic_kit-0.31.0.dist-info/METADATA +333 -0
  235. unicode_logic_kit-0.31.0.dist-info/RECORD +237 -0
  236. unicode_logic_kit-0.31.0.dist-info/WHEEL +4 -0
  237. unicode_logic_kit-0.31.0.dist-info/licenses/LICENSE +21 -0
@@ -0,0 +1,1325 @@
1
+ """Prover9 / LADR input: read Prover9-syntax formulas into the AST.
2
+
3
+ This is the inverse of :meth:`Node.to_prover9`. Prover9's surface syntax differs
4
+ from the toolkit's Unicode notation (``all X``/``exists X`` quantifiers, ``-`` for
5
+ negation, ``& | -> <->`` connectives, infix comparison predicates), so it gets
6
+ its own Lark grammar.
7
+
8
+ Variables. :meth:`Node.to_prover9` writes under ``set(prolog_style_variables)``, and this
9
+ reader reads names by Prover9's own rules (measured on Prover9 2026-8A):
10
+
11
+ * ``all x`` and ``exists x`` bind the SYMBOL they name, whatever its case. In
12
+ ``all x (man(x) -> mortal(x))`` and in ``all X (man(X) -> mortal(X))`` the three occurrences
13
+ are one :class:`Variable`, for as long as the operand that follows the variable lasts (it
14
+ ends before a ``&``, ``|``, ``->``, ``<->`` or ``<-``). A quantifier over the same spelling
15
+ inside that operand rebinds it. A name that is applied to arguments (``x(a)``) is a
16
+ predicate or a function, not an occurrence of the variable.
17
+ * A name that no quantifier binds is a variable or a constant by the convention of the text.
18
+ Under ``set(prolog_style_variables)`` a name that begins with an upper-case letter
19
+ (``A`` to ``Z``; an underscore does not count) is a variable and every other name a
20
+ constant; without that flag a name that begins with ``u`` to ``z`` is a variable.
21
+ :func:`parse_prover9_problem` reads the flag as Prover9 does: the LAST ``set`` or ``clear``
22
+ of ``prolog_style_variables`` in the file decides for every formula of it, a file without
23
+ one reads Prover9's default. :func:`parse_prover9` has no file around the formula and keeps
24
+ ``set(prolog_style_variables)`` (or pass ``prolog_style_variables=False``).
25
+ * Variables are compared as written: ``Xa`` and ``XA`` are two variables, and so are ``x``
26
+ and ``X``. The name of a :class:`Variable` is the lower-case of its spelling (the inverse of
27
+ :meth:`Variable.to_prover9`, which upper-cases), and a spelling whose lower-case another
28
+ spelling already has gets a fresh name (``x0``): two spellings are never one variable.
29
+ Constants, predicates and functions keep their case.
30
+
31
+ A name applied to arguments is a predicate (in formula position) or function (in term
32
+ position) and keeps its case; a bare name in FORMULA position is a nullary (propositional)
33
+ predicate (Prover9 itself refuses a bare upper-case proposition, which it reads as a variable,
34
+ and a bound name used as a formula; this reader is more lenient about the first and refuses
35
+ the second). Comparison operators map back to the ``=`` / ``≠`` /
36
+ ``<`` / ``>`` / ``≤`` / ``≥`` atoms and ``+ - * /`` to the arithmetic functions.
37
+
38
+ Connectives: ``->``, ``<->`` and the reverse implication ``<-`` (``p <- q`` is ``q -> p``)
39
+ bind looser than ``|`` and ``&``. Prover9 reads the three as non-associative: a chain or a mix
40
+ of two of them without parentheses (``a <- b <- c``, ``a <- b -> c``) is a syntax error there
41
+ (measured), and this reader refuses a ``<-`` in such a chain as well. It still reads
42
+ ``a -> b -> c`` as ``a -> (b -> c)``, which Prover9 refuses. ``<`` is the comparison only where
43
+ no ``-`` follows it directly: ``a <-b`` is a reverse implication, ``a < -b`` the comparison
44
+ of ``a`` and ``-b``.
45
+
46
+ Keywords: ``all`` and ``exists`` are quantifiers only as words of their own, followed by a
47
+ variable. ``allowed(a)``, ``exists_in(b)`` and ``allergic(a)`` are atoms, and so is ``all(a)``,
48
+ as in Prover9.
49
+
50
+ Prover9's own constants are read too: ``$T`` and ``$F`` in formula position are the
51
+ truth constants, the nullary atoms ``$true`` and ``$false`` (the atoms
52
+ :meth:`Node.to_prover9` writes as ``$T`` and ``$F``, so ``parse_prover9(node.to_prover9())
53
+ == node`` for them). They are formulas, not terms: ``P($T)`` is refused.
54
+
55
+ Quoted symbols: a double-quoted symbol is never a variable in Prover9 (LADR stores it
56
+ with its quote characters, so the first character it tests is the quote), which is how
57
+ :meth:`Node.to_prover9` writes a constant or a proposition whose name begins with an
58
+ upper-case letter or an underscore (``P("Gaseous")``, ``"Rain"``). This reader reads
59
+ it back as a name that is never a variable: ``"Rain"`` in formula position is the
60
+ nullary atom ``Rain``, ``"Gaseous"`` in term position the :class:`Constant`
61
+ ``Gaseous``, and a quoted name applied to arguments (``"Mother"(a)``) an atom or a
62
+ :class:`Function` of that name. Prover9 keeps ``"rain"`` and ``rain`` apart as two
63
+ symbols and this reader does not (the AST has one name), so a text that writes the
64
+ SAME symbol (the same predicate, function, constant or number, at one arity) both
65
+ with and without quotes is refused by name as a :class:`Prover9ParsingError`, rather
66
+ than read as one symbol: a file is one text, and the record of what was met is kept
67
+ across all its formulas. (``"Rain"`` as a proposition next to ``Rain(x)`` as a
68
+ predicate is not that: they are two symbols here too.) Only a quoted name made of
69
+ letters, digits and an underscore (``[A-Za-z_][A-Za-z0-9_]*``) is read, because that
70
+ is what the writer produces and the only shape :meth:`Node.to_prover9` can write
71
+ back, and a quoted NUMERAL (``"2.5"``, ``"-1"``) in term position, which is the
72
+ :class:`Number` of that value (the text must be the canonical spelling of the number,
73
+ as :meth:`Number.to_prover9` writes it, one per value: ``"2.50"`` and ``"1.0"`` are
74
+ refused, because Prover9 keeps them apart from ``"2.5"`` and ``"1"``, which are the same
75
+ numbers); any other quoted text
76
+ is refused by name as a :class:`Prover9ParsingError`. The file scanner of
77
+ :func:`parse_prover9_problem` skips a quoted symbol whole: a ``.`` or a ``%`` inside it
78
+ ends no statement and starts no comment.
79
+
80
+ A term can be written ``-(a, b)``: the function ``-`` of two arguments, which is how
81
+ :meth:`Function.to_prover9` writes a binary minus (Prover9 has no infix minus, ``(a - b)``
82
+ is a syntax error there; the infix ``a - b`` of this reader's grammar is kept for the
83
+ files it has always read). A prefix minus is the function ``-`` of one argument, and binds
84
+ as Prover9's does (priority 350, tighter than the comparisons and the sums): ``-(a)``,
85
+ ``-a`` and ``- f(a)`` are terms, ``-a = b`` is the equation of the terms ``-a`` and ``b``
86
+ (Prover9 echoes ``-a = b.``), ``-a + b = c`` is ``(-a) + b = c``, and only a minus in front
87
+ of a formula that is not a comparison, or of a parenthesised formula, is a negation
88
+ (``-P(a)``, ``-(a = b)``, which Prover9 echoes ``a != b``). A bare ``-1`` is read as the
89
+ number minus one, also in front of a comparison (``-1 < x``), although Prover9 reads it as
90
+ ``-`` applied to the constant ``1``: the writer of this kit writes the number as ``"-1"``.
91
+
92
+ A text that spells one numeral two ways is refused by name: Prover9 keeps ``01`` and ``1``,
93
+ ``1.0`` and ``1``, ``2.50`` and ``2.5`` apart as symbols, a :class:`Number` has one text per
94
+ value, and the file ``P(01). -P(1).`` (consistent for Prover9) would be read as ``P(1)`` and
95
+ ``¬P(1)``. Every formula of a file counts for it, as for a symbol written both quoted and bare.
96
+ A numeral that stands alone in a text is read as the number it spells.
97
+
98
+ Depth: the parse tree is transformed with an explicit stack, so a chain of operands or a stack
99
+ of negations or quantifiers is read at any depth the parser itself reads. A formula that
100
+ still exhausts the interpreter's recursion limit is a :class:`Prover9ParsingError`, never a
101
+ bare ``RecursionError``.
102
+
103
+ A free variable: Prover9 closes each formula of a file universally, so ``P(X)`` there says
104
+ ``∀X P(X)``. This reader reads the text and does not close it: the variable stays free in the
105
+ AST, and the kit's provers read a free variable of a problem as ONE unknown element, the same
106
+ in every formula (a parameter). A problem built from such a file therefore asks another
107
+ question than Prover9 does for the file; wrap the formula in ``∀`` first when the closure is
108
+ what is meant.
109
+
110
+ Note: :meth:`Node.to_prover9` desugars exclusive-or to ``(a | b) & -(a & b)`` (Prover9
111
+ has no xor operator), so an :class:`Xor` round-trips to that conjunctive form, not to
112
+ ``Xor``.
113
+
114
+ op(...) declarations: :func:`parse_prover9_problem` applies a genuinely NEW
115
+ ``op(precedence, type, symbol)`` directive (Prover9's own syntax for declaring an
116
+ operator; the manual's page on parsing declarations —
117
+ https://www.cs.unm.edu/~mccune/prover9/manual/2009-11A/syntax.html, mirroring
118
+ ``ladr/parse.c``'s ``declare_standard_parse_types()`` in the Prover9/LADR source —
119
+ is the citation for every precedence number and type keyword below) to every
120
+ formula that follows it in the same file. Two things are refused outright, by
121
+ name, as a :class:`Prover9ParsingError`, regardless of whether the declared
122
+ operator is ever used — because applying them could silently change the
123
+ meaning of other, unrelated text already in the file:
124
+
125
+ - **Redeclaring a built-in** (any symbol already in :data:`_DEFAULT_OPS`, e.g.
126
+ ``op(500, infix, "+")``) is refused: doing this properly would mean making the
127
+ *entire* grammar table-driven, a much larger and riskier change this reader does
128
+ not make (see ``tests/test_prover9_ops.py`` and
129
+ ``tests/test_prover9_entailment.py``'s module docstrings for how the reader is
130
+ checked against an independent route and, where a binary exists, against a real
131
+ Prover9).
132
+ - **Redeclaring a symbol this same file already declared** is refused the same way.
133
+
134
+ A **malformed** ``op(...)`` (wrong arity, a non-integer precedence, an unknown
135
+ type keyword, a symbol that is not a bare or double-quoted identifier, or a
136
+ precedence Prover9 itself would not accept — outside 1-998) is also refused by
137
+ name; these are syntax problems in the directive itself, independent of placement.
138
+
139
+ Everything else syntactically well-formed is *accepted* (matching real Prover9,
140
+ which does not reject any of it either) and applied where this reader has a
141
+ splice point for it, left harmlessly **inert** otherwise — see
142
+ :func:`_classify_and_splice` for exactly which placements splice and which are
143
+ inert (``type="ordinary"``; a precedence tying a built-in tier; a precedence at
144
+ or above the quantifier tier 750; prefix/postfix at an atom-tier precedence). An
145
+ inert declaration still blocks a later redeclaration of the same name, but a
146
+ formula that tries to *use* it as an operator sees an ordinary undeclared name
147
+ and fails to parse — loudly, just at that point rather than at the declaration
148
+ (this mirrors every op() directive's behaviour before this feature existed, for
149
+ exactly the declarations this reader still cannot safely place).
150
+
151
+ What DOES splice in: a new symbol at a **term-tier** precedence (< 500) becomes a
152
+ :class:`Function`-producing operator spliced next to ``unit_term`` (so its
153
+ operands are atomic terms — parenthesize an arithmetic sub-expression used as an
154
+ operand); a new symbol at an **atom-tier** precedence (501-699 or 701-749) becomes
155
+ an :class:`Atom`-producing operator spliced next to the existing comparisons, with
156
+ :class:`Node` operands taken from the ``term`` grammar. ``infix`` (Prover9's
157
+ ``xfx``, non-associative) chains only inside explicit parentheses — exactly as in
158
+ Prover9 itself; ``infix_left``/``infix_right`` (``yfx``/``xfy``) chain without
159
+ parentheses in that direction. ``prefix``/``prefix_paren`` and
160
+ ``postfix``/``postfix_paren`` splice in at the term tier only; this reader does
161
+ not distinguish ``fy`` from ``fx`` self-chaining (both may chain without
162
+ parentheses) — a narrower distinction real Prover9 makes and this one does not
163
+ need for any formula it still *accepts* (it only widens what parses, never what
164
+ a given input means). Only the 3-argument ``op(...)`` form is read; Prover9's
165
+ 2-argument ``op(type, symbol)`` shorthand (valid only for ``type="ordinary"``)
166
+ is a malformed-arity refusal here.
167
+
168
+ Public API: :func:`parse_prover9` (a single formula; a trailing ``.`` is accepted).
169
+ """
170
+
171
+ import re
172
+ from dataclasses import dataclass
173
+ from functools import lru_cache
174
+
175
+ from lark import Lark, Transformer, Tree
176
+
177
+ from ._fol_nodes import NumeralTextError, _numeral_from_text
178
+ from ._identifiers import fresh_variable_like
179
+ from ._numeral_symbols import numeral_name
180
+ from .nodes import (
181
+ Node, Variable, Constant, Number, Function,
182
+ Atom, Not, And, Or, Implies, Iff, Quantifier,
183
+ )
184
+ from .naming import ParsingError
185
+
186
+
187
+ class Prover9ParsingError(ParsingError):
188
+ """A Prover9 import failure carrying a plain message (subclasses ParsingError)."""
189
+
190
+ def __init__(self, message: str):
191
+ self.args = (message,)
192
+
193
+ def __str__(self):
194
+ return self.args[0]
195
+
196
+
197
+ def _where(token) -> str:
198
+ """Where ``token`` stands in the text that was parsed, for a message (``""`` when unknown)."""
199
+ line, column = getattr(token, "line", None), getattr(token, "column", None)
200
+ if line is None or column is None:
201
+ return ""
202
+ return f" (at line {line}, column {column} of the formula)"
203
+
204
+
205
+ def _numeral(text: str, token=None):
206
+ """The value of the decimal numeral ``text``, read exactly (see
207
+ :func:`~unicode_logic_kit.fol._fol_nodes._numeral_from_text`); a
208
+ :class:`Prover9ParsingError` when no number holds it, which names the position of
209
+ ``token`` (the token the numeral was read from) when one is given. Whatever the numeral
210
+ reader refuses with (a ``ValueError`` for a text of thousands of digits, an ``OverflowError``,
211
+ its own :class:`~unicode_logic_kit.fol._fol_nodes.NumeralTextError`) is that error here."""
212
+ try:
213
+ return _numeral_from_text(text)
214
+ except (ValueError, ArithmeticError, NumeralTextError) as exc:
215
+ message = str(exc)
216
+ if message.startswith("SYNTAX_ERROR: "):
217
+ message = message[len("SYNTAX_ERROR: "):]
218
+ raise Prover9ParsingError(f"SYNTAX_ERROR: {message}{_where(token)}") from None
219
+
220
+
221
+ # The formula sub-grammar, shared by the single-formula parser and the whole-file
222
+ # parser. Deliberately kept free of the file-level keyword terminals
223
+ # (set/clear/assign/formulas/end_of_list): in single-formula mode a predicate that
224
+ # happens to be named e.g. ``set`` still lexes as a NAME, preserving backward
225
+ # compatibility for :func:`parse_prover9`.
226
+ #
227
+ # ``{atom_extra}`` / ``{term_extra}`` / ``{unit_term_extra}`` are splice points for
228
+ # newly op()-declared operators (see :func:`_build_custom_grammar`); with all three
229
+ # empty (the default, and the common case — a file with no custom operators) this
230
+ # formats to byte-identical text to the fixed grammar this module has always used,
231
+ # so the shared singleton parser below is neither rebuilt nor slowed down.
232
+ _FORMULA_RULES_TEMPLATE = r"""
233
+ ?formula: equiv
234
+ ?equiv: imp
235
+ | imp "<->" imp -> iff_
236
+ | disj "<-" disj -> rimplies_
237
+ ?imp: disj
238
+ | disj "->" imp -> implies_
239
+ ?disj: conj
240
+ | disj "|" conj -> or_
241
+ ?conj: unary
242
+ | conj "&" unary -> and_
243
+ ?unary: "-" negated -> neg
244
+ | _ALL NAME unary -> forall
245
+ | _EXISTS NAME unary -> exists
246
+ | "(" formula ")"
247
+ | atom
248
+
249
+ ?negated: "-" negated -> neg
250
+ | _ALL NAME unary -> forall
251
+ | _EXISTS NAME unary -> exists
252
+ | "(" formula ")"
253
+ | plain_atom
254
+
255
+ ?atom: comparison
256
+ | plain_atom
257
+
258
+ ?comparison: term "=" term -> equality
259
+ | term "!=" term -> disequality
260
+ | term "<=" term -> le
261
+ | term ">=" term -> ge
262
+ | term _LT term -> lt
263
+ | term ">" term -> gt{atom_extra}
264
+
265
+ ?plain_atom: NAME "(" termlist ")" -> pred_app
266
+ | QNAME "(" termlist ")" -> qpred_app
267
+ | NAME -> prop_atom
268
+ | TRUTH -> truth_atom
269
+ | QNAME -> qprop_atom
270
+
271
+ ?term: sum{term_extra}
272
+ ?sum: product
273
+ | sum "+" product -> add
274
+ | sum "-" product -> sub
275
+ ?product: unit_term
276
+ | product "*" unit_term -> mul
277
+ | product "/" unit_term -> div
278
+ ?unit_term: signed_term
279
+ | NUMBER -> number{unit_term_extra}
280
+ ?signed_term: NAME "(" termlist ")" -> func_app
281
+ | QNAME "(" termlist ")" -> qfunc_app
282
+ | "-" "(" term "," termlist ")" -> minus_app
283
+ | NAME -> name_term
284
+ | QNAME -> qname_term
285
+ | "(" term ")"
286
+ | "-" signed_term -> uminus
287
+
288
+ termlist: term ("," term)*
289
+
290
+ NAME: /[A-Za-z_][A-Za-z0-9_]*/
291
+ _ALL: /all(?![A-Za-z0-9_])/
292
+ _EXISTS: /exists(?![A-Za-z0-9_])/
293
+ _LT: /<(?!-)/
294
+ QNAME: /"[^"]*"/
295
+ TRUTH: /\$[TF](?![A-Za-z0-9_])/
296
+ NUMBER: /-?[0-9]+(\.[0-9]+)?/
297
+
298
+ %import common.WS
299
+ %ignore WS
300
+ %ignore /%[^\r\n]*/
301
+ """
302
+
303
+ _FORMULA_RULES = _FORMULA_RULES_TEMPLATE.format(atom_extra="", term_extra="", unit_term_extra="")
304
+ _GRAMMAR = "?start: formula\n\n" + _FORMULA_RULES
305
+
306
+
307
+ @dataclass(frozen=True)
308
+ class Prover9Formula:
309
+ """One formula read from a Prover9 file: its ``role`` and parsed ``formula``.
310
+
311
+ ``role`` is the enclosing ``formulas(<name>)`` list name (``"sos"`` /
312
+ ``"assumptions"`` / ``"goals"`` / …), or ``""`` for a bare top-level formula.
313
+ """
314
+
315
+ role: str
316
+ formula: Node
317
+
318
+
319
+ # --- op(...) operator declarations (see the module docstring) ---
320
+
321
+ @dataclass(frozen=True)
322
+ class _CustomOp:
323
+ """One parsed ``op(precedence, type, symbol)`` record."""
324
+
325
+ precedence: int
326
+ type: str
327
+ symbol: str
328
+
329
+
330
+ # Prover9's own default operator table, from ``ladr/parse.c``'s
331
+ # ``declare_standard_parse_types()`` in the Prover9/LADR source (mirrored by the
332
+ # manual's "Clauses and Formulas" / parsing-declarations page — see the module
333
+ # docstring for the URL). Each entry is (precedence, type, symbol); type uses
334
+ # Prover9's own op()-command keywords, not Prolog's xfx/xfy/yfx names (the manual
335
+ # documents the correspondence: infix=xfx, infix_left=yfx, infix_right=xfy,
336
+ # prefix=fy, prefix_paren=fx, postfix=yf, postfix_paren=xf). Only the subset this
337
+ # reader's own grammar implements matters for parsing; the FULL real table is kept
338
+ # here anyway so that redeclaring any genuine Prover9 built-in — even one this
339
+ # reader's grammar never parses on its own, like "^" or "#" — is refused by name.
340
+ _DEFAULT_OPS = (
341
+ (810, "infix_right", "#"),
342
+ (800, "infix", "<->"),
343
+ (800, "infix", "->"),
344
+ (800, "infix", "<-"),
345
+ (790, "infix_right", "|"),
346
+ (780, "infix_right", "&"),
347
+ (700, "infix", "="),
348
+ (700, "infix", "!="),
349
+ (700, "infix", "=="),
350
+ (700, "infix", "<"),
351
+ (700, "infix", "<="),
352
+ (700, "infix", ">"),
353
+ (700, "infix", ">="),
354
+ (500, "infix", "+"),
355
+ (500, "infix", "*"),
356
+ (500, "infix", "@"),
357
+ (500, "infix", "/"),
358
+ (500, "infix", "\\"),
359
+ (500, "infix", "^"),
360
+ (500, "infix", "v"),
361
+ (350, "prefix", "-"),
362
+ (300, "postfix", "'"),
363
+ )
364
+ _DEFAULT_OP_SYMBOLS = frozenset(sym for _prec, _type, sym in _DEFAULT_OPS)
365
+ # "all"/"exists" are hard-coded grammar keywords rather than entries in
366
+ # _DEFAULT_OPS (they are not ordinary binary/unary operators — quantifiers bind a
367
+ # variable), but redeclaring them by name is refused the same way.
368
+ _RESERVED_SYMBOLS = _DEFAULT_OP_SYMBOLS | {"all", "exists"}
369
+
370
+ # Precedences that coincide with a built-in tier (every distinct precedence in
371
+ # _DEFAULT_OPS — 300, 350, 500, 700 — plus the quantifiers' 750 and the
372
+ # argument-list comma's 999, both declared standalone in
373
+ # declare_standard_parse_types() rather than through _DEFAULT_OPS's op()-style
374
+ # entries; 780/790/800/810 are already covered by _DEFAULT_OPS's own
375
+ # precedences). A NEW operator declared at exactly one of these ties an
376
+ # existing tier: this reader's placement rule (below 500 is a term operator,
377
+ # 500-750 an atom operator) cannot tell which side of the tie it belongs on —
378
+ # see _classify_and_splice for what happens then (left inert, not refused; a
379
+ # declaration this reader cannot safely place is simply never applied, exactly
380
+ # like every op() directive before this feature existed).
381
+ _RESERVED_PRECEDENCES = frozenset({300, 350, 500, 700, 750, 780, 790, 800, 810, 999})
382
+
383
+ _P9_OP_TYPE_NAMES = (
384
+ "infix", "infix_left", "infix_right",
385
+ "prefix", "prefix_paren", "postfix", "postfix_paren", "ordinary",
386
+ )
387
+ _P9_OP_BODY_RE = re.compile(r"^op\((?P<inner>.*)\)$", re.DOTALL)
388
+ _P9_INT_RE = re.compile(r"^-?[0-9]+$")
389
+ _P9_QUOTED_SYMBOL_RE = re.compile(r'^"([^"\\]*)"$')
390
+ _P9_BARE_SYMBOL_RE = re.compile(r"^[A-Za-z_][A-Za-z0-9_]*$")
391
+
392
+
393
+ def _split_top_level_commas(text: str) -> list:
394
+ """Split ``text`` on commas that are not nested inside ``[...]`` or ``"..."``.
395
+
396
+ Used both for the outer ``op(precedence, type, symbol_or_list)`` arguments
397
+ and for the inner ``[sym1, sym2, ...]`` symbol list.
398
+ """
399
+ parts = []
400
+ depth = 0
401
+ in_quotes = False
402
+ current = []
403
+ for ch in text:
404
+ if in_quotes:
405
+ current.append(ch)
406
+ if ch == '"':
407
+ in_quotes = False
408
+ continue
409
+ if ch == '"':
410
+ in_quotes = True
411
+ current.append(ch)
412
+ elif ch == "[":
413
+ depth += 1
414
+ current.append(ch)
415
+ elif ch == "]":
416
+ depth -= 1
417
+ current.append(ch)
418
+ elif ch == "," and depth == 0:
419
+ parts.append("".join(current))
420
+ current = []
421
+ else:
422
+ current.append(ch)
423
+ parts.append("".join(current))
424
+ return [p.strip() for p in parts]
425
+
426
+
427
+ def _parse_op_symbol(token: str) -> str:
428
+ """Read one op() symbol token: a bare identifier or a double-quoted one.
429
+
430
+ Only symbols shaped like the grammar's own NAME terminal
431
+ (``[A-Za-z_][A-Za-z0-9_]*``) are supported — a symbolic token such as
432
+ ``"++"`` cannot be spliced into the grammar as a new keyword-like literal
433
+ the way an identifier can (see the module docstring), so it is refused.
434
+ A symbolic token that names an existing built-in (``"+"``, ``"&"``, …) is
435
+ let through here regardless of shape, so :func:`_classify_and_splice` can
436
+ give the specific "redeclaring built-in" error instead of this generic one.
437
+ """
438
+ quoted = _P9_QUOTED_SYMBOL_RE.match(token)
439
+ name = quoted.group(1) if quoted else token
440
+ if name in _RESERVED_SYMBOLS or _P9_BARE_SYMBOL_RE.match(name):
441
+ return name
442
+ raise Prover9ParsingError(
443
+ f"SYNTAX_ERROR: op() symbol {token!r} is not supported by this reader "
444
+ "(only alphanumeric/underscore operator names are; a symbolic token "
445
+ "cannot be declared or redeclared here)")
446
+
447
+
448
+ def _parse_op_directive(body: str) -> list:
449
+ """Parse one ``op(precedence, type, symbol)`` directive body (``body`` is the
450
+ whole statement text, e.g. ``op(650, infix, "before")``, sans the trailing
451
+ ``.``) into a list of :class:`_CustomOp` records — one per symbol, all
452
+ sharing ``precedence``/``type``; ``op(N, TYPE, [s1, s2])`` yields two records.
453
+
454
+ Only the 3-argument form is supported — see the module docstring.
455
+
456
+ Raises:
457
+ Prover9ParsingError: on any malformed ``op(...)`` (bad arity, a
458
+ non-integer precedence, an unknown type keyword, or a symbol token
459
+ this reader does not support).
460
+ """
461
+ match = _P9_OP_BODY_RE.match(body)
462
+ if not match:
463
+ raise Prover9ParsingError(f"SYNTAX_ERROR: malformed op(...) directive: {body!r}")
464
+ args = _split_top_level_commas(match.group("inner"))
465
+ if len(args) != 3:
466
+ raise Prover9ParsingError(
467
+ "SYNTAX_ERROR: op(...) needs exactly 3 arguments (precedence, type, "
468
+ f"symbol[s]) — Prover9's 2-argument ordinary-only form is not "
469
+ f"supported by this reader; got {len(args)}: {body!r}")
470
+ prec_text, type_text, symbols_text = args
471
+ if not _P9_INT_RE.match(prec_text):
472
+ raise Prover9ParsingError(
473
+ f"SYNTAX_ERROR: op() precedence must be an integer, got {prec_text!r}")
474
+ try:
475
+ precedence = int(prec_text)
476
+ except ValueError: # Python's int() refuses a text of thousands of digits
477
+ raise Prover9ParsingError(
478
+ f"SYNTAX_ERROR: op() precedence {prec_text[:12]!r}... has {len(prec_text.lstrip('-'))} digits: "
479
+ "it is out of Prover9's valid range (1-998)") from None
480
+ if type_text not in _P9_OP_TYPE_NAMES:
481
+ raise Prover9ParsingError(
482
+ f"SYNTAX_ERROR: unknown op() type {type_text!r} (expected one of "
483
+ + ", ".join(_P9_OP_TYPE_NAMES) + ")")
484
+ if symbols_text.startswith("[") and symbols_text.endswith("]"):
485
+ inner_tokens = _split_top_level_commas(symbols_text[1:-1])
486
+ if inner_tokens == [""]:
487
+ raise Prover9ParsingError(f"SYNTAX_ERROR: empty op() symbol list: {body!r}")
488
+ else:
489
+ inner_tokens = [symbols_text]
490
+ symbols = [_parse_op_symbol(tok) for tok in inner_tokens]
491
+ return [_CustomOp(precedence, type_text, sym) for sym in symbols]
492
+
493
+
494
+ #: The first characters that make a bare name a variable (LADR ``variable_name``, measured on
495
+ #: Prover9 2026-8A, by every first letter and with names of several characters): under
496
+ #: ``set(prolog_style_variables)`` an upper-case ASCII letter, without it ``u`` to ``z``. Nothing
497
+ #: else is: ``_x`` is a constant in both conventions, and a quoted symbol never is.
498
+ _PROLOG_VARIABLE_INITIALS = frozenset("ABCDEFGHIJKLMNOPQRSTUVWXYZ")
499
+ _STANDARD_VARIABLE_INITIALS = frozenset("uvwxyz")
500
+
501
+
502
+ def _is_variable(name: str, prolog_style: bool = True) -> bool:
503
+ """Whether Prover9 reads the bare name ``name`` as a variable when no quantifier binds it:
504
+ its first character is ``A``..``Z`` under ``set(prolog_style_variables)``, ``u``..``z``
505
+ without it."""
506
+ return name[:1] in (_PROLOG_VARIABLE_INITIALS if prolog_style else _STANDARD_VARIABLE_INITIALS)
507
+
508
+
509
+ _P9_QUOTED_NAME_RE = re.compile(r"[A-Za-z_][A-Za-z0-9_]*")
510
+ _P9_QUOTED_NUMERAL_RE = re.compile(r"-?[0-9]+(\.[0-9]+)?")
511
+
512
+
513
+ def _unquote(token) -> str:
514
+ """The name inside a double-quoted symbol token, or a refusal by name.
515
+
516
+ Prover9 accepts any text without a double quote between the quotes. Only an
517
+ identifier-shaped name is read here (see the module docstring): the AST has no
518
+ way to carry another one that :meth:`Node.to_prover9` could write back.
519
+ """
520
+ name = str(token)[1:-1]
521
+ if not _P9_QUOTED_NAME_RE.fullmatch(name):
522
+ raise Prover9ParsingError(
523
+ f"SYNTAX_ERROR: the quoted symbol {str(token)!r} is not supported by this "
524
+ "reader: Prover9 accepts any text between double quotes, but this reader "
525
+ "reads only a quoted name made of letters, digits and underscores that "
526
+ "does not begin with a digit (the shape Node.to_prover9 writes), or a "
527
+ "quoted numeral (\"2.5\", \"-1\"), because any other name could not be "
528
+ "written back as the same symbol")
529
+ return name
530
+
531
+
532
+ def _quoted_numeral(token) -> "Number":
533
+ """The :class:`Number` a quoted numeral symbol (``"2.5"``, ``"-1"``) stands for, or a
534
+ refusal by name when the text is not the canonical spelling of that number.
535
+
536
+ Prover9 keeps ``"2.5"`` and ``"2.50"``, and ``"1"`` and ``"1.0"``, apart as two symbols;
537
+ a :class:`Number` is identified by its value and has one text, so only the spelling
538
+ :meth:`Number.to_prover9` writes is read (``"1"``, never ``"1.0"``).
539
+ """
540
+ text = str(token)[1:-1]
541
+ value = _numeral(text, token)
542
+ if text != numeral_name(value):
543
+ raise Prover9ParsingError(
544
+ f"SYNTAX_ERROR: the quoted numeral {str(token)!r} is not read: Prover9 keeps "
545
+ f"it apart from the quoted numeral {numeral_name(value)!r} that is the same "
546
+ "number, and a Number node has one text, so reading it would merge two "
547
+ "symbols. Write the number in its canonical form")
548
+ return Number(value)
549
+
550
+
551
+ class _Prover9Transformer(Transformer):
552
+ """Turn the Lark parse tree into the toolkit AST."""
553
+
554
+ def iff_(self, items):
555
+ return Iff(items[0], items[1])
556
+
557
+ def implies_(self, items):
558
+ return Implies(items[0], items[1])
559
+
560
+ def rimplies_(self, items):
561
+ # ``p <- q`` is ``q -> p``.
562
+ return Implies(items[1], items[0])
563
+
564
+ def or_(self, items):
565
+ return Or(items[0], items[1])
566
+
567
+ def and_(self, items):
568
+ return And(items[0], items[1])
569
+
570
+ def neg(self, items):
571
+ return Not(items[0])
572
+
573
+ # The binder and every occurrence of a variable carry the NAME of the AST variable already
574
+ # (see :func:`_resolve_variables`, which runs first and decides what each spelling is).
575
+ def forall(self, items):
576
+ return Quantifier("∀", Variable(str(items[0])), items[1])
577
+
578
+ def exists(self, items):
579
+ return Quantifier("∃", Variable(str(items[0])), items[1])
580
+
581
+ # --- atoms ---
582
+ def equality(self, items):
583
+ return Atom("=", [items[0], items[1]])
584
+
585
+ def disequality(self, items):
586
+ return Atom("≠", [items[0], items[1]])
587
+
588
+ def le(self, items):
589
+ return Atom("≤", [items[0], items[1]])
590
+
591
+ def ge(self, items):
592
+ return Atom("≥", [items[0], items[1]])
593
+
594
+ def lt(self, items):
595
+ return Atom("<", [items[0], items[1]])
596
+
597
+ def gt(self, items):
598
+ return Atom(">", [items[0], items[1]])
599
+
600
+ def pred_app(self, items):
601
+ return Atom(str(items[0]), items[1])
602
+
603
+ def prop_atom(self, items):
604
+ return Atom(str(items[0]), [])
605
+
606
+ # --- Prover9's own constants: $T is true, $F is false ---
607
+ def truth_atom(self, items):
608
+ return Atom("$true" if str(items[0]) == "$T" else "$false", [])
609
+
610
+ # --- quoted symbols: never a variable, whatever their first letter ---
611
+ def qpred_app(self, items):
612
+ return Atom(_unquote(items[0]), items[1])
613
+
614
+ def qprop_atom(self, items):
615
+ return Atom(_unquote(items[0]), [])
616
+
617
+ # --- terms ---
618
+ def add(self, items):
619
+ return Function("+", [items[0], items[1]])
620
+
621
+ def sub(self, items):
622
+ return Function("-", [items[0], items[1]])
623
+
624
+ def mul(self, items):
625
+ return Function("*", [items[0], items[1]])
626
+
627
+ def div(self, items):
628
+ return Function("/", [items[0], items[1]])
629
+
630
+ def func_app(self, items):
631
+ return Function(str(items[0]), items[1])
632
+
633
+ def name_term(self, items):
634
+ # A bare name that is no variable (see :func:`_resolve_variables`): a constant.
635
+ return Constant(str(items[0]))
636
+
637
+ def var_term(self, items):
638
+ return Variable(str(items[0]))
639
+
640
+ def qfunc_app(self, items):
641
+ return Function(_unquote(items[0]), items[1])
642
+
643
+ def qname_term(self, items):
644
+ if _P9_QUOTED_NUMERAL_RE.fullmatch(str(items[0])[1:-1]):
645
+ return _quoted_numeral(items[0])
646
+ return Constant(_unquote(items[0]))
647
+
648
+ def minus_app(self, items):
649
+ return Function("-", [items[0], *items[1]])
650
+
651
+ def uminus(self, items):
652
+ return Function("-", [items[0]])
653
+
654
+ def number(self, items):
655
+ return Number(_numeral(str(items[0]), items[0]))
656
+
657
+ def termlist(self, items):
658
+ return list(items)
659
+
660
+
661
+ _PARSER = Lark(_GRAMMAR, start="start", parser="earley")
662
+ _TRANSFORMER = _Prover9Transformer()
663
+
664
+
665
+ def _classify_and_splice(custom_ops: tuple) -> tuple:
666
+ """Validate ``custom_ops`` (in declaration order) and split them into the
667
+ grammar-splice buckets :func:`_build_custom_grammar` needs: atom-tier infix,
668
+ term-tier infix, term-tier prefix and term-tier postfix.
669
+
670
+ Two things are refused outright, by name, regardless of whether the
671
+ operator is ever used in a formula — because accepting them would risk
672
+ silently reinterpreting other, unrelated text in the same file:
673
+ redeclaring an existing built-in (:data:`_RESERVED_SYMBOLS`), and
674
+ redeclaring a symbol this same file already gave a custom meaning to.
675
+
676
+ Everything else that is syntactically well-formed is accepted (matching
677
+ real Prover9, which does not reject any of it either) and is spliced into
678
+ the grammar when it falls in a window this reader supports — below the
679
+ arithmetic tier (< 500) for a :class:`Function`-producing term operator,
680
+ or between the arithmetic and quantifier tiers (500-750, excluding the
681
+ comparison tier at 700) for an :class:`Atom`-producing one. A declaration
682
+ this reader has no splice point for — ``type="ordinary"``, a precedence
683
+ that exactly ties a built-in tier (:data:`_RESERVED_PRECEDENCES`), a
684
+ precedence at or above the quantifier tier (750), or prefix/postfix at an
685
+ atom-tier precedence — is simply left inert: recorded (so it still blocks
686
+ a later redeclaration) but never spliced in, exactly like every op()
687
+ directive before this feature existed. A formula that then tries to use
688
+ such an operator sees an ordinary undeclared name and fails to parse —
689
+ loudly, just at that point rather than at the declaration.
690
+ """
691
+ seen = set()
692
+ atom_ops, term_infix, term_prefix, term_postfix = [], [], [], []
693
+ for op in custom_ops:
694
+ if op.symbol in _RESERVED_SYMBOLS:
695
+ raise Prover9ParsingError(
696
+ f"SYNTAX_ERROR: redeclaring built-in Prover9 operator {op.symbol!r} "
697
+ "is not supported")
698
+ if op.symbol in seen:
699
+ raise Prover9ParsingError(
700
+ f"SYNTAX_ERROR: operator {op.symbol!r} is already declared by an "
701
+ "earlier op(...) in this file")
702
+ seen.add(op.symbol)
703
+ if not (1 <= op.precedence <= 998):
704
+ raise Prover9ParsingError(
705
+ f"SYNTAX_ERROR: op() precedence {op.precedence} is out of "
706
+ "Prover9's valid range (1-998)")
707
+ if op.type == "ordinary":
708
+ continue # not a mixfix operator: nothing to splice, just reserves the name
709
+ if op.precedence in _RESERVED_PRECEDENCES:
710
+ continue # ties a built-in tier: left inert (see docstring above)
711
+ if op.precedence < 500:
712
+ kind = "term"
713
+ elif op.precedence < 750:
714
+ kind = "atom"
715
+ else:
716
+ continue # would graft onto the quantifier/connective grammar: left inert
717
+ if kind == "atom" and op.type not in ("infix", "infix_left", "infix_right"):
718
+ continue # prefix/postfix only supported at a term-tier precedence: left inert
719
+ if kind == "atom":
720
+ atom_ops.append(op)
721
+ elif op.type in ("infix", "infix_left", "infix_right"):
722
+ term_infix.append(op)
723
+ elif op.type in ("prefix", "prefix_paren"):
724
+ term_prefix.append(op)
725
+ else:
726
+ term_postfix.append(op)
727
+ return tuple(atom_ops), tuple(term_infix), tuple(term_prefix), tuple(term_postfix)
728
+
729
+
730
+ def _infix_rule(rule: str, alias: str, sym: str, op_type: str, operand: str) -> str:
731
+ """Build one Lark rule definition for a new infix operator.
732
+
733
+ ``infix`` (xfx) is a single non-recursive alternative — deliberately: with
734
+ no self-recursive branch, a chain like ``a sym b sym c`` cannot be produced
735
+ by this rule at all, which is exactly Prover9's own non-associative
736
+ semantics (both operands of an xfx operator need strictly lower precedence,
737
+ so a bare 3-way chain needs explicit parentheses in real Prover9 too).
738
+ ``infix_right`` (xfy) recurses on the right operand, ``infix_left`` (yfx) on
739
+ the left, each with a non-recursive base alternative for the single-use case.
740
+ """
741
+ if op_type == "infix":
742
+ return f'{rule}: {operand} "{sym}" {operand} -> {alias}'
743
+ if op_type == "infix_right":
744
+ return (f'{rule}: {operand} "{sym}" {rule} -> {alias}\n'
745
+ f' | {operand} "{sym}" {operand} -> {alias}')
746
+ return (f'{rule}: {rule} "{sym}" {operand} -> {alias}\n'
747
+ f' | {operand} "{sym}" {operand} -> {alias}')
748
+
749
+
750
+ @lru_cache(maxsize=256)
751
+ def _build_custom_grammar(custom_ops: tuple):
752
+ """Build (and cache, by the exact tuple of active op() declarations) a Lark
753
+ parser + transformer pair that extends the shared grammar with newly
754
+ op()-declared operators. An empty tuple returns the shared singleton
755
+ (:data:`_PARSER`, :data:`_TRANSFORMER`) unchanged — no rebuild, no perf
756
+ regression for the common case of a file with no custom operators.
757
+
758
+ Validation (redeclaration, tie, range and placement checks — see
759
+ :func:`_classify_and_splice`) happens here, so it runs once per distinct
760
+ set of active declarations and is then free on every later call.
761
+
762
+ Generated Lark rule/alias names are built from a per-bucket numeric index,
763
+ never from ``op.symbol`` itself: Lark's own grammar meta-language requires
764
+ a RULE/alias identifier to start with a lowercase letter (or underscore)
765
+ and never contain an uppercase one (an uppercase-leading token is instead
766
+ read as a TERMINAL reference), while op() symbols accepted here
767
+ (:func:`_parse_op_symbol`) are only required to match the *formula*
768
+ grammar's own ``NAME`` terminal — which does allow uppercase. Splicing an
769
+ uppercase-containing symbol straight into a generated rule name (e.g.
770
+ ``atom_chain_Before``) would therefore be lexed as two malformed grammar
771
+ tokens and make the ``Lark(...)`` call below raise
772
+ ``lark.exceptions.UnexpectedToken`` — a bare Lark internals leak, not a
773
+ targeted :class:`Prover9ParsingError`, for a symbol shape this reader's
774
+ own validator otherwise accepts. The real symbol still appears in the
775
+ grammar, but only inside a quoted string literal (``"{op.symbol}"``),
776
+ where Lark's syntax places no case restriction.
777
+ """
778
+ if not custom_ops:
779
+ return _PARSER, _TRANSFORMER
780
+ atom_ops, term_infix, term_prefix, term_postfix = _classify_and_splice(custom_ops)
781
+
782
+ atom_extra, term_extra, unit_extra = [], [], []
783
+ extra_rules = []
784
+ handlers = {}
785
+
786
+ for idx, op in enumerate(atom_ops):
787
+ alias = f"custom_atom_{idx}"
788
+ rule = f"atom_chain_{idx}"
789
+ extra_rules.append(_infix_rule(rule, alias, op.symbol, op.type, "term"))
790
+ atom_extra.append(f"\n | {rule}")
791
+ handlers[alias] = (lambda items, _s=op.symbol: Atom(_s, [items[0], items[1]]))
792
+
793
+ for idx, op in enumerate(term_infix):
794
+ alias = f"custom_term_{idx}"
795
+ rule = f"term_chain_{idx}"
796
+ extra_rules.append(_infix_rule(rule, alias, op.symbol, op.type, "unit_term"))
797
+ term_extra.append(f"\n | {rule}")
798
+ handlers[alias] = (lambda items, _s=op.symbol: Function(_s, [items[0], items[1]]))
799
+
800
+ for idx, op in enumerate(term_prefix):
801
+ alias = f"custom_prefix_{idx}"
802
+ rule = f"unit_prefix_{idx}"
803
+ extra_rules.append(f'{rule}: "{op.symbol}" unit_term -> {alias}')
804
+ unit_extra.append(f"\n | {rule}")
805
+ handlers[alias] = (lambda items, _s=op.symbol: Function(_s, [items[0]]))
806
+
807
+ for idx, op in enumerate(term_postfix):
808
+ alias = f"custom_postfix_{idx}"
809
+ rule = f"unit_postfix_{idx}"
810
+ extra_rules.append(f'{rule}: unit_term "{op.symbol}" -> {alias}')
811
+ unit_extra.append(f"\n | {rule}")
812
+ handlers[alias] = (lambda items, _s=op.symbol: Function(_s, [items[0]]))
813
+
814
+ rules_text = _FORMULA_RULES_TEMPLATE.format(
815
+ atom_extra="".join(atom_extra),
816
+ term_extra="".join(term_extra),
817
+ unit_term_extra="".join(unit_extra),
818
+ )
819
+ grammar_text = "?start: formula\n\n" + rules_text + "\n" + "\n".join(extra_rules) + "\n"
820
+ parser = Lark(grammar_text, start="start", parser="earley")
821
+ transformer = _Prover9Transformer()
822
+ for name, handler in handlers.items():
823
+ setattr(transformer, name, handler)
824
+ return parser, transformer
825
+
826
+
827
+ def _normalize_custom_ops(custom_ops) -> tuple:
828
+ """Coerce ``custom_ops`` into a tuple of :class:`_CustomOp` (accepting plain
829
+ ``(precedence, type, symbol)`` triples, the public/documented shape, as well
830
+ as already-built :class:`_CustomOp` records)."""
831
+ result = []
832
+ for item in custom_ops:
833
+ if isinstance(item, _CustomOp):
834
+ result.append(item)
835
+ else:
836
+ precedence, op_type, symbol = item
837
+ result.append(_CustomOp(int(precedence), str(op_type), str(symbol)))
838
+ return tuple(result)
839
+
840
+
841
+ def parse_prover9(text: str, custom_ops=(), *, prolog_style_variables: bool = True) -> Node:
842
+ """Parse a single Prover9-syntax formula into a toolkit :class:`Node`.
843
+
844
+ A trailing period (Prover9 terminates each formula with ``.``) is accepted
845
+ and ignored.
846
+
847
+ A formula has no file around it to say which names are variables, so this reader
848
+ keeps the convention of :meth:`Node.to_prover9`, ``set(prolog_style_variables)``: a name
849
+ that no quantifier binds is a variable when it begins with an upper-case letter and a
850
+ constant otherwise (see the module docstring for what a quantifier binds). Pass
851
+ ``prolog_style_variables=False`` for Prover9's default, where it is a variable when it
852
+ begins with ``u`` to ``z``.
853
+
854
+ Args:
855
+ text: a Prover9 formula, e.g. ``"(all X (man(X) -> mortal(X)))"``.
856
+ custom_ops: previously-declared ``op(precedence, type, symbol)``
857
+ operators that extend the grammar for this one parse, as
858
+ ``(precedence, type, symbol)`` triples, in declaration order — see
859
+ the module docstring's "op(...) declarations" section for exactly
860
+ which operators are applied and which are refused.
861
+ :func:`parse_prover9_problem` builds and threads this automatically
862
+ from a file's own ``op(...)`` directives; most callers parsing a
863
+ single formula never need to pass it.
864
+ prolog_style_variables: which of the two conventions reads a name that is bound
865
+ by no quantifier (``True`` is the default of this function).
866
+
867
+ Returns:
868
+ The formula as a toolkit :class:`Node`.
869
+
870
+ Raises:
871
+ Prover9ParsingError: if ``text`` is not a well-formed Prover9 formula,
872
+ or ``custom_ops`` contains a declaration this reader refuses.
873
+ """
874
+ return _parse_formula(text, custom_ops, {}, prolog_style_variables)
875
+
876
+
877
+ # The tree nodes that name a symbol, with the key of the symbol they stand for.
878
+ # ``(kind, name, arity)`` as the AST keeps it: a name in formula position is a predicate, in
879
+ # term position a function (a constant is a function of arity 0), a number its own kind.
880
+ _P9_BARE_SYMBOL_RULES = {"pred_app": "predicate", "func_app": "function"}
881
+ _P9_QUOTED_SYMBOL_RULES = {"qpred_app": "predicate", "qfunc_app": "function"}
882
+
883
+
884
+ def _symbol_spellings(tree) -> list:
885
+ """Every symbol of a parse tree as ``(key, "bare" | "quoted")``, ``key`` being
886
+ ``(kind, name, arity)`` as the AST keeps a symbol (the name only, not its quotes).
887
+
888
+ A variable is no symbol: :func:`_resolve_variables` has turned every name in term position
889
+ that is a variable (bound by a quantifier, or one by the convention of the file) into a
890
+ ``var_term`` node, so the ``name_term`` nodes that are left are constants.
891
+ """
892
+ found = []
893
+ for sub in tree.iter_subtrees():
894
+ rule = str(sub.data)
895
+ if rule in _P9_BARE_SYMBOL_RULES or rule in _P9_QUOTED_SYMBOL_RULES:
896
+ bare = rule in _P9_BARE_SYMBOL_RULES
897
+ kind = _P9_BARE_SYMBOL_RULES[rule] if bare else _P9_QUOTED_SYMBOL_RULES[rule]
898
+ name = str(sub.children[0]) if bare else str(sub.children[0])[1:-1]
899
+ found.append(((kind, name, len(sub.children[1].children)), "bare" if bare else "quoted"))
900
+ elif rule == "prop_atom":
901
+ found.append((("predicate", str(sub.children[0]), 0), "bare"))
902
+ elif rule == "qprop_atom":
903
+ found.append((("predicate", str(sub.children[0])[1:-1], 0), "quoted"))
904
+ elif rule == "name_term":
905
+ found.append((("function", str(sub.children[0]), 0), "bare"))
906
+ elif rule == "qname_term":
907
+ name = str(sub.children[0])[1:-1]
908
+ if _P9_QUOTED_NUMERAL_RE.fullmatch(name):
909
+ found.append((("numeral", name, 0), "quoted"))
910
+ else:
911
+ found.append((("function", name, 0), "quoted"))
912
+ elif rule == "number":
913
+ found.append((("numeral", str(sub.children[0]), 0), "bare"))
914
+ return found
915
+
916
+
917
+ def _refuse_two_spellings(tree, spellings: dict) -> None:
918
+ """Refuse a symbol that is written both with and without double quotes.
919
+
920
+ Prover9 keeps ``"rain"`` and ``rain`` apart as two symbols, and this reader has one
921
+ name per symbol, so a text that uses both for the same predicate, function, constant
922
+ or number would be read as ONE and mean something else. ``spellings`` carries the
923
+ symbols already met (a file is one text), and is updated.
924
+ """
925
+ for key, how in _symbol_spellings(tree):
926
+ seen = spellings.setdefault(key, set())
927
+ seen.add(how)
928
+ if len(seen) == 2:
929
+ kind, name, arity = key
930
+ what = {"predicate": "predicate" if arity else "proposition",
931
+ "function": "function" if arity else "constant",
932
+ "numeral": "number"}[kind]
933
+ raise Prover9ParsingError(
934
+ f"SYNTAX_ERROR: the {what} {name!r} is written both with and without double "
935
+ f"quotes (\"{name}\" and {name}): Prover9 reads them as two symbols, this "
936
+ "reader has one name per symbol and would read them as one, so the text "
937
+ "would mean something else. Write the symbol one way.")
938
+
939
+
940
+ def _refuse_two_numeral_spellings(tree, spellings: dict) -> None:
941
+ """Refuse a text that spells one numeral value two ways (``01`` and ``1``, ``1.0`` and ``1``).
942
+
943
+ Prover9 keeps such numerals apart as symbols of their own, and a :class:`Number` is identified
944
+ by its value, so this reader would read the two as ONE numeral: ``P(01). -P(1).`` is consistent
945
+ (U = {0, 1}, 01 = 0, 1 = 1, P = {0}), and read as ``P(1), ¬P(1)`` it proves everything.
946
+ ``spellings`` is the record shared by every formula of a file (a numeral value -> the texts
947
+ it was written as), and is updated.
948
+ """
949
+ for sub in tree.iter_subtrees():
950
+ rule = str(sub.data)
951
+ if rule == "number":
952
+ written = str(sub.children[0])
953
+ elif rule == "qname_term" and _P9_QUOTED_NUMERAL_RE.fullmatch(str(sub.children[0])[1:-1]):
954
+ written = str(sub.children[0])[1:-1]
955
+ else:
956
+ continue
957
+ value = _numeral(written, sub.children[0])
958
+ texts = spellings.setdefault(("numeral value", numeral_name(value)), set())
959
+ texts.add(written)
960
+ if len(texts) > 1:
961
+ first, second = sorted(texts, key=lambda t: (len(t), t))[:2]
962
+ raise Prover9ParsingError(
963
+ f"SYNTAX_ERROR: the numerals {first} and {second} are written in one text: Prover9 "
964
+ "reads them as two symbols, this reader has one numeral per value and would read "
965
+ "them as one, so the text would mean something else. Write the number one way.")
966
+
967
+
968
+ _P9_WORD_RE = re.compile(r"[A-Za-z_][A-Za-z0-9_]*")
969
+
970
+
971
+ def _resolve_variables(tree, prolog_style: bool, record: dict, text: str) -> None:
972
+ """Decide which names of a parse tree are variables, and give each variable its AST name.
973
+
974
+ Prover9 reads a quantifier as binding the SYMBOL it names, whatever its case: in
975
+ ``all x (man(x) -> mortal(x))`` the three ``x`` are one variable, and the same text with
976
+ ``X`` says the same (measured on Prover9 2026-8A, with and without
977
+ ``set(prolog_style_variables)``). Inside the scope of the quantifier (the operand that
978
+ follows the variable) a bare name in term position is that variable; an inner quantifier
979
+ over the same spelling rebinds it, a name that is applied to arguments or stands as a
980
+ formula is no term occurrence, and outside every quantifier the convention of the file
981
+ decides (:func:`_is_variable`). The transformation builds a body before it meets the binder
982
+ that is above it, so this runs first, over the finished parse tree, with an explicit stack: a
983
+ ``name_term`` that is a variable becomes a ``var_term``, and the NAME of the binder and of
984
+ every variable occurrence is replaced by the name of the :class:`Variable` it stands for.
985
+
986
+ Variables are compared as written (``Xa`` and ``XA`` are two variables, ``x`` and ``X``
987
+ too): each spelling gets the lower-case of itself as its name, which is what
988
+ :meth:`Variable.to_prover9` (it upper-cases) is the inverse of, and a spelling whose
989
+ lower-case another spelling of the same formula already has gets a fresh name instead
990
+ (:func:`~unicode_logic_kit.fol._identifiers.fresh_variable_like`, fresh against every word of
991
+ the text). A variable that no quantifier binds is one unknown element of the whole file (the
992
+ same in every formula), so its spelling keeps its name in all of them: ``record`` is shared by
993
+ all formulas of a file and holds those names.
994
+
995
+ Raises:
996
+ Prover9ParsingError: a bound name stands as a formula (``all x (P(x) & x)``), which
997
+ Prover9 refuses as well, because a variable cannot be an atomic formula.
998
+ """
999
+ scope: dict = {} # spelling -> number of enclosing quantifiers that bind it
1000
+ sites: list = [] # (tree node, spelling, free) of every variable, in text order
1001
+ seen: set = set()
1002
+ stack = [(tree, False)]
1003
+ while stack:
1004
+ node, leaving = stack.pop()
1005
+ if leaving:
1006
+ scope[str(node.children[0])] -= 1
1007
+ continue
1008
+ if id(node) in seen:
1009
+ continue
1010
+ seen.add(id(node))
1011
+ rule = node.data
1012
+ if rule == "name_term":
1013
+ spelling = str(node.children[0])
1014
+ bound = bool(scope.get(spelling))
1015
+ if bound or _is_variable(spelling, prolog_style):
1016
+ node.data = "var_term"
1017
+ sites.append((node, spelling, not bound))
1018
+ continue
1019
+ if rule == "prop_atom":
1020
+ spelling = str(node.children[0])
1021
+ if scope.get(spelling):
1022
+ raise Prover9ParsingError(
1023
+ f"SYNTAX_ERROR: the name {spelling!r} stands as a formula inside the scope of "
1024
+ f"'all {spelling}' / 'exists {spelling}', where it is a variable: Prover9 refuses "
1025
+ "a variable as an atomic formula, and so does this reader.")
1026
+ continue
1027
+ if rule in ("forall", "exists"):
1028
+ spelling = str(node.children[0])
1029
+ sites.append((node, spelling, False))
1030
+ scope[spelling] = scope.get(spelling, 0) + 1
1031
+ stack.append((node, True))
1032
+ stack.extend((child, False) for child in reversed(node.children) if isinstance(child, Tree))
1033
+ if not sites:
1034
+ return
1035
+ free_names = record.setdefault(("variable names",), {}) # free spelling -> name, in all of the file
1036
+ free_taken = record.setdefault(("variable names taken",), set())
1037
+ free_spellings = {spelling for _, spelling, free in sites if free}
1038
+ spellings = list(dict.fromkeys(spelling for _, spelling, _ in sites))
1039
+ reserved = {s.lower() for s in spellings} | {w.lower() for w in _P9_WORD_RE.findall(text)}
1040
+ names: dict = {}
1041
+ used: set = set() # the names this formula has given out
1042
+ for free_pass in (True, False):
1043
+ for spelling in spellings:
1044
+ if (spelling in free_spellings) != free_pass:
1045
+ continue
1046
+ name = free_names.get(spelling) if free_pass else None
1047
+ if name is None:
1048
+ name = spelling.lower()
1049
+ if name in used or (free_pass and name in free_taken):
1050
+ name = fresh_variable_like(name, used | free_taken | reserved)
1051
+ if free_pass:
1052
+ free_names[spelling] = name
1053
+ free_taken.add(name)
1054
+ names[spelling] = name
1055
+ used.add(name)
1056
+ for node, spelling, _ in sites:
1057
+ node.children[0] = names[spelling]
1058
+
1059
+
1060
+ def _parse_formula(text: str, custom_ops, spellings: dict, prolog_style: bool = True) -> Node:
1061
+ """:func:`parse_prover9` for a text that is part of a bigger one: ``spellings`` is the
1062
+ record of :func:`_refuse_two_spellings` and of the names of the variables, shared by every
1063
+ formula of a file, and ``prolog_style`` the convention of the file (see :func:`_is_variable`)."""
1064
+ ops = _normalize_custom_ops(custom_ops)
1065
+ parser, transformer = _build_custom_grammar(ops) if ops else (_PARSER, _TRANSFORMER)
1066
+ stripped = text.strip()
1067
+ if stripped.endswith("."):
1068
+ stripped = stripped[:-1]
1069
+ try:
1070
+ tree = parser.parse(stripped)
1071
+ except RecursionError:
1072
+ raise Prover9ParsingError(_TOO_DEEP) from None
1073
+ except Exception as exc:
1074
+ raise Prover9ParsingError(
1075
+ f"SYNTAX_ERROR: could not parse Prover9 formula: {exc}")
1076
+ try:
1077
+ _resolve_variables(tree, prolog_style, spellings, stripped)
1078
+ _refuse_two_spellings(tree, spellings)
1079
+ _refuse_two_numeral_spellings(tree, spellings)
1080
+ return _transform_tree(tree, transformer)
1081
+ except ParsingError:
1082
+ raise
1083
+ except RecursionError:
1084
+ raise Prover9ParsingError(_TOO_DEEP) from None
1085
+ except Exception as original:
1086
+ raise Prover9ParsingError(f"SYNTAX_ERROR: in Prover9 formula: {original}")
1087
+
1088
+
1089
+ #: The refusal for a formula that the Python recursion limit stops the reader on.
1090
+ _TOO_DEEP = (
1091
+ "SYNTAX_ERROR: the formula is nested too deeply for this reader (reading it reached "
1092
+ "Python's recursion limit). Split the formula, or raise sys.setrecursionlimit.")
1093
+
1094
+
1095
+ def _transform_tree(tree, transformer):
1096
+ """``transformer.transform(tree)`` without recursion in Python.
1097
+
1098
+ The transformer of Lark visits the tree recursively, so a chain of a few hundred
1099
+ operands (``a | b | c | ...`` is a left-nested tree) overflowed the stack of the
1100
+ interpreter. This visits the same nodes in the same order (children before their
1101
+ parent), each node through the method of its name, with an explicit stack; the
1102
+ callbacks are the ones of ``transformer``, and so are the exceptions they raise.
1103
+ """
1104
+ done: dict = {}
1105
+ stack = [(tree, False)]
1106
+ while stack:
1107
+ node, ready = stack.pop()
1108
+ if id(node) in done:
1109
+ continue
1110
+ if not ready:
1111
+ stack.append((node, True))
1112
+ stack.extend((child, False) for child in node.children
1113
+ if isinstance(child, Tree) and id(child) not in done)
1114
+ continue
1115
+ children = [done[id(child)] if isinstance(child, Tree) else child
1116
+ for child in node.children]
1117
+ callback = getattr(transformer, node.data, None)
1118
+ done[id(node)] = callback(children) if callback is not None else Tree(
1119
+ node.data, children, node.meta)
1120
+ return done[id(tree)]
1121
+
1122
+
1123
+ # --- whole-file statement scanner (deterministic; see parse_prover9_problem) ---
1124
+ # A line comment runs from '%' to end of line (LF, CRLF or a bare CR). A statement is a run of text ending
1125
+ # at a '.' that terminates it — i.e. a '.' that is NOT the decimal point of a number
1126
+ # (``.`` immediately followed by a digit, preceded by a digit, stays inside the run).
1127
+ # A double-quoted symbol is skipped whole by both expressions: a '%' inside it starts no comment and a '.' inside
1128
+ # it ends no statement (LADR reads quoted text raw). A quote is a pair; an odd number of them is refused.
1129
+ _P9_COMMENT_RE = re.compile(r'("[^"]*")|%[^\r\n]*')
1130
+ _P9_STATEMENT_RE = re.compile(r'(?:"[^"]*"|[^."])*(?:\.[0-9](?:"[^"]*"|[^."])*)*\.', re.DOTALL)
1131
+ # A top-level directive is ``set``/``clear``/``assign``/``op`` applied with parens.
1132
+ _P9_DIRECTIVES = frozenset({"set", "clear", "assign", "op"})
1133
+ _P9_HEAD_RE = re.compile(r"^([A-Za-z_][A-Za-z0-9_]*)\s*\(")
1134
+ _P9_FORMULAS_RE = re.compile(r"^formulas\s*\(\s*([A-Za-z_][A-Za-z0-9_]*)\s*\)$")
1135
+ _P9_FORMULAS_CALL_RE = re.compile(r"^formulas\s*\(")
1136
+ _P9_VARIABLE_FLAG_RE = re.compile(r"^(set|clear)\s*\(\s*prolog_style_variables\s*\)$")
1137
+
1138
+
1139
+ def _formulas_header_arguments(body: str) -> int:
1140
+ """The number of arguments of ``body`` when it is ONE call ``formulas( ... )`` that ends
1141
+ with the statement, else ``0`` (a call whose parenthesis is never closed counts as ``1``: it is
1142
+ a header that is not well formed).
1143
+
1144
+ A list header has exactly one argument, so INSIDE a list a statement that is a call of
1145
+ ``formulas`` with two or more is an atom (``formulas(alpha, beta)``, which Prover9 reads there
1146
+ like any predicate, and which the writer of this kit writes for a predicate of that name), and
1147
+ so is a statement that goes on after the call (``formulas(a, b) = c``,
1148
+ ``formulas(alpha) & Q``). Outside every list the same call is a header that is not well formed
1149
+ (Prover9 stops at it with "Unrecognized command or list"; :func:`_statement_kind`). Quoted
1150
+ symbols and nested parentheses are skipped when the arguments are counted.
1151
+ """
1152
+ start = _P9_FORMULAS_CALL_RE.match(body)
1153
+ if start is None:
1154
+ return 0
1155
+ depth, arguments, quoted = 0, 1, False
1156
+ for index in range(start.end() - 1, len(body)):
1157
+ character = body[index]
1158
+ if quoted:
1159
+ quoted = character != '"'
1160
+ elif character == '"':
1161
+ quoted = True
1162
+ elif character == "(":
1163
+ depth += 1
1164
+ elif character == ")":
1165
+ depth -= 1
1166
+ if depth == 0:
1167
+ return arguments if index == len(body) - 1 else 0
1168
+ elif character == "," and depth == 1:
1169
+ arguments += 1
1170
+ return 1 # the parenthesis is never closed: a header that is not well formed
1171
+
1172
+
1173
+ def _statement_kind(body: str, in_list: bool) -> str:
1174
+ """What one statement of a file is: ``"directive"`` (``set``, ``clear``, ``assign``,
1175
+ ``op`` outside every list), ``"header"`` (``formulas(NAME)``, the call with ONE argument; also
1176
+ a call with more arguments outside every list, which is a header that is not well formed),
1177
+ ``"end"`` (``end_of_list``) or ``"formula"`` (inside a list ``formulas(alpha, beta)`` is one:
1178
+ an atom)."""
1179
+ head_m = _P9_HEAD_RE.match(body)
1180
+ head = head_m.group(1) if head_m else None
1181
+ if not in_list and head in _P9_DIRECTIVES:
1182
+ return "directive"
1183
+ if head == "formulas":
1184
+ arguments = _formulas_header_arguments(body)
1185
+ if arguments == 1 or (arguments > 1 and not in_list):
1186
+ return "header"
1187
+ if body == "end_of_list":
1188
+ return "end"
1189
+ return "formula"
1190
+
1191
+
1192
+ def _prolog_style_of(statements: list) -> bool:
1193
+ """Whether a file reads its names under ``set(prolog_style_variables)``.
1194
+
1195
+ Measured on Prover9 2026-8A: the LAST ``set(prolog_style_variables)`` or
1196
+ ``clear(prolog_style_variables)`` of the file decides for EVERY formula of it, the ones that
1197
+ come before it included (a flag is read before any formula is interpreted), and a file that
1198
+ never sets it reads Prover9's default. Only a directive outside the lists counts.
1199
+ """
1200
+ prolog, in_list = False, False
1201
+ for body in statements:
1202
+ kind = _statement_kind(body, in_list)
1203
+ if kind == "directive":
1204
+ flag = _P9_VARIABLE_FLAG_RE.match(body)
1205
+ if flag is not None:
1206
+ prolog = flag.group(1) == "set"
1207
+ elif kind == "header":
1208
+ in_list = True
1209
+ elif kind == "end":
1210
+ in_list = False
1211
+ return prolog
1212
+
1213
+
1214
+ def parse_prover9_problem(text: str) -> list:
1215
+ """Parse a whole Prover9 / LADR input file into a list of :class:`Prover9Formula`.
1216
+
1217
+ Reads ``set`` / ``clear`` / ``assign`` directives (recognised and skipped),
1218
+ ``op(precedence, type, symbol)`` directives (recognised and, when the
1219
+ declared operator is genuinely new, applied to every formula parsed after
1220
+ it — see the module docstring's "op(...) declarations" section for exactly
1221
+ what is applied and what is refused), ``formulas(LIST). … end_of_list.``
1222
+ blocks, and bare top-level ``formula.`` statements (``%`` line comments are
1223
+ ignored). Each formula is returned tagged with its list name as ``role``
1224
+ (``""`` for a bare top-level formula), in source order.
1225
+
1226
+ Statements are scanned deterministically (split on the terminating ``.``, with a
1227
+ ``.`` inside a decimal number or inside a double-quoted symbol kept; a ``%`` inside
1228
+ a quoted symbol is no comment), and each formula is parsed by
1229
+ :func:`parse_prover9` under whatever ``op(...)`` declarations are active at that
1230
+ point in the file (a formula using a name before its own ``op(...)`` directive
1231
+ sees it as an ordinary, undeclared name, not as an operator). A ``formulas(...)``
1232
+ header with no matching ``end_of_list.``, a stray ``end_of_list.``, or a
1233
+ malformed header is a hard error — unlike a grammar that could silently
1234
+ reinterpret an unterminated list as bare formulas. A header is the call
1235
+ ``formulas(NAME)`` with ONE argument: inside a list a call with two or more is an atom of the
1236
+ predicate ``formulas`` (``formulas(alpha, beta).``, which Prover9 reads as one and the writer of
1237
+ this kit writes for such a predicate), and so is a statement that goes on after the call;
1238
+ outside every list a call with two or more is a malformed header, as it is for Prover9
1239
+ ("Unrecognized command or list").
1240
+
1241
+ Names are read under the convention the file sets (see the module docstring): the last
1242
+ ``set(prolog_style_variables)`` or ``clear(prolog_style_variables)`` of the file decides for
1243
+ all of its formulas, and a file that never sets it reads Prover9's default, where a name
1244
+ that no quantifier binds is a variable when it begins with ``u`` to ``z``.
1245
+
1246
+ Args:
1247
+ text: the contents of a Prover9 problem file.
1248
+
1249
+ Returns:
1250
+ A list of :class:`Prover9Formula` ``(role, formula)`` records.
1251
+
1252
+ Raises:
1253
+ Prover9ParsingError: if the text is not a well-formed Prover9 problem (within
1254
+ the supported subset; an individual formula is parsed by
1255
+ :func:`parse_prover9`), or an ``op(...)`` directive is malformed or
1256
+ refused.
1257
+ """
1258
+ stripped = _P9_COMMENT_RE.sub(lambda m: m.group(1) or "", text)
1259
+ if stripped.count('"') % 2:
1260
+ raise Prover9ParsingError(
1261
+ "SYNTAX_ERROR: a double quote without its closing double quote "
1262
+ "(a quoted symbol is not closed)")
1263
+ statements = []
1264
+ pos = 0
1265
+ for match in _P9_STATEMENT_RE.finditer(stripped):
1266
+ if match.start() != pos:
1267
+ break # text was skipped, so a statement is open: the leftover check reports it
1268
+ pos = match.end()
1269
+ body = match.group(0).strip()[:-1].strip() # drop the terminating '.'
1270
+ if body:
1271
+ statements.append(body)
1272
+ prolog_style = _prolog_style_of(statements)
1273
+ records = []
1274
+ role = None # the open formulas(...) list name, or None at top level
1275
+ active_ops = () # _CustomOp records declared so far, in order
1276
+ spellings: dict = {} # symbol -> how it was written so far (see _refuse_two_spellings)
1277
+ for body in statements:
1278
+ kind = _statement_kind(body, role is not None)
1279
+ if kind == "directive":
1280
+ directive = _P9_HEAD_RE.match(body)
1281
+ if directive is not None and directive.group(1) == "op":
1282
+ new_ops = _parse_op_directive(body)
1283
+ active_ops = active_ops + tuple(new_ops)
1284
+ _build_custom_grammar(active_ops) # validates now; result is cached
1285
+ continue # top-level flag/op directive: skip
1286
+ if kind == "header":
1287
+ list_m = _P9_FORMULAS_RE.match(body)
1288
+ if not list_m:
1289
+ raise Prover9ParsingError(
1290
+ f"SYNTAX_ERROR: malformed 'formulas(...)' header: {body!r}")
1291
+ if role is not None:
1292
+ raise Prover9ParsingError(
1293
+ "SYNTAX_ERROR: nested 'formulas(...)' list "
1294
+ f"(the '{role}' list was not closed by 'end_of_list.')")
1295
+ role = list_m.group(1)
1296
+ continue
1297
+ if kind == "end":
1298
+ if role is None:
1299
+ raise Prover9ParsingError(
1300
+ "SYNTAX_ERROR: 'end_of_list.' without an open 'formulas(...)' list")
1301
+ role = None
1302
+ continue
1303
+ records.append(Prover9Formula(
1304
+ role or "", _parse_formula(body, active_ops, spellings, prolog_style)))
1305
+ if role is not None:
1306
+ raise Prover9ParsingError(
1307
+ f"SYNTAX_ERROR: 'formulas({role})' list not closed by 'end_of_list.'")
1308
+ leftover = stripped[pos:].strip()
1309
+ if leftover:
1310
+ raise Prover9ParsingError(
1311
+ f"SYNTAX_ERROR: unterminated Prover9 statement (missing '.'): {leftover[:60]!r}")
1312
+ return records
1313
+
1314
+
1315
+ def load_prover9(path: str) -> list:
1316
+ """Read a Prover9 / LADR input file and :func:`parse_prover9_problem` its contents.
1317
+
1318
+ Args:
1319
+ path: path to a Prover9 ``.in`` / ``.p9`` input file.
1320
+
1321
+ Returns:
1322
+ A list of :class:`Prover9Formula` records.
1323
+ """
1324
+ with open(path, "r", encoding="utf-8") as handle:
1325
+ return parse_prover9_problem(handle.read())