unicode-logic-kit 0.31.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- unicode_logic_kit/__init__.py +385 -0
- unicode_logic_kit/__main__.py +520 -0
- unicode_logic_kit/_deadline.py +219 -0
- unicode_logic_kit/ace/__init__.py +126 -0
- unicode_logic_kit/ace/_align.py +135 -0
- unicode_logic_kit/ace/chem_lexicon.py +128 -0
- unicode_logic_kit/ace/drs_reader.py +570 -0
- unicode_logic_kit/ace/mapping.py +666 -0
- unicode_logic_kit/ace/reverse_modal.py +138 -0
- unicode_logic_kit/ace/runner.py +551 -0
- unicode_logic_kit/ace/translate.py +452 -0
- unicode_logic_kit/ace/verbalize.py +1070 -0
- unicode_logic_kit/api.py +1284 -0
- unicode_logic_kit/atp/__init__.py +177 -0
- unicode_logic_kit/atp/_ascii_names.py +113 -0
- unicode_logic_kit/atp/_html.py +72 -0
- unicode_logic_kit/atp/_substructural_input.py +228 -0
- unicode_logic_kit/atp/_tff_problem.py +715 -0
- unicode_logic_kit/atp/_tptp_problem.py +1111 -0
- unicode_logic_kit/atp/_writer_support.py +289 -0
- unicode_logic_kit/atp/clingo_backend.py +1180 -0
- unicode_logic_kit/atp/cvc5_backend.py +1385 -0
- unicode_logic_kit/atp/eprover_backend.py +732 -0
- unicode_logic_kit/atp/finite_domain.py +1055 -0
- unicode_logic_kit/atp/fitch.py +1547 -0
- unicode_logic_kit/atp/fitch_search.py +551 -0
- unicode_logic_kit/atp/hets_backend.py +339 -0
- unicode_logic_kit/atp/hybrid_down.py +120 -0
- unicode_logic_kit/atp/incremental.py +250 -0
- unicode_logic_kit/atp/kripke_enum.py +741 -0
- unicode_logic_kit/atp/lambek.py +436 -0
- unicode_logic_kit/atp/leo3_backend.py +332 -0
- unicode_logic_kit/atp/linear.py +738 -0
- unicode_logic_kit/atp/lj.py +705 -0
- unicode_logic_kit/atp/logic_backends.py +566 -0
- unicode_logic_kit/atp/ltl_tableau.py +1084 -0
- unicode_logic_kit/atp/minizinc_backend.py +1402 -0
- unicode_logic_kit/atp/modal_tableau.py +1382 -0
- unicode_logic_kit/atp/nanocop_backend.py +410 -0
- unicode_logic_kit/atp/portfolio.py +489 -0
- unicode_logic_kit/atp/protocol.py +1803 -0
- unicode_logic_kit/atp/prover9_entailment.py +1153 -0
- unicode_logic_kit/atp/resolution.py +1376 -0
- unicode_logic_kit/atp/resolution_check.py +1114 -0
- unicode_logic_kit/atp/sequent.py +1050 -0
- unicode_logic_kit/atp/tableau.py +921 -0
- unicode_logic_kit/atp/tableau_check.py +543 -0
- unicode_logic_kit/atp/tptp_ncl.py +811 -0
- unicode_logic_kit/atp/tptp_tff.py +1546 -0
- unicode_logic_kit/atp/tstp.py +1333 -0
- unicode_logic_kit/atp/tstp_check.py +1096 -0
- unicode_logic_kit/atp/twee_backend.py +236 -0
- unicode_logic_kit/atp/twee_check.py +711 -0
- unicode_logic_kit/atp/twee_entailment.py +953 -0
- unicode_logic_kit/atp/vampire_entailment.py +540 -0
- unicode_logic_kit/atp/z3_arith.py +470 -0
- unicode_logic_kit/atp/z3_equivalence.py +36 -0
- unicode_logic_kit/atp/z3_fuzzy.py +362 -0
- unicode_logic_kit/atp/z3_input.py +500 -0
- unicode_logic_kit/atp/z3_models.py +208 -0
- unicode_logic_kit/chem/__init__.py +88 -0
- unicode_logic_kit/chem/_naming.py +284 -0
- unicode_logic_kit/chem/cache.py +185 -0
- unicode_logic_kit/chem/interop.py +244 -0
- unicode_logic_kit/chem/mol.py +525 -0
- unicode_logic_kit/chem/signature.py +112 -0
- unicode_logic_kit/comorphism.py +497 -0
- unicode_logic_kit/dl/__init__.py +384 -0
- unicode_logic_kit/dl/classification.py +227 -0
- unicode_logic_kit/dl/concepts.py +632 -0
- unicode_logic_kit/dl/datatypes.py +818 -0
- unicode_logic_kit/dl/owl_functional.py +2433 -0
- unicode_logic_kit/dl/owl_manchester.py +1637 -0
- unicode_logic_kit/dl/owl_reasoner.py +790 -0
- unicode_logic_kit/dl/parser.py +391 -0
- unicode_logic_kit/dl/tableau.py +4048 -0
- unicode_logic_kit/dl/translate.py +2704 -0
- unicode_logic_kit/drt/__init__.py +94 -0
- unicode_logic_kit/drt/export.py +179 -0
- unicode_logic_kit/drt/nodes.py +506 -0
- unicode_logic_kit/drt/parser.py +965 -0
- unicode_logic_kit/drt/resolve.py +195 -0
- unicode_logic_kit/drt/reverse.py +175 -0
- unicode_logic_kit/eval/__init__.py +106 -0
- unicode_logic_kit/eval/batch.py +382 -0
- unicode_logic_kit/eval/canonical.py +663 -0
- unicode_logic_kit/eval/chem_batch.py +606 -0
- unicode_logic_kit/eval/converses.py +200 -0
- unicode_logic_kit/eval/datasets/__init__.py +136 -0
- unicode_logic_kit/eval/datasets/_base.py +263 -0
- unicode_logic_kit/eval/datasets/_proofwriter_proof.py +422 -0
- unicode_logic_kit/eval/datasets/c3po.py +678 -0
- unicode_logic_kit/eval/datasets/folio.py +158 -0
- unicode_logic_kit/eval/datasets/fracas.py +418 -0
- unicode_logic_kit/eval/datasets/groves.py +191 -0
- unicode_logic_kit/eval/datasets/logicbench.py +467 -0
- unicode_logic_kit/eval/datasets/logicnli.py +303 -0
- unicode_logic_kit/eval/datasets/malls.py +133 -0
- unicode_logic_kit/eval/datasets/pfolio.py +594 -0
- unicode_logic_kit/eval/datasets/pmb.py +242 -0
- unicode_logic_kit/eval/datasets/prontoqa.py +611 -0
- unicode_logic_kit/eval/datasets/proofwriter.py +1431 -0
- unicode_logic_kit/eval/datasets/proverqa.py +674 -0
- unicode_logic_kit/eval/datasets/willow.py +478 -0
- unicode_logic_kit/eval/equivalence.py +466 -0
- unicode_logic_kit/eval/exercise_gen.py +533 -0
- unicode_logic_kit/eval/explain.py +791 -0
- unicode_logic_kit/eval/generality.py +750 -0
- unicode_logic_kit/eval/metric_hf.py +458 -0
- unicode_logic_kit/eval/predicate_match.py +343 -0
- unicode_logic_kit/eval/theory_check.py +1170 -0
- unicode_logic_kit/eval/validate.py +306 -0
- unicode_logic_kit/fol/__init__.py +177 -0
- unicode_logic_kit/fol/_atom_keys.py +510 -0
- unicode_logic_kit/fol/_fol_nodes.py +3586 -0
- unicode_logic_kit/fol/_free_parameters.py +105 -0
- unicode_logic_kit/fol/_ho_nodes.py +448 -0
- unicode_logic_kit/fol/_hybrid_nodes.py +308 -0
- unicode_logic_kit/fol/_identifiers.py +1091 -0
- unicode_logic_kit/fol/_lambek_nodes.py +112 -0
- unicode_logic_kit/fol/_linear_nodes.py +352 -0
- unicode_logic_kit/fol/_modal_nodes.py +1467 -0
- unicode_logic_kit/fol/_msfl_nodes.py +2196 -0
- unicode_logic_kit/fol/_numeral_symbols.py +231 -0
- unicode_logic_kit/fol/_so_nodes.py +200 -0
- unicode_logic_kit/fol/_symbol_names.py +81 -0
- unicode_logic_kit/fol/_team_nodes.py +181 -0
- unicode_logic_kit/fol/_tptp_symbols.py +551 -0
- unicode_logic_kit/fol/_truth_constants.py +117 -0
- unicode_logic_kit/fol/casl_export.py +1135 -0
- unicode_logic_kit/fol/casl_import.py +929 -0
- unicode_logic_kit/fol/derivation.py +367 -0
- unicode_logic_kit/fol/dialect_detect.py +70 -0
- unicode_logic_kit/fol/dialect_repair.py +537 -0
- unicode_logic_kit/fol/frames.py +637 -0
- unicode_logic_kit/fol/grammars/terminals.lark +31 -0
- unicode_logic_kit/fol/lambda_tools.py +297 -0
- unicode_logic_kit/fol/latex_input.py +429 -0
- unicode_logic_kit/fol/modal_translation.py +944 -0
- unicode_logic_kit/fol/msflparser.py +1033 -0
- unicode_logic_kit/fol/naming.py +422 -0
- unicode_logic_kit/fol/nodes.py +241 -0
- unicode_logic_kit/fol/normalforms.py +492 -0
- unicode_logic_kit/fol/pal.py +287 -0
- unicode_logic_kit/fol/prolog_export.py +566 -0
- unicode_logic_kit/fol/prolog_input.py +505 -0
- unicode_logic_kit/fol/prover9_input.py +1325 -0
- unicode_logic_kit/fol/qml.py +1760 -0
- unicode_logic_kit/fol/qmltp_input.py +525 -0
- unicode_logic_kit/fol/sanitize.py +221 -0
- unicode_logic_kit/fol/serialize.py +79 -0
- unicode_logic_kit/fol/signature.py +1290 -0
- unicode_logic_kit/fol/simplify_check.py +544 -0
- unicode_logic_kit/fol/spans.py +594 -0
- unicode_logic_kit/fol/tptp_input.py +1503 -0
- unicode_logic_kit/fol/tptp_repair.py +941 -0
- unicode_logic_kit/fol/unification.py +157 -0
- unicode_logic_kit/fol/verbalize.py +263 -0
- unicode_logic_kit/hets/__init__.py +163 -0
- unicode_logic_kit/hets/bridge.py +142 -0
- unicode_logic_kit/hets/client.py +748 -0
- unicode_logic_kit/hets/docker.py +420 -0
- unicode_logic_kit/hets/dol.py +712 -0
- unicode_logic_kit/hets/haskell_json.py +355 -0
- unicode_logic_kit/hets/owl_backend.py +794 -0
- unicode_logic_kit/hets/owl_cli.py +598 -0
- unicode_logic_kit/hets/symbols.py +512 -0
- unicode_logic_kit/hol/__init__.py +140 -0
- unicode_logic_kit/hol/_ho_common.py +323 -0
- unicode_logic_kit/hol/_isabelle_binders.py +125 -0
- unicode_logic_kit/hol/classical.py +812 -0
- unicode_logic_kit/hol/deepshallow/__init__.py +45 -0
- unicode_logic_kit/hol/deepshallow/_common.py +177 -0
- unicode_logic_kit/hol/deepshallow/conditional.py +225 -0
- unicode_logic_kit/hol/deepshallow/intuitionistic.py +181 -0
- unicode_logic_kit/hol/deepshallow/modal.py +217 -0
- unicode_logic_kit/hol/deepshallow/qml.py +406 -0
- unicode_logic_kit/hol/deepshallow/relevant.py +206 -0
- unicode_logic_kit/hol/free.py +753 -0
- unicode_logic_kit/hol/goedel.py +336 -0
- unicode_logic_kit/hol/ho_modal.py +1743 -0
- unicode_logic_kit/hol/intuitionistic.py +403 -0
- unicode_logic_kit/hol/isabelle_conditional.py +593 -0
- unicode_logic_kit/hol/isabelle_modal.py +1908 -0
- unicode_logic_kit/hol/isabelle_relevant.py +412 -0
- unicode_logic_kit/hol/isabelle_runner.py +1147 -0
- unicode_logic_kit/hol/isabelle_substructural.py +884 -0
- unicode_logic_kit/hol/lean.py +1018 -0
- unicode_logic_kit/hol/manyvalued.py +921 -0
- unicode_logic_kit/hol/secondorder.py +687 -0
- unicode_logic_kit/hol/thf_modal.py +941 -0
- unicode_logic_kit/hol/thirdorder.py +397 -0
- unicode_logic_kit/ilp/__init__.py +89 -0
- unicode_logic_kit/ilp/readback.py +389 -0
- unicode_logic_kit/ilp/separation.py +153 -0
- unicode_logic_kit/ilp/task.py +730 -0
- unicode_logic_kit/logic.py +163 -0
- unicode_logic_kit/mcp/__init__.py +28 -0
- unicode_logic_kit/mcp/__main__.py +5 -0
- unicode_logic_kit/mcp/chem_tools.py +1031 -0
- unicode_logic_kit/mcp/server.py +2453 -0
- unicode_logic_kit/mcp/syntax_spec.py +681 -0
- unicode_logic_kit/prob/__init__.py +53 -0
- unicode_logic_kit/prob/_bdd.py +225 -0
- unicode_logic_kit/prob/_column_gen.py +668 -0
- unicode_logic_kit/prob/distribution.py +686 -0
- unicode_logic_kit/prob/nilsson.py +470 -0
- unicode_logic_kit/py.typed +0 -0
- unicode_logic_kit/semantics/__init__.py +137 -0
- unicode_logic_kit/semantics/_modal_reject.py +156 -0
- unicode_logic_kit/semantics/action_models.py +466 -0
- unicode_logic_kit/semantics/asp_models.py +1200 -0
- unicode_logic_kit/semantics/conditional.py +580 -0
- unicode_logic_kit/semantics/dynamic_epistemic.py +95 -0
- unicode_logic_kit/semantics/free_logic.py +913 -0
- unicode_logic_kit/semantics/fuzzy.py +384 -0
- unicode_logic_kit/semantics/fuzzy_kripke.py +442 -0
- unicode_logic_kit/semantics/intuitionistic.py +581 -0
- unicode_logic_kit/semantics/kripke.py +1139 -0
- unicode_logic_kit/semantics/manyvalued.py +580 -0
- unicode_logic_kit/semantics/matrix.py +342 -0
- unicode_logic_kit/semantics/model_eval.py +1135 -0
- unicode_logic_kit/semantics/modelfinder.py +1036 -0
- unicode_logic_kit/semantics/nonmonotonic.py +372 -0
- unicode_logic_kit/semantics/relevant.py +331 -0
- unicode_logic_kit/semantics/secondorder.py +657 -0
- unicode_logic_kit/semantics/structures.py +352 -0
- unicode_logic_kit/semantics/tarski.py +975 -0
- unicode_logic_kit/semantics/team.py +315 -0
- unicode_logic_kit/semantics/team_translation.py +416 -0
- unicode_logic_kit/semantics/thirdorder.py +358 -0
- unicode_logic_kit/semantics/tnorm.py +85 -0
- unicode_logic_kit/semantics/truthtable.py +201 -0
- unicode_logic_kit-0.31.0.dist-info/METADATA +333 -0
- unicode_logic_kit-0.31.0.dist-info/RECORD +237 -0
- unicode_logic_kit-0.31.0.dist-info/WHEEL +4 -0
- unicode_logic_kit-0.31.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,1325 @@
|
|
|
1
|
+
"""Prover9 / LADR input: read Prover9-syntax formulas into the AST.
|
|
2
|
+
|
|
3
|
+
This is the inverse of :meth:`Node.to_prover9`. Prover9's surface syntax differs
|
|
4
|
+
from the toolkit's Unicode notation (``all X``/``exists X`` quantifiers, ``-`` for
|
|
5
|
+
negation, ``& | -> <->`` connectives, infix comparison predicates), so it gets
|
|
6
|
+
its own Lark grammar.
|
|
7
|
+
|
|
8
|
+
Variables. :meth:`Node.to_prover9` writes under ``set(prolog_style_variables)``, and this
|
|
9
|
+
reader reads names by Prover9's own rules (measured on Prover9 2026-8A):
|
|
10
|
+
|
|
11
|
+
* ``all x`` and ``exists x`` bind the SYMBOL they name, whatever its case. In
|
|
12
|
+
``all x (man(x) -> mortal(x))`` and in ``all X (man(X) -> mortal(X))`` the three occurrences
|
|
13
|
+
are one :class:`Variable`, for as long as the operand that follows the variable lasts (it
|
|
14
|
+
ends before a ``&``, ``|``, ``->``, ``<->`` or ``<-``). A quantifier over the same spelling
|
|
15
|
+
inside that operand rebinds it. A name that is applied to arguments (``x(a)``) is a
|
|
16
|
+
predicate or a function, not an occurrence of the variable.
|
|
17
|
+
* A name that no quantifier binds is a variable or a constant by the convention of the text.
|
|
18
|
+
Under ``set(prolog_style_variables)`` a name that begins with an upper-case letter
|
|
19
|
+
(``A`` to ``Z``; an underscore does not count) is a variable and every other name a
|
|
20
|
+
constant; without that flag a name that begins with ``u`` to ``z`` is a variable.
|
|
21
|
+
:func:`parse_prover9_problem` reads the flag as Prover9 does: the LAST ``set`` or ``clear``
|
|
22
|
+
of ``prolog_style_variables`` in the file decides for every formula of it, a file without
|
|
23
|
+
one reads Prover9's default. :func:`parse_prover9` has no file around the formula and keeps
|
|
24
|
+
``set(prolog_style_variables)`` (or pass ``prolog_style_variables=False``).
|
|
25
|
+
* Variables are compared as written: ``Xa`` and ``XA`` are two variables, and so are ``x``
|
|
26
|
+
and ``X``. The name of a :class:`Variable` is the lower-case of its spelling (the inverse of
|
|
27
|
+
:meth:`Variable.to_prover9`, which upper-cases), and a spelling whose lower-case another
|
|
28
|
+
spelling already has gets a fresh name (``x0``): two spellings are never one variable.
|
|
29
|
+
Constants, predicates and functions keep their case.
|
|
30
|
+
|
|
31
|
+
A name applied to arguments is a predicate (in formula position) or function (in term
|
|
32
|
+
position) and keeps its case; a bare name in FORMULA position is a nullary (propositional)
|
|
33
|
+
predicate (Prover9 itself refuses a bare upper-case proposition, which it reads as a variable,
|
|
34
|
+
and a bound name used as a formula; this reader is more lenient about the first and refuses
|
|
35
|
+
the second). Comparison operators map back to the ``=`` / ``≠`` /
|
|
36
|
+
``<`` / ``>`` / ``≤`` / ``≥`` atoms and ``+ - * /`` to the arithmetic functions.
|
|
37
|
+
|
|
38
|
+
Connectives: ``->``, ``<->`` and the reverse implication ``<-`` (``p <- q`` is ``q -> p``)
|
|
39
|
+
bind looser than ``|`` and ``&``. Prover9 reads the three as non-associative: a chain or a mix
|
|
40
|
+
of two of them without parentheses (``a <- b <- c``, ``a <- b -> c``) is a syntax error there
|
|
41
|
+
(measured), and this reader refuses a ``<-`` in such a chain as well. It still reads
|
|
42
|
+
``a -> b -> c`` as ``a -> (b -> c)``, which Prover9 refuses. ``<`` is the comparison only where
|
|
43
|
+
no ``-`` follows it directly: ``a <-b`` is a reverse implication, ``a < -b`` the comparison
|
|
44
|
+
of ``a`` and ``-b``.
|
|
45
|
+
|
|
46
|
+
Keywords: ``all`` and ``exists`` are quantifiers only as words of their own, followed by a
|
|
47
|
+
variable. ``allowed(a)``, ``exists_in(b)`` and ``allergic(a)`` are atoms, and so is ``all(a)``,
|
|
48
|
+
as in Prover9.
|
|
49
|
+
|
|
50
|
+
Prover9's own constants are read too: ``$T`` and ``$F`` in formula position are the
|
|
51
|
+
truth constants, the nullary atoms ``$true`` and ``$false`` (the atoms
|
|
52
|
+
:meth:`Node.to_prover9` writes as ``$T`` and ``$F``, so ``parse_prover9(node.to_prover9())
|
|
53
|
+
== node`` for them). They are formulas, not terms: ``P($T)`` is refused.
|
|
54
|
+
|
|
55
|
+
Quoted symbols: a double-quoted symbol is never a variable in Prover9 (LADR stores it
|
|
56
|
+
with its quote characters, so the first character it tests is the quote), which is how
|
|
57
|
+
:meth:`Node.to_prover9` writes a constant or a proposition whose name begins with an
|
|
58
|
+
upper-case letter or an underscore (``P("Gaseous")``, ``"Rain"``). This reader reads
|
|
59
|
+
it back as a name that is never a variable: ``"Rain"`` in formula position is the
|
|
60
|
+
nullary atom ``Rain``, ``"Gaseous"`` in term position the :class:`Constant`
|
|
61
|
+
``Gaseous``, and a quoted name applied to arguments (``"Mother"(a)``) an atom or a
|
|
62
|
+
:class:`Function` of that name. Prover9 keeps ``"rain"`` and ``rain`` apart as two
|
|
63
|
+
symbols and this reader does not (the AST has one name), so a text that writes the
|
|
64
|
+
SAME symbol (the same predicate, function, constant or number, at one arity) both
|
|
65
|
+
with and without quotes is refused by name as a :class:`Prover9ParsingError`, rather
|
|
66
|
+
than read as one symbol: a file is one text, and the record of what was met is kept
|
|
67
|
+
across all its formulas. (``"Rain"`` as a proposition next to ``Rain(x)`` as a
|
|
68
|
+
predicate is not that: they are two symbols here too.) Only a quoted name made of
|
|
69
|
+
letters, digits and an underscore (``[A-Za-z_][A-Za-z0-9_]*``) is read, because that
|
|
70
|
+
is what the writer produces and the only shape :meth:`Node.to_prover9` can write
|
|
71
|
+
back, and a quoted NUMERAL (``"2.5"``, ``"-1"``) in term position, which is the
|
|
72
|
+
:class:`Number` of that value (the text must be the canonical spelling of the number,
|
|
73
|
+
as :meth:`Number.to_prover9` writes it, one per value: ``"2.50"`` and ``"1.0"`` are
|
|
74
|
+
refused, because Prover9 keeps them apart from ``"2.5"`` and ``"1"``, which are the same
|
|
75
|
+
numbers); any other quoted text
|
|
76
|
+
is refused by name as a :class:`Prover9ParsingError`. The file scanner of
|
|
77
|
+
:func:`parse_prover9_problem` skips a quoted symbol whole: a ``.`` or a ``%`` inside it
|
|
78
|
+
ends no statement and starts no comment.
|
|
79
|
+
|
|
80
|
+
A term can be written ``-(a, b)``: the function ``-`` of two arguments, which is how
|
|
81
|
+
:meth:`Function.to_prover9` writes a binary minus (Prover9 has no infix minus, ``(a - b)``
|
|
82
|
+
is a syntax error there; the infix ``a - b`` of this reader's grammar is kept for the
|
|
83
|
+
files it has always read). A prefix minus is the function ``-`` of one argument, and binds
|
|
84
|
+
as Prover9's does (priority 350, tighter than the comparisons and the sums): ``-(a)``,
|
|
85
|
+
``-a`` and ``- f(a)`` are terms, ``-a = b`` is the equation of the terms ``-a`` and ``b``
|
|
86
|
+
(Prover9 echoes ``-a = b.``), ``-a + b = c`` is ``(-a) + b = c``, and only a minus in front
|
|
87
|
+
of a formula that is not a comparison, or of a parenthesised formula, is a negation
|
|
88
|
+
(``-P(a)``, ``-(a = b)``, which Prover9 echoes ``a != b``). A bare ``-1`` is read as the
|
|
89
|
+
number minus one, also in front of a comparison (``-1 < x``), although Prover9 reads it as
|
|
90
|
+
``-`` applied to the constant ``1``: the writer of this kit writes the number as ``"-1"``.
|
|
91
|
+
|
|
92
|
+
A text that spells one numeral two ways is refused by name: Prover9 keeps ``01`` and ``1``,
|
|
93
|
+
``1.0`` and ``1``, ``2.50`` and ``2.5`` apart as symbols, a :class:`Number` has one text per
|
|
94
|
+
value, and the file ``P(01). -P(1).`` (consistent for Prover9) would be read as ``P(1)`` and
|
|
95
|
+
``¬P(1)``. Every formula of a file counts for it, as for a symbol written both quoted and bare.
|
|
96
|
+
A numeral that stands alone in a text is read as the number it spells.
|
|
97
|
+
|
|
98
|
+
Depth: the parse tree is transformed with an explicit stack, so a chain of operands or a stack
|
|
99
|
+
of negations or quantifiers is read at any depth the parser itself reads. A formula that
|
|
100
|
+
still exhausts the interpreter's recursion limit is a :class:`Prover9ParsingError`, never a
|
|
101
|
+
bare ``RecursionError``.
|
|
102
|
+
|
|
103
|
+
A free variable: Prover9 closes each formula of a file universally, so ``P(X)`` there says
|
|
104
|
+
``∀X P(X)``. This reader reads the text and does not close it: the variable stays free in the
|
|
105
|
+
AST, and the kit's provers read a free variable of a problem as ONE unknown element, the same
|
|
106
|
+
in every formula (a parameter). A problem built from such a file therefore asks another
|
|
107
|
+
question than Prover9 does for the file; wrap the formula in ``∀`` first when the closure is
|
|
108
|
+
what is meant.
|
|
109
|
+
|
|
110
|
+
Note: :meth:`Node.to_prover9` desugars exclusive-or to ``(a | b) & -(a & b)`` (Prover9
|
|
111
|
+
has no xor operator), so an :class:`Xor` round-trips to that conjunctive form, not to
|
|
112
|
+
``Xor``.
|
|
113
|
+
|
|
114
|
+
op(...) declarations: :func:`parse_prover9_problem` applies a genuinely NEW
|
|
115
|
+
``op(precedence, type, symbol)`` directive (Prover9's own syntax for declaring an
|
|
116
|
+
operator; the manual's page on parsing declarations —
|
|
117
|
+
https://www.cs.unm.edu/~mccune/prover9/manual/2009-11A/syntax.html, mirroring
|
|
118
|
+
``ladr/parse.c``'s ``declare_standard_parse_types()`` in the Prover9/LADR source —
|
|
119
|
+
is the citation for every precedence number and type keyword below) to every
|
|
120
|
+
formula that follows it in the same file. Two things are refused outright, by
|
|
121
|
+
name, as a :class:`Prover9ParsingError`, regardless of whether the declared
|
|
122
|
+
operator is ever used — because applying them could silently change the
|
|
123
|
+
meaning of other, unrelated text already in the file:
|
|
124
|
+
|
|
125
|
+
- **Redeclaring a built-in** (any symbol already in :data:`_DEFAULT_OPS`, e.g.
|
|
126
|
+
``op(500, infix, "+")``) is refused: doing this properly would mean making the
|
|
127
|
+
*entire* grammar table-driven, a much larger and riskier change this reader does
|
|
128
|
+
not make (see ``tests/test_prover9_ops.py`` and
|
|
129
|
+
``tests/test_prover9_entailment.py``'s module docstrings for how the reader is
|
|
130
|
+
checked against an independent route and, where a binary exists, against a real
|
|
131
|
+
Prover9).
|
|
132
|
+
- **Redeclaring a symbol this same file already declared** is refused the same way.
|
|
133
|
+
|
|
134
|
+
A **malformed** ``op(...)`` (wrong arity, a non-integer precedence, an unknown
|
|
135
|
+
type keyword, a symbol that is not a bare or double-quoted identifier, or a
|
|
136
|
+
precedence Prover9 itself would not accept — outside 1-998) is also refused by
|
|
137
|
+
name; these are syntax problems in the directive itself, independent of placement.
|
|
138
|
+
|
|
139
|
+
Everything else syntactically well-formed is *accepted* (matching real Prover9,
|
|
140
|
+
which does not reject any of it either) and applied where this reader has a
|
|
141
|
+
splice point for it, left harmlessly **inert** otherwise — see
|
|
142
|
+
:func:`_classify_and_splice` for exactly which placements splice and which are
|
|
143
|
+
inert (``type="ordinary"``; a precedence tying a built-in tier; a precedence at
|
|
144
|
+
or above the quantifier tier 750; prefix/postfix at an atom-tier precedence). An
|
|
145
|
+
inert declaration still blocks a later redeclaration of the same name, but a
|
|
146
|
+
formula that tries to *use* it as an operator sees an ordinary undeclared name
|
|
147
|
+
and fails to parse — loudly, just at that point rather than at the declaration
|
|
148
|
+
(this mirrors every op() directive's behaviour before this feature existed, for
|
|
149
|
+
exactly the declarations this reader still cannot safely place).
|
|
150
|
+
|
|
151
|
+
What DOES splice in: a new symbol at a **term-tier** precedence (< 500) becomes a
|
|
152
|
+
:class:`Function`-producing operator spliced next to ``unit_term`` (so its
|
|
153
|
+
operands are atomic terms — parenthesize an arithmetic sub-expression used as an
|
|
154
|
+
operand); a new symbol at an **atom-tier** precedence (501-699 or 701-749) becomes
|
|
155
|
+
an :class:`Atom`-producing operator spliced next to the existing comparisons, with
|
|
156
|
+
:class:`Node` operands taken from the ``term`` grammar. ``infix`` (Prover9's
|
|
157
|
+
``xfx``, non-associative) chains only inside explicit parentheses — exactly as in
|
|
158
|
+
Prover9 itself; ``infix_left``/``infix_right`` (``yfx``/``xfy``) chain without
|
|
159
|
+
parentheses in that direction. ``prefix``/``prefix_paren`` and
|
|
160
|
+
``postfix``/``postfix_paren`` splice in at the term tier only; this reader does
|
|
161
|
+
not distinguish ``fy`` from ``fx`` self-chaining (both may chain without
|
|
162
|
+
parentheses) — a narrower distinction real Prover9 makes and this one does not
|
|
163
|
+
need for any formula it still *accepts* (it only widens what parses, never what
|
|
164
|
+
a given input means). Only the 3-argument ``op(...)`` form is read; Prover9's
|
|
165
|
+
2-argument ``op(type, symbol)`` shorthand (valid only for ``type="ordinary"``)
|
|
166
|
+
is a malformed-arity refusal here.
|
|
167
|
+
|
|
168
|
+
Public API: :func:`parse_prover9` (a single formula; a trailing ``.`` is accepted).
|
|
169
|
+
"""
|
|
170
|
+
|
|
171
|
+
import re
|
|
172
|
+
from dataclasses import dataclass
|
|
173
|
+
from functools import lru_cache
|
|
174
|
+
|
|
175
|
+
from lark import Lark, Transformer, Tree
|
|
176
|
+
|
|
177
|
+
from ._fol_nodes import NumeralTextError, _numeral_from_text
|
|
178
|
+
from ._identifiers import fresh_variable_like
|
|
179
|
+
from ._numeral_symbols import numeral_name
|
|
180
|
+
from .nodes import (
|
|
181
|
+
Node, Variable, Constant, Number, Function,
|
|
182
|
+
Atom, Not, And, Or, Implies, Iff, Quantifier,
|
|
183
|
+
)
|
|
184
|
+
from .naming import ParsingError
|
|
185
|
+
|
|
186
|
+
|
|
187
|
+
class Prover9ParsingError(ParsingError):
|
|
188
|
+
"""A Prover9 import failure carrying a plain message (subclasses ParsingError)."""
|
|
189
|
+
|
|
190
|
+
def __init__(self, message: str):
|
|
191
|
+
self.args = (message,)
|
|
192
|
+
|
|
193
|
+
def __str__(self):
|
|
194
|
+
return self.args[0]
|
|
195
|
+
|
|
196
|
+
|
|
197
|
+
def _where(token) -> str:
|
|
198
|
+
"""Where ``token`` stands in the text that was parsed, for a message (``""`` when unknown)."""
|
|
199
|
+
line, column = getattr(token, "line", None), getattr(token, "column", None)
|
|
200
|
+
if line is None or column is None:
|
|
201
|
+
return ""
|
|
202
|
+
return f" (at line {line}, column {column} of the formula)"
|
|
203
|
+
|
|
204
|
+
|
|
205
|
+
def _numeral(text: str, token=None):
|
|
206
|
+
"""The value of the decimal numeral ``text``, read exactly (see
|
|
207
|
+
:func:`~unicode_logic_kit.fol._fol_nodes._numeral_from_text`); a
|
|
208
|
+
:class:`Prover9ParsingError` when no number holds it, which names the position of
|
|
209
|
+
``token`` (the token the numeral was read from) when one is given. Whatever the numeral
|
|
210
|
+
reader refuses with (a ``ValueError`` for a text of thousands of digits, an ``OverflowError``,
|
|
211
|
+
its own :class:`~unicode_logic_kit.fol._fol_nodes.NumeralTextError`) is that error here."""
|
|
212
|
+
try:
|
|
213
|
+
return _numeral_from_text(text)
|
|
214
|
+
except (ValueError, ArithmeticError, NumeralTextError) as exc:
|
|
215
|
+
message = str(exc)
|
|
216
|
+
if message.startswith("SYNTAX_ERROR: "):
|
|
217
|
+
message = message[len("SYNTAX_ERROR: "):]
|
|
218
|
+
raise Prover9ParsingError(f"SYNTAX_ERROR: {message}{_where(token)}") from None
|
|
219
|
+
|
|
220
|
+
|
|
221
|
+
# The formula sub-grammar, shared by the single-formula parser and the whole-file
|
|
222
|
+
# parser. Deliberately kept free of the file-level keyword terminals
|
|
223
|
+
# (set/clear/assign/formulas/end_of_list): in single-formula mode a predicate that
|
|
224
|
+
# happens to be named e.g. ``set`` still lexes as a NAME, preserving backward
|
|
225
|
+
# compatibility for :func:`parse_prover9`.
|
|
226
|
+
#
|
|
227
|
+
# ``{atom_extra}`` / ``{term_extra}`` / ``{unit_term_extra}`` are splice points for
|
|
228
|
+
# newly op()-declared operators (see :func:`_build_custom_grammar`); with all three
|
|
229
|
+
# empty (the default, and the common case — a file with no custom operators) this
|
|
230
|
+
# formats to byte-identical text to the fixed grammar this module has always used,
|
|
231
|
+
# so the shared singleton parser below is neither rebuilt nor slowed down.
|
|
232
|
+
_FORMULA_RULES_TEMPLATE = r"""
|
|
233
|
+
?formula: equiv
|
|
234
|
+
?equiv: imp
|
|
235
|
+
| imp "<->" imp -> iff_
|
|
236
|
+
| disj "<-" disj -> rimplies_
|
|
237
|
+
?imp: disj
|
|
238
|
+
| disj "->" imp -> implies_
|
|
239
|
+
?disj: conj
|
|
240
|
+
| disj "|" conj -> or_
|
|
241
|
+
?conj: unary
|
|
242
|
+
| conj "&" unary -> and_
|
|
243
|
+
?unary: "-" negated -> neg
|
|
244
|
+
| _ALL NAME unary -> forall
|
|
245
|
+
| _EXISTS NAME unary -> exists
|
|
246
|
+
| "(" formula ")"
|
|
247
|
+
| atom
|
|
248
|
+
|
|
249
|
+
?negated: "-" negated -> neg
|
|
250
|
+
| _ALL NAME unary -> forall
|
|
251
|
+
| _EXISTS NAME unary -> exists
|
|
252
|
+
| "(" formula ")"
|
|
253
|
+
| plain_atom
|
|
254
|
+
|
|
255
|
+
?atom: comparison
|
|
256
|
+
| plain_atom
|
|
257
|
+
|
|
258
|
+
?comparison: term "=" term -> equality
|
|
259
|
+
| term "!=" term -> disequality
|
|
260
|
+
| term "<=" term -> le
|
|
261
|
+
| term ">=" term -> ge
|
|
262
|
+
| term _LT term -> lt
|
|
263
|
+
| term ">" term -> gt{atom_extra}
|
|
264
|
+
|
|
265
|
+
?plain_atom: NAME "(" termlist ")" -> pred_app
|
|
266
|
+
| QNAME "(" termlist ")" -> qpred_app
|
|
267
|
+
| NAME -> prop_atom
|
|
268
|
+
| TRUTH -> truth_atom
|
|
269
|
+
| QNAME -> qprop_atom
|
|
270
|
+
|
|
271
|
+
?term: sum{term_extra}
|
|
272
|
+
?sum: product
|
|
273
|
+
| sum "+" product -> add
|
|
274
|
+
| sum "-" product -> sub
|
|
275
|
+
?product: unit_term
|
|
276
|
+
| product "*" unit_term -> mul
|
|
277
|
+
| product "/" unit_term -> div
|
|
278
|
+
?unit_term: signed_term
|
|
279
|
+
| NUMBER -> number{unit_term_extra}
|
|
280
|
+
?signed_term: NAME "(" termlist ")" -> func_app
|
|
281
|
+
| QNAME "(" termlist ")" -> qfunc_app
|
|
282
|
+
| "-" "(" term "," termlist ")" -> minus_app
|
|
283
|
+
| NAME -> name_term
|
|
284
|
+
| QNAME -> qname_term
|
|
285
|
+
| "(" term ")"
|
|
286
|
+
| "-" signed_term -> uminus
|
|
287
|
+
|
|
288
|
+
termlist: term ("," term)*
|
|
289
|
+
|
|
290
|
+
NAME: /[A-Za-z_][A-Za-z0-9_]*/
|
|
291
|
+
_ALL: /all(?![A-Za-z0-9_])/
|
|
292
|
+
_EXISTS: /exists(?![A-Za-z0-9_])/
|
|
293
|
+
_LT: /<(?!-)/
|
|
294
|
+
QNAME: /"[^"]*"/
|
|
295
|
+
TRUTH: /\$[TF](?![A-Za-z0-9_])/
|
|
296
|
+
NUMBER: /-?[0-9]+(\.[0-9]+)?/
|
|
297
|
+
|
|
298
|
+
%import common.WS
|
|
299
|
+
%ignore WS
|
|
300
|
+
%ignore /%[^\r\n]*/
|
|
301
|
+
"""
|
|
302
|
+
|
|
303
|
+
_FORMULA_RULES = _FORMULA_RULES_TEMPLATE.format(atom_extra="", term_extra="", unit_term_extra="")
|
|
304
|
+
_GRAMMAR = "?start: formula\n\n" + _FORMULA_RULES
|
|
305
|
+
|
|
306
|
+
|
|
307
|
+
@dataclass(frozen=True)
|
|
308
|
+
class Prover9Formula:
|
|
309
|
+
"""One formula read from a Prover9 file: its ``role`` and parsed ``formula``.
|
|
310
|
+
|
|
311
|
+
``role`` is the enclosing ``formulas(<name>)`` list name (``"sos"`` /
|
|
312
|
+
``"assumptions"`` / ``"goals"`` / …), or ``""`` for a bare top-level formula.
|
|
313
|
+
"""
|
|
314
|
+
|
|
315
|
+
role: str
|
|
316
|
+
formula: Node
|
|
317
|
+
|
|
318
|
+
|
|
319
|
+
# --- op(...) operator declarations (see the module docstring) ---
|
|
320
|
+
|
|
321
|
+
@dataclass(frozen=True)
|
|
322
|
+
class _CustomOp:
|
|
323
|
+
"""One parsed ``op(precedence, type, symbol)`` record."""
|
|
324
|
+
|
|
325
|
+
precedence: int
|
|
326
|
+
type: str
|
|
327
|
+
symbol: str
|
|
328
|
+
|
|
329
|
+
|
|
330
|
+
# Prover9's own default operator table, from ``ladr/parse.c``'s
|
|
331
|
+
# ``declare_standard_parse_types()`` in the Prover9/LADR source (mirrored by the
|
|
332
|
+
# manual's "Clauses and Formulas" / parsing-declarations page — see the module
|
|
333
|
+
# docstring for the URL). Each entry is (precedence, type, symbol); type uses
|
|
334
|
+
# Prover9's own op()-command keywords, not Prolog's xfx/xfy/yfx names (the manual
|
|
335
|
+
# documents the correspondence: infix=xfx, infix_left=yfx, infix_right=xfy,
|
|
336
|
+
# prefix=fy, prefix_paren=fx, postfix=yf, postfix_paren=xf). Only the subset this
|
|
337
|
+
# reader's own grammar implements matters for parsing; the FULL real table is kept
|
|
338
|
+
# here anyway so that redeclaring any genuine Prover9 built-in — even one this
|
|
339
|
+
# reader's grammar never parses on its own, like "^" or "#" — is refused by name.
|
|
340
|
+
_DEFAULT_OPS = (
|
|
341
|
+
(810, "infix_right", "#"),
|
|
342
|
+
(800, "infix", "<->"),
|
|
343
|
+
(800, "infix", "->"),
|
|
344
|
+
(800, "infix", "<-"),
|
|
345
|
+
(790, "infix_right", "|"),
|
|
346
|
+
(780, "infix_right", "&"),
|
|
347
|
+
(700, "infix", "="),
|
|
348
|
+
(700, "infix", "!="),
|
|
349
|
+
(700, "infix", "=="),
|
|
350
|
+
(700, "infix", "<"),
|
|
351
|
+
(700, "infix", "<="),
|
|
352
|
+
(700, "infix", ">"),
|
|
353
|
+
(700, "infix", ">="),
|
|
354
|
+
(500, "infix", "+"),
|
|
355
|
+
(500, "infix", "*"),
|
|
356
|
+
(500, "infix", "@"),
|
|
357
|
+
(500, "infix", "/"),
|
|
358
|
+
(500, "infix", "\\"),
|
|
359
|
+
(500, "infix", "^"),
|
|
360
|
+
(500, "infix", "v"),
|
|
361
|
+
(350, "prefix", "-"),
|
|
362
|
+
(300, "postfix", "'"),
|
|
363
|
+
)
|
|
364
|
+
_DEFAULT_OP_SYMBOLS = frozenset(sym for _prec, _type, sym in _DEFAULT_OPS)
|
|
365
|
+
# "all"/"exists" are hard-coded grammar keywords rather than entries in
|
|
366
|
+
# _DEFAULT_OPS (they are not ordinary binary/unary operators — quantifiers bind a
|
|
367
|
+
# variable), but redeclaring them by name is refused the same way.
|
|
368
|
+
_RESERVED_SYMBOLS = _DEFAULT_OP_SYMBOLS | {"all", "exists"}
|
|
369
|
+
|
|
370
|
+
# Precedences that coincide with a built-in tier (every distinct precedence in
|
|
371
|
+
# _DEFAULT_OPS — 300, 350, 500, 700 — plus the quantifiers' 750 and the
|
|
372
|
+
# argument-list comma's 999, both declared standalone in
|
|
373
|
+
# declare_standard_parse_types() rather than through _DEFAULT_OPS's op()-style
|
|
374
|
+
# entries; 780/790/800/810 are already covered by _DEFAULT_OPS's own
|
|
375
|
+
# precedences). A NEW operator declared at exactly one of these ties an
|
|
376
|
+
# existing tier: this reader's placement rule (below 500 is a term operator,
|
|
377
|
+
# 500-750 an atom operator) cannot tell which side of the tie it belongs on —
|
|
378
|
+
# see _classify_and_splice for what happens then (left inert, not refused; a
|
|
379
|
+
# declaration this reader cannot safely place is simply never applied, exactly
|
|
380
|
+
# like every op() directive before this feature existed).
|
|
381
|
+
_RESERVED_PRECEDENCES = frozenset({300, 350, 500, 700, 750, 780, 790, 800, 810, 999})
|
|
382
|
+
|
|
383
|
+
_P9_OP_TYPE_NAMES = (
|
|
384
|
+
"infix", "infix_left", "infix_right",
|
|
385
|
+
"prefix", "prefix_paren", "postfix", "postfix_paren", "ordinary",
|
|
386
|
+
)
|
|
387
|
+
_P9_OP_BODY_RE = re.compile(r"^op\((?P<inner>.*)\)$", re.DOTALL)
|
|
388
|
+
_P9_INT_RE = re.compile(r"^-?[0-9]+$")
|
|
389
|
+
_P9_QUOTED_SYMBOL_RE = re.compile(r'^"([^"\\]*)"$')
|
|
390
|
+
_P9_BARE_SYMBOL_RE = re.compile(r"^[A-Za-z_][A-Za-z0-9_]*$")
|
|
391
|
+
|
|
392
|
+
|
|
393
|
+
def _split_top_level_commas(text: str) -> list:
|
|
394
|
+
"""Split ``text`` on commas that are not nested inside ``[...]`` or ``"..."``.
|
|
395
|
+
|
|
396
|
+
Used both for the outer ``op(precedence, type, symbol_or_list)`` arguments
|
|
397
|
+
and for the inner ``[sym1, sym2, ...]`` symbol list.
|
|
398
|
+
"""
|
|
399
|
+
parts = []
|
|
400
|
+
depth = 0
|
|
401
|
+
in_quotes = False
|
|
402
|
+
current = []
|
|
403
|
+
for ch in text:
|
|
404
|
+
if in_quotes:
|
|
405
|
+
current.append(ch)
|
|
406
|
+
if ch == '"':
|
|
407
|
+
in_quotes = False
|
|
408
|
+
continue
|
|
409
|
+
if ch == '"':
|
|
410
|
+
in_quotes = True
|
|
411
|
+
current.append(ch)
|
|
412
|
+
elif ch == "[":
|
|
413
|
+
depth += 1
|
|
414
|
+
current.append(ch)
|
|
415
|
+
elif ch == "]":
|
|
416
|
+
depth -= 1
|
|
417
|
+
current.append(ch)
|
|
418
|
+
elif ch == "," and depth == 0:
|
|
419
|
+
parts.append("".join(current))
|
|
420
|
+
current = []
|
|
421
|
+
else:
|
|
422
|
+
current.append(ch)
|
|
423
|
+
parts.append("".join(current))
|
|
424
|
+
return [p.strip() for p in parts]
|
|
425
|
+
|
|
426
|
+
|
|
427
|
+
def _parse_op_symbol(token: str) -> str:
|
|
428
|
+
"""Read one op() symbol token: a bare identifier or a double-quoted one.
|
|
429
|
+
|
|
430
|
+
Only symbols shaped like the grammar's own NAME terminal
|
|
431
|
+
(``[A-Za-z_][A-Za-z0-9_]*``) are supported — a symbolic token such as
|
|
432
|
+
``"++"`` cannot be spliced into the grammar as a new keyword-like literal
|
|
433
|
+
the way an identifier can (see the module docstring), so it is refused.
|
|
434
|
+
A symbolic token that names an existing built-in (``"+"``, ``"&"``, …) is
|
|
435
|
+
let through here regardless of shape, so :func:`_classify_and_splice` can
|
|
436
|
+
give the specific "redeclaring built-in" error instead of this generic one.
|
|
437
|
+
"""
|
|
438
|
+
quoted = _P9_QUOTED_SYMBOL_RE.match(token)
|
|
439
|
+
name = quoted.group(1) if quoted else token
|
|
440
|
+
if name in _RESERVED_SYMBOLS or _P9_BARE_SYMBOL_RE.match(name):
|
|
441
|
+
return name
|
|
442
|
+
raise Prover9ParsingError(
|
|
443
|
+
f"SYNTAX_ERROR: op() symbol {token!r} is not supported by this reader "
|
|
444
|
+
"(only alphanumeric/underscore operator names are; a symbolic token "
|
|
445
|
+
"cannot be declared or redeclared here)")
|
|
446
|
+
|
|
447
|
+
|
|
448
|
+
def _parse_op_directive(body: str) -> list:
|
|
449
|
+
"""Parse one ``op(precedence, type, symbol)`` directive body (``body`` is the
|
|
450
|
+
whole statement text, e.g. ``op(650, infix, "before")``, sans the trailing
|
|
451
|
+
``.``) into a list of :class:`_CustomOp` records — one per symbol, all
|
|
452
|
+
sharing ``precedence``/``type``; ``op(N, TYPE, [s1, s2])`` yields two records.
|
|
453
|
+
|
|
454
|
+
Only the 3-argument form is supported — see the module docstring.
|
|
455
|
+
|
|
456
|
+
Raises:
|
|
457
|
+
Prover9ParsingError: on any malformed ``op(...)`` (bad arity, a
|
|
458
|
+
non-integer precedence, an unknown type keyword, or a symbol token
|
|
459
|
+
this reader does not support).
|
|
460
|
+
"""
|
|
461
|
+
match = _P9_OP_BODY_RE.match(body)
|
|
462
|
+
if not match:
|
|
463
|
+
raise Prover9ParsingError(f"SYNTAX_ERROR: malformed op(...) directive: {body!r}")
|
|
464
|
+
args = _split_top_level_commas(match.group("inner"))
|
|
465
|
+
if len(args) != 3:
|
|
466
|
+
raise Prover9ParsingError(
|
|
467
|
+
"SYNTAX_ERROR: op(...) needs exactly 3 arguments (precedence, type, "
|
|
468
|
+
f"symbol[s]) — Prover9's 2-argument ordinary-only form is not "
|
|
469
|
+
f"supported by this reader; got {len(args)}: {body!r}")
|
|
470
|
+
prec_text, type_text, symbols_text = args
|
|
471
|
+
if not _P9_INT_RE.match(prec_text):
|
|
472
|
+
raise Prover9ParsingError(
|
|
473
|
+
f"SYNTAX_ERROR: op() precedence must be an integer, got {prec_text!r}")
|
|
474
|
+
try:
|
|
475
|
+
precedence = int(prec_text)
|
|
476
|
+
except ValueError: # Python's int() refuses a text of thousands of digits
|
|
477
|
+
raise Prover9ParsingError(
|
|
478
|
+
f"SYNTAX_ERROR: op() precedence {prec_text[:12]!r}... has {len(prec_text.lstrip('-'))} digits: "
|
|
479
|
+
"it is out of Prover9's valid range (1-998)") from None
|
|
480
|
+
if type_text not in _P9_OP_TYPE_NAMES:
|
|
481
|
+
raise Prover9ParsingError(
|
|
482
|
+
f"SYNTAX_ERROR: unknown op() type {type_text!r} (expected one of "
|
|
483
|
+
+ ", ".join(_P9_OP_TYPE_NAMES) + ")")
|
|
484
|
+
if symbols_text.startswith("[") and symbols_text.endswith("]"):
|
|
485
|
+
inner_tokens = _split_top_level_commas(symbols_text[1:-1])
|
|
486
|
+
if inner_tokens == [""]:
|
|
487
|
+
raise Prover9ParsingError(f"SYNTAX_ERROR: empty op() symbol list: {body!r}")
|
|
488
|
+
else:
|
|
489
|
+
inner_tokens = [symbols_text]
|
|
490
|
+
symbols = [_parse_op_symbol(tok) for tok in inner_tokens]
|
|
491
|
+
return [_CustomOp(precedence, type_text, sym) for sym in symbols]
|
|
492
|
+
|
|
493
|
+
|
|
494
|
+
#: The first characters that make a bare name a variable (LADR ``variable_name``, measured on
|
|
495
|
+
#: Prover9 2026-8A, by every first letter and with names of several characters): under
|
|
496
|
+
#: ``set(prolog_style_variables)`` an upper-case ASCII letter, without it ``u`` to ``z``. Nothing
|
|
497
|
+
#: else is: ``_x`` is a constant in both conventions, and a quoted symbol never is.
|
|
498
|
+
_PROLOG_VARIABLE_INITIALS = frozenset("ABCDEFGHIJKLMNOPQRSTUVWXYZ")
|
|
499
|
+
_STANDARD_VARIABLE_INITIALS = frozenset("uvwxyz")
|
|
500
|
+
|
|
501
|
+
|
|
502
|
+
def _is_variable(name: str, prolog_style: bool = True) -> bool:
|
|
503
|
+
"""Whether Prover9 reads the bare name ``name`` as a variable when no quantifier binds it:
|
|
504
|
+
its first character is ``A``..``Z`` under ``set(prolog_style_variables)``, ``u``..``z``
|
|
505
|
+
without it."""
|
|
506
|
+
return name[:1] in (_PROLOG_VARIABLE_INITIALS if prolog_style else _STANDARD_VARIABLE_INITIALS)
|
|
507
|
+
|
|
508
|
+
|
|
509
|
+
_P9_QUOTED_NAME_RE = re.compile(r"[A-Za-z_][A-Za-z0-9_]*")
|
|
510
|
+
_P9_QUOTED_NUMERAL_RE = re.compile(r"-?[0-9]+(\.[0-9]+)?")
|
|
511
|
+
|
|
512
|
+
|
|
513
|
+
def _unquote(token) -> str:
|
|
514
|
+
"""The name inside a double-quoted symbol token, or a refusal by name.
|
|
515
|
+
|
|
516
|
+
Prover9 accepts any text without a double quote between the quotes. Only an
|
|
517
|
+
identifier-shaped name is read here (see the module docstring): the AST has no
|
|
518
|
+
way to carry another one that :meth:`Node.to_prover9` could write back.
|
|
519
|
+
"""
|
|
520
|
+
name = str(token)[1:-1]
|
|
521
|
+
if not _P9_QUOTED_NAME_RE.fullmatch(name):
|
|
522
|
+
raise Prover9ParsingError(
|
|
523
|
+
f"SYNTAX_ERROR: the quoted symbol {str(token)!r} is not supported by this "
|
|
524
|
+
"reader: Prover9 accepts any text between double quotes, but this reader "
|
|
525
|
+
"reads only a quoted name made of letters, digits and underscores that "
|
|
526
|
+
"does not begin with a digit (the shape Node.to_prover9 writes), or a "
|
|
527
|
+
"quoted numeral (\"2.5\", \"-1\"), because any other name could not be "
|
|
528
|
+
"written back as the same symbol")
|
|
529
|
+
return name
|
|
530
|
+
|
|
531
|
+
|
|
532
|
+
def _quoted_numeral(token) -> "Number":
|
|
533
|
+
"""The :class:`Number` a quoted numeral symbol (``"2.5"``, ``"-1"``) stands for, or a
|
|
534
|
+
refusal by name when the text is not the canonical spelling of that number.
|
|
535
|
+
|
|
536
|
+
Prover9 keeps ``"2.5"`` and ``"2.50"``, and ``"1"`` and ``"1.0"``, apart as two symbols;
|
|
537
|
+
a :class:`Number` is identified by its value and has one text, so only the spelling
|
|
538
|
+
:meth:`Number.to_prover9` writes is read (``"1"``, never ``"1.0"``).
|
|
539
|
+
"""
|
|
540
|
+
text = str(token)[1:-1]
|
|
541
|
+
value = _numeral(text, token)
|
|
542
|
+
if text != numeral_name(value):
|
|
543
|
+
raise Prover9ParsingError(
|
|
544
|
+
f"SYNTAX_ERROR: the quoted numeral {str(token)!r} is not read: Prover9 keeps "
|
|
545
|
+
f"it apart from the quoted numeral {numeral_name(value)!r} that is the same "
|
|
546
|
+
"number, and a Number node has one text, so reading it would merge two "
|
|
547
|
+
"symbols. Write the number in its canonical form")
|
|
548
|
+
return Number(value)
|
|
549
|
+
|
|
550
|
+
|
|
551
|
+
class _Prover9Transformer(Transformer):
|
|
552
|
+
"""Turn the Lark parse tree into the toolkit AST."""
|
|
553
|
+
|
|
554
|
+
def iff_(self, items):
|
|
555
|
+
return Iff(items[0], items[1])
|
|
556
|
+
|
|
557
|
+
def implies_(self, items):
|
|
558
|
+
return Implies(items[0], items[1])
|
|
559
|
+
|
|
560
|
+
def rimplies_(self, items):
|
|
561
|
+
# ``p <- q`` is ``q -> p``.
|
|
562
|
+
return Implies(items[1], items[0])
|
|
563
|
+
|
|
564
|
+
def or_(self, items):
|
|
565
|
+
return Or(items[0], items[1])
|
|
566
|
+
|
|
567
|
+
def and_(self, items):
|
|
568
|
+
return And(items[0], items[1])
|
|
569
|
+
|
|
570
|
+
def neg(self, items):
|
|
571
|
+
return Not(items[0])
|
|
572
|
+
|
|
573
|
+
# The binder and every occurrence of a variable carry the NAME of the AST variable already
|
|
574
|
+
# (see :func:`_resolve_variables`, which runs first and decides what each spelling is).
|
|
575
|
+
def forall(self, items):
|
|
576
|
+
return Quantifier("∀", Variable(str(items[0])), items[1])
|
|
577
|
+
|
|
578
|
+
def exists(self, items):
|
|
579
|
+
return Quantifier("∃", Variable(str(items[0])), items[1])
|
|
580
|
+
|
|
581
|
+
# --- atoms ---
|
|
582
|
+
def equality(self, items):
|
|
583
|
+
return Atom("=", [items[0], items[1]])
|
|
584
|
+
|
|
585
|
+
def disequality(self, items):
|
|
586
|
+
return Atom("≠", [items[0], items[1]])
|
|
587
|
+
|
|
588
|
+
def le(self, items):
|
|
589
|
+
return Atom("≤", [items[0], items[1]])
|
|
590
|
+
|
|
591
|
+
def ge(self, items):
|
|
592
|
+
return Atom("≥", [items[0], items[1]])
|
|
593
|
+
|
|
594
|
+
def lt(self, items):
|
|
595
|
+
return Atom("<", [items[0], items[1]])
|
|
596
|
+
|
|
597
|
+
def gt(self, items):
|
|
598
|
+
return Atom(">", [items[0], items[1]])
|
|
599
|
+
|
|
600
|
+
def pred_app(self, items):
|
|
601
|
+
return Atom(str(items[0]), items[1])
|
|
602
|
+
|
|
603
|
+
def prop_atom(self, items):
|
|
604
|
+
return Atom(str(items[0]), [])
|
|
605
|
+
|
|
606
|
+
# --- Prover9's own constants: $T is true, $F is false ---
|
|
607
|
+
def truth_atom(self, items):
|
|
608
|
+
return Atom("$true" if str(items[0]) == "$T" else "$false", [])
|
|
609
|
+
|
|
610
|
+
# --- quoted symbols: never a variable, whatever their first letter ---
|
|
611
|
+
def qpred_app(self, items):
|
|
612
|
+
return Atom(_unquote(items[0]), items[1])
|
|
613
|
+
|
|
614
|
+
def qprop_atom(self, items):
|
|
615
|
+
return Atom(_unquote(items[0]), [])
|
|
616
|
+
|
|
617
|
+
# --- terms ---
|
|
618
|
+
def add(self, items):
|
|
619
|
+
return Function("+", [items[0], items[1]])
|
|
620
|
+
|
|
621
|
+
def sub(self, items):
|
|
622
|
+
return Function("-", [items[0], items[1]])
|
|
623
|
+
|
|
624
|
+
def mul(self, items):
|
|
625
|
+
return Function("*", [items[0], items[1]])
|
|
626
|
+
|
|
627
|
+
def div(self, items):
|
|
628
|
+
return Function("/", [items[0], items[1]])
|
|
629
|
+
|
|
630
|
+
def func_app(self, items):
|
|
631
|
+
return Function(str(items[0]), items[1])
|
|
632
|
+
|
|
633
|
+
def name_term(self, items):
|
|
634
|
+
# A bare name that is no variable (see :func:`_resolve_variables`): a constant.
|
|
635
|
+
return Constant(str(items[0]))
|
|
636
|
+
|
|
637
|
+
def var_term(self, items):
|
|
638
|
+
return Variable(str(items[0]))
|
|
639
|
+
|
|
640
|
+
def qfunc_app(self, items):
|
|
641
|
+
return Function(_unquote(items[0]), items[1])
|
|
642
|
+
|
|
643
|
+
def qname_term(self, items):
|
|
644
|
+
if _P9_QUOTED_NUMERAL_RE.fullmatch(str(items[0])[1:-1]):
|
|
645
|
+
return _quoted_numeral(items[0])
|
|
646
|
+
return Constant(_unquote(items[0]))
|
|
647
|
+
|
|
648
|
+
def minus_app(self, items):
|
|
649
|
+
return Function("-", [items[0], *items[1]])
|
|
650
|
+
|
|
651
|
+
def uminus(self, items):
|
|
652
|
+
return Function("-", [items[0]])
|
|
653
|
+
|
|
654
|
+
def number(self, items):
|
|
655
|
+
return Number(_numeral(str(items[0]), items[0]))
|
|
656
|
+
|
|
657
|
+
def termlist(self, items):
|
|
658
|
+
return list(items)
|
|
659
|
+
|
|
660
|
+
|
|
661
|
+
_PARSER = Lark(_GRAMMAR, start="start", parser="earley")
|
|
662
|
+
_TRANSFORMER = _Prover9Transformer()
|
|
663
|
+
|
|
664
|
+
|
|
665
|
+
def _classify_and_splice(custom_ops: tuple) -> tuple:
|
|
666
|
+
"""Validate ``custom_ops`` (in declaration order) and split them into the
|
|
667
|
+
grammar-splice buckets :func:`_build_custom_grammar` needs: atom-tier infix,
|
|
668
|
+
term-tier infix, term-tier prefix and term-tier postfix.
|
|
669
|
+
|
|
670
|
+
Two things are refused outright, by name, regardless of whether the
|
|
671
|
+
operator is ever used in a formula — because accepting them would risk
|
|
672
|
+
silently reinterpreting other, unrelated text in the same file:
|
|
673
|
+
redeclaring an existing built-in (:data:`_RESERVED_SYMBOLS`), and
|
|
674
|
+
redeclaring a symbol this same file already gave a custom meaning to.
|
|
675
|
+
|
|
676
|
+
Everything else that is syntactically well-formed is accepted (matching
|
|
677
|
+
real Prover9, which does not reject any of it either) and is spliced into
|
|
678
|
+
the grammar when it falls in a window this reader supports — below the
|
|
679
|
+
arithmetic tier (< 500) for a :class:`Function`-producing term operator,
|
|
680
|
+
or between the arithmetic and quantifier tiers (500-750, excluding the
|
|
681
|
+
comparison tier at 700) for an :class:`Atom`-producing one. A declaration
|
|
682
|
+
this reader has no splice point for — ``type="ordinary"``, a precedence
|
|
683
|
+
that exactly ties a built-in tier (:data:`_RESERVED_PRECEDENCES`), a
|
|
684
|
+
precedence at or above the quantifier tier (750), or prefix/postfix at an
|
|
685
|
+
atom-tier precedence — is simply left inert: recorded (so it still blocks
|
|
686
|
+
a later redeclaration) but never spliced in, exactly like every op()
|
|
687
|
+
directive before this feature existed. A formula that then tries to use
|
|
688
|
+
such an operator sees an ordinary undeclared name and fails to parse —
|
|
689
|
+
loudly, just at that point rather than at the declaration.
|
|
690
|
+
"""
|
|
691
|
+
seen = set()
|
|
692
|
+
atom_ops, term_infix, term_prefix, term_postfix = [], [], [], []
|
|
693
|
+
for op in custom_ops:
|
|
694
|
+
if op.symbol in _RESERVED_SYMBOLS:
|
|
695
|
+
raise Prover9ParsingError(
|
|
696
|
+
f"SYNTAX_ERROR: redeclaring built-in Prover9 operator {op.symbol!r} "
|
|
697
|
+
"is not supported")
|
|
698
|
+
if op.symbol in seen:
|
|
699
|
+
raise Prover9ParsingError(
|
|
700
|
+
f"SYNTAX_ERROR: operator {op.symbol!r} is already declared by an "
|
|
701
|
+
"earlier op(...) in this file")
|
|
702
|
+
seen.add(op.symbol)
|
|
703
|
+
if not (1 <= op.precedence <= 998):
|
|
704
|
+
raise Prover9ParsingError(
|
|
705
|
+
f"SYNTAX_ERROR: op() precedence {op.precedence} is out of "
|
|
706
|
+
"Prover9's valid range (1-998)")
|
|
707
|
+
if op.type == "ordinary":
|
|
708
|
+
continue # not a mixfix operator: nothing to splice, just reserves the name
|
|
709
|
+
if op.precedence in _RESERVED_PRECEDENCES:
|
|
710
|
+
continue # ties a built-in tier: left inert (see docstring above)
|
|
711
|
+
if op.precedence < 500:
|
|
712
|
+
kind = "term"
|
|
713
|
+
elif op.precedence < 750:
|
|
714
|
+
kind = "atom"
|
|
715
|
+
else:
|
|
716
|
+
continue # would graft onto the quantifier/connective grammar: left inert
|
|
717
|
+
if kind == "atom" and op.type not in ("infix", "infix_left", "infix_right"):
|
|
718
|
+
continue # prefix/postfix only supported at a term-tier precedence: left inert
|
|
719
|
+
if kind == "atom":
|
|
720
|
+
atom_ops.append(op)
|
|
721
|
+
elif op.type in ("infix", "infix_left", "infix_right"):
|
|
722
|
+
term_infix.append(op)
|
|
723
|
+
elif op.type in ("prefix", "prefix_paren"):
|
|
724
|
+
term_prefix.append(op)
|
|
725
|
+
else:
|
|
726
|
+
term_postfix.append(op)
|
|
727
|
+
return tuple(atom_ops), tuple(term_infix), tuple(term_prefix), tuple(term_postfix)
|
|
728
|
+
|
|
729
|
+
|
|
730
|
+
def _infix_rule(rule: str, alias: str, sym: str, op_type: str, operand: str) -> str:
|
|
731
|
+
"""Build one Lark rule definition for a new infix operator.
|
|
732
|
+
|
|
733
|
+
``infix`` (xfx) is a single non-recursive alternative — deliberately: with
|
|
734
|
+
no self-recursive branch, a chain like ``a sym b sym c`` cannot be produced
|
|
735
|
+
by this rule at all, which is exactly Prover9's own non-associative
|
|
736
|
+
semantics (both operands of an xfx operator need strictly lower precedence,
|
|
737
|
+
so a bare 3-way chain needs explicit parentheses in real Prover9 too).
|
|
738
|
+
``infix_right`` (xfy) recurses on the right operand, ``infix_left`` (yfx) on
|
|
739
|
+
the left, each with a non-recursive base alternative for the single-use case.
|
|
740
|
+
"""
|
|
741
|
+
if op_type == "infix":
|
|
742
|
+
return f'{rule}: {operand} "{sym}" {operand} -> {alias}'
|
|
743
|
+
if op_type == "infix_right":
|
|
744
|
+
return (f'{rule}: {operand} "{sym}" {rule} -> {alias}\n'
|
|
745
|
+
f' | {operand} "{sym}" {operand} -> {alias}')
|
|
746
|
+
return (f'{rule}: {rule} "{sym}" {operand} -> {alias}\n'
|
|
747
|
+
f' | {operand} "{sym}" {operand} -> {alias}')
|
|
748
|
+
|
|
749
|
+
|
|
750
|
+
@lru_cache(maxsize=256)
|
|
751
|
+
def _build_custom_grammar(custom_ops: tuple):
|
|
752
|
+
"""Build (and cache, by the exact tuple of active op() declarations) a Lark
|
|
753
|
+
parser + transformer pair that extends the shared grammar with newly
|
|
754
|
+
op()-declared operators. An empty tuple returns the shared singleton
|
|
755
|
+
(:data:`_PARSER`, :data:`_TRANSFORMER`) unchanged — no rebuild, no perf
|
|
756
|
+
regression for the common case of a file with no custom operators.
|
|
757
|
+
|
|
758
|
+
Validation (redeclaration, tie, range and placement checks — see
|
|
759
|
+
:func:`_classify_and_splice`) happens here, so it runs once per distinct
|
|
760
|
+
set of active declarations and is then free on every later call.
|
|
761
|
+
|
|
762
|
+
Generated Lark rule/alias names are built from a per-bucket numeric index,
|
|
763
|
+
never from ``op.symbol`` itself: Lark's own grammar meta-language requires
|
|
764
|
+
a RULE/alias identifier to start with a lowercase letter (or underscore)
|
|
765
|
+
and never contain an uppercase one (an uppercase-leading token is instead
|
|
766
|
+
read as a TERMINAL reference), while op() symbols accepted here
|
|
767
|
+
(:func:`_parse_op_symbol`) are only required to match the *formula*
|
|
768
|
+
grammar's own ``NAME`` terminal — which does allow uppercase. Splicing an
|
|
769
|
+
uppercase-containing symbol straight into a generated rule name (e.g.
|
|
770
|
+
``atom_chain_Before``) would therefore be lexed as two malformed grammar
|
|
771
|
+
tokens and make the ``Lark(...)`` call below raise
|
|
772
|
+
``lark.exceptions.UnexpectedToken`` — a bare Lark internals leak, not a
|
|
773
|
+
targeted :class:`Prover9ParsingError`, for a symbol shape this reader's
|
|
774
|
+
own validator otherwise accepts. The real symbol still appears in the
|
|
775
|
+
grammar, but only inside a quoted string literal (``"{op.symbol}"``),
|
|
776
|
+
where Lark's syntax places no case restriction.
|
|
777
|
+
"""
|
|
778
|
+
if not custom_ops:
|
|
779
|
+
return _PARSER, _TRANSFORMER
|
|
780
|
+
atom_ops, term_infix, term_prefix, term_postfix = _classify_and_splice(custom_ops)
|
|
781
|
+
|
|
782
|
+
atom_extra, term_extra, unit_extra = [], [], []
|
|
783
|
+
extra_rules = []
|
|
784
|
+
handlers = {}
|
|
785
|
+
|
|
786
|
+
for idx, op in enumerate(atom_ops):
|
|
787
|
+
alias = f"custom_atom_{idx}"
|
|
788
|
+
rule = f"atom_chain_{idx}"
|
|
789
|
+
extra_rules.append(_infix_rule(rule, alias, op.symbol, op.type, "term"))
|
|
790
|
+
atom_extra.append(f"\n | {rule}")
|
|
791
|
+
handlers[alias] = (lambda items, _s=op.symbol: Atom(_s, [items[0], items[1]]))
|
|
792
|
+
|
|
793
|
+
for idx, op in enumerate(term_infix):
|
|
794
|
+
alias = f"custom_term_{idx}"
|
|
795
|
+
rule = f"term_chain_{idx}"
|
|
796
|
+
extra_rules.append(_infix_rule(rule, alias, op.symbol, op.type, "unit_term"))
|
|
797
|
+
term_extra.append(f"\n | {rule}")
|
|
798
|
+
handlers[alias] = (lambda items, _s=op.symbol: Function(_s, [items[0], items[1]]))
|
|
799
|
+
|
|
800
|
+
for idx, op in enumerate(term_prefix):
|
|
801
|
+
alias = f"custom_prefix_{idx}"
|
|
802
|
+
rule = f"unit_prefix_{idx}"
|
|
803
|
+
extra_rules.append(f'{rule}: "{op.symbol}" unit_term -> {alias}')
|
|
804
|
+
unit_extra.append(f"\n | {rule}")
|
|
805
|
+
handlers[alias] = (lambda items, _s=op.symbol: Function(_s, [items[0]]))
|
|
806
|
+
|
|
807
|
+
for idx, op in enumerate(term_postfix):
|
|
808
|
+
alias = f"custom_postfix_{idx}"
|
|
809
|
+
rule = f"unit_postfix_{idx}"
|
|
810
|
+
extra_rules.append(f'{rule}: unit_term "{op.symbol}" -> {alias}')
|
|
811
|
+
unit_extra.append(f"\n | {rule}")
|
|
812
|
+
handlers[alias] = (lambda items, _s=op.symbol: Function(_s, [items[0]]))
|
|
813
|
+
|
|
814
|
+
rules_text = _FORMULA_RULES_TEMPLATE.format(
|
|
815
|
+
atom_extra="".join(atom_extra),
|
|
816
|
+
term_extra="".join(term_extra),
|
|
817
|
+
unit_term_extra="".join(unit_extra),
|
|
818
|
+
)
|
|
819
|
+
grammar_text = "?start: formula\n\n" + rules_text + "\n" + "\n".join(extra_rules) + "\n"
|
|
820
|
+
parser = Lark(grammar_text, start="start", parser="earley")
|
|
821
|
+
transformer = _Prover9Transformer()
|
|
822
|
+
for name, handler in handlers.items():
|
|
823
|
+
setattr(transformer, name, handler)
|
|
824
|
+
return parser, transformer
|
|
825
|
+
|
|
826
|
+
|
|
827
|
+
def _normalize_custom_ops(custom_ops) -> tuple:
|
|
828
|
+
"""Coerce ``custom_ops`` into a tuple of :class:`_CustomOp` (accepting plain
|
|
829
|
+
``(precedence, type, symbol)`` triples, the public/documented shape, as well
|
|
830
|
+
as already-built :class:`_CustomOp` records)."""
|
|
831
|
+
result = []
|
|
832
|
+
for item in custom_ops:
|
|
833
|
+
if isinstance(item, _CustomOp):
|
|
834
|
+
result.append(item)
|
|
835
|
+
else:
|
|
836
|
+
precedence, op_type, symbol = item
|
|
837
|
+
result.append(_CustomOp(int(precedence), str(op_type), str(symbol)))
|
|
838
|
+
return tuple(result)
|
|
839
|
+
|
|
840
|
+
|
|
841
|
+
def parse_prover9(text: str, custom_ops=(), *, prolog_style_variables: bool = True) -> Node:
|
|
842
|
+
"""Parse a single Prover9-syntax formula into a toolkit :class:`Node`.
|
|
843
|
+
|
|
844
|
+
A trailing period (Prover9 terminates each formula with ``.``) is accepted
|
|
845
|
+
and ignored.
|
|
846
|
+
|
|
847
|
+
A formula has no file around it to say which names are variables, so this reader
|
|
848
|
+
keeps the convention of :meth:`Node.to_prover9`, ``set(prolog_style_variables)``: a name
|
|
849
|
+
that no quantifier binds is a variable when it begins with an upper-case letter and a
|
|
850
|
+
constant otherwise (see the module docstring for what a quantifier binds). Pass
|
|
851
|
+
``prolog_style_variables=False`` for Prover9's default, where it is a variable when it
|
|
852
|
+
begins with ``u`` to ``z``.
|
|
853
|
+
|
|
854
|
+
Args:
|
|
855
|
+
text: a Prover9 formula, e.g. ``"(all X (man(X) -> mortal(X)))"``.
|
|
856
|
+
custom_ops: previously-declared ``op(precedence, type, symbol)``
|
|
857
|
+
operators that extend the grammar for this one parse, as
|
|
858
|
+
``(precedence, type, symbol)`` triples, in declaration order — see
|
|
859
|
+
the module docstring's "op(...) declarations" section for exactly
|
|
860
|
+
which operators are applied and which are refused.
|
|
861
|
+
:func:`parse_prover9_problem` builds and threads this automatically
|
|
862
|
+
from a file's own ``op(...)`` directives; most callers parsing a
|
|
863
|
+
single formula never need to pass it.
|
|
864
|
+
prolog_style_variables: which of the two conventions reads a name that is bound
|
|
865
|
+
by no quantifier (``True`` is the default of this function).
|
|
866
|
+
|
|
867
|
+
Returns:
|
|
868
|
+
The formula as a toolkit :class:`Node`.
|
|
869
|
+
|
|
870
|
+
Raises:
|
|
871
|
+
Prover9ParsingError: if ``text`` is not a well-formed Prover9 formula,
|
|
872
|
+
or ``custom_ops`` contains a declaration this reader refuses.
|
|
873
|
+
"""
|
|
874
|
+
return _parse_formula(text, custom_ops, {}, prolog_style_variables)
|
|
875
|
+
|
|
876
|
+
|
|
877
|
+
# The tree nodes that name a symbol, with the key of the symbol they stand for.
|
|
878
|
+
# ``(kind, name, arity)`` as the AST keeps it: a name in formula position is a predicate, in
|
|
879
|
+
# term position a function (a constant is a function of arity 0), a number its own kind.
|
|
880
|
+
_P9_BARE_SYMBOL_RULES = {"pred_app": "predicate", "func_app": "function"}
|
|
881
|
+
_P9_QUOTED_SYMBOL_RULES = {"qpred_app": "predicate", "qfunc_app": "function"}
|
|
882
|
+
|
|
883
|
+
|
|
884
|
+
def _symbol_spellings(tree) -> list:
|
|
885
|
+
"""Every symbol of a parse tree as ``(key, "bare" | "quoted")``, ``key`` being
|
|
886
|
+
``(kind, name, arity)`` as the AST keeps a symbol (the name only, not its quotes).
|
|
887
|
+
|
|
888
|
+
A variable is no symbol: :func:`_resolve_variables` has turned every name in term position
|
|
889
|
+
that is a variable (bound by a quantifier, or one by the convention of the file) into a
|
|
890
|
+
``var_term`` node, so the ``name_term`` nodes that are left are constants.
|
|
891
|
+
"""
|
|
892
|
+
found = []
|
|
893
|
+
for sub in tree.iter_subtrees():
|
|
894
|
+
rule = str(sub.data)
|
|
895
|
+
if rule in _P9_BARE_SYMBOL_RULES or rule in _P9_QUOTED_SYMBOL_RULES:
|
|
896
|
+
bare = rule in _P9_BARE_SYMBOL_RULES
|
|
897
|
+
kind = _P9_BARE_SYMBOL_RULES[rule] if bare else _P9_QUOTED_SYMBOL_RULES[rule]
|
|
898
|
+
name = str(sub.children[0]) if bare else str(sub.children[0])[1:-1]
|
|
899
|
+
found.append(((kind, name, len(sub.children[1].children)), "bare" if bare else "quoted"))
|
|
900
|
+
elif rule == "prop_atom":
|
|
901
|
+
found.append((("predicate", str(sub.children[0]), 0), "bare"))
|
|
902
|
+
elif rule == "qprop_atom":
|
|
903
|
+
found.append((("predicate", str(sub.children[0])[1:-1], 0), "quoted"))
|
|
904
|
+
elif rule == "name_term":
|
|
905
|
+
found.append((("function", str(sub.children[0]), 0), "bare"))
|
|
906
|
+
elif rule == "qname_term":
|
|
907
|
+
name = str(sub.children[0])[1:-1]
|
|
908
|
+
if _P9_QUOTED_NUMERAL_RE.fullmatch(name):
|
|
909
|
+
found.append((("numeral", name, 0), "quoted"))
|
|
910
|
+
else:
|
|
911
|
+
found.append((("function", name, 0), "quoted"))
|
|
912
|
+
elif rule == "number":
|
|
913
|
+
found.append((("numeral", str(sub.children[0]), 0), "bare"))
|
|
914
|
+
return found
|
|
915
|
+
|
|
916
|
+
|
|
917
|
+
def _refuse_two_spellings(tree, spellings: dict) -> None:
|
|
918
|
+
"""Refuse a symbol that is written both with and without double quotes.
|
|
919
|
+
|
|
920
|
+
Prover9 keeps ``"rain"`` and ``rain`` apart as two symbols, and this reader has one
|
|
921
|
+
name per symbol, so a text that uses both for the same predicate, function, constant
|
|
922
|
+
or number would be read as ONE and mean something else. ``spellings`` carries the
|
|
923
|
+
symbols already met (a file is one text), and is updated.
|
|
924
|
+
"""
|
|
925
|
+
for key, how in _symbol_spellings(tree):
|
|
926
|
+
seen = spellings.setdefault(key, set())
|
|
927
|
+
seen.add(how)
|
|
928
|
+
if len(seen) == 2:
|
|
929
|
+
kind, name, arity = key
|
|
930
|
+
what = {"predicate": "predicate" if arity else "proposition",
|
|
931
|
+
"function": "function" if arity else "constant",
|
|
932
|
+
"numeral": "number"}[kind]
|
|
933
|
+
raise Prover9ParsingError(
|
|
934
|
+
f"SYNTAX_ERROR: the {what} {name!r} is written both with and without double "
|
|
935
|
+
f"quotes (\"{name}\" and {name}): Prover9 reads them as two symbols, this "
|
|
936
|
+
"reader has one name per symbol and would read them as one, so the text "
|
|
937
|
+
"would mean something else. Write the symbol one way.")
|
|
938
|
+
|
|
939
|
+
|
|
940
|
+
def _refuse_two_numeral_spellings(tree, spellings: dict) -> None:
|
|
941
|
+
"""Refuse a text that spells one numeral value two ways (``01`` and ``1``, ``1.0`` and ``1``).
|
|
942
|
+
|
|
943
|
+
Prover9 keeps such numerals apart as symbols of their own, and a :class:`Number` is identified
|
|
944
|
+
by its value, so this reader would read the two as ONE numeral: ``P(01). -P(1).`` is consistent
|
|
945
|
+
(U = {0, 1}, 01 = 0, 1 = 1, P = {0}), and read as ``P(1), ¬P(1)`` it proves everything.
|
|
946
|
+
``spellings`` is the record shared by every formula of a file (a numeral value -> the texts
|
|
947
|
+
it was written as), and is updated.
|
|
948
|
+
"""
|
|
949
|
+
for sub in tree.iter_subtrees():
|
|
950
|
+
rule = str(sub.data)
|
|
951
|
+
if rule == "number":
|
|
952
|
+
written = str(sub.children[0])
|
|
953
|
+
elif rule == "qname_term" and _P9_QUOTED_NUMERAL_RE.fullmatch(str(sub.children[0])[1:-1]):
|
|
954
|
+
written = str(sub.children[0])[1:-1]
|
|
955
|
+
else:
|
|
956
|
+
continue
|
|
957
|
+
value = _numeral(written, sub.children[0])
|
|
958
|
+
texts = spellings.setdefault(("numeral value", numeral_name(value)), set())
|
|
959
|
+
texts.add(written)
|
|
960
|
+
if len(texts) > 1:
|
|
961
|
+
first, second = sorted(texts, key=lambda t: (len(t), t))[:2]
|
|
962
|
+
raise Prover9ParsingError(
|
|
963
|
+
f"SYNTAX_ERROR: the numerals {first} and {second} are written in one text: Prover9 "
|
|
964
|
+
"reads them as two symbols, this reader has one numeral per value and would read "
|
|
965
|
+
"them as one, so the text would mean something else. Write the number one way.")
|
|
966
|
+
|
|
967
|
+
|
|
968
|
+
_P9_WORD_RE = re.compile(r"[A-Za-z_][A-Za-z0-9_]*")
|
|
969
|
+
|
|
970
|
+
|
|
971
|
+
def _resolve_variables(tree, prolog_style: bool, record: dict, text: str) -> None:
|
|
972
|
+
"""Decide which names of a parse tree are variables, and give each variable its AST name.
|
|
973
|
+
|
|
974
|
+
Prover9 reads a quantifier as binding the SYMBOL it names, whatever its case: in
|
|
975
|
+
``all x (man(x) -> mortal(x))`` the three ``x`` are one variable, and the same text with
|
|
976
|
+
``X`` says the same (measured on Prover9 2026-8A, with and without
|
|
977
|
+
``set(prolog_style_variables)``). Inside the scope of the quantifier (the operand that
|
|
978
|
+
follows the variable) a bare name in term position is that variable; an inner quantifier
|
|
979
|
+
over the same spelling rebinds it, a name that is applied to arguments or stands as a
|
|
980
|
+
formula is no term occurrence, and outside every quantifier the convention of the file
|
|
981
|
+
decides (:func:`_is_variable`). The transformation builds a body before it meets the binder
|
|
982
|
+
that is above it, so this runs first, over the finished parse tree, with an explicit stack: a
|
|
983
|
+
``name_term`` that is a variable becomes a ``var_term``, and the NAME of the binder and of
|
|
984
|
+
every variable occurrence is replaced by the name of the :class:`Variable` it stands for.
|
|
985
|
+
|
|
986
|
+
Variables are compared as written (``Xa`` and ``XA`` are two variables, ``x`` and ``X``
|
|
987
|
+
too): each spelling gets the lower-case of itself as its name, which is what
|
|
988
|
+
:meth:`Variable.to_prover9` (it upper-cases) is the inverse of, and a spelling whose
|
|
989
|
+
lower-case another spelling of the same formula already has gets a fresh name instead
|
|
990
|
+
(:func:`~unicode_logic_kit.fol._identifiers.fresh_variable_like`, fresh against every word of
|
|
991
|
+
the text). A variable that no quantifier binds is one unknown element of the whole file (the
|
|
992
|
+
same in every formula), so its spelling keeps its name in all of them: ``record`` is shared by
|
|
993
|
+
all formulas of a file and holds those names.
|
|
994
|
+
|
|
995
|
+
Raises:
|
|
996
|
+
Prover9ParsingError: a bound name stands as a formula (``all x (P(x) & x)``), which
|
|
997
|
+
Prover9 refuses as well, because a variable cannot be an atomic formula.
|
|
998
|
+
"""
|
|
999
|
+
scope: dict = {} # spelling -> number of enclosing quantifiers that bind it
|
|
1000
|
+
sites: list = [] # (tree node, spelling, free) of every variable, in text order
|
|
1001
|
+
seen: set = set()
|
|
1002
|
+
stack = [(tree, False)]
|
|
1003
|
+
while stack:
|
|
1004
|
+
node, leaving = stack.pop()
|
|
1005
|
+
if leaving:
|
|
1006
|
+
scope[str(node.children[0])] -= 1
|
|
1007
|
+
continue
|
|
1008
|
+
if id(node) in seen:
|
|
1009
|
+
continue
|
|
1010
|
+
seen.add(id(node))
|
|
1011
|
+
rule = node.data
|
|
1012
|
+
if rule == "name_term":
|
|
1013
|
+
spelling = str(node.children[0])
|
|
1014
|
+
bound = bool(scope.get(spelling))
|
|
1015
|
+
if bound or _is_variable(spelling, prolog_style):
|
|
1016
|
+
node.data = "var_term"
|
|
1017
|
+
sites.append((node, spelling, not bound))
|
|
1018
|
+
continue
|
|
1019
|
+
if rule == "prop_atom":
|
|
1020
|
+
spelling = str(node.children[0])
|
|
1021
|
+
if scope.get(spelling):
|
|
1022
|
+
raise Prover9ParsingError(
|
|
1023
|
+
f"SYNTAX_ERROR: the name {spelling!r} stands as a formula inside the scope of "
|
|
1024
|
+
f"'all {spelling}' / 'exists {spelling}', where it is a variable: Prover9 refuses "
|
|
1025
|
+
"a variable as an atomic formula, and so does this reader.")
|
|
1026
|
+
continue
|
|
1027
|
+
if rule in ("forall", "exists"):
|
|
1028
|
+
spelling = str(node.children[0])
|
|
1029
|
+
sites.append((node, spelling, False))
|
|
1030
|
+
scope[spelling] = scope.get(spelling, 0) + 1
|
|
1031
|
+
stack.append((node, True))
|
|
1032
|
+
stack.extend((child, False) for child in reversed(node.children) if isinstance(child, Tree))
|
|
1033
|
+
if not sites:
|
|
1034
|
+
return
|
|
1035
|
+
free_names = record.setdefault(("variable names",), {}) # free spelling -> name, in all of the file
|
|
1036
|
+
free_taken = record.setdefault(("variable names taken",), set())
|
|
1037
|
+
free_spellings = {spelling for _, spelling, free in sites if free}
|
|
1038
|
+
spellings = list(dict.fromkeys(spelling for _, spelling, _ in sites))
|
|
1039
|
+
reserved = {s.lower() for s in spellings} | {w.lower() for w in _P9_WORD_RE.findall(text)}
|
|
1040
|
+
names: dict = {}
|
|
1041
|
+
used: set = set() # the names this formula has given out
|
|
1042
|
+
for free_pass in (True, False):
|
|
1043
|
+
for spelling in spellings:
|
|
1044
|
+
if (spelling in free_spellings) != free_pass:
|
|
1045
|
+
continue
|
|
1046
|
+
name = free_names.get(spelling) if free_pass else None
|
|
1047
|
+
if name is None:
|
|
1048
|
+
name = spelling.lower()
|
|
1049
|
+
if name in used or (free_pass and name in free_taken):
|
|
1050
|
+
name = fresh_variable_like(name, used | free_taken | reserved)
|
|
1051
|
+
if free_pass:
|
|
1052
|
+
free_names[spelling] = name
|
|
1053
|
+
free_taken.add(name)
|
|
1054
|
+
names[spelling] = name
|
|
1055
|
+
used.add(name)
|
|
1056
|
+
for node, spelling, _ in sites:
|
|
1057
|
+
node.children[0] = names[spelling]
|
|
1058
|
+
|
|
1059
|
+
|
|
1060
|
+
def _parse_formula(text: str, custom_ops, spellings: dict, prolog_style: bool = True) -> Node:
|
|
1061
|
+
""":func:`parse_prover9` for a text that is part of a bigger one: ``spellings`` is the
|
|
1062
|
+
record of :func:`_refuse_two_spellings` and of the names of the variables, shared by every
|
|
1063
|
+
formula of a file, and ``prolog_style`` the convention of the file (see :func:`_is_variable`)."""
|
|
1064
|
+
ops = _normalize_custom_ops(custom_ops)
|
|
1065
|
+
parser, transformer = _build_custom_grammar(ops) if ops else (_PARSER, _TRANSFORMER)
|
|
1066
|
+
stripped = text.strip()
|
|
1067
|
+
if stripped.endswith("."):
|
|
1068
|
+
stripped = stripped[:-1]
|
|
1069
|
+
try:
|
|
1070
|
+
tree = parser.parse(stripped)
|
|
1071
|
+
except RecursionError:
|
|
1072
|
+
raise Prover9ParsingError(_TOO_DEEP) from None
|
|
1073
|
+
except Exception as exc:
|
|
1074
|
+
raise Prover9ParsingError(
|
|
1075
|
+
f"SYNTAX_ERROR: could not parse Prover9 formula: {exc}")
|
|
1076
|
+
try:
|
|
1077
|
+
_resolve_variables(tree, prolog_style, spellings, stripped)
|
|
1078
|
+
_refuse_two_spellings(tree, spellings)
|
|
1079
|
+
_refuse_two_numeral_spellings(tree, spellings)
|
|
1080
|
+
return _transform_tree(tree, transformer)
|
|
1081
|
+
except ParsingError:
|
|
1082
|
+
raise
|
|
1083
|
+
except RecursionError:
|
|
1084
|
+
raise Prover9ParsingError(_TOO_DEEP) from None
|
|
1085
|
+
except Exception as original:
|
|
1086
|
+
raise Prover9ParsingError(f"SYNTAX_ERROR: in Prover9 formula: {original}")
|
|
1087
|
+
|
|
1088
|
+
|
|
1089
|
+
#: The refusal for a formula that the Python recursion limit stops the reader on.
|
|
1090
|
+
_TOO_DEEP = (
|
|
1091
|
+
"SYNTAX_ERROR: the formula is nested too deeply for this reader (reading it reached "
|
|
1092
|
+
"Python's recursion limit). Split the formula, or raise sys.setrecursionlimit.")
|
|
1093
|
+
|
|
1094
|
+
|
|
1095
|
+
def _transform_tree(tree, transformer):
|
|
1096
|
+
"""``transformer.transform(tree)`` without recursion in Python.
|
|
1097
|
+
|
|
1098
|
+
The transformer of Lark visits the tree recursively, so a chain of a few hundred
|
|
1099
|
+
operands (``a | b | c | ...`` is a left-nested tree) overflowed the stack of the
|
|
1100
|
+
interpreter. This visits the same nodes in the same order (children before their
|
|
1101
|
+
parent), each node through the method of its name, with an explicit stack; the
|
|
1102
|
+
callbacks are the ones of ``transformer``, and so are the exceptions they raise.
|
|
1103
|
+
"""
|
|
1104
|
+
done: dict = {}
|
|
1105
|
+
stack = [(tree, False)]
|
|
1106
|
+
while stack:
|
|
1107
|
+
node, ready = stack.pop()
|
|
1108
|
+
if id(node) in done:
|
|
1109
|
+
continue
|
|
1110
|
+
if not ready:
|
|
1111
|
+
stack.append((node, True))
|
|
1112
|
+
stack.extend((child, False) for child in node.children
|
|
1113
|
+
if isinstance(child, Tree) and id(child) not in done)
|
|
1114
|
+
continue
|
|
1115
|
+
children = [done[id(child)] if isinstance(child, Tree) else child
|
|
1116
|
+
for child in node.children]
|
|
1117
|
+
callback = getattr(transformer, node.data, None)
|
|
1118
|
+
done[id(node)] = callback(children) if callback is not None else Tree(
|
|
1119
|
+
node.data, children, node.meta)
|
|
1120
|
+
return done[id(tree)]
|
|
1121
|
+
|
|
1122
|
+
|
|
1123
|
+
# --- whole-file statement scanner (deterministic; see parse_prover9_problem) ---
|
|
1124
|
+
# A line comment runs from '%' to end of line (LF, CRLF or a bare CR). A statement is a run of text ending
|
|
1125
|
+
# at a '.' that terminates it — i.e. a '.' that is NOT the decimal point of a number
|
|
1126
|
+
# (``.`` immediately followed by a digit, preceded by a digit, stays inside the run).
|
|
1127
|
+
# A double-quoted symbol is skipped whole by both expressions: a '%' inside it starts no comment and a '.' inside
|
|
1128
|
+
# it ends no statement (LADR reads quoted text raw). A quote is a pair; an odd number of them is refused.
|
|
1129
|
+
_P9_COMMENT_RE = re.compile(r'("[^"]*")|%[^\r\n]*')
|
|
1130
|
+
_P9_STATEMENT_RE = re.compile(r'(?:"[^"]*"|[^."])*(?:\.[0-9](?:"[^"]*"|[^."])*)*\.', re.DOTALL)
|
|
1131
|
+
# A top-level directive is ``set``/``clear``/``assign``/``op`` applied with parens.
|
|
1132
|
+
_P9_DIRECTIVES = frozenset({"set", "clear", "assign", "op"})
|
|
1133
|
+
_P9_HEAD_RE = re.compile(r"^([A-Za-z_][A-Za-z0-9_]*)\s*\(")
|
|
1134
|
+
_P9_FORMULAS_RE = re.compile(r"^formulas\s*\(\s*([A-Za-z_][A-Za-z0-9_]*)\s*\)$")
|
|
1135
|
+
_P9_FORMULAS_CALL_RE = re.compile(r"^formulas\s*\(")
|
|
1136
|
+
_P9_VARIABLE_FLAG_RE = re.compile(r"^(set|clear)\s*\(\s*prolog_style_variables\s*\)$")
|
|
1137
|
+
|
|
1138
|
+
|
|
1139
|
+
def _formulas_header_arguments(body: str) -> int:
|
|
1140
|
+
"""The number of arguments of ``body`` when it is ONE call ``formulas( ... )`` that ends
|
|
1141
|
+
with the statement, else ``0`` (a call whose parenthesis is never closed counts as ``1``: it is
|
|
1142
|
+
a header that is not well formed).
|
|
1143
|
+
|
|
1144
|
+
A list header has exactly one argument, so INSIDE a list a statement that is a call of
|
|
1145
|
+
``formulas`` with two or more is an atom (``formulas(alpha, beta)``, which Prover9 reads there
|
|
1146
|
+
like any predicate, and which the writer of this kit writes for a predicate of that name), and
|
|
1147
|
+
so is a statement that goes on after the call (``formulas(a, b) = c``,
|
|
1148
|
+
``formulas(alpha) & Q``). Outside every list the same call is a header that is not well formed
|
|
1149
|
+
(Prover9 stops at it with "Unrecognized command or list"; :func:`_statement_kind`). Quoted
|
|
1150
|
+
symbols and nested parentheses are skipped when the arguments are counted.
|
|
1151
|
+
"""
|
|
1152
|
+
start = _P9_FORMULAS_CALL_RE.match(body)
|
|
1153
|
+
if start is None:
|
|
1154
|
+
return 0
|
|
1155
|
+
depth, arguments, quoted = 0, 1, False
|
|
1156
|
+
for index in range(start.end() - 1, len(body)):
|
|
1157
|
+
character = body[index]
|
|
1158
|
+
if quoted:
|
|
1159
|
+
quoted = character != '"'
|
|
1160
|
+
elif character == '"':
|
|
1161
|
+
quoted = True
|
|
1162
|
+
elif character == "(":
|
|
1163
|
+
depth += 1
|
|
1164
|
+
elif character == ")":
|
|
1165
|
+
depth -= 1
|
|
1166
|
+
if depth == 0:
|
|
1167
|
+
return arguments if index == len(body) - 1 else 0
|
|
1168
|
+
elif character == "," and depth == 1:
|
|
1169
|
+
arguments += 1
|
|
1170
|
+
return 1 # the parenthesis is never closed: a header that is not well formed
|
|
1171
|
+
|
|
1172
|
+
|
|
1173
|
+
def _statement_kind(body: str, in_list: bool) -> str:
|
|
1174
|
+
"""What one statement of a file is: ``"directive"`` (``set``, ``clear``, ``assign``,
|
|
1175
|
+
``op`` outside every list), ``"header"`` (``formulas(NAME)``, the call with ONE argument; also
|
|
1176
|
+
a call with more arguments outside every list, which is a header that is not well formed),
|
|
1177
|
+
``"end"`` (``end_of_list``) or ``"formula"`` (inside a list ``formulas(alpha, beta)`` is one:
|
|
1178
|
+
an atom)."""
|
|
1179
|
+
head_m = _P9_HEAD_RE.match(body)
|
|
1180
|
+
head = head_m.group(1) if head_m else None
|
|
1181
|
+
if not in_list and head in _P9_DIRECTIVES:
|
|
1182
|
+
return "directive"
|
|
1183
|
+
if head == "formulas":
|
|
1184
|
+
arguments = _formulas_header_arguments(body)
|
|
1185
|
+
if arguments == 1 or (arguments > 1 and not in_list):
|
|
1186
|
+
return "header"
|
|
1187
|
+
if body == "end_of_list":
|
|
1188
|
+
return "end"
|
|
1189
|
+
return "formula"
|
|
1190
|
+
|
|
1191
|
+
|
|
1192
|
+
def _prolog_style_of(statements: list) -> bool:
|
|
1193
|
+
"""Whether a file reads its names under ``set(prolog_style_variables)``.
|
|
1194
|
+
|
|
1195
|
+
Measured on Prover9 2026-8A: the LAST ``set(prolog_style_variables)`` or
|
|
1196
|
+
``clear(prolog_style_variables)`` of the file decides for EVERY formula of it, the ones that
|
|
1197
|
+
come before it included (a flag is read before any formula is interpreted), and a file that
|
|
1198
|
+
never sets it reads Prover9's default. Only a directive outside the lists counts.
|
|
1199
|
+
"""
|
|
1200
|
+
prolog, in_list = False, False
|
|
1201
|
+
for body in statements:
|
|
1202
|
+
kind = _statement_kind(body, in_list)
|
|
1203
|
+
if kind == "directive":
|
|
1204
|
+
flag = _P9_VARIABLE_FLAG_RE.match(body)
|
|
1205
|
+
if flag is not None:
|
|
1206
|
+
prolog = flag.group(1) == "set"
|
|
1207
|
+
elif kind == "header":
|
|
1208
|
+
in_list = True
|
|
1209
|
+
elif kind == "end":
|
|
1210
|
+
in_list = False
|
|
1211
|
+
return prolog
|
|
1212
|
+
|
|
1213
|
+
|
|
1214
|
+
def parse_prover9_problem(text: str) -> list:
|
|
1215
|
+
"""Parse a whole Prover9 / LADR input file into a list of :class:`Prover9Formula`.
|
|
1216
|
+
|
|
1217
|
+
Reads ``set`` / ``clear`` / ``assign`` directives (recognised and skipped),
|
|
1218
|
+
``op(precedence, type, symbol)`` directives (recognised and, when the
|
|
1219
|
+
declared operator is genuinely new, applied to every formula parsed after
|
|
1220
|
+
it — see the module docstring's "op(...) declarations" section for exactly
|
|
1221
|
+
what is applied and what is refused), ``formulas(LIST). … end_of_list.``
|
|
1222
|
+
blocks, and bare top-level ``formula.`` statements (``%`` line comments are
|
|
1223
|
+
ignored). Each formula is returned tagged with its list name as ``role``
|
|
1224
|
+
(``""`` for a bare top-level formula), in source order.
|
|
1225
|
+
|
|
1226
|
+
Statements are scanned deterministically (split on the terminating ``.``, with a
|
|
1227
|
+
``.`` inside a decimal number or inside a double-quoted symbol kept; a ``%`` inside
|
|
1228
|
+
a quoted symbol is no comment), and each formula is parsed by
|
|
1229
|
+
:func:`parse_prover9` under whatever ``op(...)`` declarations are active at that
|
|
1230
|
+
point in the file (a formula using a name before its own ``op(...)`` directive
|
|
1231
|
+
sees it as an ordinary, undeclared name, not as an operator). A ``formulas(...)``
|
|
1232
|
+
header with no matching ``end_of_list.``, a stray ``end_of_list.``, or a
|
|
1233
|
+
malformed header is a hard error — unlike a grammar that could silently
|
|
1234
|
+
reinterpret an unterminated list as bare formulas. A header is the call
|
|
1235
|
+
``formulas(NAME)`` with ONE argument: inside a list a call with two or more is an atom of the
|
|
1236
|
+
predicate ``formulas`` (``formulas(alpha, beta).``, which Prover9 reads as one and the writer of
|
|
1237
|
+
this kit writes for such a predicate), and so is a statement that goes on after the call;
|
|
1238
|
+
outside every list a call with two or more is a malformed header, as it is for Prover9
|
|
1239
|
+
("Unrecognized command or list").
|
|
1240
|
+
|
|
1241
|
+
Names are read under the convention the file sets (see the module docstring): the last
|
|
1242
|
+
``set(prolog_style_variables)`` or ``clear(prolog_style_variables)`` of the file decides for
|
|
1243
|
+
all of its formulas, and a file that never sets it reads Prover9's default, where a name
|
|
1244
|
+
that no quantifier binds is a variable when it begins with ``u`` to ``z``.
|
|
1245
|
+
|
|
1246
|
+
Args:
|
|
1247
|
+
text: the contents of a Prover9 problem file.
|
|
1248
|
+
|
|
1249
|
+
Returns:
|
|
1250
|
+
A list of :class:`Prover9Formula` ``(role, formula)`` records.
|
|
1251
|
+
|
|
1252
|
+
Raises:
|
|
1253
|
+
Prover9ParsingError: if the text is not a well-formed Prover9 problem (within
|
|
1254
|
+
the supported subset; an individual formula is parsed by
|
|
1255
|
+
:func:`parse_prover9`), or an ``op(...)`` directive is malformed or
|
|
1256
|
+
refused.
|
|
1257
|
+
"""
|
|
1258
|
+
stripped = _P9_COMMENT_RE.sub(lambda m: m.group(1) or "", text)
|
|
1259
|
+
if stripped.count('"') % 2:
|
|
1260
|
+
raise Prover9ParsingError(
|
|
1261
|
+
"SYNTAX_ERROR: a double quote without its closing double quote "
|
|
1262
|
+
"(a quoted symbol is not closed)")
|
|
1263
|
+
statements = []
|
|
1264
|
+
pos = 0
|
|
1265
|
+
for match in _P9_STATEMENT_RE.finditer(stripped):
|
|
1266
|
+
if match.start() != pos:
|
|
1267
|
+
break # text was skipped, so a statement is open: the leftover check reports it
|
|
1268
|
+
pos = match.end()
|
|
1269
|
+
body = match.group(0).strip()[:-1].strip() # drop the terminating '.'
|
|
1270
|
+
if body:
|
|
1271
|
+
statements.append(body)
|
|
1272
|
+
prolog_style = _prolog_style_of(statements)
|
|
1273
|
+
records = []
|
|
1274
|
+
role = None # the open formulas(...) list name, or None at top level
|
|
1275
|
+
active_ops = () # _CustomOp records declared so far, in order
|
|
1276
|
+
spellings: dict = {} # symbol -> how it was written so far (see _refuse_two_spellings)
|
|
1277
|
+
for body in statements:
|
|
1278
|
+
kind = _statement_kind(body, role is not None)
|
|
1279
|
+
if kind == "directive":
|
|
1280
|
+
directive = _P9_HEAD_RE.match(body)
|
|
1281
|
+
if directive is not None and directive.group(1) == "op":
|
|
1282
|
+
new_ops = _parse_op_directive(body)
|
|
1283
|
+
active_ops = active_ops + tuple(new_ops)
|
|
1284
|
+
_build_custom_grammar(active_ops) # validates now; result is cached
|
|
1285
|
+
continue # top-level flag/op directive: skip
|
|
1286
|
+
if kind == "header":
|
|
1287
|
+
list_m = _P9_FORMULAS_RE.match(body)
|
|
1288
|
+
if not list_m:
|
|
1289
|
+
raise Prover9ParsingError(
|
|
1290
|
+
f"SYNTAX_ERROR: malformed 'formulas(...)' header: {body!r}")
|
|
1291
|
+
if role is not None:
|
|
1292
|
+
raise Prover9ParsingError(
|
|
1293
|
+
"SYNTAX_ERROR: nested 'formulas(...)' list "
|
|
1294
|
+
f"(the '{role}' list was not closed by 'end_of_list.')")
|
|
1295
|
+
role = list_m.group(1)
|
|
1296
|
+
continue
|
|
1297
|
+
if kind == "end":
|
|
1298
|
+
if role is None:
|
|
1299
|
+
raise Prover9ParsingError(
|
|
1300
|
+
"SYNTAX_ERROR: 'end_of_list.' without an open 'formulas(...)' list")
|
|
1301
|
+
role = None
|
|
1302
|
+
continue
|
|
1303
|
+
records.append(Prover9Formula(
|
|
1304
|
+
role or "", _parse_formula(body, active_ops, spellings, prolog_style)))
|
|
1305
|
+
if role is not None:
|
|
1306
|
+
raise Prover9ParsingError(
|
|
1307
|
+
f"SYNTAX_ERROR: 'formulas({role})' list not closed by 'end_of_list.'")
|
|
1308
|
+
leftover = stripped[pos:].strip()
|
|
1309
|
+
if leftover:
|
|
1310
|
+
raise Prover9ParsingError(
|
|
1311
|
+
f"SYNTAX_ERROR: unterminated Prover9 statement (missing '.'): {leftover[:60]!r}")
|
|
1312
|
+
return records
|
|
1313
|
+
|
|
1314
|
+
|
|
1315
|
+
def load_prover9(path: str) -> list:
|
|
1316
|
+
"""Read a Prover9 / LADR input file and :func:`parse_prover9_problem` its contents.
|
|
1317
|
+
|
|
1318
|
+
Args:
|
|
1319
|
+
path: path to a Prover9 ``.in`` / ``.p9`` input file.
|
|
1320
|
+
|
|
1321
|
+
Returns:
|
|
1322
|
+
A list of :class:`Prover9Formula` records.
|
|
1323
|
+
"""
|
|
1324
|
+
with open(path, "r", encoding="utf-8") as handle:
|
|
1325
|
+
return parse_prover9_problem(handle.read())
|