unicode-logic-kit 0.31.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- unicode_logic_kit/__init__.py +385 -0
- unicode_logic_kit/__main__.py +520 -0
- unicode_logic_kit/_deadline.py +219 -0
- unicode_logic_kit/ace/__init__.py +126 -0
- unicode_logic_kit/ace/_align.py +135 -0
- unicode_logic_kit/ace/chem_lexicon.py +128 -0
- unicode_logic_kit/ace/drs_reader.py +570 -0
- unicode_logic_kit/ace/mapping.py +666 -0
- unicode_logic_kit/ace/reverse_modal.py +138 -0
- unicode_logic_kit/ace/runner.py +551 -0
- unicode_logic_kit/ace/translate.py +452 -0
- unicode_logic_kit/ace/verbalize.py +1070 -0
- unicode_logic_kit/api.py +1284 -0
- unicode_logic_kit/atp/__init__.py +177 -0
- unicode_logic_kit/atp/_ascii_names.py +113 -0
- unicode_logic_kit/atp/_html.py +72 -0
- unicode_logic_kit/atp/_substructural_input.py +228 -0
- unicode_logic_kit/atp/_tff_problem.py +715 -0
- unicode_logic_kit/atp/_tptp_problem.py +1111 -0
- unicode_logic_kit/atp/_writer_support.py +289 -0
- unicode_logic_kit/atp/clingo_backend.py +1180 -0
- unicode_logic_kit/atp/cvc5_backend.py +1385 -0
- unicode_logic_kit/atp/eprover_backend.py +732 -0
- unicode_logic_kit/atp/finite_domain.py +1055 -0
- unicode_logic_kit/atp/fitch.py +1547 -0
- unicode_logic_kit/atp/fitch_search.py +551 -0
- unicode_logic_kit/atp/hets_backend.py +339 -0
- unicode_logic_kit/atp/hybrid_down.py +120 -0
- unicode_logic_kit/atp/incremental.py +250 -0
- unicode_logic_kit/atp/kripke_enum.py +741 -0
- unicode_logic_kit/atp/lambek.py +436 -0
- unicode_logic_kit/atp/leo3_backend.py +332 -0
- unicode_logic_kit/atp/linear.py +738 -0
- unicode_logic_kit/atp/lj.py +705 -0
- unicode_logic_kit/atp/logic_backends.py +566 -0
- unicode_logic_kit/atp/ltl_tableau.py +1084 -0
- unicode_logic_kit/atp/minizinc_backend.py +1402 -0
- unicode_logic_kit/atp/modal_tableau.py +1382 -0
- unicode_logic_kit/atp/nanocop_backend.py +410 -0
- unicode_logic_kit/atp/portfolio.py +489 -0
- unicode_logic_kit/atp/protocol.py +1803 -0
- unicode_logic_kit/atp/prover9_entailment.py +1153 -0
- unicode_logic_kit/atp/resolution.py +1376 -0
- unicode_logic_kit/atp/resolution_check.py +1114 -0
- unicode_logic_kit/atp/sequent.py +1050 -0
- unicode_logic_kit/atp/tableau.py +921 -0
- unicode_logic_kit/atp/tableau_check.py +543 -0
- unicode_logic_kit/atp/tptp_ncl.py +811 -0
- unicode_logic_kit/atp/tptp_tff.py +1546 -0
- unicode_logic_kit/atp/tstp.py +1333 -0
- unicode_logic_kit/atp/tstp_check.py +1096 -0
- unicode_logic_kit/atp/twee_backend.py +236 -0
- unicode_logic_kit/atp/twee_check.py +711 -0
- unicode_logic_kit/atp/twee_entailment.py +953 -0
- unicode_logic_kit/atp/vampire_entailment.py +540 -0
- unicode_logic_kit/atp/z3_arith.py +470 -0
- unicode_logic_kit/atp/z3_equivalence.py +36 -0
- unicode_logic_kit/atp/z3_fuzzy.py +362 -0
- unicode_logic_kit/atp/z3_input.py +500 -0
- unicode_logic_kit/atp/z3_models.py +208 -0
- unicode_logic_kit/chem/__init__.py +88 -0
- unicode_logic_kit/chem/_naming.py +284 -0
- unicode_logic_kit/chem/cache.py +185 -0
- unicode_logic_kit/chem/interop.py +244 -0
- unicode_logic_kit/chem/mol.py +525 -0
- unicode_logic_kit/chem/signature.py +112 -0
- unicode_logic_kit/comorphism.py +497 -0
- unicode_logic_kit/dl/__init__.py +384 -0
- unicode_logic_kit/dl/classification.py +227 -0
- unicode_logic_kit/dl/concepts.py +632 -0
- unicode_logic_kit/dl/datatypes.py +818 -0
- unicode_logic_kit/dl/owl_functional.py +2433 -0
- unicode_logic_kit/dl/owl_manchester.py +1637 -0
- unicode_logic_kit/dl/owl_reasoner.py +790 -0
- unicode_logic_kit/dl/parser.py +391 -0
- unicode_logic_kit/dl/tableau.py +4048 -0
- unicode_logic_kit/dl/translate.py +2704 -0
- unicode_logic_kit/drt/__init__.py +94 -0
- unicode_logic_kit/drt/export.py +179 -0
- unicode_logic_kit/drt/nodes.py +506 -0
- unicode_logic_kit/drt/parser.py +965 -0
- unicode_logic_kit/drt/resolve.py +195 -0
- unicode_logic_kit/drt/reverse.py +175 -0
- unicode_logic_kit/eval/__init__.py +106 -0
- unicode_logic_kit/eval/batch.py +382 -0
- unicode_logic_kit/eval/canonical.py +663 -0
- unicode_logic_kit/eval/chem_batch.py +606 -0
- unicode_logic_kit/eval/converses.py +200 -0
- unicode_logic_kit/eval/datasets/__init__.py +136 -0
- unicode_logic_kit/eval/datasets/_base.py +263 -0
- unicode_logic_kit/eval/datasets/_proofwriter_proof.py +422 -0
- unicode_logic_kit/eval/datasets/c3po.py +678 -0
- unicode_logic_kit/eval/datasets/folio.py +158 -0
- unicode_logic_kit/eval/datasets/fracas.py +418 -0
- unicode_logic_kit/eval/datasets/groves.py +191 -0
- unicode_logic_kit/eval/datasets/logicbench.py +467 -0
- unicode_logic_kit/eval/datasets/logicnli.py +303 -0
- unicode_logic_kit/eval/datasets/malls.py +133 -0
- unicode_logic_kit/eval/datasets/pfolio.py +594 -0
- unicode_logic_kit/eval/datasets/pmb.py +242 -0
- unicode_logic_kit/eval/datasets/prontoqa.py +611 -0
- unicode_logic_kit/eval/datasets/proofwriter.py +1431 -0
- unicode_logic_kit/eval/datasets/proverqa.py +674 -0
- unicode_logic_kit/eval/datasets/willow.py +478 -0
- unicode_logic_kit/eval/equivalence.py +466 -0
- unicode_logic_kit/eval/exercise_gen.py +533 -0
- unicode_logic_kit/eval/explain.py +791 -0
- unicode_logic_kit/eval/generality.py +750 -0
- unicode_logic_kit/eval/metric_hf.py +458 -0
- unicode_logic_kit/eval/predicate_match.py +343 -0
- unicode_logic_kit/eval/theory_check.py +1170 -0
- unicode_logic_kit/eval/validate.py +306 -0
- unicode_logic_kit/fol/__init__.py +177 -0
- unicode_logic_kit/fol/_atom_keys.py +510 -0
- unicode_logic_kit/fol/_fol_nodes.py +3586 -0
- unicode_logic_kit/fol/_free_parameters.py +105 -0
- unicode_logic_kit/fol/_ho_nodes.py +448 -0
- unicode_logic_kit/fol/_hybrid_nodes.py +308 -0
- unicode_logic_kit/fol/_identifiers.py +1091 -0
- unicode_logic_kit/fol/_lambek_nodes.py +112 -0
- unicode_logic_kit/fol/_linear_nodes.py +352 -0
- unicode_logic_kit/fol/_modal_nodes.py +1467 -0
- unicode_logic_kit/fol/_msfl_nodes.py +2196 -0
- unicode_logic_kit/fol/_numeral_symbols.py +231 -0
- unicode_logic_kit/fol/_so_nodes.py +200 -0
- unicode_logic_kit/fol/_symbol_names.py +81 -0
- unicode_logic_kit/fol/_team_nodes.py +181 -0
- unicode_logic_kit/fol/_tptp_symbols.py +551 -0
- unicode_logic_kit/fol/_truth_constants.py +117 -0
- unicode_logic_kit/fol/casl_export.py +1135 -0
- unicode_logic_kit/fol/casl_import.py +929 -0
- unicode_logic_kit/fol/derivation.py +367 -0
- unicode_logic_kit/fol/dialect_detect.py +70 -0
- unicode_logic_kit/fol/dialect_repair.py +537 -0
- unicode_logic_kit/fol/frames.py +637 -0
- unicode_logic_kit/fol/grammars/terminals.lark +31 -0
- unicode_logic_kit/fol/lambda_tools.py +297 -0
- unicode_logic_kit/fol/latex_input.py +429 -0
- unicode_logic_kit/fol/modal_translation.py +944 -0
- unicode_logic_kit/fol/msflparser.py +1033 -0
- unicode_logic_kit/fol/naming.py +422 -0
- unicode_logic_kit/fol/nodes.py +241 -0
- unicode_logic_kit/fol/normalforms.py +492 -0
- unicode_logic_kit/fol/pal.py +287 -0
- unicode_logic_kit/fol/prolog_export.py +566 -0
- unicode_logic_kit/fol/prolog_input.py +505 -0
- unicode_logic_kit/fol/prover9_input.py +1325 -0
- unicode_logic_kit/fol/qml.py +1760 -0
- unicode_logic_kit/fol/qmltp_input.py +525 -0
- unicode_logic_kit/fol/sanitize.py +221 -0
- unicode_logic_kit/fol/serialize.py +79 -0
- unicode_logic_kit/fol/signature.py +1290 -0
- unicode_logic_kit/fol/simplify_check.py +544 -0
- unicode_logic_kit/fol/spans.py +594 -0
- unicode_logic_kit/fol/tptp_input.py +1503 -0
- unicode_logic_kit/fol/tptp_repair.py +941 -0
- unicode_logic_kit/fol/unification.py +157 -0
- unicode_logic_kit/fol/verbalize.py +263 -0
- unicode_logic_kit/hets/__init__.py +163 -0
- unicode_logic_kit/hets/bridge.py +142 -0
- unicode_logic_kit/hets/client.py +748 -0
- unicode_logic_kit/hets/docker.py +420 -0
- unicode_logic_kit/hets/dol.py +712 -0
- unicode_logic_kit/hets/haskell_json.py +355 -0
- unicode_logic_kit/hets/owl_backend.py +794 -0
- unicode_logic_kit/hets/owl_cli.py +598 -0
- unicode_logic_kit/hets/symbols.py +512 -0
- unicode_logic_kit/hol/__init__.py +140 -0
- unicode_logic_kit/hol/_ho_common.py +323 -0
- unicode_logic_kit/hol/_isabelle_binders.py +125 -0
- unicode_logic_kit/hol/classical.py +812 -0
- unicode_logic_kit/hol/deepshallow/__init__.py +45 -0
- unicode_logic_kit/hol/deepshallow/_common.py +177 -0
- unicode_logic_kit/hol/deepshallow/conditional.py +225 -0
- unicode_logic_kit/hol/deepshallow/intuitionistic.py +181 -0
- unicode_logic_kit/hol/deepshallow/modal.py +217 -0
- unicode_logic_kit/hol/deepshallow/qml.py +406 -0
- unicode_logic_kit/hol/deepshallow/relevant.py +206 -0
- unicode_logic_kit/hol/free.py +753 -0
- unicode_logic_kit/hol/goedel.py +336 -0
- unicode_logic_kit/hol/ho_modal.py +1743 -0
- unicode_logic_kit/hol/intuitionistic.py +403 -0
- unicode_logic_kit/hol/isabelle_conditional.py +593 -0
- unicode_logic_kit/hol/isabelle_modal.py +1908 -0
- unicode_logic_kit/hol/isabelle_relevant.py +412 -0
- unicode_logic_kit/hol/isabelle_runner.py +1147 -0
- unicode_logic_kit/hol/isabelle_substructural.py +884 -0
- unicode_logic_kit/hol/lean.py +1018 -0
- unicode_logic_kit/hol/manyvalued.py +921 -0
- unicode_logic_kit/hol/secondorder.py +687 -0
- unicode_logic_kit/hol/thf_modal.py +941 -0
- unicode_logic_kit/hol/thirdorder.py +397 -0
- unicode_logic_kit/ilp/__init__.py +89 -0
- unicode_logic_kit/ilp/readback.py +389 -0
- unicode_logic_kit/ilp/separation.py +153 -0
- unicode_logic_kit/ilp/task.py +730 -0
- unicode_logic_kit/logic.py +163 -0
- unicode_logic_kit/mcp/__init__.py +28 -0
- unicode_logic_kit/mcp/__main__.py +5 -0
- unicode_logic_kit/mcp/chem_tools.py +1031 -0
- unicode_logic_kit/mcp/server.py +2453 -0
- unicode_logic_kit/mcp/syntax_spec.py +681 -0
- unicode_logic_kit/prob/__init__.py +53 -0
- unicode_logic_kit/prob/_bdd.py +225 -0
- unicode_logic_kit/prob/_column_gen.py +668 -0
- unicode_logic_kit/prob/distribution.py +686 -0
- unicode_logic_kit/prob/nilsson.py +470 -0
- unicode_logic_kit/py.typed +0 -0
- unicode_logic_kit/semantics/__init__.py +137 -0
- unicode_logic_kit/semantics/_modal_reject.py +156 -0
- unicode_logic_kit/semantics/action_models.py +466 -0
- unicode_logic_kit/semantics/asp_models.py +1200 -0
- unicode_logic_kit/semantics/conditional.py +580 -0
- unicode_logic_kit/semantics/dynamic_epistemic.py +95 -0
- unicode_logic_kit/semantics/free_logic.py +913 -0
- unicode_logic_kit/semantics/fuzzy.py +384 -0
- unicode_logic_kit/semantics/fuzzy_kripke.py +442 -0
- unicode_logic_kit/semantics/intuitionistic.py +581 -0
- unicode_logic_kit/semantics/kripke.py +1139 -0
- unicode_logic_kit/semantics/manyvalued.py +580 -0
- unicode_logic_kit/semantics/matrix.py +342 -0
- unicode_logic_kit/semantics/model_eval.py +1135 -0
- unicode_logic_kit/semantics/modelfinder.py +1036 -0
- unicode_logic_kit/semantics/nonmonotonic.py +372 -0
- unicode_logic_kit/semantics/relevant.py +331 -0
- unicode_logic_kit/semantics/secondorder.py +657 -0
- unicode_logic_kit/semantics/structures.py +352 -0
- unicode_logic_kit/semantics/tarski.py +975 -0
- unicode_logic_kit/semantics/team.py +315 -0
- unicode_logic_kit/semantics/team_translation.py +416 -0
- unicode_logic_kit/semantics/thirdorder.py +358 -0
- unicode_logic_kit/semantics/tnorm.py +85 -0
- unicode_logic_kit/semantics/truthtable.py +201 -0
- unicode_logic_kit-0.31.0.dist-info/METADATA +333 -0
- unicode_logic_kit-0.31.0.dist-info/RECORD +237 -0
- unicode_logic_kit-0.31.0.dist-info/WHEEL +4 -0
- unicode_logic_kit-0.31.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,953 @@
|
|
|
1
|
+
"""Entailment checking via the Twee equational theorem prover, through WSL.
|
|
2
|
+
|
|
3
|
+
`Twee <https://nick8325.github.io/twee/>`_ is a Knuth-Bendix-completion-based
|
|
4
|
+
prover for pure UNIT EQUALITY: every axiom and the conjecture must be a
|
|
5
|
+
(possibly universally quantified) equation ``s = t``, or a conjunction of
|
|
6
|
+
such equations — nothing else (no disjunction, no implication, no predicates
|
|
7
|
+
other than ``=``). This module drives Twee exactly as far as that fragment
|
|
8
|
+
reaches: :func:`check_entailment_twee_detailed` raises ``NotImplementedError``
|
|
9
|
+
naming the offending premise/conclusion for anything outside it, mirroring
|
|
10
|
+
:mod:`atp.vampire_entailment`'s contract for the FOL-vs-non-FOL boundary.
|
|
11
|
+
|
|
12
|
+
Twee has no native Windows build; on this machine (and presumably any other
|
|
13
|
+
Windows dev box following the same install recipe) it lives in WSL at
|
|
14
|
+
``~/.local/bin/twee`` (installed via ``ghcup``/``cabal``, Twee 2.6.1). Every
|
|
15
|
+
invocation therefore goes through ``wsl.exe -e bash -lc "<cmd> ..."`` by
|
|
16
|
+
default (``use_wsl=True``) — the ``bash -lc`` wrapper is required, not
|
|
17
|
+
cosmetic: invoking ``wsl.exe ~/.local/bin/twee`` directly lets the OUTER
|
|
18
|
+
(Windows/Git-Bash) shell expand the ``~`` before ``wsl.exe`` ever sees it,
|
|
19
|
+
which resolves to nonsense on the Windows side. Wrapping the whole command in
|
|
20
|
+
a WSL-side login shell (``bash -lc "..."``) makes WSL's own bash expand the
|
|
21
|
+
``~`` correctly. The command is overridable via ``$UFK_TWEE_CMD`` (defaults
|
|
22
|
+
to ``~/.local/bin/twee``), and ``use_wsl=False`` switches to a bare
|
|
23
|
+
subprocess call for a machine with a native ``twee`` on ``PATH`` (e.g. a
|
|
24
|
+
Linux/macOS host) — the same ``use_wsl`` toggle :mod:`atp.vampire_entailment`
|
|
25
|
+
uses for Vampire, just defaulted the other way since Twee's only known
|
|
26
|
+
install on THIS kit's dev machines is the WSL one.
|
|
27
|
+
|
|
28
|
+
Windows temp-file paths are translated to their WSL ``/mnt/...`` form via
|
|
29
|
+
``wslpath`` (identical technique to :mod:`atp.vampire_entailment`'s
|
|
30
|
+
``_to_wsl_path`` — re-implemented locally here rather than imported, so this
|
|
31
|
+
module has no dependency on the Vampire module).
|
|
32
|
+
|
|
33
|
+
Every run passes ``--quiet`` (Twee's own flag: "Only print essential
|
|
34
|
+
information"). This is not just cosmetic either — WITHOUT it, Twee's stdout
|
|
35
|
+
opens with a restated/flattened echo of the input problem and a running
|
|
36
|
+
completion trace ("Here is the input problem: ... 1. f(a) -> f5 ...") before
|
|
37
|
+
the actual result, which is irrelevant noise for parsing; ``--quiet`` was
|
|
38
|
+
verified live to produce exactly the axiom/lemma/goal/proof text documented
|
|
39
|
+
below and nothing else.
|
|
40
|
+
|
|
41
|
+
Observed output shapes (every one below was produced by a real ``twee
|
|
42
|
+
--quiet`` run on this machine, Twee 2.6.1, and is what
|
|
43
|
+
:func:`parse_twee_proof` is built to read — anything else raises
|
|
44
|
+
``ValueError``, never a guess):
|
|
45
|
+
|
|
46
|
+
1. **Theorem, no lemmas** (input ``![X]: f(f(X)) = X`` proving
|
|
47
|
+
``f(f(f(f(a)))) = a``)::
|
|
48
|
+
|
|
49
|
+
The conjecture is true! Here is a proof.
|
|
50
|
+
|
|
51
|
+
Axiom 1 (a): f(f(X)) = X.
|
|
52
|
+
|
|
53
|
+
Goal 1 (g): f(f(f(f(a)))) = a.
|
|
54
|
+
Proof:
|
|
55
|
+
f(f(f(f(a))))
|
|
56
|
+
= { by axiom 1 (a) }
|
|
57
|
+
f(f(a))
|
|
58
|
+
= { by axiom 1 (a) }
|
|
59
|
+
a
|
|
60
|
+
|
|
61
|
+
RESULT: Theorem (the conjecture is true).
|
|
62
|
+
|
|
63
|
+
The opening banner line is fixed and always present on a Theorem run —
|
|
64
|
+
even under ``--quiet`` ("only print ESSENTIAL information" apparently
|
|
65
|
+
includes it); :func:`parse_twee_proof` requires it verbatim before the
|
|
66
|
+
axiom block.
|
|
67
|
+
|
|
68
|
+
2. **Theorem with a lemma, and a reverse-direction (``R->L``) step**
|
|
69
|
+
(left-group axioms proving right-identity; banner line omitted below for
|
|
70
|
+
brevity — see shape 1 for where it goes)::
|
|
71
|
+
|
|
72
|
+
Axiom 1 (left_id): mult(e, X) = X.
|
|
73
|
+
Axiom 2 (left_inv): mult(inv(X), X) = e.
|
|
74
|
+
Axiom 3 (assoc): mult(mult(X, Y), Z) = mult(X, mult(Y, Z)).
|
|
75
|
+
|
|
76
|
+
Lemma 4: mult(inv(X), mult(X, Y)) = Y.
|
|
77
|
+
Proof:
|
|
78
|
+
mult(inv(X), mult(X, Y))
|
|
79
|
+
= { by axiom 3 (assoc) R->L }
|
|
80
|
+
mult(mult(inv(X), X), Y)
|
|
81
|
+
= { by axiom 2 (left_inv) }
|
|
82
|
+
mult(e, Y)
|
|
83
|
+
= { by axiom 1 (left_id) }
|
|
84
|
+
Y
|
|
85
|
+
|
|
86
|
+
Goal 1 (right_id): mult(x, e) = x.
|
|
87
|
+
Proof:
|
|
88
|
+
mult(x, e)
|
|
89
|
+
= { by lemma 4 R->L }
|
|
90
|
+
mult(inv(inv(x)), mult(inv(x), mult(x, e)))
|
|
91
|
+
= { by lemma 4 }
|
|
92
|
+
mult(inv(inv(x)), e)
|
|
93
|
+
= { by axiom 2 (left_inv) R->L }
|
|
94
|
+
mult(inv(inv(x)), mult(inv(x), x))
|
|
95
|
+
= { by lemma 4 }
|
|
96
|
+
x
|
|
97
|
+
|
|
98
|
+
RESULT: Theorem (the conjecture is true).
|
|
99
|
+
|
|
100
|
+
Grammar notes distilled from this and further examples (``inv(inv(X))=X``
|
|
101
|
+
needing 2 chained lemma applications with mixed directions; a 2-variable
|
|
102
|
+
commutation goal; a 3-conjunct tupled goal; see ``tests/test_twee.py``
|
|
103
|
+
for the exact captured texts used as fixtures):
|
|
104
|
+
|
|
105
|
+
* A citation is either ``{ by axiom N (name) }`` or ``{ by lemma N }``
|
|
106
|
+
(lemmas are NEVER named in a citation — only axioms carry a ``(name)``,
|
|
107
|
+
because that name is the one WE assigned when generating the TPTP
|
|
108
|
+
problem, see :func:`_generate_twee_input`), optionally suffixed
|
|
109
|
+
``R->L`` (apply the equation right-to-left).
|
|
110
|
+
* Axiom variables print in a CANONICAL ``X``/``Y``/``Z``... scheme, NOT
|
|
111
|
+
the original source names — verified by feeding an axiom variable named
|
|
112
|
+
``V`` and seeing it echoed as ``X``. A goal's variables, by contrast,
|
|
113
|
+
are Skolemized to a FRESH constant that is the ORIGINAL bound-variable
|
|
114
|
+
letter, lowercased (``W`` in the conjecture prints as ``w`` in the Goal
|
|
115
|
+
line and its proof) — this is why axioms are matched to the caller's
|
|
116
|
+
original premises up to alpha-equivalence (a name-blind bijection
|
|
117
|
+
check, see :mod:`atp.twee_check`), while goal variables are matched
|
|
118
|
+
structurally instead of by any assumed naming convention.
|
|
119
|
+
* Only axioms actually USED in the proof are listed in the header block —
|
|
120
|
+
an irrelevant extra axiom is silently omitted, so axiom numbers in the
|
|
121
|
+
proof are a purely LOCAL indexing scheme, unrelated to the caller's
|
|
122
|
+
premise-list position; :func:`_generate_twee_input` names every axiom
|
|
123
|
+
``premise_<i>`` (1-based, matching the caller's premise list) precisely
|
|
124
|
+
so the checker can still recover which original premise a citation
|
|
125
|
+
means, regardless of Twee's renumbering/omission.
|
|
126
|
+
* A premise that is a top-level conjunction of equations gets clausified
|
|
127
|
+
by Twee into one clause per conjunct (a ground conjunct that repeats an
|
|
128
|
+
earlier one is dropped); the clauses are named ``premise_<i>``,
|
|
129
|
+
``premise_<i>_1``, ``premise_<i>_2``, ... but NOT in the order of the
|
|
130
|
+
source: Twee numbers them in an order of its own (measured, Twee 2.6.1:
|
|
131
|
+
for ``aa = bb ∧ ∀x ff(x) = x`` the clause ``ff(X) = X`` is
|
|
132
|
+
``premise_1`` and ``aa = bb`` is ``premise_1_1``; for ``aa = bb ∧ aa = bb
|
|
133
|
+
∧ cc = dd`` the clause ``cc = dd`` is ``premise_1_1``). A name therefore
|
|
134
|
+
tells which PREMISE a clause comes from and nothing more, and
|
|
135
|
+
:mod:`atp.twee_check` decides which conjunct of that premise a restated
|
|
136
|
+
axiom is by what it says.
|
|
137
|
+
* A conjunctive CONCLUSION is encoded by Twee as a single equation
|
|
138
|
+
between ``tuple(...)`` applications: ``![X]: (P(X)) & (Q(X))`` proves
|
|
139
|
+
as ``Goal 1 (goal): tuple(P(sk1), Q(sk2)) = tuple(true, true)``-shaped
|
|
140
|
+
(schematically) — and, importantly, each conjunct receives an
|
|
141
|
+
INDEPENDENT fresh Skolem constant even when the conjuncts share a
|
|
142
|
+
source variable name (verified: a 2-conjunct goal over a shared ``X``
|
|
143
|
+
Skolemized to ``x2`` in the first slot and ``x`` in the second) — sound,
|
|
144
|
+
since ``∀X.(P(X)∧Q(X))`` is equivalent to ``(∀X.P(X))∧(∀X.Q(X))``, two
|
|
145
|
+
independent universal claims. A single-conjunct conclusion is NEVER
|
|
146
|
+
tuple-wrapped.
|
|
147
|
+
|
|
148
|
+
3. **CounterSatisfiable** (the conjecture does not follow) — with
|
|
149
|
+
``--quiet``, this is the ENTIRE output, no axiom/proof text at all::
|
|
150
|
+
|
|
151
|
+
RESULT: CounterSatisfiable (the conjecture is false).
|
|
152
|
+
|
|
153
|
+
(Likewise ``Satisfiable``/``Unsatisfiable`` for an axiom-only problem
|
|
154
|
+
with no conjecture — out of scope for this module, which always emits
|
|
155
|
+
exactly one conjecture, but confirmed live for documentation.)
|
|
156
|
+
|
|
157
|
+
4. **GaveUp** — Twee's own resource budget (``--max-time``, ``--max-cps``,
|
|
158
|
+
etc. — none of which this module passes by default, so this is only
|
|
159
|
+
reachable if a caller forwards such a flag) was exhausted before
|
|
160
|
+
completion could decide the problem either way::
|
|
161
|
+
|
|
162
|
+
RESULT: GaveUp (couldn't solve the problem).
|
|
163
|
+
|
|
164
|
+
5. **No RESULT line at all** — a genuine subprocess timeout (Twee's own
|
|
165
|
+
resource limits are UNLIMITED by default, so an undecided problem just
|
|
166
|
+
runs forever until this module's own ``timeout`` kills the process; the
|
|
167
|
+
partial stdout at that point, if any, is empty since Twee buffers its
|
|
168
|
+
report until the very end), or a Twee-side parse/usage error (printed to
|
|
169
|
+
STDERR with a non-zero exit and nothing on stdout — verified live with a
|
|
170
|
+
malformed conjecture). Both come back as ``status="Unknown"`` from
|
|
171
|
+
:func:`check_entailment_twee_detailed`; ``result["timed_out"]``
|
|
172
|
+
distinguishes the two.
|
|
173
|
+
|
|
174
|
+
**One name at two arities.** The kit reads a name used at two arities (the constant
|
|
175
|
+
``ff`` and the unary function ``ff(x)``, or ``ff(x)`` and ``ff(x, y)``) as two symbols,
|
|
176
|
+
and so does every prover the TPTP writer feeds but Twee, which types a symbol by its name
|
|
177
|
+
alone and stops with ``Type mismatch in term 'ff': Constant ff has arity 1 but was applied
|
|
178
|
+
to 0 arguments`` (measured, Twee 2.6.1). So :func:`check_entailment_twee_detailed` writes
|
|
179
|
+
each arity but the first of such a name under a name of its own, one that no symbol of the
|
|
180
|
+
problem has (:func:`_separate_arities`), and hands everything Twee prints back under the
|
|
181
|
+
name the caller used: the parsed proof and the text of ``raw_output`` speak of the problem
|
|
182
|
+
as it was asked, and the proof checker compares it with the caller's own premises.
|
|
183
|
+
|
|
184
|
+
Public API: :func:`twee_available`, :func:`check_entailment_twee_detailed`,
|
|
185
|
+
the proof data classes (:class:`TweeEquation`, :class:`TweeCitation`,
|
|
186
|
+
:class:`TweeChain`, :class:`TweeAxiom`, :class:`TweeLemma`, :class:`TweeGoal`,
|
|
187
|
+
:class:`TweeProof`), and :func:`parse_twee_proof`.
|
|
188
|
+
"""
|
|
189
|
+
|
|
190
|
+
import os
|
|
191
|
+
import re
|
|
192
|
+
import subprocess
|
|
193
|
+
import tempfile
|
|
194
|
+
from dataclasses import dataclass
|
|
195
|
+
from typing import Callable, Dict, List, Optional, Tuple
|
|
196
|
+
|
|
197
|
+
from ..fol._fol_nodes import _numeral_from_text
|
|
198
|
+
from ..fol._identifiers import symbol_names
|
|
199
|
+
from ..fol.nodes import (
|
|
200
|
+
Atom, And, Constant, Function, Measure, Node, Number, Quantifier, SortedConstant, Variable,
|
|
201
|
+
)
|
|
202
|
+
from ._ascii_names import reverse_map_text
|
|
203
|
+
from ._tptp_problem import TptpNameMap, apply_reverse_tptp, generate_tptp_problem_with_mapping
|
|
204
|
+
|
|
205
|
+
__all__ = [
|
|
206
|
+
"twee_available", "check_entailment_twee_detailed",
|
|
207
|
+
"TweeEquation", "TweeCitation", "TweeChain",
|
|
208
|
+
"TweeAxiom", "TweeLemma", "TweeGoal", "TweeProof",
|
|
209
|
+
"parse_twee_proof", "reverse_map_twee_proof",
|
|
210
|
+
]
|
|
211
|
+
|
|
212
|
+
_DEFAULT_TWEE_CMD = "~/.local/bin/twee"
|
|
213
|
+
|
|
214
|
+
|
|
215
|
+
def _twee_cmd(explicit: Optional[str]) -> str:
|
|
216
|
+
"""Resolve the command used to invoke Twee: explicit arg > ``$UFK_TWEE_CMD`` > default."""
|
|
217
|
+
return explicit or os.environ.get("UFK_TWEE_CMD") or _DEFAULT_TWEE_CMD
|
|
218
|
+
|
|
219
|
+
|
|
220
|
+
# ---------------------------------------------------------------------------
|
|
221
|
+
# The equational fragment
|
|
222
|
+
# ---------------------------------------------------------------------------
|
|
223
|
+
|
|
224
|
+
def _is_equational(node: Node) -> bool:
|
|
225
|
+
"""True iff ``node`` is a (forall-closed) equation or conjunction of equations.
|
|
226
|
+
|
|
227
|
+
Twee decides unit equality only: an ``Atom`` with predicate ``"="`` and
|
|
228
|
+
exactly two arguments, universally quantified any number of times
|
|
229
|
+
(``∀``/``forall`` — an ``∃`` or any other quantifier is out of fragment),
|
|
230
|
+
and/or conjoined with ``And``. Anything else (disjunction, negation,
|
|
231
|
+
implication, a non-``=`` predicate, a modal/second-order/substructural
|
|
232
|
+
node, ...) is not.
|
|
233
|
+
"""
|
|
234
|
+
if isinstance(node, Quantifier):
|
|
235
|
+
return node.type in ("forall", "∀") and _is_equational(node.formula)
|
|
236
|
+
if isinstance(node, And):
|
|
237
|
+
return _is_equational(node.left) and _is_equational(node.right)
|
|
238
|
+
if isinstance(node, Atom):
|
|
239
|
+
return node.predicate == "=" and len(node.args) == 2
|
|
240
|
+
return False
|
|
241
|
+
|
|
242
|
+
|
|
243
|
+
def _require_equational(node: Node, role: str) -> None:
|
|
244
|
+
"""Raise ``NotImplementedError`` naming ``role`` if ``node`` is not equational."""
|
|
245
|
+
if not _is_equational(node):
|
|
246
|
+
raise NotImplementedError(
|
|
247
|
+
f"twee: {role} is outside the equational fragment Twee decides — "
|
|
248
|
+
f"every premise and the conclusion must be a (forall-closed) "
|
|
249
|
+
f"equation `s = t` or a conjunction of such equations, got "
|
|
250
|
+
f"{role}={node.to_unicode_str()!r}")
|
|
251
|
+
|
|
252
|
+
|
|
253
|
+
# ---------------------------------------------------------------------------
|
|
254
|
+
# TPTP problem generation
|
|
255
|
+
# ---------------------------------------------------------------------------
|
|
256
|
+
|
|
257
|
+
def _generate_twee_input(premises: List[Node], conclusion: Node) -> str:
|
|
258
|
+
"""Build a TPTP ``fof`` problem string, one axiom per premise plus one conjecture.
|
|
259
|
+
|
|
260
|
+
A thin wrapper over the shared :func:`atp._tptp_problem.generate_tptp_problem`
|
|
261
|
+
(also used by :mod:`atp.vampire_entailment` and :mod:`atp.eprover_backend`,
|
|
262
|
+
which build the identical problem shape; this module calls its sibling
|
|
263
|
+
:func:`atp._tptp_problem.generate_tptp_problem_with_mapping`, which writes the
|
|
264
|
+
same text and also returns the name map): each premise becomes
|
|
265
|
+
``fof(premise_<i>, axiom, <tptp>).`` (1-based) and the conclusion becomes
|
|
266
|
+
``fof(goal, conjecture, <tptp>).``. The ``premise_<i>`` naming is not
|
|
267
|
+
cosmetic here — it is the anchor :mod:`atp.twee_check` uses to recover
|
|
268
|
+
which original premise an axiom citation in Twee's proof refers to (see
|
|
269
|
+
the module docstring's naming-scheme notes). ``generate_tptp_problem``'s
|
|
270
|
+
cross-formula symbol-collision guard can raise ``NotImplementedError``
|
|
271
|
+
before any TPTP text is produced — see ``_tptp_problem``'s module
|
|
272
|
+
docstring. A name used at two arities is written as one name per arity
|
|
273
|
+
first (:func:`_separate_arities`), so that Twee reads each as the symbol it is.
|
|
274
|
+
"""
|
|
275
|
+
problem, _name_map, _restore = _twee_problem(premises, conclusion)
|
|
276
|
+
return problem
|
|
277
|
+
|
|
278
|
+
|
|
279
|
+
def _symbol_arities(formulas: List[Node]) -> Dict[str, List[int]]:
|
|
280
|
+
"""Every arity each function/constant name of ``formulas`` is used at, in the order the
|
|
281
|
+
arities first occur (a constant, a sorted constant and a function of no arguments are
|
|
282
|
+
the symbol of their name at arity 0; a :class:`~unicode_logic_kit.fol.nodes.Measure` is
|
|
283
|
+
the binary function ``measure``)."""
|
|
284
|
+
arities: Dict[str, List[int]] = {}
|
|
285
|
+
for formula in formulas:
|
|
286
|
+
for node in formula.walk():
|
|
287
|
+
if isinstance(node, Function):
|
|
288
|
+
name, arity = node.name, len(node.args)
|
|
289
|
+
elif isinstance(node, (Constant, SortedConstant)):
|
|
290
|
+
name, arity = node.name, 0
|
|
291
|
+
elif isinstance(node, Measure):
|
|
292
|
+
name, arity = "measure", 2
|
|
293
|
+
else:
|
|
294
|
+
continue
|
|
295
|
+
seen = arities.setdefault(name, [])
|
|
296
|
+
if arity not in seen:
|
|
297
|
+
seen.append(arity)
|
|
298
|
+
return arities
|
|
299
|
+
|
|
300
|
+
|
|
301
|
+
def _rename_by_arity(node: Node, table: Dict[Tuple[str, int], str]) -> Node:
|
|
302
|
+
"""``node`` with every function/constant ``(name, arity)`` found in ``table`` written as
|
|
303
|
+
the name ``table`` gives it. Everything else is rebuilt equal; a function of no arguments
|
|
304
|
+
that is renamed is the constant of its new name."""
|
|
305
|
+
if isinstance(node, (Variable, Number)):
|
|
306
|
+
return node
|
|
307
|
+
if isinstance(node, Function):
|
|
308
|
+
args = tuple(_rename_by_arity(a, table) for a in node.args)
|
|
309
|
+
new = table.get((node.name, len(args)))
|
|
310
|
+
if not args:
|
|
311
|
+
return node if new is None else Constant(new)
|
|
312
|
+
return Function(node.name if new is None else new, args)
|
|
313
|
+
if isinstance(node, Constant):
|
|
314
|
+
new = table.get((node.name, 0))
|
|
315
|
+
return node if new is None else Constant(new)
|
|
316
|
+
if isinstance(node, SortedConstant):
|
|
317
|
+
new = table.get((node.name, 0))
|
|
318
|
+
return node if new is None else SortedConstant(new, node.sort)
|
|
319
|
+
if isinstance(node, Measure) and ("measure", 2) in table:
|
|
320
|
+
return Function(table[("measure", 2)], (_rename_by_arity(node.entity, table),
|
|
321
|
+
_rename_by_arity(node.dimension, table)))
|
|
322
|
+
return node.map_children(lambda child: _rename_by_arity(child, table))
|
|
323
|
+
|
|
324
|
+
|
|
325
|
+
def _separate_arities(formulas: List[Node]) -> Tuple[List[Node], Dict[str, str]]:
|
|
326
|
+
"""``formulas`` with each name that is used at more than one arity written as one name per
|
|
327
|
+
arity, and the table that undoes it.
|
|
328
|
+
|
|
329
|
+
The kit reads the constant ``ff`` and the unary function ``ff(x)`` as two symbols
|
|
330
|
+
(the problem writer writes them so, and so do Vampire, E, Z3 and Prover9), and Twee
|
|
331
|
+
reads one: it types a symbol by its name and stops on ``ff`` and ``ff(X)`` in one
|
|
332
|
+
problem. The first arity in which a name occurs (premises in order, then the
|
|
333
|
+
conclusion) keeps it; every other arity is written under a name minted for it, which
|
|
334
|
+
is fresh against EVERY name of the problem, of every kind and in every spelling that
|
|
335
|
+
differs from another only in case (TPTP reads ``Ff`` and ``ff`` as one word). That is
|
|
336
|
+
a renaming of one symbol to another that is not in the problem, which changes no
|
|
337
|
+
question asked of it. A minted name is ``sym_arity<n>`` (``sym_arity<n>_<i>`` when
|
|
338
|
+
that is taken): an ASCII word that starts with a lower-case letter whatever the name
|
|
339
|
+
it stands for, so the problem writer leaves it as it is and Twee prints it as
|
|
340
|
+
written. Returns ``(formulas, restore)`` where ``restore`` maps each
|
|
341
|
+
minted name to the name it stands for, to apply to whatever Twee prints; a problem
|
|
342
|
+
without such a name comes back as the very same list and an empty table.
|
|
343
|
+
"""
|
|
344
|
+
clashing = {name: seen for name, seen in _symbol_arities(formulas).items() if len(seen) > 1}
|
|
345
|
+
if not clashing:
|
|
346
|
+
return formulas, {}
|
|
347
|
+
taken = set(symbol_names(*formulas, fold=str.casefold))
|
|
348
|
+
table: Dict[Tuple[str, int], str] = {}
|
|
349
|
+
restore: Dict[str, str] = {}
|
|
350
|
+
for name in sorted(clashing):
|
|
351
|
+
for arity in clashing[name][1:]:
|
|
352
|
+
candidate = f"sym_arity{arity}"
|
|
353
|
+
index = 1
|
|
354
|
+
while candidate.casefold() in taken:
|
|
355
|
+
candidate = f"sym_arity{arity}_{index}"
|
|
356
|
+
index += 1
|
|
357
|
+
taken.add(candidate.casefold())
|
|
358
|
+
table[(name, arity)] = candidate
|
|
359
|
+
restore[candidate] = name
|
|
360
|
+
return [_rename_by_arity(f, table) for f in formulas], restore
|
|
361
|
+
|
|
362
|
+
|
|
363
|
+
def _twee_problem(premises: List[Node], conclusion: Node) -> Tuple[str, TptpNameMap, Dict[str, str]]:
|
|
364
|
+
"""The problem text handed to Twee, the writer's name map and the table of the names
|
|
365
|
+
:func:`_separate_arities` minted (empty for nearly every problem)."""
|
|
366
|
+
formulas, restore = _separate_arities(list(premises) + [conclusion])
|
|
367
|
+
problem, name_map = generate_tptp_problem_with_mapping(formulas[:-1], formulas[-1])
|
|
368
|
+
return problem, name_map, restore
|
|
369
|
+
|
|
370
|
+
|
|
371
|
+
# ---------------------------------------------------------------------------
|
|
372
|
+
# Process plumbing (WSL by default — see module docstring)
|
|
373
|
+
# ---------------------------------------------------------------------------
|
|
374
|
+
|
|
375
|
+
def _to_wsl_path(windows_path: str) -> str:
|
|
376
|
+
"""Translate a Windows path to its WSL ``/mnt/...`` form via ``wslpath``.
|
|
377
|
+
|
|
378
|
+
A local copy of :func:`atp.vampire_entailment._to_wsl_path`'s technique
|
|
379
|
+
(backslashes to forward slashes first, since the WSL interop layer
|
|
380
|
+
swallows literal backslashes in arguments) — reimplemented here rather
|
|
381
|
+
than imported so this module carries no dependency on the Vampire one.
|
|
382
|
+
"""
|
|
383
|
+
result = subprocess.run(
|
|
384
|
+
["wsl.exe", "wslpath", "-u", windows_path.replace("\\", "/")],
|
|
385
|
+
capture_output=True, text=True, timeout=20,
|
|
386
|
+
)
|
|
387
|
+
wsl_path = result.stdout.strip()
|
|
388
|
+
if not wsl_path:
|
|
389
|
+
raise RuntimeError(
|
|
390
|
+
f"wslpath could not translate {windows_path!r} (is WSL available?): "
|
|
391
|
+
f"{result.stderr.strip()}")
|
|
392
|
+
return wsl_path
|
|
393
|
+
|
|
394
|
+
|
|
395
|
+
def _spawn_twee(input_str: str, timeout: int, use_wsl: bool,
|
|
396
|
+
twee_cmd: Optional[str]) -> Tuple[str, str, bool]:
|
|
397
|
+
"""Write the TPTP problem to a temp file, run Twee, return ``(stdout, stderr, timed_out)``.
|
|
398
|
+
|
|
399
|
+
With ``use_wsl=True`` (the default), Twee is invoked as
|
|
400
|
+
``wsl.exe -e bash -lc "<cmd> --quiet '<wsl-path>'"`` — the ``bash -lc``
|
|
401
|
+
wrapper is required so WSL's own shell (not the Windows/Git-Bash caller)
|
|
402
|
+
expands a leading ``~`` in ``<cmd>`` (see module docstring). With
|
|
403
|
+
``use_wsl=False``, ``<cmd>`` is invoked as a bare subprocess with the
|
|
404
|
+
Windows temp path directly (for a native, non-WSL Twee).
|
|
405
|
+
|
|
406
|
+
A subprocess timeout is swallowed into ``timed_out=True`` (stdout/stderr
|
|
407
|
+
both ``""``), matching :mod:`atp.vampire_entailment`'s convention; any
|
|
408
|
+
other error (e.g. ``FileNotFoundError`` for ``wsl.exe`` itself missing)
|
|
409
|
+
propagates. The temp file is always removed.
|
|
410
|
+
"""
|
|
411
|
+
cmd = _twee_cmd(twee_cmd)
|
|
412
|
+
with tempfile.NamedTemporaryFile(mode="w", suffix=".p", delete=False,
|
|
413
|
+
encoding="utf-8") as temp_file:
|
|
414
|
+
temp_file.write(input_str)
|
|
415
|
+
temp_filename = temp_file.name
|
|
416
|
+
|
|
417
|
+
try:
|
|
418
|
+
if use_wsl:
|
|
419
|
+
wsl_path = _to_wsl_path(temp_filename)
|
|
420
|
+
full_command = f"{cmd} --quiet '{wsl_path}'"
|
|
421
|
+
command = ["wsl.exe", "-e", "bash", "-lc", full_command]
|
|
422
|
+
else:
|
|
423
|
+
command = [cmd, "--quiet", temp_filename]
|
|
424
|
+
result = subprocess.run(command, capture_output=True, text=True, timeout=timeout)
|
|
425
|
+
return result.stdout, result.stderr, False
|
|
426
|
+
except subprocess.TimeoutExpired:
|
|
427
|
+
return "", "", True
|
|
428
|
+
finally:
|
|
429
|
+
try:
|
|
430
|
+
os.unlink(temp_filename)
|
|
431
|
+
except OSError:
|
|
432
|
+
pass
|
|
433
|
+
|
|
434
|
+
|
|
435
|
+
def twee_available(use_wsl: bool = True, twee_cmd: Optional[str] = None) -> bool:
|
|
436
|
+
"""``True`` iff a Twee binary responds to ``--version`` (cheap; no caching).
|
|
437
|
+
|
|
438
|
+
Discovery: ``twee_cmd`` argument > ``$UFK_TWEE_CMD`` > ``~/.local/bin/twee``
|
|
439
|
+
(see :func:`_twee_cmd`). With ``use_wsl=True`` (the default) the check runs
|
|
440
|
+
``wsl.exe -e bash -lc "<cmd> --version"``; any failure (no ``wsl.exe``, no
|
|
441
|
+
WSL distro, the command not found inside WSL, ...) is swallowed to
|
|
442
|
+
``False`` — pure discovery, never raises. Twee prints ``--version`` to
|
|
443
|
+
STDERR, not stdout (verified live), so both streams are checked.
|
|
444
|
+
"""
|
|
445
|
+
cmd = _twee_cmd(twee_cmd)
|
|
446
|
+
try:
|
|
447
|
+
if use_wsl:
|
|
448
|
+
command = ["wsl.exe", "-e", "bash", "-lc", f"{cmd} --version"]
|
|
449
|
+
else:
|
|
450
|
+
command = [cmd, "--version"]
|
|
451
|
+
result = subprocess.run(command, capture_output=True, text=True, timeout=20)
|
|
452
|
+
combined = (result.stdout + result.stderr).lower()
|
|
453
|
+
return result.returncode == 0 and "twee version" in combined
|
|
454
|
+
except Exception: # noqa: BLE001 - any failure means "not available"
|
|
455
|
+
return False
|
|
456
|
+
|
|
457
|
+
|
|
458
|
+
# ---------------------------------------------------------------------------
|
|
459
|
+
# Term parsing (Twee's own printed syntax: NAME or NAME(arg, arg, ...))
|
|
460
|
+
# ---------------------------------------------------------------------------
|
|
461
|
+
|
|
462
|
+
_TOKEN_RE = re.compile(r"\s*([A-Za-z_$][A-Za-z0-9_$]*|-?\d+(?:\.\d+)?|[(),])")
|
|
463
|
+
|
|
464
|
+
|
|
465
|
+
def _tokenize(text: str) -> List[str]:
|
|
466
|
+
"""Tokenize a Twee-printed term into identifiers, numbers, and ``( ) ,``."""
|
|
467
|
+
tokens = []
|
|
468
|
+
pos = 0
|
|
469
|
+
while pos < len(text):
|
|
470
|
+
m = _TOKEN_RE.match(text, pos)
|
|
471
|
+
if not m:
|
|
472
|
+
if text[pos:].strip() == "":
|
|
473
|
+
break
|
|
474
|
+
raise ValueError(f"twee term parser: unrecognised text at {text[pos:]!r} in {text!r}")
|
|
475
|
+
tokens.append(m.group(1))
|
|
476
|
+
pos = m.end()
|
|
477
|
+
return tokens
|
|
478
|
+
|
|
479
|
+
|
|
480
|
+
def _parse_term(text: str) -> Node:
|
|
481
|
+
"""Parse one Twee-printed term into a kit term ``Node``.
|
|
482
|
+
|
|
483
|
+
An identifier starting with an uppercase letter is a :class:`Variable`
|
|
484
|
+
(Twee's own convention for axiom/lemma rewrite variables, e.g. ``X``);
|
|
485
|
+
any other identifier is a :class:`Constant` (no args) or :class:`Function`
|
|
486
|
+
(parenthesised, comma-space-separated args, e.g. ``mult(inv(X), Y)``); a
|
|
487
|
+
bare numeral is a :class:`Number`. Raises ``ValueError`` for anything that
|
|
488
|
+
does not parse as exactly one such term (trailing tokens, unbalanced
|
|
489
|
+
parens, ...) — this is the boundary past which the module refuses to
|
|
490
|
+
guess, per the "anything unrecognised raises" contract.
|
|
491
|
+
"""
|
|
492
|
+
tokens = _tokenize(text)
|
|
493
|
+
if not tokens:
|
|
494
|
+
raise ValueError(f"twee term parser: empty term text {text!r}")
|
|
495
|
+
pos = [0]
|
|
496
|
+
|
|
497
|
+
def _peek() -> Optional[str]:
|
|
498
|
+
return tokens[pos[0]] if pos[0] < len(tokens) else None
|
|
499
|
+
|
|
500
|
+
def _numeric(tok: str) -> bool:
|
|
501
|
+
return re.fullmatch(r"-?\d+(?:\.\d+)?", tok) is not None
|
|
502
|
+
|
|
503
|
+
def parse_one() -> Node:
|
|
504
|
+
tok = _peek()
|
|
505
|
+
if tok is None:
|
|
506
|
+
raise ValueError(f"twee term parser: unexpected end of term in {text!r}")
|
|
507
|
+
if _numeric(tok):
|
|
508
|
+
pos[0] += 1
|
|
509
|
+
return Number(_numeral_from_text(tok))
|
|
510
|
+
if tok in "(),":
|
|
511
|
+
raise ValueError(f"twee term parser: unexpected {tok!r} in {text!r}")
|
|
512
|
+
name = tok
|
|
513
|
+
pos[0] += 1
|
|
514
|
+
if _peek() == "(":
|
|
515
|
+
pos[0] += 1
|
|
516
|
+
args = [parse_one()]
|
|
517
|
+
while _peek() == ",":
|
|
518
|
+
pos[0] += 1
|
|
519
|
+
args.append(parse_one())
|
|
520
|
+
if _peek() != ")":
|
|
521
|
+
raise ValueError(f"twee term parser: unbalanced parentheses in {text!r}")
|
|
522
|
+
pos[0] += 1
|
|
523
|
+
return Function(name, args)
|
|
524
|
+
if name[0].isupper():
|
|
525
|
+
return Variable(name)
|
|
526
|
+
return Constant(name)
|
|
527
|
+
|
|
528
|
+
result = parse_one()
|
|
529
|
+
if pos[0] != len(tokens):
|
|
530
|
+
raise ValueError(f"twee term parser: trailing tokens after a complete term in {text!r}")
|
|
531
|
+
return result
|
|
532
|
+
|
|
533
|
+
|
|
534
|
+
def _split_equation(text: str) -> Tuple[str, str]:
|
|
535
|
+
"""Split a header's ``"lhs = rhs"`` text on the (sole) top-level ``" = "``."""
|
|
536
|
+
if " = " not in text:
|
|
537
|
+
raise ValueError(f"twee proof parser: expected 'lhs = rhs', got {text!r}")
|
|
538
|
+
lhs, rhs = text.split(" = ", 1)
|
|
539
|
+
return lhs, rhs
|
|
540
|
+
|
|
541
|
+
|
|
542
|
+
# ---------------------------------------------------------------------------
|
|
543
|
+
# Parsed-proof data model
|
|
544
|
+
# ---------------------------------------------------------------------------
|
|
545
|
+
|
|
546
|
+
@dataclass(frozen=True)
|
|
547
|
+
class TweeEquation:
|
|
548
|
+
"""One equation as Twee restates it: ``lhs = rhs`` (both kit term Nodes)."""
|
|
549
|
+
|
|
550
|
+
lhs: Node
|
|
551
|
+
rhs: Node
|
|
552
|
+
|
|
553
|
+
def to_dict(self) -> dict:
|
|
554
|
+
return {"lhs": self.lhs.to_dict(), "rhs": self.rhs.to_dict()}
|
|
555
|
+
|
|
556
|
+
|
|
557
|
+
@dataclass(frozen=True)
|
|
558
|
+
class TweeCitation:
|
|
559
|
+
"""One ``{ by axiom N (name) [R->L] }`` / ``{ by lemma N [R->L] }`` citation.
|
|
560
|
+
|
|
561
|
+
``name`` is the axiom's ``(name)`` (always present for ``kind="axiom"``,
|
|
562
|
+
always ``None`` for ``kind="lemma"`` — lemmas are never named in Twee's
|
|
563
|
+
own citations). ``reversed`` is ``True`` iff the citation carries the
|
|
564
|
+
``R->L`` suffix (apply the equation right-to-left).
|
|
565
|
+
"""
|
|
566
|
+
|
|
567
|
+
kind: str
|
|
568
|
+
number: int
|
|
569
|
+
name: Optional[str]
|
|
570
|
+
reversed: bool = False
|
|
571
|
+
|
|
572
|
+
def __post_init__(self):
|
|
573
|
+
if self.kind not in ("axiom", "lemma"):
|
|
574
|
+
raise ValueError(f"TweeCitation: kind must be 'axiom' or 'lemma', got {self.kind!r}")
|
|
575
|
+
|
|
576
|
+
def to_dict(self) -> dict:
|
|
577
|
+
return {"kind": self.kind, "number": self.number, "name": self.name,
|
|
578
|
+
"reversed": self.reversed}
|
|
579
|
+
|
|
580
|
+
|
|
581
|
+
@dataclass(frozen=True)
|
|
582
|
+
class TweeChain:
|
|
583
|
+
"""A rewrite chain: ``terms[0]`` rewritten step by step down to ``terms[-1]``.
|
|
584
|
+
|
|
585
|
+
``len(terms) == len(citations) + 1`` — ``citations[i]`` justifies the step
|
|
586
|
+
from ``terms[i]`` to ``terms[i+1]``.
|
|
587
|
+
"""
|
|
588
|
+
|
|
589
|
+
terms: Tuple[Node, ...]
|
|
590
|
+
citations: Tuple[TweeCitation, ...] = ()
|
|
591
|
+
|
|
592
|
+
def __post_init__(self):
|
|
593
|
+
object.__setattr__(self, "terms", tuple(self.terms))
|
|
594
|
+
object.__setattr__(self, "citations", tuple(self.citations))
|
|
595
|
+
if len(self.terms) != len(self.citations) + 1:
|
|
596
|
+
raise ValueError(
|
|
597
|
+
f"TweeChain: {len(self.terms)} terms need exactly "
|
|
598
|
+
f"{max(len(self.terms) - 1, 0)} citations, got {len(self.citations)}")
|
|
599
|
+
|
|
600
|
+
def to_dict(self) -> dict:
|
|
601
|
+
return {"terms": [t.to_dict() for t in self.terms],
|
|
602
|
+
"citations": [c.to_dict() for c in self.citations]}
|
|
603
|
+
|
|
604
|
+
|
|
605
|
+
@dataclass(frozen=True)
|
|
606
|
+
class TweeAxiom:
|
|
607
|
+
"""One ``Axiom N (name): lhs = rhs.`` header line."""
|
|
608
|
+
|
|
609
|
+
number: int
|
|
610
|
+
name: str
|
|
611
|
+
equation: TweeEquation
|
|
612
|
+
|
|
613
|
+
def to_dict(self) -> dict:
|
|
614
|
+
return {"number": self.number, "name": self.name, "equation": self.equation.to_dict()}
|
|
615
|
+
|
|
616
|
+
|
|
617
|
+
@dataclass(frozen=True)
|
|
618
|
+
class TweeLemma:
|
|
619
|
+
"""One ``Lemma N: lhs = rhs.`` block with its own ``Proof:`` chain."""
|
|
620
|
+
|
|
621
|
+
number: int
|
|
622
|
+
equation: TweeEquation
|
|
623
|
+
chain: TweeChain
|
|
624
|
+
|
|
625
|
+
def to_dict(self) -> dict:
|
|
626
|
+
return {"number": self.number, "equation": self.equation.to_dict(),
|
|
627
|
+
"chain": self.chain.to_dict()}
|
|
628
|
+
|
|
629
|
+
|
|
630
|
+
@dataclass(frozen=True)
|
|
631
|
+
class TweeGoal:
|
|
632
|
+
"""The (single, always-last) ``Goal N (name): lhs = rhs.`` block."""
|
|
633
|
+
|
|
634
|
+
number: int
|
|
635
|
+
name: str
|
|
636
|
+
equation: TweeEquation
|
|
637
|
+
chain: TweeChain
|
|
638
|
+
|
|
639
|
+
def to_dict(self) -> dict:
|
|
640
|
+
return {"number": self.number, "name": self.name,
|
|
641
|
+
"equation": self.equation.to_dict(), "chain": self.chain.to_dict()}
|
|
642
|
+
|
|
643
|
+
|
|
644
|
+
@dataclass(frozen=True)
|
|
645
|
+
class TweeProof:
|
|
646
|
+
"""A full parsed Twee ``Theorem`` proof: the used axioms, lemmas, then the goal."""
|
|
647
|
+
|
|
648
|
+
axioms: Tuple[TweeAxiom, ...] = ()
|
|
649
|
+
lemmas: Tuple[TweeLemma, ...] = ()
|
|
650
|
+
goal: TweeGoal = None
|
|
651
|
+
|
|
652
|
+
def __post_init__(self):
|
|
653
|
+
object.__setattr__(self, "axioms", tuple(self.axioms))
|
|
654
|
+
object.__setattr__(self, "lemmas", tuple(self.lemmas))
|
|
655
|
+
if self.goal is None:
|
|
656
|
+
raise ValueError("TweeProof: goal is required")
|
|
657
|
+
|
|
658
|
+
def to_dict(self) -> dict:
|
|
659
|
+
return {"axioms": [a.to_dict() for a in self.axioms],
|
|
660
|
+
"lemmas": [l.to_dict() for l in self.lemmas],
|
|
661
|
+
"goal": self.goal.to_dict()}
|
|
662
|
+
|
|
663
|
+
|
|
664
|
+
# ---------------------------------------------------------------------------
|
|
665
|
+
# Proof parser
|
|
666
|
+
# ---------------------------------------------------------------------------
|
|
667
|
+
|
|
668
|
+
_AXIOM_RE = re.compile(r"^Axiom (\d+) \(([^)]*)\): (.+)\.$")
|
|
669
|
+
_LEMMA_RE = re.compile(r"^Lemma (\d+): (.+)\.$")
|
|
670
|
+
_GOAL_RE = re.compile(r"^Goal (\d+) \(([^)]*)\): (.+)\.$")
|
|
671
|
+
_CITATION_RE = re.compile(r"^= \{ by (axiom|lemma) (\d+)(?: \(([^)]*)\))?( R->L)? \}$")
|
|
672
|
+
_RESULT_RE = re.compile(r"^RESULT: (\S+) \(.*\)\.$")
|
|
673
|
+
|
|
674
|
+
|
|
675
|
+
def _extract_result_status(stdout: str) -> Optional[str]:
|
|
676
|
+
"""Return Twee's ``RESULT: <Status> (...)`` token, or ``None`` if absent.
|
|
677
|
+
|
|
678
|
+
Only the last non-blank line is checked (that is always where ``RESULT:``
|
|
679
|
+
lives in every observed shape) — a stray line elsewhere that happens to
|
|
680
|
+
start with ``RESULT:`` (not observed, never generated by Twee) is not
|
|
681
|
+
treated as the verdict.
|
|
682
|
+
"""
|
|
683
|
+
lines = [ln for ln in stdout.splitlines() if ln.strip()]
|
|
684
|
+
if not lines:
|
|
685
|
+
return None
|
|
686
|
+
m = _RESULT_RE.match(lines[-1].strip())
|
|
687
|
+
return m.group(1) if m else None
|
|
688
|
+
|
|
689
|
+
|
|
690
|
+
def _parse_chain(lines: List[str], idx: int) -> Tuple[TweeChain, int]:
|
|
691
|
+
"""Parse a ``Proof:``-following rewrite chain starting at ``lines[idx]``.
|
|
692
|
+
|
|
693
|
+
Returns ``(chain, next_idx)`` where ``next_idx`` is the first line after
|
|
694
|
+
the chain (a blank line, a new header, or end of input).
|
|
695
|
+
"""
|
|
696
|
+
if idx >= len(lines) or not lines[idx].startswith(" "):
|
|
697
|
+
raise ValueError(f"twee proof parser: expected an indented term line, got {lines[idx:idx+1]!r}")
|
|
698
|
+
terms = [_parse_term(lines[idx][2:].strip())]
|
|
699
|
+
idx += 1
|
|
700
|
+
citations: List[TweeCitation] = []
|
|
701
|
+
while idx < len(lines) and lines[idx].strip():
|
|
702
|
+
cm = _CITATION_RE.match(lines[idx])
|
|
703
|
+
if not cm:
|
|
704
|
+
break
|
|
705
|
+
kind, number, name, rev = cm.group(1), int(cm.group(2)), cm.group(3), cm.group(4) is not None
|
|
706
|
+
if kind == "axiom" and name is None:
|
|
707
|
+
raise ValueError(f"twee proof parser: axiom citation missing its (name): {lines[idx]!r}")
|
|
708
|
+
if kind == "lemma" and name is not None:
|
|
709
|
+
raise ValueError(f"twee proof parser: lemma citation must not carry a (name): {lines[idx]!r}")
|
|
710
|
+
citations.append(TweeCitation(kind, number, name, rev))
|
|
711
|
+
idx += 1
|
|
712
|
+
if idx >= len(lines) or not lines[idx].startswith(" "):
|
|
713
|
+
raise ValueError(f"twee proof parser: expected an indented term line after a citation, "
|
|
714
|
+
f"got {lines[idx:idx+1]!r}")
|
|
715
|
+
terms.append(_parse_term(lines[idx][2:].strip()))
|
|
716
|
+
idx += 1
|
|
717
|
+
return TweeChain(tuple(terms), tuple(citations)), idx
|
|
718
|
+
|
|
719
|
+
|
|
720
|
+
def parse_twee_proof(stdout: str) -> Optional[TweeProof]:
|
|
721
|
+
"""Parse Twee's ``--quiet`` stdout into a :class:`TweeProof`, or ``None``.
|
|
722
|
+
|
|
723
|
+
Returns ``None`` whenever the ``RESULT`` is not ``Theorem`` (Twee's
|
|
724
|
+
``--quiet`` output for ``CounterSatisfiable``/``Satisfiable``/etc. is just
|
|
725
|
+
the bare ``RESULT:`` line — nothing to parse — see the module docstring's
|
|
726
|
+
shape 3). Raises ``ValueError`` when ``RESULT`` IS ``Theorem`` but the
|
|
727
|
+
surrounding text does not match the grammar distilled in the module
|
|
728
|
+
docstring (shapes 1-2) — never silently produces a partial/guessed proof.
|
|
729
|
+
"""
|
|
730
|
+
if _extract_result_status(stdout) != "Theorem":
|
|
731
|
+
return None
|
|
732
|
+
|
|
733
|
+
lines = [ln.rstrip() for ln in stdout.splitlines()]
|
|
734
|
+
idx = 0
|
|
735
|
+
|
|
736
|
+
# Every observed Theorem run opens with this fixed banner line (even
|
|
737
|
+
# under --quiet — verified live; it is the one piece of "essential
|
|
738
|
+
# information" --quiet does not suppress) followed by a blank line.
|
|
739
|
+
if idx >= len(lines) or lines[idx] != "The conjecture is true! Here is a proof.":
|
|
740
|
+
raise ValueError(
|
|
741
|
+
f"twee proof parser: expected the 'conjecture is true' banner line, "
|
|
742
|
+
f"got {lines[idx:idx + 1]!r}")
|
|
743
|
+
idx += 1
|
|
744
|
+
while idx < len(lines) and not lines[idx].strip():
|
|
745
|
+
idx += 1
|
|
746
|
+
|
|
747
|
+
axioms: List[TweeAxiom] = []
|
|
748
|
+
while idx < len(lines) and lines[idx].strip():
|
|
749
|
+
m = _AXIOM_RE.match(lines[idx])
|
|
750
|
+
if not m:
|
|
751
|
+
break
|
|
752
|
+
number, name, eq_text = int(m.group(1)), m.group(2), m.group(3)
|
|
753
|
+
lhs_text, rhs_text = _split_equation(eq_text)
|
|
754
|
+
axioms.append(TweeAxiom(number, name,
|
|
755
|
+
TweeEquation(_parse_term(lhs_text), _parse_term(rhs_text))))
|
|
756
|
+
idx += 1
|
|
757
|
+
while idx < len(lines) and not lines[idx].strip():
|
|
758
|
+
idx += 1
|
|
759
|
+
|
|
760
|
+
lemmas: List[TweeLemma] = []
|
|
761
|
+
goal: Optional[TweeGoal] = None
|
|
762
|
+
while idx < len(lines):
|
|
763
|
+
if not lines[idx].strip():
|
|
764
|
+
idx += 1
|
|
765
|
+
continue
|
|
766
|
+
if _RESULT_RE.match(lines[idx]):
|
|
767
|
+
# The trailing RESULT line is a natural terminator, not a
|
|
768
|
+
# Lemma/Goal header — reached here only for a malformed proof
|
|
769
|
+
# that has no Goal block at all (see the "no Goal block" check
|
|
770
|
+
# below); the RESULT's own status was already read separately.
|
|
771
|
+
break
|
|
772
|
+
m_lemma = _LEMMA_RE.match(lines[idx])
|
|
773
|
+
m_goal = _GOAL_RE.match(lines[idx])
|
|
774
|
+
if m_lemma:
|
|
775
|
+
number, eq_text = int(m_lemma.group(1)), m_lemma.group(2)
|
|
776
|
+
lhs_text, rhs_text = _split_equation(eq_text)
|
|
777
|
+
equation = TweeEquation(_parse_term(lhs_text), _parse_term(rhs_text))
|
|
778
|
+
idx += 1
|
|
779
|
+
if idx >= len(lines) or lines[idx].strip() != "Proof:":
|
|
780
|
+
raise ValueError(f"twee proof parser: expected 'Proof:' after Lemma {number}")
|
|
781
|
+
idx += 1
|
|
782
|
+
chain, idx = _parse_chain(lines, idx)
|
|
783
|
+
lemmas.append(TweeLemma(number, equation, chain))
|
|
784
|
+
elif m_goal:
|
|
785
|
+
number, name, eq_text = int(m_goal.group(1)), m_goal.group(2), m_goal.group(3)
|
|
786
|
+
lhs_text, rhs_text = _split_equation(eq_text)
|
|
787
|
+
equation = TweeEquation(_parse_term(lhs_text), _parse_term(rhs_text))
|
|
788
|
+
idx += 1
|
|
789
|
+
if idx >= len(lines) or lines[idx].strip() != "Proof:":
|
|
790
|
+
raise ValueError(f"twee proof parser: expected 'Proof:' after Goal {number}")
|
|
791
|
+
idx += 1
|
|
792
|
+
chain, idx = _parse_chain(lines, idx)
|
|
793
|
+
goal = TweeGoal(number, name, equation, chain)
|
|
794
|
+
break # the goal is always the last block Twee prints
|
|
795
|
+
else:
|
|
796
|
+
raise ValueError(
|
|
797
|
+
f"twee proof parser: expected a Lemma/Goal header, got {lines[idx]!r}")
|
|
798
|
+
|
|
799
|
+
if goal is None:
|
|
800
|
+
raise ValueError("twee proof parser: RESULT was Theorem but no Goal block was found")
|
|
801
|
+
|
|
802
|
+
return TweeProof(tuple(axioms), tuple(lemmas), goal)
|
|
803
|
+
|
|
804
|
+
|
|
805
|
+
# ---------------------------------------------------------------------------
|
|
806
|
+
# Rückweg: translate a parsed TweeProof's terms back to kit-level names.
|
|
807
|
+
# ---------------------------------------------------------------------------
|
|
808
|
+
|
|
809
|
+
def _map_proof_terms(proof: TweeProof, fn: Callable[[Node], Node]) -> TweeProof:
|
|
810
|
+
"""``proof`` with ``fn`` applied to every term of it (the two sides of each axiom, lemma
|
|
811
|
+
and goal equation and every term of every chain); numbers, names and citations stay."""
|
|
812
|
+
def equation(eq: TweeEquation) -> TweeEquation:
|
|
813
|
+
return TweeEquation(fn(eq.lhs), fn(eq.rhs))
|
|
814
|
+
|
|
815
|
+
def chain(c: TweeChain) -> TweeChain:
|
|
816
|
+
return TweeChain(tuple(fn(t) for t in c.terms), c.citations)
|
|
817
|
+
|
|
818
|
+
return TweeProof(
|
|
819
|
+
tuple(TweeAxiom(a.number, a.name, equation(a.equation)) for a in proof.axioms),
|
|
820
|
+
tuple(TweeLemma(l.number, equation(l.equation), chain(l.chain)) for l in proof.lemmas),
|
|
821
|
+
TweeGoal(proof.goal.number, proof.goal.name, equation(proof.goal.equation),
|
|
822
|
+
chain(proof.goal.chain)))
|
|
823
|
+
|
|
824
|
+
|
|
825
|
+
def _restore_term_names(term: Node, restore: Dict[str, str]) -> Node:
|
|
826
|
+
"""``term`` with every function/constant name that ``restore`` maps written as the name it
|
|
827
|
+
maps to (see :func:`_separate_arities`)."""
|
|
828
|
+
if isinstance(term, Function):
|
|
829
|
+
args = tuple(_restore_term_names(a, restore) for a in term.args)
|
|
830
|
+
return Function(restore.get(term.name, term.name), args)
|
|
831
|
+
if isinstance(term, Constant):
|
|
832
|
+
return Constant(restore.get(term.name, term.name))
|
|
833
|
+
return term
|
|
834
|
+
|
|
835
|
+
|
|
836
|
+
def reverse_map_twee_proof(proof: TweeProof, mapping: TptpNameMap) -> TweeProof:
|
|
837
|
+
"""Rewrite every term in ``proof`` from the sanitised TPTP-ASCII function/
|
|
838
|
+
constant names :func:`atp._tptp_problem.generate_tptp_problem_with_mapping`
|
|
839
|
+
chose back to the original kit-level names (see
|
|
840
|
+
:func:`atp._tptp_problem.apply_reverse_tptp`).
|
|
841
|
+
|
|
842
|
+
Twee's equational fragment only ever uses ``=`` as a predicate (excluded
|
|
843
|
+
from renaming to begin with — see :mod:`atp._tptp_problem`'s module
|
|
844
|
+
docstring), so only function/constant names can ever have been
|
|
845
|
+
sanitised here; ``apply_reverse_tptp`` is reused as-is (it simply never
|
|
846
|
+
hits its ``Atom`` branch on a bare term). Axiom/lemma/goal NUMBERS and
|
|
847
|
+
axiom NAMES (Twee's own local proof-step indexing, and the
|
|
848
|
+
``premise_<i>`` names this module itself assigned — see
|
|
849
|
+
:func:`_generate_twee_input`) are not symbol names and are left alone.
|
|
850
|
+
"""
|
|
851
|
+
return _map_proof_terms(proof, lambda term: apply_reverse_tptp(term, mapping))
|
|
852
|
+
|
|
853
|
+
|
|
854
|
+
# ---------------------------------------------------------------------------
|
|
855
|
+
# Public entry point
|
|
856
|
+
# ---------------------------------------------------------------------------
|
|
857
|
+
|
|
858
|
+
def check_entailment_twee_detailed(premises: List[Node], conclusion: Node,
|
|
859
|
+
timeout: int = 30, use_wsl: bool = True,
|
|
860
|
+
twee_cmd: Optional[str] = None) -> dict:
|
|
861
|
+
"""Run Twee on ``premises ⊨ conclusion`` and return its status, output, and proof.
|
|
862
|
+
|
|
863
|
+
Both ``premises`` and ``conclusion`` must be in Twee's equational
|
|
864
|
+
fragment (see :func:`_is_equational`) — anything else raises
|
|
865
|
+
``NotImplementedError`` naming the offender, BEFORE any subprocess is
|
|
866
|
+
spawned (same contract as :func:`atp.vampire_entailment
|
|
867
|
+
.check_entailment_vampire_detailed`). Problem generation additionally
|
|
868
|
+
refuses (also ``NotImplementedError``, also before any subprocess) if
|
|
869
|
+
two distinct function/constant names would fold to the same TPTP
|
|
870
|
+
identifier — see :mod:`atp._tptp_problem`'s module docstring; ``=`` is
|
|
871
|
+
the only predicate this fragment ever uses, and it is excluded from that
|
|
872
|
+
check entirely (see the same docstring), so only a function/constant
|
|
873
|
+
collision can occur here.
|
|
874
|
+
|
|
875
|
+
Args:
|
|
876
|
+
premises: equational premise formulas.
|
|
877
|
+
conclusion: the equational conclusion formula.
|
|
878
|
+
timeout: seconds to allow the Twee process before giving up (a
|
|
879
|
+
subprocess timeout, not one of Twee's own ``--max-*`` flags,
|
|
880
|
+
which this function never passes).
|
|
881
|
+
use_wsl: drive Twee through WSL (default ``True`` — see module
|
|
882
|
+
docstring); ``False`` for a native, non-WSL Twee on ``PATH``.
|
|
883
|
+
twee_cmd: override the Twee command/path (default: ``$UFK_TWEE_CMD``,
|
|
884
|
+
else ``~/.local/bin/twee``).
|
|
885
|
+
|
|
886
|
+
Returns:
|
|
887
|
+
A dict with:
|
|
888
|
+
|
|
889
|
+
* ``status``: Twee's own ``RESULT:`` token verbatim (``"Theorem"``,
|
|
890
|
+
``"CounterSatisfiable"``, ``"GaveUp"``, ``"Satisfiable"``,
|
|
891
|
+
``"Unsatisfiable"``, ...), or ``"Unknown"`` when no ``RESULT:``
|
|
892
|
+
line was found at all (a subprocess timeout, or a Twee-side
|
|
893
|
+
parse/usage error — see ``timed_out`` to distinguish).
|
|
894
|
+
* ``raw_output``: Twee's full stdout; if no ``RESULT:`` line was
|
|
895
|
+
found and Twee wrote to stderr (a parse/usage error), stderr is
|
|
896
|
+
appended so the failure is diagnosable.
|
|
897
|
+
* ``proof``: the parsed :class:`TweeProof` when ``status ==
|
|
898
|
+
"Theorem"`` (call ``.to_dict()`` for a JSON-compatible form — kept
|
|
899
|
+
as the live dataclass here, not pre-serialised, because
|
|
900
|
+
:func:`atp.twee_check.check_twee_proof` consumes it directly),
|
|
901
|
+
else ``None``.
|
|
902
|
+
* ``timed_out``: ``True`` iff this function's own ``timeout`` (not
|
|
903
|
+
one of Twee's ``--max-*`` budgets) killed the subprocess.
|
|
904
|
+
|
|
905
|
+
Every function/constant name in ``raw_output`` and in ``proof``'s
|
|
906
|
+
terms has already been translated back from whatever ASCII-safe
|
|
907
|
+
token :func:`atp._tptp_problem.generate_tptp_problem_with_mapping`
|
|
908
|
+
may have substituted (a non-ASCII or digit-leading kit-level name)
|
|
909
|
+
to the ORIGINAL kit-level name (see
|
|
910
|
+
:func:`reverse_map_twee_proof`); Twee's own axiom/lemma/goal
|
|
911
|
+
numbering and the ``premise_<i>`` axiom names are untouched (they
|
|
912
|
+
were never symbol names to begin with). A name the problem uses at
|
|
913
|
+
two arities is a problem for Twee alone (it types a symbol by its name):
|
|
914
|
+
it is written as one name per arity (see the module docstring) and
|
|
915
|
+
handed back under the caller's name in both.
|
|
916
|
+
"""
|
|
917
|
+
for i, premise in enumerate(premises, start=1):
|
|
918
|
+
_require_equational(premise, f"premise {i}")
|
|
919
|
+
_require_equational(conclusion, "conclusion")
|
|
920
|
+
|
|
921
|
+
problem, name_map, restore = _twee_problem(list(premises), conclusion)
|
|
922
|
+
stdout, stderr, timed_out = _spawn_twee(problem, timeout=timeout, use_wsl=use_wsl,
|
|
923
|
+
twee_cmd=twee_cmd)
|
|
924
|
+
|
|
925
|
+
if timed_out:
|
|
926
|
+
return {"status": "Unknown", "raw_output": "", "proof": None,
|
|
927
|
+
"timed_out": True, "proof_parse_error": None}
|
|
928
|
+
|
|
929
|
+
status = _extract_result_status(stdout) or "Unknown"
|
|
930
|
+
raw_output = stdout
|
|
931
|
+
if status == "Unknown" and stderr:
|
|
932
|
+
raw_output = f"{stdout}\n{stderr}" if stdout else stderr
|
|
933
|
+
pred_rev, term_rev = name_map.reverse_rendered()
|
|
934
|
+
# The names minted for a name used at two arities come first: the writer's own table
|
|
935
|
+
# maps each of them to itself, and the first table to know a token decides.
|
|
936
|
+
raw_output = reverse_map_text(raw_output, restore, pred_rev, term_rev)
|
|
937
|
+
|
|
938
|
+
proof = None
|
|
939
|
+
proof_parse_error = None
|
|
940
|
+
if status == "Theorem":
|
|
941
|
+
try:
|
|
942
|
+
proof = reverse_map_twee_proof(parse_twee_proof(stdout), name_map)
|
|
943
|
+
proof = _map_proof_terms(proof, lambda term: _restore_term_names(term, restore))
|
|
944
|
+
except ValueError as exc:
|
|
945
|
+
# A Theorem status with proof text outside this module's
|
|
946
|
+
# distilled grammar (another Twee version, a reformat) must not
|
|
947
|
+
# CRASH the caller: the status stands, the proof is honestly
|
|
948
|
+
# absent, and the parser's complaint travels along so
|
|
949
|
+
# TweeBackend can refuse PROVED with the reason on record
|
|
950
|
+
# (review-confirmed: this previously raised out of decide()).
|
|
951
|
+
proof_parse_error = str(exc)
|
|
952
|
+
return {"status": status, "raw_output": raw_output, "proof": proof,
|
|
953
|
+
"timed_out": False, "proof_parse_error": proof_parse_error}
|