unicode-logic-kit 0.31.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- unicode_logic_kit/__init__.py +385 -0
- unicode_logic_kit/__main__.py +520 -0
- unicode_logic_kit/_deadline.py +219 -0
- unicode_logic_kit/ace/__init__.py +126 -0
- unicode_logic_kit/ace/_align.py +135 -0
- unicode_logic_kit/ace/chem_lexicon.py +128 -0
- unicode_logic_kit/ace/drs_reader.py +570 -0
- unicode_logic_kit/ace/mapping.py +666 -0
- unicode_logic_kit/ace/reverse_modal.py +138 -0
- unicode_logic_kit/ace/runner.py +551 -0
- unicode_logic_kit/ace/translate.py +452 -0
- unicode_logic_kit/ace/verbalize.py +1070 -0
- unicode_logic_kit/api.py +1284 -0
- unicode_logic_kit/atp/__init__.py +177 -0
- unicode_logic_kit/atp/_ascii_names.py +113 -0
- unicode_logic_kit/atp/_html.py +72 -0
- unicode_logic_kit/atp/_substructural_input.py +228 -0
- unicode_logic_kit/atp/_tff_problem.py +715 -0
- unicode_logic_kit/atp/_tptp_problem.py +1111 -0
- unicode_logic_kit/atp/_writer_support.py +289 -0
- unicode_logic_kit/atp/clingo_backend.py +1180 -0
- unicode_logic_kit/atp/cvc5_backend.py +1385 -0
- unicode_logic_kit/atp/eprover_backend.py +732 -0
- unicode_logic_kit/atp/finite_domain.py +1055 -0
- unicode_logic_kit/atp/fitch.py +1547 -0
- unicode_logic_kit/atp/fitch_search.py +551 -0
- unicode_logic_kit/atp/hets_backend.py +339 -0
- unicode_logic_kit/atp/hybrid_down.py +120 -0
- unicode_logic_kit/atp/incremental.py +250 -0
- unicode_logic_kit/atp/kripke_enum.py +741 -0
- unicode_logic_kit/atp/lambek.py +436 -0
- unicode_logic_kit/atp/leo3_backend.py +332 -0
- unicode_logic_kit/atp/linear.py +738 -0
- unicode_logic_kit/atp/lj.py +705 -0
- unicode_logic_kit/atp/logic_backends.py +566 -0
- unicode_logic_kit/atp/ltl_tableau.py +1084 -0
- unicode_logic_kit/atp/minizinc_backend.py +1402 -0
- unicode_logic_kit/atp/modal_tableau.py +1382 -0
- unicode_logic_kit/atp/nanocop_backend.py +410 -0
- unicode_logic_kit/atp/portfolio.py +489 -0
- unicode_logic_kit/atp/protocol.py +1803 -0
- unicode_logic_kit/atp/prover9_entailment.py +1153 -0
- unicode_logic_kit/atp/resolution.py +1376 -0
- unicode_logic_kit/atp/resolution_check.py +1114 -0
- unicode_logic_kit/atp/sequent.py +1050 -0
- unicode_logic_kit/atp/tableau.py +921 -0
- unicode_logic_kit/atp/tableau_check.py +543 -0
- unicode_logic_kit/atp/tptp_ncl.py +811 -0
- unicode_logic_kit/atp/tptp_tff.py +1546 -0
- unicode_logic_kit/atp/tstp.py +1333 -0
- unicode_logic_kit/atp/tstp_check.py +1096 -0
- unicode_logic_kit/atp/twee_backend.py +236 -0
- unicode_logic_kit/atp/twee_check.py +711 -0
- unicode_logic_kit/atp/twee_entailment.py +953 -0
- unicode_logic_kit/atp/vampire_entailment.py +540 -0
- unicode_logic_kit/atp/z3_arith.py +470 -0
- unicode_logic_kit/atp/z3_equivalence.py +36 -0
- unicode_logic_kit/atp/z3_fuzzy.py +362 -0
- unicode_logic_kit/atp/z3_input.py +500 -0
- unicode_logic_kit/atp/z3_models.py +208 -0
- unicode_logic_kit/chem/__init__.py +88 -0
- unicode_logic_kit/chem/_naming.py +284 -0
- unicode_logic_kit/chem/cache.py +185 -0
- unicode_logic_kit/chem/interop.py +244 -0
- unicode_logic_kit/chem/mol.py +525 -0
- unicode_logic_kit/chem/signature.py +112 -0
- unicode_logic_kit/comorphism.py +497 -0
- unicode_logic_kit/dl/__init__.py +384 -0
- unicode_logic_kit/dl/classification.py +227 -0
- unicode_logic_kit/dl/concepts.py +632 -0
- unicode_logic_kit/dl/datatypes.py +818 -0
- unicode_logic_kit/dl/owl_functional.py +2433 -0
- unicode_logic_kit/dl/owl_manchester.py +1637 -0
- unicode_logic_kit/dl/owl_reasoner.py +790 -0
- unicode_logic_kit/dl/parser.py +391 -0
- unicode_logic_kit/dl/tableau.py +4048 -0
- unicode_logic_kit/dl/translate.py +2704 -0
- unicode_logic_kit/drt/__init__.py +94 -0
- unicode_logic_kit/drt/export.py +179 -0
- unicode_logic_kit/drt/nodes.py +506 -0
- unicode_logic_kit/drt/parser.py +965 -0
- unicode_logic_kit/drt/resolve.py +195 -0
- unicode_logic_kit/drt/reverse.py +175 -0
- unicode_logic_kit/eval/__init__.py +106 -0
- unicode_logic_kit/eval/batch.py +382 -0
- unicode_logic_kit/eval/canonical.py +663 -0
- unicode_logic_kit/eval/chem_batch.py +606 -0
- unicode_logic_kit/eval/converses.py +200 -0
- unicode_logic_kit/eval/datasets/__init__.py +136 -0
- unicode_logic_kit/eval/datasets/_base.py +263 -0
- unicode_logic_kit/eval/datasets/_proofwriter_proof.py +422 -0
- unicode_logic_kit/eval/datasets/c3po.py +678 -0
- unicode_logic_kit/eval/datasets/folio.py +158 -0
- unicode_logic_kit/eval/datasets/fracas.py +418 -0
- unicode_logic_kit/eval/datasets/groves.py +191 -0
- unicode_logic_kit/eval/datasets/logicbench.py +467 -0
- unicode_logic_kit/eval/datasets/logicnli.py +303 -0
- unicode_logic_kit/eval/datasets/malls.py +133 -0
- unicode_logic_kit/eval/datasets/pfolio.py +594 -0
- unicode_logic_kit/eval/datasets/pmb.py +242 -0
- unicode_logic_kit/eval/datasets/prontoqa.py +611 -0
- unicode_logic_kit/eval/datasets/proofwriter.py +1431 -0
- unicode_logic_kit/eval/datasets/proverqa.py +674 -0
- unicode_logic_kit/eval/datasets/willow.py +478 -0
- unicode_logic_kit/eval/equivalence.py +466 -0
- unicode_logic_kit/eval/exercise_gen.py +533 -0
- unicode_logic_kit/eval/explain.py +791 -0
- unicode_logic_kit/eval/generality.py +750 -0
- unicode_logic_kit/eval/metric_hf.py +458 -0
- unicode_logic_kit/eval/predicate_match.py +343 -0
- unicode_logic_kit/eval/theory_check.py +1170 -0
- unicode_logic_kit/eval/validate.py +306 -0
- unicode_logic_kit/fol/__init__.py +177 -0
- unicode_logic_kit/fol/_atom_keys.py +510 -0
- unicode_logic_kit/fol/_fol_nodes.py +3586 -0
- unicode_logic_kit/fol/_free_parameters.py +105 -0
- unicode_logic_kit/fol/_ho_nodes.py +448 -0
- unicode_logic_kit/fol/_hybrid_nodes.py +308 -0
- unicode_logic_kit/fol/_identifiers.py +1091 -0
- unicode_logic_kit/fol/_lambek_nodes.py +112 -0
- unicode_logic_kit/fol/_linear_nodes.py +352 -0
- unicode_logic_kit/fol/_modal_nodes.py +1467 -0
- unicode_logic_kit/fol/_msfl_nodes.py +2196 -0
- unicode_logic_kit/fol/_numeral_symbols.py +231 -0
- unicode_logic_kit/fol/_so_nodes.py +200 -0
- unicode_logic_kit/fol/_symbol_names.py +81 -0
- unicode_logic_kit/fol/_team_nodes.py +181 -0
- unicode_logic_kit/fol/_tptp_symbols.py +551 -0
- unicode_logic_kit/fol/_truth_constants.py +117 -0
- unicode_logic_kit/fol/casl_export.py +1135 -0
- unicode_logic_kit/fol/casl_import.py +929 -0
- unicode_logic_kit/fol/derivation.py +367 -0
- unicode_logic_kit/fol/dialect_detect.py +70 -0
- unicode_logic_kit/fol/dialect_repair.py +537 -0
- unicode_logic_kit/fol/frames.py +637 -0
- unicode_logic_kit/fol/grammars/terminals.lark +31 -0
- unicode_logic_kit/fol/lambda_tools.py +297 -0
- unicode_logic_kit/fol/latex_input.py +429 -0
- unicode_logic_kit/fol/modal_translation.py +944 -0
- unicode_logic_kit/fol/msflparser.py +1033 -0
- unicode_logic_kit/fol/naming.py +422 -0
- unicode_logic_kit/fol/nodes.py +241 -0
- unicode_logic_kit/fol/normalforms.py +492 -0
- unicode_logic_kit/fol/pal.py +287 -0
- unicode_logic_kit/fol/prolog_export.py +566 -0
- unicode_logic_kit/fol/prolog_input.py +505 -0
- unicode_logic_kit/fol/prover9_input.py +1325 -0
- unicode_logic_kit/fol/qml.py +1760 -0
- unicode_logic_kit/fol/qmltp_input.py +525 -0
- unicode_logic_kit/fol/sanitize.py +221 -0
- unicode_logic_kit/fol/serialize.py +79 -0
- unicode_logic_kit/fol/signature.py +1290 -0
- unicode_logic_kit/fol/simplify_check.py +544 -0
- unicode_logic_kit/fol/spans.py +594 -0
- unicode_logic_kit/fol/tptp_input.py +1503 -0
- unicode_logic_kit/fol/tptp_repair.py +941 -0
- unicode_logic_kit/fol/unification.py +157 -0
- unicode_logic_kit/fol/verbalize.py +263 -0
- unicode_logic_kit/hets/__init__.py +163 -0
- unicode_logic_kit/hets/bridge.py +142 -0
- unicode_logic_kit/hets/client.py +748 -0
- unicode_logic_kit/hets/docker.py +420 -0
- unicode_logic_kit/hets/dol.py +712 -0
- unicode_logic_kit/hets/haskell_json.py +355 -0
- unicode_logic_kit/hets/owl_backend.py +794 -0
- unicode_logic_kit/hets/owl_cli.py +598 -0
- unicode_logic_kit/hets/symbols.py +512 -0
- unicode_logic_kit/hol/__init__.py +140 -0
- unicode_logic_kit/hol/_ho_common.py +323 -0
- unicode_logic_kit/hol/_isabelle_binders.py +125 -0
- unicode_logic_kit/hol/classical.py +812 -0
- unicode_logic_kit/hol/deepshallow/__init__.py +45 -0
- unicode_logic_kit/hol/deepshallow/_common.py +177 -0
- unicode_logic_kit/hol/deepshallow/conditional.py +225 -0
- unicode_logic_kit/hol/deepshallow/intuitionistic.py +181 -0
- unicode_logic_kit/hol/deepshallow/modal.py +217 -0
- unicode_logic_kit/hol/deepshallow/qml.py +406 -0
- unicode_logic_kit/hol/deepshallow/relevant.py +206 -0
- unicode_logic_kit/hol/free.py +753 -0
- unicode_logic_kit/hol/goedel.py +336 -0
- unicode_logic_kit/hol/ho_modal.py +1743 -0
- unicode_logic_kit/hol/intuitionistic.py +403 -0
- unicode_logic_kit/hol/isabelle_conditional.py +593 -0
- unicode_logic_kit/hol/isabelle_modal.py +1908 -0
- unicode_logic_kit/hol/isabelle_relevant.py +412 -0
- unicode_logic_kit/hol/isabelle_runner.py +1147 -0
- unicode_logic_kit/hol/isabelle_substructural.py +884 -0
- unicode_logic_kit/hol/lean.py +1018 -0
- unicode_logic_kit/hol/manyvalued.py +921 -0
- unicode_logic_kit/hol/secondorder.py +687 -0
- unicode_logic_kit/hol/thf_modal.py +941 -0
- unicode_logic_kit/hol/thirdorder.py +397 -0
- unicode_logic_kit/ilp/__init__.py +89 -0
- unicode_logic_kit/ilp/readback.py +389 -0
- unicode_logic_kit/ilp/separation.py +153 -0
- unicode_logic_kit/ilp/task.py +730 -0
- unicode_logic_kit/logic.py +163 -0
- unicode_logic_kit/mcp/__init__.py +28 -0
- unicode_logic_kit/mcp/__main__.py +5 -0
- unicode_logic_kit/mcp/chem_tools.py +1031 -0
- unicode_logic_kit/mcp/server.py +2453 -0
- unicode_logic_kit/mcp/syntax_spec.py +681 -0
- unicode_logic_kit/prob/__init__.py +53 -0
- unicode_logic_kit/prob/_bdd.py +225 -0
- unicode_logic_kit/prob/_column_gen.py +668 -0
- unicode_logic_kit/prob/distribution.py +686 -0
- unicode_logic_kit/prob/nilsson.py +470 -0
- unicode_logic_kit/py.typed +0 -0
- unicode_logic_kit/semantics/__init__.py +137 -0
- unicode_logic_kit/semantics/_modal_reject.py +156 -0
- unicode_logic_kit/semantics/action_models.py +466 -0
- unicode_logic_kit/semantics/asp_models.py +1200 -0
- unicode_logic_kit/semantics/conditional.py +580 -0
- unicode_logic_kit/semantics/dynamic_epistemic.py +95 -0
- unicode_logic_kit/semantics/free_logic.py +913 -0
- unicode_logic_kit/semantics/fuzzy.py +384 -0
- unicode_logic_kit/semantics/fuzzy_kripke.py +442 -0
- unicode_logic_kit/semantics/intuitionistic.py +581 -0
- unicode_logic_kit/semantics/kripke.py +1139 -0
- unicode_logic_kit/semantics/manyvalued.py +580 -0
- unicode_logic_kit/semantics/matrix.py +342 -0
- unicode_logic_kit/semantics/model_eval.py +1135 -0
- unicode_logic_kit/semantics/modelfinder.py +1036 -0
- unicode_logic_kit/semantics/nonmonotonic.py +372 -0
- unicode_logic_kit/semantics/relevant.py +331 -0
- unicode_logic_kit/semantics/secondorder.py +657 -0
- unicode_logic_kit/semantics/structures.py +352 -0
- unicode_logic_kit/semantics/tarski.py +975 -0
- unicode_logic_kit/semantics/team.py +315 -0
- unicode_logic_kit/semantics/team_translation.py +416 -0
- unicode_logic_kit/semantics/thirdorder.py +358 -0
- unicode_logic_kit/semantics/tnorm.py +85 -0
- unicode_logic_kit/semantics/truthtable.py +201 -0
- unicode_logic_kit-0.31.0.dist-info/METADATA +333 -0
- unicode_logic_kit-0.31.0.dist-info/RECORD +237 -0
- unicode_logic_kit-0.31.0.dist-info/WHEEL +4 -0
- unicode_logic_kit-0.31.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,1096 @@
|
|
|
1
|
+
"""Independent checker for Vampire/E TSTP derivations (:mod:`atp.tstp`'s read side).
|
|
2
|
+
|
|
3
|
+
:func:`atp.tstp.parse_tstp_derivation` turns a prover's printed proof into a
|
|
4
|
+
:class:`~atp.tstp.TstpDerivation` DAG, but nothing re-derives it — the DAG is
|
|
5
|
+
trusted verbatim. This module is that independent check, in the same spirit
|
|
6
|
+
as :mod:`atp.resolution_check` (an external searcher's derivation is only as
|
|
7
|
+
trustworthy as its checker) and :mod:`atp.twee_check`, but TILED rather than
|
|
8
|
+
uniform: a real captured proof (see ``tests/fixtures/tstp_check/`` and
|
|
9
|
+
``tests/fixtures/eprover_3_5_1_theorem.txt``) mixes three very different
|
|
10
|
+
kinds of step, and this module gives each its own honest treatment instead of
|
|
11
|
+
pretending they are all equally certifiable.
|
|
12
|
+
|
|
13
|
+
Three tiers
|
|
14
|
+
-----------
|
|
15
|
+
|
|
16
|
+
1. **Core, independently-checked rules** (:data:`VAMPIRE_CHECKED_RULES` /
|
|
17
|
+
:data:`EPROVER_CHECKED_RULES`): binary resolution, factoring,
|
|
18
|
+
superposition/paramodulation, equality resolution, forward/backward
|
|
19
|
+
demodulation, forward/backward subsumption resolution, and the removal of
|
|
20
|
+
the literals that are false in every interpretation
|
|
21
|
+
(``true_and_false_elimination``). Each step's
|
|
22
|
+
clause and its cited parents' clauses are converted from the general
|
|
23
|
+
:class:`~fol.nodes.Node` :func:`atp.tstp.parse_tstp_derivation` already
|
|
24
|
+
produced into :mod:`atp.resolution_check`'s frozenset-of-literals clause
|
|
25
|
+
shape (:func:`_node_to_clause`), and the re-derivation itself reuses that
|
|
26
|
+
module's :func:`~atp.resolution_check._unify` /
|
|
27
|
+
:func:`~atp.resolution_check._apply` (a unifier's output, whose bindings are
|
|
28
|
+
followed) / :func:`~atp.resolution_check._apply_matcher` (a one-sided
|
|
29
|
+
matcher, applied in one simultaneous step) /
|
|
30
|
+
:func:`~atp.resolution_check._is_variant` /
|
|
31
|
+
:func:`~atp.resolution_check._term_gt` primitives directly (imported as
|
|
32
|
+
``_rc.<name>`` throughout this module) — no code is shared with any
|
|
33
|
+
*searcher* (:mod:`atp.resolution`, Vampire, E themselves), preserving
|
|
34
|
+
:mod:`atp.resolution_check`'s checker-independence property.
|
|
35
|
+
|
|
36
|
+
TSTP carries no ``eq_literal``/``target_literal``/``direction``/
|
|
37
|
+
``position`` fields the way a hand-built :class:`atp.resolution_check
|
|
38
|
+
.ResolutionStep` does, so the equality-rule checkers here
|
|
39
|
+
(:func:`_check_tstp_superposition`, :func:`_check_tstp_demodulation`,
|
|
40
|
+
:func:`_check_tstp_equality_resolution`) are *generalized* from
|
|
41
|
+
:mod:`atp.resolution_check`'s "trust the stated fields, only re-derive the
|
|
42
|
+
unifier" design into a bounded SEARCH over candidate literals, subterm
|
|
43
|
+
positions (:func:`_all_positions`) and rewrite directions — this is new
|
|
44
|
+
work, not reuse, and is exercised by hand-built accept/reject fixtures in
|
|
45
|
+
``tests/test_tstp_check.py`` for each rule. Forward/backward subsumption
|
|
46
|
+
resolution (:func:`_check_tstp_subsumption_resolution`) is new work too —
|
|
47
|
+
:mod:`atp.resolution_check` has no such rule at all — implemented as a
|
|
48
|
+
bounded backtracking one-sided match (the subsumer clause's variables
|
|
49
|
+
bind, the target clause is held fixed), independently reimplemented here
|
|
50
|
+
rather than reusing :mod:`atp.resolution`'s own subsumption/matching code.
|
|
51
|
+
``true_and_false_elimination`` (Vampire's name for it, captured live in
|
|
52
|
+
``tests/fixtures/tstp_check/vampire_true_and_false_elimination.txt``; the kit's
|
|
53
|
+
own ``truth_constants`` rule is written under it) takes ONE parent and
|
|
54
|
+
licenses its clause minus some literals, each of which is ``$false`` or
|
|
55
|
+
``¬$true`` (:func:`_check_tstp_truth_constants`): dropping any other literal,
|
|
56
|
+
or keeping a literal the parent does not have, is rejected. It reads the
|
|
57
|
+
parent with its constant literals kept (:func:`_node_to_clause` with
|
|
58
|
+
``keep_constants=True``), which no other core rule sees. A step whose
|
|
59
|
+
formula is not a flat clause (Vampire also uses the rule on whole formulas)
|
|
60
|
+
comes back unchecked, never approximately checked.
|
|
61
|
+
|
|
62
|
+
2. **Clausification checked by entailment** (:data:`VAMPIRE_CLAUSIFICATION_RULES` /
|
|
63
|
+
:data:`EPROVER_CLAUSIFICATION_RULES`): steps whose rule identifies them as
|
|
64
|
+
clausification/normalisation (``negated_conjecture``, ``flattening``,
|
|
65
|
+
``(e)nnf_transformation``, ``cnf_transformation``, ``rectify``,
|
|
66
|
+
``shift_quantors``, ``variable_rename``, ``pure_predicate_removal``;
|
|
67
|
+
E's ``assume_negation``, ``fof_simplification``, ``fof_nnf``,
|
|
68
|
+
``split_conjunct``) are not re-derived transformation by transformation
|
|
69
|
+
-- any sound clausifier output passes -- but each one's stated formula
|
|
70
|
+
must be ENTAILED by its (already verified) parents, which Z3 has to
|
|
71
|
+
PROVE within a per-step budget (:func:`_entailed`; ``unknown`` or a
|
|
72
|
+
timeout is a failure, never a pass). That is exactly what a
|
|
73
|
+
refutation's soundness needs: every statement is then a consequence of
|
|
74
|
+
the caller's premises plus the negated conjecture, so a derived
|
|
75
|
+
``$false`` really refutes them.
|
|
76
|
+
|
|
77
|
+
The conjecture needs one more rule, because entailment alone would let
|
|
78
|
+
a proof ASSUME it: a leaf standing for the caller's conclusion may be
|
|
79
|
+
cited only by ``negated_conjecture``/``assume_negation`` (and those may
|
|
80
|
+
cite nothing else), whose formula must be entailed by the conjecture's
|
|
81
|
+
NEGATION. Any other step -- clausification or core rule -- that cites
|
|
82
|
+
the conjecture leaf fails, naming it.
|
|
83
|
+
|
|
84
|
+
Leaves themselves (a :class:`~atp.tstp.TstpStep` whose own ``rule is
|
|
85
|
+
None`` -- a ``file(...)``-sourced original, or any other
|
|
86
|
+
non-``inference(...)`` source) must be an ALPHA-VARIANT
|
|
87
|
+
(:func:`_formula_alpha_equal` -- ordered structural equality up to a
|
|
88
|
+
consistent bound-variable renaming, ``Quantifier``-scope aware) of one of
|
|
89
|
+
the caller's own ``premises``/``conclusion``, so a derivation cannot
|
|
90
|
+
smuggle in an extra axiom as a leaf either.
|
|
91
|
+
|
|
92
|
+
``skolemisation`` is recognised but refused by name (tier
|
|
93
|
+
``"unchecked"``): a Skolemized formula is NOT a consequence of its
|
|
94
|
+
parent -- Skolemization preserves satisfiability only -- so no
|
|
95
|
+
entailment check can license it, and certifying it needs a structural
|
|
96
|
+
"fresh symbol, right argument list" check this module does not do.
|
|
97
|
+
Vampire's Skolemization step also cites a leaf sourced
|
|
98
|
+
``introduced(definition, [], [skolem_symbol_introduction])``, which is not
|
|
99
|
+
one of the caller's premises and fails the leaf check for the same
|
|
100
|
+
reason. Any derivation that Skolemizes therefore comes back unverified;
|
|
101
|
+
see ``tests/fixtures/tstp_check/vampire_skolemisation.txt`` and its test.
|
|
102
|
+
|
|
103
|
+
An earlier version of this tier only walked each clausification step's
|
|
104
|
+
parent chain back to genuine leaves and never looked at the step's own
|
|
105
|
+
formula, so a single ``cnf_transformation`` step could state anything --
|
|
106
|
+
``p(a) |- p(b)`` came back ``verified=True`` -- and a step could cite the
|
|
107
|
+
conjecture positively. Both are pinned as regression tests.
|
|
108
|
+
|
|
109
|
+
3. **Refuse loudly on everything else**: any step whose rule is outside both
|
|
110
|
+
tables above — AVATAR splitting, global subsumption, ``definition_
|
|
111
|
+
unfolding``, equality factoring (deliberately not implemented — see
|
|
112
|
+
below), E's ``cn`` (see below), or any unrecognised/future rule name —
|
|
113
|
+
makes that step, and therefore the WHOLE derivation
|
|
114
|
+
(:attr:`TstpCheckResult.verified`), come back ``False``, naming the
|
|
115
|
+
offending rule (:attr:`TstpStepResult.detail`); never silently accepted.
|
|
116
|
+
|
|
117
|
+
Deliberate scope decisions (not oversights — each is a step this module
|
|
118
|
+
could not honestly certify without materially expanding its scope):
|
|
119
|
+
|
|
120
|
+
* **Equality factoring** is not implemented as a checked rule. It is the
|
|
121
|
+
least-used of the superposition-calculus rules, is not exercised by any
|
|
122
|
+
real fixture this module was developed against, and is not required by
|
|
123
|
+
its own test oracle. A step naming it is honestly reported unchecked
|
|
124
|
+
rather than approximately checked.
|
|
125
|
+
* **E's ``cn`` rule** is not registered in either table. In the one real E
|
|
126
|
+
fixture available (``tests/fixtures/eprover_3_5_1_theorem.txt``), the
|
|
127
|
+
final refutation step's rule is ``cn`` with a SINGLE nested
|
|
128
|
+
``inference(rw, ..., [inference(spm, ..., [...]), ...])`` parent, and an
|
|
129
|
+
earlier step (``c_0_6``, ``fof_nnf``) similarly nests a
|
|
130
|
+
``variable_rename``/``fof_nnf`` chain inside ONE compound parent-list
|
|
131
|
+
entry — :func:`atp.tstp._parse_source` (deliberately, for the proof-DAG-
|
|
132
|
+
display use case it serves) drops a parent-list entry that is itself a
|
|
133
|
+
compound ``inference(...)`` term, so both steps' own
|
|
134
|
+
:attr:`~atp.tstp.TstpStep.parents` come back EMPTY. Even setting that
|
|
135
|
+
parsing choice aside, ``cn`` ("clause normalisation") has no single fixed
|
|
136
|
+
licensing condition the way ``resolution``/``factoring``/etc. do — it is
|
|
137
|
+
E's generic wrapper for "here is the final simplified clause", not one
|
|
138
|
+
calculus rule — so there is nothing well-defined to re-derive even given
|
|
139
|
+
resolvable parents. E's own ``spm``/``rw``/``er`` abbreviations
|
|
140
|
+
(superposition/rewrite=demodulation/equality-resolution, confirmed from
|
|
141
|
+
that same nested record) ARE registered in :data:`EPROVER_CHECKED_RULES`
|
|
142
|
+
for the case where a future E configuration prints one as its own
|
|
143
|
+
top-level statement with resolvable parents. A consequence, confirmed by
|
|
144
|
+
``tests/test_tstp_check.py::test_eprover_real_fixture_is_not_fully_verified``:
|
|
145
|
+
the one available real E fixture comes back UNVERIFIED overall (the
|
|
146
|
+
``fof_nnf``/``cn`` steps come back unchecked/unresolvable), not fully
|
|
147
|
+
verified — the honest outcome given what the parser can actually recover,
|
|
148
|
+
not a bug in this module.
|
|
149
|
+
* **Wiring into ``VampireBackend``/``EProverBackend`` as an opt-in
|
|
150
|
+
``verify_tstp`` option** (this item's original 4th point) is out of scope
|
|
151
|
+
for this module — :mod:`atp.protocol` is not this change's file, see the
|
|
152
|
+
integration notes accompanying it.
|
|
153
|
+
|
|
154
|
+
Public API: :class:`TstpStepResult`, :class:`TstpCheckResult`,
|
|
155
|
+
:func:`check_tstp_derivation`, and the four rule-name tables
|
|
156
|
+
:data:`VAMPIRE_CLAUSIFICATION_RULES`, :data:`VAMPIRE_CHECKED_RULES`,
|
|
157
|
+
:data:`EPROVER_CLAUSIFICATION_RULES`, :data:`EPROVER_CHECKED_RULES`.
|
|
158
|
+
"""
|
|
159
|
+
|
|
160
|
+
from dataclasses import dataclass
|
|
161
|
+
from typing import Callable, Dict, FrozenSet, List, Optional, Sequence, Tuple
|
|
162
|
+
|
|
163
|
+
from ..fol.nodes import (
|
|
164
|
+
Node, Atom, Not, And, Or, Xor, Implies, Iff, Quantifier,
|
|
165
|
+
Variable, Constant, Number, Function, free_variables,
|
|
166
|
+
)
|
|
167
|
+
from . import resolution_check as _rc
|
|
168
|
+
from .tstp import TstpDerivation, TstpStep
|
|
169
|
+
|
|
170
|
+
__all__ = [
|
|
171
|
+
"TstpStepResult", "TstpCheckResult", "check_tstp_derivation",
|
|
172
|
+
"VAMPIRE_CLAUSIFICATION_RULES", "VAMPIRE_CHECKED_RULES",
|
|
173
|
+
"EPROVER_CLAUSIFICATION_RULES", "EPROVER_CHECKED_RULES",
|
|
174
|
+
]
|
|
175
|
+
|
|
176
|
+
|
|
177
|
+
# ---------------------------------------------------------------------------
|
|
178
|
+
# Rule-name tables — separate per prover (their TSTP vocabularies do not
|
|
179
|
+
# overlap in the sense that no name below means something different across
|
|
180
|
+
# the two — see the module docstring), merged for runtime dispatch since a
|
|
181
|
+
# rule's checking logic depends only on the rule's own semantics, not on
|
|
182
|
+
# which prover happened to print it.
|
|
183
|
+
# ---------------------------------------------------------------------------
|
|
184
|
+
|
|
185
|
+
# Confirmed against tests/fixtures/tstp_check/vampire_*.txt (captured live,
|
|
186
|
+
# `vampire --proof tptp --avatar off`, Vampire 5.0.1, during this module's
|
|
187
|
+
# development) and tests/test_tstp.py's own pre-existing _THEOREM_OUTPUT
|
|
188
|
+
# fixture. 'nnf_transformation'/'rectify'/'shift_quantors'/
|
|
189
|
+
# 'pure_predicate_removal' are Vampire's documented preprocessing rule names
|
|
190
|
+
# for constructs the small fixtures above did not individually exercise, but
|
|
191
|
+
# which get the exact same entailment check as the ones that were captured
|
|
192
|
+
# ('skolemisation' is recognised only to be refused by name: see
|
|
193
|
+
# _SATISFIABILITY_ONLY_RULES).
|
|
194
|
+
VAMPIRE_CLAUSIFICATION_RULES: FrozenSet[str] = frozenset({
|
|
195
|
+
"negated_conjecture", "flattening", "nnf_transformation",
|
|
196
|
+
"ennf_transformation", "cnf_transformation", "rectify",
|
|
197
|
+
"shift_quantors", "variable_rename", "pure_predicate_removal",
|
|
198
|
+
"skolemisation",
|
|
199
|
+
})
|
|
200
|
+
|
|
201
|
+
# 'resolution', 'superposition', 'equality_resolution' and
|
|
202
|
+
# 'forward_subsumption_resolution' are each confirmed from a real captured
|
|
203
|
+
# fixture (see tests/fixtures/tstp_check/). 'factoring',
|
|
204
|
+
# 'backward_subsumption_resolution' and 'forward_demodulation'/
|
|
205
|
+
# 'backward_demodulation' are the standard TPTP-family names for the
|
|
206
|
+
# remaining core rules (not independently captured this session) — see the
|
|
207
|
+
# module docstring for 'equality_factoring', deliberately absent.
|
|
208
|
+
# 'true_and_false_elimination' is confirmed from a real captured fixture too
|
|
209
|
+
# (tests/fixtures/tstp_check/vampire_true_and_false_elimination.txt, Vampire
|
|
210
|
+
# 5.0.1): it drops the literals $false and ~$true from a clause.
|
|
211
|
+
VAMPIRE_CHECKED_RULES: FrozenSet[str] = frozenset({
|
|
212
|
+
"resolution", "factoring", "superposition",
|
|
213
|
+
"forward_demodulation", "backward_demodulation",
|
|
214
|
+
"equality_resolution",
|
|
215
|
+
"forward_subsumption_resolution", "backward_subsumption_resolution",
|
|
216
|
+
"true_and_false_elimination",
|
|
217
|
+
})
|
|
218
|
+
|
|
219
|
+
# Confirmed against tests/fixtures/eprover_3_5_1_theorem.txt (already
|
|
220
|
+
# captured for tests/test_tstp.py / tests/test_eprover_zipperposition.py):
|
|
221
|
+
# 'assume_negation', 'fof_simplification', 'fof_nnf', 'variable_rename' and
|
|
222
|
+
# 'split_conjunct' all appear there as genuine top-level steps.
|
|
223
|
+
EPROVER_CLAUSIFICATION_RULES: FrozenSet[str] = frozenset({
|
|
224
|
+
"assume_negation", "fof_simplification", "fof_nnf", "variable_rename",
|
|
225
|
+
"split_conjunct",
|
|
226
|
+
})
|
|
227
|
+
|
|
228
|
+
# 'spm'/'rw'/'er' appear in that same fixture, but only NESTED inside its
|
|
229
|
+
# final 'cn' step's source record — never as their own top-level statement.
|
|
230
|
+
# Registered here (same generalized checkers as Vampire's 'superposition'/
|
|
231
|
+
# 'demodulation'/'equality_resolution') for when a top-level step uses one
|
|
232
|
+
# directly; see the module docstring for why 'cn' itself is not registered.
|
|
233
|
+
EPROVER_CHECKED_RULES: FrozenSet[str] = frozenset({"spm", "rw", "er"})
|
|
234
|
+
|
|
235
|
+
_CLAUSIFICATION_RULES: FrozenSet[str] = VAMPIRE_CLAUSIFICATION_RULES | EPROVER_CLAUSIFICATION_RULES
|
|
236
|
+
|
|
237
|
+
# The only rules allowed to cite the CONJECTURE itself: a refutation derives
|
|
238
|
+
# $false from premises + negated conjecture, so the conjecture enters exactly
|
|
239
|
+
# once, negated. Any other step citing it would be assuming what is to be proved.
|
|
240
|
+
_NEGATION_RULES: FrozenSet[str] = frozenset({"negated_conjecture", "assume_negation"})
|
|
241
|
+
|
|
242
|
+
# Recognised, but refused by name: Skolemisation preserves SATISFIABILITY
|
|
243
|
+
# only -- the Skolemized formula is not a logical consequence of its parent --
|
|
244
|
+
# so no entailment check can license it, and certifying it needs the
|
|
245
|
+
# structural "fresh symbol, right argument list" check this module does not do.
|
|
246
|
+
_SATISFIABILITY_ONLY_RULES: FrozenSet[str] = frozenset({"skolemisation"})
|
|
247
|
+
|
|
248
|
+
# Per-step budget for the Z3 entailment check. A timeout is NOT a pass: the
|
|
249
|
+
# step then comes back unconfirmed, never verified.
|
|
250
|
+
_ENTAILMENT_TIMEOUT_MS = 5000
|
|
251
|
+
|
|
252
|
+
|
|
253
|
+
# ---------------------------------------------------------------------------
|
|
254
|
+
# Node -> clause conversion: a TstpStep.formula (a general Node, possibly
|
|
255
|
+
# carrying a leading universal-quantifier prefix the way an un-Skolemized
|
|
256
|
+
# clausal fof step does) into resolution_check.py's frozenset-of-literals
|
|
257
|
+
# clause shape. Returns None — refuse rather than guess — for any formula
|
|
258
|
+
# shape that is not a flat, already-clausal disjunction of literals (an And,
|
|
259
|
+
# an Implies/Iff/Xor at the matrix level, a leading existential, ...): such a
|
|
260
|
+
# step simply cannot be treated as a clause by this module's core-checked
|
|
261
|
+
# tier, and the caller reports that step unverified rather than misreading it.
|
|
262
|
+
# ---------------------------------------------------------------------------
|
|
263
|
+
|
|
264
|
+
def _normalize_literal(lit: Node) -> Node:
|
|
265
|
+
"""Rewrite TPTP's dedicated disequality atom into resolution_check.py's
|
|
266
|
+
``Not(Atom("=", ...))`` shape (and the symmetric double-negative), so
|
|
267
|
+
every downstream helper (:func:`~atp.resolution_check._lit_atom_polarity`,
|
|
268
|
+
``_is_equality_atom``, ...) only ever has to recognise ONE negative-
|
|
269
|
+
equality shape. :func:`fol.tptp_input.parse_tptp_formula` maps TPTP's
|
|
270
|
+
``a != b`` to a dedicated ``Atom("≠", [a, b])`` (never to
|
|
271
|
+
``Not(Atom("=", ...))``) — see that module's docstring — so without this
|
|
272
|
+
normalization a real disequality literal (e.g. the ``a != X0`` in
|
|
273
|
+
``tests/fixtures/tstp_check/vampire_equality_resolution.txt``) would be
|
|
274
|
+
invisible to every equality-rule checker below.
|
|
275
|
+
"""
|
|
276
|
+
if isinstance(lit, Atom) and lit.predicate == "≠" and len(lit.args) == 2:
|
|
277
|
+
return Not(Atom("=", list(lit.args)))
|
|
278
|
+
if (isinstance(lit, Not) and isinstance(lit.formula, Atom)
|
|
279
|
+
and lit.formula.predicate == "≠" and len(lit.formula.args) == 2):
|
|
280
|
+
return Atom("=", list(lit.formula.args))
|
|
281
|
+
return lit
|
|
282
|
+
|
|
283
|
+
|
|
284
|
+
def _flatten_or_into(node: Node, out: List[Node], keep_constants: bool = False) -> bool:
|
|
285
|
+
"""Append ``node``'s literals (recursing through ``Or``) onto ``out``.
|
|
286
|
+
|
|
287
|
+
Returns False — leaving ``out`` in an unspecified partial state the
|
|
288
|
+
caller discards — the moment a non-literal, non-``Or`` node is reached,
|
|
289
|
+
or a literal reduces to the ``$true``/``$false`` marker atoms (neither is
|
|
290
|
+
a genuine literal within a clause; ``$false`` alone as the WHOLE matrix
|
|
291
|
+
is handled separately by :func:`_node_to_clause`, as the empty clause).
|
|
292
|
+
With ``keep_constants`` the marker atoms are kept as literals instead: the
|
|
293
|
+
one rule that reads them (``true_and_false_elimination``) needs to see
|
|
294
|
+
which ones its parent holds.
|
|
295
|
+
"""
|
|
296
|
+
if isinstance(node, Or):
|
|
297
|
+
return (_flatten_or_into(node.left, out, keep_constants)
|
|
298
|
+
and _flatten_or_into(node.right, out, keep_constants))
|
|
299
|
+
lit = _normalize_literal(node)
|
|
300
|
+
parsed = _rc._lit_atom_polarity(lit)
|
|
301
|
+
if parsed is None:
|
|
302
|
+
return False
|
|
303
|
+
atom, _is_positive = parsed
|
|
304
|
+
if atom.predicate in ("$true", "$false") and not keep_constants:
|
|
305
|
+
return False
|
|
306
|
+
out.append(lit)
|
|
307
|
+
return True
|
|
308
|
+
|
|
309
|
+
|
|
310
|
+
def _node_to_clause(node: Node, keep_constants: bool = False) -> Optional[FrozenSet[Node]]:
|
|
311
|
+
"""Convert a TSTP step's formula into a clause (frozenset of literals).
|
|
312
|
+
|
|
313
|
+
Strips a leading chain of universal (``"∀"``) quantifiers (an
|
|
314
|
+
existential anywhere in that prefix means this is not a clause — a
|
|
315
|
+
genuinely clausal TSTP step is always implicitly-or-explicitly
|
|
316
|
+
universally quantified), then reads the remaining matrix as ``$false``
|
|
317
|
+
(the empty clause) or a flat ``Or``-tree of literals
|
|
318
|
+
(:func:`_flatten_or_into`). Returns ``None`` on any other shape.
|
|
319
|
+
``keep_constants`` keeps a ``$true`` / ``$false`` literal of a longer clause
|
|
320
|
+
as a literal (see :func:`_flatten_or_into`); the formula ``$false`` alone is
|
|
321
|
+
the empty clause either way.
|
|
322
|
+
"""
|
|
323
|
+
n = node
|
|
324
|
+
while isinstance(n, Quantifier):
|
|
325
|
+
if n.type != "∀":
|
|
326
|
+
return None
|
|
327
|
+
n = n.formula
|
|
328
|
+
if isinstance(n, Atom) and n.predicate == "$false" and not n.args:
|
|
329
|
+
return frozenset()
|
|
330
|
+
literals: List[Node] = []
|
|
331
|
+
if not _flatten_or_into(n, literals, keep_constants):
|
|
332
|
+
return None
|
|
333
|
+
return frozenset(literals)
|
|
334
|
+
|
|
335
|
+
|
|
336
|
+
# ---------------------------------------------------------------------------
|
|
337
|
+
# Formula-level alpha-equivalence (leaf-boundary check, tier 2). Deliberately
|
|
338
|
+
# an ORDERED structural comparison (unlike resolution_check.py's clause-level
|
|
339
|
+
# _is_variant, which permutes an unordered literal set) -- a leaf formula is
|
|
340
|
+
# a near-verbatim restatement of ONE whole original formula, so its
|
|
341
|
+
# top-level shape (which side of an Implies, which And/Or came first, ...)
|
|
342
|
+
# is expected to match exactly; only the concrete spelling of bound variable
|
|
343
|
+
# names may differ. A free (unbound-in-this-walk) Variable must match by
|
|
344
|
+
# name literally -- premises/conclusions passed to check_tstp_derivation are
|
|
345
|
+
# expected to be closed sentences or ground atoms, so this case is not
|
|
346
|
+
# exercised by any real derivation, but it is the safe (never a false
|
|
347
|
+
# "matches") default when it is.
|
|
348
|
+
# ---------------------------------------------------------------------------
|
|
349
|
+
|
|
350
|
+
def _formula_alpha_equal(a: Node, b: Node,
|
|
351
|
+
fwd: Optional[Dict[str, str]] = None,
|
|
352
|
+
bwd: Optional[Dict[str, str]] = None) -> bool:
|
|
353
|
+
"""True iff ``a`` and ``b`` are the same formula up to a consistent
|
|
354
|
+
renaming of bound variables (Quantifier-scope aware)."""
|
|
355
|
+
if fwd is None:
|
|
356
|
+
fwd, bwd = {}, {}
|
|
357
|
+
if type(a) is not type(b):
|
|
358
|
+
return False
|
|
359
|
+
if isinstance(a, Variable):
|
|
360
|
+
mapped = fwd.get(a.name)
|
|
361
|
+
if mapped is not None:
|
|
362
|
+
return mapped == b.name
|
|
363
|
+
if b.name in bwd:
|
|
364
|
+
return False
|
|
365
|
+
return a.name == b.name
|
|
366
|
+
if isinstance(a, Constant):
|
|
367
|
+
return a.name == b.name
|
|
368
|
+
if isinstance(a, Number):
|
|
369
|
+
return a.value == b.value
|
|
370
|
+
if isinstance(a, Function):
|
|
371
|
+
if a.name != b.name or len(a.args) != len(b.args):
|
|
372
|
+
return False
|
|
373
|
+
return all(_formula_alpha_equal(x, y, fwd, bwd) for x, y in zip(a.args, b.args))
|
|
374
|
+
if isinstance(a, Atom):
|
|
375
|
+
if a.predicate != b.predicate or len(a.args) != len(b.args):
|
|
376
|
+
return False
|
|
377
|
+
return all(_formula_alpha_equal(x, y, fwd, bwd) for x, y in zip(a.args, b.args))
|
|
378
|
+
if isinstance(a, Not):
|
|
379
|
+
return _formula_alpha_equal(a.formula, b.formula, fwd, bwd)
|
|
380
|
+
if isinstance(a, (And, Or, Xor, Implies, Iff)):
|
|
381
|
+
return (_formula_alpha_equal(a.left, b.left, fwd, bwd)
|
|
382
|
+
and _formula_alpha_equal(a.right, b.right, fwd, bwd))
|
|
383
|
+
if isinstance(a, Quantifier):
|
|
384
|
+
if a.type != b.type:
|
|
385
|
+
return False
|
|
386
|
+
new_fwd = dict(fwd)
|
|
387
|
+
new_fwd[a.variable.name] = b.variable.name
|
|
388
|
+
new_bwd = dict(bwd)
|
|
389
|
+
new_bwd[b.variable.name] = a.variable.name
|
|
390
|
+
return _formula_alpha_equal(a.formula, b.formula, new_fwd, new_bwd)
|
|
391
|
+
return a == b
|
|
392
|
+
|
|
393
|
+
|
|
394
|
+
# ---------------------------------------------------------------------------
|
|
395
|
+
# Per-rule checkers (core-checked tier). Each has the signature
|
|
396
|
+
# ``(clause, *parent_clauses) -> Optional[str]`` (None = licensed, else an
|
|
397
|
+
# error string without a "step N:" prefix — the caller adds that), mirroring
|
|
398
|
+
# resolution_check.py's per-rule checkers but reused as a bounded SEARCH
|
|
399
|
+
# (TSTP names no literal/position/direction explicitly) instead of trusting
|
|
400
|
+
# caller-stated fields.
|
|
401
|
+
# ---------------------------------------------------------------------------
|
|
402
|
+
|
|
403
|
+
def _check_tstp_resolve(clause: FrozenSet[Node], ci: FrozenSet[Node],
|
|
404
|
+
cj: FrozenSet[Node]) -> Optional[str]:
|
|
405
|
+
"""``"resolution"``: some complementary-polarity literal pair's mgu must
|
|
406
|
+
give ``clause`` -- direct port of
|
|
407
|
+
:func:`atp.resolution_check._check_resolve_step`."""
|
|
408
|
+
ci2, cj2 = _rc._standardize_apart(ci, cj)
|
|
409
|
+
found_complementary_pair = False
|
|
410
|
+
for lit1 in sorted(ci2, key=_rc._lit_key):
|
|
411
|
+
parsed1 = _rc._lit_atom_polarity(lit1)
|
|
412
|
+
if parsed1 is None:
|
|
413
|
+
continue
|
|
414
|
+
atom1, pos1 = parsed1
|
|
415
|
+
for lit2 in sorted(cj2, key=_rc._lit_key):
|
|
416
|
+
parsed2 = _rc._lit_atom_polarity(lit2)
|
|
417
|
+
if parsed2 is None:
|
|
418
|
+
continue
|
|
419
|
+
atom2, pos2 = parsed2
|
|
420
|
+
if pos1 == pos2:
|
|
421
|
+
continue
|
|
422
|
+
found_complementary_pair = True
|
|
423
|
+
sigma = _rc._unify(atom1, atom2)
|
|
424
|
+
if sigma is None:
|
|
425
|
+
continue
|
|
426
|
+
resolvent = frozenset(
|
|
427
|
+
[_rc._apply_literal(l, sigma) for l in ci2 if l != lit1]
|
|
428
|
+
+ [_rc._apply_literal(l, sigma) for l in cj2 if l != lit2]
|
|
429
|
+
)
|
|
430
|
+
if _rc._is_variant(resolvent, clause):
|
|
431
|
+
return None
|
|
432
|
+
if not found_complementary_pair:
|
|
433
|
+
return "'resolution': the parent clauses contain no complementary-polarity literal pair"
|
|
434
|
+
return "'resolution': no complementary-polarity literal pair's mgu produces the stated clause"
|
|
435
|
+
|
|
436
|
+
|
|
437
|
+
def _check_tstp_factor(clause: FrozenSet[Node], ci: FrozenSet[Node]) -> Optional[str]:
|
|
438
|
+
"""``"factoring"``: some same-polarity literal pair's mgu must give
|
|
439
|
+
``clause`` -- direct port of
|
|
440
|
+
:func:`atp.resolution_check._check_factor_step`."""
|
|
441
|
+
literals = sorted(ci, key=_rc._lit_key)
|
|
442
|
+
for a_idx in range(len(literals)):
|
|
443
|
+
parsed_a = _rc._lit_atom_polarity(literals[a_idx])
|
|
444
|
+
if parsed_a is None:
|
|
445
|
+
continue
|
|
446
|
+
atom_a, pos_a = parsed_a
|
|
447
|
+
for b_idx in range(a_idx + 1, len(literals)):
|
|
448
|
+
parsed_b = _rc._lit_atom_polarity(literals[b_idx])
|
|
449
|
+
if parsed_b is None:
|
|
450
|
+
continue
|
|
451
|
+
atom_b, pos_b = parsed_b
|
|
452
|
+
if pos_a != pos_b:
|
|
453
|
+
continue
|
|
454
|
+
sigma = _rc._unify(atom_a, atom_b)
|
|
455
|
+
if sigma is None:
|
|
456
|
+
continue
|
|
457
|
+
factored = frozenset(_rc._apply_literal(l, sigma) for l in ci)
|
|
458
|
+
if _rc._is_variant(factored, clause):
|
|
459
|
+
return None
|
|
460
|
+
return "'factoring': no same-polarity literal pair's mgu produces the stated clause"
|
|
461
|
+
|
|
462
|
+
|
|
463
|
+
def _check_tstp_equality_resolution(clause: FrozenSet[Node],
|
|
464
|
+
ci: FrozenSet[Node]) -> Optional[str]:
|
|
465
|
+
"""``"equality_resolution"``/E's ``"er"``: some negative equality
|
|
466
|
+
literal's own two sides must unify, and dropping it under that unifier
|
|
467
|
+
must give ``clause`` -- generalizes
|
|
468
|
+
:func:`atp.resolution_check._check_reflexivity_step` (which trusts a
|
|
469
|
+
caller-stated ``eq_literal``) into a search over every negative equality
|
|
470
|
+
literal of ``ci``."""
|
|
471
|
+
found_negative_equality = False
|
|
472
|
+
for lit in sorted(ci, key=_rc._lit_key):
|
|
473
|
+
parsed = _rc._lit_atom_polarity(lit)
|
|
474
|
+
if parsed is None:
|
|
475
|
+
continue
|
|
476
|
+
atom, is_positive = parsed
|
|
477
|
+
if is_positive or not _rc._is_equality_atom(atom):
|
|
478
|
+
continue
|
|
479
|
+
found_negative_equality = True
|
|
480
|
+
u, v = atom.args
|
|
481
|
+
sigma = _rc._unify(u, v)
|
|
482
|
+
if sigma is None:
|
|
483
|
+
continue
|
|
484
|
+
expected = frozenset(_rc._apply_literal(l, sigma) for l in ci if l != lit)
|
|
485
|
+
if _rc._is_variant(expected, clause):
|
|
486
|
+
return None
|
|
487
|
+
if not found_negative_equality:
|
|
488
|
+
return "'equality_resolution': the parent clause has no negative equality literal"
|
|
489
|
+
return "'equality_resolution': no negative equality literal's self-unifier produces the stated clause"
|
|
490
|
+
|
|
491
|
+
|
|
492
|
+
def _all_positions(node: Node) -> List[Tuple[int, ...]]:
|
|
493
|
+
"""Every non-empty argument-index position reachable within ``node``
|
|
494
|
+
(an :class:`Atom` or :class:`Function`), depth-first, including
|
|
495
|
+
positions inside nested :class:`Function` arguments. Excludes the empty
|
|
496
|
+
position — a position always addresses a subterm reached by descending
|
|
497
|
+
at least one argument, never the whole atom itself (see
|
|
498
|
+
:func:`atp.resolution_check._term_at`'s docstring)."""
|
|
499
|
+
positions: List[Tuple[int, ...]] = []
|
|
500
|
+
|
|
501
|
+
def walk(n: Node, prefix: Tuple[int, ...]) -> None:
|
|
502
|
+
if isinstance(n, (Atom, Function)):
|
|
503
|
+
for i, arg in enumerate(n.args):
|
|
504
|
+
positions.append(prefix + (i,))
|
|
505
|
+
walk(arg, prefix + (i,))
|
|
506
|
+
|
|
507
|
+
walk(node, ())
|
|
508
|
+
return positions
|
|
509
|
+
|
|
510
|
+
|
|
511
|
+
def _check_tstp_superposition(clause: FrozenSet[Node], ci: FrozenSet[Node],
|
|
512
|
+
cj: FrozenSet[Node]) -> Optional[str]:
|
|
513
|
+
"""``"superposition"``/E's ``"spm"``: generalizes
|
|
514
|
+
:func:`atp.resolution_check._check_paramodulate_step` into a bounded
|
|
515
|
+
search over which parent supplies the positive equality literal, which
|
|
516
|
+
literal it is, its rewrite direction, which literal of the OTHER parent
|
|
517
|
+
is the target, and which subterm position of that literal is rewritten
|
|
518
|
+
(TSTP names none of these explicitly). Both parent-role assignments are
|
|
519
|
+
tried (the real captured fixtures in ``tests/fixtures/tstp_check/`` cite
|
|
520
|
+
the TARGET clause first and the EQUATION clause second — the opposite of
|
|
521
|
+
:mod:`atp.resolution_check`'s own convention — so this checker does not
|
|
522
|
+
assume either order)."""
|
|
523
|
+
for eq_side, tgt_side in ((cj, ci), (ci, cj)):
|
|
524
|
+
eq_side2, tgt_side2 = _rc._standardize_apart(eq_side, tgt_side)
|
|
525
|
+
eq_candidates = []
|
|
526
|
+
for lit in eq_side2:
|
|
527
|
+
parsed = _rc._lit_atom_polarity(lit)
|
|
528
|
+
if parsed is None:
|
|
529
|
+
continue
|
|
530
|
+
atom, is_positive = parsed
|
|
531
|
+
if is_positive and _rc._is_equality_atom(atom):
|
|
532
|
+
eq_candidates.append((lit, atom))
|
|
533
|
+
for eq_lit, eq_atom in eq_candidates:
|
|
534
|
+
for direction in ("lr", "rl"):
|
|
535
|
+
frm, to = _rc._direction_sides(eq_atom, direction)
|
|
536
|
+
for tgt_lit in tgt_side2:
|
|
537
|
+
parsed_tgt = _rc._lit_atom_polarity(tgt_lit)
|
|
538
|
+
if parsed_tgt is None:
|
|
539
|
+
continue
|
|
540
|
+
tgt_atom, tgt_is_positive = parsed_tgt
|
|
541
|
+
for position in _all_positions(tgt_atom):
|
|
542
|
+
try:
|
|
543
|
+
subterm = _rc._term_at(tgt_atom, position)
|
|
544
|
+
except (IndexError, TypeError, AttributeError):
|
|
545
|
+
continue
|
|
546
|
+
sigma = _rc._unify(frm, subterm)
|
|
547
|
+
if sigma is None:
|
|
548
|
+
continue
|
|
549
|
+
try:
|
|
550
|
+
rewritten_atom = _rc._replace_at(tgt_atom, position, to)
|
|
551
|
+
except (IndexError, TypeError, AttributeError):
|
|
552
|
+
continue
|
|
553
|
+
rewritten_lit = (rewritten_atom if tgt_is_positive
|
|
554
|
+
else Not(rewritten_atom))
|
|
555
|
+
expected = frozenset(
|
|
556
|
+
[_rc._apply_literal(l, sigma) for l in eq_side2 if l != eq_lit]
|
|
557
|
+
+ [_rc._apply_literal(l, sigma) for l in tgt_side2 if l != tgt_lit]
|
|
558
|
+
+ [_rc._apply_literal(rewritten_lit, sigma)]
|
|
559
|
+
)
|
|
560
|
+
if _rc._is_variant(expected, clause):
|
|
561
|
+
return None
|
|
562
|
+
return ("'superposition': no positive equality literal, rewrite position and "
|
|
563
|
+
"direction reproduces the stated clause")
|
|
564
|
+
|
|
565
|
+
|
|
566
|
+
def _check_tstp_demodulation(clause: FrozenSet[Node], ci: FrozenSet[Node],
|
|
567
|
+
cj: FrozenSet[Node]) -> Optional[str]:
|
|
568
|
+
"""``"forward_demodulation"``/``"backward_demodulation"``/E's ``"rw"``:
|
|
569
|
+
generalizes :func:`atp.resolution_check._check_demodulate_step` (one-
|
|
570
|
+
sided MATCHING, not unification, with the rewrite orientation re-checked
|
|
571
|
+
against the term order) into a search over which parent is the UNIT
|
|
572
|
+
equation, which literal of the other parent is the target, and which
|
|
573
|
+
subterm position is rewritten. Whichever of ``ci``/``cj`` is a unit
|
|
574
|
+
clause carrying a positive equality literal is tried as the equation
|
|
575
|
+
side; if both are, both are tried.
|
|
576
|
+
|
|
577
|
+
The matcher binds the equation's variables to subterms of the target, and the
|
|
578
|
+
two clauses are not standardized apart (a prover numbers the variables of every
|
|
579
|
+
clause from ``X0``), so an image can be spelled like a variable the matcher
|
|
580
|
+
binds. The matcher is therefore applied to the right-hand side in ONE
|
|
581
|
+
simultaneous step (:func:`atp.resolution_check._apply_matcher`), never by
|
|
582
|
+
following its bindings."""
|
|
583
|
+
for tgt_side, eq_side in ((cj, ci), (ci, cj)):
|
|
584
|
+
if len(eq_side) != 1:
|
|
585
|
+
continue
|
|
586
|
+
eq_lit = next(iter(eq_side))
|
|
587
|
+
parsed_eq = _rc._lit_atom_polarity(eq_lit)
|
|
588
|
+
if parsed_eq is None:
|
|
589
|
+
continue
|
|
590
|
+
eq_atom, eq_is_positive = parsed_eq
|
|
591
|
+
if not eq_is_positive or not _rc._is_equality_atom(eq_atom):
|
|
592
|
+
continue
|
|
593
|
+
for direction in ("lr", "rl"):
|
|
594
|
+
l, r = _rc._direction_sides(eq_atom, direction)
|
|
595
|
+
for tgt_lit in tgt_side:
|
|
596
|
+
parsed_tgt = _rc._lit_atom_polarity(tgt_lit)
|
|
597
|
+
if parsed_tgt is None:
|
|
598
|
+
continue
|
|
599
|
+
tgt_atom, tgt_is_positive = parsed_tgt
|
|
600
|
+
for position in _all_positions(tgt_atom):
|
|
601
|
+
try:
|
|
602
|
+
subterm = _rc._term_at(tgt_atom, position)
|
|
603
|
+
except (IndexError, TypeError, AttributeError):
|
|
604
|
+
continue
|
|
605
|
+
sigma = _rc._match_term(l, subterm, {})
|
|
606
|
+
if sigma is None:
|
|
607
|
+
continue
|
|
608
|
+
r_sigma = _rc._apply_matcher(r, sigma)
|
|
609
|
+
if not _rc._term_gt(subterm, r_sigma):
|
|
610
|
+
continue
|
|
611
|
+
try:
|
|
612
|
+
rewritten_atom = _rc._replace_at(tgt_atom, position, r_sigma)
|
|
613
|
+
except (IndexError, TypeError, AttributeError):
|
|
614
|
+
continue
|
|
615
|
+
rewritten_lit = (rewritten_atom if tgt_is_positive
|
|
616
|
+
else Not(rewritten_atom))
|
|
617
|
+
rest = [l2 for l2 in tgt_side if l2 != tgt_lit]
|
|
618
|
+
expected = frozenset(rest + [rewritten_lit])
|
|
619
|
+
if _rc._is_variant(expected, clause):
|
|
620
|
+
return None
|
|
621
|
+
return ("'demodulation': no unit positive equality parent orients, matches and "
|
|
622
|
+
"rewrites into the stated clause")
|
|
623
|
+
|
|
624
|
+
|
|
625
|
+
def _match_atom_onesided(pattern: Atom, target: Atom,
|
|
626
|
+
subst: Dict[str, Node]) -> Optional[Dict[str, Node]]:
|
|
627
|
+
"""One-sided match of ``pattern``'s arguments against ``target``'s
|
|
628
|
+
(``pattern``'s variables bind, ``target``'s are held fixed), extending
|
|
629
|
+
``subst`` argument by argument via
|
|
630
|
+
:func:`atp.resolution_check._match_term`."""
|
|
631
|
+
if pattern.predicate != target.predicate or len(pattern.args) != len(target.args):
|
|
632
|
+
return None
|
|
633
|
+
for p_arg, t_arg in zip(pattern.args, target.args):
|
|
634
|
+
subst = _rc._match_term(p_arg, t_arg, subst)
|
|
635
|
+
if subst is None:
|
|
636
|
+
return None
|
|
637
|
+
return subst
|
|
638
|
+
|
|
639
|
+
|
|
640
|
+
def _match_literals_into(remaining: List[Node], pool: List[Node],
|
|
641
|
+
subst: Dict[str, Node]) -> Optional[Dict[str, Node]]:
|
|
642
|
+
"""Backtracking search: extend ``subst`` (one-sided, ``remaining``'s
|
|
643
|
+
variables bind) so every literal in ``remaining`` matches SOME literal of
|
|
644
|
+
``pool`` (order-independent, repeats allowed — soundness only needs
|
|
645
|
+
membership, not an injective mapping)."""
|
|
646
|
+
if not remaining:
|
|
647
|
+
return subst
|
|
648
|
+
lit = remaining[0]
|
|
649
|
+
parsed_lit = _rc._lit_atom_polarity(lit)
|
|
650
|
+
if parsed_lit is None:
|
|
651
|
+
return None
|
|
652
|
+
atom_lit, pos_lit = parsed_lit
|
|
653
|
+
for candidate in pool:
|
|
654
|
+
parsed_cand = _rc._lit_atom_polarity(candidate)
|
|
655
|
+
if parsed_cand is None:
|
|
656
|
+
continue
|
|
657
|
+
atom_cand, pos_cand = parsed_cand
|
|
658
|
+
if pos_lit != pos_cand:
|
|
659
|
+
continue
|
|
660
|
+
new_subst = _match_atom_onesided(atom_lit, atom_cand, subst)
|
|
661
|
+
if new_subst is None:
|
|
662
|
+
continue
|
|
663
|
+
result = _match_literals_into(remaining[1:], pool, new_subst)
|
|
664
|
+
if result is not None:
|
|
665
|
+
return result
|
|
666
|
+
return None
|
|
667
|
+
|
|
668
|
+
|
|
669
|
+
def _check_tstp_subsumption_resolution(clause: FrozenSet[Node], target: FrozenSet[Node],
|
|
670
|
+
subsumer: FrozenSet[Node]) -> Optional[str]:
|
|
671
|
+
"""``"forward_subsumption_resolution"``/``"backward_subsumption_resolution"``:
|
|
672
|
+
a rule :mod:`atp.resolution_check` does not have at all. ``target`` is
|
|
673
|
+
the FIRST cited parent, ``subsumer`` the SECOND (matching the real
|
|
674
|
+
captured fixtures in ``tests/fixtures/tstp_check/``). Licensing
|
|
675
|
+
condition: some literal ``L`` of ``subsumer`` has a complement that
|
|
676
|
+
one-sided-MATCHES (subsumer's variables bind, target's are held fixed)
|
|
677
|
+
some literal ``M`` of ``target``, AND every other literal of
|
|
678
|
+
``subsumer`` also matches (under the SAME substitution) some literal of
|
|
679
|
+
``target`` OTHER THAN ``M`` — then ``target`` minus ``M`` is licensed.
|
|
680
|
+
The "other than M" restriction is required, not cosmetic: resolving
|
|
681
|
+
``L ∨ Rest`` against ``M ∨ D'`` via ordinary binary resolution gives
|
|
682
|
+
``Restσ ∨ D'``, and this inference is sound only because that resolvent
|
|
683
|
+
is already SUBSUMED by (syntactically contained in) ``D'`` when
|
|
684
|
+
``Restσ ⊆ D'`` — the rest of the subsumer must map into the target
|
|
685
|
+
clause with ``M`` itself already removed, not into the target clause
|
|
686
|
+
still carrying ``M``. Letting ``Restσ`` re-match ``M`` would license
|
|
687
|
+
dropping ``M`` using ``M`` as its own witness (e.g. a tautologous or
|
|
688
|
+
self-redundant ``subsumer`` could "explain away" an arbitrary literal of
|
|
689
|
+
``target``, which is unsound). Both are bounded backtracking searches
|
|
690
|
+
(:func:`_match_literals_into` for "every other literal"; the outer loop
|
|
691
|
+
below for the choice of ``L``/``M``)."""
|
|
692
|
+
target_lits = list(target)
|
|
693
|
+
subsumer_lits = list(subsumer)
|
|
694
|
+
for l_idx, lit_c in enumerate(subsumer_lits):
|
|
695
|
+
parsed_c = _rc._lit_atom_polarity(lit_c)
|
|
696
|
+
if parsed_c is None:
|
|
697
|
+
continue
|
|
698
|
+
atom_c, pos_c = parsed_c
|
|
699
|
+
for m_idx, lit_d in enumerate(target_lits):
|
|
700
|
+
parsed_d = _rc._lit_atom_polarity(lit_d)
|
|
701
|
+
if parsed_d is None:
|
|
702
|
+
continue
|
|
703
|
+
atom_d, pos_d = parsed_d
|
|
704
|
+
if pos_c == pos_d:
|
|
705
|
+
continue # L's COMPLEMENT must match M -- same polarity can't
|
|
706
|
+
theta0 = _match_atom_onesided(atom_c, atom_d, {})
|
|
707
|
+
if theta0 is None:
|
|
708
|
+
continue
|
|
709
|
+
rest_c = [l for i, l in enumerate(subsumer_lits) if i != l_idx]
|
|
710
|
+
rest_target = [l for i, l in enumerate(target_lits) if i != m_idx]
|
|
711
|
+
theta = _match_literals_into(rest_c, rest_target, theta0)
|
|
712
|
+
if theta is None:
|
|
713
|
+
continue
|
|
714
|
+
expected = frozenset(rest_target)
|
|
715
|
+
if _rc._is_variant(expected, clause):
|
|
716
|
+
return None
|
|
717
|
+
return ("'subsumption_resolution': no subsumer literal's complement plus a "
|
|
718
|
+
"one-sided match of its clause-mates licenses the stated clause")
|
|
719
|
+
|
|
720
|
+
|
|
721
|
+
def _is_false_constant_literal(lit: Node) -> bool:
|
|
722
|
+
"""Whether ``lit`` holds in no interpretation: the atom ``$false`` or the
|
|
723
|
+
negation of the atom ``$true``."""
|
|
724
|
+
parsed = _rc._lit_atom_polarity(lit)
|
|
725
|
+
if parsed is None:
|
|
726
|
+
return False
|
|
727
|
+
atom, is_positive = parsed
|
|
728
|
+
if atom.args:
|
|
729
|
+
return False
|
|
730
|
+
return (atom.predicate == "$false") if is_positive else (atom.predicate == "$true")
|
|
731
|
+
|
|
732
|
+
|
|
733
|
+
def _check_tstp_truth_constants(clause: FrozenSet[Node], ci: FrozenSet[Node]) -> Optional[str]:
|
|
734
|
+
"""``"true_and_false_elimination"``: the stated clause is the parent clause
|
|
735
|
+
without some literals, each of which is false in every interpretation
|
|
736
|
+
(``$false`` or ``¬$true``) — and without any other literal.
|
|
737
|
+
|
|
738
|
+
Both clauses are read with their constant literals kept
|
|
739
|
+
(:func:`_node_to_clause` with ``keep_constants=True``). The literals that are
|
|
740
|
+
not such a constant must be the same on both sides up to a renaming of
|
|
741
|
+
variables (a prover renames the variables of each statement); the constant
|
|
742
|
+
literals of the stated clause must be among the parent's, so the step can
|
|
743
|
+
neither keep a literal the parent does not have nor drop a literal that is
|
|
744
|
+
not false. ``$true`` and ``¬$false`` are NOT false: a step that drops one of
|
|
745
|
+
them (a tautology's literal) is rejected like any other dropped literal.
|
|
746
|
+
"""
|
|
747
|
+
kept_by_parent = frozenset(lit for lit in ci if not _is_false_constant_literal(lit))
|
|
748
|
+
constants_of_parent = frozenset(lit for lit in ci if _is_false_constant_literal(lit))
|
|
749
|
+
kept_by_step = frozenset(lit for lit in clause if not _is_false_constant_literal(lit))
|
|
750
|
+
constants_of_step = frozenset(lit for lit in clause if _is_false_constant_literal(lit))
|
|
751
|
+
if not constants_of_step <= constants_of_parent:
|
|
752
|
+
return "'true_and_false_elimination': the stated clause has a constant literal the parent does not have"
|
|
753
|
+
if not _rc._is_variant(kept_by_parent, kept_by_step):
|
|
754
|
+
return ("'true_and_false_elimination': the stated clause is not the parent clause "
|
|
755
|
+
"without literals that are false in every interpretation ($false or ~$true): "
|
|
756
|
+
"another literal was added or dropped")
|
|
757
|
+
return None
|
|
758
|
+
|
|
759
|
+
|
|
760
|
+
_CHECKED_DISPATCH: Dict[str, Tuple[int, Callable]] = {
|
|
761
|
+
"resolution": (2, _check_tstp_resolve),
|
|
762
|
+
"factoring": (1, _check_tstp_factor),
|
|
763
|
+
"superposition": (2, _check_tstp_superposition),
|
|
764
|
+
"forward_demodulation": (2, _check_tstp_demodulation),
|
|
765
|
+
"backward_demodulation": (2, _check_tstp_demodulation),
|
|
766
|
+
"equality_resolution": (1, _check_tstp_equality_resolution),
|
|
767
|
+
"forward_subsumption_resolution": (2, _check_tstp_subsumption_resolution),
|
|
768
|
+
"backward_subsumption_resolution": (2, _check_tstp_subsumption_resolution),
|
|
769
|
+
"spm": (2, _check_tstp_superposition),
|
|
770
|
+
"rw": (2, _check_tstp_demodulation),
|
|
771
|
+
"er": (1, _check_tstp_equality_resolution),
|
|
772
|
+
"true_and_false_elimination": (1, _check_tstp_truth_constants),
|
|
773
|
+
}
|
|
774
|
+
assert set(_CHECKED_DISPATCH) == VAMPIRE_CHECKED_RULES | EPROVER_CHECKED_RULES
|
|
775
|
+
|
|
776
|
+
#: The core rules that read the ``$true`` / ``$false`` literals of a clause, and so
|
|
777
|
+
#: are handed clauses that keep them (every other rule gets the clause without).
|
|
778
|
+
_CONSTANT_READING_RULES: FrozenSet[str] = frozenset({"true_and_false_elimination"})
|
|
779
|
+
|
|
780
|
+
|
|
781
|
+
# ---------------------------------------------------------------------------
|
|
782
|
+
# Leaves and the entailment-checked clausification tier (tier 2)
|
|
783
|
+
# ---------------------------------------------------------------------------
|
|
784
|
+
|
|
785
|
+
def _check_leaf_formula(leaf: TstpStep, premises: Sequence[Node],
|
|
786
|
+
conclusion: Optional[Node], query: str) -> Tuple[bool, Optional[str]]:
|
|
787
|
+
"""Whether ``leaf`` (a :class:`~atp.tstp.TstpStep` with ``rule is None``)
|
|
788
|
+
is an alpha-variant of one of the caller's own originals -- ``conclusion``
|
|
789
|
+
when ``query == "conjecture"`` and ``leaf.role == "conjecture"``,
|
|
790
|
+
otherwise some member of ``premises``."""
|
|
791
|
+
if leaf.formula is None:
|
|
792
|
+
return False, "formula failed to parse"
|
|
793
|
+
if query == "conjecture" and leaf.role == "conjecture":
|
|
794
|
+
if conclusion is None:
|
|
795
|
+
return False, "role is 'conjecture' but no conclusion was supplied to match against"
|
|
796
|
+
if _formula_alpha_equal(leaf.formula, conclusion):
|
|
797
|
+
return True, None
|
|
798
|
+
return False, "is not an alpha-variant of the supplied conclusion"
|
|
799
|
+
for premise in premises:
|
|
800
|
+
if _formula_alpha_equal(leaf.formula, premise):
|
|
801
|
+
return True, None
|
|
802
|
+
return False, "is not an alpha-variant of any supplied premise"
|
|
803
|
+
|
|
804
|
+
|
|
805
|
+
_TRUTH = Atom("__tstp_truth", [])
|
|
806
|
+
|
|
807
|
+
|
|
808
|
+
def _for_z3(node: Node) -> Node:
|
|
809
|
+
"""``node`` with TSTP's ``$true``/``$false`` made logical, for Z3.
|
|
810
|
+
|
|
811
|
+
:func:`fol.tptp_input.parse_tptp_formula` reads them as ordinary 0-ary
|
|
812
|
+
atoms named ``$true``/``$false``, which Z3 would treat as free
|
|
813
|
+
propositions; they become ``T ∨ ¬T`` / ``T ∧ ¬T`` over one fixed atom,
|
|
814
|
+
exactly true and exactly false in every model."""
|
|
815
|
+
if isinstance(node, Atom):
|
|
816
|
+
if node.predicate == "$true" and not node.args:
|
|
817
|
+
return Or(_TRUTH, Not(_TRUTH))
|
|
818
|
+
if node.predicate == "$false" and not node.args:
|
|
819
|
+
return And(_TRUTH, Not(_TRUTH))
|
|
820
|
+
return node
|
|
821
|
+
if isinstance(node, Not):
|
|
822
|
+
return Not(_for_z3(node.formula))
|
|
823
|
+
if isinstance(node, (And, Or, Implies, Iff, Xor)):
|
|
824
|
+
return type(node)(_for_z3(node.left), _for_z3(node.right))
|
|
825
|
+
if isinstance(node, Quantifier):
|
|
826
|
+
return Quantifier(node.type, node.variable, _for_z3(node.formula))
|
|
827
|
+
return node
|
|
828
|
+
|
|
829
|
+
|
|
830
|
+
def _closed(node: Node) -> Node:
|
|
831
|
+
"""``node`` universally closed over its free variables -- the reading TSTP
|
|
832
|
+
gives a clause's variables -- with ``$true``/``$false`` made logical."""
|
|
833
|
+
out = _for_z3(node)
|
|
834
|
+
names = sorted({v.name for v in free_variables(node) if isinstance(v, Variable)})
|
|
835
|
+
for name in reversed(names):
|
|
836
|
+
out = Quantifier("∀", Variable(name), out)
|
|
837
|
+
return out
|
|
838
|
+
|
|
839
|
+
|
|
840
|
+
def _entailed(assumptions: Sequence[Node], formula: Node) -> bool:
|
|
841
|
+
"""Whether the (each separately closed) ``assumptions`` entail ``formula``.
|
|
842
|
+
|
|
843
|
+
Z3, PROVED-only: ``unknown`` or a timeout is False, never a pass."""
|
|
844
|
+
from .z3_models import is_valid # deferred: z3_models pulls in the whole fol layer
|
|
845
|
+
goal = _closed(formula)
|
|
846
|
+
if assumptions:
|
|
847
|
+
hypothesis = _closed(assumptions[0])
|
|
848
|
+
for extra in assumptions[1:]:
|
|
849
|
+
hypothesis = And(hypothesis, _closed(extra))
|
|
850
|
+
goal = Implies(hypothesis, goal)
|
|
851
|
+
return is_valid(goal, timeout=_ENTAILMENT_TIMEOUT_MS)
|
|
852
|
+
|
|
853
|
+
|
|
854
|
+
def _conjecture_misuse(rule: str, cited: Sequence[str]) -> str:
|
|
855
|
+
return (f"rule {rule!r} cites the conjecture {cited[0]!r} itself; a refutation may "
|
|
856
|
+
f"only use the conjecture through {sorted(_NEGATION_RULES)}")
|
|
857
|
+
|
|
858
|
+
|
|
859
|
+
def _check_clausification_step(step: TstpStep, results: Dict[str, "TstpStepResult"],
|
|
860
|
+
by_name: Dict[str, TstpStep],
|
|
861
|
+
conjecture_names: FrozenSet[str]) -> Tuple[bool, Optional[str]]:
|
|
862
|
+
"""Check a clausification/normalisation step SEMANTICALLY: its formula must
|
|
863
|
+
be entailed by its parents (for a negation rule: by the negated
|
|
864
|
+
conjecture). The specific transformation is not re-derived -- any sound
|
|
865
|
+
clausifier output passes -- but a step that states something its parents
|
|
866
|
+
do not license fails, which is all a refutation's soundness needs."""
|
|
867
|
+
if step.formula is None:
|
|
868
|
+
return False, "formula failed to parse"
|
|
869
|
+
if not step.parents:
|
|
870
|
+
return False, f"rule {step.rule!r} step has no resolvable parents"
|
|
871
|
+
parents: List[TstpStep] = []
|
|
872
|
+
for pname in step.parents:
|
|
873
|
+
pres = results.get(pname)
|
|
874
|
+
if pres is None or not pres.ok:
|
|
875
|
+
return False, f"parent {pname!r} is not an earlier, successfully verified statement"
|
|
876
|
+
parent = by_name[pname]
|
|
877
|
+
if parent.formula is None:
|
|
878
|
+
return False, f"parent {pname!r}'s formula failed to parse"
|
|
879
|
+
parents.append(parent)
|
|
880
|
+
cited = [p.name for p in parents if p.name in conjecture_names]
|
|
881
|
+
if step.rule in _NEGATION_RULES:
|
|
882
|
+
if len(cited) != len(parents):
|
|
883
|
+
others = [p.name for p in parents if p.name not in conjecture_names]
|
|
884
|
+
return False, (f"rule {step.rule!r} may only negate the conjecture, "
|
|
885
|
+
f"but also cites {others[0]!r}")
|
|
886
|
+
negated = Not(_closed(parents[0].formula))
|
|
887
|
+
if _entailed([negated], step.formula):
|
|
888
|
+
return True, None
|
|
889
|
+
return False, (f"rule {step.rule!r}: the stated formula is not entailed by the "
|
|
890
|
+
f"negated conjecture (Z3, PROVED-only)")
|
|
891
|
+
if cited:
|
|
892
|
+
return False, _conjecture_misuse(step.rule, cited)
|
|
893
|
+
if _entailed([p.formula for p in parents], step.formula):
|
|
894
|
+
return True, None
|
|
895
|
+
return False, (f"rule {step.rule!r}: the stated formula is not entailed by its parents "
|
|
896
|
+
f"(Z3, PROVED-only, {_ENTAILMENT_TIMEOUT_MS} ms)")
|
|
897
|
+
|
|
898
|
+
|
|
899
|
+
# ---------------------------------------------------------------------------
|
|
900
|
+
# The checker
|
|
901
|
+
# ---------------------------------------------------------------------------
|
|
902
|
+
|
|
903
|
+
@dataclass(frozen=True)
|
|
904
|
+
class TstpStepResult:
|
|
905
|
+
"""The outcome of checking one :class:`~atp.tstp.TstpStep`.
|
|
906
|
+
|
|
907
|
+
``tier`` is one of:
|
|
908
|
+
|
|
909
|
+
- ``"leaf"``: ``step.rule is None`` (a ``file(...)``-sourced original, or
|
|
910
|
+
any other non-``inference(...)`` source); ``ok`` iff its formula is an
|
|
911
|
+
alpha-variant of one of the caller's premises/conclusion.
|
|
912
|
+
- ``"entailed"``: ``step.rule`` is a clausification/normalisation rule
|
|
913
|
+
(:data:`VAMPIRE_CLAUSIFICATION_RULES` / :data:`EPROVER_CLAUSIFICATION_RULES`);
|
|
914
|
+
``ok`` iff every parent verified, none is the conjecture itself (unless
|
|
915
|
+
the rule is ``negated_conjecture``/``assume_negation``, which may cite
|
|
916
|
+
ONLY the conjecture), and Z3 proves the parents (resp. the negated
|
|
917
|
+
conjecture) entail the stated formula.
|
|
918
|
+
- ``"checked"``: ``step.rule`` is in the core independently-checked tier
|
|
919
|
+
(:data:`VAMPIRE_CHECKED_RULES` / :data:`EPROVER_CHECKED_RULES`); ``ok``
|
|
920
|
+
iff the re-derivation licenses the stated clause.
|
|
921
|
+
- ``"unchecked"``: ``step.rule`` is outside every table above, or is
|
|
922
|
+
``skolemisation`` (satisfiability-preserving only); ``ok`` is always
|
|
923
|
+
False, and ``detail`` names the rule.
|
|
924
|
+
"""
|
|
925
|
+
|
|
926
|
+
name: str
|
|
927
|
+
tier: str
|
|
928
|
+
ok: bool
|
|
929
|
+
detail: Optional[str] = None
|
|
930
|
+
|
|
931
|
+
def to_dict(self) -> dict:
|
|
932
|
+
"""Serialise to a JSON-compatible dict."""
|
|
933
|
+
return {"name": self.name, "tier": self.tier, "ok": self.ok, "detail": self.detail}
|
|
934
|
+
|
|
935
|
+
|
|
936
|
+
@dataclass(frozen=True)
|
|
937
|
+
class TstpCheckResult:
|
|
938
|
+
"""The outcome of checking a whole :class:`~atp.tstp.TstpDerivation`.
|
|
939
|
+
|
|
940
|
+
``verified`` is True iff EVERY step — leaf, entailed, and checked alike —
|
|
941
|
+
came back ``ok``; this is deliberately all-or-nothing (see the module
|
|
942
|
+
docstring's tier 3: one unchecked/unrecognised/failing rule anywhere
|
|
943
|
+
makes the whole thing unverified), but ``steps`` carries the full
|
|
944
|
+
per-step tiered breakdown so a caller can see exactly how far
|
|
945
|
+
verification reached. ``error`` names the FIRST failing step
|
|
946
|
+
(``"step <name> (<tier>): <reason>"``), or is ``None`` on full success.
|
|
947
|
+
``refuted`` is True iff some successfully-verified step's clause is
|
|
948
|
+
empty (the empty-clause / ``$false`` sink of a refutation), tracked
|
|
949
|
+
independently of ``verified`` exactly as
|
|
950
|
+
:attr:`atp.resolution_check.ResolutionCheckResult.refuted` is.
|
|
951
|
+
"""
|
|
952
|
+
|
|
953
|
+
verified: bool
|
|
954
|
+
steps: Tuple[TstpStepResult, ...]
|
|
955
|
+
error: Optional[str]
|
|
956
|
+
refuted: bool
|
|
957
|
+
|
|
958
|
+
def __bool__(self) -> bool:
|
|
959
|
+
"""A TstpCheckResult is truthy iff the whole derivation verified."""
|
|
960
|
+
return self.verified
|
|
961
|
+
|
|
962
|
+
def to_dict(self) -> dict:
|
|
963
|
+
"""Serialise to a JSON-compatible dict."""
|
|
964
|
+
return {
|
|
965
|
+
"verified": self.verified,
|
|
966
|
+
"steps": [s.to_dict() for s in self.steps],
|
|
967
|
+
"error": self.error,
|
|
968
|
+
"refuted": self.refuted,
|
|
969
|
+
}
|
|
970
|
+
|
|
971
|
+
|
|
972
|
+
def check_tstp_derivation(derivation: TstpDerivation, premises: Sequence[Node],
|
|
973
|
+
conclusion: Optional[Node] = None, *,
|
|
974
|
+
query: str = "conjecture") -> TstpCheckResult:
|
|
975
|
+
"""Independently check whether ``derivation`` genuinely holds.
|
|
976
|
+
|
|
977
|
+
Args:
|
|
978
|
+
derivation: a parsed :class:`~atp.tstp.TstpDerivation`
|
|
979
|
+
(:func:`atp.tstp.parse_tstp_derivation`'s output).
|
|
980
|
+
premises: the caller's own original premise formulas, in the SAME
|
|
981
|
+
representation basis as the derivation's leaf statements (i.e.
|
|
982
|
+
typically :func:`fol.tptp_input.parse_tptp_formula`'d from the
|
|
983
|
+
exact TPTP problem text handed to the prover — predicate names
|
|
984
|
+
already capitalised the way that parser does; a caller that
|
|
985
|
+
renamed symbols before generating the TPTP problem must
|
|
986
|
+
:func:`atp.tstp.reverse_map_derivation` first, or pass the
|
|
987
|
+
renamed originals here).
|
|
988
|
+
conclusion: the caller's original conclusion, required when
|
|
989
|
+
``query == "conjecture"`` and the derivation has a
|
|
990
|
+
``conjecture``-role leaf (every real Vampire/E fixture this
|
|
991
|
+
module was developed against does); ``None`` is fine for a
|
|
992
|
+
``query == "refutation"`` derivation, where ``premises`` is
|
|
993
|
+
understood to already hold the caller's whole folded
|
|
994
|
+
premises-plus-negated-conclusion set (see
|
|
995
|
+
:func:`atp.tstp.szs_to_verdict_fields`'s module docstring for
|
|
996
|
+
what the two framings mean).
|
|
997
|
+
query: ``"conjecture"`` or ``"refutation"`` — see ``conclusion``
|
|
998
|
+
above and :func:`atp.tstp.szs_to_verdict_fields`.
|
|
999
|
+
|
|
1000
|
+
Returns:
|
|
1001
|
+
A :class:`TstpCheckResult` with the full per-step tiered breakdown.
|
|
1002
|
+
|
|
1003
|
+
Raises:
|
|
1004
|
+
ValueError: ``query`` is neither ``"conjecture"`` nor ``"refutation"``.
|
|
1005
|
+
"""
|
|
1006
|
+
if query not in ("conjecture", "refutation"):
|
|
1007
|
+
raise ValueError(
|
|
1008
|
+
f"check_tstp_derivation: query must be 'conjecture' or 'refutation', got {query!r}")
|
|
1009
|
+
|
|
1010
|
+
by_name: Dict[str, TstpStep] = {step.name: step for step in derivation.steps}
|
|
1011
|
+
results: Dict[str, TstpStepResult] = {}
|
|
1012
|
+
clause_by_name: Dict[str, FrozenSet[Node]] = {}
|
|
1013
|
+
clause_with_constants_by_name: Dict[str, FrozenSet[Node]] = {}
|
|
1014
|
+
step_results: List[TstpStepResult] = []
|
|
1015
|
+
refuted = False
|
|
1016
|
+
first_error: Optional[str] = None
|
|
1017
|
+
# Leaves standing for the caller's conclusion: usable only negated.
|
|
1018
|
+
conjecture_names: FrozenSet[str] = frozenset(
|
|
1019
|
+
step.name for step in derivation.steps
|
|
1020
|
+
if query == "conjecture" and step.rule is None and step.role == "conjecture")
|
|
1021
|
+
|
|
1022
|
+
for step in derivation.steps:
|
|
1023
|
+
clause = _node_to_clause(step.formula) if step.formula is not None else None
|
|
1024
|
+
if clause is not None:
|
|
1025
|
+
clause_by_name[step.name] = clause
|
|
1026
|
+
clause_with_constants = (_node_to_clause(step.formula, keep_constants=True)
|
|
1027
|
+
if step.formula is not None else None)
|
|
1028
|
+
if clause_with_constants is not None:
|
|
1029
|
+
clause_with_constants_by_name[step.name] = clause_with_constants
|
|
1030
|
+
|
|
1031
|
+
if step.rule is None:
|
|
1032
|
+
tier = "leaf"
|
|
1033
|
+
ok, detail = _check_leaf_formula(step, premises, conclusion, query)
|
|
1034
|
+
elif step.rule in _SATISFIABILITY_ONLY_RULES:
|
|
1035
|
+
tier = "unchecked"
|
|
1036
|
+
ok, detail = False, (f"rule {step.rule!r} only preserves satisfiability, not "
|
|
1037
|
+
f"entailment; this checker does not certify it")
|
|
1038
|
+
elif step.rule in _CLAUSIFICATION_RULES:
|
|
1039
|
+
tier = "entailed"
|
|
1040
|
+
ok, detail = _check_clausification_step(step, results, by_name, conjecture_names)
|
|
1041
|
+
elif step.rule in _CHECKED_DISPATCH:
|
|
1042
|
+
tier = "checked"
|
|
1043
|
+
cited = [p for p in step.parents if p in conjecture_names]
|
|
1044
|
+
if cited:
|
|
1045
|
+
ok, detail = False, _conjecture_misuse(step.rule, cited)
|
|
1046
|
+
else:
|
|
1047
|
+
arity, checker = _CHECKED_DISPATCH[step.rule]
|
|
1048
|
+
if step.rule in _CONSTANT_READING_RULES:
|
|
1049
|
+
ok, detail = _check_checked_step(step, arity, checker, results,
|
|
1050
|
+
clause_with_constants_by_name,
|
|
1051
|
+
clause_with_constants)
|
|
1052
|
+
else:
|
|
1053
|
+
ok, detail = _check_checked_step(step, arity, checker, results,
|
|
1054
|
+
clause_by_name, clause)
|
|
1055
|
+
else:
|
|
1056
|
+
tier = "unchecked"
|
|
1057
|
+
ok, detail = False, f"rule {step.rule!r} is not in this checker's clausification or core-checked tables"
|
|
1058
|
+
|
|
1059
|
+
if ok and clause is not None and not clause:
|
|
1060
|
+
refuted = True
|
|
1061
|
+
|
|
1062
|
+
result = TstpStepResult(name=step.name, tier=tier, ok=ok, detail=detail)
|
|
1063
|
+
results[step.name] = result
|
|
1064
|
+
step_results.append(result)
|
|
1065
|
+
if not ok and first_error is None:
|
|
1066
|
+
first_error = f"step {step.name!r} ({tier}): {detail}"
|
|
1067
|
+
|
|
1068
|
+
verified = all(r.ok for r in step_results)
|
|
1069
|
+
return TstpCheckResult(verified=verified, steps=tuple(step_results),
|
|
1070
|
+
error=first_error, refuted=refuted)
|
|
1071
|
+
|
|
1072
|
+
|
|
1073
|
+
def _check_checked_step(step: TstpStep, arity: int, checker,
|
|
1074
|
+
results: Dict[str, TstpStepResult],
|
|
1075
|
+
clause_by_name: Dict[str, FrozenSet[Node]],
|
|
1076
|
+
clause: Optional[FrozenSet[Node]]) -> Tuple[bool, Optional[str]]:
|
|
1077
|
+
"""Shared plumbing for every core-checked rule: parent count/arity,
|
|
1078
|
+
"parents must be earlier and already verified", "every clause (own and
|
|
1079
|
+
parents') must be flat clausal form" -- then dispatches to ``checker``."""
|
|
1080
|
+
if len(step.parents) != arity:
|
|
1081
|
+
return False, f"rule {step.rule!r} takes {arity} parent(s), got {len(step.parents)}"
|
|
1082
|
+
parent_clauses = []
|
|
1083
|
+
for pname in step.parents:
|
|
1084
|
+
pres = results.get(pname)
|
|
1085
|
+
if pres is None or not pres.ok:
|
|
1086
|
+
return False, f"parent {pname!r} is not an earlier, successfully verified statement"
|
|
1087
|
+
pclause = clause_by_name.get(pname)
|
|
1088
|
+
if pclause is None:
|
|
1089
|
+
return False, f"parent {pname!r}'s formula is not in flat clausal (disjunction-of-literals) form"
|
|
1090
|
+
parent_clauses.append(pclause)
|
|
1091
|
+
if clause is None:
|
|
1092
|
+
return False, "this step's own formula is not in flat clausal (disjunction-of-literals) form"
|
|
1093
|
+
error = checker(clause, *parent_clauses)
|
|
1094
|
+
if error is not None:
|
|
1095
|
+
return False, error
|
|
1096
|
+
return True, None
|