unicode-logic-kit 0.31.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- unicode_logic_kit/__init__.py +385 -0
- unicode_logic_kit/__main__.py +520 -0
- unicode_logic_kit/_deadline.py +219 -0
- unicode_logic_kit/ace/__init__.py +126 -0
- unicode_logic_kit/ace/_align.py +135 -0
- unicode_logic_kit/ace/chem_lexicon.py +128 -0
- unicode_logic_kit/ace/drs_reader.py +570 -0
- unicode_logic_kit/ace/mapping.py +666 -0
- unicode_logic_kit/ace/reverse_modal.py +138 -0
- unicode_logic_kit/ace/runner.py +551 -0
- unicode_logic_kit/ace/translate.py +452 -0
- unicode_logic_kit/ace/verbalize.py +1070 -0
- unicode_logic_kit/api.py +1284 -0
- unicode_logic_kit/atp/__init__.py +177 -0
- unicode_logic_kit/atp/_ascii_names.py +113 -0
- unicode_logic_kit/atp/_html.py +72 -0
- unicode_logic_kit/atp/_substructural_input.py +228 -0
- unicode_logic_kit/atp/_tff_problem.py +715 -0
- unicode_logic_kit/atp/_tptp_problem.py +1111 -0
- unicode_logic_kit/atp/_writer_support.py +289 -0
- unicode_logic_kit/atp/clingo_backend.py +1180 -0
- unicode_logic_kit/atp/cvc5_backend.py +1385 -0
- unicode_logic_kit/atp/eprover_backend.py +732 -0
- unicode_logic_kit/atp/finite_domain.py +1055 -0
- unicode_logic_kit/atp/fitch.py +1547 -0
- unicode_logic_kit/atp/fitch_search.py +551 -0
- unicode_logic_kit/atp/hets_backend.py +339 -0
- unicode_logic_kit/atp/hybrid_down.py +120 -0
- unicode_logic_kit/atp/incremental.py +250 -0
- unicode_logic_kit/atp/kripke_enum.py +741 -0
- unicode_logic_kit/atp/lambek.py +436 -0
- unicode_logic_kit/atp/leo3_backend.py +332 -0
- unicode_logic_kit/atp/linear.py +738 -0
- unicode_logic_kit/atp/lj.py +705 -0
- unicode_logic_kit/atp/logic_backends.py +566 -0
- unicode_logic_kit/atp/ltl_tableau.py +1084 -0
- unicode_logic_kit/atp/minizinc_backend.py +1402 -0
- unicode_logic_kit/atp/modal_tableau.py +1382 -0
- unicode_logic_kit/atp/nanocop_backend.py +410 -0
- unicode_logic_kit/atp/portfolio.py +489 -0
- unicode_logic_kit/atp/protocol.py +1803 -0
- unicode_logic_kit/atp/prover9_entailment.py +1153 -0
- unicode_logic_kit/atp/resolution.py +1376 -0
- unicode_logic_kit/atp/resolution_check.py +1114 -0
- unicode_logic_kit/atp/sequent.py +1050 -0
- unicode_logic_kit/atp/tableau.py +921 -0
- unicode_logic_kit/atp/tableau_check.py +543 -0
- unicode_logic_kit/atp/tptp_ncl.py +811 -0
- unicode_logic_kit/atp/tptp_tff.py +1546 -0
- unicode_logic_kit/atp/tstp.py +1333 -0
- unicode_logic_kit/atp/tstp_check.py +1096 -0
- unicode_logic_kit/atp/twee_backend.py +236 -0
- unicode_logic_kit/atp/twee_check.py +711 -0
- unicode_logic_kit/atp/twee_entailment.py +953 -0
- unicode_logic_kit/atp/vampire_entailment.py +540 -0
- unicode_logic_kit/atp/z3_arith.py +470 -0
- unicode_logic_kit/atp/z3_equivalence.py +36 -0
- unicode_logic_kit/atp/z3_fuzzy.py +362 -0
- unicode_logic_kit/atp/z3_input.py +500 -0
- unicode_logic_kit/atp/z3_models.py +208 -0
- unicode_logic_kit/chem/__init__.py +88 -0
- unicode_logic_kit/chem/_naming.py +284 -0
- unicode_logic_kit/chem/cache.py +185 -0
- unicode_logic_kit/chem/interop.py +244 -0
- unicode_logic_kit/chem/mol.py +525 -0
- unicode_logic_kit/chem/signature.py +112 -0
- unicode_logic_kit/comorphism.py +497 -0
- unicode_logic_kit/dl/__init__.py +384 -0
- unicode_logic_kit/dl/classification.py +227 -0
- unicode_logic_kit/dl/concepts.py +632 -0
- unicode_logic_kit/dl/datatypes.py +818 -0
- unicode_logic_kit/dl/owl_functional.py +2433 -0
- unicode_logic_kit/dl/owl_manchester.py +1637 -0
- unicode_logic_kit/dl/owl_reasoner.py +790 -0
- unicode_logic_kit/dl/parser.py +391 -0
- unicode_logic_kit/dl/tableau.py +4048 -0
- unicode_logic_kit/dl/translate.py +2704 -0
- unicode_logic_kit/drt/__init__.py +94 -0
- unicode_logic_kit/drt/export.py +179 -0
- unicode_logic_kit/drt/nodes.py +506 -0
- unicode_logic_kit/drt/parser.py +965 -0
- unicode_logic_kit/drt/resolve.py +195 -0
- unicode_logic_kit/drt/reverse.py +175 -0
- unicode_logic_kit/eval/__init__.py +106 -0
- unicode_logic_kit/eval/batch.py +382 -0
- unicode_logic_kit/eval/canonical.py +663 -0
- unicode_logic_kit/eval/chem_batch.py +606 -0
- unicode_logic_kit/eval/converses.py +200 -0
- unicode_logic_kit/eval/datasets/__init__.py +136 -0
- unicode_logic_kit/eval/datasets/_base.py +263 -0
- unicode_logic_kit/eval/datasets/_proofwriter_proof.py +422 -0
- unicode_logic_kit/eval/datasets/c3po.py +678 -0
- unicode_logic_kit/eval/datasets/folio.py +158 -0
- unicode_logic_kit/eval/datasets/fracas.py +418 -0
- unicode_logic_kit/eval/datasets/groves.py +191 -0
- unicode_logic_kit/eval/datasets/logicbench.py +467 -0
- unicode_logic_kit/eval/datasets/logicnli.py +303 -0
- unicode_logic_kit/eval/datasets/malls.py +133 -0
- unicode_logic_kit/eval/datasets/pfolio.py +594 -0
- unicode_logic_kit/eval/datasets/pmb.py +242 -0
- unicode_logic_kit/eval/datasets/prontoqa.py +611 -0
- unicode_logic_kit/eval/datasets/proofwriter.py +1431 -0
- unicode_logic_kit/eval/datasets/proverqa.py +674 -0
- unicode_logic_kit/eval/datasets/willow.py +478 -0
- unicode_logic_kit/eval/equivalence.py +466 -0
- unicode_logic_kit/eval/exercise_gen.py +533 -0
- unicode_logic_kit/eval/explain.py +791 -0
- unicode_logic_kit/eval/generality.py +750 -0
- unicode_logic_kit/eval/metric_hf.py +458 -0
- unicode_logic_kit/eval/predicate_match.py +343 -0
- unicode_logic_kit/eval/theory_check.py +1170 -0
- unicode_logic_kit/eval/validate.py +306 -0
- unicode_logic_kit/fol/__init__.py +177 -0
- unicode_logic_kit/fol/_atom_keys.py +510 -0
- unicode_logic_kit/fol/_fol_nodes.py +3586 -0
- unicode_logic_kit/fol/_free_parameters.py +105 -0
- unicode_logic_kit/fol/_ho_nodes.py +448 -0
- unicode_logic_kit/fol/_hybrid_nodes.py +308 -0
- unicode_logic_kit/fol/_identifiers.py +1091 -0
- unicode_logic_kit/fol/_lambek_nodes.py +112 -0
- unicode_logic_kit/fol/_linear_nodes.py +352 -0
- unicode_logic_kit/fol/_modal_nodes.py +1467 -0
- unicode_logic_kit/fol/_msfl_nodes.py +2196 -0
- unicode_logic_kit/fol/_numeral_symbols.py +231 -0
- unicode_logic_kit/fol/_so_nodes.py +200 -0
- unicode_logic_kit/fol/_symbol_names.py +81 -0
- unicode_logic_kit/fol/_team_nodes.py +181 -0
- unicode_logic_kit/fol/_tptp_symbols.py +551 -0
- unicode_logic_kit/fol/_truth_constants.py +117 -0
- unicode_logic_kit/fol/casl_export.py +1135 -0
- unicode_logic_kit/fol/casl_import.py +929 -0
- unicode_logic_kit/fol/derivation.py +367 -0
- unicode_logic_kit/fol/dialect_detect.py +70 -0
- unicode_logic_kit/fol/dialect_repair.py +537 -0
- unicode_logic_kit/fol/frames.py +637 -0
- unicode_logic_kit/fol/grammars/terminals.lark +31 -0
- unicode_logic_kit/fol/lambda_tools.py +297 -0
- unicode_logic_kit/fol/latex_input.py +429 -0
- unicode_logic_kit/fol/modal_translation.py +944 -0
- unicode_logic_kit/fol/msflparser.py +1033 -0
- unicode_logic_kit/fol/naming.py +422 -0
- unicode_logic_kit/fol/nodes.py +241 -0
- unicode_logic_kit/fol/normalforms.py +492 -0
- unicode_logic_kit/fol/pal.py +287 -0
- unicode_logic_kit/fol/prolog_export.py +566 -0
- unicode_logic_kit/fol/prolog_input.py +505 -0
- unicode_logic_kit/fol/prover9_input.py +1325 -0
- unicode_logic_kit/fol/qml.py +1760 -0
- unicode_logic_kit/fol/qmltp_input.py +525 -0
- unicode_logic_kit/fol/sanitize.py +221 -0
- unicode_logic_kit/fol/serialize.py +79 -0
- unicode_logic_kit/fol/signature.py +1290 -0
- unicode_logic_kit/fol/simplify_check.py +544 -0
- unicode_logic_kit/fol/spans.py +594 -0
- unicode_logic_kit/fol/tptp_input.py +1503 -0
- unicode_logic_kit/fol/tptp_repair.py +941 -0
- unicode_logic_kit/fol/unification.py +157 -0
- unicode_logic_kit/fol/verbalize.py +263 -0
- unicode_logic_kit/hets/__init__.py +163 -0
- unicode_logic_kit/hets/bridge.py +142 -0
- unicode_logic_kit/hets/client.py +748 -0
- unicode_logic_kit/hets/docker.py +420 -0
- unicode_logic_kit/hets/dol.py +712 -0
- unicode_logic_kit/hets/haskell_json.py +355 -0
- unicode_logic_kit/hets/owl_backend.py +794 -0
- unicode_logic_kit/hets/owl_cli.py +598 -0
- unicode_logic_kit/hets/symbols.py +512 -0
- unicode_logic_kit/hol/__init__.py +140 -0
- unicode_logic_kit/hol/_ho_common.py +323 -0
- unicode_logic_kit/hol/_isabelle_binders.py +125 -0
- unicode_logic_kit/hol/classical.py +812 -0
- unicode_logic_kit/hol/deepshallow/__init__.py +45 -0
- unicode_logic_kit/hol/deepshallow/_common.py +177 -0
- unicode_logic_kit/hol/deepshallow/conditional.py +225 -0
- unicode_logic_kit/hol/deepshallow/intuitionistic.py +181 -0
- unicode_logic_kit/hol/deepshallow/modal.py +217 -0
- unicode_logic_kit/hol/deepshallow/qml.py +406 -0
- unicode_logic_kit/hol/deepshallow/relevant.py +206 -0
- unicode_logic_kit/hol/free.py +753 -0
- unicode_logic_kit/hol/goedel.py +336 -0
- unicode_logic_kit/hol/ho_modal.py +1743 -0
- unicode_logic_kit/hol/intuitionistic.py +403 -0
- unicode_logic_kit/hol/isabelle_conditional.py +593 -0
- unicode_logic_kit/hol/isabelle_modal.py +1908 -0
- unicode_logic_kit/hol/isabelle_relevant.py +412 -0
- unicode_logic_kit/hol/isabelle_runner.py +1147 -0
- unicode_logic_kit/hol/isabelle_substructural.py +884 -0
- unicode_logic_kit/hol/lean.py +1018 -0
- unicode_logic_kit/hol/manyvalued.py +921 -0
- unicode_logic_kit/hol/secondorder.py +687 -0
- unicode_logic_kit/hol/thf_modal.py +941 -0
- unicode_logic_kit/hol/thirdorder.py +397 -0
- unicode_logic_kit/ilp/__init__.py +89 -0
- unicode_logic_kit/ilp/readback.py +389 -0
- unicode_logic_kit/ilp/separation.py +153 -0
- unicode_logic_kit/ilp/task.py +730 -0
- unicode_logic_kit/logic.py +163 -0
- unicode_logic_kit/mcp/__init__.py +28 -0
- unicode_logic_kit/mcp/__main__.py +5 -0
- unicode_logic_kit/mcp/chem_tools.py +1031 -0
- unicode_logic_kit/mcp/server.py +2453 -0
- unicode_logic_kit/mcp/syntax_spec.py +681 -0
- unicode_logic_kit/prob/__init__.py +53 -0
- unicode_logic_kit/prob/_bdd.py +225 -0
- unicode_logic_kit/prob/_column_gen.py +668 -0
- unicode_logic_kit/prob/distribution.py +686 -0
- unicode_logic_kit/prob/nilsson.py +470 -0
- unicode_logic_kit/py.typed +0 -0
- unicode_logic_kit/semantics/__init__.py +137 -0
- unicode_logic_kit/semantics/_modal_reject.py +156 -0
- unicode_logic_kit/semantics/action_models.py +466 -0
- unicode_logic_kit/semantics/asp_models.py +1200 -0
- unicode_logic_kit/semantics/conditional.py +580 -0
- unicode_logic_kit/semantics/dynamic_epistemic.py +95 -0
- unicode_logic_kit/semantics/free_logic.py +913 -0
- unicode_logic_kit/semantics/fuzzy.py +384 -0
- unicode_logic_kit/semantics/fuzzy_kripke.py +442 -0
- unicode_logic_kit/semantics/intuitionistic.py +581 -0
- unicode_logic_kit/semantics/kripke.py +1139 -0
- unicode_logic_kit/semantics/manyvalued.py +580 -0
- unicode_logic_kit/semantics/matrix.py +342 -0
- unicode_logic_kit/semantics/model_eval.py +1135 -0
- unicode_logic_kit/semantics/modelfinder.py +1036 -0
- unicode_logic_kit/semantics/nonmonotonic.py +372 -0
- unicode_logic_kit/semantics/relevant.py +331 -0
- unicode_logic_kit/semantics/secondorder.py +657 -0
- unicode_logic_kit/semantics/structures.py +352 -0
- unicode_logic_kit/semantics/tarski.py +975 -0
- unicode_logic_kit/semantics/team.py +315 -0
- unicode_logic_kit/semantics/team_translation.py +416 -0
- unicode_logic_kit/semantics/thirdorder.py +358 -0
- unicode_logic_kit/semantics/tnorm.py +85 -0
- unicode_logic_kit/semantics/truthtable.py +201 -0
- unicode_logic_kit-0.31.0.dist-info/METADATA +333 -0
- unicode_logic_kit-0.31.0.dist-info/RECORD +237 -0
- unicode_logic_kit-0.31.0.dist-info/WHEEL +4 -0
- unicode_logic_kit-0.31.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,711 @@
|
|
|
1
|
+
"""Independent verification of a parsed Twee proof (not a searcher — a checker).
|
|
2
|
+
|
|
3
|
+
Where :mod:`atp.twee_entailment` drives Twee and PARSES its stdout into a
|
|
4
|
+
:class:`~atp.twee_entailment.TweeProof`, this module re-derives, from
|
|
5
|
+
scratch, whether that proof actually holds — the same division of labour as
|
|
6
|
+
:mod:`atp.resolution_check` for resolution refutations and
|
|
7
|
+
:mod:`atp.fitch`/:mod:`atp.sequent` for their calculi: a proof produced by an
|
|
8
|
+
external tool is only as trustworthy as an INDEPENDENT re-check of it, so
|
|
9
|
+
Twee itself is trusted only for search, never for soundness.
|
|
10
|
+
|
|
11
|
+
Two things are checked, independently of each other:
|
|
12
|
+
|
|
13
|
+
1. :func:`check_twee_proof` — every rewrite step in every lemma's and the
|
|
14
|
+
goal's proof chain is a genuine single-rewrite instance of the axiom or
|
|
15
|
+
lemma it cites (in the stated direction), lemmas are only cited after they
|
|
16
|
+
are themselves proved, and — the actual trust boundary, since Twee could
|
|
17
|
+
in principle print a fabricated "Axiom N (name): ..." line — every axiom
|
|
18
|
+
the proof restates is cross-checked, up to alpha-equivalence, against the
|
|
19
|
+
ORIGINAL premise (or premise conjunct) the caller actually gave Twee.
|
|
20
|
+
2. :func:`goal_matches_conclusion` — the proof's Goal equation is actually a
|
|
21
|
+
restatement of the conclusion the caller asked Twee to prove (guards
|
|
22
|
+
against a proof that is internally perfect but answers a different
|
|
23
|
+
question — see :mod:`atp.twee_entailment`'s module docstring for the
|
|
24
|
+
``tuple(...)`` encoding a conjunctive conclusion gets rewritten into,
|
|
25
|
+
which this function has to reverse to compare against the original).
|
|
26
|
+
The two readings it relies on hold only for names of Twee's own: a universally
|
|
27
|
+
quantified variable stands for a constant that occurs in no axiom of the proof and
|
|
28
|
+
nowhere in the conclusion, and the symbol that encodes a conjunction is one that
|
|
29
|
+
occurs in neither (:func:`goal_mismatch` says which name made it refuse).
|
|
30
|
+
|
|
31
|
+
Independence, concretely: rewrite-step verification is implemented here from
|
|
32
|
+
scratch (:func:`_match`, one-directional structural matching — NOT
|
|
33
|
+
:mod:`fol.unification`'s ``unify``, which is bidirectional and would let a
|
|
34
|
+
term-side variable bind, which is unsound here — a Twee proof term's own
|
|
35
|
+
variables, whatever their surface form, are always ground/opaque data to be
|
|
36
|
+
matched literally, never something to unify against; see the module
|
|
37
|
+
docstring of :mod:`atp.twee_entailment` for why axiom-side variables are
|
|
38
|
+
canonically renamed and goal-side variables are Skolem constants). Only
|
|
39
|
+
SUBSTITUTION application reuses :func:`unicode_logic_kit.fol.unification
|
|
40
|
+
.apply_subst` — a generic, direction-agnostic tree rewrite with no bearing on
|
|
41
|
+
matching soundness, explicitly fine to share.
|
|
42
|
+
"""
|
|
43
|
+
|
|
44
|
+
from dataclasses import dataclass
|
|
45
|
+
from typing import Dict, FrozenSet, List, Optional, Sequence, Set, Tuple
|
|
46
|
+
|
|
47
|
+
from ..fol.nodes import (
|
|
48
|
+
And, Atom, Constant, Function, Node, Number, Quantifier, SortedConstant, Variable,
|
|
49
|
+
)
|
|
50
|
+
from ..fol.unification import apply_subst
|
|
51
|
+
from .twee_entailment import TweeChain, TweeCitation, TweeProof
|
|
52
|
+
|
|
53
|
+
__all__ = ["TweeCheckResult", "check_twee_proof", "goal_matches_conclusion", "goal_mismatch"]
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
@dataclass(frozen=True)
|
|
57
|
+
class TweeCheckResult:
|
|
58
|
+
"""The outcome of :func:`check_twee_proof`.
|
|
59
|
+
|
|
60
|
+
``ok`` is True iff every axiom restatement and every rewrite step checks
|
|
61
|
+
out; ``error`` names the first failure (``None`` on success).
|
|
62
|
+
"""
|
|
63
|
+
|
|
64
|
+
ok: bool
|
|
65
|
+
error: Optional[str] = None
|
|
66
|
+
|
|
67
|
+
def __bool__(self) -> bool:
|
|
68
|
+
return self.ok
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
# ---------------------------------------------------------------------------
|
|
72
|
+
# Flattening a premise/conclusion into its top-level equational conjuncts —
|
|
73
|
+
# used BOTH to build the axiom-provenance name map (see module docstring of
|
|
74
|
+
# atp.twee_entailment for the premise_<i>/premise_<i>_<k> naming Twee's own
|
|
75
|
+
# clausifier produces) and to decompose a (possibly tupled) goal for
|
|
76
|
+
# goal_matches_conclusion.
|
|
77
|
+
# ---------------------------------------------------------------------------
|
|
78
|
+
|
|
79
|
+
def _erase_sorts(term: Node) -> Node:
|
|
80
|
+
"""``term`` with every sorted constant ``c:S`` read as the constant ``c``.
|
|
81
|
+
|
|
82
|
+
Twee decides unit equality and prints plain constants. A sorted constant is the
|
|
83
|
+
constant of its name (the kit's one-universe reading: ``c:S`` and ``c`` are one
|
|
84
|
+
symbol), and what its sort adds, that ``c`` is a member of ``S``, is a premise of
|
|
85
|
+
the problem, never part of an equation: a derivation of an equation from equations
|
|
86
|
+
is a derivation whatever else is assumed. So the equations are compared with the
|
|
87
|
+
sorts read off. A function of no arguments is the constant of its name (the problem
|
|
88
|
+
writer writes it so, and Twee prints it so), and is read as one."""
|
|
89
|
+
if isinstance(term, SortedConstant):
|
|
90
|
+
return Constant(term.name)
|
|
91
|
+
if isinstance(term, Function):
|
|
92
|
+
if not term.args:
|
|
93
|
+
return Constant(term.name)
|
|
94
|
+
return Function(term.name, tuple(_erase_sorts(a) for a in term.args))
|
|
95
|
+
return term
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
def _variable_names(term: Node, into: Set[str]) -> None:
|
|
99
|
+
"""Add the name of every ``Variable`` occurring in ``term`` to ``into``."""
|
|
100
|
+
if isinstance(term, Variable):
|
|
101
|
+
into.add(term.name)
|
|
102
|
+
elif isinstance(term, Function):
|
|
103
|
+
for arg in term.args:
|
|
104
|
+
_variable_names(arg, into)
|
|
105
|
+
|
|
106
|
+
|
|
107
|
+
def _symbol_names(term: Node, into: Set[str]) -> None:
|
|
108
|
+
"""Add the name of every constant and every function symbol occurring in ``term`` to
|
|
109
|
+
``into``: ONE namespace, as in a TPTP problem, where a functor ``x`` and a constant ``x``
|
|
110
|
+
are not told apart by Twee (it renames its own symbol when either is taken)."""
|
|
111
|
+
if isinstance(term, (Constant, SortedConstant)):
|
|
112
|
+
into.add(term.name)
|
|
113
|
+
elif isinstance(term, Function):
|
|
114
|
+
into.add(term.name)
|
|
115
|
+
for arg in term.args:
|
|
116
|
+
_symbol_names(arg, into)
|
|
117
|
+
|
|
118
|
+
|
|
119
|
+
def _flatten_scoped(node: Node, bound: FrozenSet[str] = frozenset()
|
|
120
|
+
) -> List[Tuple[Node, Node, FrozenSet[str]]]:
|
|
121
|
+
"""Flatten like :func:`_flatten_equations`, and say which variables each equation
|
|
122
|
+
leaves FREE: ``[(lhs, rhs, free), ...]`` where ``free`` holds the name of every
|
|
123
|
+
variable of the equation that no quantifier above it binds."""
|
|
124
|
+
if isinstance(node, Quantifier):
|
|
125
|
+
if node.type not in ("forall", "∀"):
|
|
126
|
+
raise ValueError(f"twee_check: non-universal quantifier {node.type!r} in premise/conclusion")
|
|
127
|
+
return _flatten_scoped(node.formula, bound | {node.variable.name})
|
|
128
|
+
if isinstance(node, And):
|
|
129
|
+
return _flatten_scoped(node.left, bound) + _flatten_scoped(node.right, bound)
|
|
130
|
+
if isinstance(node, Atom) and node.predicate == "=" and len(node.args) == 2:
|
|
131
|
+
lhs, rhs = _erase_sorts(node.args[0]), _erase_sorts(node.args[1])
|
|
132
|
+
occurring: Set[str] = set()
|
|
133
|
+
_variable_names(lhs, occurring)
|
|
134
|
+
_variable_names(rhs, occurring)
|
|
135
|
+
return [(lhs, rhs, frozenset(occurring - bound))]
|
|
136
|
+
raise ValueError(f"twee_check: not an equation or conjunction of equations: "
|
|
137
|
+
f"{node.to_unicode_str()!r}")
|
|
138
|
+
|
|
139
|
+
|
|
140
|
+
def _flatten_equations(node: Node) -> List[Tuple[Node, Node]]:
|
|
141
|
+
"""Flatten a (forall-closed) equation-or-conjunction into ``[(lhs, rhs), ...]``.
|
|
142
|
+
|
|
143
|
+
Left-to-right in source order (``And`` is binary and this recurses
|
|
144
|
+
left-then-right). That is the order of the SOURCE and nothing relies on it
|
|
145
|
+
for naming: Twee numbers the clauses of a conjunctive premise in an order of
|
|
146
|
+
its own (see :func:`_expected_axiom_equations`). Quantifiers are stripped
|
|
147
|
+
(matching is variable-name-blind — see module docstring), so the returned
|
|
148
|
+
pairs are plain term ``Node`` objects, still possibly containing the original bound
|
|
149
|
+
variables. A sorted constant is read as the plain constant of its name
|
|
150
|
+
(:func:`_erase_sorts`).
|
|
151
|
+
"""
|
|
152
|
+
return [(lhs, rhs) for lhs, rhs, _ in _flatten_scoped(node)]
|
|
153
|
+
|
|
154
|
+
|
|
155
|
+
def _expected_axiom_equations(axioms: Sequence[Node]) -> Dict[str, List[Tuple[Node, Node]]]:
|
|
156
|
+
"""Map every axiom name Twee can give a clause of the premises to the equations it may be.
|
|
157
|
+
|
|
158
|
+
Twee names the clauses of premise ``i`` (1-based) ``premise_<i>``, ``premise_<i>_1``, ...
|
|
159
|
+
``premise_<i>_<n - 1>`` for a premise of ``n`` conjuncts, and which clause carries which
|
|
160
|
+
suffix is Twee's own business, not the order of the source: its clausifier numbers the
|
|
161
|
+
clauses in an order of its own and drops a ground conjunct that repeats an earlier one
|
|
162
|
+
(measured, Twee 2.6.1: for ``aa = bb ∧ ∀x ff(x) = x`` the clause ``ff(X) = X`` is
|
|
163
|
+
``premise_1`` and ``aa = bb`` is ``premise_1_1``; for ``∀x ∀y (hh(x, y) = hh(y, x) ∧
|
|
164
|
+
ff(x) = x)`` the clause ``ff(X) = X`` is ``premise_1``; for ``aa = bb ∧ aa = bb ∧ cc = dd``
|
|
165
|
+
the clause ``cc = dd`` is ``premise_1_1``). So a name determines the
|
|
166
|
+
PREMISE and nothing else: the value of ``premise_<i>`` and of every ``premise_<i>_<k>`` is
|
|
167
|
+
the list of all the conjuncts of premise ``i``, and a restated axiom is checked against
|
|
168
|
+
what it SAYS (:func:`check_twee_proof`: it must be one of them, up to the renaming of its
|
|
169
|
+
variables), never against its position. A name of another premise, or a suffix beyond
|
|
170
|
+
the last conjunct, is not a key.
|
|
171
|
+
|
|
172
|
+
A premise whose equation has a free variable is refused (``ValueError``): a free
|
|
173
|
+
variable is one unknown element of the problem, and the axiom Twee restates, with
|
|
174
|
+
its variables universal, would be a stronger statement than the premise.
|
|
175
|
+
"""
|
|
176
|
+
mapping: Dict[str, List[Tuple[Node, Node]]] = {}
|
|
177
|
+
for i, premise in enumerate(axioms, start=1):
|
|
178
|
+
equations: List[Tuple[Node, Node]] = []
|
|
179
|
+
for lhs, rhs, free in _flatten_scoped(premise):
|
|
180
|
+
if free:
|
|
181
|
+
raise ValueError(
|
|
182
|
+
f"twee_check: premise {i} has the free variable(s) {sorted(free)}: a free "
|
|
183
|
+
f"variable is one unknown element, not a universally quantified one, so the "
|
|
184
|
+
f"equation is not one a proof's axiom with universal variables restates")
|
|
185
|
+
equations.append((lhs, rhs))
|
|
186
|
+
for k in range(len(equations)):
|
|
187
|
+
mapping[f"premise_{i}" if k == 0 else f"premise_{i}_{k}"] = equations
|
|
188
|
+
return mapping
|
|
189
|
+
|
|
190
|
+
|
|
191
|
+
# ---------------------------------------------------------------------------
|
|
192
|
+
# Alpha-equivalence (variable-renaming-blind) equation matching — used ONLY
|
|
193
|
+
# for the axiom-provenance cross-check (an axiom's restated equation must be
|
|
194
|
+
# the SAME equation as the original premise, up to a variable bijection,
|
|
195
|
+
# since Twee renames axiom variables to a canonical X/Y/Z scheme — see
|
|
196
|
+
# atp.twee_entailment's module docstring).
|
|
197
|
+
# ---------------------------------------------------------------------------
|
|
198
|
+
|
|
199
|
+
def _alpha_extend(t1: Node, t2: Node, fwd: Dict[str, str], bwd: Dict[str, str]):
|
|
200
|
+
"""Extend the variable bijection ``(fwd, bwd)`` matching ``t1`` onto ``t2``.
|
|
201
|
+
|
|
202
|
+
``fwd``/``bwd`` map ``t1``-side variable names to ``t2``-side names and
|
|
203
|
+
back; returns the extended pair, or ``None`` on a genuine mismatch.
|
|
204
|
+
Neither input dict is mutated.
|
|
205
|
+
"""
|
|
206
|
+
if isinstance(t1, Variable) and isinstance(t2, Variable):
|
|
207
|
+
mapped = fwd.get(t1.name)
|
|
208
|
+
if mapped is not None:
|
|
209
|
+
return (fwd, bwd) if mapped == t2.name else None
|
|
210
|
+
if t2.name in bwd:
|
|
211
|
+
return None
|
|
212
|
+
new_fwd, new_bwd = dict(fwd), dict(bwd)
|
|
213
|
+
new_fwd[t1.name] = t2.name
|
|
214
|
+
new_bwd[t2.name] = t1.name
|
|
215
|
+
return new_fwd, new_bwd
|
|
216
|
+
if isinstance(t1, Variable) or isinstance(t2, Variable):
|
|
217
|
+
return None
|
|
218
|
+
if isinstance(t1, Constant) and isinstance(t2, Constant):
|
|
219
|
+
return (fwd, bwd) if t1.name == t2.name else None
|
|
220
|
+
if isinstance(t1, Number) and isinstance(t2, Number):
|
|
221
|
+
return (fwd, bwd) if t1.value == t2.value else None
|
|
222
|
+
if isinstance(t1, Function) and isinstance(t2, Function):
|
|
223
|
+
if t1.name != t2.name or len(t1.args) != len(t2.args):
|
|
224
|
+
return None
|
|
225
|
+
for a1, a2 in zip(t1.args, t2.args):
|
|
226
|
+
res = _alpha_extend(a1, a2, fwd, bwd)
|
|
227
|
+
if res is None:
|
|
228
|
+
return None
|
|
229
|
+
fwd, bwd = res
|
|
230
|
+
return fwd, bwd
|
|
231
|
+
return None
|
|
232
|
+
|
|
233
|
+
|
|
234
|
+
def _equation_is_variant(expected: Tuple[Node, Node], actual: Tuple[Node, Node]) -> bool:
|
|
235
|
+
"""True iff ``actual`` is ``expected`` up to a single consistent variable renaming.
|
|
236
|
+
|
|
237
|
+
Orientation-sensitive (``expected[0]`` must correspond to ``actual[0]``,
|
|
238
|
+
not either side) — Twee's ``Axiom N (name): lhs = rhs.`` header was
|
|
239
|
+
observed to always echo the premise's original left-to-right orientation
|
|
240
|
+
(never auto-flipped), so this deliberately does not also try the swapped
|
|
241
|
+
pairing.
|
|
242
|
+
"""
|
|
243
|
+
res = _alpha_extend(expected[0], actual[0], {}, {})
|
|
244
|
+
if res is None:
|
|
245
|
+
return False
|
|
246
|
+
res = _alpha_extend(expected[1], actual[1], res[0], res[1])
|
|
247
|
+
return res is not None
|
|
248
|
+
|
|
249
|
+
|
|
250
|
+
# ---------------------------------------------------------------------------
|
|
251
|
+
# One-directional structural matching + position/replacement — the rewrite-
|
|
252
|
+
# step verifier's core. See module docstring: deliberately NOT
|
|
253
|
+
# fol.unification.unify (bidirectional; a term-side variable must never bind).
|
|
254
|
+
# ---------------------------------------------------------------------------
|
|
255
|
+
|
|
256
|
+
def _match(pattern: Node, term: Node, subst: Dict[str, Node]) -> Optional[Dict[str, Node]]:
|
|
257
|
+
"""Match ``pattern`` (may contain bindable Variables) against ``term`` (ground/opaque).
|
|
258
|
+
|
|
259
|
+
Only ``pattern``'s own ``Variable`` nodes are ever bound; a ``Variable``
|
|
260
|
+
appearing on the ``term`` side (e.g. a lemma's own rewrite variable
|
|
261
|
+
showing up inside another citation's redex) is compared like any other
|
|
262
|
+
leaf — by structural equality against whatever ``term`` already is,
|
|
263
|
+
never unified against. Returns the extended substitution, or ``None``.
|
|
264
|
+
``subst`` is not mutated; already-bound pattern variables must match
|
|
265
|
+
``term`` exactly (via kit ``Node`` structural ``==``) to succeed again.
|
|
266
|
+
"""
|
|
267
|
+
if isinstance(pattern, Variable):
|
|
268
|
+
if pattern.name in subst:
|
|
269
|
+
return subst if subst[pattern.name] == term else None
|
|
270
|
+
new_subst = dict(subst)
|
|
271
|
+
new_subst[pattern.name] = term
|
|
272
|
+
return new_subst
|
|
273
|
+
if isinstance(pattern, Constant):
|
|
274
|
+
return subst if (isinstance(term, Constant) and term.name == pattern.name) else None
|
|
275
|
+
if isinstance(pattern, Number):
|
|
276
|
+
return subst if (isinstance(term, Number) and term.value == pattern.value) else None
|
|
277
|
+
if isinstance(pattern, Function):
|
|
278
|
+
if not isinstance(term, Function) or term.name != pattern.name \
|
|
279
|
+
or len(term.args) != len(pattern.args):
|
|
280
|
+
return None
|
|
281
|
+
extended = subst
|
|
282
|
+
for pa, ta in zip(pattern.args, term.args):
|
|
283
|
+
step = _match(pa, ta, extended)
|
|
284
|
+
if step is None:
|
|
285
|
+
return None
|
|
286
|
+
extended = step
|
|
287
|
+
return extended
|
|
288
|
+
return None # a pattern shape this checker does not expect (e.g. an Atom)
|
|
289
|
+
|
|
290
|
+
|
|
291
|
+
def _positions(node: Node) -> List[Tuple[int, ...]]:
|
|
292
|
+
"""All subterm positions of ``node`` (argument-index paths), including ``()`` (the root)."""
|
|
293
|
+
positions: List[Tuple[int, ...]] = [()]
|
|
294
|
+
if isinstance(node, Function):
|
|
295
|
+
for i, arg in enumerate(node.args):
|
|
296
|
+
positions.extend((i,) + p for p in _positions(arg))
|
|
297
|
+
return positions
|
|
298
|
+
|
|
299
|
+
|
|
300
|
+
def _get_at(node: Node, path: Tuple[int, ...]) -> Optional[Node]:
|
|
301
|
+
"""The subterm of ``node`` at ``path``, or ``None`` if ``path`` does not exist in it."""
|
|
302
|
+
for i in path:
|
|
303
|
+
if not isinstance(node, Function) or i >= len(node.args):
|
|
304
|
+
return None
|
|
305
|
+
node = node.args[i]
|
|
306
|
+
return node
|
|
307
|
+
|
|
308
|
+
|
|
309
|
+
def _replace_at(node: Node, path: Tuple[int, ...], replacement: Node) -> Node:
|
|
310
|
+
"""Return ``node`` with the subterm at ``path`` replaced by ``replacement``."""
|
|
311
|
+
if not path:
|
|
312
|
+
return replacement
|
|
313
|
+
if not isinstance(node, Function):
|
|
314
|
+
raise ValueError("twee_check: _replace_at called with a path into a non-Function term")
|
|
315
|
+
i, rest = path[0], path[1:]
|
|
316
|
+
new_args = list(node.args)
|
|
317
|
+
new_args[i] = _replace_at(node.args[i], rest, replacement)
|
|
318
|
+
return Function(node.name, tuple(new_args))
|
|
319
|
+
|
|
320
|
+
|
|
321
|
+
def _freshen(node: Node, prefix: str) -> Node:
|
|
322
|
+
"""Rename every ``Variable`` in ``node`` by prepending ``prefix``.
|
|
323
|
+
|
|
324
|
+
Twee names axiom/lemma rewrite variables from a canonical ``X``/``Y``/
|
|
325
|
+
``Z``/... pool that STARTS OVER independently for every equation (see
|
|
326
|
+
:mod:`atp.twee_entailment`'s module docstring) — so a citation's "X" and
|
|
327
|
+
the current proof term's own "X" (e.g. inside a LEMMA's chain, which
|
|
328
|
+
keeps its own rewrite variables live throughout — verified live: this
|
|
329
|
+
genuinely happens, e.g. Lemma 4 using X, Y while citing an axiom that
|
|
330
|
+
ALSO prints as X, Y) are two unrelated variables that merely share a
|
|
331
|
+
spelling. Without freshening, :func:`_match` would bind the citation's
|
|
332
|
+
"X" to a term-side ``Variable("X")`` and :func:`~unicode_logic_kit.fol
|
|
333
|
+
.unification.apply_subst` would loop forever resolving ``X -> X``
|
|
334
|
+
(found live while building this checker). ``prefix`` uses ``#``, a
|
|
335
|
+
character :func:`atp.twee_entailment._tokenize` never produces, so a
|
|
336
|
+
freshened name can never collide with a genuine Twee-parsed one.
|
|
337
|
+
"""
|
|
338
|
+
if isinstance(node, Variable):
|
|
339
|
+
return Variable(prefix + node.name)
|
|
340
|
+
if isinstance(node, Function):
|
|
341
|
+
return Function(node.name, tuple(_freshen(a, prefix) for a in node.args))
|
|
342
|
+
return node
|
|
343
|
+
|
|
344
|
+
|
|
345
|
+
def _verify_rewrite(term1: Node, term2: Node, from_pattern: Node, to_pattern: Node) -> bool:
|
|
346
|
+
"""True iff ``term2`` is ``term1`` with ONE subterm rewritten via ``from_pattern = to_pattern``.
|
|
347
|
+
|
|
348
|
+
Tries every subterm position ``p`` of ``term1``: if the subterm there
|
|
349
|
+
matches ``from_pattern`` (binding ``from_pattern``'s variables), and the
|
|
350
|
+
corresponding subterm of ``term2`` at the SAME position matches
|
|
351
|
+
``to_pattern`` under that (extended) substitution — needed because a
|
|
352
|
+
variable can appear only on the ``to`` side, e.g. an ``R->L`` step whose
|
|
353
|
+
cited equation's "from" side (the original right-hand side) is a bare
|
|
354
|
+
variable that does not mention every variable the "to" side does — then
|
|
355
|
+
rebuilding ``term1`` with that position replaced by the fully-substituted
|
|
356
|
+
``to_pattern`` must reproduce ``term2`` exactly. Succeeds on the first
|
|
357
|
+
matching position found.
|
|
358
|
+
|
|
359
|
+
``from_pattern``/``to_pattern`` are freshened (see :func:`_freshen`)
|
|
360
|
+
before anything else — they belong to a DIFFERENT equation's variable
|
|
361
|
+
scope than ``term1``/``term2``, and must never be confused with it.
|
|
362
|
+
"""
|
|
363
|
+
from_pattern = _freshen(from_pattern, "#")
|
|
364
|
+
to_pattern = _freshen(to_pattern, "#")
|
|
365
|
+
for path in _positions(term1):
|
|
366
|
+
sub1 = _get_at(term1, path)
|
|
367
|
+
if sub1 is None: # unreachable: every path of _positions(term1) exists in term1
|
|
368
|
+
continue
|
|
369
|
+
subst = _match(from_pattern, sub1, {})
|
|
370
|
+
if subst is None:
|
|
371
|
+
continue
|
|
372
|
+
sub2 = _get_at(term2, path)
|
|
373
|
+
if sub2 is None:
|
|
374
|
+
continue
|
|
375
|
+
subst2 = _match(to_pattern, sub2, subst)
|
|
376
|
+
if subst2 is None:
|
|
377
|
+
continue
|
|
378
|
+
candidate = _replace_at(term1, path, apply_subst(to_pattern, subst2))
|
|
379
|
+
if candidate == term2:
|
|
380
|
+
return True
|
|
381
|
+
return False
|
|
382
|
+
|
|
383
|
+
|
|
384
|
+
# ---------------------------------------------------------------------------
|
|
385
|
+
# check_twee_proof
|
|
386
|
+
# ---------------------------------------------------------------------------
|
|
387
|
+
|
|
388
|
+
def check_twee_proof(proof: TweeProof, axioms: Sequence[Node]) -> TweeCheckResult:
|
|
389
|
+
"""Independently verify ``proof`` against the original ``axioms`` (premises).
|
|
390
|
+
|
|
391
|
+
Two passes:
|
|
392
|
+
|
|
393
|
+
1. Every ``Axiom N (name): ...`` header in ``proof`` must name a premise
|
|
394
|
+
actually present in ``axioms`` (``premise_<i>``, or ``premise_<i>_<k>``
|
|
395
|
+
with ``k`` below the number of its conjuncts — see
|
|
396
|
+
:func:`_expected_axiom_equations`), and its restated equation must be
|
|
397
|
+
alpha-equivalent (see :func:`_equation_is_variant`) to that premise's
|
|
398
|
+
equation or, for a conjunctive premise, to one of its conjuncts: which
|
|
399
|
+
one is decided by what the axiom says, never by the suffix of its name,
|
|
400
|
+
because Twee numbers the clauses of a conjunction in an order of its
|
|
401
|
+
own. This is the boundary that makes the check genuinely independent
|
|
402
|
+
of trusting Twee's own transcription: every axiom a proof uses is a
|
|
403
|
+
consequence of a premise.
|
|
404
|
+
2. Every lemma's proof chain, then the goal's, is walked step by step:
|
|
405
|
+
each step must be licensed by :func:`_verify_rewrite` against the
|
|
406
|
+
cited axiom (already cross-checked in pass 1) or an EARLIER lemma in
|
|
407
|
+
the SAME proof (forward/self-citation is rejected); each chain's
|
|
408
|
+
first and last term must equal its own header's stated ``lhs``/``rhs``.
|
|
409
|
+
|
|
410
|
+
Returns the first failure found (``TweeCheckResult(False, "...")``), or
|
|
411
|
+
``TweeCheckResult(True)`` if everything checks out. Never raises for a
|
|
412
|
+
well-formed :class:`TweeProof` and any ``axioms`` sequence of ``Node``s —
|
|
413
|
+
a malformed ``axioms`` entry (not itself equational) is reported as a
|
|
414
|
+
failure, not an exception.
|
|
415
|
+
"""
|
|
416
|
+
try:
|
|
417
|
+
expected = _expected_axiom_equations(axioms)
|
|
418
|
+
except ValueError as exc:
|
|
419
|
+
return TweeCheckResult(False, f"invalid axioms argument: {exc}")
|
|
420
|
+
|
|
421
|
+
axiom_by_number: Dict[int, Tuple[str, Tuple[Node, Node]]] = {}
|
|
422
|
+
for ax in proof.axioms:
|
|
423
|
+
if ax.number in axiom_by_number:
|
|
424
|
+
return TweeCheckResult(False, f"axiom {ax.number} is restated more than once")
|
|
425
|
+
exp = expected.get(ax.name)
|
|
426
|
+
if exp is None:
|
|
427
|
+
return TweeCheckResult(
|
|
428
|
+
False, f"axiom {ax.number} ({ax.name}): no given premise (or flattened "
|
|
429
|
+
f"conjunct of one) is named {ax.name!r}")
|
|
430
|
+
actual = (ax.equation.lhs, ax.equation.rhs)
|
|
431
|
+
if not any(_equation_is_variant(candidate, actual) for candidate in exp):
|
|
432
|
+
return TweeCheckResult(
|
|
433
|
+
False, f"axiom {ax.number} ({ax.name}): restated equation is not "
|
|
434
|
+
f"alpha-equivalent to the given premise or to any conjunct of it")
|
|
435
|
+
axiom_by_number[ax.number] = (ax.name, actual)
|
|
436
|
+
|
|
437
|
+
lemma_by_number: Dict[int, Tuple[Node, Node]] = {}
|
|
438
|
+
|
|
439
|
+
def _resolve(citation: TweeCitation, enforce_before: Optional[int]):
|
|
440
|
+
"""Return ``(equation, None)`` or ``(None, error)`` for a citation."""
|
|
441
|
+
if citation.kind == "axiom":
|
|
442
|
+
entry = axiom_by_number.get(citation.number)
|
|
443
|
+
if entry is None:
|
|
444
|
+
return None, f"cites axiom {citation.number}, which was never stated in the proof header"
|
|
445
|
+
stored_name, eq = entry
|
|
446
|
+
if citation.name != stored_name:
|
|
447
|
+
return None, (f"cites axiom {citation.number} as ({citation.name}), but that "
|
|
448
|
+
f"axiom number was stated as ({stored_name})")
|
|
449
|
+
return eq, None
|
|
450
|
+
lemma_eq = lemma_by_number.get(citation.number)
|
|
451
|
+
if lemma_eq is None:
|
|
452
|
+
return None, f"cites lemma {citation.number}, which is not an earlier proven lemma"
|
|
453
|
+
if enforce_before is not None and citation.number >= enforce_before:
|
|
454
|
+
return None, f"cites lemma {citation.number}, which is not strictly earlier"
|
|
455
|
+
return lemma_eq, None
|
|
456
|
+
|
|
457
|
+
def _check_chain(chain: TweeChain, enforce_before: Optional[int], label: str) -> Optional[str]:
|
|
458
|
+
for i, citation in enumerate(chain.citations):
|
|
459
|
+
eq, err = _resolve(citation, enforce_before)
|
|
460
|
+
if err is not None:
|
|
461
|
+
return f"{label} step {i + 1}: {err}"
|
|
462
|
+
from_pattern, to_pattern = (eq[1], eq[0]) if citation.reversed else (eq[0], eq[1])
|
|
463
|
+
if not _verify_rewrite(chain.terms[i], chain.terms[i + 1], from_pattern, to_pattern):
|
|
464
|
+
cite_desc = (f"axiom {citation.number} ({citation.name})" if citation.kind == "axiom"
|
|
465
|
+
else f"lemma {citation.number}")
|
|
466
|
+
direction = " R->L" if citation.reversed else ""
|
|
467
|
+
return (f"{label} step {i + 1}: not a valid single-rewrite instance of "
|
|
468
|
+
f"{cite_desc}{direction}")
|
|
469
|
+
return None
|
|
470
|
+
|
|
471
|
+
for lemma in proof.lemmas:
|
|
472
|
+
if lemma.chain.terms[0] != lemma.equation.lhs:
|
|
473
|
+
return TweeCheckResult(
|
|
474
|
+
False, f"lemma {lemma.number}: proof chain does not start at its stated left-hand side")
|
|
475
|
+
if lemma.chain.terms[-1] != lemma.equation.rhs:
|
|
476
|
+
return TweeCheckResult(
|
|
477
|
+
False, f"lemma {lemma.number}: proof chain does not end at its stated right-hand side")
|
|
478
|
+
err = _check_chain(lemma.chain, lemma.number, f"lemma {lemma.number}")
|
|
479
|
+
if err is not None:
|
|
480
|
+
return TweeCheckResult(False, err)
|
|
481
|
+
lemma_by_number[lemma.number] = (lemma.equation.lhs, lemma.equation.rhs)
|
|
482
|
+
|
|
483
|
+
goal = proof.goal
|
|
484
|
+
if goal.chain.terms[0] != goal.equation.lhs:
|
|
485
|
+
return TweeCheckResult(False, "goal: proof chain does not start at its stated left-hand side")
|
|
486
|
+
if goal.chain.terms[-1] != goal.equation.rhs:
|
|
487
|
+
return TweeCheckResult(False, "goal: proof chain does not end at its stated right-hand side")
|
|
488
|
+
err = _check_chain(goal.chain, None, "goal") # goal may cite ANY lemma (it is always last)
|
|
489
|
+
if err is not None:
|
|
490
|
+
return TweeCheckResult(False, err)
|
|
491
|
+
|
|
492
|
+
return TweeCheckResult(True)
|
|
493
|
+
|
|
494
|
+
|
|
495
|
+
# ---------------------------------------------------------------------------
|
|
496
|
+
# goal_matches_conclusion
|
|
497
|
+
# ---------------------------------------------------------------------------
|
|
498
|
+
|
|
499
|
+
def _matches_conjunct(exp_lhs: Node, exp_rhs: Node, act_lhs: Node, act_rhs: Node,
|
|
500
|
+
taken: FrozenSet[str] = frozenset(),
|
|
501
|
+
notes: Optional[List[str]] = None) -> bool:
|
|
502
|
+
"""One conjunct's match: ``(exp_lhs, exp_rhs)`` structurally matches ``(act_lhs, act_rhs)``.
|
|
503
|
+
|
|
504
|
+
Uses the same one-directional :func:`_match` as rewrite-step checking:
|
|
505
|
+
the conclusion's own (bound) variables bind to whatever term Twee's Goal
|
|
506
|
+
line actually shows there, threading ONE substitution across both sides
|
|
507
|
+
of the equation. Two conditions make that binding the Skolemisation of
|
|
508
|
+
the universal quantifiers and nothing weaker:
|
|
509
|
+
|
|
510
|
+
* **Every variable is bound to a fresh constant.** ``Γ ⊨ ∀x φ(x)`` follows
|
|
511
|
+
from ``Γ ⊨ φ(c)`` only when ``c`` is a constant of its own: one that
|
|
512
|
+
occurs in no axiom the proof used and nowhere in the conclusion (nor is
|
|
513
|
+
the name ``taken`` by the encoding of a conjunction, see
|
|
514
|
+
:func:`goal_mismatch`). A constant that does occur there makes the goal
|
|
515
|
+
an INSTANCE of the claim, and an instance is weaker: ``f(a) = g(a)`` from
|
|
516
|
+
the premise ``f(a) = g(a)`` is not ``∀x f(x) = g(x)`` (universe ``{0, 1}``,
|
|
517
|
+
``a`` ↦ 0, ``f`` ↦ (0, 0), ``g`` ↦ (0, 1)). A compound term, a number or a
|
|
518
|
+
variable of the goal is not a constant either. ``taken`` holds the
|
|
519
|
+
names that are not fresh; a reason for a refusal is appended to ``notes``.
|
|
520
|
+
* **The binding is INJECTIVE:** two distinct conclusion variables may never
|
|
521
|
+
collapse onto the same goal constant. Skolemization maps distinct
|
|
522
|
+
universally quantified variables to distinct fresh constants, so a
|
|
523
|
+
non-injective binding means the Goal line proves a strictly WEAKER
|
|
524
|
+
statement than the conclusion (a fresh ``c`` with ``f(c) = g(c)`` would
|
|
525
|
+
otherwise "prove" ``∀X,Y. f(X) = g(Y)`` by binding both X and Y to
|
|
526
|
+
``c``) — the same bijection discipline :func:`_equation_is_variant`
|
|
527
|
+
already enforces on the axiom-restatement trust boundary.
|
|
528
|
+
"""
|
|
529
|
+
subst = _match(exp_lhs, act_lhs, {})
|
|
530
|
+
if subst is None:
|
|
531
|
+
return False
|
|
532
|
+
subst = _match(exp_rhs, act_rhs, subst)
|
|
533
|
+
if subst is None:
|
|
534
|
+
return False
|
|
535
|
+
names = []
|
|
536
|
+
for variable, value in subst.items():
|
|
537
|
+
if not isinstance(value, Constant):
|
|
538
|
+
if notes is not None:
|
|
539
|
+
notes.append(
|
|
540
|
+
f"the conclusion's variable {variable} is matched to "
|
|
541
|
+
f"{value.to_unicode_str()}, which is not a constant: only a constant of its "
|
|
542
|
+
f"own can stand for a universally quantified variable")
|
|
543
|
+
return False
|
|
544
|
+
if value.name in taken:
|
|
545
|
+
if notes is not None:
|
|
546
|
+
notes.append(
|
|
547
|
+
f"the conclusion's variable {variable} is matched to the constant "
|
|
548
|
+
f"{value.name}, which already occurs in an axiom of the proof or in the "
|
|
549
|
+
f"conclusion, so the goal is an instance of the claim, not the claim")
|
|
550
|
+
return False
|
|
551
|
+
names.append(value.name)
|
|
552
|
+
return len(set(names)) == len(names)
|
|
553
|
+
|
|
554
|
+
|
|
555
|
+
def goal_matches_conclusion(proof: TweeProof, conclusion: Node) -> bool:
|
|
556
|
+
"""True iff ``proof.goal`` actually restates ``conclusion``: :func:`goal_mismatch` has
|
|
557
|
+
nothing to say against it. Never raises."""
|
|
558
|
+
return goal_mismatch(proof, conclusion) is None
|
|
559
|
+
|
|
560
|
+
|
|
561
|
+
def goal_mismatch(proof: TweeProof, conclusion: Node) -> Optional[str]:
|
|
562
|
+
"""``None`` iff ``proof.goal`` actually restates ``conclusion``, else why it does not.
|
|
563
|
+
|
|
564
|
+
This is the second, SEPARATE trust boundary from :func:`check_twee_proof`
|
|
565
|
+
(which never sees ``conclusion`` at all): a proof can be internally
|
|
566
|
+
flawless yet prove the wrong thing if Twee's ``Goal`` header were ever
|
|
567
|
+
inconsistent with the conjecture it was asked to prove. The two checks speak
|
|
568
|
+
about ONE proof: the names that must be fresh below are fresh against the
|
|
569
|
+
axioms the proof restates, which :func:`check_twee_proof` has cross-checked
|
|
570
|
+
against the premises.
|
|
571
|
+
|
|
572
|
+
A single-equation ``conclusion`` is matched directly against the goal's
|
|
573
|
+
``lhs``/``rhs``. A conjunctive ``conclusion`` (``n > 1`` top-level
|
|
574
|
+
conjuncts after flattening) is matched against Twee's observed
|
|
575
|
+
``tuple(...)`` encoding: the goal's ``lhs`` and ``rhs`` must each be an
|
|
576
|
+
application of ONE function symbol to exactly ``n`` arguments, matched against the
|
|
577
|
+
``n`` conjuncts by a
|
|
578
|
+
backtracking SEARCH for a bijection (slot <-> conjunct), each pairing
|
|
579
|
+
using its own independent fresh substitution — Twee was observed to (a)
|
|
580
|
+
Skolemize each conjunct's variables separately even when they share a
|
|
581
|
+
source name, and (b) NOT preserve the conclusion's source conjunct order
|
|
582
|
+
in the tuple once a shared quantifier is involved (verified live: a
|
|
583
|
+
2-conjunct quantified conclusion ``![X]: (g(g(g(g(X))))=X) &
|
|
584
|
+
(f(f(X))=X)`` — g-conjunct first, f-conjunct second in the SOURCE —
|
|
585
|
+
printed its Goal as ``tuple(f(f(x)), g(g(g(g(x2))))) = tuple(x, x2)``,
|
|
586
|
+
f-conjunct FIRST) — so slot order cannot be assumed to track conjunct
|
|
587
|
+
order; see :mod:`atp.twee_entailment`'s module docstring. Returns
|
|
588
|
+
a reason (never raises) if ``conclusion`` is not itself a valid
|
|
589
|
+
equation/conjunction, or if the tuple shape/bijection does not exist.
|
|
590
|
+
|
|
591
|
+
**The encoding symbol is a name of Twee's own.** Twee calls it ``tuple``, and
|
|
592
|
+
``tuple2`` (then ``tuple3``, ...) when the problem already uses ``tuple``
|
|
593
|
+
(measured: ``tuple2(tuple(b, c), e) = tuple2(d, f)`` for the conclusion
|
|
594
|
+
``tuple(b, c) = d ∧ e = f``). Which name it chose is not assumed: the symbol must
|
|
595
|
+
occur in no axiom of the proof and nowhere in the conclusion, and it is not
|
|
596
|
+
a name a variable of the conclusion is bound to. An equation between applications
|
|
597
|
+
of a symbol the problem itself uses (a premise ``tuple(f(a), g(a)) = tuple(b, c)``)
|
|
598
|
+
is a statement about that symbol, never the conjunction ``f(a) = b ∧ g(a) = c``
|
|
599
|
+
(which does not follow: ``tuple`` need not be injective).
|
|
600
|
+
|
|
601
|
+
**The constants a variable is bound to are fresh** (see :func:`_matches_conjunct`):
|
|
602
|
+
a goal that is an instance of a universal claim at a constant of the problem is a
|
|
603
|
+
weaker statement, and is refused with the constant's name.
|
|
604
|
+
|
|
605
|
+
**A free variable of the conclusion** is one unknown element, not a universal
|
|
606
|
+
one, and no goal Twee prints can say which: it is refused by name.
|
|
607
|
+
|
|
608
|
+
**A conjunct that repeats another is one conjunct.** ``A ∧ A`` is ``A``, and Twee
|
|
609
|
+
answers it as it answers ``A``: its clausifier drops a repeated GROUND conjunct
|
|
610
|
+
(measured: ``f(a) = b ∧ f(a) = b`` is proved as the single goal ``f(a) = b``, and
|
|
611
|
+
``f(a) = b ∧ g(a) = c ∧ f(a) = b`` as ``tuple(f(a), g(a)) = tuple(b, c)``), where
|
|
612
|
+
it keeps two conjuncts that have variables apart (each is Skolemised on its own:
|
|
613
|
+
``∀x (f(x) = x ∧ f(x) = x)`` is ``tuple(f(x2), f(x)) = tuple(x2, x)``). So the goal
|
|
614
|
+
is matched against the conjuncts as they stand, and, when some of them repeat an
|
|
615
|
+
earlier one (equal up to the renaming of their own variables, in the same
|
|
616
|
+
orientation), also against the conjuncts with the repeats removed. Both are sound
|
|
617
|
+
readings of the conclusion: a proof of every conjunct of the shorter list proves
|
|
618
|
+
the longer one, whose extra members are copies of members of the shorter. Neither
|
|
619
|
+
loosens the match itself: a proof of ``A`` still does not match ``A ∧ B``, a
|
|
620
|
+
conjunct is still matched under an injective substitution, and a proof of the
|
|
621
|
+
repeated conjunct matches ``A ∧ A`` only when the goal is that conjunct.
|
|
622
|
+
"""
|
|
623
|
+
try:
|
|
624
|
+
scoped = _flatten_scoped(conclusion)
|
|
625
|
+
except ValueError as exc:
|
|
626
|
+
return str(exc)
|
|
627
|
+
|
|
628
|
+
free = sorted(set().union(*(variables for _, _, variables in scoped)))
|
|
629
|
+
if free:
|
|
630
|
+
return (f"the conclusion has the free variable(s) {free}: a free variable is one unknown "
|
|
631
|
+
f"element, and the goal of a proof, whose variables are all universal, cannot "
|
|
632
|
+
f"be the statement about it")
|
|
633
|
+
|
|
634
|
+
conjuncts = [(lhs, rhs) for lhs, rhs, _ in scoped]
|
|
635
|
+
taken: Set[str] = set()
|
|
636
|
+
for axiom in proof.axioms:
|
|
637
|
+
_symbol_names(axiom.equation.lhs, taken)
|
|
638
|
+
_symbol_names(axiom.equation.rhs, taken)
|
|
639
|
+
for lhs, rhs in conjuncts:
|
|
640
|
+
_symbol_names(lhs, taken)
|
|
641
|
+
_symbol_names(rhs, taken)
|
|
642
|
+
|
|
643
|
+
goal_lhs, goal_rhs = proof.goal.equation.lhs, proof.goal.equation.rhs
|
|
644
|
+
notes: List[str] = []
|
|
645
|
+
|
|
646
|
+
if _goal_matches_conjuncts(goal_lhs, goal_rhs, conjuncts, frozenset(taken), notes):
|
|
647
|
+
return None
|
|
648
|
+
distinct = _distinct_conjuncts(conjuncts)
|
|
649
|
+
if len(distinct) < len(conjuncts) and _goal_matches_conjuncts(
|
|
650
|
+
goal_lhs, goal_rhs, distinct, frozenset(taken), notes):
|
|
651
|
+
return None
|
|
652
|
+
return notes[0] if notes else ("the goal is neither the conclusion's equation nor, for a "
|
|
653
|
+
"conjunction, the equation between two applications of one "
|
|
654
|
+
"fresh symbol to its conjuncts")
|
|
655
|
+
|
|
656
|
+
|
|
657
|
+
def _distinct_conjuncts(conjuncts: List[Tuple[Node, Node]]) -> List[Tuple[Node, Node]]:
|
|
658
|
+
"""``conjuncts`` without any that repeats an earlier one: the same equation, in the
|
|
659
|
+
same orientation, up to a consistent renaming of its variables
|
|
660
|
+
(:func:`_equation_is_variant`). Order is kept."""
|
|
661
|
+
kept: List[Tuple[Node, Node]] = []
|
|
662
|
+
for conjunct in conjuncts:
|
|
663
|
+
if not any(_equation_is_variant(earlier, conjunct) for earlier in kept):
|
|
664
|
+
kept.append(conjunct)
|
|
665
|
+
return kept
|
|
666
|
+
|
|
667
|
+
|
|
668
|
+
def _goal_matches_conjuncts(goal_lhs: Node, goal_rhs: Node,
|
|
669
|
+
conjuncts: List[Tuple[Node, Node]],
|
|
670
|
+
taken: FrozenSet[str], notes: List[str]) -> bool:
|
|
671
|
+
"""Whether the goal ``goal_lhs = goal_rhs`` restates exactly ``conjuncts`` (see
|
|
672
|
+
:func:`goal_mismatch`): one conjunct against the goal itself, several against
|
|
673
|
+
the encoding of a conjunction, by a bijection of slots and conjuncts. ``taken`` holds
|
|
674
|
+
the names the encoding symbol and the constants of the variables must not have."""
|
|
675
|
+
if len(conjuncts) == 1:
|
|
676
|
+
exp_lhs, exp_rhs = conjuncts[0]
|
|
677
|
+
return _matches_conjunct(exp_lhs, exp_rhs, goal_lhs, goal_rhs, taken, notes)
|
|
678
|
+
|
|
679
|
+
n = len(conjuncts)
|
|
680
|
+
if not (isinstance(goal_lhs, Function) and len(goal_lhs.args) == n
|
|
681
|
+
and isinstance(goal_rhs, Function) and goal_rhs.name == goal_lhs.name
|
|
682
|
+
and len(goal_rhs.args) == n):
|
|
683
|
+
return False
|
|
684
|
+
encoding = goal_lhs.name
|
|
685
|
+
if encoding in taken:
|
|
686
|
+
notes.append(
|
|
687
|
+
f"the goal is an equation between applications of {encoding}, which is how Twee "
|
|
688
|
+
f"encodes a conjunction as one equation, but {encoding} is a symbol of an axiom of "
|
|
689
|
+
f"the proof or of the conclusion, so the goal says something about that symbol, not "
|
|
690
|
+
f"about the conjuncts")
|
|
691
|
+
return False
|
|
692
|
+
taken = taken | {encoding}
|
|
693
|
+
|
|
694
|
+
used = [False] * n
|
|
695
|
+
|
|
696
|
+
def backtrack(i: int) -> bool:
|
|
697
|
+
if i == n:
|
|
698
|
+
return True
|
|
699
|
+
exp_lhs, exp_rhs = conjuncts[i]
|
|
700
|
+
for j in range(n):
|
|
701
|
+
if used[j]:
|
|
702
|
+
continue
|
|
703
|
+
if _matches_conjunct(exp_lhs, exp_rhs, goal_lhs.args[j], goal_rhs.args[j],
|
|
704
|
+
taken, notes):
|
|
705
|
+
used[j] = True
|
|
706
|
+
if backtrack(i + 1):
|
|
707
|
+
return True
|
|
708
|
+
used[j] = False
|
|
709
|
+
return False
|
|
710
|
+
|
|
711
|
+
return backtrack(0)
|