unicode-logic-kit 0.31.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- unicode_logic_kit/__init__.py +385 -0
- unicode_logic_kit/__main__.py +520 -0
- unicode_logic_kit/_deadline.py +219 -0
- unicode_logic_kit/ace/__init__.py +126 -0
- unicode_logic_kit/ace/_align.py +135 -0
- unicode_logic_kit/ace/chem_lexicon.py +128 -0
- unicode_logic_kit/ace/drs_reader.py +570 -0
- unicode_logic_kit/ace/mapping.py +666 -0
- unicode_logic_kit/ace/reverse_modal.py +138 -0
- unicode_logic_kit/ace/runner.py +551 -0
- unicode_logic_kit/ace/translate.py +452 -0
- unicode_logic_kit/ace/verbalize.py +1070 -0
- unicode_logic_kit/api.py +1284 -0
- unicode_logic_kit/atp/__init__.py +177 -0
- unicode_logic_kit/atp/_ascii_names.py +113 -0
- unicode_logic_kit/atp/_html.py +72 -0
- unicode_logic_kit/atp/_substructural_input.py +228 -0
- unicode_logic_kit/atp/_tff_problem.py +715 -0
- unicode_logic_kit/atp/_tptp_problem.py +1111 -0
- unicode_logic_kit/atp/_writer_support.py +289 -0
- unicode_logic_kit/atp/clingo_backend.py +1180 -0
- unicode_logic_kit/atp/cvc5_backend.py +1385 -0
- unicode_logic_kit/atp/eprover_backend.py +732 -0
- unicode_logic_kit/atp/finite_domain.py +1055 -0
- unicode_logic_kit/atp/fitch.py +1547 -0
- unicode_logic_kit/atp/fitch_search.py +551 -0
- unicode_logic_kit/atp/hets_backend.py +339 -0
- unicode_logic_kit/atp/hybrid_down.py +120 -0
- unicode_logic_kit/atp/incremental.py +250 -0
- unicode_logic_kit/atp/kripke_enum.py +741 -0
- unicode_logic_kit/atp/lambek.py +436 -0
- unicode_logic_kit/atp/leo3_backend.py +332 -0
- unicode_logic_kit/atp/linear.py +738 -0
- unicode_logic_kit/atp/lj.py +705 -0
- unicode_logic_kit/atp/logic_backends.py +566 -0
- unicode_logic_kit/atp/ltl_tableau.py +1084 -0
- unicode_logic_kit/atp/minizinc_backend.py +1402 -0
- unicode_logic_kit/atp/modal_tableau.py +1382 -0
- unicode_logic_kit/atp/nanocop_backend.py +410 -0
- unicode_logic_kit/atp/portfolio.py +489 -0
- unicode_logic_kit/atp/protocol.py +1803 -0
- unicode_logic_kit/atp/prover9_entailment.py +1153 -0
- unicode_logic_kit/atp/resolution.py +1376 -0
- unicode_logic_kit/atp/resolution_check.py +1114 -0
- unicode_logic_kit/atp/sequent.py +1050 -0
- unicode_logic_kit/atp/tableau.py +921 -0
- unicode_logic_kit/atp/tableau_check.py +543 -0
- unicode_logic_kit/atp/tptp_ncl.py +811 -0
- unicode_logic_kit/atp/tptp_tff.py +1546 -0
- unicode_logic_kit/atp/tstp.py +1333 -0
- unicode_logic_kit/atp/tstp_check.py +1096 -0
- unicode_logic_kit/atp/twee_backend.py +236 -0
- unicode_logic_kit/atp/twee_check.py +711 -0
- unicode_logic_kit/atp/twee_entailment.py +953 -0
- unicode_logic_kit/atp/vampire_entailment.py +540 -0
- unicode_logic_kit/atp/z3_arith.py +470 -0
- unicode_logic_kit/atp/z3_equivalence.py +36 -0
- unicode_logic_kit/atp/z3_fuzzy.py +362 -0
- unicode_logic_kit/atp/z3_input.py +500 -0
- unicode_logic_kit/atp/z3_models.py +208 -0
- unicode_logic_kit/chem/__init__.py +88 -0
- unicode_logic_kit/chem/_naming.py +284 -0
- unicode_logic_kit/chem/cache.py +185 -0
- unicode_logic_kit/chem/interop.py +244 -0
- unicode_logic_kit/chem/mol.py +525 -0
- unicode_logic_kit/chem/signature.py +112 -0
- unicode_logic_kit/comorphism.py +497 -0
- unicode_logic_kit/dl/__init__.py +384 -0
- unicode_logic_kit/dl/classification.py +227 -0
- unicode_logic_kit/dl/concepts.py +632 -0
- unicode_logic_kit/dl/datatypes.py +818 -0
- unicode_logic_kit/dl/owl_functional.py +2433 -0
- unicode_logic_kit/dl/owl_manchester.py +1637 -0
- unicode_logic_kit/dl/owl_reasoner.py +790 -0
- unicode_logic_kit/dl/parser.py +391 -0
- unicode_logic_kit/dl/tableau.py +4048 -0
- unicode_logic_kit/dl/translate.py +2704 -0
- unicode_logic_kit/drt/__init__.py +94 -0
- unicode_logic_kit/drt/export.py +179 -0
- unicode_logic_kit/drt/nodes.py +506 -0
- unicode_logic_kit/drt/parser.py +965 -0
- unicode_logic_kit/drt/resolve.py +195 -0
- unicode_logic_kit/drt/reverse.py +175 -0
- unicode_logic_kit/eval/__init__.py +106 -0
- unicode_logic_kit/eval/batch.py +382 -0
- unicode_logic_kit/eval/canonical.py +663 -0
- unicode_logic_kit/eval/chem_batch.py +606 -0
- unicode_logic_kit/eval/converses.py +200 -0
- unicode_logic_kit/eval/datasets/__init__.py +136 -0
- unicode_logic_kit/eval/datasets/_base.py +263 -0
- unicode_logic_kit/eval/datasets/_proofwriter_proof.py +422 -0
- unicode_logic_kit/eval/datasets/c3po.py +678 -0
- unicode_logic_kit/eval/datasets/folio.py +158 -0
- unicode_logic_kit/eval/datasets/fracas.py +418 -0
- unicode_logic_kit/eval/datasets/groves.py +191 -0
- unicode_logic_kit/eval/datasets/logicbench.py +467 -0
- unicode_logic_kit/eval/datasets/logicnli.py +303 -0
- unicode_logic_kit/eval/datasets/malls.py +133 -0
- unicode_logic_kit/eval/datasets/pfolio.py +594 -0
- unicode_logic_kit/eval/datasets/pmb.py +242 -0
- unicode_logic_kit/eval/datasets/prontoqa.py +611 -0
- unicode_logic_kit/eval/datasets/proofwriter.py +1431 -0
- unicode_logic_kit/eval/datasets/proverqa.py +674 -0
- unicode_logic_kit/eval/datasets/willow.py +478 -0
- unicode_logic_kit/eval/equivalence.py +466 -0
- unicode_logic_kit/eval/exercise_gen.py +533 -0
- unicode_logic_kit/eval/explain.py +791 -0
- unicode_logic_kit/eval/generality.py +750 -0
- unicode_logic_kit/eval/metric_hf.py +458 -0
- unicode_logic_kit/eval/predicate_match.py +343 -0
- unicode_logic_kit/eval/theory_check.py +1170 -0
- unicode_logic_kit/eval/validate.py +306 -0
- unicode_logic_kit/fol/__init__.py +177 -0
- unicode_logic_kit/fol/_atom_keys.py +510 -0
- unicode_logic_kit/fol/_fol_nodes.py +3586 -0
- unicode_logic_kit/fol/_free_parameters.py +105 -0
- unicode_logic_kit/fol/_ho_nodes.py +448 -0
- unicode_logic_kit/fol/_hybrid_nodes.py +308 -0
- unicode_logic_kit/fol/_identifiers.py +1091 -0
- unicode_logic_kit/fol/_lambek_nodes.py +112 -0
- unicode_logic_kit/fol/_linear_nodes.py +352 -0
- unicode_logic_kit/fol/_modal_nodes.py +1467 -0
- unicode_logic_kit/fol/_msfl_nodes.py +2196 -0
- unicode_logic_kit/fol/_numeral_symbols.py +231 -0
- unicode_logic_kit/fol/_so_nodes.py +200 -0
- unicode_logic_kit/fol/_symbol_names.py +81 -0
- unicode_logic_kit/fol/_team_nodes.py +181 -0
- unicode_logic_kit/fol/_tptp_symbols.py +551 -0
- unicode_logic_kit/fol/_truth_constants.py +117 -0
- unicode_logic_kit/fol/casl_export.py +1135 -0
- unicode_logic_kit/fol/casl_import.py +929 -0
- unicode_logic_kit/fol/derivation.py +367 -0
- unicode_logic_kit/fol/dialect_detect.py +70 -0
- unicode_logic_kit/fol/dialect_repair.py +537 -0
- unicode_logic_kit/fol/frames.py +637 -0
- unicode_logic_kit/fol/grammars/terminals.lark +31 -0
- unicode_logic_kit/fol/lambda_tools.py +297 -0
- unicode_logic_kit/fol/latex_input.py +429 -0
- unicode_logic_kit/fol/modal_translation.py +944 -0
- unicode_logic_kit/fol/msflparser.py +1033 -0
- unicode_logic_kit/fol/naming.py +422 -0
- unicode_logic_kit/fol/nodes.py +241 -0
- unicode_logic_kit/fol/normalforms.py +492 -0
- unicode_logic_kit/fol/pal.py +287 -0
- unicode_logic_kit/fol/prolog_export.py +566 -0
- unicode_logic_kit/fol/prolog_input.py +505 -0
- unicode_logic_kit/fol/prover9_input.py +1325 -0
- unicode_logic_kit/fol/qml.py +1760 -0
- unicode_logic_kit/fol/qmltp_input.py +525 -0
- unicode_logic_kit/fol/sanitize.py +221 -0
- unicode_logic_kit/fol/serialize.py +79 -0
- unicode_logic_kit/fol/signature.py +1290 -0
- unicode_logic_kit/fol/simplify_check.py +544 -0
- unicode_logic_kit/fol/spans.py +594 -0
- unicode_logic_kit/fol/tptp_input.py +1503 -0
- unicode_logic_kit/fol/tptp_repair.py +941 -0
- unicode_logic_kit/fol/unification.py +157 -0
- unicode_logic_kit/fol/verbalize.py +263 -0
- unicode_logic_kit/hets/__init__.py +163 -0
- unicode_logic_kit/hets/bridge.py +142 -0
- unicode_logic_kit/hets/client.py +748 -0
- unicode_logic_kit/hets/docker.py +420 -0
- unicode_logic_kit/hets/dol.py +712 -0
- unicode_logic_kit/hets/haskell_json.py +355 -0
- unicode_logic_kit/hets/owl_backend.py +794 -0
- unicode_logic_kit/hets/owl_cli.py +598 -0
- unicode_logic_kit/hets/symbols.py +512 -0
- unicode_logic_kit/hol/__init__.py +140 -0
- unicode_logic_kit/hol/_ho_common.py +323 -0
- unicode_logic_kit/hol/_isabelle_binders.py +125 -0
- unicode_logic_kit/hol/classical.py +812 -0
- unicode_logic_kit/hol/deepshallow/__init__.py +45 -0
- unicode_logic_kit/hol/deepshallow/_common.py +177 -0
- unicode_logic_kit/hol/deepshallow/conditional.py +225 -0
- unicode_logic_kit/hol/deepshallow/intuitionistic.py +181 -0
- unicode_logic_kit/hol/deepshallow/modal.py +217 -0
- unicode_logic_kit/hol/deepshallow/qml.py +406 -0
- unicode_logic_kit/hol/deepshallow/relevant.py +206 -0
- unicode_logic_kit/hol/free.py +753 -0
- unicode_logic_kit/hol/goedel.py +336 -0
- unicode_logic_kit/hol/ho_modal.py +1743 -0
- unicode_logic_kit/hol/intuitionistic.py +403 -0
- unicode_logic_kit/hol/isabelle_conditional.py +593 -0
- unicode_logic_kit/hol/isabelle_modal.py +1908 -0
- unicode_logic_kit/hol/isabelle_relevant.py +412 -0
- unicode_logic_kit/hol/isabelle_runner.py +1147 -0
- unicode_logic_kit/hol/isabelle_substructural.py +884 -0
- unicode_logic_kit/hol/lean.py +1018 -0
- unicode_logic_kit/hol/manyvalued.py +921 -0
- unicode_logic_kit/hol/secondorder.py +687 -0
- unicode_logic_kit/hol/thf_modal.py +941 -0
- unicode_logic_kit/hol/thirdorder.py +397 -0
- unicode_logic_kit/ilp/__init__.py +89 -0
- unicode_logic_kit/ilp/readback.py +389 -0
- unicode_logic_kit/ilp/separation.py +153 -0
- unicode_logic_kit/ilp/task.py +730 -0
- unicode_logic_kit/logic.py +163 -0
- unicode_logic_kit/mcp/__init__.py +28 -0
- unicode_logic_kit/mcp/__main__.py +5 -0
- unicode_logic_kit/mcp/chem_tools.py +1031 -0
- unicode_logic_kit/mcp/server.py +2453 -0
- unicode_logic_kit/mcp/syntax_spec.py +681 -0
- unicode_logic_kit/prob/__init__.py +53 -0
- unicode_logic_kit/prob/_bdd.py +225 -0
- unicode_logic_kit/prob/_column_gen.py +668 -0
- unicode_logic_kit/prob/distribution.py +686 -0
- unicode_logic_kit/prob/nilsson.py +470 -0
- unicode_logic_kit/py.typed +0 -0
- unicode_logic_kit/semantics/__init__.py +137 -0
- unicode_logic_kit/semantics/_modal_reject.py +156 -0
- unicode_logic_kit/semantics/action_models.py +466 -0
- unicode_logic_kit/semantics/asp_models.py +1200 -0
- unicode_logic_kit/semantics/conditional.py +580 -0
- unicode_logic_kit/semantics/dynamic_epistemic.py +95 -0
- unicode_logic_kit/semantics/free_logic.py +913 -0
- unicode_logic_kit/semantics/fuzzy.py +384 -0
- unicode_logic_kit/semantics/fuzzy_kripke.py +442 -0
- unicode_logic_kit/semantics/intuitionistic.py +581 -0
- unicode_logic_kit/semantics/kripke.py +1139 -0
- unicode_logic_kit/semantics/manyvalued.py +580 -0
- unicode_logic_kit/semantics/matrix.py +342 -0
- unicode_logic_kit/semantics/model_eval.py +1135 -0
- unicode_logic_kit/semantics/modelfinder.py +1036 -0
- unicode_logic_kit/semantics/nonmonotonic.py +372 -0
- unicode_logic_kit/semantics/relevant.py +331 -0
- unicode_logic_kit/semantics/secondorder.py +657 -0
- unicode_logic_kit/semantics/structures.py +352 -0
- unicode_logic_kit/semantics/tarski.py +975 -0
- unicode_logic_kit/semantics/team.py +315 -0
- unicode_logic_kit/semantics/team_translation.py +416 -0
- unicode_logic_kit/semantics/thirdorder.py +358 -0
- unicode_logic_kit/semantics/tnorm.py +85 -0
- unicode_logic_kit/semantics/truthtable.py +201 -0
- unicode_logic_kit-0.31.0.dist-info/METADATA +333 -0
- unicode_logic_kit-0.31.0.dist-info/RECORD +237 -0
- unicode_logic_kit-0.31.0.dist-info/WHEEL +4 -0
- unicode_logic_kit-0.31.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,732 @@
|
|
|
1
|
+
"""E and Zipperposition as external TPTP backends (one shared runner).
|
|
2
|
+
|
|
3
|
+
Both provers speak the same protocol this kit already reads for Vampire:
|
|
4
|
+
TPTP ``fof`` in, ``SZS status`` + (for E) a TSTP derivation out — so they
|
|
5
|
+
share one runner here instead of duplicating the temp-file/WSL plumbing a
|
|
6
|
+
third and fourth time. (Vampire keeps its own module,
|
|
7
|
+
:mod:`~unicode_logic_kit.atp.vampire_entailment`, because its original
|
|
8
|
+
bool-returning route is public API.)
|
|
9
|
+
|
|
10
|
+
Why these two (roadmap ranking): E is the classic high-diversity
|
|
11
|
+
superposition prover — cheap to obtain (Ubuntu 24.04 ships ``eprover``
|
|
12
|
+
3.0.03 in universe, so CI gets LIVE coverage from ``apt``), strong on
|
|
13
|
+
problems Z3's instantiation heuristics miss. Zipperposition adds a second,
|
|
14
|
+
OCaml-built superposition engine with higher-order ambitions.
|
|
15
|
+
|
|
16
|
+
Per-backend acquisition path (the kit-wide honesty rule: every external
|
|
17
|
+
backend documents how to get it, per platform):
|
|
18
|
+
|
|
19
|
+
* **E** — ``apt install eprover`` (Debian/Ubuntu; WSL: note that Ubuntu
|
|
20
|
+
22.04 does NOT carry the package — 24.04 does), or build from source
|
|
21
|
+
(https://github.com/eprover/eprover, ``./configure && make``). Windows:
|
|
22
|
+
via WSL. Env override: ``$UFK_EPROVER_CMD`` (prefix ``wsl:`` to force the
|
|
23
|
+
WSL route, e.g. ``wsl:/usr/bin/eprover``).
|
|
24
|
+
* **Zipperposition** — ``opam install zipperposition`` (no Debian/Ubuntu
|
|
25
|
+
package exists as of 2026-08); realistically an opam-capable Linux/WSL
|
|
26
|
+
box only. Env override: ``$UFK_ZIPPERPOSITION_CMD`` (same ``wsl:``
|
|
27
|
+
convention). Where it is absent the backend reports unavailable — tests
|
|
28
|
+
gate on that and skip, they never fake a pass.
|
|
29
|
+
|
|
30
|
+
Discovery order (each backend, cached per process): env override → native
|
|
31
|
+
binary on PATH → the same name inside WSL (``wsl.exe which <name>``).
|
|
32
|
+
|
|
33
|
+
Verdicts: the SZS status line is authoritative (``extract_szs_status`` —
|
|
34
|
+
E prints ``# SZS status …`` with a hash marker, Zipperposition ``% SZS
|
|
35
|
+
status …``; the extractor accepts both), mapped through
|
|
36
|
+
``szs_to_verdict_fields(query="conjecture")`` exactly like the Vampire
|
|
37
|
+
detailed route. E is additionally asked for ``--proof-object``, and a
|
|
38
|
+
parseable TSTP derivation lands in ``Verdict.proof``; an unparseable one
|
|
39
|
+
degrades to ``proof=None`` with the parse error in ``detail`` (the status
|
|
40
|
+
line alone already carries the verdict — a broken proof printer must not
|
|
41
|
+
turn a Theorem into an error, but the degradation is never silent).
|
|
42
|
+
|
|
43
|
+
**E's own time limit is the call running out of time.** The kit hands E its call
|
|
44
|
+
budget as ``--cpu-limit=<seconds>`` and nothing else, so when E stops at that limit
|
|
45
|
+
(it prints ``Failure: Resource limit exceeded (time)`` and ``SZS status ResourceOut``)
|
|
46
|
+
the call ran out of time: UNKNOWN / ``"timeout"``, the same verdict as a run the kit
|
|
47
|
+
cut off itself, and as Vampire's. A ``ResourceOut`` that is not that limit stays
|
|
48
|
+
``"bound_hit"``: E's wording for its memory limit is ``(memory)``, and for a limit the
|
|
49
|
+
kit never passes (``--processed-clauses-limit``, ``--soft-cpu-limit``) ``User resource
|
|
50
|
+
limit exceeded!``.
|
|
51
|
+
|
|
52
|
+
**Typed arithmetic (``sort="int"`` / ``"real"``) is not E's arithmetic.** The option writes the
|
|
53
|
+
problem as typed TPTP (:func:`~unicode_logic_kit.atp._tff_problem.generate_tff_arith_problem`),
|
|
54
|
+
which Vampire reads with its own arithmetic. E 3.5.1 reads the typed text and then does
|
|
55
|
+
three things (measured on that build) that its caller must not mistake for arithmetic: it
|
|
56
|
+
types ``$sum``, ``$difference``, ``$product``, ``$quotient``, ``$quotient_e`` and ``$uminus`` as
|
|
57
|
+
functions into the individuals and stops with ``Type error`` on every term that uses one
|
|
58
|
+
(``$sum(1,1) = 2`` is a type error, not a theorem); it reads a ``$real`` literal only
|
|
59
|
+
approximately and calls two close ones equal (``1.0 = 1.0000001``, ``0.1 = 0.1000001`` and
|
|
60
|
+
``9007199254740993.0 = 9007199254740992.0`` are theorems for it, ``0.1 = 0.101`` is not), where
|
|
61
|
+
``$int`` literals are exact; and it reads ``$less``, ``$lesseq``, ``$greater`` and ``$greatereq`` as predicates it never
|
|
62
|
+
evaluates, so it proves only what holds of them as uninterpreted symbols and answers ``GaveUp``
|
|
63
|
+
(``unknown``) for the rest. So E is refused by name, as ``unknown`` / ``unsupported`` (an
|
|
64
|
+
exception on a direct call), for a problem with an arithmetic function symbol, and for a numeral
|
|
65
|
+
under ``sort="real"``. What is left is a problem that E answers soundly, usually without using any
|
|
66
|
+
arithmetic: use Vampire or Z3 (``z3_arith``) for arithmetic. Zipperposition is not refused:
|
|
67
|
+
its behaviour on typed arithmetic is not measured here.
|
|
68
|
+
"""
|
|
69
|
+
|
|
70
|
+
import os
|
|
71
|
+
import re
|
|
72
|
+
import shutil
|
|
73
|
+
import subprocess
|
|
74
|
+
import tempfile
|
|
75
|
+
import time
|
|
76
|
+
from typing import List, Optional, Sequence, Tuple
|
|
77
|
+
|
|
78
|
+
from ..fol.nodes import Function, Node, Number
|
|
79
|
+
from ._ascii_names import reverse_map_text
|
|
80
|
+
from ._tff_problem import generate_tff_arith_problem
|
|
81
|
+
from ._tptp_problem import generate_tptp_problem_for_prover, generate_tptp_problem_with_mapping
|
|
82
|
+
from ._writer_support import names_kwargs
|
|
83
|
+
from .protocol import (
|
|
84
|
+
ProverBackend, Verdict, UNKNOWN, ERROR, _binary_version, _rejection_detail,
|
|
85
|
+
)
|
|
86
|
+
|
|
87
|
+
__all__ = ["EProverBackend", "ZipperpositionBackend",
|
|
88
|
+
"check_entailment_eprover_detailed", "eprover_available",
|
|
89
|
+
"zipperposition_available", "eprover_relevant_premises"]
|
|
90
|
+
|
|
91
|
+
#: What E 3.5.1 prints when it stops at the ``--cpu-limit`` the kit passed (recorded:
|
|
92
|
+
#: ``%% Failure: Resource limit exceeded (time)`` then ``%% SZS status ResourceOut``,
|
|
93
|
+
#: after ``eprover: CPU time limit exceeded, terminating`` on stderr).
|
|
94
|
+
_E_OWN_TIME_LIMIT = re.compile(r"Failure:\s*Resource limit exceeded \(time\)")
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
def _eprover_verdict_fields(szs: str, raw_output: str) -> Tuple[str, Optional[str]]:
|
|
98
|
+
"""``(status, reason)`` of E's SZS status, with ONE difference from
|
|
99
|
+
:func:`~unicode_logic_kit.atp.tstp.szs_to_verdict_fields`: a ``ResourceOut`` that
|
|
100
|
+
is E stopping at the ``--cpu-limit`` the kit derived from the call's budget IS
|
|
101
|
+
the call running out of time (UNKNOWN / ``"timeout"``), not a bound that was hit
|
|
102
|
+
(see the module docstring). ``raw_output`` is E's output, stdout and stderr."""
|
|
103
|
+
from .tstp import szs_to_verdict_fields
|
|
104
|
+
|
|
105
|
+
if szs == "ResourceOut" and _E_OWN_TIME_LIMIT.search(raw_output):
|
|
106
|
+
return UNKNOWN, "timeout"
|
|
107
|
+
return szs_to_verdict_fields(szs, query="conjecture")
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
#: What E 3.5.1 does with a function symbol of TPTP's arithmetic (measured: ``$sum(1,1) = 2``
|
|
111
|
+
#: gives ``terms $sum(1,1): $i and 2: $int should have the same sort`` and no SZS status).
|
|
112
|
+
_E_NO_ARITHMETIC_FUNCTIONS = (
|
|
113
|
+
"eprover: the typed arithmetic text of this problem uses the operator {name!r} ({word}), "
|
|
114
|
+
"which E 3.5.1 cannot read: it types TPTP's arithmetic functions as functions into the "
|
|
115
|
+
"individuals and stops with 'Type error' on every term that uses one, whatever the sort, "
|
|
116
|
+
"so it has no answer to give. E does not evaluate arithmetic. Ask Vampire "
|
|
117
|
+
"(backends=['vampire']) or Z3 with sort=, which do."
|
|
118
|
+
)
|
|
119
|
+
|
|
120
|
+
#: What E 3.5.1 does with a ``$real`` literal (measured: ``0.1 = 0.1000001``, ``1.0 = 1.0000001``
|
|
121
|
+
#: and ``9007199254740993.0 = 9007199254740992.0`` are theorems for it; ``$int`` literals of any
|
|
122
|
+
#: size are exact).
|
|
123
|
+
_E_NO_REAL_NUMERALS = (
|
|
124
|
+
"eprover: the typed arithmetic text of this problem has the numeral {value!r} under "
|
|
125
|
+
"sort='real', and E 3.5.1 reads a $real literal only approximately: it calls two close ones "
|
|
126
|
+
"equal (0.1 = 0.1000001 and 9007199254740993.0 = 9007199254740992.0 are theorems for it), so "
|
|
127
|
+
"it could prove what is false of the real numbers. Use sort='int' (E reads integer literals "
|
|
128
|
+
"exactly), or Vampire (backends=['vampire']) or Z3 with sort=, which read a real exactly."
|
|
129
|
+
)
|
|
130
|
+
|
|
131
|
+
|
|
132
|
+
def _eprover_arithmetic_refusal(premises: Sequence[Node], conclusion: Optional[Node],
|
|
133
|
+
sort: Optional[str]) -> Optional[str]:
|
|
134
|
+
"""Why E 3.5.1 is not given the typed arithmetic text of this problem, or ``None``.
|
|
135
|
+
|
|
136
|
+
Measured on E 3.5.1 (see the module docstring): an arithmetic function symbol is a type error
|
|
137
|
+
and a ``$real`` literal is read approximately, so a problem with either is refused by name.
|
|
138
|
+
A comparison (``< > ≤ ≥``) is read as an uninterpreted predicate, which can only make E
|
|
139
|
+
prove less than the arithmetic reading does, never more, and E answers ``GaveUp`` and never
|
|
140
|
+
``CounterSatisfiable`` for a text with an interpreted symbol (measured on a battery of valid
|
|
141
|
+
and invalid comparisons, with and without integer literals), so it is not refused. ``None``
|
|
142
|
+
for ``sort=None`` (the route does not apply) and for a sort the writer itself refuses.
|
|
143
|
+
"""
|
|
144
|
+
if sort not in ("int", "real"):
|
|
145
|
+
return None
|
|
146
|
+
formulas = list(premises) + ([] if conclusion is None else [conclusion])
|
|
147
|
+
for formula in formulas:
|
|
148
|
+
for node in formula.walk():
|
|
149
|
+
if isinstance(node, Function) and node.name in Function.TPTP_ARITH_OPS:
|
|
150
|
+
return _E_NO_ARITHMETIC_FUNCTIONS.format(
|
|
151
|
+
name=node.name, word=Function.TPTP_ARITH_OPS[node.name])
|
|
152
|
+
if sort == "real" and isinstance(node, Number):
|
|
153
|
+
return _E_NO_REAL_NUMERALS.format(value=node.value)
|
|
154
|
+
return None
|
|
155
|
+
|
|
156
|
+
|
|
157
|
+
def _generate_tptp_problem(premises: List[Node], conclusion: Node,
|
|
158
|
+
*, tff: Optional[bool] = None,
|
|
159
|
+
sort: Optional[str] = None,
|
|
160
|
+
premise_names: Optional[Sequence[str]] = None) -> str:
|
|
161
|
+
"""``fof(premise_<i>, axiom, …).`` lines + one ``fof(goal, conjecture, …).``
|
|
162
|
+
— or, when ``tff``/``sort`` selects a typed route, one of its siblings.
|
|
163
|
+
|
|
164
|
+
``sort`` (``None`` by default) opts into the single-numeric-sort typed
|
|
165
|
+
arithmetic route (:func:`atp._tff_problem.generate_tff_arith_problem`,
|
|
166
|
+
``'real'`` or ``'int'``) UNCONDITIONALLY when given, taking priority
|
|
167
|
+
over ``tff``. It writes the TEXT of that route, which Vampire reads with its own
|
|
168
|
+
arithmetic on `+ - * /` and `< > ≤ ≥`; E 3.5.1 does not (see the module docstring:
|
|
169
|
+
:meth:`EProverBackend.decide` and :func:`check_entailment_eprover_detailed` refuse by name
|
|
170
|
+
what E cannot read, and this function only writes the text), and the untyped ``fof`` route
|
|
171
|
+
below cannot (see :mod:`atp._tff_problem`'s module docstring): without
|
|
172
|
+
``sort`` a numeral is a constant identified by its value and those
|
|
173
|
+
symbols are uninterpreted, written under ordinary words by the problem
|
|
174
|
+
writers (see :mod:`atp._tptp_problem`), so ``1 ≠ 2`` is not provable. With
|
|
175
|
+
``sort=None``, ``tff=None`` (the default) tries
|
|
176
|
+
:func:`atp.tptp_tff.generate_tff_problem` when ``premises``/
|
|
177
|
+
``conclusion`` contain a SortedQuantifier/SortedConstant/SortedCount
|
|
178
|
+
node (see :func:`atp.tptp_tff.problem_needs_tff`) and writes the ``fof``
|
|
179
|
+
problem instead when the typed writer refuses it
|
|
180
|
+
(a :class:`~unicode_logic_kit.atp.tptp_tff.Tf0Refusal`: its text would ask another
|
|
181
|
+
question than the kit's; or a node it does not cover that the ``fof`` writer does:
|
|
182
|
+
a counting quantifier, ``Contrast``, ``Measure``; see :func:`atp._tptp_problem
|
|
183
|
+
.generate_tptp_problem_for_prover`); without a sorted node (or with
|
|
184
|
+
``tff=False``) this is the shared ``fof`` writer
|
|
185
|
+
(:func:`atp._tptp_problem.generate_tptp_problem`, also used by the
|
|
186
|
+
Vampire route, which builds the identical ``fof`` problem shape);
|
|
187
|
+
``tff=True`` raises the typed writer's refusal.
|
|
188
|
+
Every route's own ``NotImplementedError`` (an untranslatable node, or a
|
|
189
|
+
cross-formula symbol-collision — see each module's docstring)
|
|
190
|
+
propagates; the backends below turn that into an UNKNOWN/"unsupported"
|
|
191
|
+
verdict. ``premise_names`` names the premises' ``axiom`` lines in whichever route
|
|
192
|
+
is written (see :func:`atp._tptp_problem.generate_tptp_problem_with_mapping`),
|
|
193
|
+
``premise_<i>`` when it is ``None``.
|
|
194
|
+
"""
|
|
195
|
+
if sort is not None:
|
|
196
|
+
problem, _name_map = generate_tff_arith_problem(
|
|
197
|
+
premises, conclusion, sort=sort, **names_kwargs(premise_names))
|
|
198
|
+
return problem
|
|
199
|
+
return generate_tptp_problem_for_prover(
|
|
200
|
+
premises, conclusion, tff=tff, fof_writer=generate_tptp_problem_with_mapping,
|
|
201
|
+
premise_names=premise_names).text
|
|
202
|
+
|
|
203
|
+
|
|
204
|
+
def _to_wsl_path(windows_path: str) -> str:
|
|
205
|
+
"""Windows path → ``/mnt/...`` via ``wslpath`` (see vampire_entailment)."""
|
|
206
|
+
result = subprocess.run(
|
|
207
|
+
["wsl.exe", "wslpath", "-u", windows_path.replace("\\", "/")],
|
|
208
|
+
capture_output=True, text=True, timeout=20)
|
|
209
|
+
wsl_path = result.stdout.strip()
|
|
210
|
+
if not wsl_path:
|
|
211
|
+
raise RuntimeError(
|
|
212
|
+
f"wslpath could not translate {windows_path!r} (is WSL "
|
|
213
|
+
f"available?): {result.stderr.strip()}")
|
|
214
|
+
return wsl_path
|
|
215
|
+
|
|
216
|
+
|
|
217
|
+
# (command, use_wsl) per prover name, or None when nothing was found.
|
|
218
|
+
# Cached because discovery may spawn a WSL probe process.
|
|
219
|
+
_DISCOVERY_CACHE: dict = {}
|
|
220
|
+
|
|
221
|
+
|
|
222
|
+
def _discover(name: str, env_var: str) -> Optional[Tuple[str, bool]]:
|
|
223
|
+
"""Resolve prover ``name`` to ``(command, use_wsl)``, or ``None``.
|
|
224
|
+
|
|
225
|
+
Order: ``$<env_var>`` (a ``wsl:`` prefix forces the WSL route) → native
|
|
226
|
+
``shutil.which`` → ``wsl.exe which <name>``. The env override is read
|
|
227
|
+
FRESH on every call and never cached — that is what makes "set the env
|
|
228
|
+
var to repoint" true even after an earlier miss was cached
|
|
229
|
+
(review-confirmed: caching before the env read froze a process's first
|
|
230
|
+
discovery forever). Only the binary probes (PATH/WSL) are cached.
|
|
231
|
+
"""
|
|
232
|
+
override = os.environ.get(env_var)
|
|
233
|
+
if override:
|
|
234
|
+
if override.startswith("wsl:"):
|
|
235
|
+
return (override[4:], True)
|
|
236
|
+
return (override, False)
|
|
237
|
+
|
|
238
|
+
key = (name, env_var)
|
|
239
|
+
if key in _DISCOVERY_CACHE:
|
|
240
|
+
return _DISCOVERY_CACHE[key]
|
|
241
|
+
|
|
242
|
+
resolved: Optional[Tuple[str, bool]] = None
|
|
243
|
+
if shutil.which(name):
|
|
244
|
+
resolved = (name, False)
|
|
245
|
+
else:
|
|
246
|
+
try:
|
|
247
|
+
probe = subprocess.run(["wsl.exe", "which", name],
|
|
248
|
+
capture_output=True, text=True, timeout=20)
|
|
249
|
+
path = probe.stdout.strip()
|
|
250
|
+
if probe.returncode == 0 and path:
|
|
251
|
+
resolved = (path, True)
|
|
252
|
+
except (OSError, subprocess.TimeoutExpired):
|
|
253
|
+
resolved = None
|
|
254
|
+
|
|
255
|
+
_DISCOVERY_CACHE[key] = resolved
|
|
256
|
+
return resolved
|
|
257
|
+
|
|
258
|
+
|
|
259
|
+
def eprover_available() -> bool:
|
|
260
|
+
"""Pure discovery: is an ``eprover`` binary reachable (native or WSL)?"""
|
|
261
|
+
return _discover("eprover", "UFK_EPROVER_CMD") is not None
|
|
262
|
+
|
|
263
|
+
|
|
264
|
+
def zipperposition_available() -> bool:
|
|
265
|
+
"""Pure discovery for ``zipperposition`` (see the module docstring)."""
|
|
266
|
+
return _discover("zipperposition", "UFK_ZIPPERPOSITION_CMD") is not None
|
|
267
|
+
|
|
268
|
+
|
|
269
|
+
def _run_tptp_prover(problem: str, command: str, args: Sequence[str],
|
|
270
|
+
use_wsl: bool, timeout_s: float) -> Tuple[str, bool]:
|
|
271
|
+
"""Write ``problem`` to a temp ``.p`` file and run the prover on it.
|
|
272
|
+
|
|
273
|
+
Returns ``(stdout+stderr, timed_out)``; the temp file is always removed.
|
|
274
|
+
Mirrors ``vampire_entailment._spawn_vampire`` (stderr is folded in here
|
|
275
|
+
because E writes some diagnostics there).
|
|
276
|
+
"""
|
|
277
|
+
with tempfile.NamedTemporaryFile(mode="w", suffix=".p", delete=False,
|
|
278
|
+
encoding="utf-8") as tmp:
|
|
279
|
+
tmp.write(problem)
|
|
280
|
+
path = tmp.name
|
|
281
|
+
try:
|
|
282
|
+
if use_wsl:
|
|
283
|
+
cmd = ["wsl.exe", command, *args, _to_wsl_path(path)]
|
|
284
|
+
else:
|
|
285
|
+
cmd = [command, *args, path]
|
|
286
|
+
result = subprocess.run(cmd, capture_output=True, text=True,
|
|
287
|
+
timeout=timeout_s)
|
|
288
|
+
return (result.stdout or "") + (result.stderr or ""), False
|
|
289
|
+
except subprocess.TimeoutExpired:
|
|
290
|
+
return "", True
|
|
291
|
+
finally:
|
|
292
|
+
try:
|
|
293
|
+
os.unlink(path)
|
|
294
|
+
except OSError:
|
|
295
|
+
pass
|
|
296
|
+
|
|
297
|
+
|
|
298
|
+
def check_entailment_eprover_detailed(premises: List[Node], conclusion: Node,
|
|
299
|
+
*, timeout: int = 30,
|
|
300
|
+
command: Optional[str] = None,
|
|
301
|
+
use_wsl: bool = False,
|
|
302
|
+
tff: Optional[bool] = None,
|
|
303
|
+
sort: Optional[str] = None,
|
|
304
|
+
premise_names: Optional[Sequence[str]] = None) -> dict:
|
|
305
|
+
"""Run E on ``premises ⊨ conclusion``; SZS-status + TSTP-derivation dict.
|
|
306
|
+
|
|
307
|
+
Same result contract as ``check_entailment_vampire_detailed``:
|
|
308
|
+
``{"status", "reason", "szs_status", "derivation", "raw"}``, and the two keys
|
|
309
|
+
``"dialect"`` (``"fof"``, ``"tff"`` or ``"tfa"``: which writer produced the
|
|
310
|
+
problem E was given) and ``"tff_fallback"`` (``None``, or the sentence that says
|
|
311
|
+
``tff=None`` tried the typed writer, was refused, and wrote ``fof`` instead,
|
|
312
|
+
with the typed writer's reason). For a PROVED answer ``"relevant_premises"`` holds
|
|
313
|
+
the sorted 0-based indices of the premises the proof's axiom leaves are, read through
|
|
314
|
+
the problem's name map (``None`` otherwise, and when the proof cannot be read that
|
|
315
|
+
way) and ``"background_used"`` the ``(name, meaning)`` pairs of the background axioms
|
|
316
|
+
the writer added on its own (the non-emptiness of a sort, the membership of a sorted
|
|
317
|
+
constant) that the proof used: background, not premises. ``premise_names`` names the
|
|
318
|
+
premises' ``axiom`` lines instead of ``premise_<i>`` (see
|
|
319
|
+
:func:`atp._tptp_problem.generate_tptp_problem_with_mapping`); E prints them back as
|
|
320
|
+
the names of the proof's leaves. With
|
|
321
|
+
``command=None`` discovery runs (env → PATH → WSL) and a miss raises
|
|
322
|
+
``RuntimeError`` — use :func:`eprover_available` to probe first. ``tff``/
|
|
323
|
+
``sort`` select the TPTP dialect — see :func:`_generate_tptp_problem`. ``sort`` is the typed
|
|
324
|
+
arithmetic text, which E 3.5.1 does not evaluate: a problem with an arithmetic function
|
|
325
|
+
(``+ - * /``) or, under ``sort='real'``, a numeral is refused (see the module docstring).
|
|
326
|
+
Every route (``fof``, ``tff``, ``sort``) hands back a
|
|
327
|
+
:class:`~unicode_logic_kit.atp._tptp_problem.TptpNameMap`, so ``raw`` below
|
|
328
|
+
IS reverse-mapped to original kit-level names on all of them.
|
|
329
|
+
|
|
330
|
+
Every symbol name in ``raw`` and in ``derivation``'s formulas has
|
|
331
|
+
already been translated back from whatever ASCII-safe token the problem
|
|
332
|
+
writer may have substituted (a non-ASCII or digit-leading kit-level name,
|
|
333
|
+
or a function/constant renamed because its word is a predicate's:
|
|
334
|
+
``agent`` -> ``agent_term``) to the ORIGINAL kit-level name — see
|
|
335
|
+
:func:`atp._tptp_problem.generate_tptp_problem_with_mapping` and
|
|
336
|
+
:func:`atp.tptp_tff.generate_tff_problem_with_mapping`. A name E
|
|
337
|
+
introduced itself (a clausification symbol, e.g. ``c_0_7``) was never one
|
|
338
|
+
of ours and is left as E printed it, and so is a SORT name on the ``tff``
|
|
339
|
+
route (the TF0 map covers predicates, functions and constants, not
|
|
340
|
+
sorts). ``derivation`` is CURRENTLY ALWAYS ``None`` on the ``tff`` route,
|
|
341
|
+
even for a genuine proof: :func:`atp.tstp.parse_tstp_derivation` only
|
|
342
|
+
recognises ``fof``/``cnf`` statements (see its own docstring), never the
|
|
343
|
+
``tff``/``tcf`` ones E prints — a real gap (tff/tcf-proof-line reading is
|
|
344
|
+
follow-up work in :mod:`atp.tstp`), not merely "un-reversed".
|
|
345
|
+
|
|
346
|
+
Raises:
|
|
347
|
+
NotImplementedError: a formula is outside the fragment the selected
|
|
348
|
+
route covers, or two distinct predicate (or function/constant,
|
|
349
|
+
or — ``tff`` route — sort) names would fold to the same TPTP
|
|
350
|
+
identifier (see :mod:`atp._tptp_problem`'s and
|
|
351
|
+
:mod:`atp.tptp_tff`'s module docstrings), or — ``sort`` route — the problem
|
|
352
|
+
has an arithmetic function symbol or, under ``sort='real'``, a numeral, which E
|
|
353
|
+
3.5.1 cannot read or reads approximately — surfaced before any subprocess is spawned;
|
|
354
|
+
unlike :class:`_TptpSzsBackend.decide`, this function does NOT
|
|
355
|
+
catch it into an UNKNOWN verdict. With ``tff=True`` that includes
|
|
356
|
+
the typed writer's refusal, a
|
|
357
|
+
:class:`~unicode_logic_kit.atp.tptp_tff.Tf0Refusal` (also a
|
|
358
|
+
``ValueError``).
|
|
359
|
+
RuntimeError: no ``eprover`` binary found (see above).
|
|
360
|
+
"""
|
|
361
|
+
from .tstp import (_premise_use_from_tstp, extract_szs_status, parse_tstp_derivation,
|
|
362
|
+
reverse_map_derivation, szs_to_verdict_fields)
|
|
363
|
+
|
|
364
|
+
refusal = _eprover_arithmetic_refusal(premises, conclusion, sort)
|
|
365
|
+
if refusal is not None:
|
|
366
|
+
raise NotImplementedError(refusal)
|
|
367
|
+
|
|
368
|
+
if command is None:
|
|
369
|
+
found = _discover("eprover", "UFK_EPROVER_CMD")
|
|
370
|
+
if found is None:
|
|
371
|
+
raise RuntimeError(
|
|
372
|
+
"eprover: no binary found (PATH, WSL, $UFK_EPROVER_CMD) — "
|
|
373
|
+
"apt install eprover (Ubuntu 24.04+/Debian) or build from "
|
|
374
|
+
"https://github.com/eprover/eprover")
|
|
375
|
+
command, use_wsl = found
|
|
376
|
+
|
|
377
|
+
if sort is not None:
|
|
378
|
+
problem, name_map = generate_tff_arith_problem(
|
|
379
|
+
premises, conclusion, sort=sort, **names_kwargs(premise_names))
|
|
380
|
+
dialect, fallback_note = "tfa", None
|
|
381
|
+
else:
|
|
382
|
+
built = generate_tptp_problem_for_prover(
|
|
383
|
+
premises, conclusion, tff=tff, fof_writer=generate_tptp_problem_with_mapping,
|
|
384
|
+
premise_names=premise_names)
|
|
385
|
+
problem, name_map = built.text, built.name_map
|
|
386
|
+
dialect, fallback_note = built.dialect, built.fallback_note
|
|
387
|
+
args = ["--auto", "--tstp-format", "--proof-object", "-s",
|
|
388
|
+
f"--cpu-limit={max(1, timeout)}"]
|
|
389
|
+
# Wall clock outlasts E's own cpu budget so E reports ResourceOut itself.
|
|
390
|
+
raw_output, timed_out = _run_tptp_prover(problem, command, args, use_wsl,
|
|
391
|
+
timeout_s=timeout + 10)
|
|
392
|
+
if name_map is None:
|
|
393
|
+
output = raw_output
|
|
394
|
+
else:
|
|
395
|
+
pred_rev, term_rev = name_map.reverse_rendered()
|
|
396
|
+
output = reverse_map_text(raw_output, pred_rev, term_rev)
|
|
397
|
+
if timed_out:
|
|
398
|
+
return {"status": UNKNOWN, "reason": "timeout", "szs_status": None,
|
|
399
|
+
"derivation": None, "raw": output,
|
|
400
|
+
"dialect": dialect, "tff_fallback": fallback_note,
|
|
401
|
+
"relevant_premises": None, "background_used": ()}
|
|
402
|
+
|
|
403
|
+
szs = extract_szs_status(raw_output)
|
|
404
|
+
if szs is None:
|
|
405
|
+
return {"status": ERROR, "reason": "infra", "szs_status": None,
|
|
406
|
+
"derivation": None, "raw": output,
|
|
407
|
+
"dialect": dialect, "tff_fallback": fallback_note,
|
|
408
|
+
"relevant_premises": None, "background_used": ()}
|
|
409
|
+
status, reason = _eprover_verdict_fields(szs, raw_output)
|
|
410
|
+
derivation = None
|
|
411
|
+
relevant: Optional[Tuple[int, ...]] = None
|
|
412
|
+
background_used: Tuple[Tuple[str, str], ...] = ()
|
|
413
|
+
if status == "proved":
|
|
414
|
+
# The axiom leaves by the names the problem gave them, from the text E printed
|
|
415
|
+
# (a premise name is not a symbol, so never the renamed ``output``).
|
|
416
|
+
use = _premise_use_from_tstp(raw_output, len(premises), name_map, eprover=True)
|
|
417
|
+
if use is not None:
|
|
418
|
+
relevant, background_used = use
|
|
419
|
+
try:
|
|
420
|
+
parsed = parse_tstp_derivation(raw_output)
|
|
421
|
+
resolved = parsed if name_map is None else reverse_map_derivation(parsed, name_map)
|
|
422
|
+
# Empty steps -> None, matching check_entailment_vampire_detailed's
|
|
423
|
+
# own contract (this function's docstring promises the same one):
|
|
424
|
+
# "no fof/cnf statement recognised" (always true on the tff route
|
|
425
|
+
# today, see docstring) is "no derivation to report", not "here is
|
|
426
|
+
# an empty derivation".
|
|
427
|
+
derivation = resolved.to_dict() if resolved.steps else None
|
|
428
|
+
except ValueError:
|
|
429
|
+
derivation = None # never silent: backends put this in detail
|
|
430
|
+
return {"status": status, "reason": reason, "szs_status": szs,
|
|
431
|
+
"derivation": derivation, "raw": output,
|
|
432
|
+
"dialect": dialect, "tff_fallback": fallback_note,
|
|
433
|
+
"relevant_premises": relevant, "background_used": background_used}
|
|
434
|
+
|
|
435
|
+
|
|
436
|
+
def eprover_relevant_premises(premises: List[Node], conclusion: Node, *,
|
|
437
|
+
timeout: int = 30, command: Optional[str] = None,
|
|
438
|
+
use_wsl: bool = False,
|
|
439
|
+
tff: Optional[bool] = None,
|
|
440
|
+
sort: Optional[str] = None,
|
|
441
|
+
premise_names: Optional[Sequence[str]] = None
|
|
442
|
+
) -> Optional[Tuple[int, ...]]:
|
|
443
|
+
"""Which of ``premises`` did E's proof of ``premises ⊨ conclusion``
|
|
444
|
+
actually rest on?
|
|
445
|
+
|
|
446
|
+
Runs E exactly ONCE, through :func:`check_entailment_eprover_detailed`'s
|
|
447
|
+
own subprocess plumbing (no second prover invocation) — and, only when
|
|
448
|
+
the resulting verdict is PROVED, walks the TSTP derivation backward via
|
|
449
|
+
:func:`atp.tstp.relevant_premises_from_tstp`, which descends into E's
|
|
450
|
+
own nested administrative ``inference(...)`` steps (see that function's
|
|
451
|
+
module comment) down to every reachable axiom leaf; E names those leaves
|
|
452
|
+
after the names the problem's writer gave its lines (``premise_<i>``, or
|
|
453
|
+
``premise_names``; E prints a quoted name back as it was written, except for
|
|
454
|
+
an apostrophe, which :func:`atp.tstp.relevant_premises_from_tstp` allows for), so
|
|
455
|
+
index recovery goes through the writer's record of them. A leaf that is a
|
|
456
|
+
background axiom of a sorted problem (``nonempty_sort_<i>``, ``sort_member_<i>``) is
|
|
457
|
+
not one of ``premises``: it is left out of the result, and
|
|
458
|
+
:func:`check_entailment_eprover_detailed` says which were used.
|
|
459
|
+
|
|
460
|
+
Args:
|
|
461
|
+
premises: candidate premises.
|
|
462
|
+
conclusion: the goal.
|
|
463
|
+
timeout: seconds — forwarded to
|
|
464
|
+
:func:`check_entailment_eprover_detailed` unchanged.
|
|
465
|
+
command: an explicit ``eprover`` command/path — see
|
|
466
|
+
:func:`check_entailment_eprover_detailed`; when omitted,
|
|
467
|
+
discovery runs the same way, and a miss is reported as ``None``
|
|
468
|
+
here (NOT raised — unlike ``check_entailment_eprover_detailed``,
|
|
469
|
+
this is a "can you tell me" query, so a missing binary is just
|
|
470
|
+
"no answer available", exactly like an unproved verdict).
|
|
471
|
+
|
|
472
|
+
Returns:
|
|
473
|
+
A sorted tuple of 0-based indices into ``premises``, or ``None``
|
|
474
|
+
when: E is not reachable; the verdict is not PROVED (there is no
|
|
475
|
+
"premises used" answer for a non-theorem); or the derivation walk
|
|
476
|
+
itself could not be trusted (see
|
|
477
|
+
:func:`atp.tstp.relevant_premises_from_tstp`'s ``Returns``) — always
|
|
478
|
+
the honest "don't know", never a guessed/under-approximated subset.
|
|
479
|
+
|
|
480
|
+
Raises:
|
|
481
|
+
NotImplementedError: a formula is outside the classical FOL fragment
|
|
482
|
+
``Node.to_tptp`` covers, or a cross-formula symbol-collision —
|
|
483
|
+
same contract as :func:`check_entailment_eprover_detailed`,
|
|
484
|
+
surfaced before any subprocess is spawned.
|
|
485
|
+
"""
|
|
486
|
+
try:
|
|
487
|
+
result = check_entailment_eprover_detailed(
|
|
488
|
+
premises, conclusion, timeout=timeout, command=command, use_wsl=use_wsl,
|
|
489
|
+
tff=tff, sort=sort, premise_names=premise_names)
|
|
490
|
+
except RuntimeError:
|
|
491
|
+
return None # no eprover binary -- an honest "no answer", not an error
|
|
492
|
+
if result["status"] != "proved":
|
|
493
|
+
return None
|
|
494
|
+
return result["relevant_premises"]
|
|
495
|
+
|
|
496
|
+
|
|
497
|
+
class _TptpSzsBackend(ProverBackend):
|
|
498
|
+
"""Shared decide() skeleton for the two SZS-speaking TPTP provers."""
|
|
499
|
+
|
|
500
|
+
logics = frozenset({"fol", "arith"})
|
|
501
|
+
external = True
|
|
502
|
+
|
|
503
|
+
_env_var: str = ""
|
|
504
|
+
_binary: str = ""
|
|
505
|
+
|
|
506
|
+
def _args(self, timeout_s: int) -> List[str]: # pragma: no cover
|
|
507
|
+
raise NotImplementedError
|
|
508
|
+
|
|
509
|
+
def _verdict_fields(self, szs: str, raw_output: str) -> Tuple[str, Optional[str]]:
|
|
510
|
+
"""``(status, reason)`` of this prover's SZS status; E overrides it."""
|
|
511
|
+
from .tstp import szs_to_verdict_fields
|
|
512
|
+
return szs_to_verdict_fields(szs, query="conjecture")
|
|
513
|
+
|
|
514
|
+
#: Whether the prover prints a premise name back the way E does (see
|
|
515
|
+
#: :func:`atp.tstp.relevant_premises_from_tstp`'s ``eprover``).
|
|
516
|
+
_prints_names_like_eprover: bool = False
|
|
517
|
+
|
|
518
|
+
def _background_note(self, raw_output: str, n_premises: int, name_map) -> str:
|
|
519
|
+
"""The sentence for a proved verdict's ``detail`` that names the background axioms
|
|
520
|
+
of a sorted problem the proof used (the non-emptiness of a sort, the membership
|
|
521
|
+
of a sorted constant), or ``""`` when it used none or its leaves cannot be read
|
|
522
|
+
through the problem's name map."""
|
|
523
|
+
from .tstp import _premise_use_from_tstp
|
|
524
|
+
use = _premise_use_from_tstp(raw_output, n_premises, name_map,
|
|
525
|
+
eprover=self._prints_names_like_eprover)
|
|
526
|
+
if use is None or not use[1]:
|
|
527
|
+
return ""
|
|
528
|
+
facts = ", ".join(f"{name} ({meaning})" if meaning else name for name, meaning in use[1])
|
|
529
|
+
return (f"; the proof used the background facts of the sorted reading, which are "
|
|
530
|
+
f"not premises: {facts}")
|
|
531
|
+
|
|
532
|
+
def _arithmetic_refusal(self, premises: Sequence[Node], formula: Optional[Node],
|
|
533
|
+
sort: Optional[str]) -> Optional[str]:
|
|
534
|
+
"""Why this prover is not given the typed arithmetic text of the problem (``sort=``), or
|
|
535
|
+
``None``. A prover whose reading of that text has not been measured is never refused."""
|
|
536
|
+
return None
|
|
537
|
+
|
|
538
|
+
def available(self) -> bool:
|
|
539
|
+
return _discover(self._binary, self._env_var) is not None
|
|
540
|
+
|
|
541
|
+
def solver_version(self) -> Optional[str]:
|
|
542
|
+
"""The prover's ``--version`` banner (first line), for whichever
|
|
543
|
+
binary discovery (env override, PATH, WSL — see :func:`_discover`)
|
|
544
|
+
resolves to right now — the same default :meth:`available` uses.
|
|
545
|
+
Memoized per ``(binary, use_wsl)`` for the life of the process via
|
|
546
|
+
:func:`~unicode_logic_kit.atp.protocol._binary_version` (a cache kept
|
|
547
|
+
separate from :data:`_DISCOVERY_CACHE` — see that function's own
|
|
548
|
+
docstring for why). ``None`` when no binary is currently
|
|
549
|
+
discoverable.
|
|
550
|
+
"""
|
|
551
|
+
found = _discover(self._binary, self._env_var)
|
|
552
|
+
if found is None:
|
|
553
|
+
return None
|
|
554
|
+
command, use_wsl = found
|
|
555
|
+
return _binary_version(command, use_wsl)
|
|
556
|
+
|
|
557
|
+
def decide(self, formula: Node, premises: Sequence[Node] = (),
|
|
558
|
+
timeout: int = 10000, **options) -> Verdict:
|
|
559
|
+
"""``tff``/``sort`` (options forwarded via ``**options``, mirroring
|
|
560
|
+
how other backends read e.g. ``frame=`` — see
|
|
561
|
+
``ProverBackend.decide``'s own docstring): which TPTP dialect to
|
|
562
|
+
export as. ``sort`` (``None`` by default) opts into the
|
|
563
|
+
single-numeric-sort typed arithmetic route
|
|
564
|
+
(:func:`atp._tff_problem.generate_tff_arith_problem`, ``'real'`` or
|
|
565
|
+
``'int'``) UNCONDITIONALLY when given, taking priority over ``tff``.
|
|
566
|
+
That is the text Vampire reads with its own arithmetic, which the
|
|
567
|
+
untyped ``fof``/many-sorted-``tff`` routes cannot give it (see
|
|
568
|
+
:mod:`atp._tff_problem`'s module docstring); E 3.5.1 does not evaluate
|
|
569
|
+
it, so ``EProverBackend`` answers ``unknown`` / ``"unsupported"``, naming the operator or
|
|
570
|
+
the numeral, for a problem with an arithmetic function symbol (E stops with a type
|
|
571
|
+
error on one) or, under ``sort='real'``, a numeral (E reads a ``$real`` literal
|
|
572
|
+
approximately) — see the module docstring; the other problems it answers soundly, and
|
|
573
|
+
``unknown`` for what needs arithmetic. Zipperposition is asked as it always was (its
|
|
574
|
+
reading of the typed text is not measured). With ``sort=None``,
|
|
575
|
+
``tff`` selects as before: ``None`` (the default, i.e. omitted)
|
|
576
|
+
tries the native many-sorted typed ``tff`` route whenever
|
|
577
|
+
``premises``/``formula`` use a sort — see
|
|
578
|
+
:func:`_generate_tptp_problem` — and writes ``fof`` instead when the
|
|
579
|
+
typed writer refuses the problem (its text would ask another question
|
|
580
|
+
than the kit's: :class:`~unicode_logic_kit.atp.tptp_tff.Tf0Refusal`; or it holds a
|
|
581
|
+
node the typed writer does not cover and the ``fof`` writer does), saying
|
|
582
|
+
so and why in the verdict's ``detail``; ``True``/``False`` force one route
|
|
583
|
+
(with ``True`` the refusal is the verdict, UNKNOWN / ``"unsupported"``
|
|
584
|
+
with the writer's message).
|
|
585
|
+
Every route hands back a
|
|
586
|
+
:class:`~unicode_logic_kit.atp._tptp_problem.TptpNameMap`, so
|
|
587
|
+
``proof``/``detail`` are reverse-mapped to original kit-level names
|
|
588
|
+
on the ``fof``, the many-sorted ``tff`` and the ``sort`` route alike
|
|
589
|
+
(a SORT name on the ``tff`` route is not in the TF0 map and stays as
|
|
590
|
+
the prover printed it).
|
|
591
|
+
"""
|
|
592
|
+
from .tstp import (extract_szs_status, parse_tstp_derivation,
|
|
593
|
+
reverse_map_derivation, szs_to_verdict_fields)
|
|
594
|
+
|
|
595
|
+
tff = options.pop("tff", None)
|
|
596
|
+
sort = options.pop("sort", None)
|
|
597
|
+
premise_names = options.pop("premise_names", None)
|
|
598
|
+
|
|
599
|
+
found = _discover(self._binary, self._env_var)
|
|
600
|
+
if found is None:
|
|
601
|
+
from .protocol import BackendUnavailable
|
|
602
|
+
raise BackendUnavailable(
|
|
603
|
+
f"{self.name}: no binary found — see "
|
|
604
|
+
"unicode_logic_kit/atp/eprover_backend.py for the per-platform "
|
|
605
|
+
f"acquisition paths, or set ${self._env_var}.")
|
|
606
|
+
command, use_wsl = found
|
|
607
|
+
# Memoized (see _binary_version): the first EProverBackend/
|
|
608
|
+
# ZipperpositionBackend decide() call for this (command, use_wsl)
|
|
609
|
+
# pays one subprocess spawn, every later one for the same pair is
|
|
610
|
+
# free — including across the two backend CLASSES, since the cache
|
|
611
|
+
# key is the resolved binary path, not the registry name.
|
|
612
|
+
solver_version = _binary_version(command, use_wsl)
|
|
613
|
+
|
|
614
|
+
try:
|
|
615
|
+
if sort is not None:
|
|
616
|
+
refusal = self._arithmetic_refusal(premises, formula, sort)
|
|
617
|
+
if refusal is not None:
|
|
618
|
+
raise NotImplementedError(refusal)
|
|
619
|
+
problem, name_map = generate_tff_arith_problem(
|
|
620
|
+
list(premises), formula, sort=sort, **names_kwargs(premise_names))
|
|
621
|
+
fallback_note = None
|
|
622
|
+
else:
|
|
623
|
+
built = generate_tptp_problem_for_prover(
|
|
624
|
+
list(premises), formula, tff=tff,
|
|
625
|
+
fof_writer=generate_tptp_problem_with_mapping,
|
|
626
|
+
premise_names=premise_names)
|
|
627
|
+
problem, name_map, fallback_note = built.text, built.name_map, built.fallback_note
|
|
628
|
+
except NotImplementedError as exc:
|
|
629
|
+
# tff=True: the typed writer's own refusal (Tf0Refusal is a
|
|
630
|
+
# NotImplementedError too) is the verdict, with its message.
|
|
631
|
+
return Verdict(UNKNOWN, self.name, reason="unsupported",
|
|
632
|
+
solver_version=solver_version, detail=str(exc))
|
|
633
|
+
if name_map is not None:
|
|
634
|
+
pred_rev, term_rev = name_map.reverse_rendered()
|
|
635
|
+
|
|
636
|
+
def noted(text: str) -> str:
|
|
637
|
+
# tff=None tried the typed writer for a sorted problem and it refused:
|
|
638
|
+
# the answer is that of the fof text, and the detail says so and why.
|
|
639
|
+
return text if fallback_note is None else f"{text}; {fallback_note}"
|
|
640
|
+
|
|
641
|
+
timeout_s = max(1, timeout // 1000)
|
|
642
|
+
start = time.perf_counter()
|
|
643
|
+
try:
|
|
644
|
+
raw_output, timed_out = _run_tptp_prover(
|
|
645
|
+
problem, command, self._args(timeout_s), use_wsl,
|
|
646
|
+
timeout_s=timeout_s + 10)
|
|
647
|
+
except (OSError, RuntimeError) as exc:
|
|
648
|
+
return Verdict(ERROR, self.name, reason="infra",
|
|
649
|
+
solver_version=solver_version,
|
|
650
|
+
detail=noted(f"{type(exc).__name__}: {exc}"))
|
|
651
|
+
elapsed = time.perf_counter() - start
|
|
652
|
+
|
|
653
|
+
if timed_out:
|
|
654
|
+
return Verdict(UNKNOWN, self.name, reason="timeout",
|
|
655
|
+
wall_time=elapsed, solver_version=solver_version,
|
|
656
|
+
detail=noted(
|
|
657
|
+
f"{self.name} had not stopped when the budget of this call "
|
|
658
|
+
f"({timeout_s} s) plus 10 s of grace had passed, so the kit "
|
|
659
|
+
"stopped it before it printed a verdict"))
|
|
660
|
+
szs = extract_szs_status(raw_output)
|
|
661
|
+
if szs is None:
|
|
662
|
+
# The prover read the problem and refused it (E: a parse error on
|
|
663
|
+
# stderr, exit 3), or died: a failure that carries the prover's
|
|
664
|
+
# own text, never an "unknown" that reads like a timeout.
|
|
665
|
+
shown = raw_output
|
|
666
|
+
if name_map is not None:
|
|
667
|
+
shown = reverse_map_text(shown, pred_rev, term_rev)
|
|
668
|
+
return Verdict(ERROR, self.name, reason="infra",
|
|
669
|
+
wall_time=elapsed, solver_version=solver_version,
|
|
670
|
+
detail=noted(_rejection_detail(
|
|
671
|
+
self.name, shown, "no SZS status line in its output")))
|
|
672
|
+
status, reason = self._verdict_fields(szs, raw_output)
|
|
673
|
+
proof = None
|
|
674
|
+
detail = f"SZS status {szs}"
|
|
675
|
+
if szs == "ResourceOut" and reason == "timeout":
|
|
676
|
+
detail += (f"; {self.name} stopped itself at its --cpu-limit of {timeout_s} s, "
|
|
677
|
+
"which is the budget of this call")
|
|
678
|
+
if status == ERROR:
|
|
679
|
+
# An SZS status that itself names a failure (Error, InputError,
|
|
680
|
+
# SyntaxError): quote what the prover said about it.
|
|
681
|
+
shown = raw_output
|
|
682
|
+
if name_map is not None:
|
|
683
|
+
shown = reverse_map_text(shown, pred_rev, term_rev)
|
|
684
|
+
detail = _rejection_detail(self.name, shown, f"SZS status {szs}")
|
|
685
|
+
if status == "proved":
|
|
686
|
+
try:
|
|
687
|
+
parsed = parse_tstp_derivation(raw_output)
|
|
688
|
+
if name_map is not None:
|
|
689
|
+
parsed = reverse_map_derivation(parsed, name_map)
|
|
690
|
+
# A status line without TSTP steps (Zipperposition's default
|
|
691
|
+
# output) is a verdict without a certificate — proof stays
|
|
692
|
+
# None and the detail says so, the Theorem status stands.
|
|
693
|
+
proof = parsed.to_dict() if parsed.steps else None
|
|
694
|
+
if proof is None:
|
|
695
|
+
detail += "; no TSTP derivation in output"
|
|
696
|
+
except ValueError as exc:
|
|
697
|
+
detail += f"; TSTP derivation unparseable: {exc}"
|
|
698
|
+
detail += self._background_note(raw_output, len(premises), name_map)
|
|
699
|
+
return Verdict(status, self.name, reason=reason, szs_status=szs,
|
|
700
|
+
wall_time=elapsed, solver_version=solver_version,
|
|
701
|
+
proof=proof, detail=noted(detail))
|
|
702
|
+
|
|
703
|
+
|
|
704
|
+
class EProverBackend(_TptpSzsBackend):
|
|
705
|
+
"""E (superposition, https://eprover.org) — registry name ``"eprover"``."""
|
|
706
|
+
|
|
707
|
+
name = "eprover"
|
|
708
|
+
_env_var = "UFK_EPROVER_CMD"
|
|
709
|
+
_binary = "eprover"
|
|
710
|
+
_prints_names_like_eprover = True
|
|
711
|
+
|
|
712
|
+
def _arithmetic_refusal(self, premises: Sequence[Node], formula: Optional[Node],
|
|
713
|
+
sort: Optional[str]) -> Optional[str]:
|
|
714
|
+
return _eprover_arithmetic_refusal(premises, formula, sort)
|
|
715
|
+
|
|
716
|
+
def _args(self, timeout_s: int) -> List[str]:
|
|
717
|
+
return ["--auto", "--tstp-format", "--proof-object", "-s",
|
|
718
|
+
f"--cpu-limit={timeout_s}"]
|
|
719
|
+
|
|
720
|
+
def _verdict_fields(self, szs: str, raw_output: str) -> Tuple[str, Optional[str]]:
|
|
721
|
+
return _eprover_verdict_fields(szs, raw_output)
|
|
722
|
+
|
|
723
|
+
|
|
724
|
+
class ZipperpositionBackend(_TptpSzsBackend):
|
|
725
|
+
"""Zipperposition (OCaml superposition) — registry name ``"zipperposition"``."""
|
|
726
|
+
|
|
727
|
+
name = "zipperposition"
|
|
728
|
+
_env_var = "UFK_ZIPPERPOSITION_CMD"
|
|
729
|
+
_binary = "zipperposition"
|
|
730
|
+
|
|
731
|
+
def _args(self, timeout_s: int) -> List[str]:
|
|
732
|
+
return ["--timeout", str(timeout_s)]
|