unicode-logic-kit 0.31.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- unicode_logic_kit/__init__.py +385 -0
- unicode_logic_kit/__main__.py +520 -0
- unicode_logic_kit/_deadline.py +219 -0
- unicode_logic_kit/ace/__init__.py +126 -0
- unicode_logic_kit/ace/_align.py +135 -0
- unicode_logic_kit/ace/chem_lexicon.py +128 -0
- unicode_logic_kit/ace/drs_reader.py +570 -0
- unicode_logic_kit/ace/mapping.py +666 -0
- unicode_logic_kit/ace/reverse_modal.py +138 -0
- unicode_logic_kit/ace/runner.py +551 -0
- unicode_logic_kit/ace/translate.py +452 -0
- unicode_logic_kit/ace/verbalize.py +1070 -0
- unicode_logic_kit/api.py +1284 -0
- unicode_logic_kit/atp/__init__.py +177 -0
- unicode_logic_kit/atp/_ascii_names.py +113 -0
- unicode_logic_kit/atp/_html.py +72 -0
- unicode_logic_kit/atp/_substructural_input.py +228 -0
- unicode_logic_kit/atp/_tff_problem.py +715 -0
- unicode_logic_kit/atp/_tptp_problem.py +1111 -0
- unicode_logic_kit/atp/_writer_support.py +289 -0
- unicode_logic_kit/atp/clingo_backend.py +1180 -0
- unicode_logic_kit/atp/cvc5_backend.py +1385 -0
- unicode_logic_kit/atp/eprover_backend.py +732 -0
- unicode_logic_kit/atp/finite_domain.py +1055 -0
- unicode_logic_kit/atp/fitch.py +1547 -0
- unicode_logic_kit/atp/fitch_search.py +551 -0
- unicode_logic_kit/atp/hets_backend.py +339 -0
- unicode_logic_kit/atp/hybrid_down.py +120 -0
- unicode_logic_kit/atp/incremental.py +250 -0
- unicode_logic_kit/atp/kripke_enum.py +741 -0
- unicode_logic_kit/atp/lambek.py +436 -0
- unicode_logic_kit/atp/leo3_backend.py +332 -0
- unicode_logic_kit/atp/linear.py +738 -0
- unicode_logic_kit/atp/lj.py +705 -0
- unicode_logic_kit/atp/logic_backends.py +566 -0
- unicode_logic_kit/atp/ltl_tableau.py +1084 -0
- unicode_logic_kit/atp/minizinc_backend.py +1402 -0
- unicode_logic_kit/atp/modal_tableau.py +1382 -0
- unicode_logic_kit/atp/nanocop_backend.py +410 -0
- unicode_logic_kit/atp/portfolio.py +489 -0
- unicode_logic_kit/atp/protocol.py +1803 -0
- unicode_logic_kit/atp/prover9_entailment.py +1153 -0
- unicode_logic_kit/atp/resolution.py +1376 -0
- unicode_logic_kit/atp/resolution_check.py +1114 -0
- unicode_logic_kit/atp/sequent.py +1050 -0
- unicode_logic_kit/atp/tableau.py +921 -0
- unicode_logic_kit/atp/tableau_check.py +543 -0
- unicode_logic_kit/atp/tptp_ncl.py +811 -0
- unicode_logic_kit/atp/tptp_tff.py +1546 -0
- unicode_logic_kit/atp/tstp.py +1333 -0
- unicode_logic_kit/atp/tstp_check.py +1096 -0
- unicode_logic_kit/atp/twee_backend.py +236 -0
- unicode_logic_kit/atp/twee_check.py +711 -0
- unicode_logic_kit/atp/twee_entailment.py +953 -0
- unicode_logic_kit/atp/vampire_entailment.py +540 -0
- unicode_logic_kit/atp/z3_arith.py +470 -0
- unicode_logic_kit/atp/z3_equivalence.py +36 -0
- unicode_logic_kit/atp/z3_fuzzy.py +362 -0
- unicode_logic_kit/atp/z3_input.py +500 -0
- unicode_logic_kit/atp/z3_models.py +208 -0
- unicode_logic_kit/chem/__init__.py +88 -0
- unicode_logic_kit/chem/_naming.py +284 -0
- unicode_logic_kit/chem/cache.py +185 -0
- unicode_logic_kit/chem/interop.py +244 -0
- unicode_logic_kit/chem/mol.py +525 -0
- unicode_logic_kit/chem/signature.py +112 -0
- unicode_logic_kit/comorphism.py +497 -0
- unicode_logic_kit/dl/__init__.py +384 -0
- unicode_logic_kit/dl/classification.py +227 -0
- unicode_logic_kit/dl/concepts.py +632 -0
- unicode_logic_kit/dl/datatypes.py +818 -0
- unicode_logic_kit/dl/owl_functional.py +2433 -0
- unicode_logic_kit/dl/owl_manchester.py +1637 -0
- unicode_logic_kit/dl/owl_reasoner.py +790 -0
- unicode_logic_kit/dl/parser.py +391 -0
- unicode_logic_kit/dl/tableau.py +4048 -0
- unicode_logic_kit/dl/translate.py +2704 -0
- unicode_logic_kit/drt/__init__.py +94 -0
- unicode_logic_kit/drt/export.py +179 -0
- unicode_logic_kit/drt/nodes.py +506 -0
- unicode_logic_kit/drt/parser.py +965 -0
- unicode_logic_kit/drt/resolve.py +195 -0
- unicode_logic_kit/drt/reverse.py +175 -0
- unicode_logic_kit/eval/__init__.py +106 -0
- unicode_logic_kit/eval/batch.py +382 -0
- unicode_logic_kit/eval/canonical.py +663 -0
- unicode_logic_kit/eval/chem_batch.py +606 -0
- unicode_logic_kit/eval/converses.py +200 -0
- unicode_logic_kit/eval/datasets/__init__.py +136 -0
- unicode_logic_kit/eval/datasets/_base.py +263 -0
- unicode_logic_kit/eval/datasets/_proofwriter_proof.py +422 -0
- unicode_logic_kit/eval/datasets/c3po.py +678 -0
- unicode_logic_kit/eval/datasets/folio.py +158 -0
- unicode_logic_kit/eval/datasets/fracas.py +418 -0
- unicode_logic_kit/eval/datasets/groves.py +191 -0
- unicode_logic_kit/eval/datasets/logicbench.py +467 -0
- unicode_logic_kit/eval/datasets/logicnli.py +303 -0
- unicode_logic_kit/eval/datasets/malls.py +133 -0
- unicode_logic_kit/eval/datasets/pfolio.py +594 -0
- unicode_logic_kit/eval/datasets/pmb.py +242 -0
- unicode_logic_kit/eval/datasets/prontoqa.py +611 -0
- unicode_logic_kit/eval/datasets/proofwriter.py +1431 -0
- unicode_logic_kit/eval/datasets/proverqa.py +674 -0
- unicode_logic_kit/eval/datasets/willow.py +478 -0
- unicode_logic_kit/eval/equivalence.py +466 -0
- unicode_logic_kit/eval/exercise_gen.py +533 -0
- unicode_logic_kit/eval/explain.py +791 -0
- unicode_logic_kit/eval/generality.py +750 -0
- unicode_logic_kit/eval/metric_hf.py +458 -0
- unicode_logic_kit/eval/predicate_match.py +343 -0
- unicode_logic_kit/eval/theory_check.py +1170 -0
- unicode_logic_kit/eval/validate.py +306 -0
- unicode_logic_kit/fol/__init__.py +177 -0
- unicode_logic_kit/fol/_atom_keys.py +510 -0
- unicode_logic_kit/fol/_fol_nodes.py +3586 -0
- unicode_logic_kit/fol/_free_parameters.py +105 -0
- unicode_logic_kit/fol/_ho_nodes.py +448 -0
- unicode_logic_kit/fol/_hybrid_nodes.py +308 -0
- unicode_logic_kit/fol/_identifiers.py +1091 -0
- unicode_logic_kit/fol/_lambek_nodes.py +112 -0
- unicode_logic_kit/fol/_linear_nodes.py +352 -0
- unicode_logic_kit/fol/_modal_nodes.py +1467 -0
- unicode_logic_kit/fol/_msfl_nodes.py +2196 -0
- unicode_logic_kit/fol/_numeral_symbols.py +231 -0
- unicode_logic_kit/fol/_so_nodes.py +200 -0
- unicode_logic_kit/fol/_symbol_names.py +81 -0
- unicode_logic_kit/fol/_team_nodes.py +181 -0
- unicode_logic_kit/fol/_tptp_symbols.py +551 -0
- unicode_logic_kit/fol/_truth_constants.py +117 -0
- unicode_logic_kit/fol/casl_export.py +1135 -0
- unicode_logic_kit/fol/casl_import.py +929 -0
- unicode_logic_kit/fol/derivation.py +367 -0
- unicode_logic_kit/fol/dialect_detect.py +70 -0
- unicode_logic_kit/fol/dialect_repair.py +537 -0
- unicode_logic_kit/fol/frames.py +637 -0
- unicode_logic_kit/fol/grammars/terminals.lark +31 -0
- unicode_logic_kit/fol/lambda_tools.py +297 -0
- unicode_logic_kit/fol/latex_input.py +429 -0
- unicode_logic_kit/fol/modal_translation.py +944 -0
- unicode_logic_kit/fol/msflparser.py +1033 -0
- unicode_logic_kit/fol/naming.py +422 -0
- unicode_logic_kit/fol/nodes.py +241 -0
- unicode_logic_kit/fol/normalforms.py +492 -0
- unicode_logic_kit/fol/pal.py +287 -0
- unicode_logic_kit/fol/prolog_export.py +566 -0
- unicode_logic_kit/fol/prolog_input.py +505 -0
- unicode_logic_kit/fol/prover9_input.py +1325 -0
- unicode_logic_kit/fol/qml.py +1760 -0
- unicode_logic_kit/fol/qmltp_input.py +525 -0
- unicode_logic_kit/fol/sanitize.py +221 -0
- unicode_logic_kit/fol/serialize.py +79 -0
- unicode_logic_kit/fol/signature.py +1290 -0
- unicode_logic_kit/fol/simplify_check.py +544 -0
- unicode_logic_kit/fol/spans.py +594 -0
- unicode_logic_kit/fol/tptp_input.py +1503 -0
- unicode_logic_kit/fol/tptp_repair.py +941 -0
- unicode_logic_kit/fol/unification.py +157 -0
- unicode_logic_kit/fol/verbalize.py +263 -0
- unicode_logic_kit/hets/__init__.py +163 -0
- unicode_logic_kit/hets/bridge.py +142 -0
- unicode_logic_kit/hets/client.py +748 -0
- unicode_logic_kit/hets/docker.py +420 -0
- unicode_logic_kit/hets/dol.py +712 -0
- unicode_logic_kit/hets/haskell_json.py +355 -0
- unicode_logic_kit/hets/owl_backend.py +794 -0
- unicode_logic_kit/hets/owl_cli.py +598 -0
- unicode_logic_kit/hets/symbols.py +512 -0
- unicode_logic_kit/hol/__init__.py +140 -0
- unicode_logic_kit/hol/_ho_common.py +323 -0
- unicode_logic_kit/hol/_isabelle_binders.py +125 -0
- unicode_logic_kit/hol/classical.py +812 -0
- unicode_logic_kit/hol/deepshallow/__init__.py +45 -0
- unicode_logic_kit/hol/deepshallow/_common.py +177 -0
- unicode_logic_kit/hol/deepshallow/conditional.py +225 -0
- unicode_logic_kit/hol/deepshallow/intuitionistic.py +181 -0
- unicode_logic_kit/hol/deepshallow/modal.py +217 -0
- unicode_logic_kit/hol/deepshallow/qml.py +406 -0
- unicode_logic_kit/hol/deepshallow/relevant.py +206 -0
- unicode_logic_kit/hol/free.py +753 -0
- unicode_logic_kit/hol/goedel.py +336 -0
- unicode_logic_kit/hol/ho_modal.py +1743 -0
- unicode_logic_kit/hol/intuitionistic.py +403 -0
- unicode_logic_kit/hol/isabelle_conditional.py +593 -0
- unicode_logic_kit/hol/isabelle_modal.py +1908 -0
- unicode_logic_kit/hol/isabelle_relevant.py +412 -0
- unicode_logic_kit/hol/isabelle_runner.py +1147 -0
- unicode_logic_kit/hol/isabelle_substructural.py +884 -0
- unicode_logic_kit/hol/lean.py +1018 -0
- unicode_logic_kit/hol/manyvalued.py +921 -0
- unicode_logic_kit/hol/secondorder.py +687 -0
- unicode_logic_kit/hol/thf_modal.py +941 -0
- unicode_logic_kit/hol/thirdorder.py +397 -0
- unicode_logic_kit/ilp/__init__.py +89 -0
- unicode_logic_kit/ilp/readback.py +389 -0
- unicode_logic_kit/ilp/separation.py +153 -0
- unicode_logic_kit/ilp/task.py +730 -0
- unicode_logic_kit/logic.py +163 -0
- unicode_logic_kit/mcp/__init__.py +28 -0
- unicode_logic_kit/mcp/__main__.py +5 -0
- unicode_logic_kit/mcp/chem_tools.py +1031 -0
- unicode_logic_kit/mcp/server.py +2453 -0
- unicode_logic_kit/mcp/syntax_spec.py +681 -0
- unicode_logic_kit/prob/__init__.py +53 -0
- unicode_logic_kit/prob/_bdd.py +225 -0
- unicode_logic_kit/prob/_column_gen.py +668 -0
- unicode_logic_kit/prob/distribution.py +686 -0
- unicode_logic_kit/prob/nilsson.py +470 -0
- unicode_logic_kit/py.typed +0 -0
- unicode_logic_kit/semantics/__init__.py +137 -0
- unicode_logic_kit/semantics/_modal_reject.py +156 -0
- unicode_logic_kit/semantics/action_models.py +466 -0
- unicode_logic_kit/semantics/asp_models.py +1200 -0
- unicode_logic_kit/semantics/conditional.py +580 -0
- unicode_logic_kit/semantics/dynamic_epistemic.py +95 -0
- unicode_logic_kit/semantics/free_logic.py +913 -0
- unicode_logic_kit/semantics/fuzzy.py +384 -0
- unicode_logic_kit/semantics/fuzzy_kripke.py +442 -0
- unicode_logic_kit/semantics/intuitionistic.py +581 -0
- unicode_logic_kit/semantics/kripke.py +1139 -0
- unicode_logic_kit/semantics/manyvalued.py +580 -0
- unicode_logic_kit/semantics/matrix.py +342 -0
- unicode_logic_kit/semantics/model_eval.py +1135 -0
- unicode_logic_kit/semantics/modelfinder.py +1036 -0
- unicode_logic_kit/semantics/nonmonotonic.py +372 -0
- unicode_logic_kit/semantics/relevant.py +331 -0
- unicode_logic_kit/semantics/secondorder.py +657 -0
- unicode_logic_kit/semantics/structures.py +352 -0
- unicode_logic_kit/semantics/tarski.py +975 -0
- unicode_logic_kit/semantics/team.py +315 -0
- unicode_logic_kit/semantics/team_translation.py +416 -0
- unicode_logic_kit/semantics/thirdorder.py +358 -0
- unicode_logic_kit/semantics/tnorm.py +85 -0
- unicode_logic_kit/semantics/truthtable.py +201 -0
- unicode_logic_kit-0.31.0.dist-info/METADATA +333 -0
- unicode_logic_kit-0.31.0.dist-info/RECORD +237 -0
- unicode_logic_kit-0.31.0.dist-info/WHEEL +4 -0
- unicode_logic_kit-0.31.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,1803 @@
|
|
|
1
|
+
"""Uniform prover protocol: one Verdict type over every decision route.
|
|
2
|
+
|
|
3
|
+
The kit decides validity/entailment through many routes — its OWN calculi and
|
|
4
|
+
semantic model searches (Z3-free tableau, resolution, finite model finder,
|
|
5
|
+
modal labelled tableau, the QML embedding) and EXTERNAL provers (Z3, Isabelle,
|
|
6
|
+
Prover9, Vampire). Each grew its own return convention (bare bool,
|
|
7
|
+
``"valid"/"invalid"/"unknown"`` strings, ``ModalVerdict``/``FolVerdict``).
|
|
8
|
+
This module adds the layer a pipeline needs on top, without touching any
|
|
9
|
+
existing signature:
|
|
10
|
+
|
|
11
|
+
* :class:`Verdict` — the one result type: a semantic ``status`` (proved /
|
|
12
|
+
refuted / unknown / error), a ``reason`` axis that distinguishes budget
|
|
13
|
+
exhaustion from honest incompleteness from timeouts, the SZS status string,
|
|
14
|
+
provenance (which backend, how long), and JSON-able witnesses.
|
|
15
|
+
* :class:`ProverBackend` — the adapter contract (``available()`` +
|
|
16
|
+
``decide()``), with the kit's internal calculi and semantic searches as
|
|
17
|
+
first-class backends alongside the external provers.
|
|
18
|
+
* a registry (:func:`register_backend`, :func:`get_backend`,
|
|
19
|
+
:func:`available_backends`, :func:`default_chain`) that the ``prove()``
|
|
20
|
+
facade in :mod:`unicode_logic_kit.api` dispatches over.
|
|
21
|
+
|
|
22
|
+
Error contract: requesting an UNKNOWN backend name raises ``ValueError``;
|
|
23
|
+
requesting a known backend whose prerequisites are missing (no binary, no
|
|
24
|
+
install) raises :class:`BackendUnavailable` — never a silent skip, because a
|
|
25
|
+
silently skipped backend makes evaluation results irreproducible.
|
|
26
|
+
|
|
27
|
+
The status/reason split is deliberate: ``status`` answers "what do we know
|
|
28
|
+
about the formula" (only four values, easy to branch on), ``reason`` answers
|
|
29
|
+
"why do we not know more" (``bound_hit`` ≠ ``timeout`` ≠ ``incomplete`` ≠
|
|
30
|
+
``unsupported`` — a benchmark table must not conflate them).
|
|
31
|
+
"""
|
|
32
|
+
|
|
33
|
+
import re
|
|
34
|
+
import shutil
|
|
35
|
+
import time
|
|
36
|
+
from abc import ABC, abstractmethod
|
|
37
|
+
from dataclasses import dataclass, field, replace
|
|
38
|
+
from typing import Dict, Optional, Sequence, Tuple
|
|
39
|
+
|
|
40
|
+
from ..fol.nodes import Node, And, Implies
|
|
41
|
+
|
|
42
|
+
__all__ = [
|
|
43
|
+
"PROVED", "REFUTED", "UNKNOWN", "ERROR", "STATUSES",
|
|
44
|
+
"Verdict", "BackendUnavailable", "ProverBackend",
|
|
45
|
+
"register_backend", "get_backend", "available_backends", "default_chain",
|
|
46
|
+
"run_backend",
|
|
47
|
+
"z3_relevant_premises",
|
|
48
|
+
]
|
|
49
|
+
|
|
50
|
+
# ---------------------------------------------------------------------------
|
|
51
|
+
# Statuses, reasons, and their SZS image
|
|
52
|
+
# ---------------------------------------------------------------------------
|
|
53
|
+
|
|
54
|
+
PROVED = "proved" # validity / entailment established
|
|
55
|
+
REFUTED = "refuted" # a genuine countermodel witnesses invalidity
|
|
56
|
+
UNKNOWN = "unknown" # neither, within this backend's budget/strength
|
|
57
|
+
ERROR = "error" # the backend failed for an infrastructure reason
|
|
58
|
+
STATUSES = (PROVED, REFUTED, UNKNOWN, ERROR)
|
|
59
|
+
|
|
60
|
+
# reason values (for UNKNOWN/ERROR): why we do not know more.
|
|
61
|
+
# "timeout" — a wall-clock budget expired
|
|
62
|
+
# "bound_hit" — a step/size bound expired (max_steps, max_worlds, …)
|
|
63
|
+
# "incomplete" — the method is sound but incomplete on this fragment and
|
|
64
|
+
# gave up honestly (no bound was hit)
|
|
65
|
+
# "unsupported" — the backend has no rule/translation for this fragment
|
|
66
|
+
# "infra" — subprocess/JVM/syntax failure (ERROR only). A prover that
|
|
67
|
+
# REJECTS its input (a syntax or type error, Vampire's "User
|
|
68
|
+
# error: ..." line, E's parse error, Twee's usage error,
|
|
69
|
+
# Prover9's fatal error) and prints no verdict lands here, with
|
|
70
|
+
# the prover's own message in ``detail``: refusing to read a
|
|
71
|
+
# problem is not the same as running out of time, and a
|
|
72
|
+
# benchmark table that files the two together reads a writer
|
|
73
|
+
# defect as "undecided".
|
|
74
|
+
|
|
75
|
+
_SZS = {
|
|
76
|
+
(PROVED, None): "Theorem",
|
|
77
|
+
(REFUTED, None): "CounterSatisfiable",
|
|
78
|
+
(UNKNOWN, "timeout"): "Timeout",
|
|
79
|
+
(UNKNOWN, "bound_hit"): "ResourceOut",
|
|
80
|
+
(UNKNOWN, "incomplete"): "GaveUp",
|
|
81
|
+
(UNKNOWN, "unsupported"): "Inappropriate",
|
|
82
|
+
(UNKNOWN, None): "Unknown",
|
|
83
|
+
(ERROR, "infra"): "Error",
|
|
84
|
+
(ERROR, None): "Error",
|
|
85
|
+
}
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
def _szs_for(status: str, reason: Optional[str]) -> str:
|
|
89
|
+
"""Return the SZS ontology value for a (status, reason) pair."""
|
|
90
|
+
return _SZS.get((status, reason), _SZS[(status, None)])
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
# ---------------------------------------------------------------------------
|
|
94
|
+
# Verdict
|
|
95
|
+
# ---------------------------------------------------------------------------
|
|
96
|
+
|
|
97
|
+
@dataclass(frozen=True)
|
|
98
|
+
class Verdict:
|
|
99
|
+
"""One decision result, whatever route produced it.
|
|
100
|
+
|
|
101
|
+
Fields:
|
|
102
|
+
|
|
103
|
+
* status: ``"proved"`` / ``"refuted"`` / ``"unknown"`` / ``"error"``.
|
|
104
|
+
Truthiness follows ``status == "proved"``.
|
|
105
|
+
* backend: name of the backend that produced this verdict.
|
|
106
|
+
* logic: the logic the query was decided in (``"fol"``, ``"modal"``, …).
|
|
107
|
+
* reason: the why-not-more axis for UNKNOWN/ERROR (see module docstring);
|
|
108
|
+
``None`` for definitive verdicts.
|
|
109
|
+
* szs_status: SZS ontology value; derived from (status, reason) unless a
|
|
110
|
+
backend sets it explicitly (e.g. verbatim from a TSTP line).
|
|
111
|
+
* wall_time: seconds spent inside the backend call.
|
|
112
|
+
* countermodel: JSON-able witness for REFUTED (shape is backend-specific
|
|
113
|
+
but always a dict with a ``"kind"`` key), else ``None``.
|
|
114
|
+
* proof: JSON-able proof object where the backend yields one, else None.
|
|
115
|
+
A ``"kind": "z3_unsat_core"`` or ``"kind": "cvc5_alethe"`` proof
|
|
116
|
+
carries an unsat CORE — sound (re-asserting just that subset is still
|
|
117
|
+
unsat) but not necessarily MINIMAL (the underlying solver is free to
|
|
118
|
+
track more than the smallest sufficient subset) — see
|
|
119
|
+
:func:`z3_relevant_premises` for the same caveat on the dedicated
|
|
120
|
+
premise-relevance query.
|
|
121
|
+
* detail: short free-text note (method that closed it, bound that was
|
|
122
|
+
hit, tried-backends summary, …).
|
|
123
|
+
* agreement: backend names that reported the SAME status, filled by the
|
|
124
|
+
portfolio layer; a single-backend verdict lists just its own.
|
|
125
|
+
* relevant_premises: 0-based indices into the caller's ``premises`` that
|
|
126
|
+
a premise-relevance query found necessary for a PROVED verdict (see
|
|
127
|
+
:func:`z3_relevant_premises`, :mod:`atp.tstp`'s ancestor walk, and
|
|
128
|
+
:mod:`atp.eprover_backend`'s ``eprover_relevant_premises``); ``None``
|
|
129
|
+
when nobody computed it (the default) — NOT a claim that every premise
|
|
130
|
+
was needed. Like ``proof``'s unsat core, a reported set is sound but
|
|
131
|
+
not guaranteed minimal.
|
|
132
|
+
* solver_version: the underlying external tool's own version string
|
|
133
|
+
(Vampire's/Prover9's/E's/Zipperposition's ``--version`` banner, the
|
|
134
|
+
reachable HETS server's ``GET /version`` text, the installed ``cvc5``
|
|
135
|
+
package's distribution version), captured once per process per
|
|
136
|
+
backend/binary and reported here unchanged — see
|
|
137
|
+
:meth:`ProverBackend.solver_version` and :func:`_binary_version` for
|
|
138
|
+
the memoization contract. ``None`` for the kit's own internal
|
|
139
|
+
backends (Z3/tableau/resolution/modelfinder/QML/…, which have no
|
|
140
|
+
external tool to version) and for an external backend whose version
|
|
141
|
+
lookup itself failed or found nothing (the tool's PROOF/DISPROOF
|
|
142
|
+
verdict is unaffected either way — a missing version string is never
|
|
143
|
+
grounds to downgrade a sound answer).
|
|
144
|
+
"""
|
|
145
|
+
|
|
146
|
+
status: str
|
|
147
|
+
backend: str
|
|
148
|
+
logic: str = "fol"
|
|
149
|
+
reason: Optional[str] = None
|
|
150
|
+
szs_status: Optional[str] = None
|
|
151
|
+
wall_time: float = 0.0
|
|
152
|
+
countermodel: Optional[dict] = None
|
|
153
|
+
proof: Optional[dict] = None
|
|
154
|
+
detail: Optional[str] = None
|
|
155
|
+
agreement: Tuple[str, ...] = ()
|
|
156
|
+
relevant_premises: Optional[Tuple[int, ...]] = None
|
|
157
|
+
# New fields go last: positional construction by callers must keep working.
|
|
158
|
+
solver_version: Optional[str] = None
|
|
159
|
+
|
|
160
|
+
def __post_init__(self):
|
|
161
|
+
if self.status not in STATUSES:
|
|
162
|
+
raise ValueError(f"Verdict: unknown status {self.status!r} (use one of {STATUSES})")
|
|
163
|
+
if self.szs_status is None:
|
|
164
|
+
object.__setattr__(self, "szs_status", _szs_for(self.status, self.reason))
|
|
165
|
+
if not self.agreement:
|
|
166
|
+
object.__setattr__(self, "agreement", (self.backend,))
|
|
167
|
+
|
|
168
|
+
def __bool__(self) -> bool:
|
|
169
|
+
return self.status == PROVED
|
|
170
|
+
|
|
171
|
+
@property
|
|
172
|
+
def is_definitive(self) -> bool:
|
|
173
|
+
"""True iff the verdict settles the question (proved or refuted)."""
|
|
174
|
+
return self.status in (PROVED, REFUTED)
|
|
175
|
+
|
|
176
|
+
def to_dict(self) -> dict:
|
|
177
|
+
"""Serialise to a JSON-compatible dict (all fields, names as keys)."""
|
|
178
|
+
return {
|
|
179
|
+
"status": self.status,
|
|
180
|
+
"backend": self.backend,
|
|
181
|
+
"logic": self.logic,
|
|
182
|
+
"reason": self.reason,
|
|
183
|
+
"szs_status": self.szs_status,
|
|
184
|
+
"wall_time": self.wall_time,
|
|
185
|
+
"countermodel": self.countermodel,
|
|
186
|
+
"proof": self.proof,
|
|
187
|
+
"detail": self.detail,
|
|
188
|
+
"agreement": list(self.agreement),
|
|
189
|
+
"relevant_premises": (list(self.relevant_premises)
|
|
190
|
+
if self.relevant_premises is not None else None),
|
|
191
|
+
"solver_version": self.solver_version,
|
|
192
|
+
}
|
|
193
|
+
|
|
194
|
+
|
|
195
|
+
class BackendUnavailable(RuntimeError):
|
|
196
|
+
"""A KNOWN backend was requested but its prerequisites are missing.
|
|
197
|
+
|
|
198
|
+
Raised instead of silently skipping, because a pipeline whose backend
|
|
199
|
+
quietly vanished produces irreproducible numbers. The message names the
|
|
200
|
+
backend and what discovery looked for (env var / binary / install).
|
|
201
|
+
"""
|
|
202
|
+
|
|
203
|
+
|
|
204
|
+
# ---------------------------------------------------------------------------
|
|
205
|
+
# The backend contract
|
|
206
|
+
# ---------------------------------------------------------------------------
|
|
207
|
+
|
|
208
|
+
class ProverBackend(ABC):
|
|
209
|
+
"""Adapter contract for one decision route.
|
|
210
|
+
|
|
211
|
+
``name`` is the registry key; ``logics`` the set of logic labels the
|
|
212
|
+
backend accepts (``"fol"``, ``"modal"``). ``available()`` performs
|
|
213
|
+
discovery without side effects. ``decide()`` answers "do ``premises``
|
|
214
|
+
entail ``formula``?" (validity when ``premises`` is empty) and must NEVER
|
|
215
|
+
raise for an in-contract input — undecidable/unsupported fragments come
|
|
216
|
+
back as an UNKNOWN verdict with the honest ``reason``. Extra keyword
|
|
217
|
+
options are forwarded verbatim to the underlying route (``frame=``,
|
|
218
|
+
``systems=``, ``max_steps=``, …), so the adapters stay thin and nothing
|
|
219
|
+
of the existing signatures is hidden. A backend that names the options it
|
|
220
|
+
reads (:meth:`accepted_options`) is handed only those by
|
|
221
|
+
:func:`~unicode_logic_kit.api.prove`, which refuses an option that no backend
|
|
222
|
+
of the chain reads; :meth:`available_for` answers for the route the options
|
|
223
|
+
of a call select.
|
|
224
|
+
|
|
225
|
+
For the modal backends, a non-empty ``premises`` is the LOCAL consequence
|
|
226
|
+
``⊨ (∧ premises) → φ`` — the standard finite-premise reading; global
|
|
227
|
+
consequence is out of scope here.
|
|
228
|
+
"""
|
|
229
|
+
|
|
230
|
+
name: str = ""
|
|
231
|
+
logics: frozenset = frozenset({"fol"})
|
|
232
|
+
external: bool = False # needs a binary/install outside this venv
|
|
233
|
+
|
|
234
|
+
@abstractmethod
|
|
235
|
+
def available(self) -> bool:
|
|
236
|
+
"""Return whether this backend can run right now (pure discovery)."""
|
|
237
|
+
|
|
238
|
+
def available_for(self, options: dict) -> bool:
|
|
239
|
+
"""Return whether this backend can run a call made with ``options``.
|
|
240
|
+
|
|
241
|
+
:meth:`available` answers for the defaults. An option that picks the
|
|
242
|
+
route (``use_wsl=``, the path of a binary) can change the answer, so the
|
|
243
|
+
dispatcher asks THIS method, with the options of the call, and the answer
|
|
244
|
+
and the run agree. The default ignores ``options`` and returns
|
|
245
|
+
:meth:`available`.
|
|
246
|
+
"""
|
|
247
|
+
return self.available()
|
|
248
|
+
|
|
249
|
+
def accepted_options(self, logic: Optional[str] = None) -> Optional[frozenset]:
|
|
250
|
+
"""Return the names of the keyword options :meth:`decide` reads, or ``None``.
|
|
251
|
+
|
|
252
|
+
``None`` (the default) means the backend declares nothing: the dispatcher
|
|
253
|
+
then passes it every option and cannot tell whether one is read. A
|
|
254
|
+
backend that declares its names lets :func:`~unicode_logic_kit.api.prove`
|
|
255
|
+
refuse an option that no backend of the chain reads, and pass each backend
|
|
256
|
+
only the options it reads. ``logic`` is the logic of the call, for a
|
|
257
|
+
backend whose options depend on it.
|
|
258
|
+
"""
|
|
259
|
+
return None
|
|
260
|
+
|
|
261
|
+
@abstractmethod
|
|
262
|
+
def decide(self, formula: Node, premises: Sequence[Node] = (),
|
|
263
|
+
timeout: int = 10000, **options) -> Verdict:
|
|
264
|
+
"""Decide ``premises ⊨ formula`` and return a :class:`Verdict`."""
|
|
265
|
+
|
|
266
|
+
def solver_version(self) -> Optional[str]:
|
|
267
|
+
"""The underlying external tool's own version string, or ``None``.
|
|
268
|
+
|
|
269
|
+
Default (this base implementation): always ``None`` — the kit's own
|
|
270
|
+
internal calculi and semantic searches (Z3 aside: its Python BINDING
|
|
271
|
+
version is not "the solver's version" in the sense this method
|
|
272
|
+
means, and :class:`Z3Backend` does not override this) have no
|
|
273
|
+
external tool to version. An external backend overrides this with a
|
|
274
|
+
lookup that is PROCESS-LOCAL MEMOIZED (see :func:`_binary_version`
|
|
275
|
+
for the subprocess-spawning backends, and each override's own
|
|
276
|
+
docstring for the HETS/cvc5 routes) — probed at most once per
|
|
277
|
+
distinct binary/server/package for the life of this process, so
|
|
278
|
+
neither this method nor :meth:`decide` (which populates
|
|
279
|
+
``Verdict.solver_version`` from it) ever pays a second lookup.
|
|
280
|
+
Never raises and never changes any verdict's status/reason — this
|
|
281
|
+
is provenance only, queried both from ``decide()`` and, independently,
|
|
282
|
+
from :func:`unicode_logic_kit.eval.batch.batch_decide`'s cache key (so
|
|
283
|
+
a solver upgrade invalidates stale cache entries).
|
|
284
|
+
"""
|
|
285
|
+
return None
|
|
286
|
+
|
|
287
|
+
|
|
288
|
+
# ---------------------------------------------------------------------------
|
|
289
|
+
# Solver-version provenance: a process-local memoized ``--version`` lookup,
|
|
290
|
+
# shared by every subprocess-spawning backend below (Vampire, Prover9, and —
|
|
291
|
+
# via atp.eprover_backend's own import of this function — E/Zipperposition).
|
|
292
|
+
# HETS (an HTTP server, not a spawned binary) and cvc5 (a pip binding, not a
|
|
293
|
+
# spawned binary) have their own analogous, separately-memoized routes in
|
|
294
|
+
# their own modules (hets_backend.py, cvc5_backend.py) — see
|
|
295
|
+
# ProverBackend.solver_version's docstring.
|
|
296
|
+
# ---------------------------------------------------------------------------
|
|
297
|
+
|
|
298
|
+
#: (command, use_wsl) -> the tool's version string, or None on a failed
|
|
299
|
+
#: lookup — a MISS is cached too (never retried), matching item 5's "never
|
|
300
|
+
#: spawn a subprocess per decide() call after the first" requirement.
|
|
301
|
+
#: Deliberately separate from atp.eprover_backend._DISCOVERY_CACHE: discovery
|
|
302
|
+
#: (does a binary exist at all?) and version (what does it print?) answer
|
|
303
|
+
#: different questions and can fail independently — collapsing them into one
|
|
304
|
+
#: cache would make a version-lookup failure look like the binary vanished,
|
|
305
|
+
#: or vice versa.
|
|
306
|
+
_VERSION_CACHE: Dict[Tuple[str, bool], Optional[str]] = {}
|
|
307
|
+
|
|
308
|
+
|
|
309
|
+
def _binary_version(command: str, use_wsl: bool,
|
|
310
|
+
args: Tuple[str, ...] = ("--version",)) -> Optional[str]:
|
|
311
|
+
"""Process-local memoized ``<command> <args>`` version lookup.
|
|
312
|
+
|
|
313
|
+
Spawns the subprocess (through ``wsl.exe`` when ``use_wsl``) at most ONCE
|
|
314
|
+
per ``(command, use_wsl)`` pair for the life of this process; every
|
|
315
|
+
later call for the same pair — including one that failed — returns the
|
|
316
|
+
cached result without spawning again. On a ZERO exit, returns the first
|
|
317
|
+
non-blank line of stdout, falling back to stderr (some tools print
|
|
318
|
+
version banners there), stripped. ``None`` on any failure: binary
|
|
319
|
+
missing, a timeout, WSL unreachable, a non-zero exit — an unrecognized
|
|
320
|
+
flag commonly prints an error/usage line to stdout or stderr on a
|
|
321
|
+
non-zero exit, and accepting that text as a version string would be
|
|
322
|
+
worse than reporting no provenance at all (mirrors the returncode check
|
|
323
|
+
:func:`unicode_logic_kit.atp.eprover_backend._discover` already applies to
|
|
324
|
+
its own subprocess probe) — OR the banner not being valid text under
|
|
325
|
+
``subprocess.run(..., text=True)``'s decoding (``UnicodeError``, e.g. a
|
|
326
|
+
tool that writes a non-UTF-8 locale-encoded byte in its ``--version``
|
|
327
|
+
output): decoding happens INSIDE ``subprocess.run`` here, so this is
|
|
328
|
+
caught exactly like any other failure to read the banner, never left to
|
|
329
|
+
propagate as a bare ``ValueError`` out of a caller's ``decide()``. This
|
|
330
|
+
is best-effort provenance, so nothing here ever raises or turns into an
|
|
331
|
+
ERROR verdict.
|
|
332
|
+
"""
|
|
333
|
+
key = (command, use_wsl)
|
|
334
|
+
if key in _VERSION_CACHE:
|
|
335
|
+
return _VERSION_CACHE[key]
|
|
336
|
+
|
|
337
|
+
import subprocess
|
|
338
|
+
|
|
339
|
+
result: Optional[str] = None
|
|
340
|
+
try:
|
|
341
|
+
cmd = ["wsl.exe", command, *args] if use_wsl else [command, *args]
|
|
342
|
+
# The tool never reads this process's stdin: Prover9 reads its problem from
|
|
343
|
+
# stdin for any flag it does not know (``--version`` included), and would
|
|
344
|
+
# block on an inherited pipe (an MCP server's own protocol stream) or read it.
|
|
345
|
+
proc = subprocess.run(cmd, capture_output=True, text=True, timeout=10,
|
|
346
|
+
stdin=subprocess.DEVNULL)
|
|
347
|
+
if proc.returncode == 0:
|
|
348
|
+
text = (proc.stdout or "").strip() or (proc.stderr or "").strip()
|
|
349
|
+
if text:
|
|
350
|
+
result = text.splitlines()[0].strip()
|
|
351
|
+
except (OSError, subprocess.TimeoutExpired, UnicodeError):
|
|
352
|
+
result = None
|
|
353
|
+
_VERSION_CACHE[key] = result
|
|
354
|
+
return result
|
|
355
|
+
|
|
356
|
+
|
|
357
|
+
#: binary -> whether ``wsl.exe which <binary>`` found it. An answer is cached for
|
|
358
|
+
#: the life of the process, like :data:`_VERSION_CACHE`, so a discovery that
|
|
359
|
+
#: spawns ``wsl.exe`` is paid once per binary. A probe that TIMED OUT is not an
|
|
360
|
+
#: answer and is not cached: WSL may only have been slow to start.
|
|
361
|
+
_WSL_BINARY_CACHE: Dict[str, bool] = {}
|
|
362
|
+
|
|
363
|
+
#: Seconds the WSL probe of :func:`_wsl_has_binary` may take.
|
|
364
|
+
_WSL_PROBE_TIMEOUT = 10
|
|
365
|
+
|
|
366
|
+
|
|
367
|
+
def _wsl_has_binary(binary: str) -> bool:
|
|
368
|
+
"""Whether ``binary`` (a name on the PATH inside WSL, or a path inside WSL)
|
|
369
|
+
exists there and is executable.
|
|
370
|
+
|
|
371
|
+
This is what ``available()`` has to ask when a backend will run its binary
|
|
372
|
+
through ``wsl.exe``: the host's own PATH says nothing about a Linux binary.
|
|
373
|
+
Asks ``wsl.exe which <binary>`` -- the command the runners themselves use to
|
|
374
|
+
start it -- with a short timeout and an empty standard input, so the probe
|
|
375
|
+
never blocks on, or reads, this process's input. ``False`` when ``wsl.exe``
|
|
376
|
+
cannot be started (no WSL on this host), when it times out, and when the
|
|
377
|
+
binary is not found; never raises.
|
|
378
|
+
"""
|
|
379
|
+
if binary in _WSL_BINARY_CACHE:
|
|
380
|
+
return _WSL_BINARY_CACHE[binary]
|
|
381
|
+
|
|
382
|
+
import subprocess
|
|
383
|
+
|
|
384
|
+
try:
|
|
385
|
+
proc = subprocess.run(["wsl.exe", "which", binary], capture_output=True,
|
|
386
|
+
timeout=_WSL_PROBE_TIMEOUT, stdin=subprocess.DEVNULL)
|
|
387
|
+
except subprocess.TimeoutExpired:
|
|
388
|
+
return False
|
|
389
|
+
except OSError:
|
|
390
|
+
found = False
|
|
391
|
+
else:
|
|
392
|
+
found = proc.returncode == 0 and bool(
|
|
393
|
+
(proc.stdout or b"").decode("utf-8", errors="replace").replace("\x00", "").strip())
|
|
394
|
+
_WSL_BINARY_CACHE[binary] = found
|
|
395
|
+
return found
|
|
396
|
+
|
|
397
|
+
|
|
398
|
+
def _native_command_exists(command: str) -> bool:
|
|
399
|
+
"""Whether this host can start ``command``: an executable file at that path, or a
|
|
400
|
+
name that is on ``PATH``.
|
|
401
|
+
|
|
402
|
+
``shutil.which`` answers both, except for one spelling that the run accepts: on
|
|
403
|
+
Windows a path with a directory part and no extension is looked up exactly as
|
|
404
|
+
written by ``shutil.which`` (before Python 3.12), whereas ``subprocess`` starts it
|
|
405
|
+
through ``CreateProcess``, which appends ``.exe`` to a name that has no extension.
|
|
406
|
+
A caller who names the installed ``minizinc.exe`` as ``D:/Minizinc/MiniZinc/minizinc``
|
|
407
|
+
names a binary the run starts, so that spelling is tried too. Never starts anything.
|
|
408
|
+
"""
|
|
409
|
+
import os
|
|
410
|
+
|
|
411
|
+
if shutil.which(command) is not None:
|
|
412
|
+
return True
|
|
413
|
+
if os.name == "nt" and not os.path.splitext(command)[1]:
|
|
414
|
+
return shutil.which(command + ".exe") is not None
|
|
415
|
+
return False
|
|
416
|
+
|
|
417
|
+
|
|
418
|
+
def _implication(formula: Node, premises: Sequence[Node]) -> Node:
|
|
419
|
+
"""Fold ``premises ⊨ φ`` into the single formula ``(∧ premises) → φ``."""
|
|
420
|
+
premises = list(premises)
|
|
421
|
+
if not premises:
|
|
422
|
+
return formula
|
|
423
|
+
conj = premises[0]
|
|
424
|
+
for p in premises[1:]:
|
|
425
|
+
conj = And(conj, p)
|
|
426
|
+
return Implies(conj, formula)
|
|
427
|
+
|
|
428
|
+
|
|
429
|
+
def _timed(fn):
|
|
430
|
+
"""Run ``fn()`` returning ``(result, seconds)``."""
|
|
431
|
+
start = time.perf_counter()
|
|
432
|
+
result = fn()
|
|
433
|
+
return result, time.perf_counter() - start
|
|
434
|
+
|
|
435
|
+
|
|
436
|
+
# ---------------------------------------------------------------------------
|
|
437
|
+
# A prover that REFUSES its input is an error, not a timeout.
|
|
438
|
+
#
|
|
439
|
+
# Every subprocess backend reads a verdict off its prover's output: an SZS
|
|
440
|
+
# status line (Vampire, E, Zipperposition), a ``RESULT:`` line (Twee), a
|
|
441
|
+
# ``THEOREM PROVED`` line (Prover9). A prover that cannot read the problem
|
|
442
|
+
# prints none of those -- it prints its own complaint instead (Vampire 5.0.1:
|
|
443
|
+
# ``User error: Non-boolean term agent(agent(X0)) of sort $i is used in a
|
|
444
|
+
# formula context``; E 3.5.1: ``eprover: <file>:1:(Column 45): ... expected,
|
|
445
|
+
# but Closing bracket (')') read``; Twee 2.6.1: ``Error in <file> (line 2,
|
|
446
|
+
# column 1): Unexpected fof``) -- and a backend that reads "no verdict" as
|
|
447
|
+
# "did not get there in time" reports a defect in the problem WRITER as an
|
|
448
|
+
# undecided question. One wording, shared by every backend, keeps the three
|
|
449
|
+
# readings (timeout / honest give-up / refusal) apart wherever the outcome is
|
|
450
|
+
# shown.
|
|
451
|
+
# ---------------------------------------------------------------------------
|
|
452
|
+
|
|
453
|
+
#: Lines of a prover's output that explain a failure. Used only to choose
|
|
454
|
+
#: WHERE to start quoting a long output; a short one is quoted whole.
|
|
455
|
+
_FAILURE_LINE = re.compile(
|
|
456
|
+
r"error|exception|assertion|fatal|abort|segmentation|core dumped|cannot|"
|
|
457
|
+
r"unexpected|expected|failed|invalid", re.IGNORECASE)
|
|
458
|
+
|
|
459
|
+
#: How much of the prover's own output a refusal's ``detail`` quotes.
|
|
460
|
+
_REJECTION_QUOTE_CHARS = 600
|
|
461
|
+
|
|
462
|
+
|
|
463
|
+
def _rejection_detail(prover: str, output: str, evidence: str) -> str:
|
|
464
|
+
"""The ``detail`` of an ERROR verdict for a prover that gave no verdict.
|
|
465
|
+
|
|
466
|
+
``evidence`` says what was missing (``"no SZS status line in its output"``,
|
|
467
|
+
``"no RESULT line in its output"``, ...). The prover's OWN text follows,
|
|
468
|
+
one line per ``|``-separated item, unmodified apart from stripping blank
|
|
469
|
+
lines, NUL bytes (a Windows ``wsl.exe`` diagnostic arrives UTF-16 decoded
|
|
470
|
+
as 8-bit) and trailing whitespace. An output longer than
|
|
471
|
+
:data:`_REJECTION_QUOTE_CHARS` is quoted from its first failure-looking
|
|
472
|
+
line, or from its tail when no line looks like one -- the explanation of a
|
|
473
|
+
refusal is at the start (Vampire) and that of a crash at the end (E's
|
|
474
|
+
``Assertion ... failed``), and a run that printed pages of statistics
|
|
475
|
+
first would otherwise bury either.
|
|
476
|
+
"""
|
|
477
|
+
lines = [ln.strip() for ln in output.replace("\x00", "").splitlines()]
|
|
478
|
+
lines = [ln for ln in lines if ln]
|
|
479
|
+
message = " | ".join(lines)
|
|
480
|
+
if len(message) > _REJECTION_QUOTE_CHARS:
|
|
481
|
+
start = next((i for i, ln in enumerate(lines) if _FAILURE_LINE.search(ln)), None)
|
|
482
|
+
if start is None:
|
|
483
|
+
message = "... " + message[-_REJECTION_QUOTE_CHARS:]
|
|
484
|
+
else:
|
|
485
|
+
message = " | ".join(lines[start:])
|
|
486
|
+
if len(message) > _REJECTION_QUOTE_CHARS:
|
|
487
|
+
message = message[:_REJECTION_QUOTE_CHARS] + " ..."
|
|
488
|
+
if not message:
|
|
489
|
+
message = "(it printed nothing)"
|
|
490
|
+
return (f"{prover} produced no verdict ({evidence}); this is a failure, "
|
|
491
|
+
f"not a timeout. Its own output: {message}")
|
|
492
|
+
|
|
493
|
+
|
|
494
|
+
def _no_definitive_verdict(pseudo_backend: str, logic: str,
|
|
495
|
+
verdicts: Sequence[Verdict], empty: str) -> Verdict:
|
|
496
|
+
"""The collective verdict of a chain/portfolio in which nobody settled it.
|
|
497
|
+
|
|
498
|
+
``"<backend>:<status>[/<reason>] (<detail>)"`` for EVERY member that ran:
|
|
499
|
+
its status and reason, and its own account of them when it gave one. A member
|
|
500
|
+
that FAILED (ERROR, or any member whose reason is ``"infra"`` -- the Isabelle
|
|
501
|
+
adapter reports that as UNKNOWN) quotes its ``detail``, so a prover's
|
|
502
|
+
refusal reaches the caller instead of being flattened into the bare
|
|
503
|
+
``vampire:error/infra`` that reads like any other "no answer"; so does a
|
|
504
|
+
member that REFUSED the question (reason ``"unsupported"``): the writer's
|
|
505
|
+
message says what it would not write and what to write instead, and without it
|
|
506
|
+
the caller has only the word "unsupported"; so does a member that gave up
|
|
507
|
+
(``"bound_hit"``, ``"incomplete"``, ``"timeout"``): the bound that was hit, or
|
|
508
|
+
the point where the search ended, is what tells the caller which budget to
|
|
509
|
+
raise. A member without a ``detail`` is listed by status and reason alone. When EVERY
|
|
510
|
+
member failed there is no honest ``unknown`` to report -- nothing was asked
|
|
511
|
+
and answered -- so the collective verdict is itself an ERROR; with at least
|
|
512
|
+
one member that really answered "unknown" it stays UNKNOWN and the failures
|
|
513
|
+
sit in the detail beside it.
|
|
514
|
+
"""
|
|
515
|
+
if not verdicts:
|
|
516
|
+
return Verdict(UNKNOWN, pseudo_backend, logic=logic, detail=empty)
|
|
517
|
+
summary = "; ".join(
|
|
518
|
+
f"{v.backend}:{v.status}" + (f"/{v.reason}" if v.reason else "")
|
|
519
|
+
+ (f" ({v.detail})" if v.detail else "")
|
|
520
|
+
for v in verdicts)
|
|
521
|
+
if all(v.status == ERROR for v in verdicts):
|
|
522
|
+
return Verdict(ERROR, pseudo_backend, logic=logic, reason="infra",
|
|
523
|
+
detail=f"no backend gave an answer — {summary}")
|
|
524
|
+
return Verdict(UNKNOWN, pseudo_backend, logic=logic,
|
|
525
|
+
detail=f"no definitive verdict — {summary}")
|
|
526
|
+
|
|
527
|
+
|
|
528
|
+
def _z3_sort_axioms(formula: Node, premises: Sequence[Node], env=None) -> list:
|
|
529
|
+
"""``sort_axioms(formula, *premises)``, each translated via ``to_z3(env)``.
|
|
530
|
+
|
|
531
|
+
``env`` is the :class:`~unicode_logic_kit.fol.nodes.Z3Env` the goal and the
|
|
532
|
+
premises were translated with (the caller's one environment for the whole
|
|
533
|
+
problem).
|
|
534
|
+
|
|
535
|
+
That is every sort's non-emptiness AND the membership atom ``S(c)`` of every
|
|
536
|
+
sorted constant ``c:S`` of the goal or of a premise: ``to_z3()`` reads a
|
|
537
|
+
sorted constant as the plain constant, so without the atom the guard
|
|
538
|
+
reading forgets which sort it was written in. Shared by :class:`Z3Backend`
|
|
539
|
+
and :func:`z3_relevant_premises` so both decide the exact same many-sorted
|
|
540
|
+
entailment — see :func:`_z3_track_and_check`'s ``z3_sort_axioms`` parameter
|
|
541
|
+
for why these are added UNTRACKED rather than as more ``p<i>``-tagged
|
|
542
|
+
premises. Empty for an unsorted query.
|
|
543
|
+
"""
|
|
544
|
+
from ..fol._msfl_nodes import sort_axioms
|
|
545
|
+
return [axiom.to_z3(env) for axiom in sort_axioms(formula, *premises)]
|
|
546
|
+
|
|
547
|
+
|
|
548
|
+
# ---------------------------------------------------------------------------
|
|
549
|
+
# Internal backends: the kit's own calculi and semantic searches
|
|
550
|
+
# ---------------------------------------------------------------------------
|
|
551
|
+
|
|
552
|
+
#: The tracking literals of :func:`_z3_track_and_check` are Boolean constants whose Z3
|
|
553
|
+
#: symbol is an INTEGER symbol (``Z3_mk_int_symbol``), numbered from this base. Every name
|
|
554
|
+
#: the kit writes into a Z3 expression is a STRING symbol, so no formula of any caller --
|
|
555
|
+
#: whatever its propositions, predicates and sorts are called -- can hold a symbol equal to a
|
|
556
|
+
#: tag: a proposition named ``goal`` or ``p0`` is another constant. The base keeps the numbers
|
|
557
|
+
#: clear of the ones Z3 itself gives its own fresh constants (small counters).
|
|
558
|
+
_Z3_TAG_BASE = 1 << 29
|
|
559
|
+
|
|
560
|
+
|
|
561
|
+
def _z3_tag(number: int):
|
|
562
|
+
"""The tracking literal number ``number``: 0 is the negated goal's, ``1 + i`` premise ``i``'s."""
|
|
563
|
+
from z3 import BoolRef, BoolSort, Z3_mk_const, Z3_mk_int_symbol, main_ctx
|
|
564
|
+
|
|
565
|
+
ctx = main_ctx()
|
|
566
|
+
symbol = Z3_mk_int_symbol(ctx.ref(), _Z3_TAG_BASE + number)
|
|
567
|
+
return BoolRef(Z3_mk_const(ctx.ref(), symbol, BoolSort(ctx).ast), ctx)
|
|
568
|
+
|
|
569
|
+
|
|
570
|
+
def _z3_tag_number(decl) -> Optional[int]:
|
|
571
|
+
"""The number of the tracking literal that the Z3 declaration ``decl`` is, or ``None``.
|
|
572
|
+
|
|
573
|
+
A declaration is a tracking literal exactly when it has no argument, returns ``Bool`` and
|
|
574
|
+
is named by an integer symbol of the band :data:`_Z3_TAG_BASE` and above (see there), which
|
|
575
|
+
no symbol of a kit formula is.
|
|
576
|
+
"""
|
|
577
|
+
from z3 import BoolSort, Z3_INT_SYMBOL, Z3_get_decl_name, Z3_get_symbol_int, Z3_get_symbol_kind
|
|
578
|
+
|
|
579
|
+
if decl.arity() != 0 or decl.range() != BoolSort(decl.ctx):
|
|
580
|
+
return None
|
|
581
|
+
ref = decl.ctx.ref()
|
|
582
|
+
symbol = Z3_get_decl_name(ref, decl.ast)
|
|
583
|
+
if Z3_get_symbol_kind(ref, symbol) != Z3_INT_SYMBOL:
|
|
584
|
+
return None
|
|
585
|
+
value = Z3_get_symbol_int(ref, symbol)
|
|
586
|
+
return value - _Z3_TAG_BASE if value >= _Z3_TAG_BASE else None
|
|
587
|
+
|
|
588
|
+
|
|
589
|
+
def _z3_core_numbers(solver) -> list:
|
|
590
|
+
"""The numbers of the tracking literals in ``solver.unsat_core()`` (0 is the goal, ``1 + i`` premise ``i``)."""
|
|
591
|
+
numbers = (_z3_tag_number(tag.decl()) for tag in solver.unsat_core())
|
|
592
|
+
return sorted(number for number in numbers if number is not None)
|
|
593
|
+
|
|
594
|
+
|
|
595
|
+
def _z3_core_names(solver) -> list:
|
|
596
|
+
"""The unsat core of ``solver`` as the names a proof reports: ``"goal"`` and ``"p<i>"`` (0-based)."""
|
|
597
|
+
return [("goal" if number == 0 else f"p{number - 1}") for number in _z3_core_numbers(solver)]
|
|
598
|
+
|
|
599
|
+
|
|
600
|
+
def _z3_track_and_check(z3_formula, z3_premises: Sequence, timeout: int,
|
|
601
|
+
z3_sort_axioms: Sequence = ()):
|
|
602
|
+
"""Run ONE per-call Z3 ``Solver``, tracking every assertion by name.
|
|
603
|
+
|
|
604
|
+
Asserts ``Not(z3_formula)`` under the tag ``"goal"`` and each of
|
|
605
|
+
``z3_premises`` under ``"p<i>"`` (0-based), via ``assert_and_track``,
|
|
606
|
+
with ``unsat_core=True`` set on THIS solver instance only — never
|
|
607
|
+
``z3.set_param(proof=True)``, which is a process-wide global that would
|
|
608
|
+
change solving behaviour for every other Z3 consumer in the kit
|
|
609
|
+
(semantics evaluators, dl, chem, finite-model, modal/second/third-order
|
|
610
|
+
— anything sharing the module-level ``_SORT``/default context in
|
|
611
|
+
``fol/_fol_nodes.py``) for the rest of the process.
|
|
612
|
+
|
|
613
|
+
``z3_sort_axioms`` (see
|
|
614
|
+
:func:`~unicode_logic_kit.fol._msfl_nodes.sort_axioms`: every sort is
|
|
615
|
+
non-empty, and a sorted constant ``c:S`` is in ``S``) are added with a
|
|
616
|
+
plain, UNTRACKED ``solver.add`` — they are background MSFOL convention,
|
|
617
|
+
never one of the caller's own premises, so they must never gain a ``p<i>``
|
|
618
|
+
tag: that would make an untranslatable, caller-invisible synthetic sentence
|
|
619
|
+
show up in :class:`Z3Backend`'s ``proof`` unsat core or in
|
|
620
|
+
:func:`z3_relevant_premises`'s reported indices, which are defined purely
|
|
621
|
+
over the CALLER's own ``premises`` list. They are asserted OUTSIDE the
|
|
622
|
+
negated goal: a membership atom under the goal's negation would be
|
|
623
|
+
something to prove. Empty (the default) for an unsorted query, so the
|
|
624
|
+
solver call is byte-for-byte the same as before this parameter existed.
|
|
625
|
+
|
|
626
|
+
**The tags are named by integer symbols.** A proposition of a problem can be
|
|
627
|
+
called anything, ``goal`` and ``p0`` included (the Prover9 and SMT-LIB readers
|
|
628
|
+
produce such propositions); a tag spelled like one would be the very proposition,
|
|
629
|
+
assumed true, and would prove any goal. So a tag is a Boolean constant with an
|
|
630
|
+
integer symbol (:func:`_z3_tag`), a kind of name that no string a formula carries
|
|
631
|
+
can be, and the names ``"goal"`` / ``"p<i>"`` are only what a proof REPORTS
|
|
632
|
+
(:func:`_z3_core_names`).
|
|
633
|
+
|
|
634
|
+
Built once and shared by :class:`Z3Backend` (C12: a per-verdict
|
|
635
|
+
``z3_unsat_core`` proof certificate) and :func:`z3_relevant_premises`
|
|
636
|
+
(C11: which premises a PROVED entailment actually needed) so both read
|
|
637
|
+
the exact same assert-and-track call shape rather than drifting apart —
|
|
638
|
+
including the identical sort axioms, so the two can never disagree about
|
|
639
|
+
whether a many-sorted entailment holds.
|
|
640
|
+
|
|
641
|
+
Returns ``(result, solver)`` — ``result`` is Z3's own
|
|
642
|
+
``sat``/``unsat``/``unknown``; on ``unsat``, ``solver.unsat_core()``
|
|
643
|
+
holds the tracked-name subset Z3 actually used. That subset is SOUND
|
|
644
|
+
(re-asserting just it is still unsat) but not necessarily MINIMAL (Z3's
|
|
645
|
+
core extraction is not obliged to find the smallest one) — callers that
|
|
646
|
+
need "used" language should say "relevant"/"a sufficient subset", never
|
|
647
|
+
"the minimal set".
|
|
648
|
+
"""
|
|
649
|
+
from z3 import Solver, Not as _ZNot
|
|
650
|
+
|
|
651
|
+
solver = Solver()
|
|
652
|
+
solver.set("timeout", timeout)
|
|
653
|
+
solver.set("random_seed", 42)
|
|
654
|
+
solver.set(unsat_core=True)
|
|
655
|
+
for axiom in z3_sort_axioms:
|
|
656
|
+
solver.add(axiom)
|
|
657
|
+
for i, p in enumerate(z3_premises):
|
|
658
|
+
solver.assert_and_track(p, _z3_tag(1 + i))
|
|
659
|
+
solver.assert_and_track(_ZNot(z3_formula), _z3_tag(0))
|
|
660
|
+
return solver.check(), solver
|
|
661
|
+
|
|
662
|
+
|
|
663
|
+
def _z3_model_assignment(model, n_premises: int = 0) -> Dict[str, str]:
|
|
664
|
+
"""Read a satisfying ``z3.ModelRef`` back into a ``{name: value}`` dict.
|
|
665
|
+
|
|
666
|
+
``assert_and_track``'s own tracking booleans (see :func:`_z3_track_and_check`)
|
|
667
|
+
are 0-ary Bool-sorted Z3 declarations, so they show up in ``model.decls()``
|
|
668
|
+
right alongside the formula's real symbols and would otherwise leak into a
|
|
669
|
+
REFUTED verdict's witness as spurious extra keys. They are excluded by what
|
|
670
|
+
they are, not by their spelling: a tag is a constant with an integer symbol
|
|
671
|
+
(:func:`_z3_tag_number`), and no symbol of a formula is, so a proposition
|
|
672
|
+
named ``goal`` or ``p0`` is reported like any other. ``n_premises`` is not
|
|
673
|
+
needed to recognise a tag and is accepted for the callers that pass it.
|
|
674
|
+
|
|
675
|
+
A name that the model declares more than once (``P`` at two arities, a
|
|
676
|
+
function and a predicate of one name) is two symbols and is reported under
|
|
677
|
+
``"P/1"`` / ``"P/2"`` keys instead of one overwriting the other — see
|
|
678
|
+
:func:`~unicode_logic_kit.atp.z3_models.declaration_keys`; a name declared
|
|
679
|
+
once keeps its plain name.
|
|
680
|
+
"""
|
|
681
|
+
from .z3_models import model_assignment
|
|
682
|
+
|
|
683
|
+
return model_assignment(model, skip=lambda d: _z3_tag_number(d) is not None)
|
|
684
|
+
|
|
685
|
+
|
|
686
|
+
def z3_relevant_premises(formula: Node, premises: Sequence[Node] = (),
|
|
687
|
+
timeout: int = 10000) -> Optional[Tuple[int, ...]]:
|
|
688
|
+
"""Which of ``premises`` did Z3 actually need to prove ``⊨ formula``?
|
|
689
|
+
|
|
690
|
+
Runs the same :func:`_z3_track_and_check` per-``Solver`` assert-and-track
|
|
691
|
+
call :class:`Z3Backend` uses for its own ``proof`` certificate — including
|
|
692
|
+
the same many-sorted axioms (non-emptiness of every sort, membership of
|
|
693
|
+
every sorted constant) when ``formula``/``premises`` use a sort
|
|
694
|
+
(:func:`_z3_sort_axioms`), so this always agrees with
|
|
695
|
+
:class:`Z3Backend` about whether the entailment holds — but reports only
|
|
696
|
+
the premise side of the core, as 0-based indices into ``premises`` (never
|
|
697
|
+
including the ``"goal"`` tag itself — the negated conclusion is always
|
|
698
|
+
"needed" trivially, so it carries no information about which PREMISES
|
|
699
|
+
were relevant; the sort axioms are untracked, so they can never appear in
|
|
700
|
+
the core either — see :func:`_z3_track_and_check`).
|
|
701
|
+
|
|
702
|
+
Args:
|
|
703
|
+
formula: the goal.
|
|
704
|
+
premises: candidate premises (same fragment ``Z3Backend``/
|
|
705
|
+
``Node.to_z3`` decides — uninterpreted sort + equality, no
|
|
706
|
+
arithmetic; substructural nodes are outside it).
|
|
707
|
+
timeout: milliseconds, forwarded to the solver exactly as
|
|
708
|
+
``Z3Backend.decide`` forwards its own ``timeout``.
|
|
709
|
+
|
|
710
|
+
Returns:
|
|
711
|
+
A sorted tuple of 0-based premise indices, or ``None`` when there is
|
|
712
|
+
nothing sound to report: the entailment does not hold (Z3 finds it
|
|
713
|
+
SAT or times out UNKNOWN — there is no "used premises" answer for a
|
|
714
|
+
non-theorem), or the fragment is unsupported (``to_z3`` raises
|
|
715
|
+
``NotImplementedError`` on ``formula`` or any premise). ``None`` is
|
|
716
|
+
the honest "don't know" answer here, never a guessed subset.
|
|
717
|
+
|
|
718
|
+
The returned set is SOUND but not necessarily MINIMAL — see
|
|
719
|
+
:func:`_z3_track_and_check`'s docstring; a genuinely redundant
|
|
720
|
+
premise (one that, alone, already suffices) need not appear
|
|
721
|
+
alongside the other route to the same conclusion, but two premises
|
|
722
|
+
that are each independently sufficient are not guaranteed to be
|
|
723
|
+
pruned down to a single one either — only that the returned subset
|
|
724
|
+
itself is enough.
|
|
725
|
+
"""
|
|
726
|
+
from ..fol.nodes import Z3Env
|
|
727
|
+
|
|
728
|
+
premises = list(premises)
|
|
729
|
+
try:
|
|
730
|
+
env = Z3Env()
|
|
731
|
+
z3_formula = formula.to_z3(env)
|
|
732
|
+
z3_premises = [p.to_z3(env) for p in premises]
|
|
733
|
+
z3_sorts = _z3_sort_axioms(formula, premises, env)
|
|
734
|
+
except NotImplementedError:
|
|
735
|
+
return None
|
|
736
|
+
|
|
737
|
+
from z3 import unsat
|
|
738
|
+
|
|
739
|
+
res, solver = _z3_track_and_check(z3_formula, z3_premises, timeout, z3_sorts)
|
|
740
|
+
if res != unsat:
|
|
741
|
+
return None
|
|
742
|
+
return tuple(sorted({number - 1 for number in _z3_core_numbers(solver) if number > 0}))
|
|
743
|
+
|
|
744
|
+
|
|
745
|
+
class Z3Backend(ProverBackend):
|
|
746
|
+
"""Classical FOL/MSFOL via Z3 — tri-state, with a model on refutation.
|
|
747
|
+
|
|
748
|
+
Many-sorted input: a sort's non-emptiness and the membership of every
|
|
749
|
+
sorted constant in its sort (the MSFOL convention — see the
|
|
750
|
+
classical-reasoning guide's many-sorted section) are asserted as extra,
|
|
751
|
+
untracked, UNNEGATED premises alongside ``premises`` — see
|
|
752
|
+
:func:`_z3_sort_axioms` and :func:`_z3_track_and_check` — never folded
|
|
753
|
+
inside ``Node.to_z3()`` itself, which stays polarity-blind. This is what
|
|
754
|
+
makes a REFUTED verdict's countermodel always a legal MSFOL structure (no
|
|
755
|
+
sort empty, every sorted constant inside its sort) instead of exploiting a
|
|
756
|
+
loophole ``semantics.modelfinder`` never considers, and what makes
|
|
757
|
+
``∀x:Human Mortal(x) ⊢ Mortal(socrates:Human)`` PROVED.
|
|
758
|
+
"""
|
|
759
|
+
|
|
760
|
+
name = "z3"
|
|
761
|
+
logics = frozenset({"fol"})
|
|
762
|
+
external = False # z3-solver is a hard dependency
|
|
763
|
+
|
|
764
|
+
def available(self) -> bool:
|
|
765
|
+
return True
|
|
766
|
+
|
|
767
|
+
def decide(self, formula: Node, premises: Sequence[Node] = (),
|
|
768
|
+
timeout: int = 10000, **options) -> Verdict:
|
|
769
|
+
from z3 import sat, unsat
|
|
770
|
+
from ..fol.nodes import Z3Env
|
|
771
|
+
|
|
772
|
+
premises = list(premises)
|
|
773
|
+
try:
|
|
774
|
+
# ONE environment for the goal and every premise: a numeral and a
|
|
775
|
+
# constant of the same text are refused wherever they meet, and
|
|
776
|
+
# one name at two arities stays two symbols across the problem.
|
|
777
|
+
env = Z3Env()
|
|
778
|
+
z3_formula = formula.to_z3(env)
|
|
779
|
+
z3_premises = [p.to_z3(env) for p in premises]
|
|
780
|
+
z3_sorts = _z3_sort_axioms(formula, premises, env)
|
|
781
|
+
except NotImplementedError as exc:
|
|
782
|
+
return Verdict(UNKNOWN, self.name, reason="unsupported", detail=str(exc))
|
|
783
|
+
|
|
784
|
+
(res, solver), elapsed = _timed(
|
|
785
|
+
lambda: _z3_track_and_check(z3_formula, z3_premises, timeout, z3_sorts))
|
|
786
|
+
if res == unsat:
|
|
787
|
+
# A tracked-name unsat core, per _z3_track_and_check's docstring:
|
|
788
|
+
# sound (re-asserting just these is still unsat, see the C12
|
|
789
|
+
# soundness self-check test) but not necessarily minimal.
|
|
790
|
+
core = sorted(_z3_core_names(solver))
|
|
791
|
+
proof = {"kind": "z3_unsat_core", "core": core}
|
|
792
|
+
return Verdict(PROVED, self.name, wall_time=elapsed, proof=proof)
|
|
793
|
+
if res == sat:
|
|
794
|
+
assignment = _z3_model_assignment(solver.model(), len(z3_premises))
|
|
795
|
+
return Verdict(REFUTED, self.name, wall_time=elapsed,
|
|
796
|
+
countermodel={"kind": "z3_model", "assignment": assignment})
|
|
797
|
+
why = solver.reason_unknown()
|
|
798
|
+
reason = "timeout" if ("timeout" in why or "cancel" in why) else "incomplete"
|
|
799
|
+
return Verdict(UNKNOWN, self.name, reason=reason, wall_time=elapsed, detail=why)
|
|
800
|
+
|
|
801
|
+
|
|
802
|
+
class TableauBackend(ProverBackend):
|
|
803
|
+
"""The kit's own classical analytic tableau (Z3-free).
|
|
804
|
+
|
|
805
|
+
Complete and decidable propositionally; on first-order inputs a
|
|
806
|
+
non-closure is only "no closed tableau within the bounds" → bound_hit.
|
|
807
|
+
The bounds are ``max_steps`` (which also bounds the length of a branch: the
|
|
808
|
+
search is a loop over a branch, it does not depend on the interpreter's
|
|
809
|
+
recursion limit), ``max_terms`` and the call's ``timeout`` (a run that used it
|
|
810
|
+
up is ``timeout``, not ``bound_hit``). A formula nested deeper than the helpers
|
|
811
|
+
that walk it can follow within the recursion limit is a bound too: the verdict
|
|
812
|
+
is ``unknown`` / ``bound_hit`` and its ``detail`` names the nesting depth
|
|
813
|
+
(:func:`~unicode_logic_kit.atp.tableau.nesting_depth`). Many-sorted input is
|
|
814
|
+
searched through its guard image with the sort axioms as further formulas to
|
|
815
|
+
refute (see :func:`~unicode_logic_kit.atp.tableau.tableau_closed`).
|
|
816
|
+
"""
|
|
817
|
+
|
|
818
|
+
name = "tableau"
|
|
819
|
+
logics = frozenset({"fol"})
|
|
820
|
+
external = False
|
|
821
|
+
|
|
822
|
+
def available(self) -> bool:
|
|
823
|
+
return True
|
|
824
|
+
|
|
825
|
+
def decide(self, formula: Node, premises: Sequence[Node] = (),
|
|
826
|
+
timeout: int = 10000, **options) -> Verdict:
|
|
827
|
+
# The detailed route records a TableauProof alongside the same
|
|
828
|
+
# search (recording never changes the verdict — regression-pinned
|
|
829
|
+
# in test_tableau_check.py), so a PROVED Verdict can carry the
|
|
830
|
+
# independently checkable proof dict.
|
|
831
|
+
from .tableau import _search_detailed
|
|
832
|
+
|
|
833
|
+
try:
|
|
834
|
+
(proof, nesting), elapsed = _timed(
|
|
835
|
+
lambda: _search_detailed(list(premises), formula,
|
|
836
|
+
timeout=timeout, **options))
|
|
837
|
+
except NotImplementedError as exc:
|
|
838
|
+
return Verdict(UNKNOWN, self.name, reason="unsupported", detail=str(exc))
|
|
839
|
+
if proof is not None:
|
|
840
|
+
try:
|
|
841
|
+
encoded = proof.to_dict()
|
|
842
|
+
except RecursionError:
|
|
843
|
+
encoded = None # a proof too deeply nested to serialise is still a proof
|
|
844
|
+
return Verdict(PROVED, self.name, wall_time=elapsed, proof=encoded)
|
|
845
|
+
if nesting is not None:
|
|
846
|
+
import sys
|
|
847
|
+
return Verdict(UNKNOWN, self.name, reason="bound_hit", wall_time=elapsed,
|
|
848
|
+
detail=f"a formula nested {nesting} levels deep is deeper than "
|
|
849
|
+
"the tableau's recursive helpers can walk within the "
|
|
850
|
+
f"interpreter's recursion limit ({sys.getrecursionlimit()}): "
|
|
851
|
+
"no closed tableau was found")
|
|
852
|
+
if elapsed * 1000 >= timeout:
|
|
853
|
+
return Verdict(UNKNOWN, self.name, reason="timeout", wall_time=elapsed,
|
|
854
|
+
detail=f"no closed tableau within the {timeout} ms limit")
|
|
855
|
+
return Verdict(UNKNOWN, self.name, reason="bound_hit", wall_time=elapsed,
|
|
856
|
+
detail="no closed tableau within max_steps/max_terms")
|
|
857
|
+
|
|
858
|
+
|
|
859
|
+
class ResolutionBackend(ProverBackend):
|
|
860
|
+
"""The kit's own resolution prover (given-clause saturation).
|
|
861
|
+
|
|
862
|
+
The underlying bool API cannot distinguish saturation from a hit step
|
|
863
|
+
bound, so a False is reported as UNKNOWN/bound_hit — never as REFUTED.
|
|
864
|
+
Many-sorted input gets its sort axioms (every sort non-empty, every sorted
|
|
865
|
+
constant in its sort) as premise clauses from
|
|
866
|
+
:func:`~unicode_logic_kit.atp.resolution.prove`. The call's ``timeout`` bounds the
|
|
867
|
+
saturation; a run that used it up is ``timeout``, not ``bound_hit``.
|
|
868
|
+
"""
|
|
869
|
+
|
|
870
|
+
name = "resolution"
|
|
871
|
+
logics = frozenset({"fol"})
|
|
872
|
+
external = False
|
|
873
|
+
|
|
874
|
+
def available(self) -> bool:
|
|
875
|
+
return True
|
|
876
|
+
|
|
877
|
+
def decide(self, formula: Node, premises: Sequence[Node] = (),
|
|
878
|
+
timeout: int = 10000, **options) -> Verdict:
|
|
879
|
+
from .resolution import prove as resolution_prove
|
|
880
|
+
|
|
881
|
+
try:
|
|
882
|
+
proved, elapsed = _timed(
|
|
883
|
+
lambda: resolution_prove(list(premises), formula, timeout=timeout,
|
|
884
|
+
**options))
|
|
885
|
+
except NotImplementedError as exc:
|
|
886
|
+
return Verdict(UNKNOWN, self.name, reason="unsupported", detail=str(exc))
|
|
887
|
+
if proved:
|
|
888
|
+
return Verdict(PROVED, self.name, wall_time=elapsed)
|
|
889
|
+
if elapsed * 1000 >= timeout:
|
|
890
|
+
return Verdict(UNKNOWN, self.name, reason="timeout", wall_time=elapsed,
|
|
891
|
+
detail=f"not refuted within the {timeout} ms limit")
|
|
892
|
+
return Verdict(UNKNOWN, self.name, reason="bound_hit", wall_time=elapsed,
|
|
893
|
+
detail="not refuted within max_steps (saturation not distinguished)")
|
|
894
|
+
|
|
895
|
+
|
|
896
|
+
class ModelFinderBackend(ProverBackend):
|
|
897
|
+
"""The kit's finite model finder — a refutation-only semantic backend.
|
|
898
|
+
|
|
899
|
+
Finding a finite structure that satisfies the premises but not the
|
|
900
|
+
conclusion REFUTES the entailment; exhausting the size bound proves
|
|
901
|
+
nothing (FOL has no finite model property) → bound_hit. The call's
|
|
902
|
+
``timeout`` ends the search at a candidate structure; a search that it cut
|
|
903
|
+
off is ``timeout``, not ``bound_hit``.
|
|
904
|
+
"""
|
|
905
|
+
|
|
906
|
+
name = "modelfinder"
|
|
907
|
+
logics = frozenset({"fol"})
|
|
908
|
+
external = False
|
|
909
|
+
|
|
910
|
+
def available(self) -> bool:
|
|
911
|
+
return True
|
|
912
|
+
|
|
913
|
+
def decide(self, formula: Node, premises: Sequence[Node] = (),
|
|
914
|
+
timeout: int = 10000, **options) -> Verdict:
|
|
915
|
+
from ..semantics.modelfinder import search_countermodel
|
|
916
|
+
|
|
917
|
+
try:
|
|
918
|
+
search, elapsed = _timed(
|
|
919
|
+
lambda: search_countermodel(list(premises), formula, timeout=timeout,
|
|
920
|
+
**options))
|
|
921
|
+
except NotImplementedError as exc:
|
|
922
|
+
return Verdict(UNKNOWN, self.name, reason="unsupported", detail=str(exc))
|
|
923
|
+
if search.structure is not None:
|
|
924
|
+
return Verdict(REFUTED, self.name, wall_time=elapsed,
|
|
925
|
+
countermodel={"kind": "finite_structure",
|
|
926
|
+
"repr": repr(search.structure)})
|
|
927
|
+
if search.timed_out:
|
|
928
|
+
return Verdict(UNKNOWN, self.name, reason="timeout", wall_time=elapsed,
|
|
929
|
+
detail=f"no countermodel found within the {timeout} ms limit")
|
|
930
|
+
return Verdict(UNKNOWN, self.name, reason="bound_hit", wall_time=elapsed,
|
|
931
|
+
detail="no countermodel up to the size bound")
|
|
932
|
+
|
|
933
|
+
|
|
934
|
+
class ModalTableauBackend(ProverBackend):
|
|
935
|
+
"""The kit's labelled modal tableau — tri-state with Kripke witnesses."""
|
|
936
|
+
|
|
937
|
+
name = "modal-tableau"
|
|
938
|
+
logics = frozenset({"modal"})
|
|
939
|
+
external = False
|
|
940
|
+
|
|
941
|
+
def available(self) -> bool:
|
|
942
|
+
return True
|
|
943
|
+
|
|
944
|
+
def decide(self, formula: Node, premises: Sequence[Node] = (),
|
|
945
|
+
timeout: int = 10000, **options) -> Verdict:
|
|
946
|
+
from .modal_tableau import _decide_explained, modal_countermodel
|
|
947
|
+
|
|
948
|
+
goal = _implication(formula, premises)
|
|
949
|
+
try:
|
|
950
|
+
(status, too_deep), elapsed = _timed(
|
|
951
|
+
lambda: _decide_explained(goal, timeout=timeout, **options))
|
|
952
|
+
except NotImplementedError as exc:
|
|
953
|
+
return Verdict(UNKNOWN, self.name, logic="modal",
|
|
954
|
+
reason="unsupported", detail=str(exc))
|
|
955
|
+
if status == "valid":
|
|
956
|
+
return Verdict(PROVED, self.name, logic="modal", wall_time=elapsed)
|
|
957
|
+
if status == "invalid":
|
|
958
|
+
# The witness is a second search; it gets what is left of the limit.
|
|
959
|
+
cm = modal_countermodel(goal, timeout=max(1, int(timeout - elapsed * 1000)),
|
|
960
|
+
**options)
|
|
961
|
+
witness = None
|
|
962
|
+
if cm is not None:
|
|
963
|
+
from .kripke_enum import kripke_model_to_dict
|
|
964
|
+
witness = {"kind": "kripke", "repr": repr(cm),
|
|
965
|
+
"data": kripke_model_to_dict(cm)}
|
|
966
|
+
return Verdict(REFUTED, self.name, logic="modal", wall_time=elapsed,
|
|
967
|
+
countermodel=witness)
|
|
968
|
+
# modal_decide answers "unknown" both for an exhausted budget and for
|
|
969
|
+
# the temporal-closure operators it has no rule for (G/F/U/H/O/P/S,
|
|
970
|
+
# classified "unsupported" internally) — tell them apart here so the
|
|
971
|
+
# reason axis stays honest.
|
|
972
|
+
from .modal_tableau import _TEMPORAL_CLOSURE
|
|
973
|
+
if any(isinstance(n, _TEMPORAL_CLOSURE) for n in goal.walk()):
|
|
974
|
+
return Verdict(UNKNOWN, self.name, logic="modal", reason="unsupported",
|
|
975
|
+
wall_time=elapsed,
|
|
976
|
+
detail="temporal closure operators (G/F/U/…) have no "
|
|
977
|
+
"tableau rule here for the kit's general Kripke "
|
|
978
|
+
"frame — for the STANDARD linear-time reading, "
|
|
979
|
+
"the 'ltl-tableau' backend (atp.ltl_tableau) is "
|
|
980
|
+
"a complete decision procedure and the "
|
|
981
|
+
"definitive route; 'qml' and 'isabelle' decide "
|
|
982
|
+
"the general (possibly non-linear) frame "
|
|
983
|
+
"instead, soundly but not completely")
|
|
984
|
+
if too_deep:
|
|
985
|
+
# One of the tableau's walks over the formula ran into the interpreter's recursion
|
|
986
|
+
# limit, which is one of its bounds: the formula was not read to its end.
|
|
987
|
+
import sys
|
|
988
|
+
from .tableau import nesting_depth
|
|
989
|
+
return Verdict(UNKNOWN, self.name, logic="modal", reason="bound_hit",
|
|
990
|
+
wall_time=elapsed,
|
|
991
|
+
detail=f"a formula nested {nesting_depth(goal)} levels deep is deeper "
|
|
992
|
+
"than the tableau's recursive walks can follow within the "
|
|
993
|
+
f"interpreter's recursion limit ({sys.getrecursionlimit()}): "
|
|
994
|
+
"nothing was decided")
|
|
995
|
+
if elapsed * 1000 >= timeout:
|
|
996
|
+
return Verdict(UNKNOWN, self.name, logic="modal", reason="timeout",
|
|
997
|
+
wall_time=elapsed,
|
|
998
|
+
detail=f"no verdict within the {timeout} ms limit")
|
|
999
|
+
return Verdict(UNKNOWN, self.name, logic="modal", reason="bound_hit",
|
|
1000
|
+
wall_time=elapsed,
|
|
1001
|
+
detail="tableau budget exhausted (max_worlds/max_steps)")
|
|
1002
|
+
|
|
1003
|
+
|
|
1004
|
+
class QmlBackend(ProverBackend):
|
|
1005
|
+
"""Quantified modal logic via the FO embedding + Z3 — sound, incomplete.
|
|
1006
|
+
|
|
1007
|
+
``True`` is a proof; ``False`` only means "not proven" (undecidable
|
|
1008
|
+
fragment), so it is reported as UNKNOWN/incomplete, never as REFUTED.
|
|
1009
|
+
"""
|
|
1010
|
+
|
|
1011
|
+
name = "qml"
|
|
1012
|
+
logics = frozenset({"modal"})
|
|
1013
|
+
external = False
|
|
1014
|
+
|
|
1015
|
+
def available(self) -> bool:
|
|
1016
|
+
return True
|
|
1017
|
+
|
|
1018
|
+
def decide(self, formula: Node, premises: Sequence[Node] = (),
|
|
1019
|
+
timeout: int = 10000, **options) -> Verdict:
|
|
1020
|
+
from ..fol.qml import qml_is_valid
|
|
1021
|
+
|
|
1022
|
+
goal = _implication(formula, premises)
|
|
1023
|
+
try:
|
|
1024
|
+
proved, elapsed = _timed(
|
|
1025
|
+
lambda: qml_is_valid(goal, timeout=timeout, **options))
|
|
1026
|
+
except NotImplementedError as exc:
|
|
1027
|
+
return Verdict(UNKNOWN, self.name, logic="modal",
|
|
1028
|
+
reason="unsupported", detail=str(exc))
|
|
1029
|
+
if proved:
|
|
1030
|
+
return Verdict(PROVED, self.name, logic="modal", wall_time=elapsed)
|
|
1031
|
+
return Verdict(UNKNOWN, self.name, logic="modal", reason="incomplete",
|
|
1032
|
+
wall_time=elapsed,
|
|
1033
|
+
detail="Z3 did not close the embedded goal (FO modal logic is undecidable)")
|
|
1034
|
+
|
|
1035
|
+
|
|
1036
|
+
# ---------------------------------------------------------------------------
|
|
1037
|
+
# External backends
|
|
1038
|
+
# ---------------------------------------------------------------------------
|
|
1039
|
+
|
|
1040
|
+
class IsabelleBackend(ProverBackend):
|
|
1041
|
+
"""Isabelle/HOL via the kit's runner — the most trusted external route.
|
|
1042
|
+
|
|
1043
|
+
Deliberately NOT in any default chain: one decision spawns a real
|
|
1044
|
+
``isabelle build`` (minutes of wall time), so it runs only when requested
|
|
1045
|
+
by name. ``ModalVerdict``/``FolVerdict`` map 1:1 onto :class:`Verdict`
|
|
1046
|
+
(their ``infra_error`` becomes reason="infra" detail on UNKNOWN).
|
|
1047
|
+
|
|
1048
|
+
The classical route decides with ``native_equality=True``: ``=`` is HOL
|
|
1049
|
+
identity, as in every other backend's semantics. With the embedding's
|
|
1050
|
+
uninterpreted ``feq`` a nitpick countermodel is one for FOL *without*
|
|
1051
|
+
identity, so ``∀x (x = x)`` came back REFUTED. Passing
|
|
1052
|
+
``native_equality=False`` is refused for that reason. The modal route makes
|
|
1053
|
+
the same choice by a different road: since 0.30.0 its embedding reads ``=``
|
|
1054
|
+
as RIGID identity (HOL's own, no world argument) and the Kripke evaluator it
|
|
1055
|
+
falls back on for a countermodel refuses an identity atom by name rather
|
|
1056
|
+
than reading it as an ordinary atom, so neither half of that route can
|
|
1057
|
+
answer a question about identity without interpreting it.
|
|
1058
|
+
"""
|
|
1059
|
+
|
|
1060
|
+
name = "isabelle"
|
|
1061
|
+
logics = frozenset({"fol", "modal"})
|
|
1062
|
+
external = True
|
|
1063
|
+
|
|
1064
|
+
def available(self) -> bool:
|
|
1065
|
+
from ..hol.isabelle_runner import isabelle_available
|
|
1066
|
+
return isabelle_available()
|
|
1067
|
+
|
|
1068
|
+
def available_for(self, options: dict) -> bool:
|
|
1069
|
+
"""Whether the installation :meth:`decide` will run is there: the one a call
|
|
1070
|
+
names with ``install=`` (an :class:`~unicode_logic_kit.hol.isabelle_runner.IsabelleInstall`,
|
|
1071
|
+
which the runner uses instead of looking), else :meth:`available`."""
|
|
1072
|
+
if options.get("install") is not None:
|
|
1073
|
+
return True
|
|
1074
|
+
return self.available()
|
|
1075
|
+
|
|
1076
|
+
def decide(self, formula: Node, premises: Sequence[Node] = (),
|
|
1077
|
+
timeout: int = 10000, **options) -> Verdict:
|
|
1078
|
+
goal = _implication(formula, premises)
|
|
1079
|
+
logic = options.pop("logic", None) or (
|
|
1080
|
+
"modal" if _looks_modal(goal) else "fol")
|
|
1081
|
+
|
|
1082
|
+
if logic == "modal":
|
|
1083
|
+
from ..hol.isabelle_runner import isabelle_decide_modal
|
|
1084
|
+
verdict, elapsed = _timed(lambda: isabelle_decide_modal(goal, **options))
|
|
1085
|
+
else:
|
|
1086
|
+
from ..hol.isabelle_runner import isabelle_decide_fol
|
|
1087
|
+
if not options.pop("native_equality", True):
|
|
1088
|
+
raise ValueError(
|
|
1089
|
+
"isabelle backend: native_equality=False would decide FOL without "
|
|
1090
|
+
"identity (uninterpreted feq), so a REFUTED verdict could contradict "
|
|
1091
|
+
"every other backend; call hol.isabelle_runner.isabelle_decide_fol "
|
|
1092
|
+
"directly for that reading.")
|
|
1093
|
+
verdict, elapsed = _timed(
|
|
1094
|
+
lambda: isabelle_decide_fol(goal, native_equality=True, **options))
|
|
1095
|
+
|
|
1096
|
+
if verdict.status == "valid":
|
|
1097
|
+
return Verdict(PROVED, self.name, logic=logic, wall_time=elapsed,
|
|
1098
|
+
detail=verdict.method)
|
|
1099
|
+
if verdict.status == "invalid":
|
|
1100
|
+
witness = ({"kind": "nitpick" if logic == "fol" else "kripke",
|
|
1101
|
+
"repr": verdict.countermodel}
|
|
1102
|
+
if verdict.countermodel else None)
|
|
1103
|
+
return Verdict(REFUTED, self.name, logic=logic, wall_time=elapsed,
|
|
1104
|
+
countermodel=witness)
|
|
1105
|
+
reason = "infra" if verdict.infra_error else "incomplete"
|
|
1106
|
+
return Verdict(UNKNOWN, self.name, logic=logic, reason=reason,
|
|
1107
|
+
wall_time=elapsed, detail=verdict.infra_error)
|
|
1108
|
+
|
|
1109
|
+
|
|
1110
|
+
class Prover9Backend(ProverBackend):
|
|
1111
|
+
"""Prover9 via subprocess. Discovery: $UFK_PROVER9, then PATH.
|
|
1112
|
+
|
|
1113
|
+
Set ``UFK_PROVER9_WSL=1`` when the binary is a Linux build reached through
|
|
1114
|
+
WSL on a Windows host (``$UFK_PROVER9`` then names the path INSIDE WSL, for
|
|
1115
|
+
example ``/mnt/d/prover9/Prover9-LADR-2026-8A/bin/prover9``); the ``use_wsl``
|
|
1116
|
+
option does the same for one call, as for the Vampire backend.
|
|
1117
|
+
|
|
1118
|
+
A run without ``THEOREM PROVED`` is UNKNOWN / ``"incomplete"`` (Prover9's
|
|
1119
|
+
exit does not certify invalidity), EXCEPT a problem Prover9 refused to read
|
|
1120
|
+
(its fatal-error exit) or a binary that could not be started (a path that
|
|
1121
|
+
does not exist inside WSL): that is ERROR / ``"infra"`` with Prover9's own
|
|
1122
|
+
(or the shell's) message in ``detail``, and a run the kit stopped because the
|
|
1123
|
+
``timeout`` (milliseconds, as for every backend) ran out: that is UNKNOWN /
|
|
1124
|
+
``"timeout"``.
|
|
1125
|
+
|
|
1126
|
+
A free variable of the problem is one unknown element, the same in every premise and in
|
|
1127
|
+
the conclusion (the problem writer replaces it by a constant of its own: Prover9 itself
|
|
1128
|
+
would close each formula universally, which is another question), so ``P(x) ⊢ P(alpha)``
|
|
1129
|
+
is not proved. A Łukasiewicz connective has no classical reading and is refused:
|
|
1130
|
+
UNKNOWN / ``"unsupported"`` with the name of the connective in ``detail``.
|
|
1131
|
+
"""
|
|
1132
|
+
|
|
1133
|
+
name = "prover9"
|
|
1134
|
+
logics = frozenset({"fol"})
|
|
1135
|
+
external = True
|
|
1136
|
+
|
|
1137
|
+
@staticmethod
|
|
1138
|
+
def _binary() -> Optional[str]:
|
|
1139
|
+
import os
|
|
1140
|
+
return os.environ.get("UFK_PROVER9") or shutil.which("prover9")
|
|
1141
|
+
|
|
1142
|
+
@staticmethod
|
|
1143
|
+
def _uses_wsl() -> bool:
|
|
1144
|
+
"""Whether ``$UFK_PROVER9_WSL`` says the binary lives inside WSL."""
|
|
1145
|
+
import os
|
|
1146
|
+
return os.environ.get("UFK_PROVER9_WSL") == "1"
|
|
1147
|
+
|
|
1148
|
+
def available(self) -> bool:
|
|
1149
|
+
return self.available_for({})
|
|
1150
|
+
|
|
1151
|
+
def available_for(self, options: dict) -> bool:
|
|
1152
|
+
"""Whether the binary the run will use is there: the one :meth:`decide`
|
|
1153
|
+
resolves (``prover9_path=``, else ``$UFK_PROVER9``, else ``prover9`` on
|
|
1154
|
+
PATH), looked for where :meth:`decide` will run it. With the WSL switch on
|
|
1155
|
+
(``use_wsl=True`` in ``options``, else ``$UFK_PROVER9_WSL=1``) that is
|
|
1156
|
+
inside WSL (see :func:`_wsl_has_binary`), not on this host's PATH. A native
|
|
1157
|
+
``prover9_path=`` that names no file this host can start, and no command on
|
|
1158
|
+
PATH, is refused here by name instead of failing inside the run; what
|
|
1159
|
+
``$UFK_PROVER9`` names is taken as the installation the user pointed at."""
|
|
1160
|
+
explicit = options.get("prover9_path")
|
|
1161
|
+
path = explicit or self._binary()
|
|
1162
|
+
if path is None:
|
|
1163
|
+
return False
|
|
1164
|
+
if options.get("use_wsl", self._uses_wsl()):
|
|
1165
|
+
return _wsl_has_binary(path)
|
|
1166
|
+
if explicit:
|
|
1167
|
+
return _native_command_exists(explicit)
|
|
1168
|
+
return True
|
|
1169
|
+
|
|
1170
|
+
#: The banner line Prover9 prints first: ``Prover9 (64) version 2026-8A, August 2026.``
|
|
1171
|
+
_BANNER_LINE = re.compile(r"^[ \t]*(Prover9\b[^\r\n]*\bversion\b[^\r\n]*?)[ \t]*$", re.MULTILINE)
|
|
1172
|
+
|
|
1173
|
+
@classmethod
|
|
1174
|
+
def _banner(cls, path: str, use_wsl: bool) -> Optional[str]:
|
|
1175
|
+
"""The banner line of the Prover9 at ``path`` (through ``wsl.exe`` when
|
|
1176
|
+
``use_wsl``), or ``None``.
|
|
1177
|
+
|
|
1178
|
+
Prover9 has no ``--version``: it takes the flag for a resume directory and
|
|
1179
|
+
ends with a fatal error (exit 1), so :func:`_binary_version` reads nothing
|
|
1180
|
+
from it. ``-h`` prints the banner first and exits 0 at once; the line that
|
|
1181
|
+
names the version is the one that starts with ``Prover9`` and holds the word
|
|
1182
|
+
``version``. The tool never reads this process's standard input (``stdin`` is
|
|
1183
|
+
the null device) and the probe has a time limit, so it cannot block. The
|
|
1184
|
+
result, a miss too, is memoized per ``(path, use_wsl)`` in the process-local
|
|
1185
|
+
version cache that :func:`_binary_version` fills for the other tools, never
|
|
1186
|
+
probed twice; nothing here raises.
|
|
1187
|
+
"""
|
|
1188
|
+
key = (path, use_wsl)
|
|
1189
|
+
if key in _VERSION_CACHE:
|
|
1190
|
+
return _VERSION_CACHE[key]
|
|
1191
|
+
import subprocess
|
|
1192
|
+
|
|
1193
|
+
banner: Optional[str] = None
|
|
1194
|
+
try:
|
|
1195
|
+
cmd = ["wsl.exe", path, "-h"] if use_wsl else [path, "-h"]
|
|
1196
|
+
proc = subprocess.run(cmd, capture_output=True, text=True, timeout=10,
|
|
1197
|
+
stdin=subprocess.DEVNULL)
|
|
1198
|
+
for stream in (proc.stdout, proc.stderr):
|
|
1199
|
+
found = cls._BANNER_LINE.search(stream or "")
|
|
1200
|
+
if found:
|
|
1201
|
+
banner = found.group(1)
|
|
1202
|
+
break
|
|
1203
|
+
except (OSError, subprocess.TimeoutExpired, UnicodeError):
|
|
1204
|
+
banner = None
|
|
1205
|
+
_VERSION_CACHE[key] = banner
|
|
1206
|
+
return banner
|
|
1207
|
+
|
|
1208
|
+
def solver_version(self) -> Optional[str]:
|
|
1209
|
+
"""Prover9's banner line (``Prover9 (64) version 2026-8A, August 2026.``), for
|
|
1210
|
+
whichever binary discovery (``$UFK_PROVER9``/PATH, ``$UFK_PROVER9_WSL``)
|
|
1211
|
+
resolves to right now — the same default :meth:`available` uses. Read by
|
|
1212
|
+
:meth:`_banner` (``prover9 -h``: the binary has no ``--version``), memoized
|
|
1213
|
+
per ``(binary, use_wsl)`` for the life of the process, never reading this
|
|
1214
|
+
process's standard input and never blocking. ``None`` when no binary is
|
|
1215
|
+
currently discoverable or the banner cannot be read; a per-call
|
|
1216
|
+
``prover9_path=``/``use_wsl=`` override is reflected in THAT call's own
|
|
1217
|
+
``Verdict.solver_version`` (see :meth:`decide`), not here —
|
|
1218
|
+
see :func:`_binary_version`'s module-level docstring section for why
|
|
1219
|
+
this method answers for the default binary only.
|
|
1220
|
+
"""
|
|
1221
|
+
path = self._binary()
|
|
1222
|
+
if path is None:
|
|
1223
|
+
return None
|
|
1224
|
+
return self._banner(path, self._uses_wsl())
|
|
1225
|
+
|
|
1226
|
+
def decide(self, formula: Node, premises: Sequence[Node] = (),
|
|
1227
|
+
timeout: int = 10000, **options) -> Verdict:
|
|
1228
|
+
from .prover9_entailment import (
|
|
1229
|
+
Prover9Rejected, Prover9TimedOut, check_logical_entailment,
|
|
1230
|
+
)
|
|
1231
|
+
|
|
1232
|
+
path = options.pop("prover9_path", None) or self._binary()
|
|
1233
|
+
if path is None:
|
|
1234
|
+
raise BackendUnavailable(
|
|
1235
|
+
"prover9: no binary found (set $UFK_PROVER9 or put 'prover9' on PATH)")
|
|
1236
|
+
use_wsl = options.pop("use_wsl", self._uses_wsl())
|
|
1237
|
+
solver_version = self._banner(path, use_wsl)
|
|
1238
|
+
start = time.perf_counter()
|
|
1239
|
+
try:
|
|
1240
|
+
proved, elapsed = _timed(
|
|
1241
|
+
lambda: check_logical_entailment(list(premises), formula, path,
|
|
1242
|
+
raise_on_rejection=True,
|
|
1243
|
+
timeout=max(1, timeout // 1000),
|
|
1244
|
+
raise_on_timeout=True,
|
|
1245
|
+
use_wsl=use_wsl))
|
|
1246
|
+
except NotImplementedError as exc:
|
|
1247
|
+
return Verdict(UNKNOWN, self.name, reason="unsupported",
|
|
1248
|
+
solver_version=solver_version, detail=str(exc))
|
|
1249
|
+
except Prover9TimedOut as exc:
|
|
1250
|
+
# The kit's wall-clock budget ran out and the kit stopped Prover9: a
|
|
1251
|
+
# timeout, said as one -- not a search that "found no proof".
|
|
1252
|
+
return Verdict(UNKNOWN, self.name, reason="timeout",
|
|
1253
|
+
wall_time=time.perf_counter() - start,
|
|
1254
|
+
solver_version=solver_version, detail=str(exc))
|
|
1255
|
+
except Prover9Rejected as exc:
|
|
1256
|
+
# Prover9 refused to read the problem (its documented fatal exit):
|
|
1257
|
+
# an error carrying Prover9's own message, not "found no proof".
|
|
1258
|
+
return Verdict(ERROR, self.name, reason="infra",
|
|
1259
|
+
solver_version=solver_version,
|
|
1260
|
+
detail=_rejection_detail(
|
|
1261
|
+
"prover9", exc.output,
|
|
1262
|
+
f"exit code {exc.returncode} and no THEOREM PROVED line"))
|
|
1263
|
+
except OSError as exc:
|
|
1264
|
+
return Verdict(ERROR, self.name, reason="infra",
|
|
1265
|
+
solver_version=solver_version, detail=str(exc))
|
|
1266
|
+
if proved:
|
|
1267
|
+
return Verdict(PROVED, self.name, wall_time=elapsed,
|
|
1268
|
+
solver_version=solver_version)
|
|
1269
|
+
return Verdict(UNKNOWN, self.name, reason="incomplete", wall_time=elapsed,
|
|
1270
|
+
solver_version=solver_version,
|
|
1271
|
+
detail="Prover9 found no proof (its exit does not certify invalidity)")
|
|
1272
|
+
|
|
1273
|
+
|
|
1274
|
+
class VampireBackend(ProverBackend):
|
|
1275
|
+
"""Vampire via subprocess. Discovery: $UFK_VAMPIRE, then PATH.
|
|
1276
|
+
|
|
1277
|
+
Set ``UFK_VAMPIRE_WSL=1`` when the binary is a Linux build reached
|
|
1278
|
+
through WSL on a Windows host.
|
|
1279
|
+
|
|
1280
|
+
Decides through the SZS/TSTP route
|
|
1281
|
+
(:func:`~unicode_logic_kit.atp.vampire_entailment.check_entailment_vampire_detailed`):
|
|
1282
|
+
the verdict's ``szs_status`` is Vampire's own status line verbatim, a
|
|
1283
|
+
``CounterSatisfiable`` answer becomes an honest REFUTED (Vampire's
|
|
1284
|
+
saturation certifies invalidity, though it yields no model structure —
|
|
1285
|
+
``countermodel`` stays ``None``), and a PROVED verdict carries the parsed
|
|
1286
|
+
TSTP derivation DAG in ``proof``. A problem Vampire REFUSES to read (its
|
|
1287
|
+
``User error: ...`` or parse error, with no SZS status line) is ERROR /
|
|
1288
|
+
``"infra"`` with Vampire's own message in ``detail`` — never an UNKNOWN
|
|
1289
|
+
that reads like a timeout; its own ``GaveUp`` / ``ResourceOut`` /
|
|
1290
|
+
``Unknown`` and a real subprocess timeout keep their UNKNOWN verdicts. So
|
|
1291
|
+
does a run that printed no SZS line but DID print how its search ended
|
|
1292
|
+
(``Termination reason: Refutation not found, incomplete strategy`` —
|
|
1293
|
+
Vampire's way of giving up on a non-theorem of a typed-arithmetic problem):
|
|
1294
|
+
UNKNOWN / ``"incomplete"``, with the termination reason in ``detail``.
|
|
1295
|
+
|
|
1296
|
+
Options (through ``**options``, like the E / Zipperposition backend, and
|
|
1297
|
+
read the same way): ``vampire_path`` / ``use_wsl`` pick the binary, and
|
|
1298
|
+
``tff`` / ``sort`` pick the TPTP dialect it is given. ``sort`` (``'real'``
|
|
1299
|
+
or ``'int'``; ``None`` by default) opts into the single-numeric-sort typed
|
|
1300
|
+
arithmetic route UNCONDITIONALLY when given, taking priority over ``tff``;
|
|
1301
|
+
with ``sort=None``, ``tff=None`` tries the many-sorted typed route whenever
|
|
1302
|
+
the problem uses a sort, and ``True`` / ``False`` force the typed / the
|
|
1303
|
+
``fof`` route. The typed writer refuses a problem whose typed text would not
|
|
1304
|
+
ask the kit's question (see :class:`~unicode_logic_kit.atp.tptp_tff.Tf0Refusal`):
|
|
1305
|
+
with ``tff=None`` the problem is then written as ``fof`` and the verdict's
|
|
1306
|
+
``detail`` says that it was, and why; with ``tff=True`` the refusal is the
|
|
1307
|
+
verdict, ``UNKNOWN`` / ``"unsupported"`` with the writer's message in
|
|
1308
|
+
``detail`` (never an exception). The options reach
|
|
1309
|
+
:func:`~unicode_logic_kit.atp.vampire_entailment.check_entailment_vampire_detailed`
|
|
1310
|
+
unchanged.
|
|
1311
|
+
|
|
1312
|
+
**Which premises a proof used.** Vampire prints ``unknown`` for the name of an
|
|
1313
|
+
axiom unless it is asked for the names, so the backend asks (``axiom_names=``,
|
|
1314
|
+
on unless a call turns it off) and names the premises' ``axiom`` lines
|
|
1315
|
+
(``premise_names=``, ``premise_<i>`` when not given). A PROVED verdict then
|
|
1316
|
+
carries the caller's premise indices that the proof's axiom leaves are in
|
|
1317
|
+
``relevant_premises`` (the sorted 0-based indices into the caller's ``premises``,
|
|
1318
|
+
``()`` for a proof that needs none), and its ``detail`` names the background facts
|
|
1319
|
+
of a sorted reading the proof used (the non-emptiness of a sort, the membership
|
|
1320
|
+
of a sorted constant), which are not premises. ``relevant_premises`` stays
|
|
1321
|
+
``None`` for a verdict that is not PROVED, and for a proof whose axiom leaves
|
|
1322
|
+
cannot be read back through the problem's own names (never a guess). The set is
|
|
1323
|
+
sound but not necessarily minimal, like every report of this kind.
|
|
1324
|
+
"""
|
|
1325
|
+
|
|
1326
|
+
name = "vampire"
|
|
1327
|
+
logics = frozenset({"fol"})
|
|
1328
|
+
external = True
|
|
1329
|
+
|
|
1330
|
+
@staticmethod
|
|
1331
|
+
def _binary() -> Optional[str]:
|
|
1332
|
+
import os
|
|
1333
|
+
return os.environ.get("UFK_VAMPIRE") or shutil.which("vampire")
|
|
1334
|
+
|
|
1335
|
+
def available(self) -> bool:
|
|
1336
|
+
return self.available_for({})
|
|
1337
|
+
|
|
1338
|
+
def available_for(self, options: dict) -> bool:
|
|
1339
|
+
"""Whether the binary the run will use is there: the one :meth:`decide`
|
|
1340
|
+
resolves (``vampire_path=``, else ``$UFK_VAMPIRE``, else ``vampire`` on
|
|
1341
|
+
PATH), looked for where :meth:`decide` will run it. With the WSL switch on
|
|
1342
|
+
(``use_wsl=True`` in ``options``, else ``$UFK_VAMPIRE_WSL=1``) that is
|
|
1343
|
+
inside WSL (see :func:`_wsl_has_binary`), not on this host's PATH. A native
|
|
1344
|
+
``vampire_path=`` that names no file this host can start, and no command on
|
|
1345
|
+
PATH, is refused here by name instead of failing inside the run; what
|
|
1346
|
+
``$UFK_VAMPIRE`` names is taken as the installation the user pointed at."""
|
|
1347
|
+
import os
|
|
1348
|
+
|
|
1349
|
+
explicit = options.get("vampire_path")
|
|
1350
|
+
path = explicit or self._binary()
|
|
1351
|
+
if path is None:
|
|
1352
|
+
return False
|
|
1353
|
+
if options.get("use_wsl", os.environ.get("UFK_VAMPIRE_WSL") == "1"):
|
|
1354
|
+
return _wsl_has_binary(path)
|
|
1355
|
+
if explicit:
|
|
1356
|
+
return _native_command_exists(explicit)
|
|
1357
|
+
return True
|
|
1358
|
+
|
|
1359
|
+
def solver_version(self) -> Optional[str]:
|
|
1360
|
+
"""Vampire's ``--version`` banner (first line), for whichever binary
|
|
1361
|
+
discovery (``$UFK_VAMPIRE``/PATH, ``$UFK_VAMPIRE_WSL``) resolves to
|
|
1362
|
+
right now — the same default :meth:`available` uses. Memoized per
|
|
1363
|
+
``(binary, use_wsl)`` for the life of the process via
|
|
1364
|
+
:func:`_binary_version`. ``None`` when no binary is currently
|
|
1365
|
+
discoverable; a per-call ``vampire_path=``/``use_wsl=`` override is
|
|
1366
|
+
reflected in THAT call's own ``Verdict.solver_version`` (see
|
|
1367
|
+
:meth:`decide`), not here — see :func:`_binary_version`'s
|
|
1368
|
+
module-level docstring section for why this method answers for the
|
|
1369
|
+
default binary only.
|
|
1370
|
+
"""
|
|
1371
|
+
import os
|
|
1372
|
+
|
|
1373
|
+
path = self._binary()
|
|
1374
|
+
if path is None:
|
|
1375
|
+
return None
|
|
1376
|
+
use_wsl = os.environ.get("UFK_VAMPIRE_WSL") == "1"
|
|
1377
|
+
return _binary_version(path, use_wsl)
|
|
1378
|
+
|
|
1379
|
+
def decide(self, formula: Node, premises: Sequence[Node] = (),
|
|
1380
|
+
timeout: int = 10000, **options) -> Verdict:
|
|
1381
|
+
import os
|
|
1382
|
+
from .vampire_entailment import _ended_by_itself, check_entailment_vampire_detailed
|
|
1383
|
+
from .tstp import background_use_note
|
|
1384
|
+
|
|
1385
|
+
path = options.pop("vampire_path", None) or self._binary()
|
|
1386
|
+
if path is None:
|
|
1387
|
+
raise BackendUnavailable(
|
|
1388
|
+
"vampire: no binary found (set $UFK_VAMPIRE or put 'vampire' on PATH)")
|
|
1389
|
+
use_wsl = options.pop("use_wsl", os.environ.get("UFK_VAMPIRE_WSL") == "1")
|
|
1390
|
+
tff = options.pop("tff", None)
|
|
1391
|
+
if tff not in (None, True, False):
|
|
1392
|
+
raise ValueError(
|
|
1393
|
+
f"vampire: tff= is None (decide from the formula), True or False, "
|
|
1394
|
+
f"got {tff!r}.")
|
|
1395
|
+
sort = options.pop("sort", None)
|
|
1396
|
+
premise_names = options.pop("premise_names", None)
|
|
1397
|
+
axiom_names = options.pop("axiom_names", True)
|
|
1398
|
+
solver_version = _binary_version(path, use_wsl)
|
|
1399
|
+
try:
|
|
1400
|
+
result, elapsed = _timed(lambda: check_entailment_vampire_detailed(
|
|
1401
|
+
list(premises), formula, path,
|
|
1402
|
+
timeout=max(1, timeout // 1000), use_wsl=use_wsl,
|
|
1403
|
+
tff=tff, sort=sort, premise_names=premise_names,
|
|
1404
|
+
axiom_names=axiom_names))
|
|
1405
|
+
except NotImplementedError as exc:
|
|
1406
|
+
return Verdict(UNKNOWN, self.name, reason="unsupported",
|
|
1407
|
+
solver_version=solver_version, detail=str(exc))
|
|
1408
|
+
except OSError as exc:
|
|
1409
|
+
return Verdict(ERROR, self.name, reason="infra",
|
|
1410
|
+
solver_version=solver_version, detail=str(exc))
|
|
1411
|
+
status, reason, szs = result["status"], result["reason"], result["szs_status"]
|
|
1412
|
+
if status == ERROR:
|
|
1413
|
+
# Vampire read the problem and refused it (a type or syntax error:
|
|
1414
|
+
# "User error: ..."), or failed outright. Its own message goes in
|
|
1415
|
+
# the detail -- a refusal must not read like running out of time.
|
|
1416
|
+
evidence = ("no SZS status line in its output" if szs is None
|
|
1417
|
+
else f"SZS status {szs}")
|
|
1418
|
+
detail = _rejection_detail("vampire", result.get("output_excerpt", ""),
|
|
1419
|
+
evidence)
|
|
1420
|
+
else:
|
|
1421
|
+
detail = (f"SZS status {szs}" if szs is not None
|
|
1422
|
+
else "no SZS status line in Vampire's output")
|
|
1423
|
+
if szs is None and status == UNKNOWN:
|
|
1424
|
+
# No SZS line: say what DID happen, not what is missing. Either it
|
|
1425
|
+
# read the problem and said how its search ended, or the kit's own
|
|
1426
|
+
# budget ran out and the kit stopped it before it printed a verdict.
|
|
1427
|
+
ended = _ended_by_itself(result.get("output_excerpt", ""))
|
|
1428
|
+
if ended is not None:
|
|
1429
|
+
detail = (f"Vampire's search ended on its own (Termination reason: "
|
|
1430
|
+
f"{ended[1]})")
|
|
1431
|
+
elif reason == "timeout":
|
|
1432
|
+
detail = (f"the budget of this call ({max(1, timeout // 1000)} s) ran "
|
|
1433
|
+
"out and the kit stopped Vampire before it printed a verdict")
|
|
1434
|
+
relevant = None
|
|
1435
|
+
if status == PROVED:
|
|
1436
|
+
# The caller's premises the proof's axiom leaves are, and the background
|
|
1437
|
+
# facts of a sorted reading it used (not premises), as the proof printed them.
|
|
1438
|
+
if result.get("relevant_premises") is not None:
|
|
1439
|
+
relevant = tuple(result["relevant_premises"])
|
|
1440
|
+
detail += background_use_note(tuple(result.get("background_used") or ()))
|
|
1441
|
+
if result.get("tff_fallback"):
|
|
1442
|
+
# tff=None tried the typed writer for a sorted problem and it refused:
|
|
1443
|
+
# the answer is that of the fof text, and the caller is told so and why.
|
|
1444
|
+
detail = f"{detail}; {result['tff_fallback']}"
|
|
1445
|
+
return Verdict(status, self.name, reason=reason, szs_status=szs,
|
|
1446
|
+
wall_time=elapsed, solver_version=solver_version,
|
|
1447
|
+
proof=result["derivation"] if status == PROVED else None,
|
|
1448
|
+
detail=detail, relevant_premises=relevant)
|
|
1449
|
+
|
|
1450
|
+
|
|
1451
|
+
# ---------------------------------------------------------------------------
|
|
1452
|
+
# Registry
|
|
1453
|
+
# ---------------------------------------------------------------------------
|
|
1454
|
+
|
|
1455
|
+
def _looks_modal(node: Node) -> bool:
|
|
1456
|
+
"""Route helper: does the formula contain any modal-family operator?"""
|
|
1457
|
+
from .modal_tableau import has_modal
|
|
1458
|
+
return has_modal(node)
|
|
1459
|
+
|
|
1460
|
+
|
|
1461
|
+
_REGISTRY: Dict[str, ProverBackend] = {}
|
|
1462
|
+
|
|
1463
|
+
|
|
1464
|
+
def register_backend(backend: ProverBackend) -> None:
|
|
1465
|
+
"""Register (or replace) a backend under ``backend.name``.
|
|
1466
|
+
|
|
1467
|
+
Third-party code can plug in its own :class:`ProverBackend` here and it
|
|
1468
|
+
becomes addressable by every ``prove(backends=[...])`` call.
|
|
1469
|
+
"""
|
|
1470
|
+
if not backend.name:
|
|
1471
|
+
raise ValueError("register_backend: backend must set a non-empty name")
|
|
1472
|
+
_REGISTRY[backend.name] = backend
|
|
1473
|
+
|
|
1474
|
+
|
|
1475
|
+
_b: ProverBackend # one name for both registration loops: each backend is a ProverBackend
|
|
1476
|
+
for _b in (Z3Backend(), TableauBackend(), ResolutionBackend(),
|
|
1477
|
+
ModelFinderBackend(), ModalTableauBackend(), QmlBackend(),
|
|
1478
|
+
IsabelleBackend(), Prover9Backend(), VampireBackend()):
|
|
1479
|
+
register_backend(_b)
|
|
1480
|
+
|
|
1481
|
+
# The remaining stock backends live in their own modules (an optional
|
|
1482
|
+
# dependency, a heavier search, an external prover family) and import THIS
|
|
1483
|
+
# module's public contract — so they are pulled in here, after that contract
|
|
1484
|
+
# is fully defined. A deliberate bottom-of-registration import, not an
|
|
1485
|
+
# accident: keeping the whole stock registry in one place guarantees that
|
|
1486
|
+
# `default_chain` and `_REGISTRY` can never disagree about what exists,
|
|
1487
|
+
# whichever submodule a process imports first.
|
|
1488
|
+
from .clingo_backend import ClingoBackend # noqa: E402
|
|
1489
|
+
from .cvc5_backend import Cvc5Backend # noqa: E402
|
|
1490
|
+
from .eprover_backend import EProverBackend, ZipperpositionBackend # noqa: E402
|
|
1491
|
+
from .hets_backend import HetsBackend # noqa: E402
|
|
1492
|
+
from .kripke_enum import KripkeEnumBackend # noqa: E402
|
|
1493
|
+
from .leo3_backend import Leo3Backend # noqa: E402
|
|
1494
|
+
from .ltl_tableau import LtlTableauBackend # noqa: E402
|
|
1495
|
+
from .minizinc_backend import MinizincBackend # noqa: E402
|
|
1496
|
+
from .nanocop_backend import NanocopBackend # noqa: E402
|
|
1497
|
+
from .twee_backend import TweeBackend # noqa: E402
|
|
1498
|
+
# Five individually-reasoned per-logic adapters (C10) — see logic_backends'
|
|
1499
|
+
# module docstring for why these are NOT a uniform bulk registration.
|
|
1500
|
+
from .logic_backends import ( # noqa: E402
|
|
1501
|
+
IntBackend, LambekBackend, IllBackend, RelevantBackend, HybridBackend,
|
|
1502
|
+
)
|
|
1503
|
+
|
|
1504
|
+
for _b in (ClingoBackend(), Cvc5Backend(), EProverBackend(), HetsBackend(),
|
|
1505
|
+
KripkeEnumBackend(), Leo3Backend(), LtlTableauBackend(), MinizincBackend(),
|
|
1506
|
+
NanocopBackend(), TweeBackend(), ZipperpositionBackend(),
|
|
1507
|
+
IntBackend(), LambekBackend(), IllBackend(), RelevantBackend(),
|
|
1508
|
+
HybridBackend()):
|
|
1509
|
+
register_backend(_b)
|
|
1510
|
+
|
|
1511
|
+
|
|
1512
|
+
def get_backend(name: str) -> ProverBackend:
|
|
1513
|
+
"""Return the backend registered under ``name`` (ValueError if unknown)."""
|
|
1514
|
+
if name not in _REGISTRY:
|
|
1515
|
+
raise ValueError(
|
|
1516
|
+
f"get_backend: unknown backend {name!r} (registered: {sorted(_REGISTRY)})")
|
|
1517
|
+
return _REGISTRY[name]
|
|
1518
|
+
|
|
1519
|
+
|
|
1520
|
+
def available_backends(logic: Optional[str] = None) -> Tuple[str, ...]:
|
|
1521
|
+
"""Names of the currently-available backends, optionally per logic."""
|
|
1522
|
+
names = []
|
|
1523
|
+
for name, backend in _REGISTRY.items():
|
|
1524
|
+
if logic is not None and logic not in backend.logics:
|
|
1525
|
+
continue
|
|
1526
|
+
if backend.available():
|
|
1527
|
+
names.append(name)
|
|
1528
|
+
return tuple(sorted(names))
|
|
1529
|
+
|
|
1530
|
+
|
|
1531
|
+
# Default chains: fast, deterministic, and NEVER silently expensive — the
|
|
1532
|
+
# external minutes-per-call route (isabelle) must be requested by name.
|
|
1533
|
+
# "kripke-enum" sits between the tableau and qml on purpose: it is the only
|
|
1534
|
+
# default-chain member that can REFUTE a temporal-closure formula (the
|
|
1535
|
+
# tableau reports those unsupported, qml is proof-only), and its bounded
|
|
1536
|
+
# enumeration settles the refutation side before qml's Z3 call can burn its
|
|
1537
|
+
# full timeout failing to prove an invalid goal.
|
|
1538
|
+
#
|
|
1539
|
+
# The five substructural/non-classical entries (C10) are each a SINGLETON
|
|
1540
|
+
# chain: every one of these logics currently has exactly one decision route
|
|
1541
|
+
# in the kit, so there is no actual "portfolio race" to order here (unlike
|
|
1542
|
+
# "fol"/"modal" above) — see logic_backends' module docstring for why these
|
|
1543
|
+
# five are individually-reasoned adapters, not a uniform bulk registration.
|
|
1544
|
+
_DEFAULT_CHAINS = {
|
|
1545
|
+
"fol": ("z3", "tableau", "resolution", "modelfinder"),
|
|
1546
|
+
"modal": ("modal-tableau", "kripke-enum", "qml"),
|
|
1547
|
+
"intuitionistic": ("intuitionistic",),
|
|
1548
|
+
"lambek": ("lambek",),
|
|
1549
|
+
"ill": ("ill",),
|
|
1550
|
+
"relevant": ("relevant",),
|
|
1551
|
+
"hybrid": ("hybrid",),
|
|
1552
|
+
}
|
|
1553
|
+
|
|
1554
|
+
|
|
1555
|
+
def default_chain(logic: str) -> Tuple[str, ...]:
|
|
1556
|
+
"""The default backend order for ``logic`` (ValueError if unknown).
|
|
1557
|
+
|
|
1558
|
+
The chains are static, with ONE documented availability-dependent member:
|
|
1559
|
+
when the optional ``cvc5`` extra is installed
|
|
1560
|
+
(``pip install unicode-logic-kit[cvc5]``), the FOL chain gains ``"cvc5"``
|
|
1561
|
+
directly after ``"z3"`` — a second, fully independent SMT decision
|
|
1562
|
+
procedure whose quantifier instantiation (E-matching/enumerative/CEGQI)
|
|
1563
|
+
often closes goals Z3's MBQI leaves UNKNOWN, and vice versa. It cannot be
|
|
1564
|
+
an unconditional member: a missing backend in the DEFAULT chain would
|
|
1565
|
+
make every plain ``prove()`` call raise :class:`BackendUnavailable` on a
|
|
1566
|
+
machine without the extra. Every verdict names the backend that actually
|
|
1567
|
+
answered (``backend`` / ``agreement``), so provenance stays exact even
|
|
1568
|
+
though the chain adapts to the install.
|
|
1569
|
+
"""
|
|
1570
|
+
if logic not in _DEFAULT_CHAINS:
|
|
1571
|
+
raise ValueError(
|
|
1572
|
+
f"default_chain: unknown logic {logic!r} (use one of {sorted(_DEFAULT_CHAINS)})")
|
|
1573
|
+
chain = _DEFAULT_CHAINS[logic]
|
|
1574
|
+
if logic == "fol" and _REGISTRY["cvc5"].available():
|
|
1575
|
+
chain = chain[:1] + ("cvc5",) + chain[1:]
|
|
1576
|
+
return chain
|
|
1577
|
+
|
|
1578
|
+
|
|
1579
|
+
# ---------------------------------------------------------------------------
|
|
1580
|
+
# Options: which backend of a chain reads which keyword
|
|
1581
|
+
#
|
|
1582
|
+
# ``api.prove(f, premises, frame="S4")`` hands ``frame`` to every backend of the
|
|
1583
|
+
# chain, and what a backend does with a keyword it does not know used to differ
|
|
1584
|
+
# from one to the next: Z3, E and Zipperposition ignored it, the tableau, the
|
|
1585
|
+
# resolution prover and the model finder failed with a ``TypeError`` that surfaced
|
|
1586
|
+
# as an ERROR verdict. An ignored option answers a different question than the
|
|
1587
|
+
# caller asked. Every stock backend therefore DECLARES the names it reads (below),
|
|
1588
|
+
# and the dispatcher checks a call against the declarations before anything runs.
|
|
1589
|
+
# ---------------------------------------------------------------------------
|
|
1590
|
+
|
|
1591
|
+
from typing import Callable, FrozenSet # noqa: E402
|
|
1592
|
+
|
|
1593
|
+
|
|
1594
|
+
def _keywords_of(function, skip: Sequence[str] = ()) -> FrozenSet[str]:
|
|
1595
|
+
"""The names ``function`` takes by keyword, without ``skip``.
|
|
1596
|
+
|
|
1597
|
+
A backend that forwards ``**options`` to a function with its own keyword list
|
|
1598
|
+
reads exactly that list; deriving the set from the signature keeps the
|
|
1599
|
+
declaration from drifting away from the function it describes.
|
|
1600
|
+
"""
|
|
1601
|
+
import inspect
|
|
1602
|
+
|
|
1603
|
+
return frozenset(
|
|
1604
|
+
name for name, parameter in inspect.signature(function).parameters.items()
|
|
1605
|
+
if parameter.kind in (parameter.POSITIONAL_OR_KEYWORD, parameter.KEYWORD_ONLY)
|
|
1606
|
+
and name not in skip)
|
|
1607
|
+
|
|
1608
|
+
|
|
1609
|
+
def _keywords(*targets: str, skip: Sequence[str] = ()) -> Callable[[Optional[str]], FrozenSet[str]]:
|
|
1610
|
+
"""A lazy reader of the keyword names of the functions at ``"module:name"``.
|
|
1611
|
+
|
|
1612
|
+
Several targets give the names ALL of them take (a backend that forwards the
|
|
1613
|
+
same options to two functions reads what both read). Lazy, because a backend
|
|
1614
|
+
is rarely in the chain and its module is heavy.
|
|
1615
|
+
"""
|
|
1616
|
+
def read(logic: Optional[str] = None) -> FrozenSet[str]:
|
|
1617
|
+
import importlib
|
|
1618
|
+
|
|
1619
|
+
names = None
|
|
1620
|
+
for target in targets:
|
|
1621
|
+
module, _, function = target.partition(":")
|
|
1622
|
+
found = _keywords_of(getattr(importlib.import_module(module), function), skip)
|
|
1623
|
+
names = found if names is None else names & found
|
|
1624
|
+
return names if names is not None else frozenset()
|
|
1625
|
+
return read
|
|
1626
|
+
|
|
1627
|
+
|
|
1628
|
+
def _names(*names: str) -> Callable[[Optional[str]], FrozenSet[str]]:
|
|
1629
|
+
"""A reader of a fixed set of option names."""
|
|
1630
|
+
fixed = frozenset(names)
|
|
1631
|
+
|
|
1632
|
+
def read(logic: Optional[str] = None) -> FrozenSet[str]:
|
|
1633
|
+
return fixed
|
|
1634
|
+
return read
|
|
1635
|
+
|
|
1636
|
+
|
|
1637
|
+
def _isabelle_options(logic: Optional[str] = None) -> FrozenSet[str]:
|
|
1638
|
+
"""What :class:`IsabelleBackend` reads: the keywords of the runner for the
|
|
1639
|
+
logic of the call (both when the logic is not known), plus ``native_equality``,
|
|
1640
|
+
which the backend checks itself."""
|
|
1641
|
+
from ..hol.isabelle_runner import isabelle_decide_fol, isabelle_decide_modal
|
|
1642
|
+
|
|
1643
|
+
modal = _keywords_of(isabelle_decide_modal, ("formula",))
|
|
1644
|
+
classical = _keywords_of(isabelle_decide_fol, ("formula",)) | {"native_equality"}
|
|
1645
|
+
if logic == "modal":
|
|
1646
|
+
return modal
|
|
1647
|
+
if logic == "fol":
|
|
1648
|
+
return classical
|
|
1649
|
+
return modal | classical
|
|
1650
|
+
|
|
1651
|
+
|
|
1652
|
+
#: The options each stock backend reads, as lazy readers keyed by the backend's
|
|
1653
|
+
#: class (an exact class: a subclass has to declare its own, through
|
|
1654
|
+
#: :meth:`ProverBackend.accepted_options`). Every name here is read in the
|
|
1655
|
+
#: backend's ``decide`` or by the function it forwards ``**options`` to; the test
|
|
1656
|
+
#: suite compares the two.
|
|
1657
|
+
_STOCK_OPTIONS: Dict[type, Callable[[Optional[str]], FrozenSet[str]]] = {
|
|
1658
|
+
Z3Backend: _names(),
|
|
1659
|
+
TableauBackend: _keywords("unicode_logic_kit.atp.tableau:prove_tableau_detailed",
|
|
1660
|
+
skip=("premises", "conclusion", "timeout")),
|
|
1661
|
+
ResolutionBackend: _keywords("unicode_logic_kit.atp.resolution:prove",
|
|
1662
|
+
skip=("premises", "conclusion", "timeout")),
|
|
1663
|
+
ModelFinderBackend: _keywords("unicode_logic_kit.semantics.modelfinder:find_countermodel",
|
|
1664
|
+
skip=("premises", "conclusion", "timeout")),
|
|
1665
|
+
ModalTableauBackend: _keywords("unicode_logic_kit.atp.modal_tableau:modal_decide",
|
|
1666
|
+
"unicode_logic_kit.atp.modal_tableau:modal_countermodel",
|
|
1667
|
+
skip=("formula", "timeout")),
|
|
1668
|
+
QmlBackend: _keywords("unicode_logic_kit.fol.qml:qml_is_valid", skip=("formula", "timeout")),
|
|
1669
|
+
IsabelleBackend: _isabelle_options,
|
|
1670
|
+
Prover9Backend: _names("prover9_path", "use_wsl"),
|
|
1671
|
+
VampireBackend: _keywords("unicode_logic_kit.atp.vampire_entailment:check_entailment_vampire_detailed",
|
|
1672
|
+
skip=("premises", "conclusion", "timeout")),
|
|
1673
|
+
EProverBackend: _names("tff", "sort", "premise_names"),
|
|
1674
|
+
ZipperpositionBackend: _names("tff", "sort", "premise_names"),
|
|
1675
|
+
Cvc5Backend: _names("logic", "random_seed", "proof"),
|
|
1676
|
+
ClingoBackend: _names("max_size", "all_different", "verify"),
|
|
1677
|
+
MinizincBackend: _names("minizinc_path", "max_size", "solver", "all_different"),
|
|
1678
|
+
HetsBackend: _names("reasoner", "translation", "url"),
|
|
1679
|
+
TweeBackend: _names("use_wsl", "twee_cmd"),
|
|
1680
|
+
NanocopBackend: _names("logic", "domain"),
|
|
1681
|
+
Leo3Backend: _names("frame", "domains"),
|
|
1682
|
+
KripkeEnumBackend: _keywords("unicode_logic_kit.atp.kripke_enum:modal_enum_search",
|
|
1683
|
+
skip=("formula", "timeout")),
|
|
1684
|
+
LtlTableauBackend: _names("mode", "max_atoms"),
|
|
1685
|
+
IntBackend: _keywords("unicode_logic_kit.semantics.intuitionistic:int_countermodel",
|
|
1686
|
+
skip=("formula",)),
|
|
1687
|
+
LambekBackend: _names(),
|
|
1688
|
+
IllBackend: _keywords("unicode_logic_kit.atp.linear:ill_derivable",
|
|
1689
|
+
skip=("antecedents", "goal")),
|
|
1690
|
+
RelevantBackend: _names("max_worlds"),
|
|
1691
|
+
HybridBackend: _names("frame", "systems", "temporal_closure"),
|
|
1692
|
+
}
|
|
1693
|
+
|
|
1694
|
+
#: Options that only bound how far a search goes, say how a solver is driven and
|
|
1695
|
+
#: where it lives, or ask for more to be reported about the same answer (the names
|
|
1696
|
+
#: of the premises, a proof text). A backend that does not read one still answers the same
|
|
1697
|
+
#: question without it, so it runs when another backend of the chain reads the
|
|
1698
|
+
#: option. Every other option changes WHICH question is asked (a modal frame, a
|
|
1699
|
+
#: domain regime, a bridge, the reading of numerals, a subsort edge): a backend
|
|
1700
|
+
#: that does not read one would answer a different question, and is not run.
|
|
1701
|
+
_NEUTRAL_OPTIONS = frozenset({
|
|
1702
|
+
"max_steps", "max_terms", "max_worlds", "max_atoms", "max_models", "max_depth",
|
|
1703
|
+
"max_size", "max_candidates", "symmetry_breaking", "domain_elements",
|
|
1704
|
+
"methods", "refute", "card", "prove_timeout", "refute_timeout",
|
|
1705
|
+
"random_seed", "solver", "verify", "install",
|
|
1706
|
+
"use_wsl", "vampire_path", "prover9_path", "twee_cmd", "minizinc_path",
|
|
1707
|
+
"url", "reasoner", "translation", "tff",
|
|
1708
|
+
"premise_names", "axiom_names", "proof",
|
|
1709
|
+
})
|
|
1710
|
+
|
|
1711
|
+
|
|
1712
|
+
def declared_options(backend: ProverBackend, logic: Optional[str] = None) -> Optional[FrozenSet[str]]:
|
|
1713
|
+
"""The names of the options ``backend`` reads, or ``None`` when it declares none.
|
|
1714
|
+
|
|
1715
|
+
The backend's own :meth:`~ProverBackend.accepted_options` first, then the
|
|
1716
|
+
declaration of the stock class of exactly that type. ``None`` means that
|
|
1717
|
+
nothing is known about the backend and it is handed every option.
|
|
1718
|
+
"""
|
|
1719
|
+
own = getattr(backend, "accepted_options", lambda logic=None: None)(logic)
|
|
1720
|
+
if own is not None:
|
|
1721
|
+
return frozenset(own)
|
|
1722
|
+
reader = _STOCK_OPTIONS.get(type(backend))
|
|
1723
|
+
return reader(logic) if reader is not None else None
|
|
1724
|
+
|
|
1725
|
+
|
|
1726
|
+
def plan_options(caller: str, chain: Sequence[str], logic: str,
|
|
1727
|
+
options: dict) -> Dict[str, Tuple[dict, Optional[str]]]:
|
|
1728
|
+
"""Decide which of ``options`` each backend of ``chain`` is handed.
|
|
1729
|
+
|
|
1730
|
+
Returns, per backend name, ``(options to pass, refusal)``. ``refusal`` is
|
|
1731
|
+
``None`` when the backend runs, and otherwise the text of the reason it must
|
|
1732
|
+
not: it does not read an option that changes the question (see
|
|
1733
|
+
:data:`_NEUTRAL_OPTIONS`) which another backend of the chain does read, so any
|
|
1734
|
+
answer of it would be about a different question.
|
|
1735
|
+
|
|
1736
|
+
Raises:
|
|
1737
|
+
ValueError: when an option is read by NO backend of the chain, naming the
|
|
1738
|
+
option, the chain and what its backends do read, before anything runs.
|
|
1739
|
+
A backend that declares no options counts as reading everything.
|
|
1740
|
+
"""
|
|
1741
|
+
if not options:
|
|
1742
|
+
return {name: ({}, None) for name in chain}
|
|
1743
|
+
declared = {name: declared_options(get_backend(name), logic) for name in chain}
|
|
1744
|
+
for option in options:
|
|
1745
|
+
if not any(names is None or option in names for names in declared.values()):
|
|
1746
|
+
reads = "; ".join(
|
|
1747
|
+
f"{name}: " + (", ".join(sorted(names)) if names else "no options")
|
|
1748
|
+
for name, names in declared.items())
|
|
1749
|
+
raise ValueError(
|
|
1750
|
+
f"{caller}: option {option!r} is read by no backend of the chain "
|
|
1751
|
+
f"{list(chain)} -- nothing would use it, and the answer would be "
|
|
1752
|
+
f"given as if it had not been passed. The options each backend reads: "
|
|
1753
|
+
f"{reads}.")
|
|
1754
|
+
plan: Dict[str, Tuple[dict, Optional[str]]] = {}
|
|
1755
|
+
for name in chain:
|
|
1756
|
+
names = declared[name]
|
|
1757
|
+
if names is None:
|
|
1758
|
+
plan[name] = (dict(options), None)
|
|
1759
|
+
continue
|
|
1760
|
+
passed = {key: value for key, value in options.items() if key in names}
|
|
1761
|
+
unread = sorted(key for key in options
|
|
1762
|
+
if key not in names and key not in _NEUTRAL_OPTIONS)
|
|
1763
|
+
if unread:
|
|
1764
|
+
shown = ", ".join(repr(key) for key in unread)
|
|
1765
|
+
plan[name] = ({}, f"{name} does not read the option {shown}, which "
|
|
1766
|
+
f"changes the question; another backend of the chain "
|
|
1767
|
+
f"reads it, and answering without it would decide a "
|
|
1768
|
+
f"different question, so {name} was not run")
|
|
1769
|
+
else:
|
|
1770
|
+
plan[name] = (passed, None)
|
|
1771
|
+
return plan
|
|
1772
|
+
|
|
1773
|
+
|
|
1774
|
+
def _available_for(backend: ProverBackend, options: dict) -> bool:
|
|
1775
|
+
""":meth:`ProverBackend.available_for` of ``backend``, or its ``available()``
|
|
1776
|
+
for an object that has no such method."""
|
|
1777
|
+
method = getattr(backend, "available_for", None)
|
|
1778
|
+
return bool(method(options)) if method is not None else bool(backend.available())
|
|
1779
|
+
|
|
1780
|
+
|
|
1781
|
+
def run_backend(name: str, formula: Node, premises: Sequence[Node] = (),
|
|
1782
|
+
timeout: int = 10000, **options) -> Verdict:
|
|
1783
|
+
"""Run one backend by name, enforcing the availability contract.
|
|
1784
|
+
|
|
1785
|
+
Raises ``ValueError`` for an unknown name and :class:`BackendUnavailable`
|
|
1786
|
+
for a known-but-unavailable one; whether it is available is asked for the
|
|
1787
|
+
route this call's ``options`` select (:meth:`ProverBackend.available_for`).
|
|
1788
|
+
Any unexpected exception inside the
|
|
1789
|
+
backend is converted to an ERROR verdict (reason="infra") so a batch run
|
|
1790
|
+
over thousands of formulas records the failure instead of dying.
|
|
1791
|
+
"""
|
|
1792
|
+
backend = get_backend(name)
|
|
1793
|
+
if not _available_for(backend, options):
|
|
1794
|
+
raise BackendUnavailable(
|
|
1795
|
+
f"{name}: backend is not available on this machine "
|
|
1796
|
+
f"(external={backend.external}) — install it or drop it from `backends`.")
|
|
1797
|
+
try:
|
|
1798
|
+
return backend.decide(formula, premises, timeout=timeout, **options)
|
|
1799
|
+
except (ValueError, BackendUnavailable):
|
|
1800
|
+
raise # caller errors stay loud
|
|
1801
|
+
except Exception as exc: # infra failure → recorded, not fatal
|
|
1802
|
+
return Verdict(ERROR, name, reason="infra",
|
|
1803
|
+
detail=f"{type(exc).__name__}: {exc}")
|