unicode-logic-kit 0.31.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (237) hide show
  1. unicode_logic_kit/__init__.py +385 -0
  2. unicode_logic_kit/__main__.py +520 -0
  3. unicode_logic_kit/_deadline.py +219 -0
  4. unicode_logic_kit/ace/__init__.py +126 -0
  5. unicode_logic_kit/ace/_align.py +135 -0
  6. unicode_logic_kit/ace/chem_lexicon.py +128 -0
  7. unicode_logic_kit/ace/drs_reader.py +570 -0
  8. unicode_logic_kit/ace/mapping.py +666 -0
  9. unicode_logic_kit/ace/reverse_modal.py +138 -0
  10. unicode_logic_kit/ace/runner.py +551 -0
  11. unicode_logic_kit/ace/translate.py +452 -0
  12. unicode_logic_kit/ace/verbalize.py +1070 -0
  13. unicode_logic_kit/api.py +1284 -0
  14. unicode_logic_kit/atp/__init__.py +177 -0
  15. unicode_logic_kit/atp/_ascii_names.py +113 -0
  16. unicode_logic_kit/atp/_html.py +72 -0
  17. unicode_logic_kit/atp/_substructural_input.py +228 -0
  18. unicode_logic_kit/atp/_tff_problem.py +715 -0
  19. unicode_logic_kit/atp/_tptp_problem.py +1111 -0
  20. unicode_logic_kit/atp/_writer_support.py +289 -0
  21. unicode_logic_kit/atp/clingo_backend.py +1180 -0
  22. unicode_logic_kit/atp/cvc5_backend.py +1385 -0
  23. unicode_logic_kit/atp/eprover_backend.py +732 -0
  24. unicode_logic_kit/atp/finite_domain.py +1055 -0
  25. unicode_logic_kit/atp/fitch.py +1547 -0
  26. unicode_logic_kit/atp/fitch_search.py +551 -0
  27. unicode_logic_kit/atp/hets_backend.py +339 -0
  28. unicode_logic_kit/atp/hybrid_down.py +120 -0
  29. unicode_logic_kit/atp/incremental.py +250 -0
  30. unicode_logic_kit/atp/kripke_enum.py +741 -0
  31. unicode_logic_kit/atp/lambek.py +436 -0
  32. unicode_logic_kit/atp/leo3_backend.py +332 -0
  33. unicode_logic_kit/atp/linear.py +738 -0
  34. unicode_logic_kit/atp/lj.py +705 -0
  35. unicode_logic_kit/atp/logic_backends.py +566 -0
  36. unicode_logic_kit/atp/ltl_tableau.py +1084 -0
  37. unicode_logic_kit/atp/minizinc_backend.py +1402 -0
  38. unicode_logic_kit/atp/modal_tableau.py +1382 -0
  39. unicode_logic_kit/atp/nanocop_backend.py +410 -0
  40. unicode_logic_kit/atp/portfolio.py +489 -0
  41. unicode_logic_kit/atp/protocol.py +1803 -0
  42. unicode_logic_kit/atp/prover9_entailment.py +1153 -0
  43. unicode_logic_kit/atp/resolution.py +1376 -0
  44. unicode_logic_kit/atp/resolution_check.py +1114 -0
  45. unicode_logic_kit/atp/sequent.py +1050 -0
  46. unicode_logic_kit/atp/tableau.py +921 -0
  47. unicode_logic_kit/atp/tableau_check.py +543 -0
  48. unicode_logic_kit/atp/tptp_ncl.py +811 -0
  49. unicode_logic_kit/atp/tptp_tff.py +1546 -0
  50. unicode_logic_kit/atp/tstp.py +1333 -0
  51. unicode_logic_kit/atp/tstp_check.py +1096 -0
  52. unicode_logic_kit/atp/twee_backend.py +236 -0
  53. unicode_logic_kit/atp/twee_check.py +711 -0
  54. unicode_logic_kit/atp/twee_entailment.py +953 -0
  55. unicode_logic_kit/atp/vampire_entailment.py +540 -0
  56. unicode_logic_kit/atp/z3_arith.py +470 -0
  57. unicode_logic_kit/atp/z3_equivalence.py +36 -0
  58. unicode_logic_kit/atp/z3_fuzzy.py +362 -0
  59. unicode_logic_kit/atp/z3_input.py +500 -0
  60. unicode_logic_kit/atp/z3_models.py +208 -0
  61. unicode_logic_kit/chem/__init__.py +88 -0
  62. unicode_logic_kit/chem/_naming.py +284 -0
  63. unicode_logic_kit/chem/cache.py +185 -0
  64. unicode_logic_kit/chem/interop.py +244 -0
  65. unicode_logic_kit/chem/mol.py +525 -0
  66. unicode_logic_kit/chem/signature.py +112 -0
  67. unicode_logic_kit/comorphism.py +497 -0
  68. unicode_logic_kit/dl/__init__.py +384 -0
  69. unicode_logic_kit/dl/classification.py +227 -0
  70. unicode_logic_kit/dl/concepts.py +632 -0
  71. unicode_logic_kit/dl/datatypes.py +818 -0
  72. unicode_logic_kit/dl/owl_functional.py +2433 -0
  73. unicode_logic_kit/dl/owl_manchester.py +1637 -0
  74. unicode_logic_kit/dl/owl_reasoner.py +790 -0
  75. unicode_logic_kit/dl/parser.py +391 -0
  76. unicode_logic_kit/dl/tableau.py +4048 -0
  77. unicode_logic_kit/dl/translate.py +2704 -0
  78. unicode_logic_kit/drt/__init__.py +94 -0
  79. unicode_logic_kit/drt/export.py +179 -0
  80. unicode_logic_kit/drt/nodes.py +506 -0
  81. unicode_logic_kit/drt/parser.py +965 -0
  82. unicode_logic_kit/drt/resolve.py +195 -0
  83. unicode_logic_kit/drt/reverse.py +175 -0
  84. unicode_logic_kit/eval/__init__.py +106 -0
  85. unicode_logic_kit/eval/batch.py +382 -0
  86. unicode_logic_kit/eval/canonical.py +663 -0
  87. unicode_logic_kit/eval/chem_batch.py +606 -0
  88. unicode_logic_kit/eval/converses.py +200 -0
  89. unicode_logic_kit/eval/datasets/__init__.py +136 -0
  90. unicode_logic_kit/eval/datasets/_base.py +263 -0
  91. unicode_logic_kit/eval/datasets/_proofwriter_proof.py +422 -0
  92. unicode_logic_kit/eval/datasets/c3po.py +678 -0
  93. unicode_logic_kit/eval/datasets/folio.py +158 -0
  94. unicode_logic_kit/eval/datasets/fracas.py +418 -0
  95. unicode_logic_kit/eval/datasets/groves.py +191 -0
  96. unicode_logic_kit/eval/datasets/logicbench.py +467 -0
  97. unicode_logic_kit/eval/datasets/logicnli.py +303 -0
  98. unicode_logic_kit/eval/datasets/malls.py +133 -0
  99. unicode_logic_kit/eval/datasets/pfolio.py +594 -0
  100. unicode_logic_kit/eval/datasets/pmb.py +242 -0
  101. unicode_logic_kit/eval/datasets/prontoqa.py +611 -0
  102. unicode_logic_kit/eval/datasets/proofwriter.py +1431 -0
  103. unicode_logic_kit/eval/datasets/proverqa.py +674 -0
  104. unicode_logic_kit/eval/datasets/willow.py +478 -0
  105. unicode_logic_kit/eval/equivalence.py +466 -0
  106. unicode_logic_kit/eval/exercise_gen.py +533 -0
  107. unicode_logic_kit/eval/explain.py +791 -0
  108. unicode_logic_kit/eval/generality.py +750 -0
  109. unicode_logic_kit/eval/metric_hf.py +458 -0
  110. unicode_logic_kit/eval/predicate_match.py +343 -0
  111. unicode_logic_kit/eval/theory_check.py +1170 -0
  112. unicode_logic_kit/eval/validate.py +306 -0
  113. unicode_logic_kit/fol/__init__.py +177 -0
  114. unicode_logic_kit/fol/_atom_keys.py +510 -0
  115. unicode_logic_kit/fol/_fol_nodes.py +3586 -0
  116. unicode_logic_kit/fol/_free_parameters.py +105 -0
  117. unicode_logic_kit/fol/_ho_nodes.py +448 -0
  118. unicode_logic_kit/fol/_hybrid_nodes.py +308 -0
  119. unicode_logic_kit/fol/_identifiers.py +1091 -0
  120. unicode_logic_kit/fol/_lambek_nodes.py +112 -0
  121. unicode_logic_kit/fol/_linear_nodes.py +352 -0
  122. unicode_logic_kit/fol/_modal_nodes.py +1467 -0
  123. unicode_logic_kit/fol/_msfl_nodes.py +2196 -0
  124. unicode_logic_kit/fol/_numeral_symbols.py +231 -0
  125. unicode_logic_kit/fol/_so_nodes.py +200 -0
  126. unicode_logic_kit/fol/_symbol_names.py +81 -0
  127. unicode_logic_kit/fol/_team_nodes.py +181 -0
  128. unicode_logic_kit/fol/_tptp_symbols.py +551 -0
  129. unicode_logic_kit/fol/_truth_constants.py +117 -0
  130. unicode_logic_kit/fol/casl_export.py +1135 -0
  131. unicode_logic_kit/fol/casl_import.py +929 -0
  132. unicode_logic_kit/fol/derivation.py +367 -0
  133. unicode_logic_kit/fol/dialect_detect.py +70 -0
  134. unicode_logic_kit/fol/dialect_repair.py +537 -0
  135. unicode_logic_kit/fol/frames.py +637 -0
  136. unicode_logic_kit/fol/grammars/terminals.lark +31 -0
  137. unicode_logic_kit/fol/lambda_tools.py +297 -0
  138. unicode_logic_kit/fol/latex_input.py +429 -0
  139. unicode_logic_kit/fol/modal_translation.py +944 -0
  140. unicode_logic_kit/fol/msflparser.py +1033 -0
  141. unicode_logic_kit/fol/naming.py +422 -0
  142. unicode_logic_kit/fol/nodes.py +241 -0
  143. unicode_logic_kit/fol/normalforms.py +492 -0
  144. unicode_logic_kit/fol/pal.py +287 -0
  145. unicode_logic_kit/fol/prolog_export.py +566 -0
  146. unicode_logic_kit/fol/prolog_input.py +505 -0
  147. unicode_logic_kit/fol/prover9_input.py +1325 -0
  148. unicode_logic_kit/fol/qml.py +1760 -0
  149. unicode_logic_kit/fol/qmltp_input.py +525 -0
  150. unicode_logic_kit/fol/sanitize.py +221 -0
  151. unicode_logic_kit/fol/serialize.py +79 -0
  152. unicode_logic_kit/fol/signature.py +1290 -0
  153. unicode_logic_kit/fol/simplify_check.py +544 -0
  154. unicode_logic_kit/fol/spans.py +594 -0
  155. unicode_logic_kit/fol/tptp_input.py +1503 -0
  156. unicode_logic_kit/fol/tptp_repair.py +941 -0
  157. unicode_logic_kit/fol/unification.py +157 -0
  158. unicode_logic_kit/fol/verbalize.py +263 -0
  159. unicode_logic_kit/hets/__init__.py +163 -0
  160. unicode_logic_kit/hets/bridge.py +142 -0
  161. unicode_logic_kit/hets/client.py +748 -0
  162. unicode_logic_kit/hets/docker.py +420 -0
  163. unicode_logic_kit/hets/dol.py +712 -0
  164. unicode_logic_kit/hets/haskell_json.py +355 -0
  165. unicode_logic_kit/hets/owl_backend.py +794 -0
  166. unicode_logic_kit/hets/owl_cli.py +598 -0
  167. unicode_logic_kit/hets/symbols.py +512 -0
  168. unicode_logic_kit/hol/__init__.py +140 -0
  169. unicode_logic_kit/hol/_ho_common.py +323 -0
  170. unicode_logic_kit/hol/_isabelle_binders.py +125 -0
  171. unicode_logic_kit/hol/classical.py +812 -0
  172. unicode_logic_kit/hol/deepshallow/__init__.py +45 -0
  173. unicode_logic_kit/hol/deepshallow/_common.py +177 -0
  174. unicode_logic_kit/hol/deepshallow/conditional.py +225 -0
  175. unicode_logic_kit/hol/deepshallow/intuitionistic.py +181 -0
  176. unicode_logic_kit/hol/deepshallow/modal.py +217 -0
  177. unicode_logic_kit/hol/deepshallow/qml.py +406 -0
  178. unicode_logic_kit/hol/deepshallow/relevant.py +206 -0
  179. unicode_logic_kit/hol/free.py +753 -0
  180. unicode_logic_kit/hol/goedel.py +336 -0
  181. unicode_logic_kit/hol/ho_modal.py +1743 -0
  182. unicode_logic_kit/hol/intuitionistic.py +403 -0
  183. unicode_logic_kit/hol/isabelle_conditional.py +593 -0
  184. unicode_logic_kit/hol/isabelle_modal.py +1908 -0
  185. unicode_logic_kit/hol/isabelle_relevant.py +412 -0
  186. unicode_logic_kit/hol/isabelle_runner.py +1147 -0
  187. unicode_logic_kit/hol/isabelle_substructural.py +884 -0
  188. unicode_logic_kit/hol/lean.py +1018 -0
  189. unicode_logic_kit/hol/manyvalued.py +921 -0
  190. unicode_logic_kit/hol/secondorder.py +687 -0
  191. unicode_logic_kit/hol/thf_modal.py +941 -0
  192. unicode_logic_kit/hol/thirdorder.py +397 -0
  193. unicode_logic_kit/ilp/__init__.py +89 -0
  194. unicode_logic_kit/ilp/readback.py +389 -0
  195. unicode_logic_kit/ilp/separation.py +153 -0
  196. unicode_logic_kit/ilp/task.py +730 -0
  197. unicode_logic_kit/logic.py +163 -0
  198. unicode_logic_kit/mcp/__init__.py +28 -0
  199. unicode_logic_kit/mcp/__main__.py +5 -0
  200. unicode_logic_kit/mcp/chem_tools.py +1031 -0
  201. unicode_logic_kit/mcp/server.py +2453 -0
  202. unicode_logic_kit/mcp/syntax_spec.py +681 -0
  203. unicode_logic_kit/prob/__init__.py +53 -0
  204. unicode_logic_kit/prob/_bdd.py +225 -0
  205. unicode_logic_kit/prob/_column_gen.py +668 -0
  206. unicode_logic_kit/prob/distribution.py +686 -0
  207. unicode_logic_kit/prob/nilsson.py +470 -0
  208. unicode_logic_kit/py.typed +0 -0
  209. unicode_logic_kit/semantics/__init__.py +137 -0
  210. unicode_logic_kit/semantics/_modal_reject.py +156 -0
  211. unicode_logic_kit/semantics/action_models.py +466 -0
  212. unicode_logic_kit/semantics/asp_models.py +1200 -0
  213. unicode_logic_kit/semantics/conditional.py +580 -0
  214. unicode_logic_kit/semantics/dynamic_epistemic.py +95 -0
  215. unicode_logic_kit/semantics/free_logic.py +913 -0
  216. unicode_logic_kit/semantics/fuzzy.py +384 -0
  217. unicode_logic_kit/semantics/fuzzy_kripke.py +442 -0
  218. unicode_logic_kit/semantics/intuitionistic.py +581 -0
  219. unicode_logic_kit/semantics/kripke.py +1139 -0
  220. unicode_logic_kit/semantics/manyvalued.py +580 -0
  221. unicode_logic_kit/semantics/matrix.py +342 -0
  222. unicode_logic_kit/semantics/model_eval.py +1135 -0
  223. unicode_logic_kit/semantics/modelfinder.py +1036 -0
  224. unicode_logic_kit/semantics/nonmonotonic.py +372 -0
  225. unicode_logic_kit/semantics/relevant.py +331 -0
  226. unicode_logic_kit/semantics/secondorder.py +657 -0
  227. unicode_logic_kit/semantics/structures.py +352 -0
  228. unicode_logic_kit/semantics/tarski.py +975 -0
  229. unicode_logic_kit/semantics/team.py +315 -0
  230. unicode_logic_kit/semantics/team_translation.py +416 -0
  231. unicode_logic_kit/semantics/thirdorder.py +358 -0
  232. unicode_logic_kit/semantics/tnorm.py +85 -0
  233. unicode_logic_kit/semantics/truthtable.py +201 -0
  234. unicode_logic_kit-0.31.0.dist-info/METADATA +333 -0
  235. unicode_logic_kit-0.31.0.dist-info/RECORD +237 -0
  236. unicode_logic_kit-0.31.0.dist-info/WHEEL +4 -0
  237. unicode_logic_kit-0.31.0.dist-info/licenses/LICENSE +21 -0
@@ -0,0 +1,1803 @@
1
+ """Uniform prover protocol: one Verdict type over every decision route.
2
+
3
+ The kit decides validity/entailment through many routes — its OWN calculi and
4
+ semantic model searches (Z3-free tableau, resolution, finite model finder,
5
+ modal labelled tableau, the QML embedding) and EXTERNAL provers (Z3, Isabelle,
6
+ Prover9, Vampire). Each grew its own return convention (bare bool,
7
+ ``"valid"/"invalid"/"unknown"`` strings, ``ModalVerdict``/``FolVerdict``).
8
+ This module adds the layer a pipeline needs on top, without touching any
9
+ existing signature:
10
+
11
+ * :class:`Verdict` — the one result type: a semantic ``status`` (proved /
12
+ refuted / unknown / error), a ``reason`` axis that distinguishes budget
13
+ exhaustion from honest incompleteness from timeouts, the SZS status string,
14
+ provenance (which backend, how long), and JSON-able witnesses.
15
+ * :class:`ProverBackend` — the adapter contract (``available()`` +
16
+ ``decide()``), with the kit's internal calculi and semantic searches as
17
+ first-class backends alongside the external provers.
18
+ * a registry (:func:`register_backend`, :func:`get_backend`,
19
+ :func:`available_backends`, :func:`default_chain`) that the ``prove()``
20
+ facade in :mod:`unicode_logic_kit.api` dispatches over.
21
+
22
+ Error contract: requesting an UNKNOWN backend name raises ``ValueError``;
23
+ requesting a known backend whose prerequisites are missing (no binary, no
24
+ install) raises :class:`BackendUnavailable` — never a silent skip, because a
25
+ silently skipped backend makes evaluation results irreproducible.
26
+
27
+ The status/reason split is deliberate: ``status`` answers "what do we know
28
+ about the formula" (only four values, easy to branch on), ``reason`` answers
29
+ "why do we not know more" (``bound_hit`` ≠ ``timeout`` ≠ ``incomplete`` ≠
30
+ ``unsupported`` — a benchmark table must not conflate them).
31
+ """
32
+
33
+ import re
34
+ import shutil
35
+ import time
36
+ from abc import ABC, abstractmethod
37
+ from dataclasses import dataclass, field, replace
38
+ from typing import Dict, Optional, Sequence, Tuple
39
+
40
+ from ..fol.nodes import Node, And, Implies
41
+
42
+ __all__ = [
43
+ "PROVED", "REFUTED", "UNKNOWN", "ERROR", "STATUSES",
44
+ "Verdict", "BackendUnavailable", "ProverBackend",
45
+ "register_backend", "get_backend", "available_backends", "default_chain",
46
+ "run_backend",
47
+ "z3_relevant_premises",
48
+ ]
49
+
50
+ # ---------------------------------------------------------------------------
51
+ # Statuses, reasons, and their SZS image
52
+ # ---------------------------------------------------------------------------
53
+
54
+ PROVED = "proved" # validity / entailment established
55
+ REFUTED = "refuted" # a genuine countermodel witnesses invalidity
56
+ UNKNOWN = "unknown" # neither, within this backend's budget/strength
57
+ ERROR = "error" # the backend failed for an infrastructure reason
58
+ STATUSES = (PROVED, REFUTED, UNKNOWN, ERROR)
59
+
60
+ # reason values (for UNKNOWN/ERROR): why we do not know more.
61
+ # "timeout" — a wall-clock budget expired
62
+ # "bound_hit" — a step/size bound expired (max_steps, max_worlds, …)
63
+ # "incomplete" — the method is sound but incomplete on this fragment and
64
+ # gave up honestly (no bound was hit)
65
+ # "unsupported" — the backend has no rule/translation for this fragment
66
+ # "infra" — subprocess/JVM/syntax failure (ERROR only). A prover that
67
+ # REJECTS its input (a syntax or type error, Vampire's "User
68
+ # error: ..." line, E's parse error, Twee's usage error,
69
+ # Prover9's fatal error) and prints no verdict lands here, with
70
+ # the prover's own message in ``detail``: refusing to read a
71
+ # problem is not the same as running out of time, and a
72
+ # benchmark table that files the two together reads a writer
73
+ # defect as "undecided".
74
+
75
+ _SZS = {
76
+ (PROVED, None): "Theorem",
77
+ (REFUTED, None): "CounterSatisfiable",
78
+ (UNKNOWN, "timeout"): "Timeout",
79
+ (UNKNOWN, "bound_hit"): "ResourceOut",
80
+ (UNKNOWN, "incomplete"): "GaveUp",
81
+ (UNKNOWN, "unsupported"): "Inappropriate",
82
+ (UNKNOWN, None): "Unknown",
83
+ (ERROR, "infra"): "Error",
84
+ (ERROR, None): "Error",
85
+ }
86
+
87
+
88
+ def _szs_for(status: str, reason: Optional[str]) -> str:
89
+ """Return the SZS ontology value for a (status, reason) pair."""
90
+ return _SZS.get((status, reason), _SZS[(status, None)])
91
+
92
+
93
+ # ---------------------------------------------------------------------------
94
+ # Verdict
95
+ # ---------------------------------------------------------------------------
96
+
97
+ @dataclass(frozen=True)
98
+ class Verdict:
99
+ """One decision result, whatever route produced it.
100
+
101
+ Fields:
102
+
103
+ * status: ``"proved"`` / ``"refuted"`` / ``"unknown"`` / ``"error"``.
104
+ Truthiness follows ``status == "proved"``.
105
+ * backend: name of the backend that produced this verdict.
106
+ * logic: the logic the query was decided in (``"fol"``, ``"modal"``, …).
107
+ * reason: the why-not-more axis for UNKNOWN/ERROR (see module docstring);
108
+ ``None`` for definitive verdicts.
109
+ * szs_status: SZS ontology value; derived from (status, reason) unless a
110
+ backend sets it explicitly (e.g. verbatim from a TSTP line).
111
+ * wall_time: seconds spent inside the backend call.
112
+ * countermodel: JSON-able witness for REFUTED (shape is backend-specific
113
+ but always a dict with a ``"kind"`` key), else ``None``.
114
+ * proof: JSON-able proof object where the backend yields one, else None.
115
+ A ``"kind": "z3_unsat_core"`` or ``"kind": "cvc5_alethe"`` proof
116
+ carries an unsat CORE — sound (re-asserting just that subset is still
117
+ unsat) but not necessarily MINIMAL (the underlying solver is free to
118
+ track more than the smallest sufficient subset) — see
119
+ :func:`z3_relevant_premises` for the same caveat on the dedicated
120
+ premise-relevance query.
121
+ * detail: short free-text note (method that closed it, bound that was
122
+ hit, tried-backends summary, …).
123
+ * agreement: backend names that reported the SAME status, filled by the
124
+ portfolio layer; a single-backend verdict lists just its own.
125
+ * relevant_premises: 0-based indices into the caller's ``premises`` that
126
+ a premise-relevance query found necessary for a PROVED verdict (see
127
+ :func:`z3_relevant_premises`, :mod:`atp.tstp`'s ancestor walk, and
128
+ :mod:`atp.eprover_backend`'s ``eprover_relevant_premises``); ``None``
129
+ when nobody computed it (the default) — NOT a claim that every premise
130
+ was needed. Like ``proof``'s unsat core, a reported set is sound but
131
+ not guaranteed minimal.
132
+ * solver_version: the underlying external tool's own version string
133
+ (Vampire's/Prover9's/E's/Zipperposition's ``--version`` banner, the
134
+ reachable HETS server's ``GET /version`` text, the installed ``cvc5``
135
+ package's distribution version), captured once per process per
136
+ backend/binary and reported here unchanged — see
137
+ :meth:`ProverBackend.solver_version` and :func:`_binary_version` for
138
+ the memoization contract. ``None`` for the kit's own internal
139
+ backends (Z3/tableau/resolution/modelfinder/QML/…, which have no
140
+ external tool to version) and for an external backend whose version
141
+ lookup itself failed or found nothing (the tool's PROOF/DISPROOF
142
+ verdict is unaffected either way — a missing version string is never
143
+ grounds to downgrade a sound answer).
144
+ """
145
+
146
+ status: str
147
+ backend: str
148
+ logic: str = "fol"
149
+ reason: Optional[str] = None
150
+ szs_status: Optional[str] = None
151
+ wall_time: float = 0.0
152
+ countermodel: Optional[dict] = None
153
+ proof: Optional[dict] = None
154
+ detail: Optional[str] = None
155
+ agreement: Tuple[str, ...] = ()
156
+ relevant_premises: Optional[Tuple[int, ...]] = None
157
+ # New fields go last: positional construction by callers must keep working.
158
+ solver_version: Optional[str] = None
159
+
160
+ def __post_init__(self):
161
+ if self.status not in STATUSES:
162
+ raise ValueError(f"Verdict: unknown status {self.status!r} (use one of {STATUSES})")
163
+ if self.szs_status is None:
164
+ object.__setattr__(self, "szs_status", _szs_for(self.status, self.reason))
165
+ if not self.agreement:
166
+ object.__setattr__(self, "agreement", (self.backend,))
167
+
168
+ def __bool__(self) -> bool:
169
+ return self.status == PROVED
170
+
171
+ @property
172
+ def is_definitive(self) -> bool:
173
+ """True iff the verdict settles the question (proved or refuted)."""
174
+ return self.status in (PROVED, REFUTED)
175
+
176
+ def to_dict(self) -> dict:
177
+ """Serialise to a JSON-compatible dict (all fields, names as keys)."""
178
+ return {
179
+ "status": self.status,
180
+ "backend": self.backend,
181
+ "logic": self.logic,
182
+ "reason": self.reason,
183
+ "szs_status": self.szs_status,
184
+ "wall_time": self.wall_time,
185
+ "countermodel": self.countermodel,
186
+ "proof": self.proof,
187
+ "detail": self.detail,
188
+ "agreement": list(self.agreement),
189
+ "relevant_premises": (list(self.relevant_premises)
190
+ if self.relevant_premises is not None else None),
191
+ "solver_version": self.solver_version,
192
+ }
193
+
194
+
195
+ class BackendUnavailable(RuntimeError):
196
+ """A KNOWN backend was requested but its prerequisites are missing.
197
+
198
+ Raised instead of silently skipping, because a pipeline whose backend
199
+ quietly vanished produces irreproducible numbers. The message names the
200
+ backend and what discovery looked for (env var / binary / install).
201
+ """
202
+
203
+
204
+ # ---------------------------------------------------------------------------
205
+ # The backend contract
206
+ # ---------------------------------------------------------------------------
207
+
208
+ class ProverBackend(ABC):
209
+ """Adapter contract for one decision route.
210
+
211
+ ``name`` is the registry key; ``logics`` the set of logic labels the
212
+ backend accepts (``"fol"``, ``"modal"``). ``available()`` performs
213
+ discovery without side effects. ``decide()`` answers "do ``premises``
214
+ entail ``formula``?" (validity when ``premises`` is empty) and must NEVER
215
+ raise for an in-contract input — undecidable/unsupported fragments come
216
+ back as an UNKNOWN verdict with the honest ``reason``. Extra keyword
217
+ options are forwarded verbatim to the underlying route (``frame=``,
218
+ ``systems=``, ``max_steps=``, …), so the adapters stay thin and nothing
219
+ of the existing signatures is hidden. A backend that names the options it
220
+ reads (:meth:`accepted_options`) is handed only those by
221
+ :func:`~unicode_logic_kit.api.prove`, which refuses an option that no backend
222
+ of the chain reads; :meth:`available_for` answers for the route the options
223
+ of a call select.
224
+
225
+ For the modal backends, a non-empty ``premises`` is the LOCAL consequence
226
+ ``⊨ (∧ premises) → φ`` — the standard finite-premise reading; global
227
+ consequence is out of scope here.
228
+ """
229
+
230
+ name: str = ""
231
+ logics: frozenset = frozenset({"fol"})
232
+ external: bool = False # needs a binary/install outside this venv
233
+
234
+ @abstractmethod
235
+ def available(self) -> bool:
236
+ """Return whether this backend can run right now (pure discovery)."""
237
+
238
+ def available_for(self, options: dict) -> bool:
239
+ """Return whether this backend can run a call made with ``options``.
240
+
241
+ :meth:`available` answers for the defaults. An option that picks the
242
+ route (``use_wsl=``, the path of a binary) can change the answer, so the
243
+ dispatcher asks THIS method, with the options of the call, and the answer
244
+ and the run agree. The default ignores ``options`` and returns
245
+ :meth:`available`.
246
+ """
247
+ return self.available()
248
+
249
+ def accepted_options(self, logic: Optional[str] = None) -> Optional[frozenset]:
250
+ """Return the names of the keyword options :meth:`decide` reads, or ``None``.
251
+
252
+ ``None`` (the default) means the backend declares nothing: the dispatcher
253
+ then passes it every option and cannot tell whether one is read. A
254
+ backend that declares its names lets :func:`~unicode_logic_kit.api.prove`
255
+ refuse an option that no backend of the chain reads, and pass each backend
256
+ only the options it reads. ``logic`` is the logic of the call, for a
257
+ backend whose options depend on it.
258
+ """
259
+ return None
260
+
261
+ @abstractmethod
262
+ def decide(self, formula: Node, premises: Sequence[Node] = (),
263
+ timeout: int = 10000, **options) -> Verdict:
264
+ """Decide ``premises ⊨ formula`` and return a :class:`Verdict`."""
265
+
266
+ def solver_version(self) -> Optional[str]:
267
+ """The underlying external tool's own version string, or ``None``.
268
+
269
+ Default (this base implementation): always ``None`` — the kit's own
270
+ internal calculi and semantic searches (Z3 aside: its Python BINDING
271
+ version is not "the solver's version" in the sense this method
272
+ means, and :class:`Z3Backend` does not override this) have no
273
+ external tool to version. An external backend overrides this with a
274
+ lookup that is PROCESS-LOCAL MEMOIZED (see :func:`_binary_version`
275
+ for the subprocess-spawning backends, and each override's own
276
+ docstring for the HETS/cvc5 routes) — probed at most once per
277
+ distinct binary/server/package for the life of this process, so
278
+ neither this method nor :meth:`decide` (which populates
279
+ ``Verdict.solver_version`` from it) ever pays a second lookup.
280
+ Never raises and never changes any verdict's status/reason — this
281
+ is provenance only, queried both from ``decide()`` and, independently,
282
+ from :func:`unicode_logic_kit.eval.batch.batch_decide`'s cache key (so
283
+ a solver upgrade invalidates stale cache entries).
284
+ """
285
+ return None
286
+
287
+
288
+ # ---------------------------------------------------------------------------
289
+ # Solver-version provenance: a process-local memoized ``--version`` lookup,
290
+ # shared by every subprocess-spawning backend below (Vampire, Prover9, and —
291
+ # via atp.eprover_backend's own import of this function — E/Zipperposition).
292
+ # HETS (an HTTP server, not a spawned binary) and cvc5 (a pip binding, not a
293
+ # spawned binary) have their own analogous, separately-memoized routes in
294
+ # their own modules (hets_backend.py, cvc5_backend.py) — see
295
+ # ProverBackend.solver_version's docstring.
296
+ # ---------------------------------------------------------------------------
297
+
298
+ #: (command, use_wsl) -> the tool's version string, or None on a failed
299
+ #: lookup — a MISS is cached too (never retried), matching item 5's "never
300
+ #: spawn a subprocess per decide() call after the first" requirement.
301
+ #: Deliberately separate from atp.eprover_backend._DISCOVERY_CACHE: discovery
302
+ #: (does a binary exist at all?) and version (what does it print?) answer
303
+ #: different questions and can fail independently — collapsing them into one
304
+ #: cache would make a version-lookup failure look like the binary vanished,
305
+ #: or vice versa.
306
+ _VERSION_CACHE: Dict[Tuple[str, bool], Optional[str]] = {}
307
+
308
+
309
+ def _binary_version(command: str, use_wsl: bool,
310
+ args: Tuple[str, ...] = ("--version",)) -> Optional[str]:
311
+ """Process-local memoized ``<command> <args>`` version lookup.
312
+
313
+ Spawns the subprocess (through ``wsl.exe`` when ``use_wsl``) at most ONCE
314
+ per ``(command, use_wsl)`` pair for the life of this process; every
315
+ later call for the same pair — including one that failed — returns the
316
+ cached result without spawning again. On a ZERO exit, returns the first
317
+ non-blank line of stdout, falling back to stderr (some tools print
318
+ version banners there), stripped. ``None`` on any failure: binary
319
+ missing, a timeout, WSL unreachable, a non-zero exit — an unrecognized
320
+ flag commonly prints an error/usage line to stdout or stderr on a
321
+ non-zero exit, and accepting that text as a version string would be
322
+ worse than reporting no provenance at all (mirrors the returncode check
323
+ :func:`unicode_logic_kit.atp.eprover_backend._discover` already applies to
324
+ its own subprocess probe) — OR the banner not being valid text under
325
+ ``subprocess.run(..., text=True)``'s decoding (``UnicodeError``, e.g. a
326
+ tool that writes a non-UTF-8 locale-encoded byte in its ``--version``
327
+ output): decoding happens INSIDE ``subprocess.run`` here, so this is
328
+ caught exactly like any other failure to read the banner, never left to
329
+ propagate as a bare ``ValueError`` out of a caller's ``decide()``. This
330
+ is best-effort provenance, so nothing here ever raises or turns into an
331
+ ERROR verdict.
332
+ """
333
+ key = (command, use_wsl)
334
+ if key in _VERSION_CACHE:
335
+ return _VERSION_CACHE[key]
336
+
337
+ import subprocess
338
+
339
+ result: Optional[str] = None
340
+ try:
341
+ cmd = ["wsl.exe", command, *args] if use_wsl else [command, *args]
342
+ # The tool never reads this process's stdin: Prover9 reads its problem from
343
+ # stdin for any flag it does not know (``--version`` included), and would
344
+ # block on an inherited pipe (an MCP server's own protocol stream) or read it.
345
+ proc = subprocess.run(cmd, capture_output=True, text=True, timeout=10,
346
+ stdin=subprocess.DEVNULL)
347
+ if proc.returncode == 0:
348
+ text = (proc.stdout or "").strip() or (proc.stderr or "").strip()
349
+ if text:
350
+ result = text.splitlines()[0].strip()
351
+ except (OSError, subprocess.TimeoutExpired, UnicodeError):
352
+ result = None
353
+ _VERSION_CACHE[key] = result
354
+ return result
355
+
356
+
357
+ #: binary -> whether ``wsl.exe which <binary>`` found it. An answer is cached for
358
+ #: the life of the process, like :data:`_VERSION_CACHE`, so a discovery that
359
+ #: spawns ``wsl.exe`` is paid once per binary. A probe that TIMED OUT is not an
360
+ #: answer and is not cached: WSL may only have been slow to start.
361
+ _WSL_BINARY_CACHE: Dict[str, bool] = {}
362
+
363
+ #: Seconds the WSL probe of :func:`_wsl_has_binary` may take.
364
+ _WSL_PROBE_TIMEOUT = 10
365
+
366
+
367
+ def _wsl_has_binary(binary: str) -> bool:
368
+ """Whether ``binary`` (a name on the PATH inside WSL, or a path inside WSL)
369
+ exists there and is executable.
370
+
371
+ This is what ``available()`` has to ask when a backend will run its binary
372
+ through ``wsl.exe``: the host's own PATH says nothing about a Linux binary.
373
+ Asks ``wsl.exe which <binary>`` -- the command the runners themselves use to
374
+ start it -- with a short timeout and an empty standard input, so the probe
375
+ never blocks on, or reads, this process's input. ``False`` when ``wsl.exe``
376
+ cannot be started (no WSL on this host), when it times out, and when the
377
+ binary is not found; never raises.
378
+ """
379
+ if binary in _WSL_BINARY_CACHE:
380
+ return _WSL_BINARY_CACHE[binary]
381
+
382
+ import subprocess
383
+
384
+ try:
385
+ proc = subprocess.run(["wsl.exe", "which", binary], capture_output=True,
386
+ timeout=_WSL_PROBE_TIMEOUT, stdin=subprocess.DEVNULL)
387
+ except subprocess.TimeoutExpired:
388
+ return False
389
+ except OSError:
390
+ found = False
391
+ else:
392
+ found = proc.returncode == 0 and bool(
393
+ (proc.stdout or b"").decode("utf-8", errors="replace").replace("\x00", "").strip())
394
+ _WSL_BINARY_CACHE[binary] = found
395
+ return found
396
+
397
+
398
+ def _native_command_exists(command: str) -> bool:
399
+ """Whether this host can start ``command``: an executable file at that path, or a
400
+ name that is on ``PATH``.
401
+
402
+ ``shutil.which`` answers both, except for one spelling that the run accepts: on
403
+ Windows a path with a directory part and no extension is looked up exactly as
404
+ written by ``shutil.which`` (before Python 3.12), whereas ``subprocess`` starts it
405
+ through ``CreateProcess``, which appends ``.exe`` to a name that has no extension.
406
+ A caller who names the installed ``minizinc.exe`` as ``D:/Minizinc/MiniZinc/minizinc``
407
+ names a binary the run starts, so that spelling is tried too. Never starts anything.
408
+ """
409
+ import os
410
+
411
+ if shutil.which(command) is not None:
412
+ return True
413
+ if os.name == "nt" and not os.path.splitext(command)[1]:
414
+ return shutil.which(command + ".exe") is not None
415
+ return False
416
+
417
+
418
+ def _implication(formula: Node, premises: Sequence[Node]) -> Node:
419
+ """Fold ``premises ⊨ φ`` into the single formula ``(∧ premises) → φ``."""
420
+ premises = list(premises)
421
+ if not premises:
422
+ return formula
423
+ conj = premises[0]
424
+ for p in premises[1:]:
425
+ conj = And(conj, p)
426
+ return Implies(conj, formula)
427
+
428
+
429
+ def _timed(fn):
430
+ """Run ``fn()`` returning ``(result, seconds)``."""
431
+ start = time.perf_counter()
432
+ result = fn()
433
+ return result, time.perf_counter() - start
434
+
435
+
436
+ # ---------------------------------------------------------------------------
437
+ # A prover that REFUSES its input is an error, not a timeout.
438
+ #
439
+ # Every subprocess backend reads a verdict off its prover's output: an SZS
440
+ # status line (Vampire, E, Zipperposition), a ``RESULT:`` line (Twee), a
441
+ # ``THEOREM PROVED`` line (Prover9). A prover that cannot read the problem
442
+ # prints none of those -- it prints its own complaint instead (Vampire 5.0.1:
443
+ # ``User error: Non-boolean term agent(agent(X0)) of sort $i is used in a
444
+ # formula context``; E 3.5.1: ``eprover: <file>:1:(Column 45): ... expected,
445
+ # but Closing bracket (')') read``; Twee 2.6.1: ``Error in <file> (line 2,
446
+ # column 1): Unexpected fof``) -- and a backend that reads "no verdict" as
447
+ # "did not get there in time" reports a defect in the problem WRITER as an
448
+ # undecided question. One wording, shared by every backend, keeps the three
449
+ # readings (timeout / honest give-up / refusal) apart wherever the outcome is
450
+ # shown.
451
+ # ---------------------------------------------------------------------------
452
+
453
+ #: Lines of a prover's output that explain a failure. Used only to choose
454
+ #: WHERE to start quoting a long output; a short one is quoted whole.
455
+ _FAILURE_LINE = re.compile(
456
+ r"error|exception|assertion|fatal|abort|segmentation|core dumped|cannot|"
457
+ r"unexpected|expected|failed|invalid", re.IGNORECASE)
458
+
459
+ #: How much of the prover's own output a refusal's ``detail`` quotes.
460
+ _REJECTION_QUOTE_CHARS = 600
461
+
462
+
463
+ def _rejection_detail(prover: str, output: str, evidence: str) -> str:
464
+ """The ``detail`` of an ERROR verdict for a prover that gave no verdict.
465
+
466
+ ``evidence`` says what was missing (``"no SZS status line in its output"``,
467
+ ``"no RESULT line in its output"``, ...). The prover's OWN text follows,
468
+ one line per ``|``-separated item, unmodified apart from stripping blank
469
+ lines, NUL bytes (a Windows ``wsl.exe`` diagnostic arrives UTF-16 decoded
470
+ as 8-bit) and trailing whitespace. An output longer than
471
+ :data:`_REJECTION_QUOTE_CHARS` is quoted from its first failure-looking
472
+ line, or from its tail when no line looks like one -- the explanation of a
473
+ refusal is at the start (Vampire) and that of a crash at the end (E's
474
+ ``Assertion ... failed``), and a run that printed pages of statistics
475
+ first would otherwise bury either.
476
+ """
477
+ lines = [ln.strip() for ln in output.replace("\x00", "").splitlines()]
478
+ lines = [ln for ln in lines if ln]
479
+ message = " | ".join(lines)
480
+ if len(message) > _REJECTION_QUOTE_CHARS:
481
+ start = next((i for i, ln in enumerate(lines) if _FAILURE_LINE.search(ln)), None)
482
+ if start is None:
483
+ message = "... " + message[-_REJECTION_QUOTE_CHARS:]
484
+ else:
485
+ message = " | ".join(lines[start:])
486
+ if len(message) > _REJECTION_QUOTE_CHARS:
487
+ message = message[:_REJECTION_QUOTE_CHARS] + " ..."
488
+ if not message:
489
+ message = "(it printed nothing)"
490
+ return (f"{prover} produced no verdict ({evidence}); this is a failure, "
491
+ f"not a timeout. Its own output: {message}")
492
+
493
+
494
+ def _no_definitive_verdict(pseudo_backend: str, logic: str,
495
+ verdicts: Sequence[Verdict], empty: str) -> Verdict:
496
+ """The collective verdict of a chain/portfolio in which nobody settled it.
497
+
498
+ ``"<backend>:<status>[/<reason>] (<detail>)"`` for EVERY member that ran:
499
+ its status and reason, and its own account of them when it gave one. A member
500
+ that FAILED (ERROR, or any member whose reason is ``"infra"`` -- the Isabelle
501
+ adapter reports that as UNKNOWN) quotes its ``detail``, so a prover's
502
+ refusal reaches the caller instead of being flattened into the bare
503
+ ``vampire:error/infra`` that reads like any other "no answer"; so does a
504
+ member that REFUSED the question (reason ``"unsupported"``): the writer's
505
+ message says what it would not write and what to write instead, and without it
506
+ the caller has only the word "unsupported"; so does a member that gave up
507
+ (``"bound_hit"``, ``"incomplete"``, ``"timeout"``): the bound that was hit, or
508
+ the point where the search ended, is what tells the caller which budget to
509
+ raise. A member without a ``detail`` is listed by status and reason alone. When EVERY
510
+ member failed there is no honest ``unknown`` to report -- nothing was asked
511
+ and answered -- so the collective verdict is itself an ERROR; with at least
512
+ one member that really answered "unknown" it stays UNKNOWN and the failures
513
+ sit in the detail beside it.
514
+ """
515
+ if not verdicts:
516
+ return Verdict(UNKNOWN, pseudo_backend, logic=logic, detail=empty)
517
+ summary = "; ".join(
518
+ f"{v.backend}:{v.status}" + (f"/{v.reason}" if v.reason else "")
519
+ + (f" ({v.detail})" if v.detail else "")
520
+ for v in verdicts)
521
+ if all(v.status == ERROR for v in verdicts):
522
+ return Verdict(ERROR, pseudo_backend, logic=logic, reason="infra",
523
+ detail=f"no backend gave an answer — {summary}")
524
+ return Verdict(UNKNOWN, pseudo_backend, logic=logic,
525
+ detail=f"no definitive verdict — {summary}")
526
+
527
+
528
+ def _z3_sort_axioms(formula: Node, premises: Sequence[Node], env=None) -> list:
529
+ """``sort_axioms(formula, *premises)``, each translated via ``to_z3(env)``.
530
+
531
+ ``env`` is the :class:`~unicode_logic_kit.fol.nodes.Z3Env` the goal and the
532
+ premises were translated with (the caller's one environment for the whole
533
+ problem).
534
+
535
+ That is every sort's non-emptiness AND the membership atom ``S(c)`` of every
536
+ sorted constant ``c:S`` of the goal or of a premise: ``to_z3()`` reads a
537
+ sorted constant as the plain constant, so without the atom the guard
538
+ reading forgets which sort it was written in. Shared by :class:`Z3Backend`
539
+ and :func:`z3_relevant_premises` so both decide the exact same many-sorted
540
+ entailment — see :func:`_z3_track_and_check`'s ``z3_sort_axioms`` parameter
541
+ for why these are added UNTRACKED rather than as more ``p<i>``-tagged
542
+ premises. Empty for an unsorted query.
543
+ """
544
+ from ..fol._msfl_nodes import sort_axioms
545
+ return [axiom.to_z3(env) for axiom in sort_axioms(formula, *premises)]
546
+
547
+
548
+ # ---------------------------------------------------------------------------
549
+ # Internal backends: the kit's own calculi and semantic searches
550
+ # ---------------------------------------------------------------------------
551
+
552
+ #: The tracking literals of :func:`_z3_track_and_check` are Boolean constants whose Z3
553
+ #: symbol is an INTEGER symbol (``Z3_mk_int_symbol``), numbered from this base. Every name
554
+ #: the kit writes into a Z3 expression is a STRING symbol, so no formula of any caller --
555
+ #: whatever its propositions, predicates and sorts are called -- can hold a symbol equal to a
556
+ #: tag: a proposition named ``goal`` or ``p0`` is another constant. The base keeps the numbers
557
+ #: clear of the ones Z3 itself gives its own fresh constants (small counters).
558
+ _Z3_TAG_BASE = 1 << 29
559
+
560
+
561
+ def _z3_tag(number: int):
562
+ """The tracking literal number ``number``: 0 is the negated goal's, ``1 + i`` premise ``i``'s."""
563
+ from z3 import BoolRef, BoolSort, Z3_mk_const, Z3_mk_int_symbol, main_ctx
564
+
565
+ ctx = main_ctx()
566
+ symbol = Z3_mk_int_symbol(ctx.ref(), _Z3_TAG_BASE + number)
567
+ return BoolRef(Z3_mk_const(ctx.ref(), symbol, BoolSort(ctx).ast), ctx)
568
+
569
+
570
+ def _z3_tag_number(decl) -> Optional[int]:
571
+ """The number of the tracking literal that the Z3 declaration ``decl`` is, or ``None``.
572
+
573
+ A declaration is a tracking literal exactly when it has no argument, returns ``Bool`` and
574
+ is named by an integer symbol of the band :data:`_Z3_TAG_BASE` and above (see there), which
575
+ no symbol of a kit formula is.
576
+ """
577
+ from z3 import BoolSort, Z3_INT_SYMBOL, Z3_get_decl_name, Z3_get_symbol_int, Z3_get_symbol_kind
578
+
579
+ if decl.arity() != 0 or decl.range() != BoolSort(decl.ctx):
580
+ return None
581
+ ref = decl.ctx.ref()
582
+ symbol = Z3_get_decl_name(ref, decl.ast)
583
+ if Z3_get_symbol_kind(ref, symbol) != Z3_INT_SYMBOL:
584
+ return None
585
+ value = Z3_get_symbol_int(ref, symbol)
586
+ return value - _Z3_TAG_BASE if value >= _Z3_TAG_BASE else None
587
+
588
+
589
+ def _z3_core_numbers(solver) -> list:
590
+ """The numbers of the tracking literals in ``solver.unsat_core()`` (0 is the goal, ``1 + i`` premise ``i``)."""
591
+ numbers = (_z3_tag_number(tag.decl()) for tag in solver.unsat_core())
592
+ return sorted(number for number in numbers if number is not None)
593
+
594
+
595
+ def _z3_core_names(solver) -> list:
596
+ """The unsat core of ``solver`` as the names a proof reports: ``"goal"`` and ``"p<i>"`` (0-based)."""
597
+ return [("goal" if number == 0 else f"p{number - 1}") for number in _z3_core_numbers(solver)]
598
+
599
+
600
+ def _z3_track_and_check(z3_formula, z3_premises: Sequence, timeout: int,
601
+ z3_sort_axioms: Sequence = ()):
602
+ """Run ONE per-call Z3 ``Solver``, tracking every assertion by name.
603
+
604
+ Asserts ``Not(z3_formula)`` under the tag ``"goal"`` and each of
605
+ ``z3_premises`` under ``"p<i>"`` (0-based), via ``assert_and_track``,
606
+ with ``unsat_core=True`` set on THIS solver instance only — never
607
+ ``z3.set_param(proof=True)``, which is a process-wide global that would
608
+ change solving behaviour for every other Z3 consumer in the kit
609
+ (semantics evaluators, dl, chem, finite-model, modal/second/third-order
610
+ — anything sharing the module-level ``_SORT``/default context in
611
+ ``fol/_fol_nodes.py``) for the rest of the process.
612
+
613
+ ``z3_sort_axioms`` (see
614
+ :func:`~unicode_logic_kit.fol._msfl_nodes.sort_axioms`: every sort is
615
+ non-empty, and a sorted constant ``c:S`` is in ``S``) are added with a
616
+ plain, UNTRACKED ``solver.add`` — they are background MSFOL convention,
617
+ never one of the caller's own premises, so they must never gain a ``p<i>``
618
+ tag: that would make an untranslatable, caller-invisible synthetic sentence
619
+ show up in :class:`Z3Backend`'s ``proof`` unsat core or in
620
+ :func:`z3_relevant_premises`'s reported indices, which are defined purely
621
+ over the CALLER's own ``premises`` list. They are asserted OUTSIDE the
622
+ negated goal: a membership atom under the goal's negation would be
623
+ something to prove. Empty (the default) for an unsorted query, so the
624
+ solver call is byte-for-byte the same as before this parameter existed.
625
+
626
+ **The tags are named by integer symbols.** A proposition of a problem can be
627
+ called anything, ``goal`` and ``p0`` included (the Prover9 and SMT-LIB readers
628
+ produce such propositions); a tag spelled like one would be the very proposition,
629
+ assumed true, and would prove any goal. So a tag is a Boolean constant with an
630
+ integer symbol (:func:`_z3_tag`), a kind of name that no string a formula carries
631
+ can be, and the names ``"goal"`` / ``"p<i>"`` are only what a proof REPORTS
632
+ (:func:`_z3_core_names`).
633
+
634
+ Built once and shared by :class:`Z3Backend` (C12: a per-verdict
635
+ ``z3_unsat_core`` proof certificate) and :func:`z3_relevant_premises`
636
+ (C11: which premises a PROVED entailment actually needed) so both read
637
+ the exact same assert-and-track call shape rather than drifting apart —
638
+ including the identical sort axioms, so the two can never disagree about
639
+ whether a many-sorted entailment holds.
640
+
641
+ Returns ``(result, solver)`` — ``result`` is Z3's own
642
+ ``sat``/``unsat``/``unknown``; on ``unsat``, ``solver.unsat_core()``
643
+ holds the tracked-name subset Z3 actually used. That subset is SOUND
644
+ (re-asserting just it is still unsat) but not necessarily MINIMAL (Z3's
645
+ core extraction is not obliged to find the smallest one) — callers that
646
+ need "used" language should say "relevant"/"a sufficient subset", never
647
+ "the minimal set".
648
+ """
649
+ from z3 import Solver, Not as _ZNot
650
+
651
+ solver = Solver()
652
+ solver.set("timeout", timeout)
653
+ solver.set("random_seed", 42)
654
+ solver.set(unsat_core=True)
655
+ for axiom in z3_sort_axioms:
656
+ solver.add(axiom)
657
+ for i, p in enumerate(z3_premises):
658
+ solver.assert_and_track(p, _z3_tag(1 + i))
659
+ solver.assert_and_track(_ZNot(z3_formula), _z3_tag(0))
660
+ return solver.check(), solver
661
+
662
+
663
+ def _z3_model_assignment(model, n_premises: int = 0) -> Dict[str, str]:
664
+ """Read a satisfying ``z3.ModelRef`` back into a ``{name: value}`` dict.
665
+
666
+ ``assert_and_track``'s own tracking booleans (see :func:`_z3_track_and_check`)
667
+ are 0-ary Bool-sorted Z3 declarations, so they show up in ``model.decls()``
668
+ right alongside the formula's real symbols and would otherwise leak into a
669
+ REFUTED verdict's witness as spurious extra keys. They are excluded by what
670
+ they are, not by their spelling: a tag is a constant with an integer symbol
671
+ (:func:`_z3_tag_number`), and no symbol of a formula is, so a proposition
672
+ named ``goal`` or ``p0`` is reported like any other. ``n_premises`` is not
673
+ needed to recognise a tag and is accepted for the callers that pass it.
674
+
675
+ A name that the model declares more than once (``P`` at two arities, a
676
+ function and a predicate of one name) is two symbols and is reported under
677
+ ``"P/1"`` / ``"P/2"`` keys instead of one overwriting the other — see
678
+ :func:`~unicode_logic_kit.atp.z3_models.declaration_keys`; a name declared
679
+ once keeps its plain name.
680
+ """
681
+ from .z3_models import model_assignment
682
+
683
+ return model_assignment(model, skip=lambda d: _z3_tag_number(d) is not None)
684
+
685
+
686
+ def z3_relevant_premises(formula: Node, premises: Sequence[Node] = (),
687
+ timeout: int = 10000) -> Optional[Tuple[int, ...]]:
688
+ """Which of ``premises`` did Z3 actually need to prove ``⊨ formula``?
689
+
690
+ Runs the same :func:`_z3_track_and_check` per-``Solver`` assert-and-track
691
+ call :class:`Z3Backend` uses for its own ``proof`` certificate — including
692
+ the same many-sorted axioms (non-emptiness of every sort, membership of
693
+ every sorted constant) when ``formula``/``premises`` use a sort
694
+ (:func:`_z3_sort_axioms`), so this always agrees with
695
+ :class:`Z3Backend` about whether the entailment holds — but reports only
696
+ the premise side of the core, as 0-based indices into ``premises`` (never
697
+ including the ``"goal"`` tag itself — the negated conclusion is always
698
+ "needed" trivially, so it carries no information about which PREMISES
699
+ were relevant; the sort axioms are untracked, so they can never appear in
700
+ the core either — see :func:`_z3_track_and_check`).
701
+
702
+ Args:
703
+ formula: the goal.
704
+ premises: candidate premises (same fragment ``Z3Backend``/
705
+ ``Node.to_z3`` decides — uninterpreted sort + equality, no
706
+ arithmetic; substructural nodes are outside it).
707
+ timeout: milliseconds, forwarded to the solver exactly as
708
+ ``Z3Backend.decide`` forwards its own ``timeout``.
709
+
710
+ Returns:
711
+ A sorted tuple of 0-based premise indices, or ``None`` when there is
712
+ nothing sound to report: the entailment does not hold (Z3 finds it
713
+ SAT or times out UNKNOWN — there is no "used premises" answer for a
714
+ non-theorem), or the fragment is unsupported (``to_z3`` raises
715
+ ``NotImplementedError`` on ``formula`` or any premise). ``None`` is
716
+ the honest "don't know" answer here, never a guessed subset.
717
+
718
+ The returned set is SOUND but not necessarily MINIMAL — see
719
+ :func:`_z3_track_and_check`'s docstring; a genuinely redundant
720
+ premise (one that, alone, already suffices) need not appear
721
+ alongside the other route to the same conclusion, but two premises
722
+ that are each independently sufficient are not guaranteed to be
723
+ pruned down to a single one either — only that the returned subset
724
+ itself is enough.
725
+ """
726
+ from ..fol.nodes import Z3Env
727
+
728
+ premises = list(premises)
729
+ try:
730
+ env = Z3Env()
731
+ z3_formula = formula.to_z3(env)
732
+ z3_premises = [p.to_z3(env) for p in premises]
733
+ z3_sorts = _z3_sort_axioms(formula, premises, env)
734
+ except NotImplementedError:
735
+ return None
736
+
737
+ from z3 import unsat
738
+
739
+ res, solver = _z3_track_and_check(z3_formula, z3_premises, timeout, z3_sorts)
740
+ if res != unsat:
741
+ return None
742
+ return tuple(sorted({number - 1 for number in _z3_core_numbers(solver) if number > 0}))
743
+
744
+
745
+ class Z3Backend(ProverBackend):
746
+ """Classical FOL/MSFOL via Z3 — tri-state, with a model on refutation.
747
+
748
+ Many-sorted input: a sort's non-emptiness and the membership of every
749
+ sorted constant in its sort (the MSFOL convention — see the
750
+ classical-reasoning guide's many-sorted section) are asserted as extra,
751
+ untracked, UNNEGATED premises alongside ``premises`` — see
752
+ :func:`_z3_sort_axioms` and :func:`_z3_track_and_check` — never folded
753
+ inside ``Node.to_z3()`` itself, which stays polarity-blind. This is what
754
+ makes a REFUTED verdict's countermodel always a legal MSFOL structure (no
755
+ sort empty, every sorted constant inside its sort) instead of exploiting a
756
+ loophole ``semantics.modelfinder`` never considers, and what makes
757
+ ``∀x:Human Mortal(x) ⊢ Mortal(socrates:Human)`` PROVED.
758
+ """
759
+
760
+ name = "z3"
761
+ logics = frozenset({"fol"})
762
+ external = False # z3-solver is a hard dependency
763
+
764
+ def available(self) -> bool:
765
+ return True
766
+
767
+ def decide(self, formula: Node, premises: Sequence[Node] = (),
768
+ timeout: int = 10000, **options) -> Verdict:
769
+ from z3 import sat, unsat
770
+ from ..fol.nodes import Z3Env
771
+
772
+ premises = list(premises)
773
+ try:
774
+ # ONE environment for the goal and every premise: a numeral and a
775
+ # constant of the same text are refused wherever they meet, and
776
+ # one name at two arities stays two symbols across the problem.
777
+ env = Z3Env()
778
+ z3_formula = formula.to_z3(env)
779
+ z3_premises = [p.to_z3(env) for p in premises]
780
+ z3_sorts = _z3_sort_axioms(formula, premises, env)
781
+ except NotImplementedError as exc:
782
+ return Verdict(UNKNOWN, self.name, reason="unsupported", detail=str(exc))
783
+
784
+ (res, solver), elapsed = _timed(
785
+ lambda: _z3_track_and_check(z3_formula, z3_premises, timeout, z3_sorts))
786
+ if res == unsat:
787
+ # A tracked-name unsat core, per _z3_track_and_check's docstring:
788
+ # sound (re-asserting just these is still unsat, see the C12
789
+ # soundness self-check test) but not necessarily minimal.
790
+ core = sorted(_z3_core_names(solver))
791
+ proof = {"kind": "z3_unsat_core", "core": core}
792
+ return Verdict(PROVED, self.name, wall_time=elapsed, proof=proof)
793
+ if res == sat:
794
+ assignment = _z3_model_assignment(solver.model(), len(z3_premises))
795
+ return Verdict(REFUTED, self.name, wall_time=elapsed,
796
+ countermodel={"kind": "z3_model", "assignment": assignment})
797
+ why = solver.reason_unknown()
798
+ reason = "timeout" if ("timeout" in why or "cancel" in why) else "incomplete"
799
+ return Verdict(UNKNOWN, self.name, reason=reason, wall_time=elapsed, detail=why)
800
+
801
+
802
+ class TableauBackend(ProverBackend):
803
+ """The kit's own classical analytic tableau (Z3-free).
804
+
805
+ Complete and decidable propositionally; on first-order inputs a
806
+ non-closure is only "no closed tableau within the bounds" → bound_hit.
807
+ The bounds are ``max_steps`` (which also bounds the length of a branch: the
808
+ search is a loop over a branch, it does not depend on the interpreter's
809
+ recursion limit), ``max_terms`` and the call's ``timeout`` (a run that used it
810
+ up is ``timeout``, not ``bound_hit``). A formula nested deeper than the helpers
811
+ that walk it can follow within the recursion limit is a bound too: the verdict
812
+ is ``unknown`` / ``bound_hit`` and its ``detail`` names the nesting depth
813
+ (:func:`~unicode_logic_kit.atp.tableau.nesting_depth`). Many-sorted input is
814
+ searched through its guard image with the sort axioms as further formulas to
815
+ refute (see :func:`~unicode_logic_kit.atp.tableau.tableau_closed`).
816
+ """
817
+
818
+ name = "tableau"
819
+ logics = frozenset({"fol"})
820
+ external = False
821
+
822
+ def available(self) -> bool:
823
+ return True
824
+
825
+ def decide(self, formula: Node, premises: Sequence[Node] = (),
826
+ timeout: int = 10000, **options) -> Verdict:
827
+ # The detailed route records a TableauProof alongside the same
828
+ # search (recording never changes the verdict — regression-pinned
829
+ # in test_tableau_check.py), so a PROVED Verdict can carry the
830
+ # independently checkable proof dict.
831
+ from .tableau import _search_detailed
832
+
833
+ try:
834
+ (proof, nesting), elapsed = _timed(
835
+ lambda: _search_detailed(list(premises), formula,
836
+ timeout=timeout, **options))
837
+ except NotImplementedError as exc:
838
+ return Verdict(UNKNOWN, self.name, reason="unsupported", detail=str(exc))
839
+ if proof is not None:
840
+ try:
841
+ encoded = proof.to_dict()
842
+ except RecursionError:
843
+ encoded = None # a proof too deeply nested to serialise is still a proof
844
+ return Verdict(PROVED, self.name, wall_time=elapsed, proof=encoded)
845
+ if nesting is not None:
846
+ import sys
847
+ return Verdict(UNKNOWN, self.name, reason="bound_hit", wall_time=elapsed,
848
+ detail=f"a formula nested {nesting} levels deep is deeper than "
849
+ "the tableau's recursive helpers can walk within the "
850
+ f"interpreter's recursion limit ({sys.getrecursionlimit()}): "
851
+ "no closed tableau was found")
852
+ if elapsed * 1000 >= timeout:
853
+ return Verdict(UNKNOWN, self.name, reason="timeout", wall_time=elapsed,
854
+ detail=f"no closed tableau within the {timeout} ms limit")
855
+ return Verdict(UNKNOWN, self.name, reason="bound_hit", wall_time=elapsed,
856
+ detail="no closed tableau within max_steps/max_terms")
857
+
858
+
859
+ class ResolutionBackend(ProverBackend):
860
+ """The kit's own resolution prover (given-clause saturation).
861
+
862
+ The underlying bool API cannot distinguish saturation from a hit step
863
+ bound, so a False is reported as UNKNOWN/bound_hit — never as REFUTED.
864
+ Many-sorted input gets its sort axioms (every sort non-empty, every sorted
865
+ constant in its sort) as premise clauses from
866
+ :func:`~unicode_logic_kit.atp.resolution.prove`. The call's ``timeout`` bounds the
867
+ saturation; a run that used it up is ``timeout``, not ``bound_hit``.
868
+ """
869
+
870
+ name = "resolution"
871
+ logics = frozenset({"fol"})
872
+ external = False
873
+
874
+ def available(self) -> bool:
875
+ return True
876
+
877
+ def decide(self, formula: Node, premises: Sequence[Node] = (),
878
+ timeout: int = 10000, **options) -> Verdict:
879
+ from .resolution import prove as resolution_prove
880
+
881
+ try:
882
+ proved, elapsed = _timed(
883
+ lambda: resolution_prove(list(premises), formula, timeout=timeout,
884
+ **options))
885
+ except NotImplementedError as exc:
886
+ return Verdict(UNKNOWN, self.name, reason="unsupported", detail=str(exc))
887
+ if proved:
888
+ return Verdict(PROVED, self.name, wall_time=elapsed)
889
+ if elapsed * 1000 >= timeout:
890
+ return Verdict(UNKNOWN, self.name, reason="timeout", wall_time=elapsed,
891
+ detail=f"not refuted within the {timeout} ms limit")
892
+ return Verdict(UNKNOWN, self.name, reason="bound_hit", wall_time=elapsed,
893
+ detail="not refuted within max_steps (saturation not distinguished)")
894
+
895
+
896
+ class ModelFinderBackend(ProverBackend):
897
+ """The kit's finite model finder — a refutation-only semantic backend.
898
+
899
+ Finding a finite structure that satisfies the premises but not the
900
+ conclusion REFUTES the entailment; exhausting the size bound proves
901
+ nothing (FOL has no finite model property) → bound_hit. The call's
902
+ ``timeout`` ends the search at a candidate structure; a search that it cut
903
+ off is ``timeout``, not ``bound_hit``.
904
+ """
905
+
906
+ name = "modelfinder"
907
+ logics = frozenset({"fol"})
908
+ external = False
909
+
910
+ def available(self) -> bool:
911
+ return True
912
+
913
+ def decide(self, formula: Node, premises: Sequence[Node] = (),
914
+ timeout: int = 10000, **options) -> Verdict:
915
+ from ..semantics.modelfinder import search_countermodel
916
+
917
+ try:
918
+ search, elapsed = _timed(
919
+ lambda: search_countermodel(list(premises), formula, timeout=timeout,
920
+ **options))
921
+ except NotImplementedError as exc:
922
+ return Verdict(UNKNOWN, self.name, reason="unsupported", detail=str(exc))
923
+ if search.structure is not None:
924
+ return Verdict(REFUTED, self.name, wall_time=elapsed,
925
+ countermodel={"kind": "finite_structure",
926
+ "repr": repr(search.structure)})
927
+ if search.timed_out:
928
+ return Verdict(UNKNOWN, self.name, reason="timeout", wall_time=elapsed,
929
+ detail=f"no countermodel found within the {timeout} ms limit")
930
+ return Verdict(UNKNOWN, self.name, reason="bound_hit", wall_time=elapsed,
931
+ detail="no countermodel up to the size bound")
932
+
933
+
934
+ class ModalTableauBackend(ProverBackend):
935
+ """The kit's labelled modal tableau — tri-state with Kripke witnesses."""
936
+
937
+ name = "modal-tableau"
938
+ logics = frozenset({"modal"})
939
+ external = False
940
+
941
+ def available(self) -> bool:
942
+ return True
943
+
944
+ def decide(self, formula: Node, premises: Sequence[Node] = (),
945
+ timeout: int = 10000, **options) -> Verdict:
946
+ from .modal_tableau import _decide_explained, modal_countermodel
947
+
948
+ goal = _implication(formula, premises)
949
+ try:
950
+ (status, too_deep), elapsed = _timed(
951
+ lambda: _decide_explained(goal, timeout=timeout, **options))
952
+ except NotImplementedError as exc:
953
+ return Verdict(UNKNOWN, self.name, logic="modal",
954
+ reason="unsupported", detail=str(exc))
955
+ if status == "valid":
956
+ return Verdict(PROVED, self.name, logic="modal", wall_time=elapsed)
957
+ if status == "invalid":
958
+ # The witness is a second search; it gets what is left of the limit.
959
+ cm = modal_countermodel(goal, timeout=max(1, int(timeout - elapsed * 1000)),
960
+ **options)
961
+ witness = None
962
+ if cm is not None:
963
+ from .kripke_enum import kripke_model_to_dict
964
+ witness = {"kind": "kripke", "repr": repr(cm),
965
+ "data": kripke_model_to_dict(cm)}
966
+ return Verdict(REFUTED, self.name, logic="modal", wall_time=elapsed,
967
+ countermodel=witness)
968
+ # modal_decide answers "unknown" both for an exhausted budget and for
969
+ # the temporal-closure operators it has no rule for (G/F/U/H/O/P/S,
970
+ # classified "unsupported" internally) — tell them apart here so the
971
+ # reason axis stays honest.
972
+ from .modal_tableau import _TEMPORAL_CLOSURE
973
+ if any(isinstance(n, _TEMPORAL_CLOSURE) for n in goal.walk()):
974
+ return Verdict(UNKNOWN, self.name, logic="modal", reason="unsupported",
975
+ wall_time=elapsed,
976
+ detail="temporal closure operators (G/F/U/…) have no "
977
+ "tableau rule here for the kit's general Kripke "
978
+ "frame — for the STANDARD linear-time reading, "
979
+ "the 'ltl-tableau' backend (atp.ltl_tableau) is "
980
+ "a complete decision procedure and the "
981
+ "definitive route; 'qml' and 'isabelle' decide "
982
+ "the general (possibly non-linear) frame "
983
+ "instead, soundly but not completely")
984
+ if too_deep:
985
+ # One of the tableau's walks over the formula ran into the interpreter's recursion
986
+ # limit, which is one of its bounds: the formula was not read to its end.
987
+ import sys
988
+ from .tableau import nesting_depth
989
+ return Verdict(UNKNOWN, self.name, logic="modal", reason="bound_hit",
990
+ wall_time=elapsed,
991
+ detail=f"a formula nested {nesting_depth(goal)} levels deep is deeper "
992
+ "than the tableau's recursive walks can follow within the "
993
+ f"interpreter's recursion limit ({sys.getrecursionlimit()}): "
994
+ "nothing was decided")
995
+ if elapsed * 1000 >= timeout:
996
+ return Verdict(UNKNOWN, self.name, logic="modal", reason="timeout",
997
+ wall_time=elapsed,
998
+ detail=f"no verdict within the {timeout} ms limit")
999
+ return Verdict(UNKNOWN, self.name, logic="modal", reason="bound_hit",
1000
+ wall_time=elapsed,
1001
+ detail="tableau budget exhausted (max_worlds/max_steps)")
1002
+
1003
+
1004
+ class QmlBackend(ProverBackend):
1005
+ """Quantified modal logic via the FO embedding + Z3 — sound, incomplete.
1006
+
1007
+ ``True`` is a proof; ``False`` only means "not proven" (undecidable
1008
+ fragment), so it is reported as UNKNOWN/incomplete, never as REFUTED.
1009
+ """
1010
+
1011
+ name = "qml"
1012
+ logics = frozenset({"modal"})
1013
+ external = False
1014
+
1015
+ def available(self) -> bool:
1016
+ return True
1017
+
1018
+ def decide(self, formula: Node, premises: Sequence[Node] = (),
1019
+ timeout: int = 10000, **options) -> Verdict:
1020
+ from ..fol.qml import qml_is_valid
1021
+
1022
+ goal = _implication(formula, premises)
1023
+ try:
1024
+ proved, elapsed = _timed(
1025
+ lambda: qml_is_valid(goal, timeout=timeout, **options))
1026
+ except NotImplementedError as exc:
1027
+ return Verdict(UNKNOWN, self.name, logic="modal",
1028
+ reason="unsupported", detail=str(exc))
1029
+ if proved:
1030
+ return Verdict(PROVED, self.name, logic="modal", wall_time=elapsed)
1031
+ return Verdict(UNKNOWN, self.name, logic="modal", reason="incomplete",
1032
+ wall_time=elapsed,
1033
+ detail="Z3 did not close the embedded goal (FO modal logic is undecidable)")
1034
+
1035
+
1036
+ # ---------------------------------------------------------------------------
1037
+ # External backends
1038
+ # ---------------------------------------------------------------------------
1039
+
1040
+ class IsabelleBackend(ProverBackend):
1041
+ """Isabelle/HOL via the kit's runner — the most trusted external route.
1042
+
1043
+ Deliberately NOT in any default chain: one decision spawns a real
1044
+ ``isabelle build`` (minutes of wall time), so it runs only when requested
1045
+ by name. ``ModalVerdict``/``FolVerdict`` map 1:1 onto :class:`Verdict`
1046
+ (their ``infra_error`` becomes reason="infra" detail on UNKNOWN).
1047
+
1048
+ The classical route decides with ``native_equality=True``: ``=`` is HOL
1049
+ identity, as in every other backend's semantics. With the embedding's
1050
+ uninterpreted ``feq`` a nitpick countermodel is one for FOL *without*
1051
+ identity, so ``∀x (x = x)`` came back REFUTED. Passing
1052
+ ``native_equality=False`` is refused for that reason. The modal route makes
1053
+ the same choice by a different road: since 0.30.0 its embedding reads ``=``
1054
+ as RIGID identity (HOL's own, no world argument) and the Kripke evaluator it
1055
+ falls back on for a countermodel refuses an identity atom by name rather
1056
+ than reading it as an ordinary atom, so neither half of that route can
1057
+ answer a question about identity without interpreting it.
1058
+ """
1059
+
1060
+ name = "isabelle"
1061
+ logics = frozenset({"fol", "modal"})
1062
+ external = True
1063
+
1064
+ def available(self) -> bool:
1065
+ from ..hol.isabelle_runner import isabelle_available
1066
+ return isabelle_available()
1067
+
1068
+ def available_for(self, options: dict) -> bool:
1069
+ """Whether the installation :meth:`decide` will run is there: the one a call
1070
+ names with ``install=`` (an :class:`~unicode_logic_kit.hol.isabelle_runner.IsabelleInstall`,
1071
+ which the runner uses instead of looking), else :meth:`available`."""
1072
+ if options.get("install") is not None:
1073
+ return True
1074
+ return self.available()
1075
+
1076
+ def decide(self, formula: Node, premises: Sequence[Node] = (),
1077
+ timeout: int = 10000, **options) -> Verdict:
1078
+ goal = _implication(formula, premises)
1079
+ logic = options.pop("logic", None) or (
1080
+ "modal" if _looks_modal(goal) else "fol")
1081
+
1082
+ if logic == "modal":
1083
+ from ..hol.isabelle_runner import isabelle_decide_modal
1084
+ verdict, elapsed = _timed(lambda: isabelle_decide_modal(goal, **options))
1085
+ else:
1086
+ from ..hol.isabelle_runner import isabelle_decide_fol
1087
+ if not options.pop("native_equality", True):
1088
+ raise ValueError(
1089
+ "isabelle backend: native_equality=False would decide FOL without "
1090
+ "identity (uninterpreted feq), so a REFUTED verdict could contradict "
1091
+ "every other backend; call hol.isabelle_runner.isabelle_decide_fol "
1092
+ "directly for that reading.")
1093
+ verdict, elapsed = _timed(
1094
+ lambda: isabelle_decide_fol(goal, native_equality=True, **options))
1095
+
1096
+ if verdict.status == "valid":
1097
+ return Verdict(PROVED, self.name, logic=logic, wall_time=elapsed,
1098
+ detail=verdict.method)
1099
+ if verdict.status == "invalid":
1100
+ witness = ({"kind": "nitpick" if logic == "fol" else "kripke",
1101
+ "repr": verdict.countermodel}
1102
+ if verdict.countermodel else None)
1103
+ return Verdict(REFUTED, self.name, logic=logic, wall_time=elapsed,
1104
+ countermodel=witness)
1105
+ reason = "infra" if verdict.infra_error else "incomplete"
1106
+ return Verdict(UNKNOWN, self.name, logic=logic, reason=reason,
1107
+ wall_time=elapsed, detail=verdict.infra_error)
1108
+
1109
+
1110
+ class Prover9Backend(ProverBackend):
1111
+ """Prover9 via subprocess. Discovery: $UFK_PROVER9, then PATH.
1112
+
1113
+ Set ``UFK_PROVER9_WSL=1`` when the binary is a Linux build reached through
1114
+ WSL on a Windows host (``$UFK_PROVER9`` then names the path INSIDE WSL, for
1115
+ example ``/mnt/d/prover9/Prover9-LADR-2026-8A/bin/prover9``); the ``use_wsl``
1116
+ option does the same for one call, as for the Vampire backend.
1117
+
1118
+ A run without ``THEOREM PROVED`` is UNKNOWN / ``"incomplete"`` (Prover9's
1119
+ exit does not certify invalidity), EXCEPT a problem Prover9 refused to read
1120
+ (its fatal-error exit) or a binary that could not be started (a path that
1121
+ does not exist inside WSL): that is ERROR / ``"infra"`` with Prover9's own
1122
+ (or the shell's) message in ``detail``, and a run the kit stopped because the
1123
+ ``timeout`` (milliseconds, as for every backend) ran out: that is UNKNOWN /
1124
+ ``"timeout"``.
1125
+
1126
+ A free variable of the problem is one unknown element, the same in every premise and in
1127
+ the conclusion (the problem writer replaces it by a constant of its own: Prover9 itself
1128
+ would close each formula universally, which is another question), so ``P(x) ⊢ P(alpha)``
1129
+ is not proved. A Łukasiewicz connective has no classical reading and is refused:
1130
+ UNKNOWN / ``"unsupported"`` with the name of the connective in ``detail``.
1131
+ """
1132
+
1133
+ name = "prover9"
1134
+ logics = frozenset({"fol"})
1135
+ external = True
1136
+
1137
+ @staticmethod
1138
+ def _binary() -> Optional[str]:
1139
+ import os
1140
+ return os.environ.get("UFK_PROVER9") or shutil.which("prover9")
1141
+
1142
+ @staticmethod
1143
+ def _uses_wsl() -> bool:
1144
+ """Whether ``$UFK_PROVER9_WSL`` says the binary lives inside WSL."""
1145
+ import os
1146
+ return os.environ.get("UFK_PROVER9_WSL") == "1"
1147
+
1148
+ def available(self) -> bool:
1149
+ return self.available_for({})
1150
+
1151
+ def available_for(self, options: dict) -> bool:
1152
+ """Whether the binary the run will use is there: the one :meth:`decide`
1153
+ resolves (``prover9_path=``, else ``$UFK_PROVER9``, else ``prover9`` on
1154
+ PATH), looked for where :meth:`decide` will run it. With the WSL switch on
1155
+ (``use_wsl=True`` in ``options``, else ``$UFK_PROVER9_WSL=1``) that is
1156
+ inside WSL (see :func:`_wsl_has_binary`), not on this host's PATH. A native
1157
+ ``prover9_path=`` that names no file this host can start, and no command on
1158
+ PATH, is refused here by name instead of failing inside the run; what
1159
+ ``$UFK_PROVER9`` names is taken as the installation the user pointed at."""
1160
+ explicit = options.get("prover9_path")
1161
+ path = explicit or self._binary()
1162
+ if path is None:
1163
+ return False
1164
+ if options.get("use_wsl", self._uses_wsl()):
1165
+ return _wsl_has_binary(path)
1166
+ if explicit:
1167
+ return _native_command_exists(explicit)
1168
+ return True
1169
+
1170
+ #: The banner line Prover9 prints first: ``Prover9 (64) version 2026-8A, August 2026.``
1171
+ _BANNER_LINE = re.compile(r"^[ \t]*(Prover9\b[^\r\n]*\bversion\b[^\r\n]*?)[ \t]*$", re.MULTILINE)
1172
+
1173
+ @classmethod
1174
+ def _banner(cls, path: str, use_wsl: bool) -> Optional[str]:
1175
+ """The banner line of the Prover9 at ``path`` (through ``wsl.exe`` when
1176
+ ``use_wsl``), or ``None``.
1177
+
1178
+ Prover9 has no ``--version``: it takes the flag for a resume directory and
1179
+ ends with a fatal error (exit 1), so :func:`_binary_version` reads nothing
1180
+ from it. ``-h`` prints the banner first and exits 0 at once; the line that
1181
+ names the version is the one that starts with ``Prover9`` and holds the word
1182
+ ``version``. The tool never reads this process's standard input (``stdin`` is
1183
+ the null device) and the probe has a time limit, so it cannot block. The
1184
+ result, a miss too, is memoized per ``(path, use_wsl)`` in the process-local
1185
+ version cache that :func:`_binary_version` fills for the other tools, never
1186
+ probed twice; nothing here raises.
1187
+ """
1188
+ key = (path, use_wsl)
1189
+ if key in _VERSION_CACHE:
1190
+ return _VERSION_CACHE[key]
1191
+ import subprocess
1192
+
1193
+ banner: Optional[str] = None
1194
+ try:
1195
+ cmd = ["wsl.exe", path, "-h"] if use_wsl else [path, "-h"]
1196
+ proc = subprocess.run(cmd, capture_output=True, text=True, timeout=10,
1197
+ stdin=subprocess.DEVNULL)
1198
+ for stream in (proc.stdout, proc.stderr):
1199
+ found = cls._BANNER_LINE.search(stream or "")
1200
+ if found:
1201
+ banner = found.group(1)
1202
+ break
1203
+ except (OSError, subprocess.TimeoutExpired, UnicodeError):
1204
+ banner = None
1205
+ _VERSION_CACHE[key] = banner
1206
+ return banner
1207
+
1208
+ def solver_version(self) -> Optional[str]:
1209
+ """Prover9's banner line (``Prover9 (64) version 2026-8A, August 2026.``), for
1210
+ whichever binary discovery (``$UFK_PROVER9``/PATH, ``$UFK_PROVER9_WSL``)
1211
+ resolves to right now — the same default :meth:`available` uses. Read by
1212
+ :meth:`_banner` (``prover9 -h``: the binary has no ``--version``), memoized
1213
+ per ``(binary, use_wsl)`` for the life of the process, never reading this
1214
+ process's standard input and never blocking. ``None`` when no binary is
1215
+ currently discoverable or the banner cannot be read; a per-call
1216
+ ``prover9_path=``/``use_wsl=`` override is reflected in THAT call's own
1217
+ ``Verdict.solver_version`` (see :meth:`decide`), not here —
1218
+ see :func:`_binary_version`'s module-level docstring section for why
1219
+ this method answers for the default binary only.
1220
+ """
1221
+ path = self._binary()
1222
+ if path is None:
1223
+ return None
1224
+ return self._banner(path, self._uses_wsl())
1225
+
1226
+ def decide(self, formula: Node, premises: Sequence[Node] = (),
1227
+ timeout: int = 10000, **options) -> Verdict:
1228
+ from .prover9_entailment import (
1229
+ Prover9Rejected, Prover9TimedOut, check_logical_entailment,
1230
+ )
1231
+
1232
+ path = options.pop("prover9_path", None) or self._binary()
1233
+ if path is None:
1234
+ raise BackendUnavailable(
1235
+ "prover9: no binary found (set $UFK_PROVER9 or put 'prover9' on PATH)")
1236
+ use_wsl = options.pop("use_wsl", self._uses_wsl())
1237
+ solver_version = self._banner(path, use_wsl)
1238
+ start = time.perf_counter()
1239
+ try:
1240
+ proved, elapsed = _timed(
1241
+ lambda: check_logical_entailment(list(premises), formula, path,
1242
+ raise_on_rejection=True,
1243
+ timeout=max(1, timeout // 1000),
1244
+ raise_on_timeout=True,
1245
+ use_wsl=use_wsl))
1246
+ except NotImplementedError as exc:
1247
+ return Verdict(UNKNOWN, self.name, reason="unsupported",
1248
+ solver_version=solver_version, detail=str(exc))
1249
+ except Prover9TimedOut as exc:
1250
+ # The kit's wall-clock budget ran out and the kit stopped Prover9: a
1251
+ # timeout, said as one -- not a search that "found no proof".
1252
+ return Verdict(UNKNOWN, self.name, reason="timeout",
1253
+ wall_time=time.perf_counter() - start,
1254
+ solver_version=solver_version, detail=str(exc))
1255
+ except Prover9Rejected as exc:
1256
+ # Prover9 refused to read the problem (its documented fatal exit):
1257
+ # an error carrying Prover9's own message, not "found no proof".
1258
+ return Verdict(ERROR, self.name, reason="infra",
1259
+ solver_version=solver_version,
1260
+ detail=_rejection_detail(
1261
+ "prover9", exc.output,
1262
+ f"exit code {exc.returncode} and no THEOREM PROVED line"))
1263
+ except OSError as exc:
1264
+ return Verdict(ERROR, self.name, reason="infra",
1265
+ solver_version=solver_version, detail=str(exc))
1266
+ if proved:
1267
+ return Verdict(PROVED, self.name, wall_time=elapsed,
1268
+ solver_version=solver_version)
1269
+ return Verdict(UNKNOWN, self.name, reason="incomplete", wall_time=elapsed,
1270
+ solver_version=solver_version,
1271
+ detail="Prover9 found no proof (its exit does not certify invalidity)")
1272
+
1273
+
1274
+ class VampireBackend(ProverBackend):
1275
+ """Vampire via subprocess. Discovery: $UFK_VAMPIRE, then PATH.
1276
+
1277
+ Set ``UFK_VAMPIRE_WSL=1`` when the binary is a Linux build reached
1278
+ through WSL on a Windows host.
1279
+
1280
+ Decides through the SZS/TSTP route
1281
+ (:func:`~unicode_logic_kit.atp.vampire_entailment.check_entailment_vampire_detailed`):
1282
+ the verdict's ``szs_status`` is Vampire's own status line verbatim, a
1283
+ ``CounterSatisfiable`` answer becomes an honest REFUTED (Vampire's
1284
+ saturation certifies invalidity, though it yields no model structure —
1285
+ ``countermodel`` stays ``None``), and a PROVED verdict carries the parsed
1286
+ TSTP derivation DAG in ``proof``. A problem Vampire REFUSES to read (its
1287
+ ``User error: ...`` or parse error, with no SZS status line) is ERROR /
1288
+ ``"infra"`` with Vampire's own message in ``detail`` — never an UNKNOWN
1289
+ that reads like a timeout; its own ``GaveUp`` / ``ResourceOut`` /
1290
+ ``Unknown`` and a real subprocess timeout keep their UNKNOWN verdicts. So
1291
+ does a run that printed no SZS line but DID print how its search ended
1292
+ (``Termination reason: Refutation not found, incomplete strategy`` —
1293
+ Vampire's way of giving up on a non-theorem of a typed-arithmetic problem):
1294
+ UNKNOWN / ``"incomplete"``, with the termination reason in ``detail``.
1295
+
1296
+ Options (through ``**options``, like the E / Zipperposition backend, and
1297
+ read the same way): ``vampire_path`` / ``use_wsl`` pick the binary, and
1298
+ ``tff`` / ``sort`` pick the TPTP dialect it is given. ``sort`` (``'real'``
1299
+ or ``'int'``; ``None`` by default) opts into the single-numeric-sort typed
1300
+ arithmetic route UNCONDITIONALLY when given, taking priority over ``tff``;
1301
+ with ``sort=None``, ``tff=None`` tries the many-sorted typed route whenever
1302
+ the problem uses a sort, and ``True`` / ``False`` force the typed / the
1303
+ ``fof`` route. The typed writer refuses a problem whose typed text would not
1304
+ ask the kit's question (see :class:`~unicode_logic_kit.atp.tptp_tff.Tf0Refusal`):
1305
+ with ``tff=None`` the problem is then written as ``fof`` and the verdict's
1306
+ ``detail`` says that it was, and why; with ``tff=True`` the refusal is the
1307
+ verdict, ``UNKNOWN`` / ``"unsupported"`` with the writer's message in
1308
+ ``detail`` (never an exception). The options reach
1309
+ :func:`~unicode_logic_kit.atp.vampire_entailment.check_entailment_vampire_detailed`
1310
+ unchanged.
1311
+
1312
+ **Which premises a proof used.** Vampire prints ``unknown`` for the name of an
1313
+ axiom unless it is asked for the names, so the backend asks (``axiom_names=``,
1314
+ on unless a call turns it off) and names the premises' ``axiom`` lines
1315
+ (``premise_names=``, ``premise_<i>`` when not given). A PROVED verdict then
1316
+ carries the caller's premise indices that the proof's axiom leaves are in
1317
+ ``relevant_premises`` (the sorted 0-based indices into the caller's ``premises``,
1318
+ ``()`` for a proof that needs none), and its ``detail`` names the background facts
1319
+ of a sorted reading the proof used (the non-emptiness of a sort, the membership
1320
+ of a sorted constant), which are not premises. ``relevant_premises`` stays
1321
+ ``None`` for a verdict that is not PROVED, and for a proof whose axiom leaves
1322
+ cannot be read back through the problem's own names (never a guess). The set is
1323
+ sound but not necessarily minimal, like every report of this kind.
1324
+ """
1325
+
1326
+ name = "vampire"
1327
+ logics = frozenset({"fol"})
1328
+ external = True
1329
+
1330
+ @staticmethod
1331
+ def _binary() -> Optional[str]:
1332
+ import os
1333
+ return os.environ.get("UFK_VAMPIRE") or shutil.which("vampire")
1334
+
1335
+ def available(self) -> bool:
1336
+ return self.available_for({})
1337
+
1338
+ def available_for(self, options: dict) -> bool:
1339
+ """Whether the binary the run will use is there: the one :meth:`decide`
1340
+ resolves (``vampire_path=``, else ``$UFK_VAMPIRE``, else ``vampire`` on
1341
+ PATH), looked for where :meth:`decide` will run it. With the WSL switch on
1342
+ (``use_wsl=True`` in ``options``, else ``$UFK_VAMPIRE_WSL=1``) that is
1343
+ inside WSL (see :func:`_wsl_has_binary`), not on this host's PATH. A native
1344
+ ``vampire_path=`` that names no file this host can start, and no command on
1345
+ PATH, is refused here by name instead of failing inside the run; what
1346
+ ``$UFK_VAMPIRE`` names is taken as the installation the user pointed at."""
1347
+ import os
1348
+
1349
+ explicit = options.get("vampire_path")
1350
+ path = explicit or self._binary()
1351
+ if path is None:
1352
+ return False
1353
+ if options.get("use_wsl", os.environ.get("UFK_VAMPIRE_WSL") == "1"):
1354
+ return _wsl_has_binary(path)
1355
+ if explicit:
1356
+ return _native_command_exists(explicit)
1357
+ return True
1358
+
1359
+ def solver_version(self) -> Optional[str]:
1360
+ """Vampire's ``--version`` banner (first line), for whichever binary
1361
+ discovery (``$UFK_VAMPIRE``/PATH, ``$UFK_VAMPIRE_WSL``) resolves to
1362
+ right now — the same default :meth:`available` uses. Memoized per
1363
+ ``(binary, use_wsl)`` for the life of the process via
1364
+ :func:`_binary_version`. ``None`` when no binary is currently
1365
+ discoverable; a per-call ``vampire_path=``/``use_wsl=`` override is
1366
+ reflected in THAT call's own ``Verdict.solver_version`` (see
1367
+ :meth:`decide`), not here — see :func:`_binary_version`'s
1368
+ module-level docstring section for why this method answers for the
1369
+ default binary only.
1370
+ """
1371
+ import os
1372
+
1373
+ path = self._binary()
1374
+ if path is None:
1375
+ return None
1376
+ use_wsl = os.environ.get("UFK_VAMPIRE_WSL") == "1"
1377
+ return _binary_version(path, use_wsl)
1378
+
1379
+ def decide(self, formula: Node, premises: Sequence[Node] = (),
1380
+ timeout: int = 10000, **options) -> Verdict:
1381
+ import os
1382
+ from .vampire_entailment import _ended_by_itself, check_entailment_vampire_detailed
1383
+ from .tstp import background_use_note
1384
+
1385
+ path = options.pop("vampire_path", None) or self._binary()
1386
+ if path is None:
1387
+ raise BackendUnavailable(
1388
+ "vampire: no binary found (set $UFK_VAMPIRE or put 'vampire' on PATH)")
1389
+ use_wsl = options.pop("use_wsl", os.environ.get("UFK_VAMPIRE_WSL") == "1")
1390
+ tff = options.pop("tff", None)
1391
+ if tff not in (None, True, False):
1392
+ raise ValueError(
1393
+ f"vampire: tff= is None (decide from the formula), True or False, "
1394
+ f"got {tff!r}.")
1395
+ sort = options.pop("sort", None)
1396
+ premise_names = options.pop("premise_names", None)
1397
+ axiom_names = options.pop("axiom_names", True)
1398
+ solver_version = _binary_version(path, use_wsl)
1399
+ try:
1400
+ result, elapsed = _timed(lambda: check_entailment_vampire_detailed(
1401
+ list(premises), formula, path,
1402
+ timeout=max(1, timeout // 1000), use_wsl=use_wsl,
1403
+ tff=tff, sort=sort, premise_names=premise_names,
1404
+ axiom_names=axiom_names))
1405
+ except NotImplementedError as exc:
1406
+ return Verdict(UNKNOWN, self.name, reason="unsupported",
1407
+ solver_version=solver_version, detail=str(exc))
1408
+ except OSError as exc:
1409
+ return Verdict(ERROR, self.name, reason="infra",
1410
+ solver_version=solver_version, detail=str(exc))
1411
+ status, reason, szs = result["status"], result["reason"], result["szs_status"]
1412
+ if status == ERROR:
1413
+ # Vampire read the problem and refused it (a type or syntax error:
1414
+ # "User error: ..."), or failed outright. Its own message goes in
1415
+ # the detail -- a refusal must not read like running out of time.
1416
+ evidence = ("no SZS status line in its output" if szs is None
1417
+ else f"SZS status {szs}")
1418
+ detail = _rejection_detail("vampire", result.get("output_excerpt", ""),
1419
+ evidence)
1420
+ else:
1421
+ detail = (f"SZS status {szs}" if szs is not None
1422
+ else "no SZS status line in Vampire's output")
1423
+ if szs is None and status == UNKNOWN:
1424
+ # No SZS line: say what DID happen, not what is missing. Either it
1425
+ # read the problem and said how its search ended, or the kit's own
1426
+ # budget ran out and the kit stopped it before it printed a verdict.
1427
+ ended = _ended_by_itself(result.get("output_excerpt", ""))
1428
+ if ended is not None:
1429
+ detail = (f"Vampire's search ended on its own (Termination reason: "
1430
+ f"{ended[1]})")
1431
+ elif reason == "timeout":
1432
+ detail = (f"the budget of this call ({max(1, timeout // 1000)} s) ran "
1433
+ "out and the kit stopped Vampire before it printed a verdict")
1434
+ relevant = None
1435
+ if status == PROVED:
1436
+ # The caller's premises the proof's axiom leaves are, and the background
1437
+ # facts of a sorted reading it used (not premises), as the proof printed them.
1438
+ if result.get("relevant_premises") is not None:
1439
+ relevant = tuple(result["relevant_premises"])
1440
+ detail += background_use_note(tuple(result.get("background_used") or ()))
1441
+ if result.get("tff_fallback"):
1442
+ # tff=None tried the typed writer for a sorted problem and it refused:
1443
+ # the answer is that of the fof text, and the caller is told so and why.
1444
+ detail = f"{detail}; {result['tff_fallback']}"
1445
+ return Verdict(status, self.name, reason=reason, szs_status=szs,
1446
+ wall_time=elapsed, solver_version=solver_version,
1447
+ proof=result["derivation"] if status == PROVED else None,
1448
+ detail=detail, relevant_premises=relevant)
1449
+
1450
+
1451
+ # ---------------------------------------------------------------------------
1452
+ # Registry
1453
+ # ---------------------------------------------------------------------------
1454
+
1455
+ def _looks_modal(node: Node) -> bool:
1456
+ """Route helper: does the formula contain any modal-family operator?"""
1457
+ from .modal_tableau import has_modal
1458
+ return has_modal(node)
1459
+
1460
+
1461
+ _REGISTRY: Dict[str, ProverBackend] = {}
1462
+
1463
+
1464
+ def register_backend(backend: ProverBackend) -> None:
1465
+ """Register (or replace) a backend under ``backend.name``.
1466
+
1467
+ Third-party code can plug in its own :class:`ProverBackend` here and it
1468
+ becomes addressable by every ``prove(backends=[...])`` call.
1469
+ """
1470
+ if not backend.name:
1471
+ raise ValueError("register_backend: backend must set a non-empty name")
1472
+ _REGISTRY[backend.name] = backend
1473
+
1474
+
1475
+ _b: ProverBackend # one name for both registration loops: each backend is a ProverBackend
1476
+ for _b in (Z3Backend(), TableauBackend(), ResolutionBackend(),
1477
+ ModelFinderBackend(), ModalTableauBackend(), QmlBackend(),
1478
+ IsabelleBackend(), Prover9Backend(), VampireBackend()):
1479
+ register_backend(_b)
1480
+
1481
+ # The remaining stock backends live in their own modules (an optional
1482
+ # dependency, a heavier search, an external prover family) and import THIS
1483
+ # module's public contract — so they are pulled in here, after that contract
1484
+ # is fully defined. A deliberate bottom-of-registration import, not an
1485
+ # accident: keeping the whole stock registry in one place guarantees that
1486
+ # `default_chain` and `_REGISTRY` can never disagree about what exists,
1487
+ # whichever submodule a process imports first.
1488
+ from .clingo_backend import ClingoBackend # noqa: E402
1489
+ from .cvc5_backend import Cvc5Backend # noqa: E402
1490
+ from .eprover_backend import EProverBackend, ZipperpositionBackend # noqa: E402
1491
+ from .hets_backend import HetsBackend # noqa: E402
1492
+ from .kripke_enum import KripkeEnumBackend # noqa: E402
1493
+ from .leo3_backend import Leo3Backend # noqa: E402
1494
+ from .ltl_tableau import LtlTableauBackend # noqa: E402
1495
+ from .minizinc_backend import MinizincBackend # noqa: E402
1496
+ from .nanocop_backend import NanocopBackend # noqa: E402
1497
+ from .twee_backend import TweeBackend # noqa: E402
1498
+ # Five individually-reasoned per-logic adapters (C10) — see logic_backends'
1499
+ # module docstring for why these are NOT a uniform bulk registration.
1500
+ from .logic_backends import ( # noqa: E402
1501
+ IntBackend, LambekBackend, IllBackend, RelevantBackend, HybridBackend,
1502
+ )
1503
+
1504
+ for _b in (ClingoBackend(), Cvc5Backend(), EProverBackend(), HetsBackend(),
1505
+ KripkeEnumBackend(), Leo3Backend(), LtlTableauBackend(), MinizincBackend(),
1506
+ NanocopBackend(), TweeBackend(), ZipperpositionBackend(),
1507
+ IntBackend(), LambekBackend(), IllBackend(), RelevantBackend(),
1508
+ HybridBackend()):
1509
+ register_backend(_b)
1510
+
1511
+
1512
+ def get_backend(name: str) -> ProverBackend:
1513
+ """Return the backend registered under ``name`` (ValueError if unknown)."""
1514
+ if name not in _REGISTRY:
1515
+ raise ValueError(
1516
+ f"get_backend: unknown backend {name!r} (registered: {sorted(_REGISTRY)})")
1517
+ return _REGISTRY[name]
1518
+
1519
+
1520
+ def available_backends(logic: Optional[str] = None) -> Tuple[str, ...]:
1521
+ """Names of the currently-available backends, optionally per logic."""
1522
+ names = []
1523
+ for name, backend in _REGISTRY.items():
1524
+ if logic is not None and logic not in backend.logics:
1525
+ continue
1526
+ if backend.available():
1527
+ names.append(name)
1528
+ return tuple(sorted(names))
1529
+
1530
+
1531
+ # Default chains: fast, deterministic, and NEVER silently expensive — the
1532
+ # external minutes-per-call route (isabelle) must be requested by name.
1533
+ # "kripke-enum" sits between the tableau and qml on purpose: it is the only
1534
+ # default-chain member that can REFUTE a temporal-closure formula (the
1535
+ # tableau reports those unsupported, qml is proof-only), and its bounded
1536
+ # enumeration settles the refutation side before qml's Z3 call can burn its
1537
+ # full timeout failing to prove an invalid goal.
1538
+ #
1539
+ # The five substructural/non-classical entries (C10) are each a SINGLETON
1540
+ # chain: every one of these logics currently has exactly one decision route
1541
+ # in the kit, so there is no actual "portfolio race" to order here (unlike
1542
+ # "fol"/"modal" above) — see logic_backends' module docstring for why these
1543
+ # five are individually-reasoned adapters, not a uniform bulk registration.
1544
+ _DEFAULT_CHAINS = {
1545
+ "fol": ("z3", "tableau", "resolution", "modelfinder"),
1546
+ "modal": ("modal-tableau", "kripke-enum", "qml"),
1547
+ "intuitionistic": ("intuitionistic",),
1548
+ "lambek": ("lambek",),
1549
+ "ill": ("ill",),
1550
+ "relevant": ("relevant",),
1551
+ "hybrid": ("hybrid",),
1552
+ }
1553
+
1554
+
1555
+ def default_chain(logic: str) -> Tuple[str, ...]:
1556
+ """The default backend order for ``logic`` (ValueError if unknown).
1557
+
1558
+ The chains are static, with ONE documented availability-dependent member:
1559
+ when the optional ``cvc5`` extra is installed
1560
+ (``pip install unicode-logic-kit[cvc5]``), the FOL chain gains ``"cvc5"``
1561
+ directly after ``"z3"`` — a second, fully independent SMT decision
1562
+ procedure whose quantifier instantiation (E-matching/enumerative/CEGQI)
1563
+ often closes goals Z3's MBQI leaves UNKNOWN, and vice versa. It cannot be
1564
+ an unconditional member: a missing backend in the DEFAULT chain would
1565
+ make every plain ``prove()`` call raise :class:`BackendUnavailable` on a
1566
+ machine without the extra. Every verdict names the backend that actually
1567
+ answered (``backend`` / ``agreement``), so provenance stays exact even
1568
+ though the chain adapts to the install.
1569
+ """
1570
+ if logic not in _DEFAULT_CHAINS:
1571
+ raise ValueError(
1572
+ f"default_chain: unknown logic {logic!r} (use one of {sorted(_DEFAULT_CHAINS)})")
1573
+ chain = _DEFAULT_CHAINS[logic]
1574
+ if logic == "fol" and _REGISTRY["cvc5"].available():
1575
+ chain = chain[:1] + ("cvc5",) + chain[1:]
1576
+ return chain
1577
+
1578
+
1579
+ # ---------------------------------------------------------------------------
1580
+ # Options: which backend of a chain reads which keyword
1581
+ #
1582
+ # ``api.prove(f, premises, frame="S4")`` hands ``frame`` to every backend of the
1583
+ # chain, and what a backend does with a keyword it does not know used to differ
1584
+ # from one to the next: Z3, E and Zipperposition ignored it, the tableau, the
1585
+ # resolution prover and the model finder failed with a ``TypeError`` that surfaced
1586
+ # as an ERROR verdict. An ignored option answers a different question than the
1587
+ # caller asked. Every stock backend therefore DECLARES the names it reads (below),
1588
+ # and the dispatcher checks a call against the declarations before anything runs.
1589
+ # ---------------------------------------------------------------------------
1590
+
1591
+ from typing import Callable, FrozenSet # noqa: E402
1592
+
1593
+
1594
+ def _keywords_of(function, skip: Sequence[str] = ()) -> FrozenSet[str]:
1595
+ """The names ``function`` takes by keyword, without ``skip``.
1596
+
1597
+ A backend that forwards ``**options`` to a function with its own keyword list
1598
+ reads exactly that list; deriving the set from the signature keeps the
1599
+ declaration from drifting away from the function it describes.
1600
+ """
1601
+ import inspect
1602
+
1603
+ return frozenset(
1604
+ name for name, parameter in inspect.signature(function).parameters.items()
1605
+ if parameter.kind in (parameter.POSITIONAL_OR_KEYWORD, parameter.KEYWORD_ONLY)
1606
+ and name not in skip)
1607
+
1608
+
1609
+ def _keywords(*targets: str, skip: Sequence[str] = ()) -> Callable[[Optional[str]], FrozenSet[str]]:
1610
+ """A lazy reader of the keyword names of the functions at ``"module:name"``.
1611
+
1612
+ Several targets give the names ALL of them take (a backend that forwards the
1613
+ same options to two functions reads what both read). Lazy, because a backend
1614
+ is rarely in the chain and its module is heavy.
1615
+ """
1616
+ def read(logic: Optional[str] = None) -> FrozenSet[str]:
1617
+ import importlib
1618
+
1619
+ names = None
1620
+ for target in targets:
1621
+ module, _, function = target.partition(":")
1622
+ found = _keywords_of(getattr(importlib.import_module(module), function), skip)
1623
+ names = found if names is None else names & found
1624
+ return names if names is not None else frozenset()
1625
+ return read
1626
+
1627
+
1628
+ def _names(*names: str) -> Callable[[Optional[str]], FrozenSet[str]]:
1629
+ """A reader of a fixed set of option names."""
1630
+ fixed = frozenset(names)
1631
+
1632
+ def read(logic: Optional[str] = None) -> FrozenSet[str]:
1633
+ return fixed
1634
+ return read
1635
+
1636
+
1637
+ def _isabelle_options(logic: Optional[str] = None) -> FrozenSet[str]:
1638
+ """What :class:`IsabelleBackend` reads: the keywords of the runner for the
1639
+ logic of the call (both when the logic is not known), plus ``native_equality``,
1640
+ which the backend checks itself."""
1641
+ from ..hol.isabelle_runner import isabelle_decide_fol, isabelle_decide_modal
1642
+
1643
+ modal = _keywords_of(isabelle_decide_modal, ("formula",))
1644
+ classical = _keywords_of(isabelle_decide_fol, ("formula",)) | {"native_equality"}
1645
+ if logic == "modal":
1646
+ return modal
1647
+ if logic == "fol":
1648
+ return classical
1649
+ return modal | classical
1650
+
1651
+
1652
+ #: The options each stock backend reads, as lazy readers keyed by the backend's
1653
+ #: class (an exact class: a subclass has to declare its own, through
1654
+ #: :meth:`ProverBackend.accepted_options`). Every name here is read in the
1655
+ #: backend's ``decide`` or by the function it forwards ``**options`` to; the test
1656
+ #: suite compares the two.
1657
+ _STOCK_OPTIONS: Dict[type, Callable[[Optional[str]], FrozenSet[str]]] = {
1658
+ Z3Backend: _names(),
1659
+ TableauBackend: _keywords("unicode_logic_kit.atp.tableau:prove_tableau_detailed",
1660
+ skip=("premises", "conclusion", "timeout")),
1661
+ ResolutionBackend: _keywords("unicode_logic_kit.atp.resolution:prove",
1662
+ skip=("premises", "conclusion", "timeout")),
1663
+ ModelFinderBackend: _keywords("unicode_logic_kit.semantics.modelfinder:find_countermodel",
1664
+ skip=("premises", "conclusion", "timeout")),
1665
+ ModalTableauBackend: _keywords("unicode_logic_kit.atp.modal_tableau:modal_decide",
1666
+ "unicode_logic_kit.atp.modal_tableau:modal_countermodel",
1667
+ skip=("formula", "timeout")),
1668
+ QmlBackend: _keywords("unicode_logic_kit.fol.qml:qml_is_valid", skip=("formula", "timeout")),
1669
+ IsabelleBackend: _isabelle_options,
1670
+ Prover9Backend: _names("prover9_path", "use_wsl"),
1671
+ VampireBackend: _keywords("unicode_logic_kit.atp.vampire_entailment:check_entailment_vampire_detailed",
1672
+ skip=("premises", "conclusion", "timeout")),
1673
+ EProverBackend: _names("tff", "sort", "premise_names"),
1674
+ ZipperpositionBackend: _names("tff", "sort", "premise_names"),
1675
+ Cvc5Backend: _names("logic", "random_seed", "proof"),
1676
+ ClingoBackend: _names("max_size", "all_different", "verify"),
1677
+ MinizincBackend: _names("minizinc_path", "max_size", "solver", "all_different"),
1678
+ HetsBackend: _names("reasoner", "translation", "url"),
1679
+ TweeBackend: _names("use_wsl", "twee_cmd"),
1680
+ NanocopBackend: _names("logic", "domain"),
1681
+ Leo3Backend: _names("frame", "domains"),
1682
+ KripkeEnumBackend: _keywords("unicode_logic_kit.atp.kripke_enum:modal_enum_search",
1683
+ skip=("formula", "timeout")),
1684
+ LtlTableauBackend: _names("mode", "max_atoms"),
1685
+ IntBackend: _keywords("unicode_logic_kit.semantics.intuitionistic:int_countermodel",
1686
+ skip=("formula",)),
1687
+ LambekBackend: _names(),
1688
+ IllBackend: _keywords("unicode_logic_kit.atp.linear:ill_derivable",
1689
+ skip=("antecedents", "goal")),
1690
+ RelevantBackend: _names("max_worlds"),
1691
+ HybridBackend: _names("frame", "systems", "temporal_closure"),
1692
+ }
1693
+
1694
+ #: Options that only bound how far a search goes, say how a solver is driven and
1695
+ #: where it lives, or ask for more to be reported about the same answer (the names
1696
+ #: of the premises, a proof text). A backend that does not read one still answers the same
1697
+ #: question without it, so it runs when another backend of the chain reads the
1698
+ #: option. Every other option changes WHICH question is asked (a modal frame, a
1699
+ #: domain regime, a bridge, the reading of numerals, a subsort edge): a backend
1700
+ #: that does not read one would answer a different question, and is not run.
1701
+ _NEUTRAL_OPTIONS = frozenset({
1702
+ "max_steps", "max_terms", "max_worlds", "max_atoms", "max_models", "max_depth",
1703
+ "max_size", "max_candidates", "symmetry_breaking", "domain_elements",
1704
+ "methods", "refute", "card", "prove_timeout", "refute_timeout",
1705
+ "random_seed", "solver", "verify", "install",
1706
+ "use_wsl", "vampire_path", "prover9_path", "twee_cmd", "minizinc_path",
1707
+ "url", "reasoner", "translation", "tff",
1708
+ "premise_names", "axiom_names", "proof",
1709
+ })
1710
+
1711
+
1712
+ def declared_options(backend: ProverBackend, logic: Optional[str] = None) -> Optional[FrozenSet[str]]:
1713
+ """The names of the options ``backend`` reads, or ``None`` when it declares none.
1714
+
1715
+ The backend's own :meth:`~ProverBackend.accepted_options` first, then the
1716
+ declaration of the stock class of exactly that type. ``None`` means that
1717
+ nothing is known about the backend and it is handed every option.
1718
+ """
1719
+ own = getattr(backend, "accepted_options", lambda logic=None: None)(logic)
1720
+ if own is not None:
1721
+ return frozenset(own)
1722
+ reader = _STOCK_OPTIONS.get(type(backend))
1723
+ return reader(logic) if reader is not None else None
1724
+
1725
+
1726
+ def plan_options(caller: str, chain: Sequence[str], logic: str,
1727
+ options: dict) -> Dict[str, Tuple[dict, Optional[str]]]:
1728
+ """Decide which of ``options`` each backend of ``chain`` is handed.
1729
+
1730
+ Returns, per backend name, ``(options to pass, refusal)``. ``refusal`` is
1731
+ ``None`` when the backend runs, and otherwise the text of the reason it must
1732
+ not: it does not read an option that changes the question (see
1733
+ :data:`_NEUTRAL_OPTIONS`) which another backend of the chain does read, so any
1734
+ answer of it would be about a different question.
1735
+
1736
+ Raises:
1737
+ ValueError: when an option is read by NO backend of the chain, naming the
1738
+ option, the chain and what its backends do read, before anything runs.
1739
+ A backend that declares no options counts as reading everything.
1740
+ """
1741
+ if not options:
1742
+ return {name: ({}, None) for name in chain}
1743
+ declared = {name: declared_options(get_backend(name), logic) for name in chain}
1744
+ for option in options:
1745
+ if not any(names is None or option in names for names in declared.values()):
1746
+ reads = "; ".join(
1747
+ f"{name}: " + (", ".join(sorted(names)) if names else "no options")
1748
+ for name, names in declared.items())
1749
+ raise ValueError(
1750
+ f"{caller}: option {option!r} is read by no backend of the chain "
1751
+ f"{list(chain)} -- nothing would use it, and the answer would be "
1752
+ f"given as if it had not been passed. The options each backend reads: "
1753
+ f"{reads}.")
1754
+ plan: Dict[str, Tuple[dict, Optional[str]]] = {}
1755
+ for name in chain:
1756
+ names = declared[name]
1757
+ if names is None:
1758
+ plan[name] = (dict(options), None)
1759
+ continue
1760
+ passed = {key: value for key, value in options.items() if key in names}
1761
+ unread = sorted(key for key in options
1762
+ if key not in names and key not in _NEUTRAL_OPTIONS)
1763
+ if unread:
1764
+ shown = ", ".join(repr(key) for key in unread)
1765
+ plan[name] = ({}, f"{name} does not read the option {shown}, which "
1766
+ f"changes the question; another backend of the chain "
1767
+ f"reads it, and answering without it would decide a "
1768
+ f"different question, so {name} was not run")
1769
+ else:
1770
+ plan[name] = (passed, None)
1771
+ return plan
1772
+
1773
+
1774
+ def _available_for(backend: ProverBackend, options: dict) -> bool:
1775
+ """:meth:`ProverBackend.available_for` of ``backend``, or its ``available()``
1776
+ for an object that has no such method."""
1777
+ method = getattr(backend, "available_for", None)
1778
+ return bool(method(options)) if method is not None else bool(backend.available())
1779
+
1780
+
1781
+ def run_backend(name: str, formula: Node, premises: Sequence[Node] = (),
1782
+ timeout: int = 10000, **options) -> Verdict:
1783
+ """Run one backend by name, enforcing the availability contract.
1784
+
1785
+ Raises ``ValueError`` for an unknown name and :class:`BackendUnavailable`
1786
+ for a known-but-unavailable one; whether it is available is asked for the
1787
+ route this call's ``options`` select (:meth:`ProverBackend.available_for`).
1788
+ Any unexpected exception inside the
1789
+ backend is converted to an ERROR verdict (reason="infra") so a batch run
1790
+ over thousands of formulas records the failure instead of dying.
1791
+ """
1792
+ backend = get_backend(name)
1793
+ if not _available_for(backend, options):
1794
+ raise BackendUnavailable(
1795
+ f"{name}: backend is not available on this machine "
1796
+ f"(external={backend.external}) — install it or drop it from `backends`.")
1797
+ try:
1798
+ return backend.decide(formula, premises, timeout=timeout, **options)
1799
+ except (ValueError, BackendUnavailable):
1800
+ raise # caller errors stay loud
1801
+ except Exception as exc: # infra failure → recorded, not fatal
1802
+ return Verdict(ERROR, name, reason="infra",
1803
+ detail=f"{type(exc).__name__}: {exc}")