unicode-logic-kit 0.31.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (237) hide show
  1. unicode_logic_kit/__init__.py +385 -0
  2. unicode_logic_kit/__main__.py +520 -0
  3. unicode_logic_kit/_deadline.py +219 -0
  4. unicode_logic_kit/ace/__init__.py +126 -0
  5. unicode_logic_kit/ace/_align.py +135 -0
  6. unicode_logic_kit/ace/chem_lexicon.py +128 -0
  7. unicode_logic_kit/ace/drs_reader.py +570 -0
  8. unicode_logic_kit/ace/mapping.py +666 -0
  9. unicode_logic_kit/ace/reverse_modal.py +138 -0
  10. unicode_logic_kit/ace/runner.py +551 -0
  11. unicode_logic_kit/ace/translate.py +452 -0
  12. unicode_logic_kit/ace/verbalize.py +1070 -0
  13. unicode_logic_kit/api.py +1284 -0
  14. unicode_logic_kit/atp/__init__.py +177 -0
  15. unicode_logic_kit/atp/_ascii_names.py +113 -0
  16. unicode_logic_kit/atp/_html.py +72 -0
  17. unicode_logic_kit/atp/_substructural_input.py +228 -0
  18. unicode_logic_kit/atp/_tff_problem.py +715 -0
  19. unicode_logic_kit/atp/_tptp_problem.py +1111 -0
  20. unicode_logic_kit/atp/_writer_support.py +289 -0
  21. unicode_logic_kit/atp/clingo_backend.py +1180 -0
  22. unicode_logic_kit/atp/cvc5_backend.py +1385 -0
  23. unicode_logic_kit/atp/eprover_backend.py +732 -0
  24. unicode_logic_kit/atp/finite_domain.py +1055 -0
  25. unicode_logic_kit/atp/fitch.py +1547 -0
  26. unicode_logic_kit/atp/fitch_search.py +551 -0
  27. unicode_logic_kit/atp/hets_backend.py +339 -0
  28. unicode_logic_kit/atp/hybrid_down.py +120 -0
  29. unicode_logic_kit/atp/incremental.py +250 -0
  30. unicode_logic_kit/atp/kripke_enum.py +741 -0
  31. unicode_logic_kit/atp/lambek.py +436 -0
  32. unicode_logic_kit/atp/leo3_backend.py +332 -0
  33. unicode_logic_kit/atp/linear.py +738 -0
  34. unicode_logic_kit/atp/lj.py +705 -0
  35. unicode_logic_kit/atp/logic_backends.py +566 -0
  36. unicode_logic_kit/atp/ltl_tableau.py +1084 -0
  37. unicode_logic_kit/atp/minizinc_backend.py +1402 -0
  38. unicode_logic_kit/atp/modal_tableau.py +1382 -0
  39. unicode_logic_kit/atp/nanocop_backend.py +410 -0
  40. unicode_logic_kit/atp/portfolio.py +489 -0
  41. unicode_logic_kit/atp/protocol.py +1803 -0
  42. unicode_logic_kit/atp/prover9_entailment.py +1153 -0
  43. unicode_logic_kit/atp/resolution.py +1376 -0
  44. unicode_logic_kit/atp/resolution_check.py +1114 -0
  45. unicode_logic_kit/atp/sequent.py +1050 -0
  46. unicode_logic_kit/atp/tableau.py +921 -0
  47. unicode_logic_kit/atp/tableau_check.py +543 -0
  48. unicode_logic_kit/atp/tptp_ncl.py +811 -0
  49. unicode_logic_kit/atp/tptp_tff.py +1546 -0
  50. unicode_logic_kit/atp/tstp.py +1333 -0
  51. unicode_logic_kit/atp/tstp_check.py +1096 -0
  52. unicode_logic_kit/atp/twee_backend.py +236 -0
  53. unicode_logic_kit/atp/twee_check.py +711 -0
  54. unicode_logic_kit/atp/twee_entailment.py +953 -0
  55. unicode_logic_kit/atp/vampire_entailment.py +540 -0
  56. unicode_logic_kit/atp/z3_arith.py +470 -0
  57. unicode_logic_kit/atp/z3_equivalence.py +36 -0
  58. unicode_logic_kit/atp/z3_fuzzy.py +362 -0
  59. unicode_logic_kit/atp/z3_input.py +500 -0
  60. unicode_logic_kit/atp/z3_models.py +208 -0
  61. unicode_logic_kit/chem/__init__.py +88 -0
  62. unicode_logic_kit/chem/_naming.py +284 -0
  63. unicode_logic_kit/chem/cache.py +185 -0
  64. unicode_logic_kit/chem/interop.py +244 -0
  65. unicode_logic_kit/chem/mol.py +525 -0
  66. unicode_logic_kit/chem/signature.py +112 -0
  67. unicode_logic_kit/comorphism.py +497 -0
  68. unicode_logic_kit/dl/__init__.py +384 -0
  69. unicode_logic_kit/dl/classification.py +227 -0
  70. unicode_logic_kit/dl/concepts.py +632 -0
  71. unicode_logic_kit/dl/datatypes.py +818 -0
  72. unicode_logic_kit/dl/owl_functional.py +2433 -0
  73. unicode_logic_kit/dl/owl_manchester.py +1637 -0
  74. unicode_logic_kit/dl/owl_reasoner.py +790 -0
  75. unicode_logic_kit/dl/parser.py +391 -0
  76. unicode_logic_kit/dl/tableau.py +4048 -0
  77. unicode_logic_kit/dl/translate.py +2704 -0
  78. unicode_logic_kit/drt/__init__.py +94 -0
  79. unicode_logic_kit/drt/export.py +179 -0
  80. unicode_logic_kit/drt/nodes.py +506 -0
  81. unicode_logic_kit/drt/parser.py +965 -0
  82. unicode_logic_kit/drt/resolve.py +195 -0
  83. unicode_logic_kit/drt/reverse.py +175 -0
  84. unicode_logic_kit/eval/__init__.py +106 -0
  85. unicode_logic_kit/eval/batch.py +382 -0
  86. unicode_logic_kit/eval/canonical.py +663 -0
  87. unicode_logic_kit/eval/chem_batch.py +606 -0
  88. unicode_logic_kit/eval/converses.py +200 -0
  89. unicode_logic_kit/eval/datasets/__init__.py +136 -0
  90. unicode_logic_kit/eval/datasets/_base.py +263 -0
  91. unicode_logic_kit/eval/datasets/_proofwriter_proof.py +422 -0
  92. unicode_logic_kit/eval/datasets/c3po.py +678 -0
  93. unicode_logic_kit/eval/datasets/folio.py +158 -0
  94. unicode_logic_kit/eval/datasets/fracas.py +418 -0
  95. unicode_logic_kit/eval/datasets/groves.py +191 -0
  96. unicode_logic_kit/eval/datasets/logicbench.py +467 -0
  97. unicode_logic_kit/eval/datasets/logicnli.py +303 -0
  98. unicode_logic_kit/eval/datasets/malls.py +133 -0
  99. unicode_logic_kit/eval/datasets/pfolio.py +594 -0
  100. unicode_logic_kit/eval/datasets/pmb.py +242 -0
  101. unicode_logic_kit/eval/datasets/prontoqa.py +611 -0
  102. unicode_logic_kit/eval/datasets/proofwriter.py +1431 -0
  103. unicode_logic_kit/eval/datasets/proverqa.py +674 -0
  104. unicode_logic_kit/eval/datasets/willow.py +478 -0
  105. unicode_logic_kit/eval/equivalence.py +466 -0
  106. unicode_logic_kit/eval/exercise_gen.py +533 -0
  107. unicode_logic_kit/eval/explain.py +791 -0
  108. unicode_logic_kit/eval/generality.py +750 -0
  109. unicode_logic_kit/eval/metric_hf.py +458 -0
  110. unicode_logic_kit/eval/predicate_match.py +343 -0
  111. unicode_logic_kit/eval/theory_check.py +1170 -0
  112. unicode_logic_kit/eval/validate.py +306 -0
  113. unicode_logic_kit/fol/__init__.py +177 -0
  114. unicode_logic_kit/fol/_atom_keys.py +510 -0
  115. unicode_logic_kit/fol/_fol_nodes.py +3586 -0
  116. unicode_logic_kit/fol/_free_parameters.py +105 -0
  117. unicode_logic_kit/fol/_ho_nodes.py +448 -0
  118. unicode_logic_kit/fol/_hybrid_nodes.py +308 -0
  119. unicode_logic_kit/fol/_identifiers.py +1091 -0
  120. unicode_logic_kit/fol/_lambek_nodes.py +112 -0
  121. unicode_logic_kit/fol/_linear_nodes.py +352 -0
  122. unicode_logic_kit/fol/_modal_nodes.py +1467 -0
  123. unicode_logic_kit/fol/_msfl_nodes.py +2196 -0
  124. unicode_logic_kit/fol/_numeral_symbols.py +231 -0
  125. unicode_logic_kit/fol/_so_nodes.py +200 -0
  126. unicode_logic_kit/fol/_symbol_names.py +81 -0
  127. unicode_logic_kit/fol/_team_nodes.py +181 -0
  128. unicode_logic_kit/fol/_tptp_symbols.py +551 -0
  129. unicode_logic_kit/fol/_truth_constants.py +117 -0
  130. unicode_logic_kit/fol/casl_export.py +1135 -0
  131. unicode_logic_kit/fol/casl_import.py +929 -0
  132. unicode_logic_kit/fol/derivation.py +367 -0
  133. unicode_logic_kit/fol/dialect_detect.py +70 -0
  134. unicode_logic_kit/fol/dialect_repair.py +537 -0
  135. unicode_logic_kit/fol/frames.py +637 -0
  136. unicode_logic_kit/fol/grammars/terminals.lark +31 -0
  137. unicode_logic_kit/fol/lambda_tools.py +297 -0
  138. unicode_logic_kit/fol/latex_input.py +429 -0
  139. unicode_logic_kit/fol/modal_translation.py +944 -0
  140. unicode_logic_kit/fol/msflparser.py +1033 -0
  141. unicode_logic_kit/fol/naming.py +422 -0
  142. unicode_logic_kit/fol/nodes.py +241 -0
  143. unicode_logic_kit/fol/normalforms.py +492 -0
  144. unicode_logic_kit/fol/pal.py +287 -0
  145. unicode_logic_kit/fol/prolog_export.py +566 -0
  146. unicode_logic_kit/fol/prolog_input.py +505 -0
  147. unicode_logic_kit/fol/prover9_input.py +1325 -0
  148. unicode_logic_kit/fol/qml.py +1760 -0
  149. unicode_logic_kit/fol/qmltp_input.py +525 -0
  150. unicode_logic_kit/fol/sanitize.py +221 -0
  151. unicode_logic_kit/fol/serialize.py +79 -0
  152. unicode_logic_kit/fol/signature.py +1290 -0
  153. unicode_logic_kit/fol/simplify_check.py +544 -0
  154. unicode_logic_kit/fol/spans.py +594 -0
  155. unicode_logic_kit/fol/tptp_input.py +1503 -0
  156. unicode_logic_kit/fol/tptp_repair.py +941 -0
  157. unicode_logic_kit/fol/unification.py +157 -0
  158. unicode_logic_kit/fol/verbalize.py +263 -0
  159. unicode_logic_kit/hets/__init__.py +163 -0
  160. unicode_logic_kit/hets/bridge.py +142 -0
  161. unicode_logic_kit/hets/client.py +748 -0
  162. unicode_logic_kit/hets/docker.py +420 -0
  163. unicode_logic_kit/hets/dol.py +712 -0
  164. unicode_logic_kit/hets/haskell_json.py +355 -0
  165. unicode_logic_kit/hets/owl_backend.py +794 -0
  166. unicode_logic_kit/hets/owl_cli.py +598 -0
  167. unicode_logic_kit/hets/symbols.py +512 -0
  168. unicode_logic_kit/hol/__init__.py +140 -0
  169. unicode_logic_kit/hol/_ho_common.py +323 -0
  170. unicode_logic_kit/hol/_isabelle_binders.py +125 -0
  171. unicode_logic_kit/hol/classical.py +812 -0
  172. unicode_logic_kit/hol/deepshallow/__init__.py +45 -0
  173. unicode_logic_kit/hol/deepshallow/_common.py +177 -0
  174. unicode_logic_kit/hol/deepshallow/conditional.py +225 -0
  175. unicode_logic_kit/hol/deepshallow/intuitionistic.py +181 -0
  176. unicode_logic_kit/hol/deepshallow/modal.py +217 -0
  177. unicode_logic_kit/hol/deepshallow/qml.py +406 -0
  178. unicode_logic_kit/hol/deepshallow/relevant.py +206 -0
  179. unicode_logic_kit/hol/free.py +753 -0
  180. unicode_logic_kit/hol/goedel.py +336 -0
  181. unicode_logic_kit/hol/ho_modal.py +1743 -0
  182. unicode_logic_kit/hol/intuitionistic.py +403 -0
  183. unicode_logic_kit/hol/isabelle_conditional.py +593 -0
  184. unicode_logic_kit/hol/isabelle_modal.py +1908 -0
  185. unicode_logic_kit/hol/isabelle_relevant.py +412 -0
  186. unicode_logic_kit/hol/isabelle_runner.py +1147 -0
  187. unicode_logic_kit/hol/isabelle_substructural.py +884 -0
  188. unicode_logic_kit/hol/lean.py +1018 -0
  189. unicode_logic_kit/hol/manyvalued.py +921 -0
  190. unicode_logic_kit/hol/secondorder.py +687 -0
  191. unicode_logic_kit/hol/thf_modal.py +941 -0
  192. unicode_logic_kit/hol/thirdorder.py +397 -0
  193. unicode_logic_kit/ilp/__init__.py +89 -0
  194. unicode_logic_kit/ilp/readback.py +389 -0
  195. unicode_logic_kit/ilp/separation.py +153 -0
  196. unicode_logic_kit/ilp/task.py +730 -0
  197. unicode_logic_kit/logic.py +163 -0
  198. unicode_logic_kit/mcp/__init__.py +28 -0
  199. unicode_logic_kit/mcp/__main__.py +5 -0
  200. unicode_logic_kit/mcp/chem_tools.py +1031 -0
  201. unicode_logic_kit/mcp/server.py +2453 -0
  202. unicode_logic_kit/mcp/syntax_spec.py +681 -0
  203. unicode_logic_kit/prob/__init__.py +53 -0
  204. unicode_logic_kit/prob/_bdd.py +225 -0
  205. unicode_logic_kit/prob/_column_gen.py +668 -0
  206. unicode_logic_kit/prob/distribution.py +686 -0
  207. unicode_logic_kit/prob/nilsson.py +470 -0
  208. unicode_logic_kit/py.typed +0 -0
  209. unicode_logic_kit/semantics/__init__.py +137 -0
  210. unicode_logic_kit/semantics/_modal_reject.py +156 -0
  211. unicode_logic_kit/semantics/action_models.py +466 -0
  212. unicode_logic_kit/semantics/asp_models.py +1200 -0
  213. unicode_logic_kit/semantics/conditional.py +580 -0
  214. unicode_logic_kit/semantics/dynamic_epistemic.py +95 -0
  215. unicode_logic_kit/semantics/free_logic.py +913 -0
  216. unicode_logic_kit/semantics/fuzzy.py +384 -0
  217. unicode_logic_kit/semantics/fuzzy_kripke.py +442 -0
  218. unicode_logic_kit/semantics/intuitionistic.py +581 -0
  219. unicode_logic_kit/semantics/kripke.py +1139 -0
  220. unicode_logic_kit/semantics/manyvalued.py +580 -0
  221. unicode_logic_kit/semantics/matrix.py +342 -0
  222. unicode_logic_kit/semantics/model_eval.py +1135 -0
  223. unicode_logic_kit/semantics/modelfinder.py +1036 -0
  224. unicode_logic_kit/semantics/nonmonotonic.py +372 -0
  225. unicode_logic_kit/semantics/relevant.py +331 -0
  226. unicode_logic_kit/semantics/secondorder.py +657 -0
  227. unicode_logic_kit/semantics/structures.py +352 -0
  228. unicode_logic_kit/semantics/tarski.py +975 -0
  229. unicode_logic_kit/semantics/team.py +315 -0
  230. unicode_logic_kit/semantics/team_translation.py +416 -0
  231. unicode_logic_kit/semantics/thirdorder.py +358 -0
  232. unicode_logic_kit/semantics/tnorm.py +85 -0
  233. unicode_logic_kit/semantics/truthtable.py +201 -0
  234. unicode_logic_kit-0.31.0.dist-info/METADATA +333 -0
  235. unicode_logic_kit-0.31.0.dist-info/RECORD +237 -0
  236. unicode_logic_kit-0.31.0.dist-info/WHEEL +4 -0
  237. unicode_logic_kit-0.31.0.dist-info/licenses/LICENSE +21 -0
@@ -0,0 +1,953 @@
1
+ """Entailment checking via the Twee equational theorem prover, through WSL.
2
+
3
+ `Twee <https://nick8325.github.io/twee/>`_ is a Knuth-Bendix-completion-based
4
+ prover for pure UNIT EQUALITY: every axiom and the conjecture must be a
5
+ (possibly universally quantified) equation ``s = t``, or a conjunction of
6
+ such equations — nothing else (no disjunction, no implication, no predicates
7
+ other than ``=``). This module drives Twee exactly as far as that fragment
8
+ reaches: :func:`check_entailment_twee_detailed` raises ``NotImplementedError``
9
+ naming the offending premise/conclusion for anything outside it, mirroring
10
+ :mod:`atp.vampire_entailment`'s contract for the FOL-vs-non-FOL boundary.
11
+
12
+ Twee has no native Windows build; on this machine (and presumably any other
13
+ Windows dev box following the same install recipe) it lives in WSL at
14
+ ``~/.local/bin/twee`` (installed via ``ghcup``/``cabal``, Twee 2.6.1). Every
15
+ invocation therefore goes through ``wsl.exe -e bash -lc "<cmd> ..."`` by
16
+ default (``use_wsl=True``) — the ``bash -lc`` wrapper is required, not
17
+ cosmetic: invoking ``wsl.exe ~/.local/bin/twee`` directly lets the OUTER
18
+ (Windows/Git-Bash) shell expand the ``~`` before ``wsl.exe`` ever sees it,
19
+ which resolves to nonsense on the Windows side. Wrapping the whole command in
20
+ a WSL-side login shell (``bash -lc "..."``) makes WSL's own bash expand the
21
+ ``~`` correctly. The command is overridable via ``$UFK_TWEE_CMD`` (defaults
22
+ to ``~/.local/bin/twee``), and ``use_wsl=False`` switches to a bare
23
+ subprocess call for a machine with a native ``twee`` on ``PATH`` (e.g. a
24
+ Linux/macOS host) — the same ``use_wsl`` toggle :mod:`atp.vampire_entailment`
25
+ uses for Vampire, just defaulted the other way since Twee's only known
26
+ install on THIS kit's dev machines is the WSL one.
27
+
28
+ Windows temp-file paths are translated to their WSL ``/mnt/...`` form via
29
+ ``wslpath`` (identical technique to :mod:`atp.vampire_entailment`'s
30
+ ``_to_wsl_path`` — re-implemented locally here rather than imported, so this
31
+ module has no dependency on the Vampire module).
32
+
33
+ Every run passes ``--quiet`` (Twee's own flag: "Only print essential
34
+ information"). This is not just cosmetic either — WITHOUT it, Twee's stdout
35
+ opens with a restated/flattened echo of the input problem and a running
36
+ completion trace ("Here is the input problem: ... 1. f(a) -> f5 ...") before
37
+ the actual result, which is irrelevant noise for parsing; ``--quiet`` was
38
+ verified live to produce exactly the axiom/lemma/goal/proof text documented
39
+ below and nothing else.
40
+
41
+ Observed output shapes (every one below was produced by a real ``twee
42
+ --quiet`` run on this machine, Twee 2.6.1, and is what
43
+ :func:`parse_twee_proof` is built to read — anything else raises
44
+ ``ValueError``, never a guess):
45
+
46
+ 1. **Theorem, no lemmas** (input ``![X]: f(f(X)) = X`` proving
47
+ ``f(f(f(f(a)))) = a``)::
48
+
49
+ The conjecture is true! Here is a proof.
50
+
51
+ Axiom 1 (a): f(f(X)) = X.
52
+
53
+ Goal 1 (g): f(f(f(f(a)))) = a.
54
+ Proof:
55
+ f(f(f(f(a))))
56
+ = { by axiom 1 (a) }
57
+ f(f(a))
58
+ = { by axiom 1 (a) }
59
+ a
60
+
61
+ RESULT: Theorem (the conjecture is true).
62
+
63
+ The opening banner line is fixed and always present on a Theorem run —
64
+ even under ``--quiet`` ("only print ESSENTIAL information" apparently
65
+ includes it); :func:`parse_twee_proof` requires it verbatim before the
66
+ axiom block.
67
+
68
+ 2. **Theorem with a lemma, and a reverse-direction (``R->L``) step**
69
+ (left-group axioms proving right-identity; banner line omitted below for
70
+ brevity — see shape 1 for where it goes)::
71
+
72
+ Axiom 1 (left_id): mult(e, X) = X.
73
+ Axiom 2 (left_inv): mult(inv(X), X) = e.
74
+ Axiom 3 (assoc): mult(mult(X, Y), Z) = mult(X, mult(Y, Z)).
75
+
76
+ Lemma 4: mult(inv(X), mult(X, Y)) = Y.
77
+ Proof:
78
+ mult(inv(X), mult(X, Y))
79
+ = { by axiom 3 (assoc) R->L }
80
+ mult(mult(inv(X), X), Y)
81
+ = { by axiom 2 (left_inv) }
82
+ mult(e, Y)
83
+ = { by axiom 1 (left_id) }
84
+ Y
85
+
86
+ Goal 1 (right_id): mult(x, e) = x.
87
+ Proof:
88
+ mult(x, e)
89
+ = { by lemma 4 R->L }
90
+ mult(inv(inv(x)), mult(inv(x), mult(x, e)))
91
+ = { by lemma 4 }
92
+ mult(inv(inv(x)), e)
93
+ = { by axiom 2 (left_inv) R->L }
94
+ mult(inv(inv(x)), mult(inv(x), x))
95
+ = { by lemma 4 }
96
+ x
97
+
98
+ RESULT: Theorem (the conjecture is true).
99
+
100
+ Grammar notes distilled from this and further examples (``inv(inv(X))=X``
101
+ needing 2 chained lemma applications with mixed directions; a 2-variable
102
+ commutation goal; a 3-conjunct tupled goal; see ``tests/test_twee.py``
103
+ for the exact captured texts used as fixtures):
104
+
105
+ * A citation is either ``{ by axiom N (name) }`` or ``{ by lemma N }``
106
+ (lemmas are NEVER named in a citation — only axioms carry a ``(name)``,
107
+ because that name is the one WE assigned when generating the TPTP
108
+ problem, see :func:`_generate_twee_input`), optionally suffixed
109
+ ``R->L`` (apply the equation right-to-left).
110
+ * Axiom variables print in a CANONICAL ``X``/``Y``/``Z``... scheme, NOT
111
+ the original source names — verified by feeding an axiom variable named
112
+ ``V`` and seeing it echoed as ``X``. A goal's variables, by contrast,
113
+ are Skolemized to a FRESH constant that is the ORIGINAL bound-variable
114
+ letter, lowercased (``W`` in the conjecture prints as ``w`` in the Goal
115
+ line and its proof) — this is why axioms are matched to the caller's
116
+ original premises up to alpha-equivalence (a name-blind bijection
117
+ check, see :mod:`atp.twee_check`), while goal variables are matched
118
+ structurally instead of by any assumed naming convention.
119
+ * Only axioms actually USED in the proof are listed in the header block —
120
+ an irrelevant extra axiom is silently omitted, so axiom numbers in the
121
+ proof are a purely LOCAL indexing scheme, unrelated to the caller's
122
+ premise-list position; :func:`_generate_twee_input` names every axiom
123
+ ``premise_<i>`` (1-based, matching the caller's premise list) precisely
124
+ so the checker can still recover which original premise a citation
125
+ means, regardless of Twee's renumbering/omission.
126
+ * A premise that is a top-level conjunction of equations gets clausified
127
+ by Twee into one clause per conjunct (a ground conjunct that repeats an
128
+ earlier one is dropped); the clauses are named ``premise_<i>``,
129
+ ``premise_<i>_1``, ``premise_<i>_2``, ... but NOT in the order of the
130
+ source: Twee numbers them in an order of its own (measured, Twee 2.6.1:
131
+ for ``aa = bb ∧ ∀x ff(x) = x`` the clause ``ff(X) = X`` is
132
+ ``premise_1`` and ``aa = bb`` is ``premise_1_1``; for ``aa = bb ∧ aa = bb
133
+ ∧ cc = dd`` the clause ``cc = dd`` is ``premise_1_1``). A name therefore
134
+ tells which PREMISE a clause comes from and nothing more, and
135
+ :mod:`atp.twee_check` decides which conjunct of that premise a restated
136
+ axiom is by what it says.
137
+ * A conjunctive CONCLUSION is encoded by Twee as a single equation
138
+ between ``tuple(...)`` applications: ``![X]: (P(X)) & (Q(X))`` proves
139
+ as ``Goal 1 (goal): tuple(P(sk1), Q(sk2)) = tuple(true, true)``-shaped
140
+ (schematically) — and, importantly, each conjunct receives an
141
+ INDEPENDENT fresh Skolem constant even when the conjuncts share a
142
+ source variable name (verified: a 2-conjunct goal over a shared ``X``
143
+ Skolemized to ``x2`` in the first slot and ``x`` in the second) — sound,
144
+ since ``∀X.(P(X)∧Q(X))`` is equivalent to ``(∀X.P(X))∧(∀X.Q(X))``, two
145
+ independent universal claims. A single-conjunct conclusion is NEVER
146
+ tuple-wrapped.
147
+
148
+ 3. **CounterSatisfiable** (the conjecture does not follow) — with
149
+ ``--quiet``, this is the ENTIRE output, no axiom/proof text at all::
150
+
151
+ RESULT: CounterSatisfiable (the conjecture is false).
152
+
153
+ (Likewise ``Satisfiable``/``Unsatisfiable`` for an axiom-only problem
154
+ with no conjecture — out of scope for this module, which always emits
155
+ exactly one conjecture, but confirmed live for documentation.)
156
+
157
+ 4. **GaveUp** — Twee's own resource budget (``--max-time``, ``--max-cps``,
158
+ etc. — none of which this module passes by default, so this is only
159
+ reachable if a caller forwards such a flag) was exhausted before
160
+ completion could decide the problem either way::
161
+
162
+ RESULT: GaveUp (couldn't solve the problem).
163
+
164
+ 5. **No RESULT line at all** — a genuine subprocess timeout (Twee's own
165
+ resource limits are UNLIMITED by default, so an undecided problem just
166
+ runs forever until this module's own ``timeout`` kills the process; the
167
+ partial stdout at that point, if any, is empty since Twee buffers its
168
+ report until the very end), or a Twee-side parse/usage error (printed to
169
+ STDERR with a non-zero exit and nothing on stdout — verified live with a
170
+ malformed conjecture). Both come back as ``status="Unknown"`` from
171
+ :func:`check_entailment_twee_detailed`; ``result["timed_out"]``
172
+ distinguishes the two.
173
+
174
+ **One name at two arities.** The kit reads a name used at two arities (the constant
175
+ ``ff`` and the unary function ``ff(x)``, or ``ff(x)`` and ``ff(x, y)``) as two symbols,
176
+ and so does every prover the TPTP writer feeds but Twee, which types a symbol by its name
177
+ alone and stops with ``Type mismatch in term 'ff': Constant ff has arity 1 but was applied
178
+ to 0 arguments`` (measured, Twee 2.6.1). So :func:`check_entailment_twee_detailed` writes
179
+ each arity but the first of such a name under a name of its own, one that no symbol of the
180
+ problem has (:func:`_separate_arities`), and hands everything Twee prints back under the
181
+ name the caller used: the parsed proof and the text of ``raw_output`` speak of the problem
182
+ as it was asked, and the proof checker compares it with the caller's own premises.
183
+
184
+ Public API: :func:`twee_available`, :func:`check_entailment_twee_detailed`,
185
+ the proof data classes (:class:`TweeEquation`, :class:`TweeCitation`,
186
+ :class:`TweeChain`, :class:`TweeAxiom`, :class:`TweeLemma`, :class:`TweeGoal`,
187
+ :class:`TweeProof`), and :func:`parse_twee_proof`.
188
+ """
189
+
190
+ import os
191
+ import re
192
+ import subprocess
193
+ import tempfile
194
+ from dataclasses import dataclass
195
+ from typing import Callable, Dict, List, Optional, Tuple
196
+
197
+ from ..fol._fol_nodes import _numeral_from_text
198
+ from ..fol._identifiers import symbol_names
199
+ from ..fol.nodes import (
200
+ Atom, And, Constant, Function, Measure, Node, Number, Quantifier, SortedConstant, Variable,
201
+ )
202
+ from ._ascii_names import reverse_map_text
203
+ from ._tptp_problem import TptpNameMap, apply_reverse_tptp, generate_tptp_problem_with_mapping
204
+
205
+ __all__ = [
206
+ "twee_available", "check_entailment_twee_detailed",
207
+ "TweeEquation", "TweeCitation", "TweeChain",
208
+ "TweeAxiom", "TweeLemma", "TweeGoal", "TweeProof",
209
+ "parse_twee_proof", "reverse_map_twee_proof",
210
+ ]
211
+
212
+ _DEFAULT_TWEE_CMD = "~/.local/bin/twee"
213
+
214
+
215
+ def _twee_cmd(explicit: Optional[str]) -> str:
216
+ """Resolve the command used to invoke Twee: explicit arg > ``$UFK_TWEE_CMD`` > default."""
217
+ return explicit or os.environ.get("UFK_TWEE_CMD") or _DEFAULT_TWEE_CMD
218
+
219
+
220
+ # ---------------------------------------------------------------------------
221
+ # The equational fragment
222
+ # ---------------------------------------------------------------------------
223
+
224
+ def _is_equational(node: Node) -> bool:
225
+ """True iff ``node`` is a (forall-closed) equation or conjunction of equations.
226
+
227
+ Twee decides unit equality only: an ``Atom`` with predicate ``"="`` and
228
+ exactly two arguments, universally quantified any number of times
229
+ (``∀``/``forall`` — an ``∃`` or any other quantifier is out of fragment),
230
+ and/or conjoined with ``And``. Anything else (disjunction, negation,
231
+ implication, a non-``=`` predicate, a modal/second-order/substructural
232
+ node, ...) is not.
233
+ """
234
+ if isinstance(node, Quantifier):
235
+ return node.type in ("forall", "∀") and _is_equational(node.formula)
236
+ if isinstance(node, And):
237
+ return _is_equational(node.left) and _is_equational(node.right)
238
+ if isinstance(node, Atom):
239
+ return node.predicate == "=" and len(node.args) == 2
240
+ return False
241
+
242
+
243
+ def _require_equational(node: Node, role: str) -> None:
244
+ """Raise ``NotImplementedError`` naming ``role`` if ``node`` is not equational."""
245
+ if not _is_equational(node):
246
+ raise NotImplementedError(
247
+ f"twee: {role} is outside the equational fragment Twee decides — "
248
+ f"every premise and the conclusion must be a (forall-closed) "
249
+ f"equation `s = t` or a conjunction of such equations, got "
250
+ f"{role}={node.to_unicode_str()!r}")
251
+
252
+
253
+ # ---------------------------------------------------------------------------
254
+ # TPTP problem generation
255
+ # ---------------------------------------------------------------------------
256
+
257
+ def _generate_twee_input(premises: List[Node], conclusion: Node) -> str:
258
+ """Build a TPTP ``fof`` problem string, one axiom per premise plus one conjecture.
259
+
260
+ A thin wrapper over the shared :func:`atp._tptp_problem.generate_tptp_problem`
261
+ (also used by :mod:`atp.vampire_entailment` and :mod:`atp.eprover_backend`,
262
+ which build the identical problem shape; this module calls its sibling
263
+ :func:`atp._tptp_problem.generate_tptp_problem_with_mapping`, which writes the
264
+ same text and also returns the name map): each premise becomes
265
+ ``fof(premise_<i>, axiom, <tptp>).`` (1-based) and the conclusion becomes
266
+ ``fof(goal, conjecture, <tptp>).``. The ``premise_<i>`` naming is not
267
+ cosmetic here — it is the anchor :mod:`atp.twee_check` uses to recover
268
+ which original premise an axiom citation in Twee's proof refers to (see
269
+ the module docstring's naming-scheme notes). ``generate_tptp_problem``'s
270
+ cross-formula symbol-collision guard can raise ``NotImplementedError``
271
+ before any TPTP text is produced — see ``_tptp_problem``'s module
272
+ docstring. A name used at two arities is written as one name per arity
273
+ first (:func:`_separate_arities`), so that Twee reads each as the symbol it is.
274
+ """
275
+ problem, _name_map, _restore = _twee_problem(premises, conclusion)
276
+ return problem
277
+
278
+
279
+ def _symbol_arities(formulas: List[Node]) -> Dict[str, List[int]]:
280
+ """Every arity each function/constant name of ``formulas`` is used at, in the order the
281
+ arities first occur (a constant, a sorted constant and a function of no arguments are
282
+ the symbol of their name at arity 0; a :class:`~unicode_logic_kit.fol.nodes.Measure` is
283
+ the binary function ``measure``)."""
284
+ arities: Dict[str, List[int]] = {}
285
+ for formula in formulas:
286
+ for node in formula.walk():
287
+ if isinstance(node, Function):
288
+ name, arity = node.name, len(node.args)
289
+ elif isinstance(node, (Constant, SortedConstant)):
290
+ name, arity = node.name, 0
291
+ elif isinstance(node, Measure):
292
+ name, arity = "measure", 2
293
+ else:
294
+ continue
295
+ seen = arities.setdefault(name, [])
296
+ if arity not in seen:
297
+ seen.append(arity)
298
+ return arities
299
+
300
+
301
+ def _rename_by_arity(node: Node, table: Dict[Tuple[str, int], str]) -> Node:
302
+ """``node`` with every function/constant ``(name, arity)`` found in ``table`` written as
303
+ the name ``table`` gives it. Everything else is rebuilt equal; a function of no arguments
304
+ that is renamed is the constant of its new name."""
305
+ if isinstance(node, (Variable, Number)):
306
+ return node
307
+ if isinstance(node, Function):
308
+ args = tuple(_rename_by_arity(a, table) for a in node.args)
309
+ new = table.get((node.name, len(args)))
310
+ if not args:
311
+ return node if new is None else Constant(new)
312
+ return Function(node.name if new is None else new, args)
313
+ if isinstance(node, Constant):
314
+ new = table.get((node.name, 0))
315
+ return node if new is None else Constant(new)
316
+ if isinstance(node, SortedConstant):
317
+ new = table.get((node.name, 0))
318
+ return node if new is None else SortedConstant(new, node.sort)
319
+ if isinstance(node, Measure) and ("measure", 2) in table:
320
+ return Function(table[("measure", 2)], (_rename_by_arity(node.entity, table),
321
+ _rename_by_arity(node.dimension, table)))
322
+ return node.map_children(lambda child: _rename_by_arity(child, table))
323
+
324
+
325
+ def _separate_arities(formulas: List[Node]) -> Tuple[List[Node], Dict[str, str]]:
326
+ """``formulas`` with each name that is used at more than one arity written as one name per
327
+ arity, and the table that undoes it.
328
+
329
+ The kit reads the constant ``ff`` and the unary function ``ff(x)`` as two symbols
330
+ (the problem writer writes them so, and so do Vampire, E, Z3 and Prover9), and Twee
331
+ reads one: it types a symbol by its name and stops on ``ff`` and ``ff(X)`` in one
332
+ problem. The first arity in which a name occurs (premises in order, then the
333
+ conclusion) keeps it; every other arity is written under a name minted for it, which
334
+ is fresh against EVERY name of the problem, of every kind and in every spelling that
335
+ differs from another only in case (TPTP reads ``Ff`` and ``ff`` as one word). That is
336
+ a renaming of one symbol to another that is not in the problem, which changes no
337
+ question asked of it. A minted name is ``sym_arity<n>`` (``sym_arity<n>_<i>`` when
338
+ that is taken): an ASCII word that starts with a lower-case letter whatever the name
339
+ it stands for, so the problem writer leaves it as it is and Twee prints it as
340
+ written. Returns ``(formulas, restore)`` where ``restore`` maps each
341
+ minted name to the name it stands for, to apply to whatever Twee prints; a problem
342
+ without such a name comes back as the very same list and an empty table.
343
+ """
344
+ clashing = {name: seen for name, seen in _symbol_arities(formulas).items() if len(seen) > 1}
345
+ if not clashing:
346
+ return formulas, {}
347
+ taken = set(symbol_names(*formulas, fold=str.casefold))
348
+ table: Dict[Tuple[str, int], str] = {}
349
+ restore: Dict[str, str] = {}
350
+ for name in sorted(clashing):
351
+ for arity in clashing[name][1:]:
352
+ candidate = f"sym_arity{arity}"
353
+ index = 1
354
+ while candidate.casefold() in taken:
355
+ candidate = f"sym_arity{arity}_{index}"
356
+ index += 1
357
+ taken.add(candidate.casefold())
358
+ table[(name, arity)] = candidate
359
+ restore[candidate] = name
360
+ return [_rename_by_arity(f, table) for f in formulas], restore
361
+
362
+
363
+ def _twee_problem(premises: List[Node], conclusion: Node) -> Tuple[str, TptpNameMap, Dict[str, str]]:
364
+ """The problem text handed to Twee, the writer's name map and the table of the names
365
+ :func:`_separate_arities` minted (empty for nearly every problem)."""
366
+ formulas, restore = _separate_arities(list(premises) + [conclusion])
367
+ problem, name_map = generate_tptp_problem_with_mapping(formulas[:-1], formulas[-1])
368
+ return problem, name_map, restore
369
+
370
+
371
+ # ---------------------------------------------------------------------------
372
+ # Process plumbing (WSL by default — see module docstring)
373
+ # ---------------------------------------------------------------------------
374
+
375
+ def _to_wsl_path(windows_path: str) -> str:
376
+ """Translate a Windows path to its WSL ``/mnt/...`` form via ``wslpath``.
377
+
378
+ A local copy of :func:`atp.vampire_entailment._to_wsl_path`'s technique
379
+ (backslashes to forward slashes first, since the WSL interop layer
380
+ swallows literal backslashes in arguments) — reimplemented here rather
381
+ than imported so this module carries no dependency on the Vampire one.
382
+ """
383
+ result = subprocess.run(
384
+ ["wsl.exe", "wslpath", "-u", windows_path.replace("\\", "/")],
385
+ capture_output=True, text=True, timeout=20,
386
+ )
387
+ wsl_path = result.stdout.strip()
388
+ if not wsl_path:
389
+ raise RuntimeError(
390
+ f"wslpath could not translate {windows_path!r} (is WSL available?): "
391
+ f"{result.stderr.strip()}")
392
+ return wsl_path
393
+
394
+
395
+ def _spawn_twee(input_str: str, timeout: int, use_wsl: bool,
396
+ twee_cmd: Optional[str]) -> Tuple[str, str, bool]:
397
+ """Write the TPTP problem to a temp file, run Twee, return ``(stdout, stderr, timed_out)``.
398
+
399
+ With ``use_wsl=True`` (the default), Twee is invoked as
400
+ ``wsl.exe -e bash -lc "<cmd> --quiet '<wsl-path>'"`` — the ``bash -lc``
401
+ wrapper is required so WSL's own shell (not the Windows/Git-Bash caller)
402
+ expands a leading ``~`` in ``<cmd>`` (see module docstring). With
403
+ ``use_wsl=False``, ``<cmd>`` is invoked as a bare subprocess with the
404
+ Windows temp path directly (for a native, non-WSL Twee).
405
+
406
+ A subprocess timeout is swallowed into ``timed_out=True`` (stdout/stderr
407
+ both ``""``), matching :mod:`atp.vampire_entailment`'s convention; any
408
+ other error (e.g. ``FileNotFoundError`` for ``wsl.exe`` itself missing)
409
+ propagates. The temp file is always removed.
410
+ """
411
+ cmd = _twee_cmd(twee_cmd)
412
+ with tempfile.NamedTemporaryFile(mode="w", suffix=".p", delete=False,
413
+ encoding="utf-8") as temp_file:
414
+ temp_file.write(input_str)
415
+ temp_filename = temp_file.name
416
+
417
+ try:
418
+ if use_wsl:
419
+ wsl_path = _to_wsl_path(temp_filename)
420
+ full_command = f"{cmd} --quiet '{wsl_path}'"
421
+ command = ["wsl.exe", "-e", "bash", "-lc", full_command]
422
+ else:
423
+ command = [cmd, "--quiet", temp_filename]
424
+ result = subprocess.run(command, capture_output=True, text=True, timeout=timeout)
425
+ return result.stdout, result.stderr, False
426
+ except subprocess.TimeoutExpired:
427
+ return "", "", True
428
+ finally:
429
+ try:
430
+ os.unlink(temp_filename)
431
+ except OSError:
432
+ pass
433
+
434
+
435
+ def twee_available(use_wsl: bool = True, twee_cmd: Optional[str] = None) -> bool:
436
+ """``True`` iff a Twee binary responds to ``--version`` (cheap; no caching).
437
+
438
+ Discovery: ``twee_cmd`` argument > ``$UFK_TWEE_CMD`` > ``~/.local/bin/twee``
439
+ (see :func:`_twee_cmd`). With ``use_wsl=True`` (the default) the check runs
440
+ ``wsl.exe -e bash -lc "<cmd> --version"``; any failure (no ``wsl.exe``, no
441
+ WSL distro, the command not found inside WSL, ...) is swallowed to
442
+ ``False`` — pure discovery, never raises. Twee prints ``--version`` to
443
+ STDERR, not stdout (verified live), so both streams are checked.
444
+ """
445
+ cmd = _twee_cmd(twee_cmd)
446
+ try:
447
+ if use_wsl:
448
+ command = ["wsl.exe", "-e", "bash", "-lc", f"{cmd} --version"]
449
+ else:
450
+ command = [cmd, "--version"]
451
+ result = subprocess.run(command, capture_output=True, text=True, timeout=20)
452
+ combined = (result.stdout + result.stderr).lower()
453
+ return result.returncode == 0 and "twee version" in combined
454
+ except Exception: # noqa: BLE001 - any failure means "not available"
455
+ return False
456
+
457
+
458
+ # ---------------------------------------------------------------------------
459
+ # Term parsing (Twee's own printed syntax: NAME or NAME(arg, arg, ...))
460
+ # ---------------------------------------------------------------------------
461
+
462
+ _TOKEN_RE = re.compile(r"\s*([A-Za-z_$][A-Za-z0-9_$]*|-?\d+(?:\.\d+)?|[(),])")
463
+
464
+
465
+ def _tokenize(text: str) -> List[str]:
466
+ """Tokenize a Twee-printed term into identifiers, numbers, and ``( ) ,``."""
467
+ tokens = []
468
+ pos = 0
469
+ while pos < len(text):
470
+ m = _TOKEN_RE.match(text, pos)
471
+ if not m:
472
+ if text[pos:].strip() == "":
473
+ break
474
+ raise ValueError(f"twee term parser: unrecognised text at {text[pos:]!r} in {text!r}")
475
+ tokens.append(m.group(1))
476
+ pos = m.end()
477
+ return tokens
478
+
479
+
480
+ def _parse_term(text: str) -> Node:
481
+ """Parse one Twee-printed term into a kit term ``Node``.
482
+
483
+ An identifier starting with an uppercase letter is a :class:`Variable`
484
+ (Twee's own convention for axiom/lemma rewrite variables, e.g. ``X``);
485
+ any other identifier is a :class:`Constant` (no args) or :class:`Function`
486
+ (parenthesised, comma-space-separated args, e.g. ``mult(inv(X), Y)``); a
487
+ bare numeral is a :class:`Number`. Raises ``ValueError`` for anything that
488
+ does not parse as exactly one such term (trailing tokens, unbalanced
489
+ parens, ...) — this is the boundary past which the module refuses to
490
+ guess, per the "anything unrecognised raises" contract.
491
+ """
492
+ tokens = _tokenize(text)
493
+ if not tokens:
494
+ raise ValueError(f"twee term parser: empty term text {text!r}")
495
+ pos = [0]
496
+
497
+ def _peek() -> Optional[str]:
498
+ return tokens[pos[0]] if pos[0] < len(tokens) else None
499
+
500
+ def _numeric(tok: str) -> bool:
501
+ return re.fullmatch(r"-?\d+(?:\.\d+)?", tok) is not None
502
+
503
+ def parse_one() -> Node:
504
+ tok = _peek()
505
+ if tok is None:
506
+ raise ValueError(f"twee term parser: unexpected end of term in {text!r}")
507
+ if _numeric(tok):
508
+ pos[0] += 1
509
+ return Number(_numeral_from_text(tok))
510
+ if tok in "(),":
511
+ raise ValueError(f"twee term parser: unexpected {tok!r} in {text!r}")
512
+ name = tok
513
+ pos[0] += 1
514
+ if _peek() == "(":
515
+ pos[0] += 1
516
+ args = [parse_one()]
517
+ while _peek() == ",":
518
+ pos[0] += 1
519
+ args.append(parse_one())
520
+ if _peek() != ")":
521
+ raise ValueError(f"twee term parser: unbalanced parentheses in {text!r}")
522
+ pos[0] += 1
523
+ return Function(name, args)
524
+ if name[0].isupper():
525
+ return Variable(name)
526
+ return Constant(name)
527
+
528
+ result = parse_one()
529
+ if pos[0] != len(tokens):
530
+ raise ValueError(f"twee term parser: trailing tokens after a complete term in {text!r}")
531
+ return result
532
+
533
+
534
+ def _split_equation(text: str) -> Tuple[str, str]:
535
+ """Split a header's ``"lhs = rhs"`` text on the (sole) top-level ``" = "``."""
536
+ if " = " not in text:
537
+ raise ValueError(f"twee proof parser: expected 'lhs = rhs', got {text!r}")
538
+ lhs, rhs = text.split(" = ", 1)
539
+ return lhs, rhs
540
+
541
+
542
+ # ---------------------------------------------------------------------------
543
+ # Parsed-proof data model
544
+ # ---------------------------------------------------------------------------
545
+
546
+ @dataclass(frozen=True)
547
+ class TweeEquation:
548
+ """One equation as Twee restates it: ``lhs = rhs`` (both kit term Nodes)."""
549
+
550
+ lhs: Node
551
+ rhs: Node
552
+
553
+ def to_dict(self) -> dict:
554
+ return {"lhs": self.lhs.to_dict(), "rhs": self.rhs.to_dict()}
555
+
556
+
557
+ @dataclass(frozen=True)
558
+ class TweeCitation:
559
+ """One ``{ by axiom N (name) [R->L] }`` / ``{ by lemma N [R->L] }`` citation.
560
+
561
+ ``name`` is the axiom's ``(name)`` (always present for ``kind="axiom"``,
562
+ always ``None`` for ``kind="lemma"`` — lemmas are never named in Twee's
563
+ own citations). ``reversed`` is ``True`` iff the citation carries the
564
+ ``R->L`` suffix (apply the equation right-to-left).
565
+ """
566
+
567
+ kind: str
568
+ number: int
569
+ name: Optional[str]
570
+ reversed: bool = False
571
+
572
+ def __post_init__(self):
573
+ if self.kind not in ("axiom", "lemma"):
574
+ raise ValueError(f"TweeCitation: kind must be 'axiom' or 'lemma', got {self.kind!r}")
575
+
576
+ def to_dict(self) -> dict:
577
+ return {"kind": self.kind, "number": self.number, "name": self.name,
578
+ "reversed": self.reversed}
579
+
580
+
581
+ @dataclass(frozen=True)
582
+ class TweeChain:
583
+ """A rewrite chain: ``terms[0]`` rewritten step by step down to ``terms[-1]``.
584
+
585
+ ``len(terms) == len(citations) + 1`` — ``citations[i]`` justifies the step
586
+ from ``terms[i]`` to ``terms[i+1]``.
587
+ """
588
+
589
+ terms: Tuple[Node, ...]
590
+ citations: Tuple[TweeCitation, ...] = ()
591
+
592
+ def __post_init__(self):
593
+ object.__setattr__(self, "terms", tuple(self.terms))
594
+ object.__setattr__(self, "citations", tuple(self.citations))
595
+ if len(self.terms) != len(self.citations) + 1:
596
+ raise ValueError(
597
+ f"TweeChain: {len(self.terms)} terms need exactly "
598
+ f"{max(len(self.terms) - 1, 0)} citations, got {len(self.citations)}")
599
+
600
+ def to_dict(self) -> dict:
601
+ return {"terms": [t.to_dict() for t in self.terms],
602
+ "citations": [c.to_dict() for c in self.citations]}
603
+
604
+
605
+ @dataclass(frozen=True)
606
+ class TweeAxiom:
607
+ """One ``Axiom N (name): lhs = rhs.`` header line."""
608
+
609
+ number: int
610
+ name: str
611
+ equation: TweeEquation
612
+
613
+ def to_dict(self) -> dict:
614
+ return {"number": self.number, "name": self.name, "equation": self.equation.to_dict()}
615
+
616
+
617
+ @dataclass(frozen=True)
618
+ class TweeLemma:
619
+ """One ``Lemma N: lhs = rhs.`` block with its own ``Proof:`` chain."""
620
+
621
+ number: int
622
+ equation: TweeEquation
623
+ chain: TweeChain
624
+
625
+ def to_dict(self) -> dict:
626
+ return {"number": self.number, "equation": self.equation.to_dict(),
627
+ "chain": self.chain.to_dict()}
628
+
629
+
630
+ @dataclass(frozen=True)
631
+ class TweeGoal:
632
+ """The (single, always-last) ``Goal N (name): lhs = rhs.`` block."""
633
+
634
+ number: int
635
+ name: str
636
+ equation: TweeEquation
637
+ chain: TweeChain
638
+
639
+ def to_dict(self) -> dict:
640
+ return {"number": self.number, "name": self.name,
641
+ "equation": self.equation.to_dict(), "chain": self.chain.to_dict()}
642
+
643
+
644
+ @dataclass(frozen=True)
645
+ class TweeProof:
646
+ """A full parsed Twee ``Theorem`` proof: the used axioms, lemmas, then the goal."""
647
+
648
+ axioms: Tuple[TweeAxiom, ...] = ()
649
+ lemmas: Tuple[TweeLemma, ...] = ()
650
+ goal: TweeGoal = None
651
+
652
+ def __post_init__(self):
653
+ object.__setattr__(self, "axioms", tuple(self.axioms))
654
+ object.__setattr__(self, "lemmas", tuple(self.lemmas))
655
+ if self.goal is None:
656
+ raise ValueError("TweeProof: goal is required")
657
+
658
+ def to_dict(self) -> dict:
659
+ return {"axioms": [a.to_dict() for a in self.axioms],
660
+ "lemmas": [l.to_dict() for l in self.lemmas],
661
+ "goal": self.goal.to_dict()}
662
+
663
+
664
+ # ---------------------------------------------------------------------------
665
+ # Proof parser
666
+ # ---------------------------------------------------------------------------
667
+
668
+ _AXIOM_RE = re.compile(r"^Axiom (\d+) \(([^)]*)\): (.+)\.$")
669
+ _LEMMA_RE = re.compile(r"^Lemma (\d+): (.+)\.$")
670
+ _GOAL_RE = re.compile(r"^Goal (\d+) \(([^)]*)\): (.+)\.$")
671
+ _CITATION_RE = re.compile(r"^= \{ by (axiom|lemma) (\d+)(?: \(([^)]*)\))?( R->L)? \}$")
672
+ _RESULT_RE = re.compile(r"^RESULT: (\S+) \(.*\)\.$")
673
+
674
+
675
+ def _extract_result_status(stdout: str) -> Optional[str]:
676
+ """Return Twee's ``RESULT: <Status> (...)`` token, or ``None`` if absent.
677
+
678
+ Only the last non-blank line is checked (that is always where ``RESULT:``
679
+ lives in every observed shape) — a stray line elsewhere that happens to
680
+ start with ``RESULT:`` (not observed, never generated by Twee) is not
681
+ treated as the verdict.
682
+ """
683
+ lines = [ln for ln in stdout.splitlines() if ln.strip()]
684
+ if not lines:
685
+ return None
686
+ m = _RESULT_RE.match(lines[-1].strip())
687
+ return m.group(1) if m else None
688
+
689
+
690
+ def _parse_chain(lines: List[str], idx: int) -> Tuple[TweeChain, int]:
691
+ """Parse a ``Proof:``-following rewrite chain starting at ``lines[idx]``.
692
+
693
+ Returns ``(chain, next_idx)`` where ``next_idx`` is the first line after
694
+ the chain (a blank line, a new header, or end of input).
695
+ """
696
+ if idx >= len(lines) or not lines[idx].startswith(" "):
697
+ raise ValueError(f"twee proof parser: expected an indented term line, got {lines[idx:idx+1]!r}")
698
+ terms = [_parse_term(lines[idx][2:].strip())]
699
+ idx += 1
700
+ citations: List[TweeCitation] = []
701
+ while idx < len(lines) and lines[idx].strip():
702
+ cm = _CITATION_RE.match(lines[idx])
703
+ if not cm:
704
+ break
705
+ kind, number, name, rev = cm.group(1), int(cm.group(2)), cm.group(3), cm.group(4) is not None
706
+ if kind == "axiom" and name is None:
707
+ raise ValueError(f"twee proof parser: axiom citation missing its (name): {lines[idx]!r}")
708
+ if kind == "lemma" and name is not None:
709
+ raise ValueError(f"twee proof parser: lemma citation must not carry a (name): {lines[idx]!r}")
710
+ citations.append(TweeCitation(kind, number, name, rev))
711
+ idx += 1
712
+ if idx >= len(lines) or not lines[idx].startswith(" "):
713
+ raise ValueError(f"twee proof parser: expected an indented term line after a citation, "
714
+ f"got {lines[idx:idx+1]!r}")
715
+ terms.append(_parse_term(lines[idx][2:].strip()))
716
+ idx += 1
717
+ return TweeChain(tuple(terms), tuple(citations)), idx
718
+
719
+
720
+ def parse_twee_proof(stdout: str) -> Optional[TweeProof]:
721
+ """Parse Twee's ``--quiet`` stdout into a :class:`TweeProof`, or ``None``.
722
+
723
+ Returns ``None`` whenever the ``RESULT`` is not ``Theorem`` (Twee's
724
+ ``--quiet`` output for ``CounterSatisfiable``/``Satisfiable``/etc. is just
725
+ the bare ``RESULT:`` line — nothing to parse — see the module docstring's
726
+ shape 3). Raises ``ValueError`` when ``RESULT`` IS ``Theorem`` but the
727
+ surrounding text does not match the grammar distilled in the module
728
+ docstring (shapes 1-2) — never silently produces a partial/guessed proof.
729
+ """
730
+ if _extract_result_status(stdout) != "Theorem":
731
+ return None
732
+
733
+ lines = [ln.rstrip() for ln in stdout.splitlines()]
734
+ idx = 0
735
+
736
+ # Every observed Theorem run opens with this fixed banner line (even
737
+ # under --quiet — verified live; it is the one piece of "essential
738
+ # information" --quiet does not suppress) followed by a blank line.
739
+ if idx >= len(lines) or lines[idx] != "The conjecture is true! Here is a proof.":
740
+ raise ValueError(
741
+ f"twee proof parser: expected the 'conjecture is true' banner line, "
742
+ f"got {lines[idx:idx + 1]!r}")
743
+ idx += 1
744
+ while idx < len(lines) and not lines[idx].strip():
745
+ idx += 1
746
+
747
+ axioms: List[TweeAxiom] = []
748
+ while idx < len(lines) and lines[idx].strip():
749
+ m = _AXIOM_RE.match(lines[idx])
750
+ if not m:
751
+ break
752
+ number, name, eq_text = int(m.group(1)), m.group(2), m.group(3)
753
+ lhs_text, rhs_text = _split_equation(eq_text)
754
+ axioms.append(TweeAxiom(number, name,
755
+ TweeEquation(_parse_term(lhs_text), _parse_term(rhs_text))))
756
+ idx += 1
757
+ while idx < len(lines) and not lines[idx].strip():
758
+ idx += 1
759
+
760
+ lemmas: List[TweeLemma] = []
761
+ goal: Optional[TweeGoal] = None
762
+ while idx < len(lines):
763
+ if not lines[idx].strip():
764
+ idx += 1
765
+ continue
766
+ if _RESULT_RE.match(lines[idx]):
767
+ # The trailing RESULT line is a natural terminator, not a
768
+ # Lemma/Goal header — reached here only for a malformed proof
769
+ # that has no Goal block at all (see the "no Goal block" check
770
+ # below); the RESULT's own status was already read separately.
771
+ break
772
+ m_lemma = _LEMMA_RE.match(lines[idx])
773
+ m_goal = _GOAL_RE.match(lines[idx])
774
+ if m_lemma:
775
+ number, eq_text = int(m_lemma.group(1)), m_lemma.group(2)
776
+ lhs_text, rhs_text = _split_equation(eq_text)
777
+ equation = TweeEquation(_parse_term(lhs_text), _parse_term(rhs_text))
778
+ idx += 1
779
+ if idx >= len(lines) or lines[idx].strip() != "Proof:":
780
+ raise ValueError(f"twee proof parser: expected 'Proof:' after Lemma {number}")
781
+ idx += 1
782
+ chain, idx = _parse_chain(lines, idx)
783
+ lemmas.append(TweeLemma(number, equation, chain))
784
+ elif m_goal:
785
+ number, name, eq_text = int(m_goal.group(1)), m_goal.group(2), m_goal.group(3)
786
+ lhs_text, rhs_text = _split_equation(eq_text)
787
+ equation = TweeEquation(_parse_term(lhs_text), _parse_term(rhs_text))
788
+ idx += 1
789
+ if idx >= len(lines) or lines[idx].strip() != "Proof:":
790
+ raise ValueError(f"twee proof parser: expected 'Proof:' after Goal {number}")
791
+ idx += 1
792
+ chain, idx = _parse_chain(lines, idx)
793
+ goal = TweeGoal(number, name, equation, chain)
794
+ break # the goal is always the last block Twee prints
795
+ else:
796
+ raise ValueError(
797
+ f"twee proof parser: expected a Lemma/Goal header, got {lines[idx]!r}")
798
+
799
+ if goal is None:
800
+ raise ValueError("twee proof parser: RESULT was Theorem but no Goal block was found")
801
+
802
+ return TweeProof(tuple(axioms), tuple(lemmas), goal)
803
+
804
+
805
+ # ---------------------------------------------------------------------------
806
+ # Rückweg: translate a parsed TweeProof's terms back to kit-level names.
807
+ # ---------------------------------------------------------------------------
808
+
809
+ def _map_proof_terms(proof: TweeProof, fn: Callable[[Node], Node]) -> TweeProof:
810
+ """``proof`` with ``fn`` applied to every term of it (the two sides of each axiom, lemma
811
+ and goal equation and every term of every chain); numbers, names and citations stay."""
812
+ def equation(eq: TweeEquation) -> TweeEquation:
813
+ return TweeEquation(fn(eq.lhs), fn(eq.rhs))
814
+
815
+ def chain(c: TweeChain) -> TweeChain:
816
+ return TweeChain(tuple(fn(t) for t in c.terms), c.citations)
817
+
818
+ return TweeProof(
819
+ tuple(TweeAxiom(a.number, a.name, equation(a.equation)) for a in proof.axioms),
820
+ tuple(TweeLemma(l.number, equation(l.equation), chain(l.chain)) for l in proof.lemmas),
821
+ TweeGoal(proof.goal.number, proof.goal.name, equation(proof.goal.equation),
822
+ chain(proof.goal.chain)))
823
+
824
+
825
+ def _restore_term_names(term: Node, restore: Dict[str, str]) -> Node:
826
+ """``term`` with every function/constant name that ``restore`` maps written as the name it
827
+ maps to (see :func:`_separate_arities`)."""
828
+ if isinstance(term, Function):
829
+ args = tuple(_restore_term_names(a, restore) for a in term.args)
830
+ return Function(restore.get(term.name, term.name), args)
831
+ if isinstance(term, Constant):
832
+ return Constant(restore.get(term.name, term.name))
833
+ return term
834
+
835
+
836
+ def reverse_map_twee_proof(proof: TweeProof, mapping: TptpNameMap) -> TweeProof:
837
+ """Rewrite every term in ``proof`` from the sanitised TPTP-ASCII function/
838
+ constant names :func:`atp._tptp_problem.generate_tptp_problem_with_mapping`
839
+ chose back to the original kit-level names (see
840
+ :func:`atp._tptp_problem.apply_reverse_tptp`).
841
+
842
+ Twee's equational fragment only ever uses ``=`` as a predicate (excluded
843
+ from renaming to begin with — see :mod:`atp._tptp_problem`'s module
844
+ docstring), so only function/constant names can ever have been
845
+ sanitised here; ``apply_reverse_tptp`` is reused as-is (it simply never
846
+ hits its ``Atom`` branch on a bare term). Axiom/lemma/goal NUMBERS and
847
+ axiom NAMES (Twee's own local proof-step indexing, and the
848
+ ``premise_<i>`` names this module itself assigned — see
849
+ :func:`_generate_twee_input`) are not symbol names and are left alone.
850
+ """
851
+ return _map_proof_terms(proof, lambda term: apply_reverse_tptp(term, mapping))
852
+
853
+
854
+ # ---------------------------------------------------------------------------
855
+ # Public entry point
856
+ # ---------------------------------------------------------------------------
857
+
858
+ def check_entailment_twee_detailed(premises: List[Node], conclusion: Node,
859
+ timeout: int = 30, use_wsl: bool = True,
860
+ twee_cmd: Optional[str] = None) -> dict:
861
+ """Run Twee on ``premises ⊨ conclusion`` and return its status, output, and proof.
862
+
863
+ Both ``premises`` and ``conclusion`` must be in Twee's equational
864
+ fragment (see :func:`_is_equational`) — anything else raises
865
+ ``NotImplementedError`` naming the offender, BEFORE any subprocess is
866
+ spawned (same contract as :func:`atp.vampire_entailment
867
+ .check_entailment_vampire_detailed`). Problem generation additionally
868
+ refuses (also ``NotImplementedError``, also before any subprocess) if
869
+ two distinct function/constant names would fold to the same TPTP
870
+ identifier — see :mod:`atp._tptp_problem`'s module docstring; ``=`` is
871
+ the only predicate this fragment ever uses, and it is excluded from that
872
+ check entirely (see the same docstring), so only a function/constant
873
+ collision can occur here.
874
+
875
+ Args:
876
+ premises: equational premise formulas.
877
+ conclusion: the equational conclusion formula.
878
+ timeout: seconds to allow the Twee process before giving up (a
879
+ subprocess timeout, not one of Twee's own ``--max-*`` flags,
880
+ which this function never passes).
881
+ use_wsl: drive Twee through WSL (default ``True`` — see module
882
+ docstring); ``False`` for a native, non-WSL Twee on ``PATH``.
883
+ twee_cmd: override the Twee command/path (default: ``$UFK_TWEE_CMD``,
884
+ else ``~/.local/bin/twee``).
885
+
886
+ Returns:
887
+ A dict with:
888
+
889
+ * ``status``: Twee's own ``RESULT:`` token verbatim (``"Theorem"``,
890
+ ``"CounterSatisfiable"``, ``"GaveUp"``, ``"Satisfiable"``,
891
+ ``"Unsatisfiable"``, ...), or ``"Unknown"`` when no ``RESULT:``
892
+ line was found at all (a subprocess timeout, or a Twee-side
893
+ parse/usage error — see ``timed_out`` to distinguish).
894
+ * ``raw_output``: Twee's full stdout; if no ``RESULT:`` line was
895
+ found and Twee wrote to stderr (a parse/usage error), stderr is
896
+ appended so the failure is diagnosable.
897
+ * ``proof``: the parsed :class:`TweeProof` when ``status ==
898
+ "Theorem"`` (call ``.to_dict()`` for a JSON-compatible form — kept
899
+ as the live dataclass here, not pre-serialised, because
900
+ :func:`atp.twee_check.check_twee_proof` consumes it directly),
901
+ else ``None``.
902
+ * ``timed_out``: ``True`` iff this function's own ``timeout`` (not
903
+ one of Twee's ``--max-*`` budgets) killed the subprocess.
904
+
905
+ Every function/constant name in ``raw_output`` and in ``proof``'s
906
+ terms has already been translated back from whatever ASCII-safe
907
+ token :func:`atp._tptp_problem.generate_tptp_problem_with_mapping`
908
+ may have substituted (a non-ASCII or digit-leading kit-level name)
909
+ to the ORIGINAL kit-level name (see
910
+ :func:`reverse_map_twee_proof`); Twee's own axiom/lemma/goal
911
+ numbering and the ``premise_<i>`` axiom names are untouched (they
912
+ were never symbol names to begin with). A name the problem uses at
913
+ two arities is a problem for Twee alone (it types a symbol by its name):
914
+ it is written as one name per arity (see the module docstring) and
915
+ handed back under the caller's name in both.
916
+ """
917
+ for i, premise in enumerate(premises, start=1):
918
+ _require_equational(premise, f"premise {i}")
919
+ _require_equational(conclusion, "conclusion")
920
+
921
+ problem, name_map, restore = _twee_problem(list(premises), conclusion)
922
+ stdout, stderr, timed_out = _spawn_twee(problem, timeout=timeout, use_wsl=use_wsl,
923
+ twee_cmd=twee_cmd)
924
+
925
+ if timed_out:
926
+ return {"status": "Unknown", "raw_output": "", "proof": None,
927
+ "timed_out": True, "proof_parse_error": None}
928
+
929
+ status = _extract_result_status(stdout) or "Unknown"
930
+ raw_output = stdout
931
+ if status == "Unknown" and stderr:
932
+ raw_output = f"{stdout}\n{stderr}" if stdout else stderr
933
+ pred_rev, term_rev = name_map.reverse_rendered()
934
+ # The names minted for a name used at two arities come first: the writer's own table
935
+ # maps each of them to itself, and the first table to know a token decides.
936
+ raw_output = reverse_map_text(raw_output, restore, pred_rev, term_rev)
937
+
938
+ proof = None
939
+ proof_parse_error = None
940
+ if status == "Theorem":
941
+ try:
942
+ proof = reverse_map_twee_proof(parse_twee_proof(stdout), name_map)
943
+ proof = _map_proof_terms(proof, lambda term: _restore_term_names(term, restore))
944
+ except ValueError as exc:
945
+ # A Theorem status with proof text outside this module's
946
+ # distilled grammar (another Twee version, a reformat) must not
947
+ # CRASH the caller: the status stands, the proof is honestly
948
+ # absent, and the parser's complaint travels along so
949
+ # TweeBackend can refuse PROVED with the reason on record
950
+ # (review-confirmed: this previously raised out of decide()).
951
+ proof_parse_error = str(exc)
952
+ return {"status": status, "raw_output": raw_output, "proof": proof,
953
+ "timed_out": False, "proof_parse_error": proof_parse_error}