unicode-logic-kit 0.31.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (237) hide show
  1. unicode_logic_kit/__init__.py +385 -0
  2. unicode_logic_kit/__main__.py +520 -0
  3. unicode_logic_kit/_deadline.py +219 -0
  4. unicode_logic_kit/ace/__init__.py +126 -0
  5. unicode_logic_kit/ace/_align.py +135 -0
  6. unicode_logic_kit/ace/chem_lexicon.py +128 -0
  7. unicode_logic_kit/ace/drs_reader.py +570 -0
  8. unicode_logic_kit/ace/mapping.py +666 -0
  9. unicode_logic_kit/ace/reverse_modal.py +138 -0
  10. unicode_logic_kit/ace/runner.py +551 -0
  11. unicode_logic_kit/ace/translate.py +452 -0
  12. unicode_logic_kit/ace/verbalize.py +1070 -0
  13. unicode_logic_kit/api.py +1284 -0
  14. unicode_logic_kit/atp/__init__.py +177 -0
  15. unicode_logic_kit/atp/_ascii_names.py +113 -0
  16. unicode_logic_kit/atp/_html.py +72 -0
  17. unicode_logic_kit/atp/_substructural_input.py +228 -0
  18. unicode_logic_kit/atp/_tff_problem.py +715 -0
  19. unicode_logic_kit/atp/_tptp_problem.py +1111 -0
  20. unicode_logic_kit/atp/_writer_support.py +289 -0
  21. unicode_logic_kit/atp/clingo_backend.py +1180 -0
  22. unicode_logic_kit/atp/cvc5_backend.py +1385 -0
  23. unicode_logic_kit/atp/eprover_backend.py +732 -0
  24. unicode_logic_kit/atp/finite_domain.py +1055 -0
  25. unicode_logic_kit/atp/fitch.py +1547 -0
  26. unicode_logic_kit/atp/fitch_search.py +551 -0
  27. unicode_logic_kit/atp/hets_backend.py +339 -0
  28. unicode_logic_kit/atp/hybrid_down.py +120 -0
  29. unicode_logic_kit/atp/incremental.py +250 -0
  30. unicode_logic_kit/atp/kripke_enum.py +741 -0
  31. unicode_logic_kit/atp/lambek.py +436 -0
  32. unicode_logic_kit/atp/leo3_backend.py +332 -0
  33. unicode_logic_kit/atp/linear.py +738 -0
  34. unicode_logic_kit/atp/lj.py +705 -0
  35. unicode_logic_kit/atp/logic_backends.py +566 -0
  36. unicode_logic_kit/atp/ltl_tableau.py +1084 -0
  37. unicode_logic_kit/atp/minizinc_backend.py +1402 -0
  38. unicode_logic_kit/atp/modal_tableau.py +1382 -0
  39. unicode_logic_kit/atp/nanocop_backend.py +410 -0
  40. unicode_logic_kit/atp/portfolio.py +489 -0
  41. unicode_logic_kit/atp/protocol.py +1803 -0
  42. unicode_logic_kit/atp/prover9_entailment.py +1153 -0
  43. unicode_logic_kit/atp/resolution.py +1376 -0
  44. unicode_logic_kit/atp/resolution_check.py +1114 -0
  45. unicode_logic_kit/atp/sequent.py +1050 -0
  46. unicode_logic_kit/atp/tableau.py +921 -0
  47. unicode_logic_kit/atp/tableau_check.py +543 -0
  48. unicode_logic_kit/atp/tptp_ncl.py +811 -0
  49. unicode_logic_kit/atp/tptp_tff.py +1546 -0
  50. unicode_logic_kit/atp/tstp.py +1333 -0
  51. unicode_logic_kit/atp/tstp_check.py +1096 -0
  52. unicode_logic_kit/atp/twee_backend.py +236 -0
  53. unicode_logic_kit/atp/twee_check.py +711 -0
  54. unicode_logic_kit/atp/twee_entailment.py +953 -0
  55. unicode_logic_kit/atp/vampire_entailment.py +540 -0
  56. unicode_logic_kit/atp/z3_arith.py +470 -0
  57. unicode_logic_kit/atp/z3_equivalence.py +36 -0
  58. unicode_logic_kit/atp/z3_fuzzy.py +362 -0
  59. unicode_logic_kit/atp/z3_input.py +500 -0
  60. unicode_logic_kit/atp/z3_models.py +208 -0
  61. unicode_logic_kit/chem/__init__.py +88 -0
  62. unicode_logic_kit/chem/_naming.py +284 -0
  63. unicode_logic_kit/chem/cache.py +185 -0
  64. unicode_logic_kit/chem/interop.py +244 -0
  65. unicode_logic_kit/chem/mol.py +525 -0
  66. unicode_logic_kit/chem/signature.py +112 -0
  67. unicode_logic_kit/comorphism.py +497 -0
  68. unicode_logic_kit/dl/__init__.py +384 -0
  69. unicode_logic_kit/dl/classification.py +227 -0
  70. unicode_logic_kit/dl/concepts.py +632 -0
  71. unicode_logic_kit/dl/datatypes.py +818 -0
  72. unicode_logic_kit/dl/owl_functional.py +2433 -0
  73. unicode_logic_kit/dl/owl_manchester.py +1637 -0
  74. unicode_logic_kit/dl/owl_reasoner.py +790 -0
  75. unicode_logic_kit/dl/parser.py +391 -0
  76. unicode_logic_kit/dl/tableau.py +4048 -0
  77. unicode_logic_kit/dl/translate.py +2704 -0
  78. unicode_logic_kit/drt/__init__.py +94 -0
  79. unicode_logic_kit/drt/export.py +179 -0
  80. unicode_logic_kit/drt/nodes.py +506 -0
  81. unicode_logic_kit/drt/parser.py +965 -0
  82. unicode_logic_kit/drt/resolve.py +195 -0
  83. unicode_logic_kit/drt/reverse.py +175 -0
  84. unicode_logic_kit/eval/__init__.py +106 -0
  85. unicode_logic_kit/eval/batch.py +382 -0
  86. unicode_logic_kit/eval/canonical.py +663 -0
  87. unicode_logic_kit/eval/chem_batch.py +606 -0
  88. unicode_logic_kit/eval/converses.py +200 -0
  89. unicode_logic_kit/eval/datasets/__init__.py +136 -0
  90. unicode_logic_kit/eval/datasets/_base.py +263 -0
  91. unicode_logic_kit/eval/datasets/_proofwriter_proof.py +422 -0
  92. unicode_logic_kit/eval/datasets/c3po.py +678 -0
  93. unicode_logic_kit/eval/datasets/folio.py +158 -0
  94. unicode_logic_kit/eval/datasets/fracas.py +418 -0
  95. unicode_logic_kit/eval/datasets/groves.py +191 -0
  96. unicode_logic_kit/eval/datasets/logicbench.py +467 -0
  97. unicode_logic_kit/eval/datasets/logicnli.py +303 -0
  98. unicode_logic_kit/eval/datasets/malls.py +133 -0
  99. unicode_logic_kit/eval/datasets/pfolio.py +594 -0
  100. unicode_logic_kit/eval/datasets/pmb.py +242 -0
  101. unicode_logic_kit/eval/datasets/prontoqa.py +611 -0
  102. unicode_logic_kit/eval/datasets/proofwriter.py +1431 -0
  103. unicode_logic_kit/eval/datasets/proverqa.py +674 -0
  104. unicode_logic_kit/eval/datasets/willow.py +478 -0
  105. unicode_logic_kit/eval/equivalence.py +466 -0
  106. unicode_logic_kit/eval/exercise_gen.py +533 -0
  107. unicode_logic_kit/eval/explain.py +791 -0
  108. unicode_logic_kit/eval/generality.py +750 -0
  109. unicode_logic_kit/eval/metric_hf.py +458 -0
  110. unicode_logic_kit/eval/predicate_match.py +343 -0
  111. unicode_logic_kit/eval/theory_check.py +1170 -0
  112. unicode_logic_kit/eval/validate.py +306 -0
  113. unicode_logic_kit/fol/__init__.py +177 -0
  114. unicode_logic_kit/fol/_atom_keys.py +510 -0
  115. unicode_logic_kit/fol/_fol_nodes.py +3586 -0
  116. unicode_logic_kit/fol/_free_parameters.py +105 -0
  117. unicode_logic_kit/fol/_ho_nodes.py +448 -0
  118. unicode_logic_kit/fol/_hybrid_nodes.py +308 -0
  119. unicode_logic_kit/fol/_identifiers.py +1091 -0
  120. unicode_logic_kit/fol/_lambek_nodes.py +112 -0
  121. unicode_logic_kit/fol/_linear_nodes.py +352 -0
  122. unicode_logic_kit/fol/_modal_nodes.py +1467 -0
  123. unicode_logic_kit/fol/_msfl_nodes.py +2196 -0
  124. unicode_logic_kit/fol/_numeral_symbols.py +231 -0
  125. unicode_logic_kit/fol/_so_nodes.py +200 -0
  126. unicode_logic_kit/fol/_symbol_names.py +81 -0
  127. unicode_logic_kit/fol/_team_nodes.py +181 -0
  128. unicode_logic_kit/fol/_tptp_symbols.py +551 -0
  129. unicode_logic_kit/fol/_truth_constants.py +117 -0
  130. unicode_logic_kit/fol/casl_export.py +1135 -0
  131. unicode_logic_kit/fol/casl_import.py +929 -0
  132. unicode_logic_kit/fol/derivation.py +367 -0
  133. unicode_logic_kit/fol/dialect_detect.py +70 -0
  134. unicode_logic_kit/fol/dialect_repair.py +537 -0
  135. unicode_logic_kit/fol/frames.py +637 -0
  136. unicode_logic_kit/fol/grammars/terminals.lark +31 -0
  137. unicode_logic_kit/fol/lambda_tools.py +297 -0
  138. unicode_logic_kit/fol/latex_input.py +429 -0
  139. unicode_logic_kit/fol/modal_translation.py +944 -0
  140. unicode_logic_kit/fol/msflparser.py +1033 -0
  141. unicode_logic_kit/fol/naming.py +422 -0
  142. unicode_logic_kit/fol/nodes.py +241 -0
  143. unicode_logic_kit/fol/normalforms.py +492 -0
  144. unicode_logic_kit/fol/pal.py +287 -0
  145. unicode_logic_kit/fol/prolog_export.py +566 -0
  146. unicode_logic_kit/fol/prolog_input.py +505 -0
  147. unicode_logic_kit/fol/prover9_input.py +1325 -0
  148. unicode_logic_kit/fol/qml.py +1760 -0
  149. unicode_logic_kit/fol/qmltp_input.py +525 -0
  150. unicode_logic_kit/fol/sanitize.py +221 -0
  151. unicode_logic_kit/fol/serialize.py +79 -0
  152. unicode_logic_kit/fol/signature.py +1290 -0
  153. unicode_logic_kit/fol/simplify_check.py +544 -0
  154. unicode_logic_kit/fol/spans.py +594 -0
  155. unicode_logic_kit/fol/tptp_input.py +1503 -0
  156. unicode_logic_kit/fol/tptp_repair.py +941 -0
  157. unicode_logic_kit/fol/unification.py +157 -0
  158. unicode_logic_kit/fol/verbalize.py +263 -0
  159. unicode_logic_kit/hets/__init__.py +163 -0
  160. unicode_logic_kit/hets/bridge.py +142 -0
  161. unicode_logic_kit/hets/client.py +748 -0
  162. unicode_logic_kit/hets/docker.py +420 -0
  163. unicode_logic_kit/hets/dol.py +712 -0
  164. unicode_logic_kit/hets/haskell_json.py +355 -0
  165. unicode_logic_kit/hets/owl_backend.py +794 -0
  166. unicode_logic_kit/hets/owl_cli.py +598 -0
  167. unicode_logic_kit/hets/symbols.py +512 -0
  168. unicode_logic_kit/hol/__init__.py +140 -0
  169. unicode_logic_kit/hol/_ho_common.py +323 -0
  170. unicode_logic_kit/hol/_isabelle_binders.py +125 -0
  171. unicode_logic_kit/hol/classical.py +812 -0
  172. unicode_logic_kit/hol/deepshallow/__init__.py +45 -0
  173. unicode_logic_kit/hol/deepshallow/_common.py +177 -0
  174. unicode_logic_kit/hol/deepshallow/conditional.py +225 -0
  175. unicode_logic_kit/hol/deepshallow/intuitionistic.py +181 -0
  176. unicode_logic_kit/hol/deepshallow/modal.py +217 -0
  177. unicode_logic_kit/hol/deepshallow/qml.py +406 -0
  178. unicode_logic_kit/hol/deepshallow/relevant.py +206 -0
  179. unicode_logic_kit/hol/free.py +753 -0
  180. unicode_logic_kit/hol/goedel.py +336 -0
  181. unicode_logic_kit/hol/ho_modal.py +1743 -0
  182. unicode_logic_kit/hol/intuitionistic.py +403 -0
  183. unicode_logic_kit/hol/isabelle_conditional.py +593 -0
  184. unicode_logic_kit/hol/isabelle_modal.py +1908 -0
  185. unicode_logic_kit/hol/isabelle_relevant.py +412 -0
  186. unicode_logic_kit/hol/isabelle_runner.py +1147 -0
  187. unicode_logic_kit/hol/isabelle_substructural.py +884 -0
  188. unicode_logic_kit/hol/lean.py +1018 -0
  189. unicode_logic_kit/hol/manyvalued.py +921 -0
  190. unicode_logic_kit/hol/secondorder.py +687 -0
  191. unicode_logic_kit/hol/thf_modal.py +941 -0
  192. unicode_logic_kit/hol/thirdorder.py +397 -0
  193. unicode_logic_kit/ilp/__init__.py +89 -0
  194. unicode_logic_kit/ilp/readback.py +389 -0
  195. unicode_logic_kit/ilp/separation.py +153 -0
  196. unicode_logic_kit/ilp/task.py +730 -0
  197. unicode_logic_kit/logic.py +163 -0
  198. unicode_logic_kit/mcp/__init__.py +28 -0
  199. unicode_logic_kit/mcp/__main__.py +5 -0
  200. unicode_logic_kit/mcp/chem_tools.py +1031 -0
  201. unicode_logic_kit/mcp/server.py +2453 -0
  202. unicode_logic_kit/mcp/syntax_spec.py +681 -0
  203. unicode_logic_kit/prob/__init__.py +53 -0
  204. unicode_logic_kit/prob/_bdd.py +225 -0
  205. unicode_logic_kit/prob/_column_gen.py +668 -0
  206. unicode_logic_kit/prob/distribution.py +686 -0
  207. unicode_logic_kit/prob/nilsson.py +470 -0
  208. unicode_logic_kit/py.typed +0 -0
  209. unicode_logic_kit/semantics/__init__.py +137 -0
  210. unicode_logic_kit/semantics/_modal_reject.py +156 -0
  211. unicode_logic_kit/semantics/action_models.py +466 -0
  212. unicode_logic_kit/semantics/asp_models.py +1200 -0
  213. unicode_logic_kit/semantics/conditional.py +580 -0
  214. unicode_logic_kit/semantics/dynamic_epistemic.py +95 -0
  215. unicode_logic_kit/semantics/free_logic.py +913 -0
  216. unicode_logic_kit/semantics/fuzzy.py +384 -0
  217. unicode_logic_kit/semantics/fuzzy_kripke.py +442 -0
  218. unicode_logic_kit/semantics/intuitionistic.py +581 -0
  219. unicode_logic_kit/semantics/kripke.py +1139 -0
  220. unicode_logic_kit/semantics/manyvalued.py +580 -0
  221. unicode_logic_kit/semantics/matrix.py +342 -0
  222. unicode_logic_kit/semantics/model_eval.py +1135 -0
  223. unicode_logic_kit/semantics/modelfinder.py +1036 -0
  224. unicode_logic_kit/semantics/nonmonotonic.py +372 -0
  225. unicode_logic_kit/semantics/relevant.py +331 -0
  226. unicode_logic_kit/semantics/secondorder.py +657 -0
  227. unicode_logic_kit/semantics/structures.py +352 -0
  228. unicode_logic_kit/semantics/tarski.py +975 -0
  229. unicode_logic_kit/semantics/team.py +315 -0
  230. unicode_logic_kit/semantics/team_translation.py +416 -0
  231. unicode_logic_kit/semantics/thirdorder.py +358 -0
  232. unicode_logic_kit/semantics/tnorm.py +85 -0
  233. unicode_logic_kit/semantics/truthtable.py +201 -0
  234. unicode_logic_kit-0.31.0.dist-info/METADATA +333 -0
  235. unicode_logic_kit-0.31.0.dist-info/RECORD +237 -0
  236. unicode_logic_kit-0.31.0.dist-info/WHEEL +4 -0
  237. unicode_logic_kit-0.31.0.dist-info/licenses/LICENSE +21 -0
@@ -0,0 +1,466 @@
1
+ """Graded equivalence checking for NL→FOL evaluation.
2
+
3
+ ``equivalent(prediction, reference)`` is the one entry point that answers "does
4
+ the predicted formula mean the same as the reference?" at every level the kit
5
+ can decide, from cheap structural checks to a solver call:
6
+
7
+ * ``exact`` — structural equality of the frozen ASTs (``==``);
8
+ * ``canonical`` — :func:`~unicode_logic_kit.eval.canonical.exact_match`
9
+ (α-renaming, commutativity/associativity, operand
10
+ de-duplication, double negation quotiented out);
11
+ * ``predicate_align`` — :func:`~unicode_logic_kit.eval.predicate_match.aligned_exact_match`
12
+ (vocabulary differences quotiented out too:
13
+ namespace- and arity-aware injective renaming,
14
+ then canonical match);
15
+ * ``solver`` — genuine logical equivalence. Classical formulas go
16
+ to Z3 with a TRI-STATE result (``True`` proved /
17
+ ``False`` refuted with a counterexample /
18
+ ``None`` unknown — the existing
19
+ ``formulas_are_equivalent`` collapses unknown to
20
+ False, which a metric must not do). Modal formulas
21
+ go to ``modal_decide`` (tri-state, with a Kripke
22
+ counterexample on refutation) and fall back to the
23
+ sound-but-bounded-incomplete ``qml_equivalent``
24
+ (``True`` / ``None``) for fragments the tableau
25
+ rejects;
26
+ * ``auto`` — the ladder above, cheapest first, stopping at the
27
+ first level that answers ``True``.
28
+
29
+ An OPT-IN ``converses`` argument additionally lets a caller declare
30
+ argument-permutation bridging axioms — e.g. ``LovedBy(x, y) ↔ Loves(y, x)`` —
31
+ that the SOLVER level asserts as extra premises (see
32
+ :mod:`unicode_logic_kit.eval.converses`). Only ``method="solver"``/``"auto"``
33
+ ever honour it (a structural level would silently ignore a declared axiom it
34
+ cannot consume, which is refused rather than allowed); the result is tagged
35
+ with its own ``method_used`` value, ``"solver_modulo_converses"``, never
36
+ merged into plain ``"solver"`` so a consumer can always tell whether a
37
+ verdict relied on caller-declared bridges. This kit's whole classical Z3
38
+ export uses exactly one Z3 sort for every term (see
39
+ ``eval.converses``'s module docstring), so a declared axiom for a
40
+ ``(name, arity)`` predicate pair interns to the identical Z3 function the
41
+ compared formulas themselves use — many-sorted (``SortedQuantifier``) input
42
+ included; ``tests/test_converses.py`` proves this end-to-end rather than just
43
+ asserting it.
44
+
45
+ Every result is an :class:`EquivalenceResult` whose ``to_dict()`` is
46
+ JSON-compatible, so pipelines and LLM repair loops can consume it without
47
+ extra parsing.
48
+
49
+ ``partial_credit`` (``method="auto"`` / ``"solver"`` only) is a heuristic
50
+ RANKING signal for use when the headline verdict is not a clean ``True`` —
51
+ NOT a probability, just a coarse {0, 0.25, 0.5, 0.75, 1.0} ordinal built from
52
+ four binary sub-checks (well-formedness, vocabulary alignability, aligned
53
+ structural match, logical equivalence). See :class:`EquivalenceResult` for
54
+ the exact rubric.
55
+ """
56
+
57
+ from dataclasses import dataclass
58
+ from typing import Optional
59
+
60
+ from unicode_logic_kit.fol._msfl_nodes import sort_axioms
61
+ from unicode_logic_kit.fol.nodes import (
62
+ Node, Iff, Quantifier, SortedQuantifier,
63
+ )
64
+
65
+ __all__ = ["EquivalenceResult", "equivalent"]
66
+
67
+ _METHODS = ("exact", "canonical", "predicate_align", "solver", "auto")
68
+
69
+
70
+ @dataclass(frozen=True)
71
+ class EquivalenceResult:
72
+ """Outcome of :func:`equivalent` — one Optional[bool] per level.
73
+
74
+ ``None`` always means "not computed" for the structural levels, and
75
+ "could not decide within budget" for ``logically_equivalent``.
76
+
77
+ Fields:
78
+
79
+ ``equivalent``
80
+ the headline verdict — ``True`` iff the requested method (or, for
81
+ ``auto``, any level of the ladder) established equivalence; ``False``
82
+ iff the strongest level that ran refuted it; ``None`` if undecided.
83
+ Truthiness follows this field.
84
+ ``method_used``
85
+ the level that produced the headline verdict. Almost always one of
86
+ ``"exact"`` / ``"canonical"`` / ``"predicate_align"`` / ``"solver"``
87
+ (matching the ``method`` argument's own vocabulary); the one
88
+ exception is ``"solver_modulo_converses"`` — set instead of
89
+ ``"solver"`` exactly when the caller passed a non-empty
90
+ ``converses`` AND the solver level ran, regardless of the tri-state
91
+ outcome, so a consumer can always tell a plain solver verdict from
92
+ one that relied on caller-declared converse axioms (see
93
+ :func:`equivalent`'s ``converses`` parameter and
94
+ :mod:`unicode_logic_kit.eval.converses`).
95
+ ``syntax_equal`` / ``structurally_equal`` / ``aligned_equal``
96
+ the three rungs of the structural ladder: ``==``, canonical form,
97
+ and alignment followed by canonical form.
98
+ ``logically_equivalent``
99
+ the solver's tri-state verdict.
100
+ ``counterexample``
101
+ THE COUNTERMODEL, as an accessible, documented, JSON-able field —
102
+ set exactly when the solver stage REFUTED equivalence (i.e. whenever
103
+ ``equivalent is False``; for ``method="solver"``/``"auto"`` that is
104
+ the same moment ``logically_equivalent is False``), ``None`` in
105
+ every other case (proved, undecided, or a structural level never
106
+ reached the solver). A caller checks it with
107
+ ``if result.counterexample is not None: ...`` — no separate lookup
108
+ or solver re-invocation needed; the witness that falsified
109
+ equivalence is already attached to the result that reports the
110
+ refutation. Two shapes, keyed by ``"kind"``:
111
+ ``{"kind": "z3_model", "assignment": {var_name: value_repr, …}}``
112
+ for classical formulas (every declared Z3 symbol's value under the
113
+ model that satisfies ``¬(φ ↔ ψ)``), and
114
+ ``{"kind": "kripke", "repr": …}`` for modal ones (the ``repr()`` of
115
+ the Kripke countermodel :func:`~unicode_logic_kit.atp.modal_tableau.modal_countermodel`
116
+ found). Both are plain JSON-compatible dicts of strings, so the
117
+ whole ``EquivalenceResult`` — countermodel included — survives a
118
+ ``pickle`` round-trip and a process boundary undisturbed (it is a
119
+ frozen dataclass over ``Optional[bool]``/``str``/``dict`` fields
120
+ only), which matters for a caller that evaluates candidates in a
121
+ worker pool and inspects failures back in the parent process.
122
+ ``reason``
123
+ why the solver level gave no verdict, when it REFUSED the input: the
124
+ refusal's own text (a family with no first-order image, a numeral and a
125
+ constant that are one symbol, a construct the modal route does not
126
+ decide), so that ``equivalent is None`` can be told from "undecided within
127
+ the budget". ``None`` in every other case: a verdict, an undecided search,
128
+ or a call that never reached the solver.
129
+ ``partial_credit``
130
+ a heuristic RANKING signal in ``{0.0, 0.25, 0.5, 0.75, 1.0}`` —
131
+ explicitly NOT a probability of correctness, just a coarse ordinal
132
+ score for sorting or averaging predictions when the headline verdict
133
+ is not a clean ``True``. The rubric: ``equivalent is True`` at ANY
134
+ level of the ladder scores ``1.0`` (the headline already gives the
135
+ strongest possible signal, so the sub-component breakdown is not
136
+ computed and ``partial_credit_components`` stays ``None``); otherwise
137
+ the score is the sum of four independent 0.25-weighted binary checks,
138
+ see ``partial_credit_components``.
139
+
140
+ Only computed for ``method="auto"`` and ``method="solver"``: the
141
+ rubric's ``s4`` component needs the solver's tri-state verdict, which
142
+ the purely structural methods (``exact`` / ``canonical`` /
143
+ ``predicate_align``) never obtain — for those, ``partial_credit``
144
+ stays ``None`` (an honest "not computed", never a score of 0).
145
+ ``partial_credit_components``
146
+ the four binary sub-checks behind a summed (non-``1.0``,
147
+ non-``None``) ``partial_credit``, as a JSON-able
148
+ ``{"s1": bool, "s2": bool, "s3": bool, "s4": bool}``. ``None``
149
+ whenever ``partial_credit`` itself is ``None`` or ``1.0`` (see
150
+ above). The four keys:
151
+
152
+ * ``s1`` "wellformed" — BOTH prediction and reference are
153
+ ``eval.validate.is_wellformed`` (closed, arity-consistent,
154
+ lambda-free).
155
+ * ``s2`` "vocabulary_alignable" — after
156
+ ``align_symbols(prediction, reference)`` the aligned prediction's
157
+ symbol inventories (predicate name/arity, function name/arity,
158
+ constant names) equal the reference's.
159
+ * ``s3`` "aligned_equal" — ``aligned_exact_match`` holds of the pair.
160
+ This practically implies ``s2`` (a successful vocabulary alignment
161
+ is a prerequisite for the canonical match that follows it) but is
162
+ tracked as an independent bit because it is the STRICTLY stronger
163
+ condition — matching vocabularies is necessary but not sufficient
164
+ for the aligned canonical forms to agree.
165
+ * ``s4`` "logically_equivalent" — the solver's tri-state verdict is
166
+ True; both ``None`` (undecided) and ``False`` (refuted) count as 0,
167
+ per the tri-state discipline the rest of this module enforces (see
168
+ the module docstring). Note: for the two methods that ever reach
169
+ the summed branch, the headline verdict IS the solver verdict, so
170
+ by the time ``s1``–``s4`` are computed that verdict has already
171
+ failed to be ``True`` — meaning ``s4`` is always ``False`` in every
172
+ case this field is actually populated today. It is still evaluated
173
+ against the real verdict (never hard-coded) so it stays correct
174
+ should a future call site ever decouple the two.
175
+ """
176
+
177
+ equivalent: Optional[bool]
178
+ method_used: str
179
+ syntax_equal: Optional[bool] = None
180
+ structurally_equal: Optional[bool] = None
181
+ aligned_equal: Optional[bool] = None
182
+ logically_equivalent: Optional[bool] = None
183
+ counterexample: Optional[dict] = None
184
+ partial_credit: Optional[float] = None
185
+ partial_credit_components: Optional[dict] = None
186
+ reason: Optional[str] = None
187
+
188
+ def __bool__(self) -> bool:
189
+ return self.equivalent is True
190
+
191
+ def to_dict(self) -> dict:
192
+ """Serialise to a JSON-compatible dict (all fields, field names as keys)."""
193
+ return {
194
+ "equivalent": self.equivalent,
195
+ "method_used": self.method_used,
196
+ "syntax_equal": self.syntax_equal,
197
+ "structurally_equal": self.structurally_equal,
198
+ "aligned_equal": self.aligned_equal,
199
+ "logically_equivalent": self.logically_equivalent,
200
+ "counterexample": self.counterexample,
201
+ "partial_credit": self.partial_credit,
202
+ "partial_credit_components": self.partial_credit_components,
203
+ "reason": self.reason,
204
+ }
205
+
206
+
207
+ def _has_object_quantifier(node: Node) -> bool:
208
+ """True iff ``node`` binds a first-order object variable anywhere."""
209
+ return any(isinstance(n, (Quantifier, SortedQuantifier)) for n in node.walk())
210
+
211
+
212
+ def _solver_tristate(f1: Node, f2: Node, timeout: int, frame: str, systems,
213
+ converse_axioms: tuple = ()):
214
+ """Return ``(verdict, counterexample, reason)`` for genuine logical equivalence.
215
+
216
+ ``reason`` is ``None`` unless the solver level REFUSED the input (a
217
+ ``NotImplementedError`` from the translation, which says why: a family with no
218
+ first-order image, a numeral and a constant that are one symbol): then it is the
219
+ text of the refusal and the verdict is ``None``.
220
+
221
+ Classical route: Z3 on ``¬(φ ↔ ψ)`` — ``unsat`` proves equivalence, ``sat``
222
+ refutes it (the model is the counterexample), ``unknown`` stays ``None``.
223
+ Many-sorted input: ``sort_axioms(f1, f2)`` — every sort of either formula
224
+ is non-empty, and every sorted constant ``c:S`` of either formula is in
225
+ ``S`` — is asserted as extra, UNNEGATED premises alongside ``¬(φ ↔ ψ)``:
226
+ the same soundness fix :mod:`unicode_logic_kit.atp.z3_models`/
227
+ ``atp.protocol.Z3Backend`` apply, needed here for the identical reason.
228
+ MSFOL, by convention, never gives a sort an empty universe, and a sorted
229
+ constant denotes an element of its sort, but ``to_z3()``'s relativisation
230
+ alone carries neither guarantee (see the classical-reasoning guide's
231
+ many-sorted section). The facts of BOTH formulas are asserted: a legal
232
+ structure fixes ``c`` in ``S`` for either side. Empty for an unsorted pair,
233
+ so behaviour there is unchanged.
234
+ ``converse_axioms`` (see :mod:`unicode_logic_kit.eval.converses`) are
235
+ asserted the same way — extra, UNNEGATED premises alongside
236
+ ``¬(φ ↔ ψ)`` — before that negated goal is added, so ``solver.add`` sees
237
+ every premise (sort facts AND converse) ahead of the goal it bridges.
238
+ Empty by default, so behaviour is unchanged unless a caller opts in.
239
+ Modal route: ``modal_decide(Iff(φ, ψ))`` for the propositional fragment
240
+ (tri-state; a Kripke counter-model witnesses refutation); quantified or
241
+ tableau-rejected modal formulas fall back to ``qml_equivalent`` — sound but
242
+ bounded-incomplete, so its ``False`` is reported as ``None`` (not proven),
243
+ never as a refutation. ``converse_axioms`` has no modal route at all —
244
+ see the ``NotImplementedError`` below, raised before either modal branch
245
+ runs.
246
+ """
247
+ from unicode_logic_kit.atp.modal_tableau import has_modal
248
+
249
+ iff = Iff(f1, f2)
250
+
251
+ if has_modal(iff):
252
+ if converse_axioms:
253
+ raise NotImplementedError(
254
+ "equivalent: converses is not supported for modal formulas "
255
+ "-- declared converse bridging axioms are only honoured by "
256
+ "the classical/MSFOL Z3 route")
257
+ from unicode_logic_kit.fol.qml import qml_equivalent
258
+
259
+ if not _has_object_quantifier(iff):
260
+ from unicode_logic_kit.atp.modal_tableau import (
261
+ modal_decide, modal_countermodel,
262
+ )
263
+ try:
264
+ status = modal_decide(iff, frame=frame, systems=systems)
265
+ except NotImplementedError:
266
+ status = None # fragment the tableau rejects
267
+ if status == "valid":
268
+ return True, None, None
269
+ if status == "invalid":
270
+ cm = modal_countermodel(iff, frame=frame, systems=systems)
271
+ witness = {"kind": "kripke", "repr": repr(cm)} if cm is not None else None
272
+ return False, witness, None
273
+ # "unknown" or rejected: fall through to the QML embedding.
274
+ try:
275
+ proved = qml_equivalent(f1, f2, frame=frame, systems=systems,
276
+ timeout=timeout)
277
+ except NotImplementedError as exc:
278
+ return None, None, str(exc)
279
+ return (True, None, None) if proved else (None, None, None)
280
+
281
+ # Classical route: tri-state Z3 (deliberately NOT formulas_are_equivalent,
282
+ # which collapses unknown to False — unusable as a metric verdict).
283
+ from z3 import Solver, Not as _ZNot, sat, unsat
284
+
285
+ from unicode_logic_kit.fol.nodes import Z3Env
286
+
287
+ try:
288
+ env = Z3Env() # one environment for both formulas and every extra assertion
289
+ phi, psi = f1.to_z3(env), f2.to_z3(env)
290
+ sort_facts = [axiom.to_z3(env) for axiom in sort_axioms(f1, f2)]
291
+ converses_z3 = [axiom.to_z3(env) for axiom in converse_axioms]
292
+ except NotImplementedError as exc:
293
+ return None, None, str(exc) # no Z3 image for this input: the refusal says why
294
+ solver = Solver()
295
+ solver.set("timeout", timeout)
296
+ solver.set("random_seed", 42)
297
+ for axiom in sort_facts:
298
+ solver.add(axiom)
299
+ for axiom in converses_z3:
300
+ solver.add(axiom)
301
+ solver.add(_ZNot(phi == psi))
302
+ res = solver.check()
303
+ if res == unsat:
304
+ return True, None, None
305
+ if res == sat:
306
+ from unicode_logic_kit.atp.z3_models import model_assignment
307
+ assignment = model_assignment(solver.model())
308
+ return False, {"kind": "z3_model", "assignment": assignment}, None
309
+ return None, None, None
310
+
311
+
312
+ def _partial_credit(prediction: Node, reference: Node, verdict: Optional[bool],
313
+ max_norm_distance: float):
314
+ """Return ``(partial_credit, partial_credit_components)`` for the rubric.
315
+
316
+ ``verdict`` is the headline equivalence verdict for the call site — which,
317
+ for both places this is invoked (``method="solver"`` and the auto ladder's
318
+ solver fallthrough), IS the solver's tri-state verdict, i.e. the same
319
+ value that ends up in ``EquivalenceResult.equivalent`` /
320
+ ``.logically_equivalent``. See the ``EquivalenceResult`` docstring for the
321
+ full rubric this implements; this function is deliberately dumb (no
322
+ control flow beyond the ``verdict is True`` short-circuit) so the rubric
323
+ stays easy to audit against that docstring.
324
+ """
325
+ if verdict is True:
326
+ return 1.0, None
327
+
328
+ from .validate import is_wellformed
329
+ from .predicate_match import align_symbols, aligned_exact_match, _symbol_inventory
330
+
331
+ s1 = is_wellformed(prediction) and is_wellformed(reference)
332
+
333
+ aligned = align_symbols(prediction, reference, max_norm_distance)
334
+ ap, af, ac = _symbol_inventory(aligned)
335
+ rp, rf, rc = _symbol_inventory(reference)
336
+ s2 = ap == rp and af == rf and ac == rc
337
+
338
+ s3 = aligned_exact_match(prediction, reference, max_norm_distance)
339
+
340
+ s4 = verdict is True # never True here; see docstring
341
+
342
+ components = {"s1": s1, "s2": s2, "s3": s3, "s4": s4}
343
+ score = 0.25 * sum(components.values())
344
+ return score, components
345
+
346
+
347
+ def equivalent(prediction: Node, reference: Node, *, method: str = "auto",
348
+ timeout: int = 10000, max_norm_distance: float = 0.6,
349
+ frame: str = "K", systems=None,
350
+ converses=None) -> EquivalenceResult:
351
+ """Decide whether ``prediction`` and ``reference`` are equivalent.
352
+
353
+ Args:
354
+ prediction / reference: parsed formulas (any family the requested
355
+ level supports; the solver level covers classical FOL/MSFOL and
356
+ the modal family).
357
+ method: one of ``"exact"``, ``"canonical"``, ``"predicate_align"``,
358
+ ``"solver"``, ``"auto"``. ``auto`` runs the ladder cheapest-first
359
+ and stops at the first ``True``; the solver only runs when the
360
+ structural levels all fail.
361
+ timeout: solver budget in milliseconds.
362
+ max_norm_distance: threshold for the ``predicate_align`` level (see
363
+ :func:`~unicode_logic_kit.eval.predicate_match.align_symbols`).
364
+ frame / systems: modal frame class and per-family systems, forwarded
365
+ to the modal deciders; ignored for classical formulas.
366
+ converses: an OPT-IN, default-``None`` sequence of
367
+ :data:`~unicode_logic_kit.eval.converses.ConverseDeclaration`
368
+ (``(a_key, b_key, permutation)`` triples — see that module) —
369
+ e.g. ``[(("LovedBy", 2), ("Loves", 2), (1, 0))]`` to bridge
370
+ ``LovedBy(x, y) ↔ Loves(y, x)``. ``None``/empty is a strict no-op
371
+ (byte-identical behaviour to calling without this argument).
372
+ Non-empty requires ``method`` in ``{"solver", "auto"}`` —
373
+ ``ValueError`` immediately otherwise, since the structural levels
374
+ never reach the solver and would silently ignore a declared
375
+ axiom instead of honouring it. When the solver level runs with
376
+ non-empty ``converses``, the axioms are asserted as extra
377
+ premises and the result's ``method_used`` is
378
+ ``"solver_modulo_converses"`` instead of ``"solver"`` — its own
379
+ category, never merged into plain solver verdicts. A modal
380
+ formula pair with non-empty ``converses`` raises
381
+ :class:`NotImplementedError` (no modal bridging route exists).
382
+
383
+ Returns:
384
+ An :class:`EquivalenceResult`. Note the asymmetry of the levels: the
385
+ structural levels can only *establish* equivalence (their ``False``
386
+ just means "this quotient did not close the gap"), so only the solver
387
+ level can make the headline verdict ``False`` — and only with a
388
+ counterexample in hand.
389
+ """
390
+ if method not in _METHODS:
391
+ raise ValueError(f"equivalent: unknown method {method!r} (use one of {_METHODS})")
392
+
393
+ from .canonical import exact_match
394
+ from .predicate_match import aligned_exact_match
395
+
396
+ axioms: tuple = ()
397
+ if converses:
398
+ if method in ("exact", "canonical", "predicate_align"):
399
+ raise ValueError(
400
+ f"equivalent: converses requires method in "
401
+ f"{{'solver', 'auto'}} (got {method!r}) — a declared "
402
+ "converse axiom is honoured only by the solver level; a "
403
+ "structural method would silently ignore it")
404
+ from .converses import converse_axioms as _converse_axioms
405
+ axioms = _converse_axioms(converses)
406
+
407
+ if method == "exact":
408
+ eq = prediction == reference
409
+ return EquivalenceResult(
410
+ equivalent=True if eq else None, method_used="exact", syntax_equal=eq)
411
+
412
+ if method == "canonical":
413
+ eq = exact_match(prediction, reference)
414
+ return EquivalenceResult(
415
+ equivalent=True if eq else None, method_used="canonical",
416
+ structurally_equal=eq)
417
+
418
+ if method == "predicate_align":
419
+ eq = aligned_exact_match(prediction, reference, max_norm_distance)
420
+ return EquivalenceResult(
421
+ equivalent=True if eq else None, method_used="predicate_align",
422
+ aligned_equal=eq)
423
+
424
+ if method == "solver":
425
+ verdict, cex, reason = _solver_tristate(prediction, reference, timeout, frame,
426
+ systems, axioms)
427
+ pc, components = _partial_credit(prediction, reference, verdict, max_norm_distance)
428
+ method_used = "solver_modulo_converses" if axioms else "solver"
429
+ return EquivalenceResult(
430
+ equivalent=verdict, method_used=method_used,
431
+ logically_equivalent=verdict, counterexample=cex,
432
+ partial_credit=pc, partial_credit_components=components, reason=reason)
433
+
434
+ # auto: cheapest first, stop at the first True; solver decides the rest.
435
+ # Every branch below computes partial_credit (method="auto" is one of the
436
+ # two rubric-eligible methods) — the early-True branches all score 1.0
437
+ # per the rubric's headline short-circuit.
438
+ syntax_equal = prediction == reference
439
+ if syntax_equal:
440
+ return EquivalenceResult(
441
+ equivalent=True, method_used="exact", syntax_equal=True,
442
+ partial_credit=1.0)
443
+
444
+ structurally_equal = exact_match(prediction, reference)
445
+ if structurally_equal:
446
+ return EquivalenceResult(
447
+ equivalent=True, method_used="canonical",
448
+ syntax_equal=False, structurally_equal=True,
449
+ partial_credit=1.0)
450
+
451
+ aligned_equal = aligned_exact_match(prediction, reference, max_norm_distance)
452
+ if aligned_equal:
453
+ return EquivalenceResult(
454
+ equivalent=True, method_used="predicate_align",
455
+ syntax_equal=False, structurally_equal=False, aligned_equal=True,
456
+ partial_credit=1.0)
457
+
458
+ verdict, cex, reason = _solver_tristate(prediction, reference, timeout, frame,
459
+ systems, axioms)
460
+ pc, components = _partial_credit(prediction, reference, verdict, max_norm_distance)
461
+ method_used = "solver_modulo_converses" if axioms else "solver"
462
+ return EquivalenceResult(
463
+ equivalent=verdict, method_used=method_used,
464
+ syntax_equal=False, structurally_equal=False, aligned_equal=False,
465
+ logically_equivalent=verdict, counterexample=cex,
466
+ partial_credit=pc, partial_credit_components=components, reason=reason)