unicode-logic-kit 0.31.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (237) hide show
  1. unicode_logic_kit/__init__.py +385 -0
  2. unicode_logic_kit/__main__.py +520 -0
  3. unicode_logic_kit/_deadline.py +219 -0
  4. unicode_logic_kit/ace/__init__.py +126 -0
  5. unicode_logic_kit/ace/_align.py +135 -0
  6. unicode_logic_kit/ace/chem_lexicon.py +128 -0
  7. unicode_logic_kit/ace/drs_reader.py +570 -0
  8. unicode_logic_kit/ace/mapping.py +666 -0
  9. unicode_logic_kit/ace/reverse_modal.py +138 -0
  10. unicode_logic_kit/ace/runner.py +551 -0
  11. unicode_logic_kit/ace/translate.py +452 -0
  12. unicode_logic_kit/ace/verbalize.py +1070 -0
  13. unicode_logic_kit/api.py +1284 -0
  14. unicode_logic_kit/atp/__init__.py +177 -0
  15. unicode_logic_kit/atp/_ascii_names.py +113 -0
  16. unicode_logic_kit/atp/_html.py +72 -0
  17. unicode_logic_kit/atp/_substructural_input.py +228 -0
  18. unicode_logic_kit/atp/_tff_problem.py +715 -0
  19. unicode_logic_kit/atp/_tptp_problem.py +1111 -0
  20. unicode_logic_kit/atp/_writer_support.py +289 -0
  21. unicode_logic_kit/atp/clingo_backend.py +1180 -0
  22. unicode_logic_kit/atp/cvc5_backend.py +1385 -0
  23. unicode_logic_kit/atp/eprover_backend.py +732 -0
  24. unicode_logic_kit/atp/finite_domain.py +1055 -0
  25. unicode_logic_kit/atp/fitch.py +1547 -0
  26. unicode_logic_kit/atp/fitch_search.py +551 -0
  27. unicode_logic_kit/atp/hets_backend.py +339 -0
  28. unicode_logic_kit/atp/hybrid_down.py +120 -0
  29. unicode_logic_kit/atp/incremental.py +250 -0
  30. unicode_logic_kit/atp/kripke_enum.py +741 -0
  31. unicode_logic_kit/atp/lambek.py +436 -0
  32. unicode_logic_kit/atp/leo3_backend.py +332 -0
  33. unicode_logic_kit/atp/linear.py +738 -0
  34. unicode_logic_kit/atp/lj.py +705 -0
  35. unicode_logic_kit/atp/logic_backends.py +566 -0
  36. unicode_logic_kit/atp/ltl_tableau.py +1084 -0
  37. unicode_logic_kit/atp/minizinc_backend.py +1402 -0
  38. unicode_logic_kit/atp/modal_tableau.py +1382 -0
  39. unicode_logic_kit/atp/nanocop_backend.py +410 -0
  40. unicode_logic_kit/atp/portfolio.py +489 -0
  41. unicode_logic_kit/atp/protocol.py +1803 -0
  42. unicode_logic_kit/atp/prover9_entailment.py +1153 -0
  43. unicode_logic_kit/atp/resolution.py +1376 -0
  44. unicode_logic_kit/atp/resolution_check.py +1114 -0
  45. unicode_logic_kit/atp/sequent.py +1050 -0
  46. unicode_logic_kit/atp/tableau.py +921 -0
  47. unicode_logic_kit/atp/tableau_check.py +543 -0
  48. unicode_logic_kit/atp/tptp_ncl.py +811 -0
  49. unicode_logic_kit/atp/tptp_tff.py +1546 -0
  50. unicode_logic_kit/atp/tstp.py +1333 -0
  51. unicode_logic_kit/atp/tstp_check.py +1096 -0
  52. unicode_logic_kit/atp/twee_backend.py +236 -0
  53. unicode_logic_kit/atp/twee_check.py +711 -0
  54. unicode_logic_kit/atp/twee_entailment.py +953 -0
  55. unicode_logic_kit/atp/vampire_entailment.py +540 -0
  56. unicode_logic_kit/atp/z3_arith.py +470 -0
  57. unicode_logic_kit/atp/z3_equivalence.py +36 -0
  58. unicode_logic_kit/atp/z3_fuzzy.py +362 -0
  59. unicode_logic_kit/atp/z3_input.py +500 -0
  60. unicode_logic_kit/atp/z3_models.py +208 -0
  61. unicode_logic_kit/chem/__init__.py +88 -0
  62. unicode_logic_kit/chem/_naming.py +284 -0
  63. unicode_logic_kit/chem/cache.py +185 -0
  64. unicode_logic_kit/chem/interop.py +244 -0
  65. unicode_logic_kit/chem/mol.py +525 -0
  66. unicode_logic_kit/chem/signature.py +112 -0
  67. unicode_logic_kit/comorphism.py +497 -0
  68. unicode_logic_kit/dl/__init__.py +384 -0
  69. unicode_logic_kit/dl/classification.py +227 -0
  70. unicode_logic_kit/dl/concepts.py +632 -0
  71. unicode_logic_kit/dl/datatypes.py +818 -0
  72. unicode_logic_kit/dl/owl_functional.py +2433 -0
  73. unicode_logic_kit/dl/owl_manchester.py +1637 -0
  74. unicode_logic_kit/dl/owl_reasoner.py +790 -0
  75. unicode_logic_kit/dl/parser.py +391 -0
  76. unicode_logic_kit/dl/tableau.py +4048 -0
  77. unicode_logic_kit/dl/translate.py +2704 -0
  78. unicode_logic_kit/drt/__init__.py +94 -0
  79. unicode_logic_kit/drt/export.py +179 -0
  80. unicode_logic_kit/drt/nodes.py +506 -0
  81. unicode_logic_kit/drt/parser.py +965 -0
  82. unicode_logic_kit/drt/resolve.py +195 -0
  83. unicode_logic_kit/drt/reverse.py +175 -0
  84. unicode_logic_kit/eval/__init__.py +106 -0
  85. unicode_logic_kit/eval/batch.py +382 -0
  86. unicode_logic_kit/eval/canonical.py +663 -0
  87. unicode_logic_kit/eval/chem_batch.py +606 -0
  88. unicode_logic_kit/eval/converses.py +200 -0
  89. unicode_logic_kit/eval/datasets/__init__.py +136 -0
  90. unicode_logic_kit/eval/datasets/_base.py +263 -0
  91. unicode_logic_kit/eval/datasets/_proofwriter_proof.py +422 -0
  92. unicode_logic_kit/eval/datasets/c3po.py +678 -0
  93. unicode_logic_kit/eval/datasets/folio.py +158 -0
  94. unicode_logic_kit/eval/datasets/fracas.py +418 -0
  95. unicode_logic_kit/eval/datasets/groves.py +191 -0
  96. unicode_logic_kit/eval/datasets/logicbench.py +467 -0
  97. unicode_logic_kit/eval/datasets/logicnli.py +303 -0
  98. unicode_logic_kit/eval/datasets/malls.py +133 -0
  99. unicode_logic_kit/eval/datasets/pfolio.py +594 -0
  100. unicode_logic_kit/eval/datasets/pmb.py +242 -0
  101. unicode_logic_kit/eval/datasets/prontoqa.py +611 -0
  102. unicode_logic_kit/eval/datasets/proofwriter.py +1431 -0
  103. unicode_logic_kit/eval/datasets/proverqa.py +674 -0
  104. unicode_logic_kit/eval/datasets/willow.py +478 -0
  105. unicode_logic_kit/eval/equivalence.py +466 -0
  106. unicode_logic_kit/eval/exercise_gen.py +533 -0
  107. unicode_logic_kit/eval/explain.py +791 -0
  108. unicode_logic_kit/eval/generality.py +750 -0
  109. unicode_logic_kit/eval/metric_hf.py +458 -0
  110. unicode_logic_kit/eval/predicate_match.py +343 -0
  111. unicode_logic_kit/eval/theory_check.py +1170 -0
  112. unicode_logic_kit/eval/validate.py +306 -0
  113. unicode_logic_kit/fol/__init__.py +177 -0
  114. unicode_logic_kit/fol/_atom_keys.py +510 -0
  115. unicode_logic_kit/fol/_fol_nodes.py +3586 -0
  116. unicode_logic_kit/fol/_free_parameters.py +105 -0
  117. unicode_logic_kit/fol/_ho_nodes.py +448 -0
  118. unicode_logic_kit/fol/_hybrid_nodes.py +308 -0
  119. unicode_logic_kit/fol/_identifiers.py +1091 -0
  120. unicode_logic_kit/fol/_lambek_nodes.py +112 -0
  121. unicode_logic_kit/fol/_linear_nodes.py +352 -0
  122. unicode_logic_kit/fol/_modal_nodes.py +1467 -0
  123. unicode_logic_kit/fol/_msfl_nodes.py +2196 -0
  124. unicode_logic_kit/fol/_numeral_symbols.py +231 -0
  125. unicode_logic_kit/fol/_so_nodes.py +200 -0
  126. unicode_logic_kit/fol/_symbol_names.py +81 -0
  127. unicode_logic_kit/fol/_team_nodes.py +181 -0
  128. unicode_logic_kit/fol/_tptp_symbols.py +551 -0
  129. unicode_logic_kit/fol/_truth_constants.py +117 -0
  130. unicode_logic_kit/fol/casl_export.py +1135 -0
  131. unicode_logic_kit/fol/casl_import.py +929 -0
  132. unicode_logic_kit/fol/derivation.py +367 -0
  133. unicode_logic_kit/fol/dialect_detect.py +70 -0
  134. unicode_logic_kit/fol/dialect_repair.py +537 -0
  135. unicode_logic_kit/fol/frames.py +637 -0
  136. unicode_logic_kit/fol/grammars/terminals.lark +31 -0
  137. unicode_logic_kit/fol/lambda_tools.py +297 -0
  138. unicode_logic_kit/fol/latex_input.py +429 -0
  139. unicode_logic_kit/fol/modal_translation.py +944 -0
  140. unicode_logic_kit/fol/msflparser.py +1033 -0
  141. unicode_logic_kit/fol/naming.py +422 -0
  142. unicode_logic_kit/fol/nodes.py +241 -0
  143. unicode_logic_kit/fol/normalforms.py +492 -0
  144. unicode_logic_kit/fol/pal.py +287 -0
  145. unicode_logic_kit/fol/prolog_export.py +566 -0
  146. unicode_logic_kit/fol/prolog_input.py +505 -0
  147. unicode_logic_kit/fol/prover9_input.py +1325 -0
  148. unicode_logic_kit/fol/qml.py +1760 -0
  149. unicode_logic_kit/fol/qmltp_input.py +525 -0
  150. unicode_logic_kit/fol/sanitize.py +221 -0
  151. unicode_logic_kit/fol/serialize.py +79 -0
  152. unicode_logic_kit/fol/signature.py +1290 -0
  153. unicode_logic_kit/fol/simplify_check.py +544 -0
  154. unicode_logic_kit/fol/spans.py +594 -0
  155. unicode_logic_kit/fol/tptp_input.py +1503 -0
  156. unicode_logic_kit/fol/tptp_repair.py +941 -0
  157. unicode_logic_kit/fol/unification.py +157 -0
  158. unicode_logic_kit/fol/verbalize.py +263 -0
  159. unicode_logic_kit/hets/__init__.py +163 -0
  160. unicode_logic_kit/hets/bridge.py +142 -0
  161. unicode_logic_kit/hets/client.py +748 -0
  162. unicode_logic_kit/hets/docker.py +420 -0
  163. unicode_logic_kit/hets/dol.py +712 -0
  164. unicode_logic_kit/hets/haskell_json.py +355 -0
  165. unicode_logic_kit/hets/owl_backend.py +794 -0
  166. unicode_logic_kit/hets/owl_cli.py +598 -0
  167. unicode_logic_kit/hets/symbols.py +512 -0
  168. unicode_logic_kit/hol/__init__.py +140 -0
  169. unicode_logic_kit/hol/_ho_common.py +323 -0
  170. unicode_logic_kit/hol/_isabelle_binders.py +125 -0
  171. unicode_logic_kit/hol/classical.py +812 -0
  172. unicode_logic_kit/hol/deepshallow/__init__.py +45 -0
  173. unicode_logic_kit/hol/deepshallow/_common.py +177 -0
  174. unicode_logic_kit/hol/deepshallow/conditional.py +225 -0
  175. unicode_logic_kit/hol/deepshallow/intuitionistic.py +181 -0
  176. unicode_logic_kit/hol/deepshallow/modal.py +217 -0
  177. unicode_logic_kit/hol/deepshallow/qml.py +406 -0
  178. unicode_logic_kit/hol/deepshallow/relevant.py +206 -0
  179. unicode_logic_kit/hol/free.py +753 -0
  180. unicode_logic_kit/hol/goedel.py +336 -0
  181. unicode_logic_kit/hol/ho_modal.py +1743 -0
  182. unicode_logic_kit/hol/intuitionistic.py +403 -0
  183. unicode_logic_kit/hol/isabelle_conditional.py +593 -0
  184. unicode_logic_kit/hol/isabelle_modal.py +1908 -0
  185. unicode_logic_kit/hol/isabelle_relevant.py +412 -0
  186. unicode_logic_kit/hol/isabelle_runner.py +1147 -0
  187. unicode_logic_kit/hol/isabelle_substructural.py +884 -0
  188. unicode_logic_kit/hol/lean.py +1018 -0
  189. unicode_logic_kit/hol/manyvalued.py +921 -0
  190. unicode_logic_kit/hol/secondorder.py +687 -0
  191. unicode_logic_kit/hol/thf_modal.py +941 -0
  192. unicode_logic_kit/hol/thirdorder.py +397 -0
  193. unicode_logic_kit/ilp/__init__.py +89 -0
  194. unicode_logic_kit/ilp/readback.py +389 -0
  195. unicode_logic_kit/ilp/separation.py +153 -0
  196. unicode_logic_kit/ilp/task.py +730 -0
  197. unicode_logic_kit/logic.py +163 -0
  198. unicode_logic_kit/mcp/__init__.py +28 -0
  199. unicode_logic_kit/mcp/__main__.py +5 -0
  200. unicode_logic_kit/mcp/chem_tools.py +1031 -0
  201. unicode_logic_kit/mcp/server.py +2453 -0
  202. unicode_logic_kit/mcp/syntax_spec.py +681 -0
  203. unicode_logic_kit/prob/__init__.py +53 -0
  204. unicode_logic_kit/prob/_bdd.py +225 -0
  205. unicode_logic_kit/prob/_column_gen.py +668 -0
  206. unicode_logic_kit/prob/distribution.py +686 -0
  207. unicode_logic_kit/prob/nilsson.py +470 -0
  208. unicode_logic_kit/py.typed +0 -0
  209. unicode_logic_kit/semantics/__init__.py +137 -0
  210. unicode_logic_kit/semantics/_modal_reject.py +156 -0
  211. unicode_logic_kit/semantics/action_models.py +466 -0
  212. unicode_logic_kit/semantics/asp_models.py +1200 -0
  213. unicode_logic_kit/semantics/conditional.py +580 -0
  214. unicode_logic_kit/semantics/dynamic_epistemic.py +95 -0
  215. unicode_logic_kit/semantics/free_logic.py +913 -0
  216. unicode_logic_kit/semantics/fuzzy.py +384 -0
  217. unicode_logic_kit/semantics/fuzzy_kripke.py +442 -0
  218. unicode_logic_kit/semantics/intuitionistic.py +581 -0
  219. unicode_logic_kit/semantics/kripke.py +1139 -0
  220. unicode_logic_kit/semantics/manyvalued.py +580 -0
  221. unicode_logic_kit/semantics/matrix.py +342 -0
  222. unicode_logic_kit/semantics/model_eval.py +1135 -0
  223. unicode_logic_kit/semantics/modelfinder.py +1036 -0
  224. unicode_logic_kit/semantics/nonmonotonic.py +372 -0
  225. unicode_logic_kit/semantics/relevant.py +331 -0
  226. unicode_logic_kit/semantics/secondorder.py +657 -0
  227. unicode_logic_kit/semantics/structures.py +352 -0
  228. unicode_logic_kit/semantics/tarski.py +975 -0
  229. unicode_logic_kit/semantics/team.py +315 -0
  230. unicode_logic_kit/semantics/team_translation.py +416 -0
  231. unicode_logic_kit/semantics/thirdorder.py +358 -0
  232. unicode_logic_kit/semantics/tnorm.py +85 -0
  233. unicode_logic_kit/semantics/truthtable.py +201 -0
  234. unicode_logic_kit-0.31.0.dist-info/METADATA +333 -0
  235. unicode_logic_kit-0.31.0.dist-info/RECORD +237 -0
  236. unicode_logic_kit-0.31.0.dist-info/WHEEL +4 -0
  237. unicode_logic_kit-0.31.0.dist-info/licenses/LICENSE +21 -0
@@ -0,0 +1,458 @@
1
+ """A HuggingFace ``evaluate``-compatible NL→FOL metric.
2
+
3
+ The `evaluate <https://github.com/huggingface/evaluate>`_ library ships
4
+ metrics for classification, translation (BLEU/ROUGE/...), and generic
5
+ sequence tasks, but nothing that understands first-order logic: two
6
+ syntactically different formulas can denote the very same claim (``P ∧
7
+ Q`` / ``Q ∧ P``; ``Winner(x)`` / ``IsWinner(x)``), and a plain
8
+ string-equality metric scores that as wrong. This module plugs that gap by
9
+ wrapping :func:`unicode_logic_kit.eval.equivalence.equivalent` — the kit's
10
+ graded structural→solver equivalence ladder — as a metric.
11
+
12
+ Per-pair semantics
13
+ -------------------
14
+ Every ``(prediction, reference)`` row is scored **independently**: the two
15
+ strings are parsed on their own (via
16
+ :func:`unicode_logic_kit.api.parse_any`, which auto-detects the dialect —
17
+ the kit's unicode surface syntax, TPTP, Prover9, LaTeX, or SMT-LIB) and then
18
+ compared with :func:`~unicode_logic_kit.eval.equivalence.equivalent`. There is
19
+ no cross-row state (no shared vocabulary, no corpus-level normalisation) —
20
+ this mirrors how the kit's other evaluation entry points
21
+ (:func:`~unicode_logic_kit.eval.batch.batch_decide`) treat a batch as
22
+ independent tasks, and it means a caller can freely shuffle, subset, or
23
+ parallelise rows without changing any pair's score.
24
+
25
+ What "equivalent" means
26
+ -------------------------
27
+ See :mod:`unicode_logic_kit.eval.equivalence` for the full ladder — in
28
+ short, ``exact`` (AST ``==``) → ``canonical`` (α-renaming,
29
+ commutativity/associativity, double-negation) → ``predicate_align``
30
+ (vocabulary renaming on top of canonical) → ``solver`` (genuine logical
31
+ equivalence via Z3 or the modal decider, TRI-STATE: proved / refuted /
32
+ unknown). ``method="auto"`` (this module's default, matching
33
+ :func:`equivalent`'s own default) runs the ladder cheapest-first and stops
34
+ at the first level that proves equivalence, falling through to the solver
35
+ only when every structural level fails.
36
+
37
+ The honesty contract
38
+ ----------------------
39
+ The solver's tri-state verdict is the reason this module reports several
40
+ numbers instead of one "accuracy":
41
+
42
+ * ``equivalence_accuracy`` counts ONLY a definitive ``equivalent is True``
43
+ as correct. A refutation (``False``, with a counterexample) and an
44
+ undecided solver call (``None``, the budget/completeness limit was hit
45
+ without an answer either way) are both counted as *not correct* — but
46
+ they are NOT THE SAME THING, and collapsing them together (the defect
47
+ :func:`unicode_logic_kit.atp.formulas_are_equivalent` has, which
48
+ :func:`equivalent` exists to avoid) would silently misreport how often
49
+ the metric actually *knows* the prediction is wrong.
50
+ * ``solver_unknown_rate`` makes that hidden mass visible: the fraction of
51
+ pairs where the ladder reached the solver level (i.e. no structural level
52
+ decided the pair, so :class:`~unicode_logic_kit.eval.equivalence
53
+ .EquivalenceResult`'s ``method_used`` is ``"solver"``) and the solver
54
+ came back ``None``. It is deliberately reported alongside
55
+ ``equivalence_accuracy`` rather than folded into it, so a consumer can
56
+ compute honest bounds: the true accuracy lies in
57
+ ``[equivalence_accuracy, equivalence_accuracy + solver_unknown_rate]``.
58
+ (With ``method`` values other than ``"auto"``/``"solver"`` the solver
59
+ never runs at all, so this rate is always ``0.0`` — that is not a claim
60
+ every pair was decided, just that this particular signal has nothing to
61
+ report; see :func:`compute_fol_metrics`'s docstring.)
62
+ * A parse failure on either side of a pair is scored ``0.0`` for that pair
63
+ (it cannot be "equivalent" to anything, having never become a formula)
64
+ and is tallied into ``parse_failure_rate`` — never silently dropped
65
+ from the batch (which would inflate every other rate by shrinking the
66
+ denominator) and never counted as a solver "unknown" (it never reached
67
+ the solver, or any level of the ladder, at all).
68
+ """
69
+
70
+ import importlib.util
71
+ from typing import List
72
+
73
+ from .equivalence import equivalent
74
+
75
+ __all__ = ["compute_fol_metrics", "FolEquivalence", "load"]
76
+
77
+
78
+ # ---------------------------------------------------------------------------
79
+ # compute_fol_metrics -- works with or without the `evaluate` package.
80
+ # ---------------------------------------------------------------------------
81
+
82
+ def _score_pair(prediction: str, reference: str, method: str, timeout_ms: int,
83
+ converses=None) -> dict:
84
+ """Score ONE ``(prediction, reference)`` pair. Never raises.
85
+
86
+ Returns ``{"parse_failure", "syntax_equal", "equivalent_true",
87
+ "partial_credit", "solver_unknown", "method_used"}`` (all bool except
88
+ ``partial_credit``, a float, and ``method_used``, a str or ``None``). A
89
+ parse failure on either side reports every bool/float field at its floor
90
+ (``False`` / ``0.0``, ``method_used=None``) — see the module docstring's
91
+ honesty-contract section for why that floor, not a skip, is the correct
92
+ score for text that never became a formula.
93
+
94
+ ``converses`` (see :mod:`unicode_logic_kit.eval.converses`) is forwarded
95
+ to :func:`~unicode_logic_kit.eval.equivalence.equivalent` unchanged;
96
+ ``method_used`` is carried back out so :func:`compute_fol_metrics` can
97
+ compute ``converse_matched_rate`` without recomputing the verdict.
98
+
99
+ ``syntax_equal`` is computed directly as ``prediction_formula ==
100
+ reference_formula`` rather than read off
101
+ ``EquivalenceResult.syntax_equal``: that field is only populated by the
102
+ ``"exact"`` and ``"auto"`` methods (see ``equivalence.py``'s
103
+ ``equivalent()`` — the ``"canonical"``/``"predicate_align"``/``"solver"``
104
+ branches never set it), so reading it directly would silently under-count
105
+ ``exact_match`` for every other ``method`` value. Computing it ourselves
106
+ keeps ``exact_match`` well-defined for any ``method``.
107
+
108
+ ``partial_credit`` reads ``EquivalenceResult.partial_credit`` when the
109
+ ladder computed one (``method="auto"``/``"solver"`` only — see that
110
+ field's docstring) and floors to ``0.0`` otherwise: the metric cannot
111
+ report a score the ladder never computed, so "not computed" and "computed
112
+ as the minimum" must not be conflated into a fabricated high average, and
113
+ ``0.0`` is the conservative choice consistent with the parse-failure floor
114
+ above.
115
+ """
116
+ from .. import api # lazy: avoid import-time cost/cycles (mirrors eval/batch.py)
117
+
118
+ pred_parsed = api.parse_any(prediction)
119
+ ref_parsed = api.parse_any(reference)
120
+ if not pred_parsed.ok or not ref_parsed.ok:
121
+ return {
122
+ "parse_failure": True,
123
+ "syntax_equal": False,
124
+ "equivalent_true": False,
125
+ "partial_credit": 0.0,
126
+ "solver_unknown": False,
127
+ "method_used": None,
128
+ }
129
+
130
+ result = equivalent(pred_parsed.formula, ref_parsed.formula,
131
+ method=method, timeout=timeout_ms, converses=converses)
132
+
133
+ # method_used in {"solver", "solver_modulo_converses"} happens exactly
134
+ # when the ladder reached the solver level -- both for method="solver"
135
+ # directly and for the "auto" ladder's fallthrough (see equivalent()'s
136
+ # auto branch: it ALWAYS labels the solver fallthrough this way, whether
137
+ # the tri-state verdict came back True, False, or None) -- the latter
138
+ # tag only ever appears when `converses` was non-empty.
139
+ solver_unknown = (result.method_used in ("solver", "solver_modulo_converses")
140
+ and result.equivalent is None)
141
+
142
+ return {
143
+ "parse_failure": False,
144
+ "syntax_equal": pred_parsed.formula == ref_parsed.formula,
145
+ "equivalent_true": result.equivalent is True,
146
+ "partial_credit": result.partial_credit if result.partial_credit is not None else 0.0,
147
+ "solver_unknown": solver_unknown,
148
+ "method_used": result.method_used,
149
+ }
150
+
151
+
152
+ def compute_fol_metrics(predictions: List[str], references: List[str], *,
153
+ method: str = "auto", timeout_ms: int = 10000,
154
+ converses=None) -> dict:
155
+ """Score a batch of NL→FOL predictions against references, per-pair.
156
+
157
+ Args:
158
+ predictions: predicted formula strings, any dialect
159
+ :func:`unicode_logic_kit.api.parse_any` can detect.
160
+ references: reference (gold) formula strings, same dialect rules.
161
+ Must be the same length as ``predictions``.
162
+ method: forwarded to :func:`~unicode_logic_kit.eval.equivalence
163
+ .equivalent` for every pair — one of ``"exact"``,
164
+ ``"canonical"``, ``"predicate_align"``, ``"solver"``, ``"auto"``
165
+ (the default: the full ladder, cheapest level first).
166
+ timeout_ms: forwarded as ``equivalent()``'s ``timeout`` (milliseconds)
167
+ for every pair's solver call.
168
+ converses: OPT-IN, default ``None`` — forwarded unchanged to
169
+ :func:`~unicode_logic_kit.eval.equivalence.equivalent` for every
170
+ pair (see that function's ``converses`` parameter and
171
+ :mod:`unicode_logic_kit.eval.converses`). ``None``/empty leaves the
172
+ returned dict's KEY SET unchanged (required — see
173
+ ``tests/test_metric_hf.py``'s full-dict-equality asserts); a
174
+ non-empty sequence adds the ``converse_matched_rate`` key below.
175
+
176
+ Returns:
177
+ A dict with six keys (seven when ``converses`` is non-empty):
178
+
179
+ * ``exact_match`` — fraction of pairs whose parsed formulas are
180
+ AST-equal (``==``); ``0.0`` for an unparseable pair.
181
+ * ``equivalence_accuracy`` — fraction of pairs where
182
+ ``equivalent(...).equivalent is True``. See the module docstring's
183
+ honesty-contract section: this is NOT ``1 - (fraction refuted)``,
184
+ because undecided pairs are excluded from the numerator without
185
+ being counted as refuted either.
186
+ * ``mean_partial_credit`` — mean of
187
+ ``equivalent(...).partial_credit`` over all pairs, treating an
188
+ unparseable pair or a ``partial_credit`` the requested ``method``
189
+ never computes (see that field's docstring) as ``0.0``.
190
+ * ``parse_failure_rate`` — fraction of pairs where either side
191
+ failed to parse.
192
+ * ``solver_unknown_rate`` — fraction of pairs where the ladder
193
+ reached the solver level and it returned ``None`` (undecided).
194
+ Always ``0.0`` for ``method`` values that never invoke the solver
195
+ (``"exact"``/``"canonical"``/``"predicate_align"``).
196
+ * ``n`` — the batch size (``len(predictions)``).
197
+ * ``converse_matched_rate`` — ONLY present when ``converses`` is
198
+ non-empty: the fraction of pairs where the solver level ran WITH
199
+ the declared axioms (``method_used == "solver_modulo_converses"``)
200
+ AND proved equivalence (``equivalent is True``). A separately
201
+ visible, subtractable slice of ``equivalence_accuracy`` — never
202
+ folded into it silently, matching this module's honesty-contract
203
+ convention for ``solver_unknown_rate`` above.
204
+
205
+ Raises:
206
+ ValueError: ``predictions`` and ``references`` have different
207
+ lengths; a malformed ``converses`` declaration (checked
208
+ unconditionally, including for an empty batch); or a non-empty
209
+ ``converses`` combined with a ``method`` that cannot honour it
210
+ (``"exact"``/``"canonical"``/``"predicate_align"`` — also
211
+ checked unconditionally, matching :func:`~unicode_logic_kit.eval
212
+ .equivalence.equivalent`'s own check for ``n >= 1``).
213
+ """
214
+ if len(predictions) != len(references):
215
+ raise ValueError(
216
+ "compute_fol_metrics: predictions and references must have the "
217
+ f"same length (got {len(predictions)} and {len(references)})")
218
+
219
+ if converses:
220
+ # Validate unconditionally, even for an empty batch: _score_pair (the
221
+ # only other call site that would reach validate_converses, via
222
+ # equivalent()) never runs when n == 0, so without this a malformed
223
+ # declaration would silently pass through the n == 0 branch below
224
+ # instead of raising -- see this function's own documented contract
225
+ # ("Raises: ValueError ... a malformed converses declaration").
226
+ from .converses import validate_converses
227
+ validate_converses(converses)
228
+ # Mirror equivalent()'s own method-gating check (equivalence.py:
229
+ # "converses requires method in {'solver', 'auto'}") unconditionally
230
+ # too, for the same reason: that check normally runs inside
231
+ # equivalent() via _score_pair, which never executes when n == 0, so
232
+ # without this a solver-incompatible method combined with a
233
+ # (structurally valid) converses declaration would raise for n >= 1
234
+ # but silently return zeroed metrics for n == 0 -- same declaration,
235
+ # same method, different n, different behaviour. Kept as an exact
236
+ # duplicate of equivalent()'s wording (not just "the same class of
237
+ # error") so a caller sees one consistent message regardless of
238
+ # which code path raised it.
239
+ if method in ("exact", "canonical", "predicate_align"):
240
+ raise ValueError(
241
+ f"equivalent: converses requires method in "
242
+ f"{{'solver', 'auto'}} (got {method!r}) — a declared "
243
+ "converse axiom is honoured only by the solver level; a "
244
+ "structural method would silently ignore it")
245
+
246
+ n = len(predictions)
247
+ if n == 0:
248
+ result = {
249
+ "exact_match": 0.0,
250
+ "equivalence_accuracy": 0.0,
251
+ "mean_partial_credit": 0.0,
252
+ "parse_failure_rate": 0.0,
253
+ "solver_unknown_rate": 0.0,
254
+ "n": 0,
255
+ }
256
+ if converses:
257
+ result["converse_matched_rate"] = 0.0
258
+ return result
259
+
260
+ scores = [_score_pair(p, r, method, timeout_ms, converses)
261
+ for p, r in zip(predictions, references)]
262
+
263
+ result = {
264
+ "exact_match": sum(s["syntax_equal"] for s in scores) / n,
265
+ "equivalence_accuracy": sum(s["equivalent_true"] for s in scores) / n,
266
+ "mean_partial_credit": sum(s["partial_credit"] for s in scores) / n,
267
+ "parse_failure_rate": sum(s["parse_failure"] for s in scores) / n,
268
+ "solver_unknown_rate": sum(s["solver_unknown"] for s in scores) / n,
269
+ "n": n,
270
+ }
271
+ if converses:
272
+ result["converse_matched_rate"] = sum(
273
+ s["method_used"] == "solver_modulo_converses" and s["equivalent_true"]
274
+ for s in scores) / n
275
+ return result
276
+
277
+
278
+ # ---------------------------------------------------------------------------
279
+ # FolEquivalence -- the evaluate.Metric wrapper (optional dependency).
280
+ # ---------------------------------------------------------------------------
281
+ #
282
+ # `evaluate` is NOT a hard dependency of this module: importing metric_hf.py
283
+ # must succeed on a machine that never installed it (mirroring how
284
+ # atp/cvc5_backend.py's Cvc5Backend stays importable without cvc5 -- see that
285
+ # module's docstring). Only *instantiating* FolEquivalence requires it -- and,
286
+ # unlike the eager `try: import evaluate / except ImportError` this used to
287
+ # do, that import must not happen just because metric_hf.py itself was
288
+ # imported (`evaluate` drags in `datasets`, together well over a second of
289
+ # import time -- see the module docstring's own claim, which this section
290
+ # makes actually true). `_HAS_EVALUATE` below is pure discovery -- no import
291
+ # -- mirroring atp/cvc5_backend.py's `Cvc5Backend.available()`.
292
+
293
+ _HAS_EVALUATE = (importlib.util.find_spec("evaluate") is not None and
294
+ importlib.util.find_spec("datasets") is not None)
295
+
296
+
297
+ _INSTALL_HINT = (
298
+ "FolEquivalence requires the optional 'evaluate' package "
299
+ "(pip install evaluate) -- it is not installed in this environment. "
300
+ "unicode_logic_kit.eval.metric_hf.compute_fol_metrics(predictions, "
301
+ "references) gives the same scores without that dependency."
302
+ )
303
+
304
+ _DESCRIPTION = (
305
+ "Graded NL→FOL equivalence: scores each (prediction, reference) pair "
306
+ "with unicode_logic_kit's structural→solver equivalence ladder "
307
+ "(unicode_logic_kit.eval.equivalence.equivalent). Reports exact_match, "
308
+ "equivalence_accuracy (solver-tri-state-aware -- undecided pairs are "
309
+ "counted as neither correct nor refuted), mean_partial_credit, "
310
+ "parse_failure_rate, and solver_unknown_rate (the undecided mass, kept "
311
+ "separate from equivalence_accuracy on purpose -- see "
312
+ "unicode_logic_kit.eval.metric_hf's module docstring)."
313
+ )
314
+
315
+ _KWARGS_DESCRIPTION = """
316
+ Args:
317
+ predictions (list of str): predicted FOL formula strings.
318
+ references (list of str): reference (gold) FOL formula strings, same
319
+ length as `predictions`.
320
+ method (str, optional): equivalence-ladder level, one of "exact",
321
+ "canonical", "predicate_align", "solver", "auto" (default).
322
+ timeout_ms (int, optional): per-pair solver budget in milliseconds
323
+ (default 10000).
324
+
325
+ Returns:
326
+ exact_match (float): fraction of pairs with AST-equal parsed formulas.
327
+ equivalence_accuracy (float): fraction of pairs the ladder proved
328
+ equivalent (a definitive True only -- see the module docstring).
329
+ mean_partial_credit (float): mean heuristic partial-credit score.
330
+ parse_failure_rate (float): fraction of pairs where either side failed
331
+ to parse.
332
+ solver_unknown_rate (float): fraction of pairs where the solver was
333
+ reached and returned an undecided verdict.
334
+ n (int): batch size.
335
+
336
+ Examples:
337
+ >>> import unicode_logic_kit.eval.metric_hf as metric_hf
338
+ >>> m = metric_hf.load()
339
+ >>> m.compute(predictions=["P(a) ∧ Q(a)"], references=["Q(a) ∧ P(a)"])
340
+ {'exact_match': 0.0, 'equivalence_accuracy': 1.0, ...}
341
+ """
342
+
343
+ # `FolEquivalence` itself is built lazily: the class body subclasses
344
+ # `evaluate.Metric`, so merely *defining* it needs `evaluate` (and, for
345
+ # `_info`'s `datasets.Features`, `datasets`) already imported. Building it
346
+ # eagerly at module scope -- even behind an `if _HAS_EVALUATE:` -- would
347
+ # import both packages the moment metric_hf.py is imported, which is
348
+ # exactly the cost this module's docstring already claims does NOT happen.
349
+ # `_build_fol_equivalence_class` does the real import (and only it pays that
350
+ # cost, once, on first use) and caches the resulting class in
351
+ # `_fol_equivalence_class`; the module-level `__getattr__` below (PEP 562)
352
+ # routes both `metric_hf.FolEquivalence` and
353
+ # `from ... import FolEquivalence` through it transparently.
354
+
355
+ _fol_equivalence_class = None
356
+
357
+
358
+ def _build_fol_equivalence_class():
359
+ """Import ``evaluate``/``datasets`` and return the real ``FolEquivalence``
360
+ class, building it once and caching it for every later call.
361
+
362
+ Raises:
363
+ ImportError: with the install hint, if ``evaluate``/``datasets`` are
364
+ not installed (or fail to import for any other reason) --
365
+ mirroring the message the pre-lazy fallback stub used to raise
366
+ from ``FolEquivalence.__init__``.
367
+ """
368
+ global _fol_equivalence_class
369
+ if _fol_equivalence_class is not None:
370
+ return _fol_equivalence_class
371
+
372
+ try:
373
+ import evaluate as _evaluate
374
+ import datasets as _hf_datasets
375
+ except ImportError:
376
+ raise ImportError(_INSTALL_HINT) from None
377
+
378
+ class FolEquivalence(_evaluate.Metric):
379
+ """``evaluate.Metric`` wrapper around :func:`compute_fol_metrics`.
380
+
381
+ A thin adapter: ``_compute`` delegates entirely to
382
+ :func:`compute_fol_metrics`, so this class adds nothing but the
383
+ ``evaluate``/``datasets`` plumbing (feature schema, description,
384
+ the ``evaluate.load``-shaped ``load()`` entry point below) around
385
+ logic that already works standalone. See the module docstring for
386
+ what the returned numbers mean.
387
+ """
388
+
389
+ def _info(self):
390
+ # Returns evaluate.MetricInfo. Deliberately NOT annotated with a
391
+ # `-> "_evaluate.MetricInfo"` string forward reference: `_evaluate`
392
+ # is a local variable of the enclosing `_build_fol_equivalence_class`
393
+ # (that is what keeps the import lazy -- see the section comment
394
+ # above), not a module global, so `typing.get_type_hints()` would
395
+ # raise NameError trying to resolve it. Nothing in this codebase
396
+ # or its test suite calls `get_type_hints` on this method; the
397
+ # type is documented here in prose instead.
398
+ return _evaluate.MetricInfo(
399
+ description=_DESCRIPTION,
400
+ citation="",
401
+ inputs_description=_KWARGS_DESCRIPTION,
402
+ features=_hf_datasets.Features({
403
+ "predictions": _hf_datasets.Value("string", id="sequence"),
404
+ "references": _hf_datasets.Value("string", id="sequence"),
405
+ }),
406
+ reference_urls=[
407
+ "https://unicode-logic-kit.readthedocs.io/",
408
+ ],
409
+ )
410
+
411
+ def _compute(self, predictions, references, method: str = "auto",
412
+ timeout_ms: int = 10000) -> dict:
413
+ return compute_fol_metrics(list(predictions), list(references),
414
+ method=method, timeout_ms=timeout_ms)
415
+
416
+ # Built inside a function, so Python's default __qualname__ would be
417
+ # "_build_fol_equivalence_class.<locals>.FolEquivalence" -- reset it to
418
+ # the plain module-level name so anything that reads it (repr, pickling
419
+ # by reference, which resolves "module.qualname" via getattr and so
420
+ # works fine through __getattr__ below) sees the same identifier a
421
+ # module-scoped class definition would have had.
422
+ FolEquivalence.__qualname__ = "FolEquivalence"
423
+
424
+ _fol_equivalence_class = FolEquivalence
425
+ return _fol_equivalence_class
426
+
427
+
428
+ def __getattr__(name: str):
429
+ """PEP 562 module-level attribute hook.
430
+
431
+ Only ``FolEquivalence`` is handled specially: accessing
432
+ ``metric_hf.FolEquivalence`` (attribute access, ``from ... import
433
+ FolEquivalence``, or ``dir()``-following tools) builds -- and, on every
434
+ call after the first, just returns the cached -- real class via
435
+ :func:`_build_fol_equivalence_class`, so `evaluate`/`datasets` are only
436
+ ever imported when this name is actually touched, never merely because
437
+ this module was.
438
+ """
439
+ if name == "FolEquivalence":
440
+ return _build_fol_equivalence_class()
441
+ raise AttributeError(f"module {__name__!r} has no attribute {name!r}")
442
+
443
+
444
+ def load():
445
+ """Return a ready-to-use :class:`FolEquivalence` instance.
446
+
447
+ Mirrors the shape of ``evaluate.load("metric_name")`` for callers used
448
+ to that entry point. Raises :class:`ImportError` (with the install
449
+ hint) if the optional ``evaluate`` package is not installed -- use
450
+ :func:`compute_fol_metrics` directly in that case.
451
+
452
+ Not annotated ``-> "FolEquivalence"``: that name is only ever reachable
453
+ through the module-level ``__getattr__`` above (never bound as a true
454
+ module global -- that is what keeps it lazy), so a string forward
455
+ reference to it would raise NameError from ``typing.get_type_hints()``.
456
+ The return type is documented here in prose instead.
457
+ """
458
+ return _build_fol_equivalence_class()()