unicode-logic-kit 0.31.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (237) hide show
  1. unicode_logic_kit/__init__.py +385 -0
  2. unicode_logic_kit/__main__.py +520 -0
  3. unicode_logic_kit/_deadline.py +219 -0
  4. unicode_logic_kit/ace/__init__.py +126 -0
  5. unicode_logic_kit/ace/_align.py +135 -0
  6. unicode_logic_kit/ace/chem_lexicon.py +128 -0
  7. unicode_logic_kit/ace/drs_reader.py +570 -0
  8. unicode_logic_kit/ace/mapping.py +666 -0
  9. unicode_logic_kit/ace/reverse_modal.py +138 -0
  10. unicode_logic_kit/ace/runner.py +551 -0
  11. unicode_logic_kit/ace/translate.py +452 -0
  12. unicode_logic_kit/ace/verbalize.py +1070 -0
  13. unicode_logic_kit/api.py +1284 -0
  14. unicode_logic_kit/atp/__init__.py +177 -0
  15. unicode_logic_kit/atp/_ascii_names.py +113 -0
  16. unicode_logic_kit/atp/_html.py +72 -0
  17. unicode_logic_kit/atp/_substructural_input.py +228 -0
  18. unicode_logic_kit/atp/_tff_problem.py +715 -0
  19. unicode_logic_kit/atp/_tptp_problem.py +1111 -0
  20. unicode_logic_kit/atp/_writer_support.py +289 -0
  21. unicode_logic_kit/atp/clingo_backend.py +1180 -0
  22. unicode_logic_kit/atp/cvc5_backend.py +1385 -0
  23. unicode_logic_kit/atp/eprover_backend.py +732 -0
  24. unicode_logic_kit/atp/finite_domain.py +1055 -0
  25. unicode_logic_kit/atp/fitch.py +1547 -0
  26. unicode_logic_kit/atp/fitch_search.py +551 -0
  27. unicode_logic_kit/atp/hets_backend.py +339 -0
  28. unicode_logic_kit/atp/hybrid_down.py +120 -0
  29. unicode_logic_kit/atp/incremental.py +250 -0
  30. unicode_logic_kit/atp/kripke_enum.py +741 -0
  31. unicode_logic_kit/atp/lambek.py +436 -0
  32. unicode_logic_kit/atp/leo3_backend.py +332 -0
  33. unicode_logic_kit/atp/linear.py +738 -0
  34. unicode_logic_kit/atp/lj.py +705 -0
  35. unicode_logic_kit/atp/logic_backends.py +566 -0
  36. unicode_logic_kit/atp/ltl_tableau.py +1084 -0
  37. unicode_logic_kit/atp/minizinc_backend.py +1402 -0
  38. unicode_logic_kit/atp/modal_tableau.py +1382 -0
  39. unicode_logic_kit/atp/nanocop_backend.py +410 -0
  40. unicode_logic_kit/atp/portfolio.py +489 -0
  41. unicode_logic_kit/atp/protocol.py +1803 -0
  42. unicode_logic_kit/atp/prover9_entailment.py +1153 -0
  43. unicode_logic_kit/atp/resolution.py +1376 -0
  44. unicode_logic_kit/atp/resolution_check.py +1114 -0
  45. unicode_logic_kit/atp/sequent.py +1050 -0
  46. unicode_logic_kit/atp/tableau.py +921 -0
  47. unicode_logic_kit/atp/tableau_check.py +543 -0
  48. unicode_logic_kit/atp/tptp_ncl.py +811 -0
  49. unicode_logic_kit/atp/tptp_tff.py +1546 -0
  50. unicode_logic_kit/atp/tstp.py +1333 -0
  51. unicode_logic_kit/atp/tstp_check.py +1096 -0
  52. unicode_logic_kit/atp/twee_backend.py +236 -0
  53. unicode_logic_kit/atp/twee_check.py +711 -0
  54. unicode_logic_kit/atp/twee_entailment.py +953 -0
  55. unicode_logic_kit/atp/vampire_entailment.py +540 -0
  56. unicode_logic_kit/atp/z3_arith.py +470 -0
  57. unicode_logic_kit/atp/z3_equivalence.py +36 -0
  58. unicode_logic_kit/atp/z3_fuzzy.py +362 -0
  59. unicode_logic_kit/atp/z3_input.py +500 -0
  60. unicode_logic_kit/atp/z3_models.py +208 -0
  61. unicode_logic_kit/chem/__init__.py +88 -0
  62. unicode_logic_kit/chem/_naming.py +284 -0
  63. unicode_logic_kit/chem/cache.py +185 -0
  64. unicode_logic_kit/chem/interop.py +244 -0
  65. unicode_logic_kit/chem/mol.py +525 -0
  66. unicode_logic_kit/chem/signature.py +112 -0
  67. unicode_logic_kit/comorphism.py +497 -0
  68. unicode_logic_kit/dl/__init__.py +384 -0
  69. unicode_logic_kit/dl/classification.py +227 -0
  70. unicode_logic_kit/dl/concepts.py +632 -0
  71. unicode_logic_kit/dl/datatypes.py +818 -0
  72. unicode_logic_kit/dl/owl_functional.py +2433 -0
  73. unicode_logic_kit/dl/owl_manchester.py +1637 -0
  74. unicode_logic_kit/dl/owl_reasoner.py +790 -0
  75. unicode_logic_kit/dl/parser.py +391 -0
  76. unicode_logic_kit/dl/tableau.py +4048 -0
  77. unicode_logic_kit/dl/translate.py +2704 -0
  78. unicode_logic_kit/drt/__init__.py +94 -0
  79. unicode_logic_kit/drt/export.py +179 -0
  80. unicode_logic_kit/drt/nodes.py +506 -0
  81. unicode_logic_kit/drt/parser.py +965 -0
  82. unicode_logic_kit/drt/resolve.py +195 -0
  83. unicode_logic_kit/drt/reverse.py +175 -0
  84. unicode_logic_kit/eval/__init__.py +106 -0
  85. unicode_logic_kit/eval/batch.py +382 -0
  86. unicode_logic_kit/eval/canonical.py +663 -0
  87. unicode_logic_kit/eval/chem_batch.py +606 -0
  88. unicode_logic_kit/eval/converses.py +200 -0
  89. unicode_logic_kit/eval/datasets/__init__.py +136 -0
  90. unicode_logic_kit/eval/datasets/_base.py +263 -0
  91. unicode_logic_kit/eval/datasets/_proofwriter_proof.py +422 -0
  92. unicode_logic_kit/eval/datasets/c3po.py +678 -0
  93. unicode_logic_kit/eval/datasets/folio.py +158 -0
  94. unicode_logic_kit/eval/datasets/fracas.py +418 -0
  95. unicode_logic_kit/eval/datasets/groves.py +191 -0
  96. unicode_logic_kit/eval/datasets/logicbench.py +467 -0
  97. unicode_logic_kit/eval/datasets/logicnli.py +303 -0
  98. unicode_logic_kit/eval/datasets/malls.py +133 -0
  99. unicode_logic_kit/eval/datasets/pfolio.py +594 -0
  100. unicode_logic_kit/eval/datasets/pmb.py +242 -0
  101. unicode_logic_kit/eval/datasets/prontoqa.py +611 -0
  102. unicode_logic_kit/eval/datasets/proofwriter.py +1431 -0
  103. unicode_logic_kit/eval/datasets/proverqa.py +674 -0
  104. unicode_logic_kit/eval/datasets/willow.py +478 -0
  105. unicode_logic_kit/eval/equivalence.py +466 -0
  106. unicode_logic_kit/eval/exercise_gen.py +533 -0
  107. unicode_logic_kit/eval/explain.py +791 -0
  108. unicode_logic_kit/eval/generality.py +750 -0
  109. unicode_logic_kit/eval/metric_hf.py +458 -0
  110. unicode_logic_kit/eval/predicate_match.py +343 -0
  111. unicode_logic_kit/eval/theory_check.py +1170 -0
  112. unicode_logic_kit/eval/validate.py +306 -0
  113. unicode_logic_kit/fol/__init__.py +177 -0
  114. unicode_logic_kit/fol/_atom_keys.py +510 -0
  115. unicode_logic_kit/fol/_fol_nodes.py +3586 -0
  116. unicode_logic_kit/fol/_free_parameters.py +105 -0
  117. unicode_logic_kit/fol/_ho_nodes.py +448 -0
  118. unicode_logic_kit/fol/_hybrid_nodes.py +308 -0
  119. unicode_logic_kit/fol/_identifiers.py +1091 -0
  120. unicode_logic_kit/fol/_lambek_nodes.py +112 -0
  121. unicode_logic_kit/fol/_linear_nodes.py +352 -0
  122. unicode_logic_kit/fol/_modal_nodes.py +1467 -0
  123. unicode_logic_kit/fol/_msfl_nodes.py +2196 -0
  124. unicode_logic_kit/fol/_numeral_symbols.py +231 -0
  125. unicode_logic_kit/fol/_so_nodes.py +200 -0
  126. unicode_logic_kit/fol/_symbol_names.py +81 -0
  127. unicode_logic_kit/fol/_team_nodes.py +181 -0
  128. unicode_logic_kit/fol/_tptp_symbols.py +551 -0
  129. unicode_logic_kit/fol/_truth_constants.py +117 -0
  130. unicode_logic_kit/fol/casl_export.py +1135 -0
  131. unicode_logic_kit/fol/casl_import.py +929 -0
  132. unicode_logic_kit/fol/derivation.py +367 -0
  133. unicode_logic_kit/fol/dialect_detect.py +70 -0
  134. unicode_logic_kit/fol/dialect_repair.py +537 -0
  135. unicode_logic_kit/fol/frames.py +637 -0
  136. unicode_logic_kit/fol/grammars/terminals.lark +31 -0
  137. unicode_logic_kit/fol/lambda_tools.py +297 -0
  138. unicode_logic_kit/fol/latex_input.py +429 -0
  139. unicode_logic_kit/fol/modal_translation.py +944 -0
  140. unicode_logic_kit/fol/msflparser.py +1033 -0
  141. unicode_logic_kit/fol/naming.py +422 -0
  142. unicode_logic_kit/fol/nodes.py +241 -0
  143. unicode_logic_kit/fol/normalforms.py +492 -0
  144. unicode_logic_kit/fol/pal.py +287 -0
  145. unicode_logic_kit/fol/prolog_export.py +566 -0
  146. unicode_logic_kit/fol/prolog_input.py +505 -0
  147. unicode_logic_kit/fol/prover9_input.py +1325 -0
  148. unicode_logic_kit/fol/qml.py +1760 -0
  149. unicode_logic_kit/fol/qmltp_input.py +525 -0
  150. unicode_logic_kit/fol/sanitize.py +221 -0
  151. unicode_logic_kit/fol/serialize.py +79 -0
  152. unicode_logic_kit/fol/signature.py +1290 -0
  153. unicode_logic_kit/fol/simplify_check.py +544 -0
  154. unicode_logic_kit/fol/spans.py +594 -0
  155. unicode_logic_kit/fol/tptp_input.py +1503 -0
  156. unicode_logic_kit/fol/tptp_repair.py +941 -0
  157. unicode_logic_kit/fol/unification.py +157 -0
  158. unicode_logic_kit/fol/verbalize.py +263 -0
  159. unicode_logic_kit/hets/__init__.py +163 -0
  160. unicode_logic_kit/hets/bridge.py +142 -0
  161. unicode_logic_kit/hets/client.py +748 -0
  162. unicode_logic_kit/hets/docker.py +420 -0
  163. unicode_logic_kit/hets/dol.py +712 -0
  164. unicode_logic_kit/hets/haskell_json.py +355 -0
  165. unicode_logic_kit/hets/owl_backend.py +794 -0
  166. unicode_logic_kit/hets/owl_cli.py +598 -0
  167. unicode_logic_kit/hets/symbols.py +512 -0
  168. unicode_logic_kit/hol/__init__.py +140 -0
  169. unicode_logic_kit/hol/_ho_common.py +323 -0
  170. unicode_logic_kit/hol/_isabelle_binders.py +125 -0
  171. unicode_logic_kit/hol/classical.py +812 -0
  172. unicode_logic_kit/hol/deepshallow/__init__.py +45 -0
  173. unicode_logic_kit/hol/deepshallow/_common.py +177 -0
  174. unicode_logic_kit/hol/deepshallow/conditional.py +225 -0
  175. unicode_logic_kit/hol/deepshallow/intuitionistic.py +181 -0
  176. unicode_logic_kit/hol/deepshallow/modal.py +217 -0
  177. unicode_logic_kit/hol/deepshallow/qml.py +406 -0
  178. unicode_logic_kit/hol/deepshallow/relevant.py +206 -0
  179. unicode_logic_kit/hol/free.py +753 -0
  180. unicode_logic_kit/hol/goedel.py +336 -0
  181. unicode_logic_kit/hol/ho_modal.py +1743 -0
  182. unicode_logic_kit/hol/intuitionistic.py +403 -0
  183. unicode_logic_kit/hol/isabelle_conditional.py +593 -0
  184. unicode_logic_kit/hol/isabelle_modal.py +1908 -0
  185. unicode_logic_kit/hol/isabelle_relevant.py +412 -0
  186. unicode_logic_kit/hol/isabelle_runner.py +1147 -0
  187. unicode_logic_kit/hol/isabelle_substructural.py +884 -0
  188. unicode_logic_kit/hol/lean.py +1018 -0
  189. unicode_logic_kit/hol/manyvalued.py +921 -0
  190. unicode_logic_kit/hol/secondorder.py +687 -0
  191. unicode_logic_kit/hol/thf_modal.py +941 -0
  192. unicode_logic_kit/hol/thirdorder.py +397 -0
  193. unicode_logic_kit/ilp/__init__.py +89 -0
  194. unicode_logic_kit/ilp/readback.py +389 -0
  195. unicode_logic_kit/ilp/separation.py +153 -0
  196. unicode_logic_kit/ilp/task.py +730 -0
  197. unicode_logic_kit/logic.py +163 -0
  198. unicode_logic_kit/mcp/__init__.py +28 -0
  199. unicode_logic_kit/mcp/__main__.py +5 -0
  200. unicode_logic_kit/mcp/chem_tools.py +1031 -0
  201. unicode_logic_kit/mcp/server.py +2453 -0
  202. unicode_logic_kit/mcp/syntax_spec.py +681 -0
  203. unicode_logic_kit/prob/__init__.py +53 -0
  204. unicode_logic_kit/prob/_bdd.py +225 -0
  205. unicode_logic_kit/prob/_column_gen.py +668 -0
  206. unicode_logic_kit/prob/distribution.py +686 -0
  207. unicode_logic_kit/prob/nilsson.py +470 -0
  208. unicode_logic_kit/py.typed +0 -0
  209. unicode_logic_kit/semantics/__init__.py +137 -0
  210. unicode_logic_kit/semantics/_modal_reject.py +156 -0
  211. unicode_logic_kit/semantics/action_models.py +466 -0
  212. unicode_logic_kit/semantics/asp_models.py +1200 -0
  213. unicode_logic_kit/semantics/conditional.py +580 -0
  214. unicode_logic_kit/semantics/dynamic_epistemic.py +95 -0
  215. unicode_logic_kit/semantics/free_logic.py +913 -0
  216. unicode_logic_kit/semantics/fuzzy.py +384 -0
  217. unicode_logic_kit/semantics/fuzzy_kripke.py +442 -0
  218. unicode_logic_kit/semantics/intuitionistic.py +581 -0
  219. unicode_logic_kit/semantics/kripke.py +1139 -0
  220. unicode_logic_kit/semantics/manyvalued.py +580 -0
  221. unicode_logic_kit/semantics/matrix.py +342 -0
  222. unicode_logic_kit/semantics/model_eval.py +1135 -0
  223. unicode_logic_kit/semantics/modelfinder.py +1036 -0
  224. unicode_logic_kit/semantics/nonmonotonic.py +372 -0
  225. unicode_logic_kit/semantics/relevant.py +331 -0
  226. unicode_logic_kit/semantics/secondorder.py +657 -0
  227. unicode_logic_kit/semantics/structures.py +352 -0
  228. unicode_logic_kit/semantics/tarski.py +975 -0
  229. unicode_logic_kit/semantics/team.py +315 -0
  230. unicode_logic_kit/semantics/team_translation.py +416 -0
  231. unicode_logic_kit/semantics/thirdorder.py +358 -0
  232. unicode_logic_kit/semantics/tnorm.py +85 -0
  233. unicode_logic_kit/semantics/truthtable.py +201 -0
  234. unicode_logic_kit-0.31.0.dist-info/METADATA +333 -0
  235. unicode_logic_kit-0.31.0.dist-info/RECORD +237 -0
  236. unicode_logic_kit-0.31.0.dist-info/WHEEL +4 -0
  237. unicode_logic_kit-0.31.0.dist-info/licenses/LICENSE +21 -0
@@ -0,0 +1,191 @@
1
+ """Adapter for the GROVES dataset (Vossel, "Groves" dataset card,
2
+ https://huggingface.co/datasets/fvossel/groves) — local JSONL only, no
3
+ network access.
4
+
5
+ Source and verified schema
6
+ ---------------------------
7
+ Source: https://huggingface.co/datasets/fvossel/groves (unauthenticated;
8
+ verified 2026-08-12 directly via the Hugging Face datasets-server
9
+ ``/splits``, ``/first-rows``, and ``/size`` APIs, and cross-checked against
10
+ the raw upstream ``train.json`` file via ``huggingface.co/.../resolve/main/``).
11
+ Config ``default`` has three splits — ``train`` (35910 rows), ``validation``
12
+ (3989 rows), ``test`` (9971 rows) — each row a flat JSON object with exactly
13
+ two keys:
14
+
15
+ * ``"NL"`` — ``str``, one natural-language statement.
16
+ * ``"FOL"`` — ``str``, its FOL translation, already written in THIS KIT'S OWN
17
+ unicode surface syntax (``∀``/``∃``/``∧``/``∨``/``¬``/``→``/``↔``/``⊕``/
18
+ ``≠``/``=`` all appear in the 100-row verified sample and ALL parse under
19
+ this kit's default ``fol`` mode via :func:`unicode_logic_kit.api.parse_any` —
20
+ confirmed directly, not assumed, by parsing and :func:`~unicode_logic_kit.api.check`-
21
+ ing a 26-row spot sample spanning every connective/quantifier combination
22
+ seen in the sample, all of which came back ``ok=True``/``is_closed=True``/
23
+ ``arity_consistent=True``/``has_lambdas=False``). GROVES is the only
24
+ adapter in this subpackage whose FOL column needs no dialect coaxing at
25
+ all — it was generated directly in this kit's notation.
26
+
27
+ GROVES has NO id field, NO premises/conclusion structure, and NO entailment
28
+ label — the SAME flat (NL, FOL) translation-pair shape as
29
+ :mod:`~unicode_logic_kit.eval.datasets.malls`, mapped onto
30
+ :class:`~unicode_logic_kit.eval.datasets.DatasetExample` the same way (see
31
+ ``malls.py``'s docstring for the field-mapping rationale, reused verbatim
32
+ here):
33
+
34
+ * ``nl_conclusion`` / ``fol_conclusion`` carry ``"NL"`` / ``"FOL"``.
35
+ * ``nl_premises`` / ``fol_premises`` are always ``()``.
36
+ * ``label`` is always ``None``.
37
+
38
+ Provenance (per the dataset card's "Description"/"Licensing" sections,
39
+ verified 2026-08-12): GROVES's natural-language inputs are drawn from
40
+ WillowNLtoFOL (https://huggingface.co/datasets/iedeveci/WillowNLtoFOL,
41
+ originally CC BY-NC-ND 4.0, "included with permission from the original
42
+ authors" per the card) and MALLS-v0
43
+ (https://huggingface.co/datasets/yuan-yang/MALLS-v0, CC BY-NC 4.0); the FOL
44
+ expressions themselves were newly generated for GROVES and the dataset was
45
+ "subsequently filtered" — the card gives no further detail on that
46
+ generation/filtering pipeline, and this adapter does not invent any (no
47
+ independent semantic re-verification of the FOL against the NL is claimed
48
+ here beyond "it parses and is well-formed under this kit").
49
+
50
+ Upstream GROVES is distributed as JSON ARRAY files (``train.json`` /
51
+ ``val.json`` / ``test.json`` — one big ``[...]`` list per split, confirmed
52
+ directly from the raw file's first bytes), NOT as JSONL. Like
53
+ :mod:`~unicode_logic_kit.eval.datasets.malls`, this loader reads local JSONL
54
+ (one JSON object per line) for a uniform streaming adapter surface; convert
55
+ an upstream split file first (e.g. ``jq -c '.[]' train.json >
56
+ groves_train.jsonl``) before calling :func:`load_groves`.
57
+
58
+ License: **CC-BY-NC-4.0** (non-commercial), per the dataset card's ``license``
59
+ front-matter and its "Licensing" section — which additionally requires
60
+ downstream users to also comply with the licenses of the two datasets GROVES
61
+ is built from (WillowNLtoFOL, CC BY-NC-ND 4.0; MALLS-v0, CC BY-NC 4.0). This
62
+ loader itself has no license implications beyond reading a local file the
63
+ caller already obtained; it never downloads or redistributes GROVES data.
64
+
65
+ Citation: as of 2026-08-12 the dataset card's "Notes" section states only
66
+ that "[f]urther details about dataset construction and evaluation will be
67
+ provided in a forthcoming publication" — it does NOT name a paper. A
68
+ plausible companion paper, matching both topic (NL-to-FOL formalization with
69
+ fine-tuned LLMs) and authorship (the dataset's Hugging Face account is
70
+ ``fvossel``): Vossel, Felix, Till Mossakowski, and Bjoern Gehrke. "Advancing
71
+ Natural Language Formalization to First Order Logic with Fine-tuned LLMs."
72
+ arXiv:2509.22338. This link is NOT asserted by the dataset card itself —
73
+ :data:`DATASET_INFO`'s ``citation_hint`` for ``"groves"`` flags it as
74
+ plausible-but-unconfirmed rather than presenting it as settled.
75
+
76
+ What GROVES does NOT have (spelled out so nothing here is silently assumed):
77
+ no premises/entailment structure (a straight translation-pair dataset, like
78
+ MALLS, unlike FOLIO); no entailment label; no id field of any kind; no
79
+ train/validation/test SPLIT FILE bundling — each split is its own upstream
80
+ JSON array file, and this loader (like ``load_malls``) takes exactly one
81
+ already-converted local JSONL path per call, so loading multiple splits
82
+ means calling :func:`load_groves` once per split file; no independent
83
+ human/semantic verification of the FOL column beyond "the dataset was
84
+ filtered" per the card — this adapter's own :func:`~unicode_logic_kit.eval.datasets.audit_examples`
85
+ only checks parseability and well-formedness (closed, arity-consistent,
86
+ lambda-free), never whether a given FOL formula is a *correct* translation
87
+ of its NL sentence.
88
+
89
+ This module never downloads anything — obtain and convert the data yourself
90
+ and pass its local JSONL path to :func:`load_groves`.
91
+ """
92
+
93
+ import json
94
+ from pathlib import Path
95
+ from typing import FrozenSet, Iterator, Union
96
+
97
+ from ._base import DatasetExample, _register_dataset_info
98
+
99
+ __all__ = ["load_groves"]
100
+
101
+ _register_dataset_info(
102
+ "groves",
103
+ license=(
104
+ "CC-BY-NC-4.0 (non-commercial); GROVES is built from WillowNLtoFOL "
105
+ "(CC BY-NC-ND 4.0, used with permission from its original authors "
106
+ "per the GROVES dataset card) and MALLS-v0 (CC BY-NC 4.0) "
107
+ "natural-language inputs, so downstream use must also honour both "
108
+ "of those source licenses per the dataset card's Licensing section"
109
+ ),
110
+ source_url="https://huggingface.co/datasets/fvossel/groves",
111
+ citation_hint=(
112
+ "Vossel, Felix. \"Groves\" (dataset card, Hugging Face, "
113
+ "https://huggingface.co/datasets/fvossel/groves) — the card states "
114
+ "only that \"further details ... will be provided in a forthcoming "
115
+ "publication\" and names no paper. Plausible (topic- and "
116
+ "author-matched, but NOT card-confirmed) companion paper: Vossel, "
117
+ "Felix, Till Mossakowski, and Bjoern Gehrke. \"Advancing Natural "
118
+ "Language Formalization to First Order Logic with Fine-tuned "
119
+ "LLMs.\" arXiv:2509.22338."
120
+ ),
121
+ )
122
+
123
+
124
+ def _example_from_record(record: dict, line_no: int,
125
+ known_bad_ids: FrozenSet[str]) -> DatasetExample:
126
+ nl = record.get("NL")
127
+ fol = record.get("FOL")
128
+ # GROVES has no native id field (see module docstring): every verified
129
+ # row is exactly {"NL": ..., "FOL": ...}. Mirrors malls.py's stance —
130
+ # honour an "id" key opportunistically should some downstream re-export
131
+ # add one, otherwise fall back to a positional id.
132
+ raw_id = record.get("id")
133
+ example_id = str(raw_id) if raw_id is not None else f"groves:{line_no}"
134
+
135
+ meta = {k: v for k, v in record.items() if k not in ("NL", "FOL", "id")}
136
+ meta["line_no"] = line_no
137
+
138
+ return DatasetExample(
139
+ id=example_id,
140
+ nl_premises=(),
141
+ fol_premises=(),
142
+ nl_conclusion=nl,
143
+ fol_conclusion=fol,
144
+ label=None,
145
+ known_bad=example_id in known_bad_ids,
146
+ meta=meta,
147
+ )
148
+
149
+
150
+ def load_groves(path: Union[str, Path], *,
151
+ known_bad_ids: FrozenSet[str] = frozenset()) -> Iterator[DatasetExample]:
152
+ """Stream :class:`~unicode_logic_kit.eval.datasets.DatasetExample` from a
153
+ local GROVES JSONL file (one already-converted split — see module
154
+ docstring for converting the upstream JSON-array split files).
155
+
156
+ Args:
157
+ path: path to a local ``.jsonl`` file — one ``{"NL": ..., "FOL": ...}``
158
+ object per non-blank line. NEVER downloaded by this function;
159
+ obtain and convert the split file yourself from
160
+ https://huggingface.co/datasets/fvossel/groves.
161
+ known_bad_ids: ids (the record's own ``"id"`` if present, else the
162
+ positional fallback ``f"groves:{line_no}"``) whose ``"FOL"``
163
+ translation is known to be broken. Every yielded example with a
164
+ matching id gets ``known_bad=True``. Defaults to an empty set.
165
+
166
+ Yields:
167
+ One :class:`~unicode_logic_kit.eval.datasets.DatasetExample` per
168
+ non-blank JSONL line, in file order, with ``nl_conclusion``/
169
+ ``fol_conclusion`` set from ``"NL"``/``"FOL"`` and
170
+ ``nl_premises``/``fol_premises`` empty (see module docstring for why).
171
+ A record missing ``"NL"`` and/or ``"FOL"`` yields an example with
172
+ ``None`` in the corresponding field rather than raising — the same
173
+ documented "missing key -> None/() default" behaviour as
174
+ :func:`~unicode_logic_kit.eval.datasets.folio.load_folio` and
175
+ :func:`~unicode_logic_kit.eval.datasets.malls.load_malls` (a structurally
176
+ malformed FILE is a loud failure, see below; a single record missing
177
+ an optional-looking key is not).
178
+
179
+ Raises:
180
+ FileNotFoundError: ``path`` does not exist.
181
+ json.JSONDecodeError: a non-blank line is not valid JSON — raised,
182
+ not swallowed (a malformed dataset file must fail loudly).
183
+ """
184
+ path = Path(path)
185
+ with path.open("r", encoding="utf-8") as fh:
186
+ for line_no, raw_line in enumerate(fh):
187
+ line = raw_line.strip()
188
+ if not line:
189
+ continue
190
+ record = json.loads(line)
191
+ yield _example_from_record(record, line_no, known_bad_ids)
@@ -0,0 +1,467 @@
1
+ """Adapter for LogicBench (Parmar, Patel, Varshney, Nakamura, Luo, Mashetty,
2
+ Mitra, Baral, "Towards Systematic Evaluation of Logical Reasoning Ability of
3
+ Large Language Models", 2024, arXiv:2404.15522) — natural-language
4
+ question-answering over 25 single-inference-rule reasoning patterns spanning
5
+ propositional, first-order and non-monotonic logic. Modeled directly on
6
+ :mod:`~unicode_logic_kit.eval.datasets.fracas`, since LogicBench ships NO gold
7
+ FOL either: every row is natural-language context plus a question and a
8
+ yes/no or multiple-choice answer, nothing more — so, exactly as for FraCaS,
9
+ the translation step lives OUTSIDE this library (:func:`solve_example`'s
10
+ ``translate``) and this package only decides.
11
+
12
+ Source and verified schema
13
+ ---------------------------
14
+ Verified 2026-09-17 directly against a local clone of the repository the
15
+ paper names, ``https://github.com/Mihir3009/LogicBench``. ``data/`` holds
16
+ two releases:
17
+
18
+ * **LogicBench(Eval)** — the human-verified evaluation set this adapter
19
+ reads, split into ``BQA`` (Binary Question-Answering) and ``MCQA``
20
+ (Multiple-Choice Question-Answering), each further split by
21
+ ``propositional_logic`` / ``first_order_logic`` / ``nm_logic`` and then by
22
+ one JSON file per inference rule ("axiom"), e.g.
23
+ ``BQA/propositional_logic/modus_tollens/data_instances.json``.
24
+ * **LogicBench(Aug)** — a synthetically augmented TRAINING split with a
25
+ DIFFERENT schema (top-level key ``"data_samples"`` instead of
26
+ ``"samples"``, no per-row ``"id"``, and 4 ``qa_pairs`` per row — both
27
+ polarities of both directions — instead of 2). Inspected while verifying
28
+ this adapter's schema, but NOT read by it: the build spec this module
29
+ implements only covers ``LogicBench(Eval)``, and Aug's different shape
30
+ would need its own field mapping, not this one.
31
+
32
+ One **Eval** JSON file — :func:`load_logicbench` reads exactly one, like
33
+ every other loader in this package; nothing here downloads anything — has
34
+ the shape ``{"type": str, "axiom": str, "samples": [...]}``:
35
+
36
+ * ``"type"`` — the file's own logic-type label, kept VERBATIM in
37
+ ``meta["logic_type"]``. Confirmed by reading every real Eval file: it is
38
+ ``"propositional_logic"`` / ``"first_order_logic"`` under those two
39
+ directories (matching the directory name), but ``"non_monotonic_logic"``
40
+ under the ``nm_logic`` DIRECTORY — never the literal string
41
+ ``"nm_logic"``. :func:`solve_example` routes on the ACTUAL file value
42
+ (``"non_monotonic_logic"``, :data:`NM_LOGIC_TYPE`), not on the directory
43
+ name a caller happened to read the file from.
44
+ * ``"axiom"`` — the inference-rule name (e.g. ``"modus_tollens"``), kept
45
+ verbatim in ``meta["axiom"]``.
46
+ * ``"samples"`` — a list of ``{"id": int, "context": str, ...}``. ``id`` is
47
+ 1-based WITHIN THIS ONE FILE, not globally unique (kept verbatim in
48
+ ``meta["sample_id"]``); ``context`` is one NL paragraph, never pre-split
49
+ into sentences the way FraCaS's ``<p>`` elements are, so it maps to the
50
+ single-element ``nl_premises = (context,)`` — a caller's ``translate``
51
+ sees the whole paragraph at once and is free to return a single
52
+ conjunctive formula for it.
53
+
54
+ * **BQA** (``split="BQA"``): each sample additionally carries
55
+ ``"qa_pairs": [{"question": str, "answer": "yes"|"no"}, ...]`` — 2 to 4
56
+ pairs per sample in every real Eval BQA file (every ``answer`` verified
57
+ to be exactly the lowercase string ``"yes"`` or ``"no"``, nothing else).
58
+ A sample with N qa_pairs yields N separate :class:`DatasetExample`\\ s
59
+ (one per question, all sharing the same ``nl_premises``), since each
60
+ pair asks about a DIFFERENT proposition with its OWN gold answer —
61
+ collapsing them into one example would silently keep only one label.
62
+ ``meta["qa_index"]`` (0-based, within the sample) makes the split point
63
+ reconstructable, and the synthetic id embeds it too.
64
+ * **MCQA** (``split="MCQA"``): each sample instead carries
65
+ ``"question": str`` (a FIXED meta-question, e.g. "What would be the most
66
+ appropriate conclusion based on the given context?" — not itself a
67
+ provable proposition; see :func:`solve_example`), ``"choices": dict``
68
+ (``"choice_1"``, … — 4 or 5 entries depending on the file, verified) and
69
+ ``"answer": str`` (verified, in every real Eval MCQA file, to always be
70
+ a key of that SAME sample's ``choices``). One :class:`DatasetExample`
71
+ per sample. ``choices`` survives verbatim in ``meta["choices"]``.
72
+
73
+ Field mapping (either split): ``nl_premises = (context,)``, ``nl_conclusion``
74
+ = the question text, ``label`` = the answer — ``"yes"``/``"no"`` verbatim
75
+ for BQA, the chosen ``"choice_N"`` key verbatim for MCQA. Kept AS-IS, not
76
+ smoothed into FraCaS's yes/no/unknown three-way scale: a binary
77
+ question-answering task and a 4-or-5-way multiple choice are different task
78
+ shapes, and forcing one vocabulary onto both would invent structure that is
79
+ not in the data. ``fol_premises = ()`` and ``fol_conclusion = None``
80
+ always — see "Honest limitations".
81
+
82
+ Honest limitations
83
+ -------------------
84
+ * No gold FOL anywhere in the source, so — exactly as for
85
+ :mod:`~unicode_logic_kit.eval.datasets.fracas` —
86
+ :func:`~unicode_logic_kit.eval.datasets.audit_examples` is vacuous on every
87
+ LogicBench example (nothing to audit), and deciding one needs an
88
+ externally injected translation.
89
+ * **MCQA rows are not decided by this module at all.** ``"question"`` in an
90
+ MCQA sample is a generic meta-question ("What would be the most
91
+ appropriate conclusion...?"), not a standalone proposition — there is
92
+ nothing there for a prover to prove or refute. :func:`solve_example`
93
+ refuses every MCQA row by name rather than silently running
94
+ ``api.prove`` on a sentence that was never meant to be one; a caller who
95
+ wants to score MCQA has to translate one of ``example.meta["choices"]``
96
+ itself and decide it directly. Consequently the DISTRACTOR choices are
97
+ never logic-checked by this adapter at all, chosen or not.
98
+ * **The non-monotonic route is a real, narrow fragment, not a general
99
+ solver.** :mod:`~unicode_logic_kit.semantics.nonmonotonic`'s
100
+ ``minimal_models``/``minimal_entails`` implement circumscription with
101
+ every predicate either CIRCUMSCRIBED (minimised) or FIXED — there is no
102
+ third "varied" category (that module's own ``circumscription_formula``
103
+ explicitly leaves it unimplemented). A translation that leaves an
104
+ "abnormality" predicate free for some individual (rather than pinning it
105
+ with an explicit ground fact, positive or negative) will generally admit
106
+ several incomparable minimal models that DISAGREE on the goal.
107
+ :func:`solve_example` does **not** detect that disagreement and does
108
+ **not** refuse it: ``minimal_entails`` only ever returns a bool, with no
109
+ way to report that its minimal models disagree, so the route falls
110
+ through to that bool's own SKEPTICAL reading — "yes" iff the goal holds
111
+ in *every* minimal model found, "no" otherwise — and reports it exactly
112
+ like any other answer, with no flag that several readings were possible.
113
+ That skeptical bool is a real, well-defined answer to a real, precisely
114
+ bounded question (minimal-model entailment up to ``max_size``), not a
115
+ guess or an approximation of one — but it is only as trustworthy as the
116
+ translation's discipline in pinning every abnormality predicate with an
117
+ explicit ground fact for every named individual; a translation that
118
+ skips one gets a confident-looking "yes"/"no" out of this route with no
119
+ signal that the question was underspecified. The ONE case this route
120
+ does detect and refuse is the *empty* minimal-model set (see
121
+ :func:`solve_example`'s own docstring) — genuinely unsatisfiable
122
+ premises, or a search bound that is simply too small.
123
+
124
+ License
125
+ -------
126
+ The roadmap build spec that requested this adapter named CC BY 4.0. That is
127
+ WRONG for this repository: the cloned ``LogicBench`` repository's
128
+ ``LICENSE`` file is the plain MIT License (``Copyright (c) 2024 Mihir``),
129
+ and its ``README.md`` states "**Licence:** MIT License" directly under its
130
+ "Data Release" heading — both read directly, 2026-09-17.
131
+ :data:`~unicode_logic_kit.eval.datasets.DATASET_INFO` records MIT, not the
132
+ spec's CC BY 4.0.
133
+ """
134
+
135
+ import json
136
+ from pathlib import Path
137
+ from typing import Callable, FrozenSet, Iterator, Optional, Set, Union
138
+
139
+ from ._base import DatasetExample, _register_dataset_info
140
+ from ...semantics.modelfinder import MAX_CANDIDATES
141
+
142
+ __all__ = ["load_logicbench", "solve_example", "LOGIC_TYPES", "NM_LOGIC_TYPE"]
143
+
144
+
145
+ #: The three ``"type"`` values a real LogicBench(Eval) file carries — see the
146
+ #: module docstring for why the non-monotonic one is NOT the string
147
+ #: ``"nm_logic"`` despite that being the directory name upstream.
148
+ LOGIC_TYPES = ("propositional_logic", "first_order_logic", "non_monotonic_logic")
149
+
150
+ #: The ``"type"`` value :func:`solve_example` routes through
151
+ #: :mod:`~unicode_logic_kit.semantics.nonmonotonic` instead of ``api.prove``.
152
+ NM_LOGIC_TYPE = "non_monotonic_logic"
153
+
154
+ _CLASSICAL_LOGIC_TYPES = frozenset(LOGIC_TYPES) - {NM_LOGIC_TYPE}
155
+ _SPLITS = ("BQA", "MCQA")
156
+ _BQA_ANSWERS = frozenset({"yes", "no"})
157
+
158
+
159
+ _register_dataset_info(
160
+ "logicbench",
161
+ license=("MIT License (the repository's own LICENSE file and its "
162
+ "README's 'Licence: MIT License' agree; verified 2026-09-17 — "
163
+ "NOT CC BY 4.0, which is what an earlier, unverified roadmap "
164
+ "entry for this adapter had assumed)"),
165
+ source_url="https://github.com/Mihir3009/LogicBench",
166
+ citation_hint=('Parmar, Patel, Varshney, Nakamura, Luo, Mashetty, Mitra, '
167
+ 'Baral, "Towards Systematic Evaluation of Logical '
168
+ 'Reasoning Ability of Large Language Models", 2024, '
169
+ 'arXiv:2404.15522.'),
170
+ )
171
+
172
+
173
+ # ---------------------------------------------------------------------------
174
+ # Reading
175
+ # ---------------------------------------------------------------------------
176
+
177
+ def load_logicbench(path: Union[str, Path], *, split: str,
178
+ known_bad_ids: FrozenSet[str] = frozenset(),
179
+ ) -> Iterator[DatasetExample]:
180
+ """Read one LogicBench(Eval) ``data_instances.json`` into
181
+ :class:`DatasetExample` objects, in file order.
182
+
183
+ See the module docstring for the field mapping and the BQA
184
+ qa_pairs-flattening this loader does. ``split`` must be ``"BQA"`` or
185
+ ``"MCQA"`` and must match the file's ACTUAL sample shape — a BQA sample
186
+ without ``qa_pairs`` (or an MCQA sample without ``question``/``choices``)
187
+ is refused by name rather than silently misread, so passing the wrong
188
+ ``split`` for a file cannot produce garbage examples.
189
+
190
+ Args:
191
+ path: the local ``data_instances.json`` (one axiom, one logic type,
192
+ one of BQA/MCQA).
193
+ split: ``"BQA"`` or ``"MCQA"`` — which of LogicBench(Eval)'s two
194
+ task shapes this file holds.
195
+ known_bad_ids: ids (in this adapter's own prefixed form, e.g.
196
+ ``"logicbench:BQA:propositional_logic:modus_tollens:1:0"``) to
197
+ flag as ``known_bad`` — the same caller-curated mechanic every
198
+ adapter has.
199
+
200
+ Raises:
201
+ ValueError: ``split`` is not one of ``"BQA"``/``"MCQA"``, the file
202
+ is missing ``type``/``axiom``/``samples``, ``type`` is outside
203
+ :data:`LOGIC_TYPES`, a sample id repeats, a BQA sample has no
204
+ ``qa_pairs`` (or an answer outside ``{"yes", "no"}``), or an
205
+ MCQA sample has no ``question``/``choices`` (or an ``answer``
206
+ that is not itself a key of its own ``choices``) — every
207
+ malformed-input case is named, never worked around.
208
+ """
209
+ if split not in _SPLITS:
210
+ raise ValueError(
211
+ f"logicbench: split must be one of {_SPLITS}, got {split!r}")
212
+
213
+ with open(path, encoding="utf-8") as f:
214
+ data = json.load(f)
215
+
216
+ if not isinstance(data, dict) or not {"type", "axiom", "samples"} <= data.keys():
217
+ raise ValueError(
218
+ f"logicbench: {path} is missing 'type'/'axiom'/'samples' — is "
219
+ "this a LogicBench(Eval) data_instances.json?")
220
+
221
+ logic_type = data["type"]
222
+ if logic_type not in LOGIC_TYPES:
223
+ raise ValueError(
224
+ f"logicbench: {path} has type={logic_type!r}, outside "
225
+ f"{list(LOGIC_TYPES)}")
226
+ axiom = data["axiom"]
227
+ if not axiom:
228
+ raise ValueError(f"logicbench: {path} has an empty 'axiom'")
229
+
230
+ samples = data["samples"]
231
+ if not isinstance(samples, list):
232
+ raise ValueError(f"logicbench: {path}: 'samples' is not a list")
233
+
234
+ seen_sample_ids: Set = set()
235
+ for sample in samples:
236
+ sample_id = sample.get("id")
237
+ if sample_id is None:
238
+ raise ValueError(f"logicbench: {path}: a sample has no 'id'")
239
+ if sample_id in seen_sample_ids:
240
+ raise ValueError(
241
+ f"logicbench: {path}: duplicate sample id {sample_id!r}")
242
+ seen_sample_ids.add(sample_id)
243
+
244
+ context = sample.get("context")
245
+ if not context:
246
+ raise ValueError(
247
+ f"logicbench: {path}: sample {sample_id} has an empty "
248
+ "'context'")
249
+
250
+ base_meta = {"axiom": axiom, "logic_type": logic_type, "task": split,
251
+ "sample_id": sample_id}
252
+
253
+ if split == "BQA":
254
+ qa_pairs = sample.get("qa_pairs")
255
+ if not qa_pairs:
256
+ raise ValueError(
257
+ f"logicbench: {path}: sample {sample_id} has no "
258
+ "'qa_pairs' — is split='BQA' correct for this file?")
259
+ for qa_index, qa in enumerate(qa_pairs):
260
+ question = qa.get("question")
261
+ answer = qa.get("answer")
262
+ if not question:
263
+ raise ValueError(
264
+ f"logicbench: {path}: sample {sample_id} "
265
+ f"qa_pairs[{qa_index}] has no 'question'")
266
+ if answer not in _BQA_ANSWERS:
267
+ raise ValueError(
268
+ f"logicbench: {path}: sample {sample_id} "
269
+ f"qa_pairs[{qa_index}] has answer={answer!r}, "
270
+ f"outside {sorted(_BQA_ANSWERS)}")
271
+ example_id = (f"logicbench:BQA:{logic_type}:{axiom}:"
272
+ f"{sample_id}:{qa_index}")
273
+ yield DatasetExample(
274
+ id=example_id,
275
+ nl_premises=(context,),
276
+ fol_premises=(),
277
+ nl_conclusion=question,
278
+ fol_conclusion=None,
279
+ label=answer,
280
+ known_bad=example_id in known_bad_ids,
281
+ meta=dict(base_meta, qa_index=qa_index),
282
+ )
283
+ else: # "MCQA"
284
+ question = sample.get("question")
285
+ choices = sample.get("choices")
286
+ answer = sample.get("answer")
287
+ if not question:
288
+ raise ValueError(
289
+ f"logicbench: {path}: sample {sample_id} has no "
290
+ "'question' — is split='MCQA' correct for this file?")
291
+ if not isinstance(choices, dict) or not choices:
292
+ raise ValueError(
293
+ f"logicbench: {path}: sample {sample_id} has no "
294
+ "'choices' dict")
295
+ if answer not in choices:
296
+ raise ValueError(
297
+ f"logicbench: {path}: sample {sample_id} has "
298
+ f"answer={answer!r}, not a key of its own 'choices' "
299
+ f"{sorted(choices)}")
300
+ example_id = f"logicbench:MCQA:{logic_type}:{axiom}:{sample_id}"
301
+ yield DatasetExample(
302
+ id=example_id,
303
+ nl_premises=(context,),
304
+ fol_premises=(),
305
+ nl_conclusion=question,
306
+ fol_conclusion=None,
307
+ label=answer,
308
+ known_bad=example_id in known_bad_ids,
309
+ meta=dict(base_meta, choices=dict(choices)),
310
+ )
311
+
312
+
313
+ # ---------------------------------------------------------------------------
314
+ # Deciding — with the translation injected by the caller
315
+ # ---------------------------------------------------------------------------
316
+
317
+ def solve_example(example: DatasetExample, *, translate: Callable[[str], object],
318
+ circumscribed: Optional[Set] = None,
319
+ max_size: int = 4, max_candidates: int = MAX_CANDIDATES,
320
+ **prove_kwargs) -> dict:
321
+ """Decide one LogicBench BQA row end-to-end — the translation is YOURS.
322
+
323
+ LogicBench ships no formulas, so this helper takes ``translate``: a
324
+ callable mapping one natural-language sentence to either a formula
325
+ string (parsed with :func:`unicode_logic_kit.api.parse_any`) or an
326
+ already-built kit node — the same seam
327
+ :func:`~unicode_logic_kit.eval.datasets.fracas.solve_example` uses. This
328
+ package calls no LLM and no external system itself.
329
+
330
+ Routing is on ``example.meta["logic_type"]``:
331
+
332
+ * :data:`NM_LOGIC_TYPE` (``"non_monotonic_logic"``): decided via
333
+ :func:`unicode_logic_kit.semantics.nonmonotonic.minimal_entails` —
334
+ ``circumscribed`` (``None`` minimises every predicate in the
335
+ translated theory, matching that module's own default reading),
336
+ ``max_size`` and ``max_candidates`` reach it verbatim. Before trusting
337
+ the answer, this function ALSO calls
338
+ :func:`~unicode_logic_kit.semantics.nonmonotonic.minimal_models`
339
+ directly and checks it is non-empty: ``minimal_entails`` returns
340
+ ``True`` VACUOUSLY when no minimal model exists within the bound
341
+ (nothing to check the conclusion against — for-loop over an empty
342
+ list), which would silently misreport either genuinely unsatisfiable
343
+ premises or a search bound that is simply too small as a confident
344
+ "yes". Finding no minimal model raises instead, naming the reason,
345
+ rather than ever returning that vacuous "yes". This is the ONLY
346
+ case this route detects and refuses: a translation that leaves an
347
+ abnormality predicate's value free for some individual (instead of
348
+ pinning it, positive or negative, with an explicit ground fact) will
349
+ typically produce several incomparable minimal models that DISAGREE
350
+ on the goal rather than an empty set, so it is NOT caught here —
351
+ :func:`minimal_entails` only ever returns a bool, with no way to
352
+ report that its minimal models disagree, so this function silently
353
+ reports that bool's own skeptical reading ("yes" iff the goal holds
354
+ in every minimal model found) with no signal that several readings
355
+ were possible. See the module docstring's "Honest limitations"
356
+ section for the discipline a translation needs (an explicit ground
357
+ fact for every abnormality predicate application) to avoid that
358
+ silent case.
359
+ * :data:`~unicode_logic_kit.eval.datasets.logicbench.LOGIC_TYPES`'s other
360
+ two values (``"propositional_logic"``, ``"first_order_logic"``):
361
+ decided via :func:`unicode_logic_kit.api.prove` — ``"yes"`` iff PROVED,
362
+ ``"no"`` iff REFUTED (a genuine countermodel, not merely "could not
363
+ prove"), and ``predicted=None`` on an inconclusive UNKNOWN verdict:
364
+ LogicBench's label vocabulary is only ``{"yes", "no"}``, so — unlike
365
+ FraCaS, whose own three-way scale has an ``"unknown"`` label to fall
366
+ back on — forcing an indefinite prover outcome into either binary
367
+ label would invent an answer LogicBench never asked for. Extra
368
+ ``prove_kwargs`` reach :func:`unicode_logic_kit.api.prove` verbatim (and
369
+ are IGNORED on a non-monotonic-logic row — that route takes
370
+ ``circumscribed``/``max_size``/``max_candidates`` instead).
371
+
372
+ Only decides BQA rows. An MCQA row's ``nl_conclusion`` is a generic
373
+ meta-question, not a standalone proposition (see the module docstring),
374
+ so this function refuses it by name instead of running a prover on a
375
+ sentence that was never meant to be one.
376
+
377
+ Returns a dict with ``predicted`` (``"yes"``/``"no"``/``None``),
378
+ ``label`` (the gold answer, untouched — scoring against it is the
379
+ caller's decision), ``route`` (``"classical"``/``"nonmonotonic"``), the
380
+ translated ``premises``/``hypothesis`` in kit notation (so a wrong
381
+ prediction can be traced back to the translation that caused it), and
382
+ either ``verdict`` (the classical route's full
383
+ :class:`~unicode_logic_kit.atp.protocol.Verdict` dict) or
384
+ ``minimal_model_count`` (the non-monotonic route's model count).
385
+
386
+ Raises:
387
+ ValueError: ``example`` is an MCQA row, its ``logic_type`` is
388
+ outside :data:`LOGIC_TYPES`, a translated string does not parse,
389
+ or (non-monotonic route only) no minimal model of the
390
+ translated premises was found within ``max_size``.
391
+ """
392
+ from ... import api
393
+ from ...fol.nodes import Node
394
+ from ...semantics.nonmonotonic import minimal_entails, minimal_models
395
+
396
+ if example.meta.get("task") != "BQA":
397
+ raise ValueError(
398
+ f"logicbench: example {example.id}: solve_example only decides "
399
+ "BQA rows — an MCQA row's 'question' field is a generic "
400
+ "meta-question ('What would be the most appropriate "
401
+ "conclusion...?'), not a standalone provable proposition; "
402
+ "translate one of example.meta['choices'] yourself and decide "
403
+ "it directly instead.")
404
+
405
+ def _formula(sentence: str) -> "Node":
406
+ produced = translate(sentence)
407
+ if isinstance(produced, Node):
408
+ return produced
409
+ if not isinstance(produced, str):
410
+ raise ValueError(
411
+ f"logicbench: example {example.id}: translate({sentence!r}) "
412
+ f"returned {type(produced).__name__}, expected a formula "
413
+ "string or a kit node")
414
+ parsed = api.parse_any(produced)
415
+ if not parsed.ok:
416
+ raise ValueError(
417
+ f"logicbench: example {example.id}: the translation "
418
+ f"{produced!r} of {sentence!r} does not parse")
419
+ return parsed.formula
420
+
421
+ premises = [_formula(sentence) for sentence in example.nl_premises]
422
+ hypothesis = _formula(example.nl_conclusion)
423
+
424
+ result = {
425
+ "label": example.label,
426
+ "premises": [p.to_unicode_str() for p in premises],
427
+ "hypothesis": hypothesis.to_unicode_str(),
428
+ }
429
+
430
+ logic_type = example.meta.get("logic_type")
431
+ if logic_type == NM_LOGIC_TYPE:
432
+ found = minimal_models(premises, circumscribed, max_size=max_size,
433
+ max_candidates=max_candidates,
434
+ extra_signature=[hypothesis])
435
+ if not found:
436
+ raise ValueError(
437
+ f"logicbench: example {example.id}: "
438
+ "semantics.nonmonotonic found NO minimal model of the "
439
+ f"translated premises within max_size={max_size} — this "
440
+ "route refuses rather than trust minimal_entails's vacuous "
441
+ "'True' for an empty model set (which could mean the "
442
+ "premises are unsatisfiable, or just that the bound is too "
443
+ "small to tell). Widen max_size, or check the translation.")
444
+ entailed = minimal_entails(premises, hypothesis, circumscribed,
445
+ max_size=max_size,
446
+ max_candidates=max_candidates)
447
+ result.update(predicted=("yes" if entailed else "no"),
448
+ route="nonmonotonic", minimal_model_count=len(found))
449
+ return result
450
+
451
+ if logic_type not in _CLASSICAL_LOGIC_TYPES:
452
+ raise ValueError(
453
+ f"logicbench: example {example.id}: unsupported "
454
+ f"logic_type {logic_type!r} — solve_example decides "
455
+ f"{sorted(_CLASSICAL_LOGIC_TYPES)} via api.prove and "
456
+ f"{NM_LOGIC_TYPE!r} via semantics.nonmonotonic, nothing else.")
457
+
458
+ verdict = api.prove(hypothesis, premises, **prove_kwargs)
459
+ if verdict.status == "proved":
460
+ predicted: Optional[str] = "yes"
461
+ elif verdict.status == "refuted":
462
+ predicted = "no"
463
+ else: # "unknown"
464
+ predicted = None
465
+ result.update(predicted=predicted, route="classical",
466
+ verdict=verdict.to_dict())
467
+ return result