unicode-logic-kit 0.31.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (237) hide show
  1. unicode_logic_kit/__init__.py +385 -0
  2. unicode_logic_kit/__main__.py +520 -0
  3. unicode_logic_kit/_deadline.py +219 -0
  4. unicode_logic_kit/ace/__init__.py +126 -0
  5. unicode_logic_kit/ace/_align.py +135 -0
  6. unicode_logic_kit/ace/chem_lexicon.py +128 -0
  7. unicode_logic_kit/ace/drs_reader.py +570 -0
  8. unicode_logic_kit/ace/mapping.py +666 -0
  9. unicode_logic_kit/ace/reverse_modal.py +138 -0
  10. unicode_logic_kit/ace/runner.py +551 -0
  11. unicode_logic_kit/ace/translate.py +452 -0
  12. unicode_logic_kit/ace/verbalize.py +1070 -0
  13. unicode_logic_kit/api.py +1284 -0
  14. unicode_logic_kit/atp/__init__.py +177 -0
  15. unicode_logic_kit/atp/_ascii_names.py +113 -0
  16. unicode_logic_kit/atp/_html.py +72 -0
  17. unicode_logic_kit/atp/_substructural_input.py +228 -0
  18. unicode_logic_kit/atp/_tff_problem.py +715 -0
  19. unicode_logic_kit/atp/_tptp_problem.py +1111 -0
  20. unicode_logic_kit/atp/_writer_support.py +289 -0
  21. unicode_logic_kit/atp/clingo_backend.py +1180 -0
  22. unicode_logic_kit/atp/cvc5_backend.py +1385 -0
  23. unicode_logic_kit/atp/eprover_backend.py +732 -0
  24. unicode_logic_kit/atp/finite_domain.py +1055 -0
  25. unicode_logic_kit/atp/fitch.py +1547 -0
  26. unicode_logic_kit/atp/fitch_search.py +551 -0
  27. unicode_logic_kit/atp/hets_backend.py +339 -0
  28. unicode_logic_kit/atp/hybrid_down.py +120 -0
  29. unicode_logic_kit/atp/incremental.py +250 -0
  30. unicode_logic_kit/atp/kripke_enum.py +741 -0
  31. unicode_logic_kit/atp/lambek.py +436 -0
  32. unicode_logic_kit/atp/leo3_backend.py +332 -0
  33. unicode_logic_kit/atp/linear.py +738 -0
  34. unicode_logic_kit/atp/lj.py +705 -0
  35. unicode_logic_kit/atp/logic_backends.py +566 -0
  36. unicode_logic_kit/atp/ltl_tableau.py +1084 -0
  37. unicode_logic_kit/atp/minizinc_backend.py +1402 -0
  38. unicode_logic_kit/atp/modal_tableau.py +1382 -0
  39. unicode_logic_kit/atp/nanocop_backend.py +410 -0
  40. unicode_logic_kit/atp/portfolio.py +489 -0
  41. unicode_logic_kit/atp/protocol.py +1803 -0
  42. unicode_logic_kit/atp/prover9_entailment.py +1153 -0
  43. unicode_logic_kit/atp/resolution.py +1376 -0
  44. unicode_logic_kit/atp/resolution_check.py +1114 -0
  45. unicode_logic_kit/atp/sequent.py +1050 -0
  46. unicode_logic_kit/atp/tableau.py +921 -0
  47. unicode_logic_kit/atp/tableau_check.py +543 -0
  48. unicode_logic_kit/atp/tptp_ncl.py +811 -0
  49. unicode_logic_kit/atp/tptp_tff.py +1546 -0
  50. unicode_logic_kit/atp/tstp.py +1333 -0
  51. unicode_logic_kit/atp/tstp_check.py +1096 -0
  52. unicode_logic_kit/atp/twee_backend.py +236 -0
  53. unicode_logic_kit/atp/twee_check.py +711 -0
  54. unicode_logic_kit/atp/twee_entailment.py +953 -0
  55. unicode_logic_kit/atp/vampire_entailment.py +540 -0
  56. unicode_logic_kit/atp/z3_arith.py +470 -0
  57. unicode_logic_kit/atp/z3_equivalence.py +36 -0
  58. unicode_logic_kit/atp/z3_fuzzy.py +362 -0
  59. unicode_logic_kit/atp/z3_input.py +500 -0
  60. unicode_logic_kit/atp/z3_models.py +208 -0
  61. unicode_logic_kit/chem/__init__.py +88 -0
  62. unicode_logic_kit/chem/_naming.py +284 -0
  63. unicode_logic_kit/chem/cache.py +185 -0
  64. unicode_logic_kit/chem/interop.py +244 -0
  65. unicode_logic_kit/chem/mol.py +525 -0
  66. unicode_logic_kit/chem/signature.py +112 -0
  67. unicode_logic_kit/comorphism.py +497 -0
  68. unicode_logic_kit/dl/__init__.py +384 -0
  69. unicode_logic_kit/dl/classification.py +227 -0
  70. unicode_logic_kit/dl/concepts.py +632 -0
  71. unicode_logic_kit/dl/datatypes.py +818 -0
  72. unicode_logic_kit/dl/owl_functional.py +2433 -0
  73. unicode_logic_kit/dl/owl_manchester.py +1637 -0
  74. unicode_logic_kit/dl/owl_reasoner.py +790 -0
  75. unicode_logic_kit/dl/parser.py +391 -0
  76. unicode_logic_kit/dl/tableau.py +4048 -0
  77. unicode_logic_kit/dl/translate.py +2704 -0
  78. unicode_logic_kit/drt/__init__.py +94 -0
  79. unicode_logic_kit/drt/export.py +179 -0
  80. unicode_logic_kit/drt/nodes.py +506 -0
  81. unicode_logic_kit/drt/parser.py +965 -0
  82. unicode_logic_kit/drt/resolve.py +195 -0
  83. unicode_logic_kit/drt/reverse.py +175 -0
  84. unicode_logic_kit/eval/__init__.py +106 -0
  85. unicode_logic_kit/eval/batch.py +382 -0
  86. unicode_logic_kit/eval/canonical.py +663 -0
  87. unicode_logic_kit/eval/chem_batch.py +606 -0
  88. unicode_logic_kit/eval/converses.py +200 -0
  89. unicode_logic_kit/eval/datasets/__init__.py +136 -0
  90. unicode_logic_kit/eval/datasets/_base.py +263 -0
  91. unicode_logic_kit/eval/datasets/_proofwriter_proof.py +422 -0
  92. unicode_logic_kit/eval/datasets/c3po.py +678 -0
  93. unicode_logic_kit/eval/datasets/folio.py +158 -0
  94. unicode_logic_kit/eval/datasets/fracas.py +418 -0
  95. unicode_logic_kit/eval/datasets/groves.py +191 -0
  96. unicode_logic_kit/eval/datasets/logicbench.py +467 -0
  97. unicode_logic_kit/eval/datasets/logicnli.py +303 -0
  98. unicode_logic_kit/eval/datasets/malls.py +133 -0
  99. unicode_logic_kit/eval/datasets/pfolio.py +594 -0
  100. unicode_logic_kit/eval/datasets/pmb.py +242 -0
  101. unicode_logic_kit/eval/datasets/prontoqa.py +611 -0
  102. unicode_logic_kit/eval/datasets/proofwriter.py +1431 -0
  103. unicode_logic_kit/eval/datasets/proverqa.py +674 -0
  104. unicode_logic_kit/eval/datasets/willow.py +478 -0
  105. unicode_logic_kit/eval/equivalence.py +466 -0
  106. unicode_logic_kit/eval/exercise_gen.py +533 -0
  107. unicode_logic_kit/eval/explain.py +791 -0
  108. unicode_logic_kit/eval/generality.py +750 -0
  109. unicode_logic_kit/eval/metric_hf.py +458 -0
  110. unicode_logic_kit/eval/predicate_match.py +343 -0
  111. unicode_logic_kit/eval/theory_check.py +1170 -0
  112. unicode_logic_kit/eval/validate.py +306 -0
  113. unicode_logic_kit/fol/__init__.py +177 -0
  114. unicode_logic_kit/fol/_atom_keys.py +510 -0
  115. unicode_logic_kit/fol/_fol_nodes.py +3586 -0
  116. unicode_logic_kit/fol/_free_parameters.py +105 -0
  117. unicode_logic_kit/fol/_ho_nodes.py +448 -0
  118. unicode_logic_kit/fol/_hybrid_nodes.py +308 -0
  119. unicode_logic_kit/fol/_identifiers.py +1091 -0
  120. unicode_logic_kit/fol/_lambek_nodes.py +112 -0
  121. unicode_logic_kit/fol/_linear_nodes.py +352 -0
  122. unicode_logic_kit/fol/_modal_nodes.py +1467 -0
  123. unicode_logic_kit/fol/_msfl_nodes.py +2196 -0
  124. unicode_logic_kit/fol/_numeral_symbols.py +231 -0
  125. unicode_logic_kit/fol/_so_nodes.py +200 -0
  126. unicode_logic_kit/fol/_symbol_names.py +81 -0
  127. unicode_logic_kit/fol/_team_nodes.py +181 -0
  128. unicode_logic_kit/fol/_tptp_symbols.py +551 -0
  129. unicode_logic_kit/fol/_truth_constants.py +117 -0
  130. unicode_logic_kit/fol/casl_export.py +1135 -0
  131. unicode_logic_kit/fol/casl_import.py +929 -0
  132. unicode_logic_kit/fol/derivation.py +367 -0
  133. unicode_logic_kit/fol/dialect_detect.py +70 -0
  134. unicode_logic_kit/fol/dialect_repair.py +537 -0
  135. unicode_logic_kit/fol/frames.py +637 -0
  136. unicode_logic_kit/fol/grammars/terminals.lark +31 -0
  137. unicode_logic_kit/fol/lambda_tools.py +297 -0
  138. unicode_logic_kit/fol/latex_input.py +429 -0
  139. unicode_logic_kit/fol/modal_translation.py +944 -0
  140. unicode_logic_kit/fol/msflparser.py +1033 -0
  141. unicode_logic_kit/fol/naming.py +422 -0
  142. unicode_logic_kit/fol/nodes.py +241 -0
  143. unicode_logic_kit/fol/normalforms.py +492 -0
  144. unicode_logic_kit/fol/pal.py +287 -0
  145. unicode_logic_kit/fol/prolog_export.py +566 -0
  146. unicode_logic_kit/fol/prolog_input.py +505 -0
  147. unicode_logic_kit/fol/prover9_input.py +1325 -0
  148. unicode_logic_kit/fol/qml.py +1760 -0
  149. unicode_logic_kit/fol/qmltp_input.py +525 -0
  150. unicode_logic_kit/fol/sanitize.py +221 -0
  151. unicode_logic_kit/fol/serialize.py +79 -0
  152. unicode_logic_kit/fol/signature.py +1290 -0
  153. unicode_logic_kit/fol/simplify_check.py +544 -0
  154. unicode_logic_kit/fol/spans.py +594 -0
  155. unicode_logic_kit/fol/tptp_input.py +1503 -0
  156. unicode_logic_kit/fol/tptp_repair.py +941 -0
  157. unicode_logic_kit/fol/unification.py +157 -0
  158. unicode_logic_kit/fol/verbalize.py +263 -0
  159. unicode_logic_kit/hets/__init__.py +163 -0
  160. unicode_logic_kit/hets/bridge.py +142 -0
  161. unicode_logic_kit/hets/client.py +748 -0
  162. unicode_logic_kit/hets/docker.py +420 -0
  163. unicode_logic_kit/hets/dol.py +712 -0
  164. unicode_logic_kit/hets/haskell_json.py +355 -0
  165. unicode_logic_kit/hets/owl_backend.py +794 -0
  166. unicode_logic_kit/hets/owl_cli.py +598 -0
  167. unicode_logic_kit/hets/symbols.py +512 -0
  168. unicode_logic_kit/hol/__init__.py +140 -0
  169. unicode_logic_kit/hol/_ho_common.py +323 -0
  170. unicode_logic_kit/hol/_isabelle_binders.py +125 -0
  171. unicode_logic_kit/hol/classical.py +812 -0
  172. unicode_logic_kit/hol/deepshallow/__init__.py +45 -0
  173. unicode_logic_kit/hol/deepshallow/_common.py +177 -0
  174. unicode_logic_kit/hol/deepshallow/conditional.py +225 -0
  175. unicode_logic_kit/hol/deepshallow/intuitionistic.py +181 -0
  176. unicode_logic_kit/hol/deepshallow/modal.py +217 -0
  177. unicode_logic_kit/hol/deepshallow/qml.py +406 -0
  178. unicode_logic_kit/hol/deepshallow/relevant.py +206 -0
  179. unicode_logic_kit/hol/free.py +753 -0
  180. unicode_logic_kit/hol/goedel.py +336 -0
  181. unicode_logic_kit/hol/ho_modal.py +1743 -0
  182. unicode_logic_kit/hol/intuitionistic.py +403 -0
  183. unicode_logic_kit/hol/isabelle_conditional.py +593 -0
  184. unicode_logic_kit/hol/isabelle_modal.py +1908 -0
  185. unicode_logic_kit/hol/isabelle_relevant.py +412 -0
  186. unicode_logic_kit/hol/isabelle_runner.py +1147 -0
  187. unicode_logic_kit/hol/isabelle_substructural.py +884 -0
  188. unicode_logic_kit/hol/lean.py +1018 -0
  189. unicode_logic_kit/hol/manyvalued.py +921 -0
  190. unicode_logic_kit/hol/secondorder.py +687 -0
  191. unicode_logic_kit/hol/thf_modal.py +941 -0
  192. unicode_logic_kit/hol/thirdorder.py +397 -0
  193. unicode_logic_kit/ilp/__init__.py +89 -0
  194. unicode_logic_kit/ilp/readback.py +389 -0
  195. unicode_logic_kit/ilp/separation.py +153 -0
  196. unicode_logic_kit/ilp/task.py +730 -0
  197. unicode_logic_kit/logic.py +163 -0
  198. unicode_logic_kit/mcp/__init__.py +28 -0
  199. unicode_logic_kit/mcp/__main__.py +5 -0
  200. unicode_logic_kit/mcp/chem_tools.py +1031 -0
  201. unicode_logic_kit/mcp/server.py +2453 -0
  202. unicode_logic_kit/mcp/syntax_spec.py +681 -0
  203. unicode_logic_kit/prob/__init__.py +53 -0
  204. unicode_logic_kit/prob/_bdd.py +225 -0
  205. unicode_logic_kit/prob/_column_gen.py +668 -0
  206. unicode_logic_kit/prob/distribution.py +686 -0
  207. unicode_logic_kit/prob/nilsson.py +470 -0
  208. unicode_logic_kit/py.typed +0 -0
  209. unicode_logic_kit/semantics/__init__.py +137 -0
  210. unicode_logic_kit/semantics/_modal_reject.py +156 -0
  211. unicode_logic_kit/semantics/action_models.py +466 -0
  212. unicode_logic_kit/semantics/asp_models.py +1200 -0
  213. unicode_logic_kit/semantics/conditional.py +580 -0
  214. unicode_logic_kit/semantics/dynamic_epistemic.py +95 -0
  215. unicode_logic_kit/semantics/free_logic.py +913 -0
  216. unicode_logic_kit/semantics/fuzzy.py +384 -0
  217. unicode_logic_kit/semantics/fuzzy_kripke.py +442 -0
  218. unicode_logic_kit/semantics/intuitionistic.py +581 -0
  219. unicode_logic_kit/semantics/kripke.py +1139 -0
  220. unicode_logic_kit/semantics/manyvalued.py +580 -0
  221. unicode_logic_kit/semantics/matrix.py +342 -0
  222. unicode_logic_kit/semantics/model_eval.py +1135 -0
  223. unicode_logic_kit/semantics/modelfinder.py +1036 -0
  224. unicode_logic_kit/semantics/nonmonotonic.py +372 -0
  225. unicode_logic_kit/semantics/relevant.py +331 -0
  226. unicode_logic_kit/semantics/secondorder.py +657 -0
  227. unicode_logic_kit/semantics/structures.py +352 -0
  228. unicode_logic_kit/semantics/tarski.py +975 -0
  229. unicode_logic_kit/semantics/team.py +315 -0
  230. unicode_logic_kit/semantics/team_translation.py +416 -0
  231. unicode_logic_kit/semantics/thirdorder.py +358 -0
  232. unicode_logic_kit/semantics/tnorm.py +85 -0
  233. unicode_logic_kit/semantics/truthtable.py +201 -0
  234. unicode_logic_kit-0.31.0.dist-info/METADATA +333 -0
  235. unicode_logic_kit-0.31.0.dist-info/RECORD +237 -0
  236. unicode_logic_kit-0.31.0.dist-info/WHEEL +4 -0
  237. unicode_logic_kit-0.31.0.dist-info/licenses/LICENSE +21 -0
@@ -0,0 +1,674 @@
1
+ """Adapter for the ProverQA dataset (Qi et al., "Large Language Models Meet
2
+ Symbolic Provers for Logical Reasoning Evaluation", ICLR 2025; dataset built
3
+ by the ProverGen framework) — local JSONL only, no network access.
4
+
5
+ Source and verified schema
6
+ ---------------------------
7
+ Source verified 2026-08-12 directly against
8
+ https://huggingface.co/datasets/opendatalab/ProverQA (repo file listing via
9
+ ``https://huggingface.co/api/datasets/opendatalab/ProverQA``, record schema
10
+ via the raw files fetched from
11
+ ``https://huggingface.co/datasets/opendatalab/ProverQA/resolve/main/<path>``,
12
+ cross-checked against the field descriptions and JSON example in the
13
+ repository's own ``README.md``). Code for the generation pipeline (not
14
+ consulted for this loader, provided for context only):
15
+ https://github.com/opendatalab/ProverGen.
16
+
17
+ The repository has exactly four data files, in TWO INCOMPATIBLE shapes:
18
+
19
+ * ``dev/easy.json``, ``dev/medium.json``, ``dev/hard.json`` — the paper's
20
+ 1,500-instance evaluation benchmark (500 per difficulty tier: easy = 1-2
21
+ reasoning steps, medium = 3-5, hard = 6-9), each a JSON ARRAY of flat
22
+ objects. Verified by downloading and inspecting all 1,500 rows: every
23
+ single row has EXACTLY these 8 keys, no more, no fewer —
24
+
25
+ * ``"id"`` — ``int``, 0-based, unique WITHIN one tier file
26
+ (0..499 in each of the three files) but NOT globally unique — the same
27
+ id appears in all three tiers. A caller loading more than one tier must
28
+ pass a distinguishing ``tier`` to :func:`load_proverqa` (see below) or
29
+ the resulting ids collide.
30
+ * ``"options"`` — ``list[str]``, always exactly 3 entries in every
31
+ verified row: ``["A) True", "B) False", "C) Uncertain"]`` verbatim
32
+ (letter-and-text order never varies across the 1,500 rows checked).
33
+ * ``"answer"`` — ``str``, one of ``"A"``/``"B"``/``"C"``; verified
34
+ for all 1,500 rows that ``options[ord(answer) - ord("A")]`` starts with
35
+ ``f"{answer})"`` — the letter always indexes its own option correctly.
36
+ * ``"question"`` — ``str``, the FULL interrogative prompt, e.g.
37
+ ``"Based on the above information, is the following statement true,
38
+ false, or uncertain? Brecken has never experienced heartbreak."`` —
39
+ UNLIKE FOLIO's ``conclusion`` field, this is NOT a bare declarative
40
+ sentence; the bare statement is not separately provided upstream, and
41
+ this adapter does not attempt to extract it with a regex heuristic (that
42
+ would be an invented field, not a verified one).
43
+ * ``"reasoning"`` — ``str``, a free-text chain-of-thought trace.
44
+ * ``"context"`` — ``str``, the natural-language premises,
45
+ concatenated. Verified equal to ``" ".join(nl2fol.keys())`` for 1,497 of
46
+ the 1,500 rows; the 3 exceptions (e.g. ``dev/easy.json`` id 361) have a
47
+ premise sentence repeated VERBATIM TWICE in ``context`` but only once in
48
+ ``nl2fol`` — ``nl2fol`` is a JSON object keyed by sentence text, so a
49
+ literal duplicate premise silently collapses to one entry. This is an
50
+ upstream data quirk (verified present in the raw file), not introduced
51
+ by this adapter; it means ``fol_premises`` can have fewer entries than
52
+ ``context`` has sentences for a handful of rows.
53
+ * ``"nl2fol"`` — ``dict[str, str]``, NL premise sentence -> its
54
+ gold FOL translation, insertion-ordered (Python/JSON preserve object key
55
+ order, and ``dict.keys()``/``dict.values()`` iterate in matched,
56
+ corresponding order for the same dict) — this is the exact
57
+ ``nl_premises``/``fol_premises`` pairing this adapter uses.
58
+ * ``"conclusion_fol"`` — ``str``, the gold FOL for the statement embedded
59
+ in ``question``.
60
+
61
+ * ``train/provergen-5000.json`` — the paper's 5,000-instance fine-tuning
62
+ split, a JSON array of flat objects. Verified by downloading and
63
+ inspecting all 5,000 rows: every row has EXACTLY 4 keys —
64
+ ``"system"``/``"instruction"``/``"input"``/``"output"``, an
65
+ instruction-tuning prompt/response pair (the context, question, and
66
+ options are embedded as free NATURAL-LANGUAGE TEXT inside ``instruction``,
67
+ and the gold reasoning/answer letter as free text inside ``output``, as a
68
+ JSON-in-a-string). It carries **NO** structured ``id``/``nl2fol``/
69
+ ``conclusion_fol``/``answer``/``options`` fields at all — no FOL
70
+ annotation of any kind is recoverable from this file without re-parsing
71
+ free text. **This loader does NOT support the train split** — only
72
+ ``dev/easy.json``/``dev/medium.json``/``dev/hard.json`` carry the
73
+ structured FOL fields :class:`~unicode_logic_kit.eval.datasets.DatasetExample`
74
+ needs.
75
+
76
+ CRITICAL notation finding (parse rate is 0%, verified, not a bug)
77
+ -------------------------------------------------------------------
78
+ Every one of the 17,342 real FOL strings in the three ``dev/*.json`` files
79
+ (every ``nl2fol`` value plus every ``conclusion_fol``, across all 1,500 rows)
80
+ was run through :func:`unicode_logic_kit.api.parse_any` as part of verifying
81
+ this adapter. **0 of 17,342 parsed** under any dialect. The cause is a
82
+ NOTATION mismatch, not a data-quality defect in ProverQA: ProverQA's gold
83
+ FOL uses lower-case ``snake_case`` predicate identifiers with underscores
84
+ (e.g. ``has_experienced_heartbreak``) and Capitalized proper-noun constants
85
+ (e.g. ``Brecken``) — the OPPOSITE convention from this kit's own grammar
86
+ (``unicode_logic_kit/fol/grammars/terminals.lark``: ``PREDICATE`` must start
87
+ uppercase with no underscore, e.g. ``Cat``; the constant/function ``NAME``
88
+ terminal must start lowercase with no underscore, e.g. ``tom`` — the
89
+ convention FOLIO's and MALLS's gold data already happen to follow, which is
90
+ WHY those two adapters' fixtures mostly parse and this one's does not).
91
+ ProverQA's own README states its FOL was "validated through automated
92
+ symbolic provers (Prover9)" — a different tool with a different accepted
93
+ surface syntax than this kit's parser.
94
+
95
+ This adapter resolves the mismatch with a DEDICATED IMPORT-TIME GRAMMAR
96
+ rather than a lexical rewrite: :data:`_PROVERQA_GRAMMAR` parses the dataset's
97
+ own notation (operators verified against the real corpus), and the resulting
98
+ tree is converted into kit nodes with identifiers renamed into the kit's
99
+ convention (:class:`_NameConverter` — injective per namespace, constants
100
+ that would become variables refused; see the dialect comment block above the
101
+ grammar). With the default ``convert_fol=True``, ``fol_premises`` /
102
+ ``fol_conclusion`` therefore hold KIT-notation strings that parse under
103
+ ``api.parse_any``, while ``meta`` keeps the verbatim upstream strings
104
+ (``original_fol_premises``/``original_fol_conclusion``) and the
105
+ changed-names mapping (``fol_name_mapping``) — the gold data remains fully
106
+ recoverable and every rename is on the record. ``convert_fol=False``
107
+ restores the raw pass-through, whose 0% kit-parse rate is pinned by
108
+ ``tests/test_datasets_proverqa.py``. :func:`solve_example` then decides a
109
+ converted example end-to-end (premises ⊨ conclusion → ``"A"``, premises ⊨
110
+ ¬conclusion → ``"B"``, else ``"C"``) against the gold ``answer``.
111
+
112
+ Also observed, and left as-is (not fixed): ``dev/easy.json`` row ``id=7``
113
+ uses the predicate ``resolves_conflict_peacefully`` (singular "conflict") in
114
+ one ``nl2fol`` rule but ``resolves_conflicts_peacefully`` (plural
115
+ "conflicts") in ``conclusion_fol`` — an internal predicate-name
116
+ inconsistency verified present in the raw upstream file itself, not
117
+ introduced by this adapter. It is exactly the kind of defect
118
+ :func:`~unicode_logic_kit.eval.datasets.audit_examples` exists to surface (as
119
+ an ``arity_conflict``/mismatched-signature symptom) IF the formulas parsed
120
+ at all — here it is masked by the notation mismatch above, since neither
121
+ formula parses in the first place.
122
+
123
+ Upstream distribution format and this loader
124
+ -----------------------------------------------
125
+ Each ``dev/<tier>.json`` file is a JSON ARRAY (like MALLS's upstream
126
+ distribution), NOT JSONL. This loader, like
127
+ :func:`~unicode_logic_kit.eval.datasets.folio.load_folio` and
128
+ :func:`~unicode_logic_kit.eval.datasets.malls.load_malls`, reads local JSONL
129
+ (one JSON object per line) for a uniform, streaming-friendly adapter surface
130
+ across this subpackage — convert an upstream ``dev/<tier>.json`` file to
131
+ JSONL first (e.g. ``jq -c '.[]' dev/easy.json > proverqa_easy.jsonl``) before
132
+ calling :func:`load_proverqa`.
133
+
134
+ Because the tier (easy/medium/hard) is which FILE a row came from, not a
135
+ field inside the row, :func:`load_proverqa` takes an optional ``tier``
136
+ keyword purely so the loader can (a) namespace ids so multiple tiers can be
137
+ loaded into one corpus without id collisions (see the ``"id"`` bullet
138
+ above), and (b) record the tier in ``meta``. It is caller-supplied, not
139
+ inferred from the file content — passing the wrong ``tier`` for a given file
140
+ is a caller error this loader cannot detect.
141
+
142
+ Field mapping onto :class:`~unicode_logic_kit.eval.datasets.DatasetExample`
143
+ (a deliberate design choice for the fields the source has no direct
144
+ equivalent for, exactly like :mod:`~unicode_logic_kit.eval.datasets.malls`
145
+ documents its own mapping choices):
146
+
147
+ * ``id``: ``f"proverqa:{tier}:{record['id']}"`` when ``tier`` is given,
148
+ else ``f"proverqa:{record['id']}"`` (see the id-collision note above).
149
+ Falls back to a positional ``f"proverqa:{tier or 'untiered'}:pos{line_no}"``
150
+ ONLY if the record has no ``"id"`` key at all (defensive; never observed
151
+ in verified real data — every one of the 1,500 rows has it).
152
+ * ``nl_premises`` / ``fol_premises``: ``tuple(nl2fol.keys())`` /
153
+ ``tuple(nl2fol.values())`` — see the ``"nl2fol"`` bullet above for why
154
+ this pairing is safe. ``()`` if ``"nl2fol"`` is absent or not a mapping.
155
+ * ``nl_conclusion``: the verbatim ``"question"`` string (the interrogative
156
+ prompt, NOT a bare declarative — see the ``"question"`` bullet above).
157
+ * ``fol_conclusion``: the verbatim ``"conclusion_fol"`` string.
158
+ * ``label``: the verbatim ``"answer"`` letter (``"A"``/``"B"``/``"C"``) —
159
+ kept as the dataset's own raw vocabulary rather than resolved to
160
+ ``"True"``/``"False"``/``"Uncertain"`` text, so nothing is synthesised
161
+ that is not literally the ``"answer"`` field; a caller who wants the
162
+ semantic text can resolve it themselves from ``meta["options"]``.
163
+ * ``meta``: every other record key verbatim (``"options"``, ``"context"``,
164
+ ``"reasoning"``, and any future/unrecognised key), plus ``"line_no"`` and
165
+ (when given) ``"tier"``.
166
+
167
+ License
168
+ -------
169
+ **UNSPECIFIED** by the authors — verified against both the Hugging Face
170
+ dataset card's metadata (``cardData`` has no ``license`` key at all) and the
171
+ README's own "License and Ethics" section, which states only that the
172
+ INPUT SOURCES used to build the dataset comply with their own licenses
173
+ ("MIT for names, WordNet for keywords") — that is a statement about the
174
+ dataset's INGREDIENTS, not a license grant for the ProverQA data itself. No
175
+ CC/MIT/Apache/etc. license is declared for the dataset artifact. Treat as
176
+ all-rights-reserved / contact the authors before any redistribution beyond
177
+ the kind of small local research fixture this module's own test fixture is.
178
+ This loader itself never downloads or redistributes ProverQA data — it only
179
+ reads a LOCAL file the caller already obtained.
180
+
181
+ Citation (from the repository's ``README.md``)::
182
+
183
+ @inproceedings{qi2025large,
184
+ title={Large Language Models Meet Symbolic Provers for Logical
185
+ Reasoning Evaluation},
186
+ author={Chengwen Qi and Ren Ma and Bowen Li and He Du and
187
+ Binyuan Hui and Jinwang Wu and Yuanjun Laili and Conghui He},
188
+ booktitle={The Thirteenth International Conference on Learning
189
+ Representations},
190
+ year={2025},
191
+ url={https://openreview.net/forum?id=C25SgeXWjE}
192
+ }
193
+
194
+ This module never downloads anything — obtain and convert the data yourself
195
+ and pass its local JSONL path to :func:`load_proverqa`.
196
+ """
197
+
198
+ import json
199
+ import re
200
+ from pathlib import Path
201
+ from typing import Dict, FrozenSet, Iterator, List, Optional, Sequence, Tuple, Union
202
+
203
+ from lark import Lark, Transformer
204
+ from lark.exceptions import LarkError, VisitError
205
+
206
+ from ...fol.nodes import (
207
+ Node, Atom, Not, And, Or, Xor, Implies, Iff, Quantifier,
208
+ Variable, Constant, free_variables,
209
+ )
210
+ from ._base import DatasetExample, _register_dataset_info
211
+
212
+ __all__ = [
213
+ "load_proverqa",
214
+ "parse_proverqa_formula", "convert_proverqa_formulas",
215
+ "solve_example",
216
+ ]
217
+
218
+ _VALID_TIERS = ("easy", "medium", "hard")
219
+
220
+ _register_dataset_info(
221
+ "proverqa",
222
+ license=(
223
+ "UNSPECIFIED by the authors (no license key in the HF dataset card; "
224
+ "the README's 'License and Ethics' section only states that the "
225
+ "dataset's INPUT SOURCES — names/keywords — comply with their own "
226
+ "licenses, which is not a license grant for the ProverQA data "
227
+ "itself). Treat as all-rights-reserved until the authors clarify."
228
+ ),
229
+ source_url="https://huggingface.co/datasets/opendatalab/ProverQA",
230
+ citation_hint=(
231
+ "Qi, Chengwen, et al. \"Large Language Models Meet Symbolic Provers "
232
+ "for Logical Reasoning Evaluation.\" ICLR 2025. "
233
+ "https://openreview.net/forum?id=C25SgeXWjE"
234
+ ),
235
+ )
236
+
237
+
238
+ # --------------------------------------------------------------------------- #
239
+ # The ProverQA FOL dialect: an import-time grammar + conversion to kit ASTs
240
+ #
241
+ # ProverQA's gold FOL is honest first-order logic in a NOTATION this kit's own
242
+ # grammar refuses (snake_case predicates, Capitalized constants — see the
243
+ # module docstring's "CRITICAL notation finding"). Instead of lexically
244
+ # rewriting the strings, the loader parses them with THIS dedicated grammar
245
+ # (operators verified against the real corpus: ¬ ∧ ∨ ⊕ → ↔ ∀ ∃, standard
246
+ # precedence ¬ > ∧ > ∨ > ⊕ > → (right-assoc) > ↔; ProverQA parenthesises
247
+ # mixed nesting anyway) and converts the resulting tree into kit nodes,
248
+ # renaming identifiers into the kit's convention along the way:
249
+ #
250
+ # * predicate ``has_experienced_heartbreak`` → ``HasExperiencedHeartbreak``
251
+ # (underscore parts joined, each part's FIRST letter upcased, interior
252
+ # capitalisation preserved — minimal change, not a re-styling),
253
+ # * constant ``Brecken`` → ``brecken`` (first letter downcased; snake_case
254
+ # constants fold the same way with camelCase joints),
255
+ # * a term that lexes as a kit VARIABLE (``[a-z][0-9]*``) stays a variable.
256
+ #
257
+ # Two hard refusals keep the conversion honest: a constant whose converted
258
+ # form would lex as a VARIABLE (e.g. the constant ``A`` → ``a``) raises
259
+ # instead of silently changing quantification semantics, and the mapping is
260
+ # INJECTIVE per namespace across one whole example — two source names that
261
+ # would collapse onto one target (``p_a`` and ``pA`` → ``PA``) raise rather
262
+ # than merging distinct symbols. The loader stores the original strings and
263
+ # the (changed-entries-only) mapping in ``meta``, so nothing is hidden.
264
+ # --------------------------------------------------------------------------- #
265
+
266
+ _PROVERQA_GRAMMAR = r"""
267
+ ?start: formula
268
+ ?formula: iff
269
+ ?iff: implies ("↔" implies)*
270
+ ?implies: xor "→" implies -> implies
271
+ | xor
272
+ ?xor: disj ("⊕" disj)*
273
+ ?disj: conj ("∨" conj)*
274
+ ?conj: unary ("∧" unary)*
275
+ ?unary: "¬" unary -> neg
276
+ | "∀" IDENT unary -> forall
277
+ | "∃" IDENT unary -> exists
278
+ | atom
279
+ | "(" formula ")"
280
+ atom: IDENT "(" IDENT ("," IDENT)* ")"
281
+ IDENT: /[A-Za-z][A-Za-z0-9_]*/
282
+ %import common.WS
283
+ %ignore WS
284
+ """
285
+
286
+ _KIT_PREDICATE_RE = re.compile(r"[A-Z][a-zA-Z0-9]*\Z")
287
+ _KIT_CONSTANT_RE = re.compile(r"[a-z][a-zA-Z0-9]*[a-zA-Z][a-zA-Z0-9]*\Z")
288
+ _KIT_VARIABLE_RE = re.compile(r"[a-z][0-9]*\Z")
289
+
290
+
291
+ class _NameConverter:
292
+ """Kit-convention renaming with a per-namespace injectivity guarantee.
293
+
294
+ One instance spans ONE example (all premises + the conclusion), so a
295
+ predicate keeps the same converted name everywhere it occurs and two
296
+ distinct source names can never collapse onto one target. ``pred_map`` /
297
+ ``const_map`` hold only the names that actually changed — exactly what
298
+ the loader records in ``meta["fol_name_mapping"]``.
299
+ """
300
+
301
+ def __init__(self) -> None:
302
+ self.pred_map: Dict[str, str] = {}
303
+ self.const_map: Dict[str, str] = {}
304
+ self._final: Dict[str, Dict[str, str]] = {"predicate": {}, "constant": {}}
305
+
306
+ def _claim(self, namespace: str, source: str, target: str) -> str:
307
+ owner = self._final[namespace].setdefault(target, source)
308
+ if owner != source:
309
+ raise ValueError(
310
+ f"proverqa: {namespace} names {owner!r} and {source!r} would "
311
+ f"both convert to {target!r} — the renaming must stay "
312
+ "injective, so this example cannot be converted.")
313
+ return target
314
+
315
+ def predicate(self, name: str) -> str:
316
+ if _KIT_PREDICATE_RE.match(name):
317
+ return self._claim("predicate", name, name)
318
+ parts = [p for p in name.split("_") if p]
319
+ if not parts:
320
+ raise ValueError(f"proverqa: predicate {name!r} is underscores only.")
321
+ target = "".join(p[0].upper() + p[1:] for p in parts)
322
+ if not _KIT_PREDICATE_RE.match(target):
323
+ raise ValueError(
324
+ f"proverqa: predicate {name!r} does not convert to a legal "
325
+ f"kit predicate (got {target!r}).")
326
+ self._claim("predicate", name, target)
327
+ self.pred_map[name] = target
328
+ return target
329
+
330
+ def constant(self, name: str) -> str:
331
+ if _KIT_CONSTANT_RE.match(name):
332
+ return self._claim("constant", name, name)
333
+ parts = [p for p in name.split("_") if p]
334
+ if not parts:
335
+ raise ValueError(f"proverqa: constant {name!r} is underscores only.")
336
+ head = parts[0][0].lower() + parts[0][1:]
337
+ target = head + "".join(p[0].upper() + p[1:] for p in parts[1:])
338
+ if _KIT_VARIABLE_RE.match(target):
339
+ raise ValueError(
340
+ f"proverqa: constant {name!r} would convert to {target!r}, "
341
+ "which this kit lexes as a VARIABLE — refusing the silent "
342
+ "semantics change.")
343
+ if not _KIT_CONSTANT_RE.match(target):
344
+ raise ValueError(
345
+ f"proverqa: constant {name!r} does not convert to a legal "
346
+ f"kit constant (got {target!r}).")
347
+ self._claim("constant", name, target)
348
+ self.const_map[name] = target
349
+ return target
350
+
351
+ def mapping(self) -> Dict[str, Dict[str, str]]:
352
+ return {"predicates": dict(self.pred_map), "constants": dict(self.const_map)}
353
+
354
+
355
+ class _ProverQATransformer(Transformer):
356
+ """Lark tree → kit :class:`~unicode_logic_kit.fol.nodes.Node`."""
357
+
358
+ def __init__(self, converter: _NameConverter) -> None:
359
+ super().__init__()
360
+ self._conv = converter
361
+
362
+ def _term(self, token) -> Node:
363
+ name = str(token)
364
+ if _KIT_VARIABLE_RE.match(name):
365
+ return Variable(name)
366
+ return Constant(self._conv.constant(name))
367
+
368
+ def _binder(self, token) -> Variable:
369
+ name = str(token)
370
+ if not _KIT_VARIABLE_RE.match(name):
371
+ raise ValueError(
372
+ f"proverqa: quantified variable {name!r} is not a kit "
373
+ "variable token ([a-z][0-9]*) — refusing to guess a rename "
374
+ "for a BOUND name.")
375
+ return Variable(name)
376
+
377
+ def atom(self, children) -> Node:
378
+ pred = self._conv.predicate(str(children[0]))
379
+ return Atom(pred, tuple(self._term(t) for t in children[1:]))
380
+
381
+ def neg(self, children) -> Node:
382
+ return Not(children[0])
383
+
384
+ def forall(self, children) -> Node:
385
+ return Quantifier("∀", self._binder(children[0]), children[1])
386
+
387
+ def exists(self, children) -> Node:
388
+ return Quantifier("∃", self._binder(children[0]), children[1])
389
+
390
+ def _fold(self, children, cls) -> Node:
391
+ node = children[0]
392
+ for right in children[1:]:
393
+ node = cls(node, right)
394
+ return node
395
+
396
+ def conj(self, children) -> Node:
397
+ return self._fold(children, And)
398
+
399
+ def disj(self, children) -> Node:
400
+ return self._fold(children, Or)
401
+
402
+ def xor(self, children) -> Node:
403
+ return self._fold(children, Xor)
404
+
405
+ def implies(self, children) -> Node:
406
+ if len(children) == 1:
407
+ return children[0]
408
+ return Implies(children[0], children[1])
409
+
410
+ def iff(self, children) -> Node:
411
+ return self._fold(children, Iff)
412
+
413
+
414
+ _proverqa_parser = Lark(_PROVERQA_GRAMMAR, parser="lalr")
415
+
416
+
417
+ def parse_proverqa_formula(text: str, *,
418
+ converter: Optional[_NameConverter] = None) -> Node:
419
+ """Parse ONE formula in ProverQA's FOL notation into a kit AST.
420
+
421
+ Identifiers are converted into this kit's naming convention (see the
422
+ dialect comment block above). Pass a shared ``converter`` to keep the
423
+ renaming consistent and injective across several formulas of one example
424
+ — :func:`convert_proverqa_formulas` does exactly that.
425
+
426
+ Raises:
427
+ lark.exceptions.LarkError: ``text`` is not in the dialect.
428
+ ValueError: an identifier cannot be converted without a collision or
429
+ a constant→variable semantics change.
430
+ """
431
+ conv = converter if converter is not None else _NameConverter()
432
+ tree = _proverqa_parser.parse(text)
433
+ try:
434
+ return _ProverQATransformer(conv).transform(tree)
435
+ except VisitError as exc:
436
+ # Lark wraps transformer exceptions; surface the converter's own
437
+ # ValueError unchanged so the documented error contract holds.
438
+ if isinstance(exc.orig_exc, ValueError):
439
+ raise exc.orig_exc from None
440
+ raise
441
+
442
+
443
+ def convert_proverqa_formulas(
444
+ texts: Sequence[str]) -> Tuple[Tuple[Node, ...], Dict[str, Dict[str, str]]]:
445
+ """Convert several ProverQA formulas with ONE shared, injective renaming.
446
+
447
+ Returns ``(nodes, mapping)`` where ``mapping`` is
448
+ ``{"predicates": {original: new}, "constants": {original: new}}`` holding
449
+ only the names that actually changed. Every returned node must be CLOSED
450
+ — a free variable (e.g. from a mis-scoped quantifier reading) raises
451
+ rather than yielding a silently defective formula.
452
+ """
453
+ conv = _NameConverter()
454
+ nodes: List[Node] = []
455
+ for text in texts:
456
+ node = parse_proverqa_formula(text, converter=conv)
457
+ free = free_variables(node)
458
+ if free:
459
+ raise ValueError(
460
+ f"proverqa: converted formula has free variable(s) "
461
+ f"{sorted(free)!r} — refusing (source: {text!r}).")
462
+ nodes.append(node)
463
+ return tuple(nodes), conv.mapping()
464
+
465
+
466
+ def solve_example(example: DatasetExample, *, on_indefinite: str = "label",
467
+ **prove_kwargs) -> dict:
468
+ """Decide one converted ProverQA example end-to-end via ``api.prove``.
469
+
470
+ ProverQA's three-way gold label maps onto classical entailment exactly:
471
+ ``"A"`` (True) iff premises ⊨ conclusion, ``"B"`` (False) iff premises ⊨
472
+ ¬conclusion, ``"C"`` (Uncertain) otherwise. This helper proves the
473
+ positive direction first and the negated one only when needed, and
474
+ returns ``{"predicted": letter, "verdict": ..., "verdict_negated": ...}``
475
+ (verdicts as dicts; ``verdict_negated`` is ``None`` when the positive
476
+ direction already settled it). Extra ``prove_kwargs`` go verbatim to
477
+ :func:`unicode_logic_kit.api.prove`, so the ATP is the caller's choice
478
+ (``backends=["vampire"]``, ``timeout=…``). It requires an example whose
479
+ formulas are in KIT notation — i.e. loaded with the default
480
+ ``convert_fol=True`` and without a recorded
481
+ ``meta["fol_conversion_error"]``; anything else raises ``ValueError``
482
+ rather than silently scoring garbage.
483
+
484
+ ``on_indefinite`` controls how a NON-DEFINITIVE prover outcome (status
485
+ ``unknown``/``error`` — timeout, hit bound, honest incompleteness) is
486
+ interpreted when neither direction was proved:
487
+
488
+ - ``"label"`` (default): predict ``"C"`` — correct whenever the chosen
489
+ prover is decisive on the fragment (z3 on ProverQA's dev tiers is).
490
+ - ``"abstain"``: ``"C"`` only when BOTH directions are definitively
491
+ REFUTED (underdetermination ESTABLISHED by countermodels); any
492
+ indefinite leg yields ``predicted=None``, so a prover timeout can
493
+ never be silently scored as a correct "Uncertain".
494
+ - ``"raise"``: like ``"abstain"`` but an indefinite leg raises
495
+ ``ValueError``.
496
+ """
497
+ from ... import api
498
+
499
+ if on_indefinite not in ("label", "abstain", "raise"):
500
+ raise ValueError(
501
+ f"proverqa: on_indefinite must be 'label', 'abstain' or 'raise', "
502
+ f"got {on_indefinite!r}")
503
+ if example.meta.get("fol_conversion_error"):
504
+ raise ValueError(
505
+ f"proverqa: example {example.id} carries a conversion error "
506
+ f"({example.meta['fol_conversion_error']}) — cannot solve it.")
507
+ if example.fol_conclusion is None:
508
+ raise ValueError(f"proverqa: example {example.id} has no conclusion.")
509
+
510
+ def _parse(text: str) -> Node:
511
+ parsed = api.parse_any(text)
512
+ if not parsed.ok:
513
+ raise ValueError(
514
+ f"proverqa: example {example.id}: {text!r} does not parse "
515
+ "under the kit grammar — was the example loaded with "
516
+ "convert_fol=False?")
517
+ return parsed.formula
518
+
519
+ premises = [_parse(p) for p in example.fol_premises]
520
+ conclusion = _parse(example.fol_conclusion)
521
+
522
+ verdict = api.prove(conclusion, premises, **prove_kwargs)
523
+ if verdict.status == "proved":
524
+ return {"predicted": "A", "verdict": verdict.to_dict(),
525
+ "verdict_negated": None}
526
+ negated = api.prove(Not(conclusion), premises, **prove_kwargs)
527
+ if negated.status == "proved":
528
+ predicted: Optional[str] = "B"
529
+ elif on_indefinite == "label":
530
+ predicted = "C"
531
+ elif verdict.status == "refuted" and negated.status == "refuted":
532
+ # Underdetermination ESTABLISHED (countermodels both ways): "C" is a
533
+ # definitive answer here, so abstain/raise modes still label it.
534
+ predicted = "C"
535
+ elif on_indefinite == "raise":
536
+ raise ValueError(
537
+ f"proverqa: example {example.id}: indefinite prover outcome "
538
+ f"(goal: {verdict.status}/{verdict.reason}, negated: "
539
+ f"{negated.status}/{negated.reason}) with on_indefinite='raise'.")
540
+ else: # "abstain"
541
+ predicted = None
542
+ return {"predicted": predicted, "verdict": verdict.to_dict(),
543
+ "verdict_negated": negated.to_dict()}
544
+
545
+
546
+ def _resolve_id(record: dict, line_no: int, tier: Optional[str]) -> str:
547
+ """The record's own ``"id"`` when present (namespaced by ``tier`` if
548
+ given, to avoid the cross-tier collision documented in the module
549
+ docstring), else a positional fallback.
550
+
551
+ The positional fallback never fires against verified real data (every
552
+ one of the 1,500 checked rows carries ``"id"``); it exists only so a
553
+ malformed/hand-edited record still yields an addressable example instead
554
+ of crashing on a missing key.
555
+ """
556
+ raw_id = record.get("id")
557
+ if raw_id is None:
558
+ raw_id = f"pos{line_no}"
559
+ if tier is not None:
560
+ return f"proverqa:{tier}:{raw_id}"
561
+ return f"proverqa:{raw_id}"
562
+
563
+
564
+ def _example_from_record(record: dict, line_no: int, tier: Optional[str],
565
+ known_bad_ids: FrozenSet[str],
566
+ convert_fol: bool) -> DatasetExample:
567
+ nl2fol = record.get("nl2fol") or {}
568
+ nl_premises = tuple(nl2fol.keys())
569
+ fol_premises = tuple(nl2fol.values())
570
+ question = record.get("question")
571
+ conclusion_fol = record.get("conclusion_fol")
572
+ answer = record.get("answer")
573
+ example_id = _resolve_id(record, line_no, tier)
574
+
575
+ meta = {
576
+ k: v for k, v in record.items()
577
+ if k not in ("id", "nl2fol", "conclusion_fol", "question", "answer")
578
+ }
579
+ meta["line_no"] = line_no
580
+ if tier is not None:
581
+ meta["tier"] = tier
582
+
583
+ if convert_fol:
584
+ texts = list(fol_premises)
585
+ if conclusion_fol is not None:
586
+ texts.append(conclusion_fol)
587
+ try:
588
+ nodes, mapping = convert_proverqa_formulas(texts)
589
+ except (LarkError, ValueError) as exc:
590
+ # Honest per-example fallback: the verbatim upstream strings stay
591
+ # in place and the failure is recorded, never swallowed.
592
+ meta["fol_conversion_error"] = f"{type(exc).__name__}: {exc}"
593
+ else:
594
+ meta["original_fol_premises"] = list(fol_premises)
595
+ meta["original_fol_conclusion"] = conclusion_fol
596
+ meta["fol_name_mapping"] = mapping
597
+ rendered = tuple(node.to_unicode_str() for node in nodes)
598
+ if conclusion_fol is not None:
599
+ fol_premises = rendered[:-1]
600
+ conclusion_fol = rendered[-1]
601
+ else:
602
+ fol_premises = rendered
603
+
604
+ return DatasetExample(
605
+ id=example_id,
606
+ nl_premises=nl_premises,
607
+ fol_premises=fol_premises,
608
+ nl_conclusion=question,
609
+ fol_conclusion=conclusion_fol,
610
+ label=answer,
611
+ known_bad=example_id in known_bad_ids,
612
+ meta=meta,
613
+ )
614
+
615
+
616
+ def load_proverqa(path: Union[str, Path], *, tier: Optional[str] = None,
617
+ known_bad_ids: FrozenSet[str] = frozenset(),
618
+ convert_fol: bool = True) -> Iterator[DatasetExample]:
619
+ """Stream :class:`~unicode_logic_kit.eval.datasets.DatasetExample` from a
620
+ local ProverQA ``dev/<tier>.json``-derived JSONL file.
621
+
622
+ Args:
623
+ path: path to a local ``.jsonl`` file — one ``dev/<tier>.json``
624
+ record object per non-blank line (see module docstring for
625
+ converting the upstream JSON-array distribution to this format).
626
+ NEVER downloaded by this function. The upstream train split
627
+ (``train/provergen-5000.json``) is NOT supported — it carries no
628
+ structured FOL fields at all (see module docstring).
629
+ tier: ``"easy"``, ``"medium"``, ``"hard"``, or ``None`` (default).
630
+ Purely caller-supplied metadata (the source file itself does not
631
+ self-report which tier a row belongs to) used to namespace ids
632
+ (see module docstring for why that matters across tiers) and
633
+ recorded verbatim in ``meta["tier"]`` when given.
634
+ known_bad_ids: ids (see :func:`_resolve_id`) whose gold FOL is known
635
+ to be broken beyond the notation mismatch documented in the
636
+ module docstring (e.g. a curated finding on top of that). Every
637
+ yielded example with a matching id gets ``known_bad=True``.
638
+ Defaults to an empty set.
639
+ convert_fol: with the default ``True``, every example's gold FOL is
640
+ parsed with the dedicated ProverQA dialect grammar and re-emitted
641
+ in KIT notation (``fol_premises``/``fol_conclusion`` then parse
642
+ under ``api.parse_any``); the verbatim upstream strings and the
643
+ injective name mapping land in ``meta["original_fol_premises"]``
644
+ / ``meta["original_fol_conclusion"]`` /
645
+ ``meta["fol_name_mapping"]``, and a per-example conversion
646
+ failure is recorded in ``meta["fol_conversion_error"]`` with the
647
+ verbatim strings kept in place — never swallowed, never a crash
648
+ of the whole load. ``False`` restores the raw pass-through
649
+ (0% kit-parse rate, see module docstring).
650
+
651
+ Yields:
652
+ One :class:`~unicode_logic_kit.eval.datasets.DatasetExample` per
653
+ non-blank JSONL line, in file order.
654
+
655
+ Raises:
656
+ ValueError: ``tier`` is given but is not one of ``"easy"``,
657
+ ``"medium"``, ``"hard"``.
658
+ FileNotFoundError: ``path`` does not exist.
659
+ json.JSONDecodeError: a non-blank line is not valid JSON — this is
660
+ NOT swallowed; a malformed dataset file is a loud failure, not a
661
+ silently-skipped row.
662
+ """
663
+ if tier is not None and tier not in _VALID_TIERS:
664
+ raise ValueError(f"tier must be one of {_VALID_TIERS!r} or None, got {tier!r}")
665
+
666
+ path = Path(path)
667
+ with path.open("r", encoding="utf-8") as fh:
668
+ for line_no, raw_line in enumerate(fh):
669
+ line = raw_line.strip()
670
+ if not line:
671
+ continue
672
+ record = json.loads(line)
673
+ yield _example_from_record(record, line_no, tier, known_bad_ids,
674
+ convert_fol)