unicode-logic-kit 0.31.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (237) hide show
  1. unicode_logic_kit/__init__.py +385 -0
  2. unicode_logic_kit/__main__.py +520 -0
  3. unicode_logic_kit/_deadline.py +219 -0
  4. unicode_logic_kit/ace/__init__.py +126 -0
  5. unicode_logic_kit/ace/_align.py +135 -0
  6. unicode_logic_kit/ace/chem_lexicon.py +128 -0
  7. unicode_logic_kit/ace/drs_reader.py +570 -0
  8. unicode_logic_kit/ace/mapping.py +666 -0
  9. unicode_logic_kit/ace/reverse_modal.py +138 -0
  10. unicode_logic_kit/ace/runner.py +551 -0
  11. unicode_logic_kit/ace/translate.py +452 -0
  12. unicode_logic_kit/ace/verbalize.py +1070 -0
  13. unicode_logic_kit/api.py +1284 -0
  14. unicode_logic_kit/atp/__init__.py +177 -0
  15. unicode_logic_kit/atp/_ascii_names.py +113 -0
  16. unicode_logic_kit/atp/_html.py +72 -0
  17. unicode_logic_kit/atp/_substructural_input.py +228 -0
  18. unicode_logic_kit/atp/_tff_problem.py +715 -0
  19. unicode_logic_kit/atp/_tptp_problem.py +1111 -0
  20. unicode_logic_kit/atp/_writer_support.py +289 -0
  21. unicode_logic_kit/atp/clingo_backend.py +1180 -0
  22. unicode_logic_kit/atp/cvc5_backend.py +1385 -0
  23. unicode_logic_kit/atp/eprover_backend.py +732 -0
  24. unicode_logic_kit/atp/finite_domain.py +1055 -0
  25. unicode_logic_kit/atp/fitch.py +1547 -0
  26. unicode_logic_kit/atp/fitch_search.py +551 -0
  27. unicode_logic_kit/atp/hets_backend.py +339 -0
  28. unicode_logic_kit/atp/hybrid_down.py +120 -0
  29. unicode_logic_kit/atp/incremental.py +250 -0
  30. unicode_logic_kit/atp/kripke_enum.py +741 -0
  31. unicode_logic_kit/atp/lambek.py +436 -0
  32. unicode_logic_kit/atp/leo3_backend.py +332 -0
  33. unicode_logic_kit/atp/linear.py +738 -0
  34. unicode_logic_kit/atp/lj.py +705 -0
  35. unicode_logic_kit/atp/logic_backends.py +566 -0
  36. unicode_logic_kit/atp/ltl_tableau.py +1084 -0
  37. unicode_logic_kit/atp/minizinc_backend.py +1402 -0
  38. unicode_logic_kit/atp/modal_tableau.py +1382 -0
  39. unicode_logic_kit/atp/nanocop_backend.py +410 -0
  40. unicode_logic_kit/atp/portfolio.py +489 -0
  41. unicode_logic_kit/atp/protocol.py +1803 -0
  42. unicode_logic_kit/atp/prover9_entailment.py +1153 -0
  43. unicode_logic_kit/atp/resolution.py +1376 -0
  44. unicode_logic_kit/atp/resolution_check.py +1114 -0
  45. unicode_logic_kit/atp/sequent.py +1050 -0
  46. unicode_logic_kit/atp/tableau.py +921 -0
  47. unicode_logic_kit/atp/tableau_check.py +543 -0
  48. unicode_logic_kit/atp/tptp_ncl.py +811 -0
  49. unicode_logic_kit/atp/tptp_tff.py +1546 -0
  50. unicode_logic_kit/atp/tstp.py +1333 -0
  51. unicode_logic_kit/atp/tstp_check.py +1096 -0
  52. unicode_logic_kit/atp/twee_backend.py +236 -0
  53. unicode_logic_kit/atp/twee_check.py +711 -0
  54. unicode_logic_kit/atp/twee_entailment.py +953 -0
  55. unicode_logic_kit/atp/vampire_entailment.py +540 -0
  56. unicode_logic_kit/atp/z3_arith.py +470 -0
  57. unicode_logic_kit/atp/z3_equivalence.py +36 -0
  58. unicode_logic_kit/atp/z3_fuzzy.py +362 -0
  59. unicode_logic_kit/atp/z3_input.py +500 -0
  60. unicode_logic_kit/atp/z3_models.py +208 -0
  61. unicode_logic_kit/chem/__init__.py +88 -0
  62. unicode_logic_kit/chem/_naming.py +284 -0
  63. unicode_logic_kit/chem/cache.py +185 -0
  64. unicode_logic_kit/chem/interop.py +244 -0
  65. unicode_logic_kit/chem/mol.py +525 -0
  66. unicode_logic_kit/chem/signature.py +112 -0
  67. unicode_logic_kit/comorphism.py +497 -0
  68. unicode_logic_kit/dl/__init__.py +384 -0
  69. unicode_logic_kit/dl/classification.py +227 -0
  70. unicode_logic_kit/dl/concepts.py +632 -0
  71. unicode_logic_kit/dl/datatypes.py +818 -0
  72. unicode_logic_kit/dl/owl_functional.py +2433 -0
  73. unicode_logic_kit/dl/owl_manchester.py +1637 -0
  74. unicode_logic_kit/dl/owl_reasoner.py +790 -0
  75. unicode_logic_kit/dl/parser.py +391 -0
  76. unicode_logic_kit/dl/tableau.py +4048 -0
  77. unicode_logic_kit/dl/translate.py +2704 -0
  78. unicode_logic_kit/drt/__init__.py +94 -0
  79. unicode_logic_kit/drt/export.py +179 -0
  80. unicode_logic_kit/drt/nodes.py +506 -0
  81. unicode_logic_kit/drt/parser.py +965 -0
  82. unicode_logic_kit/drt/resolve.py +195 -0
  83. unicode_logic_kit/drt/reverse.py +175 -0
  84. unicode_logic_kit/eval/__init__.py +106 -0
  85. unicode_logic_kit/eval/batch.py +382 -0
  86. unicode_logic_kit/eval/canonical.py +663 -0
  87. unicode_logic_kit/eval/chem_batch.py +606 -0
  88. unicode_logic_kit/eval/converses.py +200 -0
  89. unicode_logic_kit/eval/datasets/__init__.py +136 -0
  90. unicode_logic_kit/eval/datasets/_base.py +263 -0
  91. unicode_logic_kit/eval/datasets/_proofwriter_proof.py +422 -0
  92. unicode_logic_kit/eval/datasets/c3po.py +678 -0
  93. unicode_logic_kit/eval/datasets/folio.py +158 -0
  94. unicode_logic_kit/eval/datasets/fracas.py +418 -0
  95. unicode_logic_kit/eval/datasets/groves.py +191 -0
  96. unicode_logic_kit/eval/datasets/logicbench.py +467 -0
  97. unicode_logic_kit/eval/datasets/logicnli.py +303 -0
  98. unicode_logic_kit/eval/datasets/malls.py +133 -0
  99. unicode_logic_kit/eval/datasets/pfolio.py +594 -0
  100. unicode_logic_kit/eval/datasets/pmb.py +242 -0
  101. unicode_logic_kit/eval/datasets/prontoqa.py +611 -0
  102. unicode_logic_kit/eval/datasets/proofwriter.py +1431 -0
  103. unicode_logic_kit/eval/datasets/proverqa.py +674 -0
  104. unicode_logic_kit/eval/datasets/willow.py +478 -0
  105. unicode_logic_kit/eval/equivalence.py +466 -0
  106. unicode_logic_kit/eval/exercise_gen.py +533 -0
  107. unicode_logic_kit/eval/explain.py +791 -0
  108. unicode_logic_kit/eval/generality.py +750 -0
  109. unicode_logic_kit/eval/metric_hf.py +458 -0
  110. unicode_logic_kit/eval/predicate_match.py +343 -0
  111. unicode_logic_kit/eval/theory_check.py +1170 -0
  112. unicode_logic_kit/eval/validate.py +306 -0
  113. unicode_logic_kit/fol/__init__.py +177 -0
  114. unicode_logic_kit/fol/_atom_keys.py +510 -0
  115. unicode_logic_kit/fol/_fol_nodes.py +3586 -0
  116. unicode_logic_kit/fol/_free_parameters.py +105 -0
  117. unicode_logic_kit/fol/_ho_nodes.py +448 -0
  118. unicode_logic_kit/fol/_hybrid_nodes.py +308 -0
  119. unicode_logic_kit/fol/_identifiers.py +1091 -0
  120. unicode_logic_kit/fol/_lambek_nodes.py +112 -0
  121. unicode_logic_kit/fol/_linear_nodes.py +352 -0
  122. unicode_logic_kit/fol/_modal_nodes.py +1467 -0
  123. unicode_logic_kit/fol/_msfl_nodes.py +2196 -0
  124. unicode_logic_kit/fol/_numeral_symbols.py +231 -0
  125. unicode_logic_kit/fol/_so_nodes.py +200 -0
  126. unicode_logic_kit/fol/_symbol_names.py +81 -0
  127. unicode_logic_kit/fol/_team_nodes.py +181 -0
  128. unicode_logic_kit/fol/_tptp_symbols.py +551 -0
  129. unicode_logic_kit/fol/_truth_constants.py +117 -0
  130. unicode_logic_kit/fol/casl_export.py +1135 -0
  131. unicode_logic_kit/fol/casl_import.py +929 -0
  132. unicode_logic_kit/fol/derivation.py +367 -0
  133. unicode_logic_kit/fol/dialect_detect.py +70 -0
  134. unicode_logic_kit/fol/dialect_repair.py +537 -0
  135. unicode_logic_kit/fol/frames.py +637 -0
  136. unicode_logic_kit/fol/grammars/terminals.lark +31 -0
  137. unicode_logic_kit/fol/lambda_tools.py +297 -0
  138. unicode_logic_kit/fol/latex_input.py +429 -0
  139. unicode_logic_kit/fol/modal_translation.py +944 -0
  140. unicode_logic_kit/fol/msflparser.py +1033 -0
  141. unicode_logic_kit/fol/naming.py +422 -0
  142. unicode_logic_kit/fol/nodes.py +241 -0
  143. unicode_logic_kit/fol/normalforms.py +492 -0
  144. unicode_logic_kit/fol/pal.py +287 -0
  145. unicode_logic_kit/fol/prolog_export.py +566 -0
  146. unicode_logic_kit/fol/prolog_input.py +505 -0
  147. unicode_logic_kit/fol/prover9_input.py +1325 -0
  148. unicode_logic_kit/fol/qml.py +1760 -0
  149. unicode_logic_kit/fol/qmltp_input.py +525 -0
  150. unicode_logic_kit/fol/sanitize.py +221 -0
  151. unicode_logic_kit/fol/serialize.py +79 -0
  152. unicode_logic_kit/fol/signature.py +1290 -0
  153. unicode_logic_kit/fol/simplify_check.py +544 -0
  154. unicode_logic_kit/fol/spans.py +594 -0
  155. unicode_logic_kit/fol/tptp_input.py +1503 -0
  156. unicode_logic_kit/fol/tptp_repair.py +941 -0
  157. unicode_logic_kit/fol/unification.py +157 -0
  158. unicode_logic_kit/fol/verbalize.py +263 -0
  159. unicode_logic_kit/hets/__init__.py +163 -0
  160. unicode_logic_kit/hets/bridge.py +142 -0
  161. unicode_logic_kit/hets/client.py +748 -0
  162. unicode_logic_kit/hets/docker.py +420 -0
  163. unicode_logic_kit/hets/dol.py +712 -0
  164. unicode_logic_kit/hets/haskell_json.py +355 -0
  165. unicode_logic_kit/hets/owl_backend.py +794 -0
  166. unicode_logic_kit/hets/owl_cli.py +598 -0
  167. unicode_logic_kit/hets/symbols.py +512 -0
  168. unicode_logic_kit/hol/__init__.py +140 -0
  169. unicode_logic_kit/hol/_ho_common.py +323 -0
  170. unicode_logic_kit/hol/_isabelle_binders.py +125 -0
  171. unicode_logic_kit/hol/classical.py +812 -0
  172. unicode_logic_kit/hol/deepshallow/__init__.py +45 -0
  173. unicode_logic_kit/hol/deepshallow/_common.py +177 -0
  174. unicode_logic_kit/hol/deepshallow/conditional.py +225 -0
  175. unicode_logic_kit/hol/deepshallow/intuitionistic.py +181 -0
  176. unicode_logic_kit/hol/deepshallow/modal.py +217 -0
  177. unicode_logic_kit/hol/deepshallow/qml.py +406 -0
  178. unicode_logic_kit/hol/deepshallow/relevant.py +206 -0
  179. unicode_logic_kit/hol/free.py +753 -0
  180. unicode_logic_kit/hol/goedel.py +336 -0
  181. unicode_logic_kit/hol/ho_modal.py +1743 -0
  182. unicode_logic_kit/hol/intuitionistic.py +403 -0
  183. unicode_logic_kit/hol/isabelle_conditional.py +593 -0
  184. unicode_logic_kit/hol/isabelle_modal.py +1908 -0
  185. unicode_logic_kit/hol/isabelle_relevant.py +412 -0
  186. unicode_logic_kit/hol/isabelle_runner.py +1147 -0
  187. unicode_logic_kit/hol/isabelle_substructural.py +884 -0
  188. unicode_logic_kit/hol/lean.py +1018 -0
  189. unicode_logic_kit/hol/manyvalued.py +921 -0
  190. unicode_logic_kit/hol/secondorder.py +687 -0
  191. unicode_logic_kit/hol/thf_modal.py +941 -0
  192. unicode_logic_kit/hol/thirdorder.py +397 -0
  193. unicode_logic_kit/ilp/__init__.py +89 -0
  194. unicode_logic_kit/ilp/readback.py +389 -0
  195. unicode_logic_kit/ilp/separation.py +153 -0
  196. unicode_logic_kit/ilp/task.py +730 -0
  197. unicode_logic_kit/logic.py +163 -0
  198. unicode_logic_kit/mcp/__init__.py +28 -0
  199. unicode_logic_kit/mcp/__main__.py +5 -0
  200. unicode_logic_kit/mcp/chem_tools.py +1031 -0
  201. unicode_logic_kit/mcp/server.py +2453 -0
  202. unicode_logic_kit/mcp/syntax_spec.py +681 -0
  203. unicode_logic_kit/prob/__init__.py +53 -0
  204. unicode_logic_kit/prob/_bdd.py +225 -0
  205. unicode_logic_kit/prob/_column_gen.py +668 -0
  206. unicode_logic_kit/prob/distribution.py +686 -0
  207. unicode_logic_kit/prob/nilsson.py +470 -0
  208. unicode_logic_kit/py.typed +0 -0
  209. unicode_logic_kit/semantics/__init__.py +137 -0
  210. unicode_logic_kit/semantics/_modal_reject.py +156 -0
  211. unicode_logic_kit/semantics/action_models.py +466 -0
  212. unicode_logic_kit/semantics/asp_models.py +1200 -0
  213. unicode_logic_kit/semantics/conditional.py +580 -0
  214. unicode_logic_kit/semantics/dynamic_epistemic.py +95 -0
  215. unicode_logic_kit/semantics/free_logic.py +913 -0
  216. unicode_logic_kit/semantics/fuzzy.py +384 -0
  217. unicode_logic_kit/semantics/fuzzy_kripke.py +442 -0
  218. unicode_logic_kit/semantics/intuitionistic.py +581 -0
  219. unicode_logic_kit/semantics/kripke.py +1139 -0
  220. unicode_logic_kit/semantics/manyvalued.py +580 -0
  221. unicode_logic_kit/semantics/matrix.py +342 -0
  222. unicode_logic_kit/semantics/model_eval.py +1135 -0
  223. unicode_logic_kit/semantics/modelfinder.py +1036 -0
  224. unicode_logic_kit/semantics/nonmonotonic.py +372 -0
  225. unicode_logic_kit/semantics/relevant.py +331 -0
  226. unicode_logic_kit/semantics/secondorder.py +657 -0
  227. unicode_logic_kit/semantics/structures.py +352 -0
  228. unicode_logic_kit/semantics/tarski.py +975 -0
  229. unicode_logic_kit/semantics/team.py +315 -0
  230. unicode_logic_kit/semantics/team_translation.py +416 -0
  231. unicode_logic_kit/semantics/thirdorder.py +358 -0
  232. unicode_logic_kit/semantics/tnorm.py +85 -0
  233. unicode_logic_kit/semantics/truthtable.py +201 -0
  234. unicode_logic_kit-0.31.0.dist-info/METADATA +333 -0
  235. unicode_logic_kit-0.31.0.dist-info/RECORD +237 -0
  236. unicode_logic_kit-0.31.0.dist-info/WHEEL +4 -0
  237. unicode_logic_kit-0.31.0.dist-info/licenses/LICENSE +21 -0
@@ -0,0 +1,1431 @@
1
+ """Adapter for the ProofWriter dataset (Tafjord, Dalvi Mishra, Clark,
2
+ "ProofWriter: Generating Implications, Proofs, and Abductive Statements over
3
+ Natural Language", Findings of ACL 2021, arXiv:2012.13048) — local JSONL
4
+ only, no network access.
5
+
6
+ Source and verified schema
7
+ ---------------------------
8
+ The primary distribution named by the paper, https://allenai.org/data/proofwriter,
9
+ does NOT serve a metadata page: fetching it returns a bare HTTP 307 redirect
10
+ straight to ``proofwriter-dataset-V2020.12.3.zip`` (~214 MB) on
11
+ ``aristo-data-public.s3.amazonaws.com``, with no separate page describing its
12
+ internal JSONL field names, and the current https://allenai.org/data catalog
13
+ no longer lists a ProofWriter entry at all (both checked directly,
14
+ 2026-08-12). Downloading and unpacking a 214 MB archive was out of scope for
15
+ schema verification here, so — per this adapter's instructions, which allow
16
+ falling back to "the best-documented HF mirror" when the original
17
+ distribution's internal structure cannot be confirmed via the rows API — the
18
+ schema below was verified instead against
19
+ https://huggingface.co/datasets/tasksource/proofwriter (unauthenticated,
20
+ 11,898+ downloads, the most-used ProofWriter mirror on the Hub), directly via
21
+ the Hugging Face ``datasets-server`` ``/first-rows``, ``/statistics``, and
22
+ ``/search`` APIs across all three of its splits (``train`` — 585,552 rows,
23
+ ``validation`` — 85,468 rows, ``test`` — 174,476 rows), 2026-08-12.
24
+ :func:`load_proofwriter` reads that verified schema, NOT the original
25
+ AllenAI ZIP's internal format (unverified here) — see "Honesty" below for
26
+ what this means in practice.
27
+
28
+ Each verified row is a flat JSON object with exactly these keys (confirmed
29
+ via the dataset's declared ``dataset_info.features`` and by inspecting many
30
+ real fetched rows):
31
+
32
+ * ``"id"`` — ``str``. NOT a per-row unique id (see "Id resolution"
33
+ below) — it identifies one GENERATED THEORY, and every question asked
34
+ against that theory (2-6 in the fetched sample) repeats the same ``"id"``.
35
+ Its own substructure (e.g. ``"AttNeg-OWA-D0-1778"``) encodes, dash-separated:
36
+ a generation-template family (``AttNeg``/``AttNoneg``/``RelNeg``/``RelNoneg``
37
+ — attribute- or relation-typed facts, with or without negated facts), an
38
+ open/closed-world tag (``OWA`` in every single row observed here — see
39
+ "CWA vs OWA" below), a depth bucket (``D0`` etc., redundant with ``maxD``),
40
+ and a generator seed/index. This adapter does not parse that substring; it
41
+ is preserved verbatim in ``meta["theory_id"]``.
42
+ * ``"maxD"`` — ``int``, the maximum proof depth reachable from this
43
+ theory's rule set (0 in every fetched sample row; the dataset overall
44
+ ranges 0-10 per the mirror's column statistics).
45
+ * ``"NFact"`` — ``int``, number of atomic facts in ``"theory"``.
46
+ * ``"NRule"`` — ``int``, number of conditional rules in ``"theory"``.
47
+ * ``"theory"`` — ``str``, ALL of this theory's facts and rules concatenated
48
+ into one string, one sentence per fact/rule, each ending in ``"."`` and
49
+ separated by a single space — e.g. ``"Anne is smart. Dave is round. If
50
+ someone is cold then they are blue."``. Verified: in every fetched row,
51
+ ``NFact + NRule`` equals exactly the number of ``". "``-delimited sentences
52
+ in ``"theory"`` (hand-checked below in the test suite), so this adapter's
53
+ ``nl_premises`` sentence split (see "Field mapping") is not a guess.
54
+ * ``"question"`` — ``str``, one NL sentence being asked about (e.g. ``"Dave
55
+ is round."`` or its negation ``"Dave is not round."``).
56
+ * ``"answer"`` — ``str``, one of exactly ``{"True", "False", "Unknown"}``
57
+ (confirmed via the mirror's train-split column statistics: 158,805 /
58
+ 158,805 / 267,942 rows respectively — no fourth value exists in this
59
+ mirror).
60
+ * ``"QDep"`` — ``int``, the depth of proof needed to answer this
61
+ specific question (0 in every "directly stated fact" question observed).
62
+ * ``"QLen"`` — ``float`` or ``null``. In every row fetched here, ``null``
63
+ exactly when ``"answer" == "Unknown"`` and ``1.0`` whenever the answer is
64
+ ``"True"``/``"False"`` — an OBSERVED correlation from the sample fetched
65
+ for this verification, not a guarantee re-derived from a spec, so it is
66
+ not relied on by this loader beyond passing the raw value through in
67
+ ``meta``.
68
+ * ``"allProofs"`` — ``str``, an opaque proof-forest annotation in the
69
+ dataset's own ``triple``/``rule``-reference notation (e.g. ``"@0: Anne is
70
+ smart.[(triple1)] ..."``), kept verbatim in ``meta`` — this adapter does
71
+ not parse it (it is not FOL, and parsing its internal proof-tree grammar
72
+ is out of scope here).
73
+ * ``"config"`` — ``str``, one of exactly 8 values confirmed via the
74
+ mirror's train-split column statistics: ``"depth-0"``, ``"depth-1"``,
75
+ ``"depth-2"``, ``"depth-3"``, ``"depth-3ext"``, ``"depth-3ext-NatLang"``,
76
+ ``"depth-5"``, ``"NatLang"`` (the paper's D0-D5 depth staircase, plus the
77
+ hand-authored "NatLang"/"birds-electricity"-style natural-language subsets
78
+ described in the paper; note ``"depth-4"`` does not appear as a distinct
79
+ ``config`` value in this mirror).
80
+
81
+ Honesty: what this adapter does NOT give you
82
+ ----------------------------------------------
83
+ * **No FOL annotation exists.** ProofWriter's ``"theory"``/``"question"``
84
+ are natural-language sentences ("If someone is red then they are kind.");
85
+ there is no gold first-order-logic formula anywhere in the source data.
86
+ Accordingly ``fol_premises`` is ALWAYS ``()`` and ``fol_conclusion`` is
87
+ ALWAYS ``None`` for every example this loader yields — never guessed,
88
+ never back-translated. A consequence: :func:`~unicode_logic_kit.eval.datasets.audit_examples`
89
+ run over ProofWriter examples is VACUOUSLY ``ok=True`` for all of them (no
90
+ FOL string means nothing to parse or fail ``check()`` on) — it is not a
91
+ meaningful signal for this dataset and callers should not read "0 defects"
92
+ as "ProofWriter's data is fine", only as "there is nothing here to audit".
93
+ * **CWA vs OWA, and why it is NOT classical FOL entailment.** ProofWriter
94
+ was released in TWO reasoning-assumption variants per the paper: **CWA**
95
+ (closed-world assumption — a fact not provable from the theory is assumed
96
+ FALSE, i.e. negation-as-failure) and **OWA** (open-world assumption — a
97
+ fact not provable either way is genuinely ``"Unknown"``, distinct from
98
+ ``"False"``). The verified mirror used here contains **ONLY the OWA
99
+ variant** — every single ``"id"`` in the fetched sample, and a full-text
100
+ search for the substring ``"CWA"`` across the ENTIRE train split, returned
101
+ zero matches (``num_rows_total: 0``, confirmed 2026-08-12) — so this
102
+ adapter's field mapping and every claim above describes OWA data only. The
103
+ presence of the three-valued ``answer`` (``"Unknown"`` as a genuine third
104
+ value, not collapsed into ``"False"``) is itself the observable signature
105
+ of OWA rather than CWA.
106
+
107
+ How the two variants relate to classical FOL — stated precisely, because
108
+ a sloppy version of this claim is easy to make and wrong: **CWA is
109
+ ordinary two-valued FOL** — evaluation of the question in ONE canonical
110
+ model, the closed (minimal) model, where exactly the derivable atoms hold
111
+ and everything else is plainly false; there is no third value. **OWA's
112
+ three-way label is the ENTAILMENT split** — ``True`` iff the theory
113
+ entails the question, ``False`` iff it entails its negation, ``Unknown``
114
+ otherwise — and "Unknown" is a statement ABOUT entailment, not an FOL
115
+ truth value inside any model. Both are classical; they answer different
116
+ questions ("true in the closed model?" vs "true in every model?"), and
117
+ the labels visibly diverge exactly where a fact is underdetermined
118
+ (closed model: false; entailment: unknown). For the STRUCTURED route
119
+ below, both are decidable with any registered ATP:
120
+ ``solve_structured_example(..., semantics="owa")`` runs the entailment
121
+ cascade, ``semantics="cwa"`` runs closed-model checking with the ATP as
122
+ the per-atom derivability oracle (exact for the definite, negation-free
123
+ theories; a rule with negation in its body would mean
124
+ negation-as-failure inside the theory and is refused). The FLAT
125
+ tasksource mirror THIS loader reads carries no representations, so for
126
+ it these labels remain data to report, not something this adapter
127
+ recomputes.
128
+ * **Only one split's worth of assumption is covered.** This loader's field
129
+ mapping was verified against OWA rows only (see above); if a caller
130
+ obtains genuine CWA-variant ProofWriter data (e.g. from the original
131
+ AllenAI ZIP, unverified here), the row shape is very likely structurally
132
+ identical (same ``id`` substring convention, same field names) but that
133
+ has NOT been independently confirmed by this adapter.
134
+ * **``"theory"`` is one blob string, not pre-split sentences.** ``nl_premises``
135
+ below is DERIVED by this adapter (splitting on ``". "``), not a field that
136
+ exists upstream — see "Field mapping".
137
+
138
+ License
139
+ -------
140
+ **UNVERIFIED** as of 2026-08-12. No ProofWriter-specific license text could
141
+ be confirmed from any live, unauthenticated source: the AllenAI dataset page
142
+ is a bare redirect straight to the ZIP with no accompanying license file
143
+ reachable without downloading it, the current AllenAI data catalog no longer
144
+ lists a ProofWriter entry to check, and the verified HF mirror
145
+ (tasksource/proofwriter)'s dataset card carries no ``license`` tag and reads
146
+ literally "More Information needed". The sibling AI2 RuleTaker CODE
147
+ repository (https://github.com/allenai/ruletaker, whose legacy JSONL example
148
+ format ProofWriter's ``theory``/``question`` sentences continue) is
149
+ Apache-2.0-licensed, but that governs that repository's CODE, not
150
+ ProofWriter's separately-hosted data ZIP, and is not treated here as a
151
+ substitute for a verified data license. Treat ProofWriter as all-rights-
152
+ reserved research data pending confirmation directly from the paper's
153
+ authors or AI2, and do not redistribute this loader's *inputs* (the JSONL
154
+ file itself) without resolving that.
155
+
156
+ Field mapping
157
+ --------------
158
+ ProofWriter has no premises/conclusion ENTAILMENT structure quite like
159
+ FOLIO's (no separate gold "this follows" formula), but its
160
+ theory-facts-and-rules / queried-question / True-False-Unknown shape maps
161
+ onto :class:`~unicode_logic_kit.eval.datasets.DatasetExample` as an entailment
162
+ example nonetheless:
163
+
164
+ * ``nl_premises`` — ``"theory"`` SPLIT into individual sentences on ``". "``
165
+ (each fragment re-terminated with ``"."`` if the split ate it) by this
166
+ module's :func:`_split_theory_sentences`. This is a LOCAL heuristic of
167
+ this adapter, not part of the verified upstream schema (upstream gives you
168
+ one string) — verified safe against every fetched sample row only insofar
169
+ as none of them contain a sentence-internal period, abbreviation, or
170
+ ellipsis (a synthetic, template-generated corpus, so this holds by
171
+ construction for the OWA rows checked). The RAW, unsplit ``"theory"``
172
+ string is ALSO kept verbatim in ``meta["theory"]`` so nothing is lost to
173
+ the split.
174
+ * ``fol_premises`` — ALWAYS ``()`` (no FOL exists; see "Honesty" above).
175
+ * ``nl_conclusion`` — ``"question"`` verbatim.
176
+ * ``fol_conclusion`` — ALWAYS ``None`` (no FOL exists; see "Honesty" above).
177
+ * ``label`` — ``"answer"`` verbatim (``"True"``/``"False"``/``"Unknown"``).
178
+ * ``meta`` — every other record key (``maxD``, ``NFact``, ``NRule``, ``QDep``,
179
+ ``QLen``, ``allProofs``, ``config``), PLUS the raw ``"theory"`` string,
180
+ PLUS ``"theory_id"`` (renamed from the record's own ``"id"`` — see "Id
181
+ resolution"), PLUS ``"line_no"``.
182
+
183
+ Id resolution
184
+ --------------
185
+ ProofWriter's own ``"id"`` field identifies a GENERATED THEORY, not a single
186
+ row — exactly the same shape of problem as FOLIO's ``"story-id"`` (see
187
+ ``folio.py``'s docstring): several consecutive rows in the verified sample
188
+ share one ``"id"`` while asking different questions about the same theory,
189
+ so using it directly as this adapter's per-example id would collide. This
190
+ loader therefore ALWAYS uses the positional id ``f"proofwriter:{line_no}"``
191
+ (0-based line number in the local file) and preserves the original,
192
+ non-unique ``"id"`` value verbatim in ``meta["theory_id"]`` for anyone who
193
+ wants to group rows back into their source theory.
194
+
195
+ This module never downloads anything — obtain a local JSONL file yourself
196
+ (e.g. by exporting the verified ``tasksource/proofwriter`` split with
197
+ ``datasets.load_dataset("tasksource/proofwriter", split="train").to_json(path, orient="records", lines=True)``,
198
+ or via the HF ``rows``/``first-rows`` API) and pass its local path to
199
+ :func:`load_proofwriter`.
200
+ """
201
+
202
+ import itertools
203
+ import json
204
+ import re
205
+ from pathlib import Path
206
+ from typing import Dict, FrozenSet, Iterator, List, Optional, Tuple, Union
207
+
208
+ from ...fol.nodes import (
209
+ Node, Atom, Not, And, Or, Xor, Implies, Iff, Quantifier,
210
+ Variable, Constant, substitute,
211
+ )
212
+ from ...fol._msfl_nodes import key_text
213
+ from ._base import DatasetExample, _register_dataset_info
214
+ from . import _proofwriter_proof as _proof
215
+
216
+ __all__ = [
217
+ "load_proofwriter",
218
+ "load_proofwriter_structured",
219
+ "parse_proofwriter_representation",
220
+ "solve_structured_example",
221
+ "check_gold_proof",
222
+ ]
223
+
224
+ _register_dataset_info(
225
+ "proofwriter",
226
+ license=(
227
+ "UNVERIFIED as of 2026-08-12 -- no ProofWriter-specific license text "
228
+ "could be confirmed from any live, unauthenticated source (AllenAI's "
229
+ "dataset page redirects directly to a ZIP with no reachable license "
230
+ "file; the verified HF mirror's dataset card has no license tag); "
231
+ "treat as all-rights-reserved research data pending confirmation "
232
+ "from the paper's authors or AI2 -- see module docstring"
233
+ ),
234
+ source_url="https://huggingface.co/datasets/tasksource/proofwriter",
235
+ citation_hint=(
236
+ "Tafjord, Oyvind, Bhavana Dalvi Mishra, and Peter Clark. \"ProofWriter: "
237
+ "Generating Implications, Proofs, and Abductive Statements over Natural "
238
+ "Language.\" Findings of the Association for Computational Linguistics: "
239
+ "ACL-IJCNLP 2021. arXiv:2012.13048."
240
+ ),
241
+ )
242
+
243
+
244
+ def _split_theory_sentences(theory: Optional[str]) -> Tuple[str, ...]:
245
+ """Split a raw ``"theory"`` blob into individual NL fact/rule sentences.
246
+
247
+ A LOCAL heuristic of this adapter (see the module docstring's "Field
248
+ mapping" section for why this is safe over the verified sample but is
249
+ not itself part of the upstream schema): splits on the literal
250
+ substring ``". "``, then re-appends a trailing ``"."`` to any fragment
251
+ the split consumed it from (every fragment except possibly the last).
252
+ ``None`` or ``""`` (e.g. a record missing ``"theory"`` entirely) yields
253
+ ``()`` rather than raising or fabricating a one-element tuple of empty
254
+ string.
255
+ """
256
+ if not theory:
257
+ return ()
258
+ sentences = []
259
+ for part in theory.strip().split(". "):
260
+ part = part.strip()
261
+ if not part:
262
+ continue
263
+ if not part.endswith("."):
264
+ part = part + "."
265
+ sentences.append(part)
266
+ return tuple(sentences)
267
+
268
+
269
+ def _example_from_record(record: dict, line_no: int,
270
+ known_bad_ids: FrozenSet[str]) -> DatasetExample:
271
+ theory = record.get("theory")
272
+ question = record.get("question")
273
+ answer = record.get("answer")
274
+ example_id = f"proofwriter:{line_no}"
275
+
276
+ # Everything except question/answer stays in meta (including "theory"
277
+ # itself, verbatim -- see module docstring: nl_premises below is a
278
+ # DERIVED split, not a byte-identical copy). The record's own "id" is
279
+ # renamed to "theory_id" here: see "Id resolution" in the module
280
+ # docstring for why it is deliberately NOT used as this example's id.
281
+ meta = {k: v for k, v in record.items() if k not in ("question", "answer")}
282
+ meta["theory_id"] = meta.pop("id", None)
283
+ meta["line_no"] = line_no
284
+
285
+ return DatasetExample(
286
+ id=example_id,
287
+ nl_premises=_split_theory_sentences(theory),
288
+ fol_premises=(),
289
+ nl_conclusion=question,
290
+ fol_conclusion=None,
291
+ label=answer,
292
+ known_bad=example_id in known_bad_ids,
293
+ meta=meta,
294
+ )
295
+
296
+
297
+ def load_proofwriter(path: Union[str, Path], *,
298
+ known_bad_ids: FrozenSet[str] = frozenset()) -> Iterator[DatasetExample]:
299
+ """Stream :class:`~unicode_logic_kit.eval.datasets.DatasetExample` from a
300
+ local ProofWriter JSONL file (verified ``tasksource/proofwriter`` schema
301
+ -- see module docstring).
302
+
303
+ Args:
304
+ path: path to a local ``.jsonl`` file — one
305
+ ``{"id", "maxD", "NFact", "NRule", "theory", "question",
306
+ "answer", "QDep", "QLen", "allProofs", "config"}`` object per
307
+ non-blank line (see module docstring for how to produce this
308
+ from the verified HF mirror). NEVER downloaded by this function.
309
+ known_bad_ids: ids (the positional ``f"proofwriter:{line_no}"`` form
310
+ -- see "Id resolution" in the module docstring) whose ``answer``
311
+ is known to be broken (e.g. from a prior human review). Every
312
+ yielded example with a matching id gets ``known_bad=True``.
313
+ Defaults to an empty set.
314
+
315
+ Yields:
316
+ One :class:`~unicode_logic_kit.eval.datasets.DatasetExample` per
317
+ non-blank JSONL line, in file order. ``fol_premises`` is always
318
+ ``()`` and ``fol_conclusion`` is always ``None`` (ProofWriter has no
319
+ FOL gold annotation -- see the module docstring's "Honesty"
320
+ section). Fields missing from a record (e.g. no ``"theory"``, no
321
+ ``"answer"``) map to ``()``/``None`` rather than raising -- this
322
+ mirrors :mod:`~unicode_logic_kit.eval.datasets.folio`'s defensive
323
+ ``dict.get`` style, documented, not an exception.
324
+
325
+ Raises:
326
+ FileNotFoundError: ``path`` does not exist.
327
+ json.JSONDecodeError: a non-blank line is not valid JSON -- this is
328
+ NOT swallowed; a malformed dataset file is a loud failure, not a
329
+ silently-skipped row.
330
+ """
331
+ path = Path(path)
332
+ with path.open("r", encoding="utf-8") as fh:
333
+ for line_no, raw_line in enumerate(fh):
334
+ line = raw_line.strip()
335
+ if not line:
336
+ continue
337
+ record = json.loads(line)
338
+ yield _example_from_record(record, line_no, known_bad_ids)
339
+
340
+
341
+ # --------------------------------------------------------------------------- #
342
+ # Structured OWA distribution: deterministic FOL GENERATION from the dataset's
343
+ # own triple/rule representations
344
+ #
345
+ # The AllenAI ProofWriter release (mirrored with all metadata intact at
346
+ # https://huggingface.co/datasets/hitachi-nlp/proofwriter_processed_OWA —
347
+ # configs depth-0/1/2/3/3ext/5, NatLang, birds-electricity; schema verified
348
+ # against real depth-2 rows on 2026-08-12) carries, next to every NL
349
+ # sentence, the ORIGINAL RuleTaker-style symbolic representation:
350
+ #
351
+ # fact triple: ("Anne" "is" "white" "+") [attribute]
352
+ # ("bear" "chases" "dog" "+") [relation]
353
+ # rule: ((("something" "is" "young" "+"))
354
+ # -> ("something" "is" "white" "+"))
355
+ # question: ("Charlie" "is" "cold" "-") [negative polarity]
356
+ #
357
+ # Unlike the flat tasksource mirror load_proofwriter reads (NL text only —
358
+ # no FOL gold, as documented above), these representations admit a
359
+ # DETERMINISTIC, rule-based translation to FOL — no LLM, no heuristics on NL
360
+ # text. The convention implemented here (the standard reading of RuleTaker
361
+ # triples, Clark et al. 2020):
362
+ #
363
+ # * an attribute triple (E "is" A pol) becomes A'(e) — predicate = the
364
+ # attribute, capitalised (white → White); entity = a constant in kit
365
+ # casing (Anne → anne, multi-word "bald eagle" → baldEagle);
366
+ # * a relation triple (E V F pol) becomes V'(e, f) (chases → Chases);
367
+ # * polarity "-" (or "~") wraps the atom in ¬;
368
+ # * the placeholder words something/someone/somebody are RULE VARIABLES:
369
+ # every rule quantifies universally over each distinct placeholder it
370
+ # uses — ∀x (Young(x) → White(x)); a rule without placeholders stays a
371
+ # ground implication;
372
+ # * a rule ((c1 … cn) -> d) becomes (c1 ∧ … ∧ cn) → d under those
373
+ # quantifiers.
374
+ #
375
+ # HONESTY: the resulting fol_premises/fol_conclusion are GENERATED BY THIS
376
+ # KIT from the dataset's own symbolic annotations — ProofWriter itself ships
377
+ # no FOL strings; meta["fol_generated"] marks every such example and meta
378
+ # keeps the verbatim source representations. The OWA labels are the
379
+ # classical ENTAILMENT split (True iff premises ⊨ q, False iff premises ⊨
380
+ # ¬q, Unknown otherwise) — solve_structured_example(semantics="owa") decides
381
+ # exactly that, verified 24/24 on the fixture. The CWA labels are ordinary
382
+ # two-valued truth in the CLOSED (minimal) model — plain FOL model checking,
383
+ # no third value — and solve_structured_example(semantics="cwa") decides
384
+ # that via per-atom derivability with the caller's chosen ATP (see its
385
+ # docstring for the exact fragment contract). The verified hitachi-nlp
386
+ # mirror ships only OWA data; CWA gold labels live in the original AllenAI
387
+ # zip (proofwriter-dataset-V2020.12.3.zip), whose per-question schema is
388
+ # identical — the CWA route was verified against it (2026-08-12): first 100
389
+ # theories of CWA/depth-2/meta-dev.jsonl = 1078/1078 questions correct
390
+ # (29 AttNoneg / 31 AttNeg / 19 RelNoneg / 21 RelNeg; z3 cross-check on the
391
+ # definite ones). Several of those theories NEED local stratification:
392
+ # their predicate graphs cycle through negation while their ground graphs
393
+ # do not — predicate-level stratification would refuse them.
394
+ # --------------------------------------------------------------------------- #
395
+
396
+ _REPR_TOKEN_RE = re.compile(r'\(|\)|->|"[^"]*"')
397
+
398
+ #: RuleTaker's rule-variable placeholder words (subject/object position of a
399
+ #: rule triple). Every DISTINCT placeholder in one rule gets its own
400
+ #: universally quantified variable, in order of first appearance: x, y, z.
401
+ _PLACEHOLDER_WORDS = ("something", "someone", "somebody")
402
+
403
+ _KIT_PRED_RE = re.compile(r"[A-Z][a-zA-Z0-9]*\Z")
404
+ _KIT_CONST_RE = re.compile(r"[a-z][a-zA-Z0-9]*[a-zA-Z][a-zA-Z0-9]*\Z")
405
+
406
+
407
+ def _tokenise_representation(rep: str) -> List[str]:
408
+ tokens = _REPR_TOKEN_RE.findall(rep)
409
+ remainder = _REPR_TOKEN_RE.sub("", rep).strip()
410
+ if remainder:
411
+ raise ValueError(
412
+ f"proofwriter: unrecognised material {remainder!r} in "
413
+ f"representation {rep!r}")
414
+ return tokens
415
+
416
+
417
+ def _parse_sexpr(tokens: List[str], pos: int):
418
+ """One S-expression starting at ``pos`` → (parsed, next_pos).
419
+
420
+ Strings stay strings (quotes stripped); ``(`` … ``)`` becomes a list;
421
+ ``->`` stays the literal marker token.
422
+ """
423
+ token = tokens[pos]
424
+ if token == "(":
425
+ items = []
426
+ pos += 1
427
+ while pos < len(tokens) and tokens[pos] != ")":
428
+ item, pos = _parse_sexpr(tokens, pos)
429
+ items.append(item)
430
+ if pos >= len(tokens):
431
+ raise ValueError("proofwriter: unbalanced '(' in representation")
432
+ return items, pos + 1
433
+ if token == ")":
434
+ raise ValueError("proofwriter: unbalanced ')' in representation")
435
+ if token == "->":
436
+ return "->", pos + 1
437
+ return token[1:-1], pos + 1 # strip the quotes
438
+
439
+
440
+ def _camel(words: List[str]) -> str:
441
+ return words[0] + "".join(w[0].upper() + w[1:] for w in words[1:] if w)
442
+
443
+
444
+ def _to_predicate_name(word: str) -> str:
445
+ parts = [p for p in re.split(r"[\s\-]+", word.strip()) if p]
446
+ if not parts:
447
+ raise ValueError("proofwriter: empty predicate word")
448
+ joined = _camel(parts)
449
+ name = joined[0].upper() + joined[1:]
450
+ if not _KIT_PRED_RE.match(name):
451
+ raise ValueError(
452
+ f"proofwriter: {word!r} does not map to a legal kit predicate "
453
+ f"(got {name!r})")
454
+ return name
455
+
456
+
457
+ def _to_constant_name(word: str) -> str:
458
+ parts = [p for p in re.split(r"[\s\-]+", word.strip()) if p]
459
+ if not parts:
460
+ raise ValueError("proofwriter: empty entity word")
461
+ joined = _camel(parts)
462
+ name = joined[0].lower() + joined[1:]
463
+ if not _KIT_CONST_RE.match(name):
464
+ raise ValueError(
465
+ f"proofwriter: entity {word!r} does not map to a legal kit "
466
+ f"constant (got {name!r} — single-letter entities would lex as "
467
+ "variables and are refused)")
468
+ return name
469
+
470
+
471
+ def _term_for(word: str, variables: "Dict[str, Variable]") -> Node:
472
+ lowered = word.strip().lower()
473
+ if lowered in _PLACEHOLDER_WORDS:
474
+ if lowered not in variables:
475
+ if len(variables) >= 3:
476
+ raise ValueError(
477
+ "proofwriter: more than three distinct placeholder "
478
+ "words in one rule — outside the documented convention")
479
+ variables[lowered] = Variable("xyz"[len(variables)])
480
+ return variables[lowered]
481
+ return Constant(_to_constant_name(word))
482
+
483
+
484
+ def _triple_to_node(parsed, variables: "Dict[str, Variable]") -> Node:
485
+ if (not isinstance(parsed, list) or len(parsed) != 4
486
+ or any(not isinstance(p, str) for p in parsed)):
487
+ raise ValueError(
488
+ f"proofwriter: not a (subject predicate object polarity) triple: "
489
+ f"{parsed!r}")
490
+ subject, predicate, obj, polarity = parsed
491
+ if polarity not in ("+", "-", "~"):
492
+ raise ValueError(f"proofwriter: unknown polarity {polarity!r}")
493
+ if predicate.strip().lower() == "is":
494
+ atom = Atom(_to_predicate_name(obj), (_term_for(subject, variables),))
495
+ else:
496
+ atom = Atom(_to_predicate_name(predicate),
497
+ (_term_for(subject, variables), _term_for(obj, variables)))
498
+ return Not(atom) if polarity in ("-", "~") else atom
499
+
500
+
501
+ def parse_proofwriter_representation(rep: str) -> Node:
502
+ """One ProofWriter ``representation`` string → a closed kit formula.
503
+
504
+ Accepts both shapes the OWA distribution uses: a bare triple (fact or
505
+ question) and a rule ``((cond …) -> conclusion)``. See the comment block
506
+ above for the exact translation convention; every distinct placeholder
507
+ word in a rule is universally quantified, so the result is always closed.
508
+
509
+ Raises:
510
+ ValueError: the string is not a well-formed representation, or a
511
+ name in it has no legal kit rendering.
512
+ """
513
+ tokens = _tokenise_representation(rep)
514
+ if not tokens:
515
+ raise ValueError("proofwriter: empty representation")
516
+ parsed, next_pos = _parse_sexpr(tokens, 0)
517
+ if next_pos != len(tokens):
518
+ raise ValueError(
519
+ f"proofwriter: trailing tokens after the representation: {rep!r}")
520
+
521
+ variables: "Dict[str, Variable]" = {}
522
+ if isinstance(parsed, list) and len(parsed) == 3 and parsed[1] == "->":
523
+ conditions, _, conclusion = parsed
524
+ if not isinstance(conditions, list) or not conditions:
525
+ raise ValueError(
526
+ f"proofwriter: rule without conditions: {rep!r}")
527
+ cond_nodes = [_triple_to_node(c, variables) for c in conditions]
528
+ body = cond_nodes[0]
529
+ for extra in cond_nodes[1:]:
530
+ body = And(body, extra)
531
+ node: Node = Implies(body, _triple_to_node(conclusion, variables))
532
+ for var in reversed(list(variables.values())):
533
+ node = Quantifier("∀", var, node)
534
+ return node
535
+ return _triple_to_node(parsed, variables)
536
+
537
+
538
+ def _named_entries(section: Optional[dict], prefix: str) -> "List[Tuple[str, dict]]":
539
+ """Non-null ``triple<N>``/``rule<N>``/``Q<N>`` entries in numeric order."""
540
+ if not isinstance(section, dict):
541
+ return []
542
+ entries = []
543
+ for key, value in section.items():
544
+ if value is None or not isinstance(value, dict):
545
+ continue
546
+ match = re.fullmatch(re.escape(prefix) + r"(\d+)", key)
547
+ order = int(match.group(1)) if match else float("inf")
548
+ entries.append((order, key, value))
549
+ entries.sort(key=lambda item: (item[0], item[1]))
550
+ return [(key, value) for _, key, value in entries]
551
+
552
+
553
+ def load_proofwriter_structured(
554
+ path: Union[str, Path], *,
555
+ known_bad_ids: FrozenSet[str] = frozenset(),
556
+ convert_fol: bool = True) -> Iterator[DatasetExample]:
557
+ """Stream ONE example PER QUESTION from a structured-OWA JSONL file.
558
+
559
+ Args:
560
+ path: local ``.jsonl`` file with one theory record per line in the
561
+ ``hitachi-nlp/proofwriter_processed_OWA`` schema (``id``,
562
+ ``theory``, ``triples``, ``rules``, ``questions``, …; obtain it
563
+ e.g. via ``load_dataset("hitachi-nlp/proofwriter_processed_OWA",
564
+ "depth-2", split="train").to_json(...)``). NEVER downloaded here.
565
+ known_bad_ids: example ids (``proofwriter:<row id>:<Qn>``) to flag
566
+ ``known_bad=True``.
567
+ convert_fol: with the default ``True``, ``fol_premises`` /
568
+ ``fol_conclusion`` are GENERATED from the record's own symbolic
569
+ representations via :func:`parse_proofwriter_representation`
570
+ (``meta["fol_generated"]`` is set, the verbatim representations
571
+ stay in ``meta``); a record whose representations cannot be
572
+ converted yields its questions with empty FOL fields plus
573
+ ``meta["fol_conversion_error"]``. ``False`` skips generation
574
+ entirely (NL + label only, like :func:`load_proofwriter`).
575
+
576
+ Yields:
577
+ One :class:`DatasetExample` per non-null question, in file order and
578
+ numeric question order; ``nl_premises`` are the fact/rule sentence
579
+ texts, ``label`` is the OWA answer verbatim
580
+ (``"True"``/``"False"``/``"Unknown"``). ``meta["proofs"]`` (the
581
+ record's own, unparsed ``question["proofs"]`` string) and
582
+ ``meta["strategy"]`` were already carried verbatim before this
583
+ docstring was written; ``meta["premise_keys"]`` — the source
584
+ record's ``"tripleN"``/``"ruleN"`` keys, parallel to
585
+ ``meta["premise_representations"]`` — is additive, resolving those
586
+ proof references back to a premise index for
587
+ :func:`check_gold_proof`. Parse ``meta["proofs"]`` with
588
+ :func:`~._proofwriter_proof.parse_question_proof` and verify it
589
+ against this module's own forward-chaining fixpoint with
590
+ :func:`check_gold_proof`.
591
+
592
+ Raises:
593
+ FileNotFoundError / json.JSONDecodeError: as in the other loaders —
594
+ a missing or corrupt FILE fails loudly; per-record conversion
595
+ problems are recorded per example instead.
596
+ """
597
+ path = Path(path)
598
+ with path.open("r", encoding="utf-8") as fh:
599
+ for line_no, raw_line in enumerate(fh):
600
+ line = raw_line.strip()
601
+ if not line:
602
+ continue
603
+ record = json.loads(line)
604
+ row_id = record.get("id", f"pos{line_no}")
605
+
606
+ triples = _named_entries(record.get("triples"), "triple")
607
+ rules = _named_entries(record.get("rules"), "rule")
608
+ premise_entries = triples + rules
609
+ nl_premises = tuple(entry.get("text", "")
610
+ for _, entry in premise_entries)
611
+ premise_reps = tuple(entry.get("representation", "")
612
+ for _, entry in premise_entries)
613
+ premise_keys = tuple(key for key, _ in premise_entries)
614
+
615
+ fol_premises: "Tuple[str, ...]" = ()
616
+ conversion_error: Optional[str] = None
617
+ if convert_fol:
618
+ try:
619
+ fol_premises = tuple(
620
+ parse_proofwriter_representation(rep).to_unicode_str()
621
+ for rep in premise_reps)
622
+ except ValueError as exc:
623
+ conversion_error = f"{type(exc).__name__}: {exc}"
624
+
625
+ for q_key, question in _named_entries(record.get("questions"), "Q"):
626
+ example_id = f"proofwriter:{row_id}:{q_key}"
627
+ q_rep = question.get("representation", "")
628
+ fol_conclusion: Optional[str] = None
629
+ q_error = conversion_error
630
+ if convert_fol and q_error is None:
631
+ try:
632
+ fol_conclusion = parse_proofwriter_representation(
633
+ q_rep).to_unicode_str()
634
+ except ValueError as exc:
635
+ q_error = f"{type(exc).__name__}: {exc}"
636
+
637
+ meta = {
638
+ "row_id": row_id,
639
+ "question_key": q_key,
640
+ "theory": record.get("theory"),
641
+ "premise_representations": list(premise_reps),
642
+ # Parallel to premise_representations/fol_premises: the
643
+ # source record's own "tripleN"/"ruleN" keys, in the SAME
644
+ # order -- added purely so a caller can resolve a
645
+ # question["proofs"] annotation's "tripleN"/"ruleN"
646
+ # references back to a premise index (see
647
+ # check_gold_proof). Every OTHER field here was already
648
+ # present before that addition; this key is new and
649
+ # purely additive, nothing existing was removed/renamed.
650
+ "premise_keys": list(premise_keys),
651
+ "question_representation": q_rep,
652
+ "QDep": question.get("QDep"),
653
+ "strategy": question.get("strategy"),
654
+ "proofs": question.get("proofs"),
655
+ "line_no": line_no,
656
+ }
657
+ if convert_fol and q_error is None:
658
+ meta["fol_generated"] = True
659
+ if q_error is not None:
660
+ meta["fol_conversion_error"] = q_error
661
+
662
+ yield DatasetExample(
663
+ id=example_id,
664
+ nl_premises=nl_premises,
665
+ fol_premises=fol_premises if q_error is None else (),
666
+ nl_conclusion=question.get("question"),
667
+ fol_conclusion=fol_conclusion,
668
+ label=question.get("answer"),
669
+ known_bad=example_id in known_bad_ids,
670
+ meta=meta,
671
+ )
672
+
673
+
674
+ def _collect_constants(nodes) -> "List[Constant]":
675
+ """Every distinct :class:`Constant` in ``nodes``, in a fixed name order."""
676
+ by_name: "Dict[str, Constant]" = {}
677
+ for node in nodes:
678
+ for sub in node.walk():
679
+ if isinstance(sub, Constant):
680
+ by_name.setdefault(sub.name, sub)
681
+ return [by_name[name] for name in sorted(by_name)]
682
+
683
+
684
+ def _as_literal(node: Node):
685
+ """``(atom, positive)`` for a literal node, else ``ValueError``."""
686
+ if isinstance(node, Atom):
687
+ return node, True
688
+ if isinstance(node, Not) and isinstance(node.formula, Atom):
689
+ return node.formula, False
690
+ raise ValueError(
691
+ f"proofwriter: {node.to_unicode_str()!r} is not a literal — the CWA "
692
+ "theory fragment is ground literals and (∀-quantified) rules "
693
+ "'literal ∧ … ∧ literal → literal'.")
694
+
695
+
696
+ def _as_rule(premise: Node):
697
+ """Decompose one premise into ``(variables, body_literals, head_literal)``.
698
+
699
+ A ground literal comes back with an empty body (a fact). Anything outside
700
+ the fact/rule shape raises ``ValueError``.
701
+ """
702
+ variables: "List[Variable]" = []
703
+ node = premise
704
+ while isinstance(node, Quantifier):
705
+ if node.type not in ("∀", "forall"):
706
+ raise ValueError(
707
+ f"proofwriter: CWA premises must be universally quantified, "
708
+ f"got {node.type!r} in {premise.to_unicode_str()!r}")
709
+ variables.append(node.variable)
710
+ node = node.formula
711
+ if isinstance(node, Implies):
712
+ body: "List[Node]" = []
713
+ stack = [node.left]
714
+ while stack:
715
+ item = stack.pop()
716
+ if isinstance(item, And):
717
+ stack.append(item.right)
718
+ stack.append(item.left)
719
+ else:
720
+ body.append(item)
721
+ body_literals = [_as_literal(b) for b in body]
722
+ head_literal = _as_literal(node.right)
723
+ return variables, body_literals, head_literal
724
+ if variables:
725
+ raise ValueError(
726
+ f"proofwriter: quantified premise without an implication is "
727
+ f"outside the CWA fragment: {premise.to_unicode_str()!r}")
728
+ return [], [], _as_literal(node)
729
+
730
+
731
+ def _ground_rule(rule, constants: "List[Constant]"):
732
+ """One ``(variables, body, head)`` (see :func:`_as_rule`) → its list of
733
+ GROUND ``(body, head)`` instances over ``constants`` (``[]`` if the rule
734
+ is variable-free — a plain ground fact/implication grounds to itself)."""
735
+ variables, body, head = rule
736
+ if not variables:
737
+ return [(body, head)]
738
+ if not constants:
739
+ return []
740
+ groundings = []
741
+ for values in itertools.product(constants, repeat=len(variables)):
742
+ g_body = []
743
+ for atom, positive in body:
744
+ g_atom = atom
745
+ for var, value in zip(variables, values):
746
+ g_atom = substitute(g_atom, var, value)
747
+ g_body.append((g_atom, positive))
748
+ h_atom, h_positive = head
749
+ for var, value in zip(variables, values):
750
+ h_atom = substitute(h_atom, var, value)
751
+ groundings.append((g_body, (h_atom, h_positive)))
752
+ return groundings
753
+
754
+
755
+ def _closed_model(premises: "List[Node]", constants: "List[Constant]", *,
756
+ record_provenance: bool = False):
757
+ """The theory's closed model, by STRATIFIED forward chaining.
758
+
759
+ Returns ``(true_atoms, has_naf)`` — the set of ground-atom keys
760
+ (``key_text``: the text of the atom with every constant written by its name,
761
+ which is not the text of the formula where the formula writes a constant in
762
+ quotes) true in the perfect model, and whether any rule
763
+ body used negation (i.e. the theory is beyond the definite fragment, so
764
+ membership is NOT classical entailment and must not be cross-checked
765
+ against a classical prover). With ``record_provenance=True`` a THIRD
766
+ element is returned, ``provenance`` (see below); with the default
767
+ ``False`` the return shape is EXACTLY the 2-tuple above, unchanged —
768
+ every existing caller (:func:`solve_structured_example`) is untouched.
769
+
770
+ Semantics: negation in a rule body is NEGATION AS FAILURE, evaluated
771
+ stratum by stratum over the GROUND dependency graph — LOCAL
772
+ stratification: a ground atom that is tested negatively is fully
773
+ fixpointed in a strictly LOWER stratum before any ground rule reading
774
+ its absence may fire, which is exactly the perfect-model semantics of
775
+ locally stratified logic programs. Stratifying ground atoms rather than
776
+ predicate symbols is strictly more general and is required in practice:
777
+ real ProofWriter theories contain rules like ``¬Likes(mouse, dog) →
778
+ Likes(dog, rabbit)`` whose predicate graph has a negative self-loop but
779
+ whose ground graph is acyclic. Only a GROUND cycle through negation
780
+ (e.g. ``¬P(a) → P(a)``) has no local stratification (it would need
781
+ well-founded/stable-model semantics) and raises ``ValueError``. A
782
+ negative HEAD (a rule or fact concluding ``¬X``) derives no positive
783
+ atom; it is evaluated once against the COMPLETED perfect model and
784
+ tracked only for the final consistency check — a theory that derives
785
+ some atom both positively and negatively is inconsistent under CWA and
786
+ raises rather than answering arbitrarily.
787
+
788
+ ``provenance`` (only computed when ``record_provenance=True``, for
789
+ :func:`check_gold_proof`): ``{signed_key: [(premise_index,
790
+ antecedent_keys), …]}`` — ``signed_key`` is the SIGNED rendering of a
791
+ derived atom (``"Foo(a)"`` for a positive derivation, ``"¬Foo(a)"`` for
792
+ a negative one, via a negative-headed rule or fact — real ProofWriter
793
+ theories have both, see the module's ``check_gold_proof`` docstring);
794
+ each ``premise_index`` is the position, in the ORIGINAL ``premises``
795
+ list, of the fact/rule whose grounding fired; ``antecedent_keys`` is the
796
+ ordered tuple of each required body literal's OWN SIGNED key (bare for a
797
+ positive requirement, ``"¬"``-prefixed for a negation-as-failure one) —
798
+ ``()`` for a fact. Real ProofWriter theories DO use negation-as-failure
799
+ conditions (209 of 2401 real rules fetched from ``hitachi-nlp/
800
+ proofwriter_processed_OWA`` — depths 0/1/2/3/3ext/3ext-NatLang/5/
801
+ NatLang/birds-electricity — have one; ProofWriter's OWN grammar spells
802
+ this ``"~"``, distinct from the ``"-"`` a FACT or rule HEAD uses for a
803
+ flat negative assertion — see :func:`parse_proofwriter_representation`'s
804
+ ``_triple_to_node``), and its own gold ``proofs`` annotation cites an
805
+ EXPLICIT derivation of ``¬X`` for such a condition whenever the theory
806
+ has one (rather than leaving it uncited as bare absence) — signing
807
+ ``antecedent_keys`` is what lets :func:`check_gold_proof` look each one
808
+ up as a recursive ``signed_key`` regardless of polarity. A
809
+ ``signed_key`` can have SEVERAL entries (several grounded rules/facts
810
+ derived it — the "OR-forest" a gold proof may cite any one branch of).
811
+ Signing ``antecedent_keys`` does NOT change ``true_atoms``/``stratum``
812
+ below: NAF satisfaction there is still checked by plain ABSENCE from
813
+ ``true_atoms`` (``(key_text(atom) in true_atoms) == positive``),
814
+ exactly ProofWriter's own stratified-fixpoint semantics; ``provenance``
815
+ is a side record of what ALSO fired explicitly, not a different
816
+ satisfaction rule.
817
+ """
818
+ parsed = [_as_rule(p) for p in premises]
819
+ has_naf = any(not positive for _, body, _ in parsed
820
+ for _, positive in body)
821
+
822
+ # Ground every rule over the finite constant set, remembering which
823
+ # ORIGINAL premise (index into `premises`) each grounding came from --
824
+ # needed only for `provenance`, but cheap enough to always compute.
825
+ ground_rules: "List[Tuple[list, tuple]]" = []
826
+ origin: "List[int]" = []
827
+ for premise_index, rule in enumerate(parsed):
828
+ for grounding in _ground_rule(rule, constants):
829
+ ground_rules.append(grounding)
830
+ origin.append(premise_index)
831
+
832
+ # LOCAL stratification: strata live on GROUND ATOMS. A positive body
833
+ # atom forces its head onto the same-or-higher stratum, a negated one
834
+ # onto a strictly higher stratum. The iteration is Bellman-Ford-like:
835
+ # each atom's stratum is bounded by #atoms on any locally stratifiable
836
+ # program, so non-convergence within #atoms+1 rounds means a ground
837
+ # cycle through negation. Negative-head rules derive nothing and are
838
+ # left out of the graph (they are evaluated against the finished model
839
+ # below).
840
+ stratum: "Dict[str, int]" = {}
841
+ for body, (head_atom, _) in ground_rules:
842
+ stratum.setdefault(key_text(head_atom), 0)
843
+ for atom, _ in body:
844
+ stratum.setdefault(key_text(atom), 0)
845
+ for _ in range(len(stratum) + 1):
846
+ changed = False
847
+ for body, (head_atom, head_positive) in ground_rules:
848
+ if not head_positive:
849
+ continue
850
+ head_key = key_text(head_atom)
851
+ for atom, positive in body:
852
+ required = (stratum[key_text(atom)]
853
+ + (0 if positive else 1))
854
+ if stratum[head_key] < required:
855
+ stratum[head_key] = required
856
+ changed = True
857
+ if not changed:
858
+ break
859
+ else:
860
+ raise ValueError(
861
+ "proofwriter: the CWA theory has a GROUND cycle through negation "
862
+ "(not even locally stratifiable) — its negation-as-failure "
863
+ "semantics is not well-defined by stratified forward chaining; "
864
+ "refusing rather than guessing (well-founded semantics is out of "
865
+ "scope).")
866
+
867
+ provenance: "Optional[Dict[str, set]]" = {} if record_provenance else None
868
+
869
+ def _record(signed_key: str, gi: int, body) -> None:
870
+ if provenance is None:
871
+ return
872
+ # SIGNED per literal (bare for a positive requirement, "¬"-prefixed
873
+ # for a negation-as-failure one) -- a NAF antecedent's signed key is
874
+ # what a gold proof cites when it names an EXPLICIT negative
875
+ # derivation for it (see check_gold_proof's docstring on why real
876
+ # ProofWriter proofs prefer a constructive ¬X derivation over silent
877
+ # absence whenever one exists).
878
+ ante_keys = tuple(
879
+ key_text(atom) if positive else key_text(Not(atom))
880
+ for atom, positive in body)
881
+ provenance.setdefault(signed_key, set()).add((origin[gi], ante_keys))
882
+
883
+ true_atoms: set = set()
884
+ negative_atoms: set = set()
885
+ positive_rules = [
886
+ (gi, body, head) for gi, (body, head) in enumerate(ground_rules)
887
+ if head[1]
888
+ ]
889
+ max_stratum = max(
890
+ (stratum[key_text(head[0])] for _, _, head in positive_rules),
891
+ default=0)
892
+ for level in range(max_stratum + 1):
893
+ level_rules = [
894
+ (gi, body, head) for gi, body, head in positive_rules
895
+ if stratum[key_text(head[0])] == level
896
+ ]
897
+ changed = True
898
+ while changed:
899
+ changed = False
900
+ for gi, body, (head_atom, _) in level_rules:
901
+ fires = all(
902
+ (key_text(atom) in true_atoms) == positive
903
+ for atom, positive in body
904
+ )
905
+ if not fires:
906
+ continue
907
+ key = key_text(head_atom)
908
+ _record(key, gi, body)
909
+ if key not in true_atoms:
910
+ true_atoms.add(key)
911
+ changed = True
912
+
913
+ # Negative heads read the COMPLETED model (previously they fired at
914
+ # their head's level, silently missing body atoms derived later).
915
+ for gi, (body, (head_atom, head_positive)) in enumerate(ground_rules):
916
+ if head_positive:
917
+ continue
918
+ if all((key_text(atom) in true_atoms) == positive
919
+ for atom, positive in body):
920
+ negative_atoms.add(key_text(head_atom))
921
+ _record(key_text(Not(head_atom)), gi, body)
922
+
923
+ contradictions = sorted(true_atoms & negative_atoms)
924
+ if contradictions:
925
+ raise ValueError(
926
+ f"proofwriter: the CWA theory derives {contradictions[0]!r} both "
927
+ "positively and negatively — inconsistent under the closed-world "
928
+ "reading; refusing rather than answering arbitrarily.")
929
+ if record_provenance:
930
+ ordered_provenance = {
931
+ key: sorted(entries) for key, entries in provenance.items()
932
+ }
933
+ return frozenset(true_atoms), has_naf, ordered_provenance
934
+ return frozenset(true_atoms), has_naf
935
+
936
+
937
+ def _cwa_holds(formula: Node, atom_oracle, constants) -> bool:
938
+ """Two-valued truth of ``formula`` in the CLOSED model.
939
+
940
+ The closed (minimal) model is represented by its atom valuation:
941
+ ``atom_oracle(atom)`` answers "is this ground atom derivable from the
942
+ premises?" — everything not derivable is FALSE, which is exactly the
943
+ closed-world reading. Connectives are ordinary two-valued FOL on top of
944
+ that valuation, and quantifiers range over the theory's (finite) set of
945
+ constants — plain model checking, not entailment.
946
+ """
947
+ if isinstance(formula, Atom):
948
+ return atom_oracle(formula)
949
+ if isinstance(formula, Not):
950
+ return not _cwa_holds(formula.formula, atom_oracle, constants)
951
+ if isinstance(formula, And):
952
+ return (_cwa_holds(formula.left, atom_oracle, constants)
953
+ and _cwa_holds(formula.right, atom_oracle, constants))
954
+ if isinstance(formula, Or):
955
+ return (_cwa_holds(formula.left, atom_oracle, constants)
956
+ or _cwa_holds(formula.right, atom_oracle, constants))
957
+ if isinstance(formula, Xor):
958
+ return (_cwa_holds(formula.left, atom_oracle, constants)
959
+ != _cwa_holds(formula.right, atom_oracle, constants))
960
+ if isinstance(formula, Implies):
961
+ return ((not _cwa_holds(formula.left, atom_oracle, constants))
962
+ or _cwa_holds(formula.right, atom_oracle, constants))
963
+ if isinstance(formula, Iff):
964
+ return (_cwa_holds(formula.left, atom_oracle, constants)
965
+ == _cwa_holds(formula.right, atom_oracle, constants))
966
+ if isinstance(formula, Quantifier):
967
+ instances = (
968
+ _cwa_holds(substitute(formula.formula, formula.variable, c),
969
+ atom_oracle, constants)
970
+ for c in constants)
971
+ if formula.type in ("∀", "forall"):
972
+ return all(instances)
973
+ if formula.type in ("∃", "exists"):
974
+ return any(instances)
975
+ raise ValueError(
976
+ f"proofwriter: unknown quantifier {formula.type!r} in a CWA query")
977
+ raise ValueError(
978
+ f"proofwriter: node {type(formula).__name__} is outside the CWA "
979
+ "query fragment (atoms, ¬ ∧ ∨ ⊕ → ↔, ∀/∃ over the constants)")
980
+
981
+
982
+ def solve_structured_example(example: DatasetExample, *,
983
+ semantics: str = "owa",
984
+ on_indefinite: str = "label",
985
+ **prove_kwargs) -> dict:
986
+ """Decide one generated-FOL ProofWriter example against its gold label.
987
+
988
+ The prover is the CALLER'S choice: every keyword in ``prove_kwargs`` goes
989
+ verbatim to :func:`unicode_logic_kit.api.prove` — e.g.
990
+ ``backends=["vampire"]`` or ``backends=["z3"], timeout=5000`` — so any
991
+ registered ATP can drive either semantics.
992
+
993
+ ``semantics="owa"`` (default) is the open-world three-way cascade over
994
+ ENTAILMENT: premises ⊨ q → ``"True"``; else premises ⊨ ¬q → ``"False"``;
995
+ else ``"Unknown"``. This maps the OWA labels exactly.
996
+
997
+ ``on_indefinite`` controls how a NON-DEFINITIVE prover outcome (a
998
+ ``Verdict`` with status ``unknown`` or ``error`` — timeouts, hit bounds,
999
+ honest incompleteness; ``Verdict.is_definitive`` is the exact test) is
1000
+ interpreted when the OWA cascade reaches its third arm. The distinction
1001
+ it preserves: ``"Unknown"`` can be ESTABLISHED (both directions
1002
+ definitively REFUTED — the prover found countermodels both ways, so the
1003
+ question is provably underdetermined) or merely DEFAULTED to (some leg
1004
+ timed out / gave up — the prover failed to tell).
1005
+
1006
+ - ``"label"`` (default): any not-proved outcome flows into the dataset's
1007
+ ``"Unknown"`` label — the pragmatic scoring mode, correct whenever the
1008
+ chosen prover is decisive on the fragment (z3 on these ground/Horn
1009
+ theories is).
1010
+ - ``"abstain"``: ``"Unknown"`` only when BOTH legs are definitively
1011
+ REFUTED; if any leg is indefinite, ``predicted`` is ``None`` — so
1012
+ evaluation numbers cannot silently credit a prover timeout as a
1013
+ correct "Unknown" prediction. The verdict dicts show which leg failed
1014
+ and why (``status``/``reason``, e.g. ``timeout``).
1015
+ - ``"raise"``: like ``"abstain"``, but an indefinite leg raises
1016
+ ``ValueError`` — for pipelines that must not contain holes.
1017
+
1018
+ Under ``semantics="cwa"`` the fixpoint decides every atom, so
1019
+ ``on_indefinite`` has no effect there (the cross-check already records,
1020
+ and never alarms on, an indefinite prover verdict); the argument is
1021
+ still validated.
1022
+
1023
+ ``semantics="cwa"`` is closed-world MODEL CHECKING — ordinary two-valued
1024
+ FOL evaluation in the closed model, which is COMPUTED exactly by
1025
+ stratified forward chaining over the grounded theory
1026
+ (:func:`_closed_model`): a ground atom is true iff it is in the perfect
1027
+ model, and negation/connectives/quantifiers in the QUERY are evaluated
1028
+ compositionally on top (¬q is true iff q is not in the model; ∀/∃ range
1029
+ over the theory's constants). There is no ``"Unknown"``: the result is
1030
+ ``"True"`` or ``"False"``. Rules with NEGATED BODY literals are
1031
+ supported with their standard negation-as-failure reading via LOCAL
1032
+ stratification over the ground dependency graph (the negatively-tested
1033
+ ground atom is fully fixpointed in a lower stratum first); only a
1034
+ GROUND cycle through negation (no local stratification exists) or a
1035
+ theory that derives an atom both positively and negatively
1036
+ (inconsistent under CWA) raises ``ValueError``. On
1037
+ DEFINITE theories (no negated bodies) the least model coincides with
1038
+ classical entailment, so every queried atom is additionally
1039
+ CROSS-CHECKED against the caller's chosen ATP — a definitive
1040
+ disagreement (prover proves an atom the fixpoint excludes, or refutes
1041
+ one it contains) raises a soundness alarm instead of returning either
1042
+ answer; an honest prover UNKNOWN is recorded, never alarmed on.
1043
+
1044
+ Returns ``{"predicted": ..., "verdict": ..., "verdict_negated": ...}``;
1045
+ under ``"cwa"`` the two verdict slots are ``None`` and every per-atom
1046
+ oracle call is recorded in an additional ``"atom_calls"`` list
1047
+ (atom, derivable, full verdict dict). Requires an example produced with
1048
+ ``convert_fol=True`` and without a recorded conversion error; anything
1049
+ else raises ``ValueError``.
1050
+ """
1051
+ from ... import api
1052
+
1053
+ if semantics not in ("owa", "cwa"):
1054
+ raise ValueError(
1055
+ f"proofwriter: semantics must be 'owa' or 'cwa', got {semantics!r}")
1056
+ if on_indefinite not in ("label", "abstain", "raise"):
1057
+ raise ValueError(
1058
+ f"proofwriter: on_indefinite must be 'label', 'abstain' or "
1059
+ f"'raise', got {on_indefinite!r}")
1060
+ if example.meta.get("fol_conversion_error"):
1061
+ raise ValueError(
1062
+ f"proofwriter: example {example.id} carries a conversion error "
1063
+ f"({example.meta['fol_conversion_error']}) — cannot solve it.")
1064
+ if example.fol_conclusion is None:
1065
+ raise ValueError(
1066
+ f"proofwriter: example {example.id} has no generated conclusion "
1067
+ "— was it loaded with convert_fol=False?")
1068
+
1069
+ def _parse(text: str) -> Node:
1070
+ parsed = api.parse_any(text)
1071
+ if not parsed.ok:
1072
+ raise ValueError(
1073
+ f"proofwriter: example {example.id}: generated formula "
1074
+ f"{text!r} does not parse under the kit grammar")
1075
+ return parsed.formula
1076
+
1077
+ premises = [_parse(p) for p in example.fol_premises]
1078
+ conclusion = _parse(example.fol_conclusion)
1079
+
1080
+ if semantics == "owa":
1081
+ verdict = api.prove(conclusion, premises, **prove_kwargs)
1082
+ if verdict.status == "proved":
1083
+ return {"predicted": "True", "verdict": verdict.to_dict(),
1084
+ "verdict_negated": None}
1085
+ negated = api.prove(Not(conclusion), premises, **prove_kwargs)
1086
+ if negated.status == "proved":
1087
+ predicted: "Optional[str]" = "False"
1088
+ elif on_indefinite == "label":
1089
+ predicted = "Unknown"
1090
+ elif verdict.status == "refuted" and negated.status == "refuted":
1091
+ # Underdetermination ESTABLISHED: countermodels exist against
1092
+ # both directions — "Unknown" is a definitive answer here, not
1093
+ # a fallback, so abstain/raise modes still label it.
1094
+ predicted = "Unknown"
1095
+ elif on_indefinite == "raise":
1096
+ raise ValueError(
1097
+ f"proofwriter: example {example.id}: indefinite prover "
1098
+ f"outcome (goal: {verdict.status}/{verdict.reason}, negated: "
1099
+ f"{negated.status}/{negated.reason}) with "
1100
+ "on_indefinite='raise' — the cascade cannot honestly assign "
1101
+ "a label.")
1102
+ else: # "abstain"
1103
+ predicted = None
1104
+ return {"predicted": predicted, "verdict": verdict.to_dict(),
1105
+ "verdict_negated": negated.to_dict()}
1106
+
1107
+ constants = _collect_constants(premises + [conclusion])
1108
+ model, has_naf = _closed_model(premises, constants)
1109
+ atom_calls: "List[dict]" = []
1110
+ cache: "Dict[str, bool]" = {}
1111
+
1112
+ def atom_oracle(atom: Atom) -> bool:
1113
+ key = key_text(atom)
1114
+ if key not in cache:
1115
+ in_model = key in model
1116
+ record: dict = {"atom": key, "derivable": in_model}
1117
+ if not has_naf:
1118
+ # Definite theory: least model ⟺ classical entailment, so
1119
+ # the caller's ATP serves as an independent cross-check. Only
1120
+ # a DEFINITIVE disagreement is a soundness alarm; an honest
1121
+ # UNKNOWN (timeout, bound) is recorded, not alarmed on.
1122
+ verdict = api.prove(atom, premises, **prove_kwargs)
1123
+ record["verdict"] = verdict.to_dict()
1124
+ if ((verdict.status == "proved" and not in_model)
1125
+ or (verdict.status == "refuted" and in_model)):
1126
+ raise ValueError(
1127
+ f"proofwriter: soundness alarm on {example.id}: the "
1128
+ f"closed-model fixpoint says {key!r} is "
1129
+ f"{'in' if in_model else 'NOT in'} the least model, "
1130
+ f"but backend {verdict.backend!r} definitively says "
1131
+ "the opposite — on a definite theory these must "
1132
+ "coincide; refusing to answer.")
1133
+ cache[key] = in_model
1134
+ atom_calls.append(record)
1135
+ return cache[key]
1136
+
1137
+ holds = _cwa_holds(conclusion, atom_oracle, constants)
1138
+ return {"predicted": "True" if holds else "False",
1139
+ "verdict": None, "verdict_negated": None,
1140
+ "atom_calls": atom_calls}
1141
+
1142
+
1143
+ # --------------------------------------------------------------------------- #
1144
+ # check_gold_proof: verify a structured-route example's OWN question["proofs"]
1145
+ # annotation against the kit's own forward-chaining fixpoint — see
1146
+ # _proofwriter_proof.py for the annotation grammar this parses.
1147
+ # --------------------------------------------------------------------------- #
1148
+
1149
+ def _negate(node: Node) -> Node:
1150
+ """Logical negation WITHOUT double-negating: ``¬X`` → ``X``, else ``¬``."""
1151
+ return node.formula if isinstance(node, Not) else Not(node)
1152
+
1153
+
1154
+ #: Every strategy tag observed across 390 real rows / 5452 questions fetched
1155
+ #: from ``hitachi-nlp/proofwriter_processed_OWA`` (every published config)
1156
+ #: plus the real AllenAI CWA fixture — see :func:`_target_for_strategy`.
1157
+ _KNOWN_STRATEGIES = frozenset({
1158
+ "proof", "inv-proof", "rconc", "inv-rconc", "random", "inv-random",
1159
+ })
1160
+
1161
+
1162
+ def _target_for_strategy(conclusion: Node, strategy: "Optional[str]",
1163
+ example_id: str) -> Node:
1164
+ """The literal a gold ``proofs`` annotation is ABOUT, given the
1165
+ question's own strategy tag.
1166
+
1167
+ ProofWriter's ``"proof"``/``"rconc"``/``"random"`` strategies derive (or
1168
+ fail to derive) the question's conclusion EXACTLY as generated —
1169
+ whatever polarity the question itself has (a question can be phrased
1170
+ negatively, e.g. ``"The rabbit is not round."``, and be settled by
1171
+ directly citing a NEGATIVE fact — confirmed in real data: 96 of 1329
1172
+ real ``"proof"``-strategy questions fetched here are phrased negatively
1173
+ and their proof directly cites a negative-headed fact/rule). The
1174
+ ``"inv-*"`` strategies derive the OPPOSITE of the question instead
1175
+ (that is what makes the label ``"False"``/the "inv-" failure a
1176
+ not-entailed positive/negative pair) — never a double negation, since
1177
+ the question's own polarity is stripped, not added to.
1178
+ """
1179
+ if strategy not in _KNOWN_STRATEGIES:
1180
+ raise ValueError(
1181
+ f"proofwriter: example {example_id} has meta['strategy'] "
1182
+ f"{strategy!r}, not one of {sorted(_KNOWN_STRATEGIES)} — cannot "
1183
+ "tell which literal the gold proof is about")
1184
+ if strategy.startswith("inv-"):
1185
+ return _negate(conclusion)
1186
+ return conclusion
1187
+
1188
+
1189
+ def _rule_body_holds(rule_node: Node, target_key: str,
1190
+ true_atoms: "FrozenSet[str]",
1191
+ constants: "List[Constant]") -> "Optional[bool]":
1192
+ """Is there SOME grounding of ``rule_node`` whose HEAD is ``target_key``
1193
+ (a SIGNED atom key — see :func:`_closed_model`'s ``provenance``) with a
1194
+ satisfied body?
1195
+
1196
+ This is an EXISTENTIAL question over every grounding whose head matches
1197
+ (a rule with a variable that occurs in the body but not the head, or
1198
+ otherwise sharing its head atom across more than one grounding, can have
1199
+ several) — all matching groundings are checked, not just the first one
1200
+ ``itertools.product`` happens to visit, so a non-firing grounding never
1201
+ masks a later firing one.
1202
+
1203
+ Returns ``None`` when no grounding of ``rule_node`` concludes
1204
+ ``target_key`` at all (the rule cannot structurally produce this atom,
1205
+ e.g. under any constant substitution its head is a different
1206
+ predicate/arguments) — a gold "deepest failure" witness naming such a
1207
+ rule cannot be confirmed or refuted this way, which
1208
+ :func:`check_gold_proof` surfaces rather than silently treating as
1209
+ either outcome.
1210
+ """
1211
+ rule = _as_rule(rule_node)
1212
+ matches = []
1213
+ for g_body, (g_head_atom, g_head_positive) in _ground_rule(rule, constants):
1214
+ signed = (key_text(g_head_atom) if g_head_positive
1215
+ else key_text(Not(g_head_atom)))
1216
+ if signed != target_key:
1217
+ continue
1218
+ matches.append(g_body)
1219
+ if not matches:
1220
+ return None
1221
+ return any(all((key_text(atom) in true_atoms) == positive
1222
+ for atom, positive in g_body)
1223
+ for g_body in matches)
1224
+
1225
+
1226
+ def _match_gold_derivation(node: "_proof.ProofNode", target_key: str,
1227
+ key_to_index: "Dict[str, int]",
1228
+ provenance: "Dict[str, list]") -> bool:
1229
+ """Does ``node`` (a :class:`~._proofwriter_proof.Leaf` /
1230
+ :class:`~._proofwriter_proof.Naf` / :class:`~._proofwriter_proof.Apply` /
1231
+ :class:`~._proofwriter_proof.Or`) explain how the fixpoint's OWN
1232
+ ``provenance`` derived ``target_key``?
1233
+
1234
+ A :class:`~._proofwriter_proof.Leaf` matches iff the fact it cites fired
1235
+ with no antecedents for exactly ``target_key``; a
1236
+ :class:`~._proofwriter_proof.Naf` matches iff ``target_key`` is a
1237
+ NEGATIVE requirement (starts with ``"¬"``) whose bare positive form has
1238
+ NO provenance entry at all — genuine absence, matching negation-as-
1239
+ failure with no explicit ``¬X`` derivation to cite (see
1240
+ :class:`~._proofwriter_proof.Naf`); an :class:`~._proofwriter_proof.Apply`
1241
+ matches iff SOME provenance entry for ``target_key`` used the SAME rule
1242
+ with the SAME antecedent count, each antecedent recursively matching the
1243
+ corresponding gold sub-term (in order — see :func:`_closed_model`'s
1244
+ ``provenance`` docstring on why body order is preserved and safe to rely
1245
+ on positionally); an :class:`~._proofwriter_proof.Or` matches iff ANY
1246
+ alternative does (the OR-forest offers several valid supports —
1247
+ matching any one is enough).
1248
+ """
1249
+ if isinstance(node, _proof.Leaf):
1250
+ index = key_to_index.get(node.ref)
1251
+ if index is None:
1252
+ raise ValueError(
1253
+ f"proofwriter: gold proof cites unknown fact reference "
1254
+ f"{node.ref!r} — not among this example's premise_keys")
1255
+ return (index, ()) in provenance.get(target_key, ())
1256
+ if isinstance(node, _proof.Naf):
1257
+ if not target_key.startswith("¬"):
1258
+ raise ValueError(
1259
+ f"proofwriter: gold proof cites NAF (negation-as-failure) "
1260
+ f"for {target_key!r}, which is not itself a negative "
1261
+ "requirement — outside the documented grammar (NAF only "
1262
+ "ever justifies a '~'-polarity body condition)")
1263
+ return target_key[1:] not in provenance
1264
+ if isinstance(node, _proof.Or):
1265
+ return any(_match_gold_derivation(alt, target_key, key_to_index,
1266
+ provenance)
1267
+ for alt in node.alts)
1268
+ if isinstance(node, _proof.Apply):
1269
+ index = key_to_index.get(node.rule)
1270
+ if index is None:
1271
+ raise ValueError(
1272
+ f"proofwriter: gold proof cites unknown rule reference "
1273
+ f"{node.rule!r} — not among this example's premise_keys")
1274
+ parts = node.args.parts
1275
+ for origin_index, ante_keys in provenance.get(target_key, ()):
1276
+ if origin_index != index or len(ante_keys) != len(parts):
1277
+ continue
1278
+ if all(_match_gold_derivation(sub, ante_keys[i], key_to_index,
1279
+ provenance)
1280
+ for i, sub in enumerate(parts)):
1281
+ return True
1282
+ return False
1283
+ raise ValueError(
1284
+ f"proofwriter: gold proof node {type(node).__name__} is not a "
1285
+ "derivation node (Leaf/Naf/Apply/Or) — a FailWitness at this "
1286
+ "position is handled separately by check_gold_proof, never "
1287
+ "recursed into")
1288
+
1289
+
1290
+ def check_gold_proof(example: DatasetExample) -> dict:
1291
+ """Verify a structured-route example's gold ``question["proofs"]``
1292
+ against the kit's OWN forward-chaining fixpoint (:func:`_closed_model`
1293
+ with ``record_provenance=True``) — a genuine "same derivation" check,
1294
+ because ProofWriter's own generator and this fixpoint are both doing
1295
+ forward chaining over the SAME ground theory (see the module docstring's
1296
+ CWA section). Unlike :func:`solve_structured_example`, no ATP is
1297
+ invoked — the whole point is a SECOND, independent route to the same
1298
+ ground theory's derivable atoms, so this takes no ``prove_kwargs``.
1299
+ Requires an ``example`` produced by :func:`load_proofwriter_structured`
1300
+ with ``convert_fol=True`` (so ``meta["premise_keys"]``/
1301
+ ``meta["proofs"]``/``meta["strategy"]`` and the generated FOL are all
1302
+ present) and without a recorded conversion error.
1303
+
1304
+ Known, narrow disagreement (report, do not repair — see this module's
1305
+ "Independent verification" contract): a real rule body's ``"~"``
1306
+ (negation-as-failure) condition and a fact/rule head's ``"-"`` (strong
1307
+ negation) both lower to the SAME kit ``Not()`` in the generated FOL (see
1308
+ :func:`parse_proofwriter_representation`'s ``_triple_to_node`` —
1309
+ pre-existing, unrelated to this function), so when a ``"~"`` condition
1310
+ is genuinely UNDETERMINED under ProofWriter's own open-world reading
1311
+ (never asserted true OR false) rather than absent-under-closed-world,
1312
+ :func:`_closed_model`'s NAF-as-absence semantics (pre-existing,
1313
+ unrelated to this function, verified 1078/1078 against genuinely
1314
+ CWA-labelled data) can let a rule fire that ProofWriter's own OWA-
1315
+ consistent annotation says should not. Confirmed on 7 of 4550
1316
+ checkable real questions fetched here (0.15%) — every one traced to a
1317
+ rule using ``"~"``; :func:`check_gold_proof` correctly reports these as
1318
+ ``ok=False`` rather than silently agreeing, which is the intended
1319
+ behaviour, not a bug in the parser or this checker.
1320
+
1321
+ Two shapes, per :mod:`._proofwriter_proof`'s grammar:
1322
+
1323
+ * A DERIVATION (``strategy`` ``"proof"``/``"inv-proof"``/``"rconc"``/
1324
+ ``"random"``/``"inv-rconc"``/``"inv-random"`` whose ``proofs`` parses
1325
+ to a :class:`~._proofwriter_proof.Leaf`/:class:`~._proofwriter_proof.Apply`/
1326
+ :class:`~._proofwriter_proof.Or`): the target literal (see
1327
+ :func:`_target_for_strategy`) must be among the atoms the fixpoint's
1328
+ own provenance says were derived by exactly that named chain of
1329
+ facts/rules (:func:`_match_gold_derivation`).
1330
+ * A :class:`~._proofwriter_proof.FailWitness` (``"Unknown"`` answers):
1331
+ confirms the target literal is genuinely NOT derivable (absent from
1332
+ ``provenance``), and — when the witness names a first candidate rule
1333
+ (``rule_chain[0]``) — that THAT rule's own grounding for this target
1334
+ has an unsatisfied body in the fixpoint's COMPLETED perfect model
1335
+ (:func:`_rule_body_holds`). ``true_atoms`` IS the theory's perfect
1336
+ model already (:func:`_closed_model` fully stratifies before
1337
+ returning), so this is the exact ground truth regardless of whether
1338
+ the rule's own body has a negation-as-failure condition — there is no
1339
+ "which round" ambiguity to approximate. **Explicitly NOT verified**:
1340
+ deeper links of a multi-rule failure chain (``rule_chain[1:]`` — real
1341
+ witnesses go up to 5 links deep, see
1342
+ :class:`~._proofwriter_proof.FailWitness`); only the first, named
1343
+ "could this have produced the target" candidate is checked.
1344
+
1345
+ Returns a dict with ``"kind"`` (``"derivation"`` or ``"fail_witness"``),
1346
+ ``"ok"`` (bool), ``"atom"`` (the target's signed key) and ``"gold"``
1347
+ (the parsed :data:`~._proofwriter_proof.ProofNode`); a
1348
+ ``"fail_witness"`` result additionally carries ``"derivable"`` and
1349
+ ``"rule_body_holds"`` (``None`` when ``rule_chain`` is empty or its
1350
+ first rule cannot structurally conclude the target atom).
1351
+
1352
+ Raises:
1353
+ ValueError: the example is missing generated FOL, its
1354
+ ``meta["proofs"]``/``meta["premise_keys"]``/``meta["strategy"]``
1355
+ (i.e. it was not produced by :func:`load_proofwriter_structured`
1356
+ with ``convert_fol=True``), the generated conclusion is not a
1357
+ (possibly negated) atom, or the gold annotation cites a
1358
+ ``tripleN``/``ruleN`` reference outside this example's own
1359
+ premises — never silently ignored or guessed at.
1360
+ """
1361
+ from ... import api
1362
+
1363
+ if example.meta.get("fol_conversion_error"):
1364
+ raise ValueError(
1365
+ f"proofwriter: example {example.id} carries a conversion error "
1366
+ f"({example.meta['fol_conversion_error']}) — cannot check its "
1367
+ "proof.")
1368
+ if example.fol_conclusion is None:
1369
+ raise ValueError(
1370
+ f"proofwriter: example {example.id} has no generated conclusion "
1371
+ "— was it loaded with convert_fol=False?")
1372
+ proofs_text = example.meta.get("proofs")
1373
+ if not proofs_text:
1374
+ raise ValueError(
1375
+ f"proofwriter: example {example.id} has no recorded "
1376
+ "meta['proofs'] annotation to check.")
1377
+ premise_keys = example.meta.get("premise_keys")
1378
+ if not premise_keys:
1379
+ raise ValueError(
1380
+ f"proofwriter: example {example.id} has no meta['premise_keys'] "
1381
+ "— only load_proofwriter_structured's output names its "
1382
+ "triples/rules.")
1383
+
1384
+ def _parse(text: str) -> Node:
1385
+ parsed = api.parse_any(text)
1386
+ if not parsed.ok:
1387
+ raise ValueError(
1388
+ f"proofwriter: example {example.id}: generated formula "
1389
+ f"{text!r} does not parse under the kit grammar")
1390
+ return parsed.formula
1391
+
1392
+ premises = [_parse(p) for p in example.fol_premises]
1393
+ conclusion = _parse(example.fol_conclusion)
1394
+ target = _target_for_strategy(conclusion, example.meta.get("strategy"),
1395
+ example.id)
1396
+ target_atom = target.formula if isinstance(target, Not) else target
1397
+ if not isinstance(target_atom, Atom):
1398
+ raise ValueError(
1399
+ f"proofwriter: example {example.id}: target "
1400
+ f"{target.to_unicode_str()!r} is not a (possibly negated) atom "
1401
+ "— cannot check its proof against a ground fixpoint")
1402
+ # The signed key of the target (``¬Foo(a)`` for a negated atom), the form the
1403
+ # fixpoint's provenance is filed under.
1404
+ target_key = key_text(target)
1405
+
1406
+ constants = _collect_constants(premises + [conclusion])
1407
+ true_atoms, _has_naf, provenance = _closed_model(
1408
+ premises, constants, record_provenance=True)
1409
+ key_to_index = {key: i for i, key in enumerate(premise_keys)}
1410
+ gold = _proof.parse_question_proof(proofs_text)
1411
+
1412
+ if isinstance(gold, _proof.FailWitness):
1413
+ derivable = target_key in provenance
1414
+ rule_body_holds: "Optional[bool]" = None
1415
+ if gold.rule_chain:
1416
+ rule_ref = gold.rule_chain[0]
1417
+ rule_index = key_to_index.get(rule_ref)
1418
+ if rule_index is None:
1419
+ raise ValueError(
1420
+ f"proofwriter: gold failure witness cites unknown rule "
1421
+ f"reference {rule_ref!r} — not among this example's "
1422
+ "premise_keys")
1423
+ rule_body_holds = _rule_body_holds(
1424
+ premises[rule_index], target_key, true_atoms, constants)
1425
+ ok = (not derivable) and (rule_body_holds is not True)
1426
+ return {"kind": "fail_witness", "ok": ok, "atom": target_key,
1427
+ "gold": gold, "derivable": derivable,
1428
+ "rule_body_holds": rule_body_holds}
1429
+
1430
+ ok = _match_gold_derivation(gold, target_key, key_to_index, provenance)
1431
+ return {"kind": "derivation", "ok": ok, "atom": target_key, "gold": gold}