unicode-logic-kit 0.31.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (237) hide show
  1. unicode_logic_kit/__init__.py +385 -0
  2. unicode_logic_kit/__main__.py +520 -0
  3. unicode_logic_kit/_deadline.py +219 -0
  4. unicode_logic_kit/ace/__init__.py +126 -0
  5. unicode_logic_kit/ace/_align.py +135 -0
  6. unicode_logic_kit/ace/chem_lexicon.py +128 -0
  7. unicode_logic_kit/ace/drs_reader.py +570 -0
  8. unicode_logic_kit/ace/mapping.py +666 -0
  9. unicode_logic_kit/ace/reverse_modal.py +138 -0
  10. unicode_logic_kit/ace/runner.py +551 -0
  11. unicode_logic_kit/ace/translate.py +452 -0
  12. unicode_logic_kit/ace/verbalize.py +1070 -0
  13. unicode_logic_kit/api.py +1284 -0
  14. unicode_logic_kit/atp/__init__.py +177 -0
  15. unicode_logic_kit/atp/_ascii_names.py +113 -0
  16. unicode_logic_kit/atp/_html.py +72 -0
  17. unicode_logic_kit/atp/_substructural_input.py +228 -0
  18. unicode_logic_kit/atp/_tff_problem.py +715 -0
  19. unicode_logic_kit/atp/_tptp_problem.py +1111 -0
  20. unicode_logic_kit/atp/_writer_support.py +289 -0
  21. unicode_logic_kit/atp/clingo_backend.py +1180 -0
  22. unicode_logic_kit/atp/cvc5_backend.py +1385 -0
  23. unicode_logic_kit/atp/eprover_backend.py +732 -0
  24. unicode_logic_kit/atp/finite_domain.py +1055 -0
  25. unicode_logic_kit/atp/fitch.py +1547 -0
  26. unicode_logic_kit/atp/fitch_search.py +551 -0
  27. unicode_logic_kit/atp/hets_backend.py +339 -0
  28. unicode_logic_kit/atp/hybrid_down.py +120 -0
  29. unicode_logic_kit/atp/incremental.py +250 -0
  30. unicode_logic_kit/atp/kripke_enum.py +741 -0
  31. unicode_logic_kit/atp/lambek.py +436 -0
  32. unicode_logic_kit/atp/leo3_backend.py +332 -0
  33. unicode_logic_kit/atp/linear.py +738 -0
  34. unicode_logic_kit/atp/lj.py +705 -0
  35. unicode_logic_kit/atp/logic_backends.py +566 -0
  36. unicode_logic_kit/atp/ltl_tableau.py +1084 -0
  37. unicode_logic_kit/atp/minizinc_backend.py +1402 -0
  38. unicode_logic_kit/atp/modal_tableau.py +1382 -0
  39. unicode_logic_kit/atp/nanocop_backend.py +410 -0
  40. unicode_logic_kit/atp/portfolio.py +489 -0
  41. unicode_logic_kit/atp/protocol.py +1803 -0
  42. unicode_logic_kit/atp/prover9_entailment.py +1153 -0
  43. unicode_logic_kit/atp/resolution.py +1376 -0
  44. unicode_logic_kit/atp/resolution_check.py +1114 -0
  45. unicode_logic_kit/atp/sequent.py +1050 -0
  46. unicode_logic_kit/atp/tableau.py +921 -0
  47. unicode_logic_kit/atp/tableau_check.py +543 -0
  48. unicode_logic_kit/atp/tptp_ncl.py +811 -0
  49. unicode_logic_kit/atp/tptp_tff.py +1546 -0
  50. unicode_logic_kit/atp/tstp.py +1333 -0
  51. unicode_logic_kit/atp/tstp_check.py +1096 -0
  52. unicode_logic_kit/atp/twee_backend.py +236 -0
  53. unicode_logic_kit/atp/twee_check.py +711 -0
  54. unicode_logic_kit/atp/twee_entailment.py +953 -0
  55. unicode_logic_kit/atp/vampire_entailment.py +540 -0
  56. unicode_logic_kit/atp/z3_arith.py +470 -0
  57. unicode_logic_kit/atp/z3_equivalence.py +36 -0
  58. unicode_logic_kit/atp/z3_fuzzy.py +362 -0
  59. unicode_logic_kit/atp/z3_input.py +500 -0
  60. unicode_logic_kit/atp/z3_models.py +208 -0
  61. unicode_logic_kit/chem/__init__.py +88 -0
  62. unicode_logic_kit/chem/_naming.py +284 -0
  63. unicode_logic_kit/chem/cache.py +185 -0
  64. unicode_logic_kit/chem/interop.py +244 -0
  65. unicode_logic_kit/chem/mol.py +525 -0
  66. unicode_logic_kit/chem/signature.py +112 -0
  67. unicode_logic_kit/comorphism.py +497 -0
  68. unicode_logic_kit/dl/__init__.py +384 -0
  69. unicode_logic_kit/dl/classification.py +227 -0
  70. unicode_logic_kit/dl/concepts.py +632 -0
  71. unicode_logic_kit/dl/datatypes.py +818 -0
  72. unicode_logic_kit/dl/owl_functional.py +2433 -0
  73. unicode_logic_kit/dl/owl_manchester.py +1637 -0
  74. unicode_logic_kit/dl/owl_reasoner.py +790 -0
  75. unicode_logic_kit/dl/parser.py +391 -0
  76. unicode_logic_kit/dl/tableau.py +4048 -0
  77. unicode_logic_kit/dl/translate.py +2704 -0
  78. unicode_logic_kit/drt/__init__.py +94 -0
  79. unicode_logic_kit/drt/export.py +179 -0
  80. unicode_logic_kit/drt/nodes.py +506 -0
  81. unicode_logic_kit/drt/parser.py +965 -0
  82. unicode_logic_kit/drt/resolve.py +195 -0
  83. unicode_logic_kit/drt/reverse.py +175 -0
  84. unicode_logic_kit/eval/__init__.py +106 -0
  85. unicode_logic_kit/eval/batch.py +382 -0
  86. unicode_logic_kit/eval/canonical.py +663 -0
  87. unicode_logic_kit/eval/chem_batch.py +606 -0
  88. unicode_logic_kit/eval/converses.py +200 -0
  89. unicode_logic_kit/eval/datasets/__init__.py +136 -0
  90. unicode_logic_kit/eval/datasets/_base.py +263 -0
  91. unicode_logic_kit/eval/datasets/_proofwriter_proof.py +422 -0
  92. unicode_logic_kit/eval/datasets/c3po.py +678 -0
  93. unicode_logic_kit/eval/datasets/folio.py +158 -0
  94. unicode_logic_kit/eval/datasets/fracas.py +418 -0
  95. unicode_logic_kit/eval/datasets/groves.py +191 -0
  96. unicode_logic_kit/eval/datasets/logicbench.py +467 -0
  97. unicode_logic_kit/eval/datasets/logicnli.py +303 -0
  98. unicode_logic_kit/eval/datasets/malls.py +133 -0
  99. unicode_logic_kit/eval/datasets/pfolio.py +594 -0
  100. unicode_logic_kit/eval/datasets/pmb.py +242 -0
  101. unicode_logic_kit/eval/datasets/prontoqa.py +611 -0
  102. unicode_logic_kit/eval/datasets/proofwriter.py +1431 -0
  103. unicode_logic_kit/eval/datasets/proverqa.py +674 -0
  104. unicode_logic_kit/eval/datasets/willow.py +478 -0
  105. unicode_logic_kit/eval/equivalence.py +466 -0
  106. unicode_logic_kit/eval/exercise_gen.py +533 -0
  107. unicode_logic_kit/eval/explain.py +791 -0
  108. unicode_logic_kit/eval/generality.py +750 -0
  109. unicode_logic_kit/eval/metric_hf.py +458 -0
  110. unicode_logic_kit/eval/predicate_match.py +343 -0
  111. unicode_logic_kit/eval/theory_check.py +1170 -0
  112. unicode_logic_kit/eval/validate.py +306 -0
  113. unicode_logic_kit/fol/__init__.py +177 -0
  114. unicode_logic_kit/fol/_atom_keys.py +510 -0
  115. unicode_logic_kit/fol/_fol_nodes.py +3586 -0
  116. unicode_logic_kit/fol/_free_parameters.py +105 -0
  117. unicode_logic_kit/fol/_ho_nodes.py +448 -0
  118. unicode_logic_kit/fol/_hybrid_nodes.py +308 -0
  119. unicode_logic_kit/fol/_identifiers.py +1091 -0
  120. unicode_logic_kit/fol/_lambek_nodes.py +112 -0
  121. unicode_logic_kit/fol/_linear_nodes.py +352 -0
  122. unicode_logic_kit/fol/_modal_nodes.py +1467 -0
  123. unicode_logic_kit/fol/_msfl_nodes.py +2196 -0
  124. unicode_logic_kit/fol/_numeral_symbols.py +231 -0
  125. unicode_logic_kit/fol/_so_nodes.py +200 -0
  126. unicode_logic_kit/fol/_symbol_names.py +81 -0
  127. unicode_logic_kit/fol/_team_nodes.py +181 -0
  128. unicode_logic_kit/fol/_tptp_symbols.py +551 -0
  129. unicode_logic_kit/fol/_truth_constants.py +117 -0
  130. unicode_logic_kit/fol/casl_export.py +1135 -0
  131. unicode_logic_kit/fol/casl_import.py +929 -0
  132. unicode_logic_kit/fol/derivation.py +367 -0
  133. unicode_logic_kit/fol/dialect_detect.py +70 -0
  134. unicode_logic_kit/fol/dialect_repair.py +537 -0
  135. unicode_logic_kit/fol/frames.py +637 -0
  136. unicode_logic_kit/fol/grammars/terminals.lark +31 -0
  137. unicode_logic_kit/fol/lambda_tools.py +297 -0
  138. unicode_logic_kit/fol/latex_input.py +429 -0
  139. unicode_logic_kit/fol/modal_translation.py +944 -0
  140. unicode_logic_kit/fol/msflparser.py +1033 -0
  141. unicode_logic_kit/fol/naming.py +422 -0
  142. unicode_logic_kit/fol/nodes.py +241 -0
  143. unicode_logic_kit/fol/normalforms.py +492 -0
  144. unicode_logic_kit/fol/pal.py +287 -0
  145. unicode_logic_kit/fol/prolog_export.py +566 -0
  146. unicode_logic_kit/fol/prolog_input.py +505 -0
  147. unicode_logic_kit/fol/prover9_input.py +1325 -0
  148. unicode_logic_kit/fol/qml.py +1760 -0
  149. unicode_logic_kit/fol/qmltp_input.py +525 -0
  150. unicode_logic_kit/fol/sanitize.py +221 -0
  151. unicode_logic_kit/fol/serialize.py +79 -0
  152. unicode_logic_kit/fol/signature.py +1290 -0
  153. unicode_logic_kit/fol/simplify_check.py +544 -0
  154. unicode_logic_kit/fol/spans.py +594 -0
  155. unicode_logic_kit/fol/tptp_input.py +1503 -0
  156. unicode_logic_kit/fol/tptp_repair.py +941 -0
  157. unicode_logic_kit/fol/unification.py +157 -0
  158. unicode_logic_kit/fol/verbalize.py +263 -0
  159. unicode_logic_kit/hets/__init__.py +163 -0
  160. unicode_logic_kit/hets/bridge.py +142 -0
  161. unicode_logic_kit/hets/client.py +748 -0
  162. unicode_logic_kit/hets/docker.py +420 -0
  163. unicode_logic_kit/hets/dol.py +712 -0
  164. unicode_logic_kit/hets/haskell_json.py +355 -0
  165. unicode_logic_kit/hets/owl_backend.py +794 -0
  166. unicode_logic_kit/hets/owl_cli.py +598 -0
  167. unicode_logic_kit/hets/symbols.py +512 -0
  168. unicode_logic_kit/hol/__init__.py +140 -0
  169. unicode_logic_kit/hol/_ho_common.py +323 -0
  170. unicode_logic_kit/hol/_isabelle_binders.py +125 -0
  171. unicode_logic_kit/hol/classical.py +812 -0
  172. unicode_logic_kit/hol/deepshallow/__init__.py +45 -0
  173. unicode_logic_kit/hol/deepshallow/_common.py +177 -0
  174. unicode_logic_kit/hol/deepshallow/conditional.py +225 -0
  175. unicode_logic_kit/hol/deepshallow/intuitionistic.py +181 -0
  176. unicode_logic_kit/hol/deepshallow/modal.py +217 -0
  177. unicode_logic_kit/hol/deepshallow/qml.py +406 -0
  178. unicode_logic_kit/hol/deepshallow/relevant.py +206 -0
  179. unicode_logic_kit/hol/free.py +753 -0
  180. unicode_logic_kit/hol/goedel.py +336 -0
  181. unicode_logic_kit/hol/ho_modal.py +1743 -0
  182. unicode_logic_kit/hol/intuitionistic.py +403 -0
  183. unicode_logic_kit/hol/isabelle_conditional.py +593 -0
  184. unicode_logic_kit/hol/isabelle_modal.py +1908 -0
  185. unicode_logic_kit/hol/isabelle_relevant.py +412 -0
  186. unicode_logic_kit/hol/isabelle_runner.py +1147 -0
  187. unicode_logic_kit/hol/isabelle_substructural.py +884 -0
  188. unicode_logic_kit/hol/lean.py +1018 -0
  189. unicode_logic_kit/hol/manyvalued.py +921 -0
  190. unicode_logic_kit/hol/secondorder.py +687 -0
  191. unicode_logic_kit/hol/thf_modal.py +941 -0
  192. unicode_logic_kit/hol/thirdorder.py +397 -0
  193. unicode_logic_kit/ilp/__init__.py +89 -0
  194. unicode_logic_kit/ilp/readback.py +389 -0
  195. unicode_logic_kit/ilp/separation.py +153 -0
  196. unicode_logic_kit/ilp/task.py +730 -0
  197. unicode_logic_kit/logic.py +163 -0
  198. unicode_logic_kit/mcp/__init__.py +28 -0
  199. unicode_logic_kit/mcp/__main__.py +5 -0
  200. unicode_logic_kit/mcp/chem_tools.py +1031 -0
  201. unicode_logic_kit/mcp/server.py +2453 -0
  202. unicode_logic_kit/mcp/syntax_spec.py +681 -0
  203. unicode_logic_kit/prob/__init__.py +53 -0
  204. unicode_logic_kit/prob/_bdd.py +225 -0
  205. unicode_logic_kit/prob/_column_gen.py +668 -0
  206. unicode_logic_kit/prob/distribution.py +686 -0
  207. unicode_logic_kit/prob/nilsson.py +470 -0
  208. unicode_logic_kit/py.typed +0 -0
  209. unicode_logic_kit/semantics/__init__.py +137 -0
  210. unicode_logic_kit/semantics/_modal_reject.py +156 -0
  211. unicode_logic_kit/semantics/action_models.py +466 -0
  212. unicode_logic_kit/semantics/asp_models.py +1200 -0
  213. unicode_logic_kit/semantics/conditional.py +580 -0
  214. unicode_logic_kit/semantics/dynamic_epistemic.py +95 -0
  215. unicode_logic_kit/semantics/free_logic.py +913 -0
  216. unicode_logic_kit/semantics/fuzzy.py +384 -0
  217. unicode_logic_kit/semantics/fuzzy_kripke.py +442 -0
  218. unicode_logic_kit/semantics/intuitionistic.py +581 -0
  219. unicode_logic_kit/semantics/kripke.py +1139 -0
  220. unicode_logic_kit/semantics/manyvalued.py +580 -0
  221. unicode_logic_kit/semantics/matrix.py +342 -0
  222. unicode_logic_kit/semantics/model_eval.py +1135 -0
  223. unicode_logic_kit/semantics/modelfinder.py +1036 -0
  224. unicode_logic_kit/semantics/nonmonotonic.py +372 -0
  225. unicode_logic_kit/semantics/relevant.py +331 -0
  226. unicode_logic_kit/semantics/secondorder.py +657 -0
  227. unicode_logic_kit/semantics/structures.py +352 -0
  228. unicode_logic_kit/semantics/tarski.py +975 -0
  229. unicode_logic_kit/semantics/team.py +315 -0
  230. unicode_logic_kit/semantics/team_translation.py +416 -0
  231. unicode_logic_kit/semantics/thirdorder.py +358 -0
  232. unicode_logic_kit/semantics/tnorm.py +85 -0
  233. unicode_logic_kit/semantics/truthtable.py +201 -0
  234. unicode_logic_kit-0.31.0.dist-info/METADATA +333 -0
  235. unicode_logic_kit-0.31.0.dist-info/RECORD +237 -0
  236. unicode_logic_kit-0.31.0.dist-info/WHEEL +4 -0
  237. unicode_logic_kit-0.31.0.dist-info/licenses/LICENSE +21 -0
@@ -0,0 +1,594 @@
1
+ """Adapter for P-FOLIO (Han, Simeng, et al., "P-FOLIO: Evaluating and
2
+ Improving Logical Reasoning with Abundant Human-Written Reasoning Chains",
3
+ Findings of the Association for Computational Linguistics: EMNLP 2024) —
4
+ local CSV files only, no network access.
5
+
6
+ Source and verified schema
7
+ ---------------------------
8
+ Canonical source: https://huggingface.co/datasets/yale-nlp/P-FOLIO (access
9
+ gated — a Hugging Face login is required to download it; this loader never
10
+ downloads anything itself). P-FOLIO extends FOLIO (Han et al., "FOLIO:
11
+ Natural Language Reasoning with First-Order Logic", arXiv:2209.00840; see
12
+ also :mod:`~unicode_logic_kit.eval.datasets.folio`, which reads a DIFFERENT,
13
+ JSONL, distribution of FOLIO — the ``FOLIO.csv`` this module joins against
14
+ is P-FOLIO's OWN bundled spreadsheet export of the same underlying stories,
15
+ not that JSONL file, and the two do not share a row format) with a
16
+ human-written, step-by-step derivation for every (story, conclusion) pair.
17
+ The repository ships the dataset as two CSV files, and this adapter's own
18
+ schema assumptions below were checked directly against a real download
19
+ (one file each, 2026-09) — not assumed from the paper or the HF card:
20
+
21
+ ``P-FOLIO.csv`` (16071 data rows, columns ``story_id``, ``Truth Value``,
22
+ ``Premises used``, ``Derivation``, ``Derivation - Corrected``,
23
+ ``Derivation index``, ``Inference rule``) is a spreadsheet export where **a
24
+ row with a non-blank, digit-only ``story_id`` opens one conclusion's
25
+ block**: its ``Truth Value`` is that conclusion's gold label, and every
26
+ following row with a blank ``story_id`` is one derivation step of that
27
+ block (``Derivation index`` ``D1``, ``D2``, ... referencing earlier steps or
28
+ ``Premises used`` indices into the story's premise list). A story with
29
+ several conclusions has several consecutive blocks; **the block's position
30
+ among its story's blocks, counted in file order starting at 0, is the only
31
+ thing that says WHICH conclusion it is** — the file carries no conclusion
32
+ id of its own. Many rows are blank padding between blocks.
33
+
34
+ ``FOLIO.csv`` (487 data rows, columns ``''``, ``Premises - NL``,
35
+ ``Conclusions - NL``, ``Truth Values``, ``Premises - FOL``,
36
+ ``Conclusions - FOL``, ``Comments``, ``Verified by Prover``) has one row per
37
+ STORY: its unnamed first column is the story id (verified: every id ``0``
38
+ … ``486`` present exactly once, matching ``P-FOLIO.csv``'s ``story_id``
39
+ range exactly), and ``Premises - NL``/``Conclusions - NL``/
40
+ ``Truth Values``/``Premises - FOL``/``Conclusions - FOL`` each hold a
41
+ NEWLINE-separated list — one entry per premise (first two columns) or per
42
+ conclusion (last three), in the SAME order the P-FOLIO blocks for that
43
+ story appear in.
44
+
45
+ This adapter's join, and what it refuses
46
+ ------------------------------------------
47
+ The two files carry no shared conclusion id, so — exactly as this module's
48
+ build spec requires — a P-FOLIO block's ``nl_conclusion``/``fol_conclusion``
49
+ are resolved by joining ``FOLIO.csv`` on **story id AND the block's
50
+ position among its story's blocks**, and every join is CROSS-CHECKED: the
51
+ block's own ``Truth Value`` must equal ``FOLIO.csv``'s truth value at that
52
+ same position, or the block is refused rather than trusted. This was
53
+ verified by hand against several real stories before being encoded as a
54
+ blanket check (see ``tests/test_datasets_pfolio.py`` for the same
55
+ cross-check run against the small fixture below): story 5's two ``T``
56
+ blocks' last derivation steps read, verbatim, "If the Hulk does not wake up,
57
+ then Thor is not happy." and "If Thor is happy, then Peter Parker wears a
58
+ uniform" — exactly ``FOLIO.csv`` story 5's first two conclusions, in order,
59
+ both labelled ``T`` on both sides.
60
+
61
+ Of the real, once-downloaded ``P-FOLIO.csv``'s 1431 blocks (2026-09), **1420
62
+ load cleanly** through :func:`load_pfolio` and **11 are refused** — a 99.2%
63
+ yield, all 11 individually accounted for below rather than swallowed into an
64
+ aggregate. Re-measured directly through the loader by
65
+ ``test_the_real_files_join_as_measured`` in ``tests/test_datasets_pfolio.py``
66
+ (opt-in, since the source files are access-gated — see below).
67
+
68
+ A block is REFUSED (excluded from :func:`load_pfolio`'s default strict
69
+ pass, listed with a reason by :func:`pfolio_refusals`) — never guessed —
70
+ for any of:
71
+
72
+ * ``"unparseable_truth_value"`` — the block's ``Truth Value`` cell, after
73
+ stripping whitespace, is not exactly ``"T"``/``"F"``/``"U"``. The real
74
+ file's two worst offenders are reviewer notes left IN the truth-value
75
+ cell instead of a real ``T``/``F``/``U``, e.g. ``"F -> should be U?\\n
76
+ rui_comment: F is correct."`` — guessing which of the two conflicting
77
+ reviewers is right is exactly the guessing this adapter declines to do.
78
+ A bare trailing newline (``"T\\n"``, common throughout the real file) is
79
+ NOT an unparseable value — it is stripped, not refused.
80
+ * ``"folio_story_ambiguous"`` — that story's ``FOLIO.csv`` row cannot be
81
+ read unambiguously in the first place (see :func:`_read_folio`: its
82
+ conclusions/truth-values/conclusion-FOL lists disagree in length, or its
83
+ premises/premises-FOL lists do, even after trimming wholly-blank leading
84
+ or trailing list entries — a known export artifact, see below). Verified
85
+ in the real file: of 487 stories, only story 249 is genuinely ambiguous
86
+ this way (its ``Conclusions - NL`` column holds 6 lines that read like
87
+ PREMISES, not conclusions, while ``Truth Values``/``Conclusions - FOL``
88
+ hold only 2 — a real upstream data defect, not a formatting artifact).
89
+ * ``"story_not_in_folio"`` — the block's ``story_id`` has no row in
90
+ ``FOLIO.csv`` at all (defensive; does not occur in the verified real
91
+ files, where the id ranges match exactly, but a caller-supplied fixture
92
+ or a future release could still have this).
93
+ * ``"position_out_of_range"`` — the block's position exceeds how many
94
+ conclusions that story's ``FOLIO.csv`` row actually has. Verified once in
95
+ the real file: story 135 has three P-FOLIO blocks but only two resolved
96
+ FOLIO conclusions.
97
+ * ``"truth_value_mismatch"`` — the block's own truth value parses fine and
98
+ the position resolves, but disagrees with ``FOLIO.csv``'s truth value at
99
+ that position. Verified five times in the real file (stories 34, 50
100
+ twice, 258, 409) — genuine annotation disagreements between the two
101
+ files, not something this adapter is in a position to adjudicate.
102
+
103
+ Three further export artifacts, all handled WITHOUT guessing at their
104
+ content, are worth naming explicitly because they could otherwise look like
105
+ corruption:
106
+
107
+ * A handful of rows in ``P-FOLIO.csv`` have a non-blank, non-digit
108
+ ``story_id`` (e.g. a stray ``"("`` character, or a whole comment row like
109
+ ``"Need to change xor"`` sitting alone between blank padding rows). Only
110
+ a BLANK or DIGIT-ONLY ``story_id`` is read as starting a new block; any
111
+ other value is never interpreted as an id OR as a truth value — the row
112
+ is treated exactly like a blank-``story_id`` row, i.e. as one more
113
+ derivation-step candidate of whichever block is currently open (real
114
+ content columns intact) or as ignorable padding (nothing else on the
115
+ row). This is the same "only the recognised column means anything on
116
+ this row" treatment already used for a stray reviewer comment landing in
117
+ a derivation row's ``Truth Value`` cell (see the fixture and
118
+ ``tests/test_datasets_pfolio.py`` for both real-shaped cases).
119
+ * Several ``FOLIO.csv`` cells hold one extra wholly-blank leading or
120
+ trailing line inside an otherwise-consistent newline-separated list
121
+ (verified: stories 55, 113, 135). :func:`_split_field` trims ONLY
122
+ wholly-blank entries at the very start/end of such a list — never a
123
+ blank in the middle, and never anything from a non-blank entry — before
124
+ the length cross-check above runs, so these three stories load normally
125
+ rather than being flagged ``folio_story_ambiguous``.
126
+ * A block's FIRST derivation step's own content (``Premises used``/
127
+ ``Derivation``/``Derivation - Corrected``/``Derivation index``/
128
+ ``Inference rule``) sometimes sits on the SAME row as the block's own
129
+ ``story_id``/``Truth Value`` header, rather than on a following
130
+ blank-``story_id`` row (verified: real story 246's ``D1``, and 164 of the
131
+ real file's 1431 blocks overall, 33 of them with that step as their ONLY
132
+ one — indistinguishable from a genuinely derivation-less block unless
133
+ this row is also inspected). :func:`_iter_pfolio_blocks` checks a
134
+ header row's own columns 2–6 the same way it checks any other row's, so
135
+ that first step is kept, not silently dropped.
136
+
137
+ ``Derivation`` vs. ``Derivation - Corrected`` are two DIFFERENT upstream
138
+ columns (the corrected text is only sometimes present, and sometimes reads
139
+ quite differently from the original — both were observed verbatim in the
140
+ real file) and this adapter keeps them as two separate keys in every
141
+ ``meta["proof_steps"]`` entry, never merged into one. ``Premises used`` is
142
+ likewise kept VERBATIM as a single string rather than parsed into a list of
143
+ indices: real values mix plain premise numbers, ``D``-prefixed references
144
+ to earlier steps, comma- and even full-width-comma (``,``)-separated lists
145
+ (``"3,5"``) in the same column — parsing that into a clean structure would
146
+ be exactly the kind of guess this adapter's build spec forbids.
147
+
148
+ Truth-value vocabulary — a real, documented difference from
149
+ :mod:`~unicode_logic_kit.eval.datasets.folio`
150
+ ------------------------------------------------------------------------
151
+ ``FOLIO.csv``'s ``Truth Values`` column (and P-FOLIO's ``Truth Value``
152
+ column) use the single letters ``"T"``/``"F"``/``"U"`` (:data:`PFOLIO_TRUTH_VALUES`)
153
+ — verified directly, not the ``"True"``/``"False"``/``"Uncertain"`` words
154
+ the OTHER FOLIO adapter's JSONL source uses. ``label`` here is that letter,
155
+ UNCHANGED — this loader does not translate between the two vocabularies
156
+ (that would be presenting an invented value as gold data).
157
+
158
+ License: **MIT**, per the yale-nlp/P-FOLIO Hugging Face dataset card.
159
+
160
+ This module never downloads anything — obtain both CSV files yourself from
161
+ https://huggingface.co/datasets/yale-nlp/P-FOLIO (a Hugging Face account and
162
+ accepted access request are required) and pass their local paths to
163
+ :func:`load_pfolio`.
164
+ """
165
+
166
+ import csv
167
+ from pathlib import Path
168
+ from typing import Dict, FrozenSet, Iterator, List, NamedTuple, Optional, Tuple, Union
169
+
170
+ from ._base import DatasetExample, _register_dataset_info
171
+
172
+ __all__ = ["load_pfolio", "pfolio_refusals", "PFOLIO_TRUTH_VALUES",
173
+ "PFOLIO_REFUSAL_REASONS"]
174
+
175
+ #: The verified vocabulary of P-FOLIO's/FOLIO.csv's own truth-value tokens
176
+ #: (single letters — see the module docstring for how this differs from
177
+ #: :mod:`~unicode_logic_kit.eval.datasets.folio`'s word-form labels).
178
+ PFOLIO_TRUTH_VALUES = ("T", "F", "U")
179
+
180
+ #: The reasons :func:`load_pfolio`/:func:`pfolio_refusals` can report for a
181
+ #: block that was NOT turned into a :class:`~unicode_logic_kit.eval.datasets.DatasetExample`
182
+ #: — see the module docstring's "This adapter's join, and what it refuses"
183
+ #: section for what each one means and how often it fires on the real file.
184
+ PFOLIO_REFUSAL_REASONS = (
185
+ "unparseable_truth_value", "folio_story_ambiguous", "story_not_in_folio",
186
+ "position_out_of_range", "truth_value_mismatch",
187
+ )
188
+
189
+ _register_dataset_info(
190
+ "pfolio",
191
+ license="MIT, per the yale-nlp/P-FOLIO Hugging Face dataset card.",
192
+ source_url="https://huggingface.co/datasets/yale-nlp/P-FOLIO",
193
+ citation_hint=(
194
+ "Han, Simeng, et al. \"P-FOLIO: Evaluating and Improving Logical "
195
+ "Reasoning with Abundant Human-Written Reasoning Chains.\" Findings "
196
+ "of the Association for Computational Linguistics: EMNLP 2024, "
197
+ "pages 16553-16565. arXiv:2410.09207."
198
+ ),
199
+ )
200
+
201
+
202
+ # ---------------------------------------------------------------------------
203
+ # FOLIO.csv — one row per story
204
+ # ---------------------------------------------------------------------------
205
+
206
+ _FOLIO_HEADER = (
207
+ "", "Premises - NL", "Conclusions - NL", "Truth Values",
208
+ "Premises - FOL", "Conclusions - FOL", "Comments", "Verified by Prover",
209
+ )
210
+
211
+
212
+ class _FolioStory(NamedTuple):
213
+ premises_nl: Tuple[str, ...]
214
+ premises_fol: Tuple[str, ...]
215
+ conclusions_nl: Tuple[str, ...]
216
+ conclusions_fol: Tuple[str, ...]
217
+ truth_values: Tuple[str, ...]
218
+ comments: Optional[str]
219
+ verified_by_prover: Optional[str]
220
+
221
+
222
+ def _lf_cells(row: List[str]) -> List[str]:
223
+ """``row`` with every line break inside a cell rewritten to ``\\n``.
224
+
225
+ ``csv`` (read with ``newline=""``, as it must be) hands a quoted cell's
226
+ embedded line breaks through VERBATIM. The real files break lines inside
227
+ cells with a bare ``\\n``, but a copy that went through a CRLF-converting
228
+ tool (a git checkout with ``core.autocrlf``, a Windows editor) carries
229
+ ``\\r\\n`` there, and splitting that on ``\\n`` leaves a ``\\r`` glued
230
+ to every premise and conclusion. A line break inside a cell means the
231
+ same thing in either convention, so both files read identically."""
232
+ return [cell.replace("\r\n", "\n").replace("\r", "\n") for cell in row]
233
+
234
+
235
+ def _split_field(cell: str) -> List[str]:
236
+ """Newline-split ``cell``, trimming only WHOLLY-BLANK leading/trailing
237
+ entries (a verified export artifact — see the module docstring). A
238
+ blank entry anywhere else in the list, or any whitespace inside a
239
+ non-blank entry, is left untouched."""
240
+ lines = cell.split("\n")
241
+ while lines and lines[0].strip() == "":
242
+ lines.pop(0)
243
+ while lines and lines[-1].strip() == "":
244
+ lines.pop()
245
+ return lines
246
+
247
+
248
+ def _read_folio(path: Union[str, Path]) -> Tuple[Dict[int, _FolioStory], Dict[int, str]]:
249
+ """Read ``FOLIO.csv`` into per-story records.
250
+
251
+ Returns ``(stories, ambiguous)``: ``stories`` maps a story id to a
252
+ :class:`_FolioStory` for every story whose premise/conclusion lists are
253
+ internally consistent; ``ambiguous`` maps every OTHER story id to a
254
+ human-readable reason it could not be read unambiguously (see the
255
+ module docstring's ``"folio_story_ambiguous"`` entry) — such a story is
256
+ never silently dropped, only excluded from ``stories``.
257
+
258
+ Raises:
259
+ ValueError: the header does not match the verified ``FOLIO.csv``
260
+ column layout, a row's own id does not parse as a non-negative
261
+ integer, or a story id repeats — this is STRUCTURAL corruption
262
+ of the reference file itself, refused unconditionally (unlike
263
+ the per-story ambiguity above, there is no safe partial reading
264
+ of a file whose id scheme is broken).
265
+ """
266
+ path = Path(path)
267
+ stories: Dict[int, _FolioStory] = {}
268
+ ambiguous: Dict[int, str] = {}
269
+ with path.open("r", encoding="utf-8-sig", newline="") as fh:
270
+ reader = csv.reader(fh)
271
+ header = tuple(next(reader))
272
+ if header != _FOLIO_HEADER:
273
+ raise ValueError(
274
+ f"pfolio: {path} has header {header!r}, expected "
275
+ f"{_FOLIO_HEADER!r} — is this FOLIO.csv?")
276
+
277
+ seen: Dict[int, int] = {}
278
+ for row_no, row in enumerate(reader):
279
+ if len(row) != len(_FOLIO_HEADER):
280
+ raise ValueError(
281
+ f"pfolio: {path} row {row_no} has {len(row)} fields, "
282
+ f"expected {len(_FOLIO_HEADER)}")
283
+ row = _lf_cells(row)
284
+ raw_id = row[0].strip()
285
+ if not raw_id.isdigit():
286
+ raise ValueError(
287
+ f"pfolio: {path} row {row_no} has story id {row[0]!r}, "
288
+ "not a non-negative integer — is this FOLIO.csv?")
289
+ story_id = int(raw_id)
290
+ if story_id in seen:
291
+ raise ValueError(
292
+ f"pfolio: {path} has story id {story_id} at both rows "
293
+ f"{seen[story_id]} and {row_no} — ids must be unique")
294
+ seen[story_id] = row_no
295
+
296
+ premises_nl = _split_field(row[1])
297
+ conclusions_nl = _split_field(row[2])
298
+ truth_values = _split_field(row[3])
299
+ premises_fol = _split_field(row[4])
300
+ conclusions_fol = _split_field(row[5])
301
+ comments = row[6].strip() or None
302
+ verified_by_prover = row[7].strip() or None
303
+
304
+ if len(premises_nl) != len(premises_fol):
305
+ ambiguous[story_id] = (
306
+ f"premises-NL/premises-FOL length mismatch "
307
+ f"({len(premises_nl)} vs {len(premises_fol)})")
308
+ continue
309
+ if not (len(conclusions_nl) == len(truth_values) == len(conclusions_fol)):
310
+ ambiguous[story_id] = (
311
+ "conclusions-NL/truth-values/conclusions-FOL length "
312
+ f"mismatch ({len(conclusions_nl)}, {len(truth_values)}, "
313
+ f"{len(conclusions_fol)})")
314
+ continue
315
+ bad_tv = [t for t in truth_values if t.strip() not in PFOLIO_TRUTH_VALUES]
316
+ if bad_tv:
317
+ ambiguous[story_id] = f"unparseable truth value(s) {bad_tv!r}"
318
+ continue
319
+
320
+ stories[story_id] = _FolioStory(
321
+ premises_nl=tuple(premises_nl),
322
+ premises_fol=tuple(premises_fol),
323
+ conclusions_nl=tuple(conclusions_nl),
324
+ conclusions_fol=tuple(conclusions_fol),
325
+ truth_values=tuple(t.strip() for t in truth_values),
326
+ comments=comments,
327
+ verified_by_prover=verified_by_prover,
328
+ )
329
+ return stories, ambiguous
330
+
331
+
332
+ # ---------------------------------------------------------------------------
333
+ # P-FOLIO.csv — several derivation-step rows per block, several blocks per story
334
+ # ---------------------------------------------------------------------------
335
+
336
+ _PFOLIO_HEADER = (
337
+ "story_id", "Truth Value", "Premises used", "Derivation",
338
+ "Derivation - Corrected", "Derivation index", "Inference rule",
339
+ )
340
+
341
+
342
+ def _none_if_blank(text: str) -> Optional[str]:
343
+ stripped = text.strip()
344
+ return stripped if stripped else None
345
+
346
+
347
+ class _RawBlock(NamedTuple):
348
+ story_id: int
349
+ truth_raw: str
350
+ steps: Tuple[dict, ...]
351
+ header_row_no: int
352
+
353
+
354
+ def _iter_pfolio_blocks(path: Union[str, Path]) -> Iterator[_RawBlock]:
355
+ """Group ``P-FOLIO.csv``'s rows into blocks, in file order.
356
+
357
+ A row STARTS a new block iff its ``story_id`` cell, stripped, is
358
+ non-empty and digit-only; every other row either extends the currently
359
+ open block (if it carries any real step content — see the module
360
+ docstring for why a stray non-digit ``story_id`` or a stray comment in
361
+ the ``Truth Value`` column of such a row is never interpreted) or is
362
+ ignored as padding.
363
+
364
+ Raises:
365
+ ValueError: the header does not match the verified ``P-FOLIO.csv``
366
+ column layout, or a row has the wrong field count — structural
367
+ corruption of the file itself.
368
+ """
369
+ path = Path(path)
370
+ current: Optional[dict] = None
371
+ with path.open("r", encoding="utf-8-sig", newline="") as fh:
372
+ reader = csv.reader(fh)
373
+ header = tuple(next(reader))
374
+ if header != _PFOLIO_HEADER:
375
+ raise ValueError(
376
+ f"pfolio: {path} has header {header!r}, expected "
377
+ f"{_PFOLIO_HEADER!r} — is this P-FOLIO.csv?")
378
+
379
+ for row_no, row in enumerate(reader):
380
+ if len(row) != len(_PFOLIO_HEADER):
381
+ raise ValueError(
382
+ f"pfolio: {path} row {row_no} has {len(row)} fields, "
383
+ f"expected {len(_PFOLIO_HEADER)}")
384
+ row = _lf_cells(row)
385
+ story_id_raw = row[0].strip()
386
+ if story_id_raw.isdigit():
387
+ if current is not None:
388
+ yield _RawBlock(current["story_id"], current["truth_raw"],
389
+ tuple(current["steps"]), current["header_row_no"])
390
+ current = {"story_id": int(story_id_raw), "truth_raw": row[1],
391
+ "steps": [], "header_row_no": row_no}
392
+ if any(cell.strip() for cell in row[2:7]):
393
+ # A real export artifact: the block's FIRST derivation
394
+ # step's content sometimes sits on the very same row as
395
+ # the story_id/Truth Value header, rather than on a
396
+ # following blank-story_id row (verified: real story
397
+ # 246's D1). Not capturing it here would silently drop
398
+ # that step — see the module docstring.
399
+ current["steps"].append({
400
+ "premises_used": _none_if_blank(row[2]),
401
+ "derivation": _none_if_blank(row[3]),
402
+ "derivation_corrected": _none_if_blank(row[4]),
403
+ "step_id": _none_if_blank(row[5]),
404
+ "inference_rule": _none_if_blank(row[6]),
405
+ "row_no": row_no,
406
+ })
407
+ continue
408
+ if current is None:
409
+ continue # padding before any block
410
+ if any(cell.strip() for cell in row[2:7]):
411
+ current["steps"].append({
412
+ "premises_used": _none_if_blank(row[2]),
413
+ "derivation": _none_if_blank(row[3]),
414
+ "derivation_corrected": _none_if_blank(row[4]),
415
+ "step_id": _none_if_blank(row[5]),
416
+ "inference_rule": _none_if_blank(row[6]),
417
+ "row_no": row_no,
418
+ })
419
+ if current is not None:
420
+ yield _RawBlock(current["story_id"], current["truth_raw"],
421
+ tuple(current["steps"]), current["header_row_no"])
422
+
423
+
424
+ def _parse_truth_value(raw: str) -> Optional[str]:
425
+ stripped = raw.strip()
426
+ return stripped if stripped in PFOLIO_TRUTH_VALUES else None
427
+
428
+
429
+ # ---------------------------------------------------------------------------
430
+ # The join
431
+ # ---------------------------------------------------------------------------
432
+
433
+ def _iter_joined(pfolio_path: Union[str, Path], folio_path: Union[str, Path],
434
+ known_bad_ids: FrozenSet[str]) -> Iterator[Tuple[str, object]]:
435
+ """Yield ``("ok", DatasetExample)`` or ``("refused", dict)`` per P-FOLIO
436
+ block, in file order — the shared engine behind :func:`load_pfolio` and
437
+ :func:`pfolio_refusals`, so the two never disagree about what loads.
438
+
439
+ A refusal dict is ``{"story_id", "position", "reason" (one of
440
+ :data:`PFOLIO_REFUSAL_REASONS`), "detail", "row_no"}`` — ``row_no`` is
441
+ the block's own header row's 0-based data-row number in ``P-FOLIO.csv``,
442
+ for tracing a refusal back to the source file.
443
+ """
444
+ stories, ambiguous = _read_folio(folio_path)
445
+ positions: Dict[int, int] = {}
446
+
447
+ for block in _iter_pfolio_blocks(pfolio_path):
448
+ story_id = block.story_id
449
+ position = positions.get(story_id, 0)
450
+ positions[story_id] = position + 1
451
+
452
+ def _refused(reason: str, detail: Optional[str]) -> Tuple[str, dict]:
453
+ return ("refused", {"story_id": story_id, "position": position,
454
+ "reason": reason, "detail": detail,
455
+ "row_no": block.header_row_no})
456
+
457
+ truth_value = _parse_truth_value(block.truth_raw)
458
+ if truth_value is None:
459
+ yield _refused("unparseable_truth_value", block.truth_raw)
460
+ continue
461
+ if story_id in ambiguous:
462
+ yield _refused("folio_story_ambiguous", ambiguous[story_id])
463
+ continue
464
+ story = stories.get(story_id)
465
+ if story is None:
466
+ yield _refused("story_not_in_folio", None)
467
+ continue
468
+ if position >= len(story.conclusions_nl):
469
+ yield _refused(
470
+ "position_out_of_range",
471
+ f"story {story_id} has only {len(story.conclusions_nl)} "
472
+ "FOLIO.csv conclusion(s)")
473
+ continue
474
+ folio_truth_value = story.truth_values[position]
475
+ if folio_truth_value != truth_value:
476
+ yield _refused(
477
+ "truth_value_mismatch",
478
+ f"P-FOLIO.csv={truth_value!r} FOLIO.csv={folio_truth_value!r}")
479
+ continue
480
+
481
+ example_id = f"pfolio:{story_id}:{position}"
482
+ proof_steps = [
483
+ {
484
+ "step_id": step["step_id"],
485
+ "premises_used": step["premises_used"],
486
+ "derivation": step["derivation"],
487
+ "derivation_corrected": step["derivation_corrected"],
488
+ "inference_rule": step["inference_rule"],
489
+ }
490
+ for step in block.steps
491
+ ]
492
+ meta = {
493
+ "story_id": story_id,
494
+ "position": position,
495
+ "folio_comments": story.comments,
496
+ "folio_verified_by_prover": story.verified_by_prover,
497
+ "proof_steps": proof_steps,
498
+ }
499
+ yield ("ok", DatasetExample(
500
+ id=example_id,
501
+ nl_premises=story.premises_nl,
502
+ fol_premises=story.premises_fol,
503
+ nl_conclusion=story.conclusions_nl[position],
504
+ fol_conclusion=story.conclusions_fol[position],
505
+ label=folio_truth_value,
506
+ known_bad=example_id in known_bad_ids,
507
+ meta=meta,
508
+ ))
509
+
510
+
511
+ def load_pfolio(pfolio_path: Union[str, Path], folio_path: Union[str, Path], *,
512
+ known_bad_ids: FrozenSet[str] = frozenset(),
513
+ on_refused: str = "raise") -> Iterator[DatasetExample]:
514
+ """Stream :class:`~unicode_logic_kit.eval.datasets.DatasetExample` from a
515
+ local pair of P-FOLIO/FOLIO CSV files, joined per the module docstring.
516
+
517
+ Field mapping: ``nl_premises``/``fol_premises`` = the joined story's
518
+ ``FOLIO.csv`` premises (repeated across every conclusion of that story,
519
+ the same "several consecutive examples share identical premises"
520
+ situation :mod:`~unicode_logic_kit.eval.datasets.folio`'s docstring
521
+ documents for its own story grouping); ``nl_conclusion``/
522
+ ``fol_conclusion`` = that story's ``FOLIO.csv`` conclusion/FOL at the
523
+ block's position; ``label`` = the (agreeing) truth value, one of
524
+ :data:`PFOLIO_TRUTH_VALUES`. ``meta`` carries ``story_id``, ``position``
525
+ (both 0-based/int), ``folio_comments``/``folio_verified_by_prover``
526
+ (``FOLIO.csv``'s own free-text columns, kept at STORY granularity —
527
+ verified too sparse and inconsistently shaped to align per-conclusion,
528
+ see the module docstring), and ``proof_steps``: the block's derivation
529
+ rows verbatim, each ``{"step_id", "premises_used", "derivation",
530
+ "derivation_corrected", "inference_rule"}`` (``Derivation`` and
531
+ ``Derivation - Corrected`` kept as two distinct keys, never merged;
532
+ ``premises_used`` kept as one raw string, never parsed into indices —
533
+ see the module docstring for why).
534
+
535
+ Args:
536
+ pfolio_path: local path to ``P-FOLIO.csv``.
537
+ folio_path: local path to P-FOLIO's own bundled ``FOLIO.csv`` (NOT
538
+ the JSONL file :func:`~unicode_logic_kit.eval.datasets.folio.load_folio`
539
+ reads — see the module docstring).
540
+ known_bad_ids: ids (``f"pfolio:{story_id}:{position}"``) whose gold
541
+ annotation is known to be broken. Every yielded example with a
542
+ matching id gets ``known_bad=True``.
543
+ on_refused: what to do with a block this adapter cannot safely join
544
+ (see the module docstring's refusal reasons). ``"raise"``
545
+ (default): raise :class:`ValueError` naming the story, position
546
+ and reason at the first one reached — the loud, no-guessing
547
+ default. ``"skip"``: drop it from the yielded stream instead
548
+ (use :func:`pfolio_refusals` to see what was dropped and why —
549
+ nothing is silently lost either way, only excluded from this
550
+ call's own output).
551
+
552
+ Yields:
553
+ One :class:`~unicode_logic_kit.eval.datasets.DatasetExample` per
554
+ successfully-joined P-FOLIO block, in file order.
555
+
556
+ Raises:
557
+ ValueError: ``on_refused`` is not ``"raise"``/``"skip"``; either CSV
558
+ file has the wrong header or a malformed row (structural
559
+ corruption, refused unconditionally regardless of
560
+ ``on_refused``); or, with ``on_refused="raise"``, the first
561
+ block that cannot be safely joined.
562
+ FileNotFoundError: either path does not exist.
563
+ """
564
+ if on_refused not in ("raise", "skip"):
565
+ raise ValueError(
566
+ f"pfolio: on_refused must be 'raise' or 'skip', got {on_refused!r}")
567
+
568
+ for kind, payload in _iter_joined(pfolio_path, folio_path, known_bad_ids):
569
+ if kind == "ok":
570
+ yield payload
571
+ continue
572
+ if on_refused == "raise":
573
+ raise ValueError(
574
+ f"pfolio: story {payload['story_id']} conclusion #"
575
+ f"{payload['position']} (P-FOLIO.csv row {payload['row_no']}) "
576
+ f"refused ({payload['reason']}): {payload['detail']}")
577
+ # on_refused == "skip": drop it, recoverable via pfolio_refusals().
578
+
579
+
580
+ def pfolio_refusals(pfolio_path: Union[str, Path],
581
+ folio_path: Union[str, Path]) -> List[dict]:
582
+ """Every block :func:`load_pfolio` would refuse, with why — a read-only
583
+ diagnostic pass, never raising for a refusal itself (only for the same
584
+ structural-corruption cases :func:`load_pfolio` always raises for).
585
+
586
+ Returns one ``{"story_id", "position", "reason", "detail", "row_no"}``
587
+ dict per refused block, in file order (see :func:`_iter_joined`'s
588
+ docstring for the field meanings, and the module docstring's "This
589
+ adapter's join, and what it refuses" section for what each ``reason``
590
+ means). Empty iff every block in ``pfolio_path`` joins cleanly.
591
+ """
592
+ return [payload for kind, payload
593
+ in _iter_joined(pfolio_path, folio_path, frozenset())
594
+ if kind == "refused"]