unicode-logic-kit 0.31.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- unicode_logic_kit/__init__.py +385 -0
- unicode_logic_kit/__main__.py +520 -0
- unicode_logic_kit/_deadline.py +219 -0
- unicode_logic_kit/ace/__init__.py +126 -0
- unicode_logic_kit/ace/_align.py +135 -0
- unicode_logic_kit/ace/chem_lexicon.py +128 -0
- unicode_logic_kit/ace/drs_reader.py +570 -0
- unicode_logic_kit/ace/mapping.py +666 -0
- unicode_logic_kit/ace/reverse_modal.py +138 -0
- unicode_logic_kit/ace/runner.py +551 -0
- unicode_logic_kit/ace/translate.py +452 -0
- unicode_logic_kit/ace/verbalize.py +1070 -0
- unicode_logic_kit/api.py +1284 -0
- unicode_logic_kit/atp/__init__.py +177 -0
- unicode_logic_kit/atp/_ascii_names.py +113 -0
- unicode_logic_kit/atp/_html.py +72 -0
- unicode_logic_kit/atp/_substructural_input.py +228 -0
- unicode_logic_kit/atp/_tff_problem.py +715 -0
- unicode_logic_kit/atp/_tptp_problem.py +1111 -0
- unicode_logic_kit/atp/_writer_support.py +289 -0
- unicode_logic_kit/atp/clingo_backend.py +1180 -0
- unicode_logic_kit/atp/cvc5_backend.py +1385 -0
- unicode_logic_kit/atp/eprover_backend.py +732 -0
- unicode_logic_kit/atp/finite_domain.py +1055 -0
- unicode_logic_kit/atp/fitch.py +1547 -0
- unicode_logic_kit/atp/fitch_search.py +551 -0
- unicode_logic_kit/atp/hets_backend.py +339 -0
- unicode_logic_kit/atp/hybrid_down.py +120 -0
- unicode_logic_kit/atp/incremental.py +250 -0
- unicode_logic_kit/atp/kripke_enum.py +741 -0
- unicode_logic_kit/atp/lambek.py +436 -0
- unicode_logic_kit/atp/leo3_backend.py +332 -0
- unicode_logic_kit/atp/linear.py +738 -0
- unicode_logic_kit/atp/lj.py +705 -0
- unicode_logic_kit/atp/logic_backends.py +566 -0
- unicode_logic_kit/atp/ltl_tableau.py +1084 -0
- unicode_logic_kit/atp/minizinc_backend.py +1402 -0
- unicode_logic_kit/atp/modal_tableau.py +1382 -0
- unicode_logic_kit/atp/nanocop_backend.py +410 -0
- unicode_logic_kit/atp/portfolio.py +489 -0
- unicode_logic_kit/atp/protocol.py +1803 -0
- unicode_logic_kit/atp/prover9_entailment.py +1153 -0
- unicode_logic_kit/atp/resolution.py +1376 -0
- unicode_logic_kit/atp/resolution_check.py +1114 -0
- unicode_logic_kit/atp/sequent.py +1050 -0
- unicode_logic_kit/atp/tableau.py +921 -0
- unicode_logic_kit/atp/tableau_check.py +543 -0
- unicode_logic_kit/atp/tptp_ncl.py +811 -0
- unicode_logic_kit/atp/tptp_tff.py +1546 -0
- unicode_logic_kit/atp/tstp.py +1333 -0
- unicode_logic_kit/atp/tstp_check.py +1096 -0
- unicode_logic_kit/atp/twee_backend.py +236 -0
- unicode_logic_kit/atp/twee_check.py +711 -0
- unicode_logic_kit/atp/twee_entailment.py +953 -0
- unicode_logic_kit/atp/vampire_entailment.py +540 -0
- unicode_logic_kit/atp/z3_arith.py +470 -0
- unicode_logic_kit/atp/z3_equivalence.py +36 -0
- unicode_logic_kit/atp/z3_fuzzy.py +362 -0
- unicode_logic_kit/atp/z3_input.py +500 -0
- unicode_logic_kit/atp/z3_models.py +208 -0
- unicode_logic_kit/chem/__init__.py +88 -0
- unicode_logic_kit/chem/_naming.py +284 -0
- unicode_logic_kit/chem/cache.py +185 -0
- unicode_logic_kit/chem/interop.py +244 -0
- unicode_logic_kit/chem/mol.py +525 -0
- unicode_logic_kit/chem/signature.py +112 -0
- unicode_logic_kit/comorphism.py +497 -0
- unicode_logic_kit/dl/__init__.py +384 -0
- unicode_logic_kit/dl/classification.py +227 -0
- unicode_logic_kit/dl/concepts.py +632 -0
- unicode_logic_kit/dl/datatypes.py +818 -0
- unicode_logic_kit/dl/owl_functional.py +2433 -0
- unicode_logic_kit/dl/owl_manchester.py +1637 -0
- unicode_logic_kit/dl/owl_reasoner.py +790 -0
- unicode_logic_kit/dl/parser.py +391 -0
- unicode_logic_kit/dl/tableau.py +4048 -0
- unicode_logic_kit/dl/translate.py +2704 -0
- unicode_logic_kit/drt/__init__.py +94 -0
- unicode_logic_kit/drt/export.py +179 -0
- unicode_logic_kit/drt/nodes.py +506 -0
- unicode_logic_kit/drt/parser.py +965 -0
- unicode_logic_kit/drt/resolve.py +195 -0
- unicode_logic_kit/drt/reverse.py +175 -0
- unicode_logic_kit/eval/__init__.py +106 -0
- unicode_logic_kit/eval/batch.py +382 -0
- unicode_logic_kit/eval/canonical.py +663 -0
- unicode_logic_kit/eval/chem_batch.py +606 -0
- unicode_logic_kit/eval/converses.py +200 -0
- unicode_logic_kit/eval/datasets/__init__.py +136 -0
- unicode_logic_kit/eval/datasets/_base.py +263 -0
- unicode_logic_kit/eval/datasets/_proofwriter_proof.py +422 -0
- unicode_logic_kit/eval/datasets/c3po.py +678 -0
- unicode_logic_kit/eval/datasets/folio.py +158 -0
- unicode_logic_kit/eval/datasets/fracas.py +418 -0
- unicode_logic_kit/eval/datasets/groves.py +191 -0
- unicode_logic_kit/eval/datasets/logicbench.py +467 -0
- unicode_logic_kit/eval/datasets/logicnli.py +303 -0
- unicode_logic_kit/eval/datasets/malls.py +133 -0
- unicode_logic_kit/eval/datasets/pfolio.py +594 -0
- unicode_logic_kit/eval/datasets/pmb.py +242 -0
- unicode_logic_kit/eval/datasets/prontoqa.py +611 -0
- unicode_logic_kit/eval/datasets/proofwriter.py +1431 -0
- unicode_logic_kit/eval/datasets/proverqa.py +674 -0
- unicode_logic_kit/eval/datasets/willow.py +478 -0
- unicode_logic_kit/eval/equivalence.py +466 -0
- unicode_logic_kit/eval/exercise_gen.py +533 -0
- unicode_logic_kit/eval/explain.py +791 -0
- unicode_logic_kit/eval/generality.py +750 -0
- unicode_logic_kit/eval/metric_hf.py +458 -0
- unicode_logic_kit/eval/predicate_match.py +343 -0
- unicode_logic_kit/eval/theory_check.py +1170 -0
- unicode_logic_kit/eval/validate.py +306 -0
- unicode_logic_kit/fol/__init__.py +177 -0
- unicode_logic_kit/fol/_atom_keys.py +510 -0
- unicode_logic_kit/fol/_fol_nodes.py +3586 -0
- unicode_logic_kit/fol/_free_parameters.py +105 -0
- unicode_logic_kit/fol/_ho_nodes.py +448 -0
- unicode_logic_kit/fol/_hybrid_nodes.py +308 -0
- unicode_logic_kit/fol/_identifiers.py +1091 -0
- unicode_logic_kit/fol/_lambek_nodes.py +112 -0
- unicode_logic_kit/fol/_linear_nodes.py +352 -0
- unicode_logic_kit/fol/_modal_nodes.py +1467 -0
- unicode_logic_kit/fol/_msfl_nodes.py +2196 -0
- unicode_logic_kit/fol/_numeral_symbols.py +231 -0
- unicode_logic_kit/fol/_so_nodes.py +200 -0
- unicode_logic_kit/fol/_symbol_names.py +81 -0
- unicode_logic_kit/fol/_team_nodes.py +181 -0
- unicode_logic_kit/fol/_tptp_symbols.py +551 -0
- unicode_logic_kit/fol/_truth_constants.py +117 -0
- unicode_logic_kit/fol/casl_export.py +1135 -0
- unicode_logic_kit/fol/casl_import.py +929 -0
- unicode_logic_kit/fol/derivation.py +367 -0
- unicode_logic_kit/fol/dialect_detect.py +70 -0
- unicode_logic_kit/fol/dialect_repair.py +537 -0
- unicode_logic_kit/fol/frames.py +637 -0
- unicode_logic_kit/fol/grammars/terminals.lark +31 -0
- unicode_logic_kit/fol/lambda_tools.py +297 -0
- unicode_logic_kit/fol/latex_input.py +429 -0
- unicode_logic_kit/fol/modal_translation.py +944 -0
- unicode_logic_kit/fol/msflparser.py +1033 -0
- unicode_logic_kit/fol/naming.py +422 -0
- unicode_logic_kit/fol/nodes.py +241 -0
- unicode_logic_kit/fol/normalforms.py +492 -0
- unicode_logic_kit/fol/pal.py +287 -0
- unicode_logic_kit/fol/prolog_export.py +566 -0
- unicode_logic_kit/fol/prolog_input.py +505 -0
- unicode_logic_kit/fol/prover9_input.py +1325 -0
- unicode_logic_kit/fol/qml.py +1760 -0
- unicode_logic_kit/fol/qmltp_input.py +525 -0
- unicode_logic_kit/fol/sanitize.py +221 -0
- unicode_logic_kit/fol/serialize.py +79 -0
- unicode_logic_kit/fol/signature.py +1290 -0
- unicode_logic_kit/fol/simplify_check.py +544 -0
- unicode_logic_kit/fol/spans.py +594 -0
- unicode_logic_kit/fol/tptp_input.py +1503 -0
- unicode_logic_kit/fol/tptp_repair.py +941 -0
- unicode_logic_kit/fol/unification.py +157 -0
- unicode_logic_kit/fol/verbalize.py +263 -0
- unicode_logic_kit/hets/__init__.py +163 -0
- unicode_logic_kit/hets/bridge.py +142 -0
- unicode_logic_kit/hets/client.py +748 -0
- unicode_logic_kit/hets/docker.py +420 -0
- unicode_logic_kit/hets/dol.py +712 -0
- unicode_logic_kit/hets/haskell_json.py +355 -0
- unicode_logic_kit/hets/owl_backend.py +794 -0
- unicode_logic_kit/hets/owl_cli.py +598 -0
- unicode_logic_kit/hets/symbols.py +512 -0
- unicode_logic_kit/hol/__init__.py +140 -0
- unicode_logic_kit/hol/_ho_common.py +323 -0
- unicode_logic_kit/hol/_isabelle_binders.py +125 -0
- unicode_logic_kit/hol/classical.py +812 -0
- unicode_logic_kit/hol/deepshallow/__init__.py +45 -0
- unicode_logic_kit/hol/deepshallow/_common.py +177 -0
- unicode_logic_kit/hol/deepshallow/conditional.py +225 -0
- unicode_logic_kit/hol/deepshallow/intuitionistic.py +181 -0
- unicode_logic_kit/hol/deepshallow/modal.py +217 -0
- unicode_logic_kit/hol/deepshallow/qml.py +406 -0
- unicode_logic_kit/hol/deepshallow/relevant.py +206 -0
- unicode_logic_kit/hol/free.py +753 -0
- unicode_logic_kit/hol/goedel.py +336 -0
- unicode_logic_kit/hol/ho_modal.py +1743 -0
- unicode_logic_kit/hol/intuitionistic.py +403 -0
- unicode_logic_kit/hol/isabelle_conditional.py +593 -0
- unicode_logic_kit/hol/isabelle_modal.py +1908 -0
- unicode_logic_kit/hol/isabelle_relevant.py +412 -0
- unicode_logic_kit/hol/isabelle_runner.py +1147 -0
- unicode_logic_kit/hol/isabelle_substructural.py +884 -0
- unicode_logic_kit/hol/lean.py +1018 -0
- unicode_logic_kit/hol/manyvalued.py +921 -0
- unicode_logic_kit/hol/secondorder.py +687 -0
- unicode_logic_kit/hol/thf_modal.py +941 -0
- unicode_logic_kit/hol/thirdorder.py +397 -0
- unicode_logic_kit/ilp/__init__.py +89 -0
- unicode_logic_kit/ilp/readback.py +389 -0
- unicode_logic_kit/ilp/separation.py +153 -0
- unicode_logic_kit/ilp/task.py +730 -0
- unicode_logic_kit/logic.py +163 -0
- unicode_logic_kit/mcp/__init__.py +28 -0
- unicode_logic_kit/mcp/__main__.py +5 -0
- unicode_logic_kit/mcp/chem_tools.py +1031 -0
- unicode_logic_kit/mcp/server.py +2453 -0
- unicode_logic_kit/mcp/syntax_spec.py +681 -0
- unicode_logic_kit/prob/__init__.py +53 -0
- unicode_logic_kit/prob/_bdd.py +225 -0
- unicode_logic_kit/prob/_column_gen.py +668 -0
- unicode_logic_kit/prob/distribution.py +686 -0
- unicode_logic_kit/prob/nilsson.py +470 -0
- unicode_logic_kit/py.typed +0 -0
- unicode_logic_kit/semantics/__init__.py +137 -0
- unicode_logic_kit/semantics/_modal_reject.py +156 -0
- unicode_logic_kit/semantics/action_models.py +466 -0
- unicode_logic_kit/semantics/asp_models.py +1200 -0
- unicode_logic_kit/semantics/conditional.py +580 -0
- unicode_logic_kit/semantics/dynamic_epistemic.py +95 -0
- unicode_logic_kit/semantics/free_logic.py +913 -0
- unicode_logic_kit/semantics/fuzzy.py +384 -0
- unicode_logic_kit/semantics/fuzzy_kripke.py +442 -0
- unicode_logic_kit/semantics/intuitionistic.py +581 -0
- unicode_logic_kit/semantics/kripke.py +1139 -0
- unicode_logic_kit/semantics/manyvalued.py +580 -0
- unicode_logic_kit/semantics/matrix.py +342 -0
- unicode_logic_kit/semantics/model_eval.py +1135 -0
- unicode_logic_kit/semantics/modelfinder.py +1036 -0
- unicode_logic_kit/semantics/nonmonotonic.py +372 -0
- unicode_logic_kit/semantics/relevant.py +331 -0
- unicode_logic_kit/semantics/secondorder.py +657 -0
- unicode_logic_kit/semantics/structures.py +352 -0
- unicode_logic_kit/semantics/tarski.py +975 -0
- unicode_logic_kit/semantics/team.py +315 -0
- unicode_logic_kit/semantics/team_translation.py +416 -0
- unicode_logic_kit/semantics/thirdorder.py +358 -0
- unicode_logic_kit/semantics/tnorm.py +85 -0
- unicode_logic_kit/semantics/truthtable.py +201 -0
- unicode_logic_kit-0.31.0.dist-info/METADATA +333 -0
- unicode_logic_kit-0.31.0.dist-info/RECORD +237 -0
- unicode_logic_kit-0.31.0.dist-info/WHEEL +4 -0
- unicode_logic_kit-0.31.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,594 @@
|
|
|
1
|
+
"""Adapter for P-FOLIO (Han, Simeng, et al., "P-FOLIO: Evaluating and
|
|
2
|
+
Improving Logical Reasoning with Abundant Human-Written Reasoning Chains",
|
|
3
|
+
Findings of the Association for Computational Linguistics: EMNLP 2024) —
|
|
4
|
+
local CSV files only, no network access.
|
|
5
|
+
|
|
6
|
+
Source and verified schema
|
|
7
|
+
---------------------------
|
|
8
|
+
Canonical source: https://huggingface.co/datasets/yale-nlp/P-FOLIO (access
|
|
9
|
+
gated — a Hugging Face login is required to download it; this loader never
|
|
10
|
+
downloads anything itself). P-FOLIO extends FOLIO (Han et al., "FOLIO:
|
|
11
|
+
Natural Language Reasoning with First-Order Logic", arXiv:2209.00840; see
|
|
12
|
+
also :mod:`~unicode_logic_kit.eval.datasets.folio`, which reads a DIFFERENT,
|
|
13
|
+
JSONL, distribution of FOLIO — the ``FOLIO.csv`` this module joins against
|
|
14
|
+
is P-FOLIO's OWN bundled spreadsheet export of the same underlying stories,
|
|
15
|
+
not that JSONL file, and the two do not share a row format) with a
|
|
16
|
+
human-written, step-by-step derivation for every (story, conclusion) pair.
|
|
17
|
+
The repository ships the dataset as two CSV files, and this adapter's own
|
|
18
|
+
schema assumptions below were checked directly against a real download
|
|
19
|
+
(one file each, 2026-09) — not assumed from the paper or the HF card:
|
|
20
|
+
|
|
21
|
+
``P-FOLIO.csv`` (16071 data rows, columns ``story_id``, ``Truth Value``,
|
|
22
|
+
``Premises used``, ``Derivation``, ``Derivation - Corrected``,
|
|
23
|
+
``Derivation index``, ``Inference rule``) is a spreadsheet export where **a
|
|
24
|
+
row with a non-blank, digit-only ``story_id`` opens one conclusion's
|
|
25
|
+
block**: its ``Truth Value`` is that conclusion's gold label, and every
|
|
26
|
+
following row with a blank ``story_id`` is one derivation step of that
|
|
27
|
+
block (``Derivation index`` ``D1``, ``D2``, ... referencing earlier steps or
|
|
28
|
+
``Premises used`` indices into the story's premise list). A story with
|
|
29
|
+
several conclusions has several consecutive blocks; **the block's position
|
|
30
|
+
among its story's blocks, counted in file order starting at 0, is the only
|
|
31
|
+
thing that says WHICH conclusion it is** — the file carries no conclusion
|
|
32
|
+
id of its own. Many rows are blank padding between blocks.
|
|
33
|
+
|
|
34
|
+
``FOLIO.csv`` (487 data rows, columns ``''``, ``Premises - NL``,
|
|
35
|
+
``Conclusions - NL``, ``Truth Values``, ``Premises - FOL``,
|
|
36
|
+
``Conclusions - FOL``, ``Comments``, ``Verified by Prover``) has one row per
|
|
37
|
+
STORY: its unnamed first column is the story id (verified: every id ``0``
|
|
38
|
+
… ``486`` present exactly once, matching ``P-FOLIO.csv``'s ``story_id``
|
|
39
|
+
range exactly), and ``Premises - NL``/``Conclusions - NL``/
|
|
40
|
+
``Truth Values``/``Premises - FOL``/``Conclusions - FOL`` each hold a
|
|
41
|
+
NEWLINE-separated list — one entry per premise (first two columns) or per
|
|
42
|
+
conclusion (last three), in the SAME order the P-FOLIO blocks for that
|
|
43
|
+
story appear in.
|
|
44
|
+
|
|
45
|
+
This adapter's join, and what it refuses
|
|
46
|
+
------------------------------------------
|
|
47
|
+
The two files carry no shared conclusion id, so — exactly as this module's
|
|
48
|
+
build spec requires — a P-FOLIO block's ``nl_conclusion``/``fol_conclusion``
|
|
49
|
+
are resolved by joining ``FOLIO.csv`` on **story id AND the block's
|
|
50
|
+
position among its story's blocks**, and every join is CROSS-CHECKED: the
|
|
51
|
+
block's own ``Truth Value`` must equal ``FOLIO.csv``'s truth value at that
|
|
52
|
+
same position, or the block is refused rather than trusted. This was
|
|
53
|
+
verified by hand against several real stories before being encoded as a
|
|
54
|
+
blanket check (see ``tests/test_datasets_pfolio.py`` for the same
|
|
55
|
+
cross-check run against the small fixture below): story 5's two ``T``
|
|
56
|
+
blocks' last derivation steps read, verbatim, "If the Hulk does not wake up,
|
|
57
|
+
then Thor is not happy." and "If Thor is happy, then Peter Parker wears a
|
|
58
|
+
uniform" — exactly ``FOLIO.csv`` story 5's first two conclusions, in order,
|
|
59
|
+
both labelled ``T`` on both sides.
|
|
60
|
+
|
|
61
|
+
Of the real, once-downloaded ``P-FOLIO.csv``'s 1431 blocks (2026-09), **1420
|
|
62
|
+
load cleanly** through :func:`load_pfolio` and **11 are refused** — a 99.2%
|
|
63
|
+
yield, all 11 individually accounted for below rather than swallowed into an
|
|
64
|
+
aggregate. Re-measured directly through the loader by
|
|
65
|
+
``test_the_real_files_join_as_measured`` in ``tests/test_datasets_pfolio.py``
|
|
66
|
+
(opt-in, since the source files are access-gated — see below).
|
|
67
|
+
|
|
68
|
+
A block is REFUSED (excluded from :func:`load_pfolio`'s default strict
|
|
69
|
+
pass, listed with a reason by :func:`pfolio_refusals`) — never guessed —
|
|
70
|
+
for any of:
|
|
71
|
+
|
|
72
|
+
* ``"unparseable_truth_value"`` — the block's ``Truth Value`` cell, after
|
|
73
|
+
stripping whitespace, is not exactly ``"T"``/``"F"``/``"U"``. The real
|
|
74
|
+
file's two worst offenders are reviewer notes left IN the truth-value
|
|
75
|
+
cell instead of a real ``T``/``F``/``U``, e.g. ``"F -> should be U?\\n
|
|
76
|
+
rui_comment: F is correct."`` — guessing which of the two conflicting
|
|
77
|
+
reviewers is right is exactly the guessing this adapter declines to do.
|
|
78
|
+
A bare trailing newline (``"T\\n"``, common throughout the real file) is
|
|
79
|
+
NOT an unparseable value — it is stripped, not refused.
|
|
80
|
+
* ``"folio_story_ambiguous"`` — that story's ``FOLIO.csv`` row cannot be
|
|
81
|
+
read unambiguously in the first place (see :func:`_read_folio`: its
|
|
82
|
+
conclusions/truth-values/conclusion-FOL lists disagree in length, or its
|
|
83
|
+
premises/premises-FOL lists do, even after trimming wholly-blank leading
|
|
84
|
+
or trailing list entries — a known export artifact, see below). Verified
|
|
85
|
+
in the real file: of 487 stories, only story 249 is genuinely ambiguous
|
|
86
|
+
this way (its ``Conclusions - NL`` column holds 6 lines that read like
|
|
87
|
+
PREMISES, not conclusions, while ``Truth Values``/``Conclusions - FOL``
|
|
88
|
+
hold only 2 — a real upstream data defect, not a formatting artifact).
|
|
89
|
+
* ``"story_not_in_folio"`` — the block's ``story_id`` has no row in
|
|
90
|
+
``FOLIO.csv`` at all (defensive; does not occur in the verified real
|
|
91
|
+
files, where the id ranges match exactly, but a caller-supplied fixture
|
|
92
|
+
or a future release could still have this).
|
|
93
|
+
* ``"position_out_of_range"`` — the block's position exceeds how many
|
|
94
|
+
conclusions that story's ``FOLIO.csv`` row actually has. Verified once in
|
|
95
|
+
the real file: story 135 has three P-FOLIO blocks but only two resolved
|
|
96
|
+
FOLIO conclusions.
|
|
97
|
+
* ``"truth_value_mismatch"`` — the block's own truth value parses fine and
|
|
98
|
+
the position resolves, but disagrees with ``FOLIO.csv``'s truth value at
|
|
99
|
+
that position. Verified five times in the real file (stories 34, 50
|
|
100
|
+
twice, 258, 409) — genuine annotation disagreements between the two
|
|
101
|
+
files, not something this adapter is in a position to adjudicate.
|
|
102
|
+
|
|
103
|
+
Three further export artifacts, all handled WITHOUT guessing at their
|
|
104
|
+
content, are worth naming explicitly because they could otherwise look like
|
|
105
|
+
corruption:
|
|
106
|
+
|
|
107
|
+
* A handful of rows in ``P-FOLIO.csv`` have a non-blank, non-digit
|
|
108
|
+
``story_id`` (e.g. a stray ``"("`` character, or a whole comment row like
|
|
109
|
+
``"Need to change xor"`` sitting alone between blank padding rows). Only
|
|
110
|
+
a BLANK or DIGIT-ONLY ``story_id`` is read as starting a new block; any
|
|
111
|
+
other value is never interpreted as an id OR as a truth value — the row
|
|
112
|
+
is treated exactly like a blank-``story_id`` row, i.e. as one more
|
|
113
|
+
derivation-step candidate of whichever block is currently open (real
|
|
114
|
+
content columns intact) or as ignorable padding (nothing else on the
|
|
115
|
+
row). This is the same "only the recognised column means anything on
|
|
116
|
+
this row" treatment already used for a stray reviewer comment landing in
|
|
117
|
+
a derivation row's ``Truth Value`` cell (see the fixture and
|
|
118
|
+
``tests/test_datasets_pfolio.py`` for both real-shaped cases).
|
|
119
|
+
* Several ``FOLIO.csv`` cells hold one extra wholly-blank leading or
|
|
120
|
+
trailing line inside an otherwise-consistent newline-separated list
|
|
121
|
+
(verified: stories 55, 113, 135). :func:`_split_field` trims ONLY
|
|
122
|
+
wholly-blank entries at the very start/end of such a list — never a
|
|
123
|
+
blank in the middle, and never anything from a non-blank entry — before
|
|
124
|
+
the length cross-check above runs, so these three stories load normally
|
|
125
|
+
rather than being flagged ``folio_story_ambiguous``.
|
|
126
|
+
* A block's FIRST derivation step's own content (``Premises used``/
|
|
127
|
+
``Derivation``/``Derivation - Corrected``/``Derivation index``/
|
|
128
|
+
``Inference rule``) sometimes sits on the SAME row as the block's own
|
|
129
|
+
``story_id``/``Truth Value`` header, rather than on a following
|
|
130
|
+
blank-``story_id`` row (verified: real story 246's ``D1``, and 164 of the
|
|
131
|
+
real file's 1431 blocks overall, 33 of them with that step as their ONLY
|
|
132
|
+
one — indistinguishable from a genuinely derivation-less block unless
|
|
133
|
+
this row is also inspected). :func:`_iter_pfolio_blocks` checks a
|
|
134
|
+
header row's own columns 2–6 the same way it checks any other row's, so
|
|
135
|
+
that first step is kept, not silently dropped.
|
|
136
|
+
|
|
137
|
+
``Derivation`` vs. ``Derivation - Corrected`` are two DIFFERENT upstream
|
|
138
|
+
columns (the corrected text is only sometimes present, and sometimes reads
|
|
139
|
+
quite differently from the original — both were observed verbatim in the
|
|
140
|
+
real file) and this adapter keeps them as two separate keys in every
|
|
141
|
+
``meta["proof_steps"]`` entry, never merged into one. ``Premises used`` is
|
|
142
|
+
likewise kept VERBATIM as a single string rather than parsed into a list of
|
|
143
|
+
indices: real values mix plain premise numbers, ``D``-prefixed references
|
|
144
|
+
to earlier steps, comma- and even full-width-comma (``,``)-separated lists
|
|
145
|
+
(``"3,5"``) in the same column — parsing that into a clean structure would
|
|
146
|
+
be exactly the kind of guess this adapter's build spec forbids.
|
|
147
|
+
|
|
148
|
+
Truth-value vocabulary — a real, documented difference from
|
|
149
|
+
:mod:`~unicode_logic_kit.eval.datasets.folio`
|
|
150
|
+
------------------------------------------------------------------------
|
|
151
|
+
``FOLIO.csv``'s ``Truth Values`` column (and P-FOLIO's ``Truth Value``
|
|
152
|
+
column) use the single letters ``"T"``/``"F"``/``"U"`` (:data:`PFOLIO_TRUTH_VALUES`)
|
|
153
|
+
— verified directly, not the ``"True"``/``"False"``/``"Uncertain"`` words
|
|
154
|
+
the OTHER FOLIO adapter's JSONL source uses. ``label`` here is that letter,
|
|
155
|
+
UNCHANGED — this loader does not translate between the two vocabularies
|
|
156
|
+
(that would be presenting an invented value as gold data).
|
|
157
|
+
|
|
158
|
+
License: **MIT**, per the yale-nlp/P-FOLIO Hugging Face dataset card.
|
|
159
|
+
|
|
160
|
+
This module never downloads anything — obtain both CSV files yourself from
|
|
161
|
+
https://huggingface.co/datasets/yale-nlp/P-FOLIO (a Hugging Face account and
|
|
162
|
+
accepted access request are required) and pass their local paths to
|
|
163
|
+
:func:`load_pfolio`.
|
|
164
|
+
"""
|
|
165
|
+
|
|
166
|
+
import csv
|
|
167
|
+
from pathlib import Path
|
|
168
|
+
from typing import Dict, FrozenSet, Iterator, List, NamedTuple, Optional, Tuple, Union
|
|
169
|
+
|
|
170
|
+
from ._base import DatasetExample, _register_dataset_info
|
|
171
|
+
|
|
172
|
+
__all__ = ["load_pfolio", "pfolio_refusals", "PFOLIO_TRUTH_VALUES",
|
|
173
|
+
"PFOLIO_REFUSAL_REASONS"]
|
|
174
|
+
|
|
175
|
+
#: The verified vocabulary of P-FOLIO's/FOLIO.csv's own truth-value tokens
|
|
176
|
+
#: (single letters — see the module docstring for how this differs from
|
|
177
|
+
#: :mod:`~unicode_logic_kit.eval.datasets.folio`'s word-form labels).
|
|
178
|
+
PFOLIO_TRUTH_VALUES = ("T", "F", "U")
|
|
179
|
+
|
|
180
|
+
#: The reasons :func:`load_pfolio`/:func:`pfolio_refusals` can report for a
|
|
181
|
+
#: block that was NOT turned into a :class:`~unicode_logic_kit.eval.datasets.DatasetExample`
|
|
182
|
+
#: — see the module docstring's "This adapter's join, and what it refuses"
|
|
183
|
+
#: section for what each one means and how often it fires on the real file.
|
|
184
|
+
PFOLIO_REFUSAL_REASONS = (
|
|
185
|
+
"unparseable_truth_value", "folio_story_ambiguous", "story_not_in_folio",
|
|
186
|
+
"position_out_of_range", "truth_value_mismatch",
|
|
187
|
+
)
|
|
188
|
+
|
|
189
|
+
_register_dataset_info(
|
|
190
|
+
"pfolio",
|
|
191
|
+
license="MIT, per the yale-nlp/P-FOLIO Hugging Face dataset card.",
|
|
192
|
+
source_url="https://huggingface.co/datasets/yale-nlp/P-FOLIO",
|
|
193
|
+
citation_hint=(
|
|
194
|
+
"Han, Simeng, et al. \"P-FOLIO: Evaluating and Improving Logical "
|
|
195
|
+
"Reasoning with Abundant Human-Written Reasoning Chains.\" Findings "
|
|
196
|
+
"of the Association for Computational Linguistics: EMNLP 2024, "
|
|
197
|
+
"pages 16553-16565. arXiv:2410.09207."
|
|
198
|
+
),
|
|
199
|
+
)
|
|
200
|
+
|
|
201
|
+
|
|
202
|
+
# ---------------------------------------------------------------------------
|
|
203
|
+
# FOLIO.csv — one row per story
|
|
204
|
+
# ---------------------------------------------------------------------------
|
|
205
|
+
|
|
206
|
+
_FOLIO_HEADER = (
|
|
207
|
+
"", "Premises - NL", "Conclusions - NL", "Truth Values",
|
|
208
|
+
"Premises - FOL", "Conclusions - FOL", "Comments", "Verified by Prover",
|
|
209
|
+
)
|
|
210
|
+
|
|
211
|
+
|
|
212
|
+
class _FolioStory(NamedTuple):
|
|
213
|
+
premises_nl: Tuple[str, ...]
|
|
214
|
+
premises_fol: Tuple[str, ...]
|
|
215
|
+
conclusions_nl: Tuple[str, ...]
|
|
216
|
+
conclusions_fol: Tuple[str, ...]
|
|
217
|
+
truth_values: Tuple[str, ...]
|
|
218
|
+
comments: Optional[str]
|
|
219
|
+
verified_by_prover: Optional[str]
|
|
220
|
+
|
|
221
|
+
|
|
222
|
+
def _lf_cells(row: List[str]) -> List[str]:
|
|
223
|
+
"""``row`` with every line break inside a cell rewritten to ``\\n``.
|
|
224
|
+
|
|
225
|
+
``csv`` (read with ``newline=""``, as it must be) hands a quoted cell's
|
|
226
|
+
embedded line breaks through VERBATIM. The real files break lines inside
|
|
227
|
+
cells with a bare ``\\n``, but a copy that went through a CRLF-converting
|
|
228
|
+
tool (a git checkout with ``core.autocrlf``, a Windows editor) carries
|
|
229
|
+
``\\r\\n`` there, and splitting that on ``\\n`` leaves a ``\\r`` glued
|
|
230
|
+
to every premise and conclusion. A line break inside a cell means the
|
|
231
|
+
same thing in either convention, so both files read identically."""
|
|
232
|
+
return [cell.replace("\r\n", "\n").replace("\r", "\n") for cell in row]
|
|
233
|
+
|
|
234
|
+
|
|
235
|
+
def _split_field(cell: str) -> List[str]:
|
|
236
|
+
"""Newline-split ``cell``, trimming only WHOLLY-BLANK leading/trailing
|
|
237
|
+
entries (a verified export artifact — see the module docstring). A
|
|
238
|
+
blank entry anywhere else in the list, or any whitespace inside a
|
|
239
|
+
non-blank entry, is left untouched."""
|
|
240
|
+
lines = cell.split("\n")
|
|
241
|
+
while lines and lines[0].strip() == "":
|
|
242
|
+
lines.pop(0)
|
|
243
|
+
while lines and lines[-1].strip() == "":
|
|
244
|
+
lines.pop()
|
|
245
|
+
return lines
|
|
246
|
+
|
|
247
|
+
|
|
248
|
+
def _read_folio(path: Union[str, Path]) -> Tuple[Dict[int, _FolioStory], Dict[int, str]]:
|
|
249
|
+
"""Read ``FOLIO.csv`` into per-story records.
|
|
250
|
+
|
|
251
|
+
Returns ``(stories, ambiguous)``: ``stories`` maps a story id to a
|
|
252
|
+
:class:`_FolioStory` for every story whose premise/conclusion lists are
|
|
253
|
+
internally consistent; ``ambiguous`` maps every OTHER story id to a
|
|
254
|
+
human-readable reason it could not be read unambiguously (see the
|
|
255
|
+
module docstring's ``"folio_story_ambiguous"`` entry) — such a story is
|
|
256
|
+
never silently dropped, only excluded from ``stories``.
|
|
257
|
+
|
|
258
|
+
Raises:
|
|
259
|
+
ValueError: the header does not match the verified ``FOLIO.csv``
|
|
260
|
+
column layout, a row's own id does not parse as a non-negative
|
|
261
|
+
integer, or a story id repeats — this is STRUCTURAL corruption
|
|
262
|
+
of the reference file itself, refused unconditionally (unlike
|
|
263
|
+
the per-story ambiguity above, there is no safe partial reading
|
|
264
|
+
of a file whose id scheme is broken).
|
|
265
|
+
"""
|
|
266
|
+
path = Path(path)
|
|
267
|
+
stories: Dict[int, _FolioStory] = {}
|
|
268
|
+
ambiguous: Dict[int, str] = {}
|
|
269
|
+
with path.open("r", encoding="utf-8-sig", newline="") as fh:
|
|
270
|
+
reader = csv.reader(fh)
|
|
271
|
+
header = tuple(next(reader))
|
|
272
|
+
if header != _FOLIO_HEADER:
|
|
273
|
+
raise ValueError(
|
|
274
|
+
f"pfolio: {path} has header {header!r}, expected "
|
|
275
|
+
f"{_FOLIO_HEADER!r} — is this FOLIO.csv?")
|
|
276
|
+
|
|
277
|
+
seen: Dict[int, int] = {}
|
|
278
|
+
for row_no, row in enumerate(reader):
|
|
279
|
+
if len(row) != len(_FOLIO_HEADER):
|
|
280
|
+
raise ValueError(
|
|
281
|
+
f"pfolio: {path} row {row_no} has {len(row)} fields, "
|
|
282
|
+
f"expected {len(_FOLIO_HEADER)}")
|
|
283
|
+
row = _lf_cells(row)
|
|
284
|
+
raw_id = row[0].strip()
|
|
285
|
+
if not raw_id.isdigit():
|
|
286
|
+
raise ValueError(
|
|
287
|
+
f"pfolio: {path} row {row_no} has story id {row[0]!r}, "
|
|
288
|
+
"not a non-negative integer — is this FOLIO.csv?")
|
|
289
|
+
story_id = int(raw_id)
|
|
290
|
+
if story_id in seen:
|
|
291
|
+
raise ValueError(
|
|
292
|
+
f"pfolio: {path} has story id {story_id} at both rows "
|
|
293
|
+
f"{seen[story_id]} and {row_no} — ids must be unique")
|
|
294
|
+
seen[story_id] = row_no
|
|
295
|
+
|
|
296
|
+
premises_nl = _split_field(row[1])
|
|
297
|
+
conclusions_nl = _split_field(row[2])
|
|
298
|
+
truth_values = _split_field(row[3])
|
|
299
|
+
premises_fol = _split_field(row[4])
|
|
300
|
+
conclusions_fol = _split_field(row[5])
|
|
301
|
+
comments = row[6].strip() or None
|
|
302
|
+
verified_by_prover = row[7].strip() or None
|
|
303
|
+
|
|
304
|
+
if len(premises_nl) != len(premises_fol):
|
|
305
|
+
ambiguous[story_id] = (
|
|
306
|
+
f"premises-NL/premises-FOL length mismatch "
|
|
307
|
+
f"({len(premises_nl)} vs {len(premises_fol)})")
|
|
308
|
+
continue
|
|
309
|
+
if not (len(conclusions_nl) == len(truth_values) == len(conclusions_fol)):
|
|
310
|
+
ambiguous[story_id] = (
|
|
311
|
+
"conclusions-NL/truth-values/conclusions-FOL length "
|
|
312
|
+
f"mismatch ({len(conclusions_nl)}, {len(truth_values)}, "
|
|
313
|
+
f"{len(conclusions_fol)})")
|
|
314
|
+
continue
|
|
315
|
+
bad_tv = [t for t in truth_values if t.strip() not in PFOLIO_TRUTH_VALUES]
|
|
316
|
+
if bad_tv:
|
|
317
|
+
ambiguous[story_id] = f"unparseable truth value(s) {bad_tv!r}"
|
|
318
|
+
continue
|
|
319
|
+
|
|
320
|
+
stories[story_id] = _FolioStory(
|
|
321
|
+
premises_nl=tuple(premises_nl),
|
|
322
|
+
premises_fol=tuple(premises_fol),
|
|
323
|
+
conclusions_nl=tuple(conclusions_nl),
|
|
324
|
+
conclusions_fol=tuple(conclusions_fol),
|
|
325
|
+
truth_values=tuple(t.strip() for t in truth_values),
|
|
326
|
+
comments=comments,
|
|
327
|
+
verified_by_prover=verified_by_prover,
|
|
328
|
+
)
|
|
329
|
+
return stories, ambiguous
|
|
330
|
+
|
|
331
|
+
|
|
332
|
+
# ---------------------------------------------------------------------------
|
|
333
|
+
# P-FOLIO.csv — several derivation-step rows per block, several blocks per story
|
|
334
|
+
# ---------------------------------------------------------------------------
|
|
335
|
+
|
|
336
|
+
_PFOLIO_HEADER = (
|
|
337
|
+
"story_id", "Truth Value", "Premises used", "Derivation",
|
|
338
|
+
"Derivation - Corrected", "Derivation index", "Inference rule",
|
|
339
|
+
)
|
|
340
|
+
|
|
341
|
+
|
|
342
|
+
def _none_if_blank(text: str) -> Optional[str]:
|
|
343
|
+
stripped = text.strip()
|
|
344
|
+
return stripped if stripped else None
|
|
345
|
+
|
|
346
|
+
|
|
347
|
+
class _RawBlock(NamedTuple):
|
|
348
|
+
story_id: int
|
|
349
|
+
truth_raw: str
|
|
350
|
+
steps: Tuple[dict, ...]
|
|
351
|
+
header_row_no: int
|
|
352
|
+
|
|
353
|
+
|
|
354
|
+
def _iter_pfolio_blocks(path: Union[str, Path]) -> Iterator[_RawBlock]:
|
|
355
|
+
"""Group ``P-FOLIO.csv``'s rows into blocks, in file order.
|
|
356
|
+
|
|
357
|
+
A row STARTS a new block iff its ``story_id`` cell, stripped, is
|
|
358
|
+
non-empty and digit-only; every other row either extends the currently
|
|
359
|
+
open block (if it carries any real step content — see the module
|
|
360
|
+
docstring for why a stray non-digit ``story_id`` or a stray comment in
|
|
361
|
+
the ``Truth Value`` column of such a row is never interpreted) or is
|
|
362
|
+
ignored as padding.
|
|
363
|
+
|
|
364
|
+
Raises:
|
|
365
|
+
ValueError: the header does not match the verified ``P-FOLIO.csv``
|
|
366
|
+
column layout, or a row has the wrong field count — structural
|
|
367
|
+
corruption of the file itself.
|
|
368
|
+
"""
|
|
369
|
+
path = Path(path)
|
|
370
|
+
current: Optional[dict] = None
|
|
371
|
+
with path.open("r", encoding="utf-8-sig", newline="") as fh:
|
|
372
|
+
reader = csv.reader(fh)
|
|
373
|
+
header = tuple(next(reader))
|
|
374
|
+
if header != _PFOLIO_HEADER:
|
|
375
|
+
raise ValueError(
|
|
376
|
+
f"pfolio: {path} has header {header!r}, expected "
|
|
377
|
+
f"{_PFOLIO_HEADER!r} — is this P-FOLIO.csv?")
|
|
378
|
+
|
|
379
|
+
for row_no, row in enumerate(reader):
|
|
380
|
+
if len(row) != len(_PFOLIO_HEADER):
|
|
381
|
+
raise ValueError(
|
|
382
|
+
f"pfolio: {path} row {row_no} has {len(row)} fields, "
|
|
383
|
+
f"expected {len(_PFOLIO_HEADER)}")
|
|
384
|
+
row = _lf_cells(row)
|
|
385
|
+
story_id_raw = row[0].strip()
|
|
386
|
+
if story_id_raw.isdigit():
|
|
387
|
+
if current is not None:
|
|
388
|
+
yield _RawBlock(current["story_id"], current["truth_raw"],
|
|
389
|
+
tuple(current["steps"]), current["header_row_no"])
|
|
390
|
+
current = {"story_id": int(story_id_raw), "truth_raw": row[1],
|
|
391
|
+
"steps": [], "header_row_no": row_no}
|
|
392
|
+
if any(cell.strip() for cell in row[2:7]):
|
|
393
|
+
# A real export artifact: the block's FIRST derivation
|
|
394
|
+
# step's content sometimes sits on the very same row as
|
|
395
|
+
# the story_id/Truth Value header, rather than on a
|
|
396
|
+
# following blank-story_id row (verified: real story
|
|
397
|
+
# 246's D1). Not capturing it here would silently drop
|
|
398
|
+
# that step — see the module docstring.
|
|
399
|
+
current["steps"].append({
|
|
400
|
+
"premises_used": _none_if_blank(row[2]),
|
|
401
|
+
"derivation": _none_if_blank(row[3]),
|
|
402
|
+
"derivation_corrected": _none_if_blank(row[4]),
|
|
403
|
+
"step_id": _none_if_blank(row[5]),
|
|
404
|
+
"inference_rule": _none_if_blank(row[6]),
|
|
405
|
+
"row_no": row_no,
|
|
406
|
+
})
|
|
407
|
+
continue
|
|
408
|
+
if current is None:
|
|
409
|
+
continue # padding before any block
|
|
410
|
+
if any(cell.strip() for cell in row[2:7]):
|
|
411
|
+
current["steps"].append({
|
|
412
|
+
"premises_used": _none_if_blank(row[2]),
|
|
413
|
+
"derivation": _none_if_blank(row[3]),
|
|
414
|
+
"derivation_corrected": _none_if_blank(row[4]),
|
|
415
|
+
"step_id": _none_if_blank(row[5]),
|
|
416
|
+
"inference_rule": _none_if_blank(row[6]),
|
|
417
|
+
"row_no": row_no,
|
|
418
|
+
})
|
|
419
|
+
if current is not None:
|
|
420
|
+
yield _RawBlock(current["story_id"], current["truth_raw"],
|
|
421
|
+
tuple(current["steps"]), current["header_row_no"])
|
|
422
|
+
|
|
423
|
+
|
|
424
|
+
def _parse_truth_value(raw: str) -> Optional[str]:
|
|
425
|
+
stripped = raw.strip()
|
|
426
|
+
return stripped if stripped in PFOLIO_TRUTH_VALUES else None
|
|
427
|
+
|
|
428
|
+
|
|
429
|
+
# ---------------------------------------------------------------------------
|
|
430
|
+
# The join
|
|
431
|
+
# ---------------------------------------------------------------------------
|
|
432
|
+
|
|
433
|
+
def _iter_joined(pfolio_path: Union[str, Path], folio_path: Union[str, Path],
|
|
434
|
+
known_bad_ids: FrozenSet[str]) -> Iterator[Tuple[str, object]]:
|
|
435
|
+
"""Yield ``("ok", DatasetExample)`` or ``("refused", dict)`` per P-FOLIO
|
|
436
|
+
block, in file order — the shared engine behind :func:`load_pfolio` and
|
|
437
|
+
:func:`pfolio_refusals`, so the two never disagree about what loads.
|
|
438
|
+
|
|
439
|
+
A refusal dict is ``{"story_id", "position", "reason" (one of
|
|
440
|
+
:data:`PFOLIO_REFUSAL_REASONS`), "detail", "row_no"}`` — ``row_no`` is
|
|
441
|
+
the block's own header row's 0-based data-row number in ``P-FOLIO.csv``,
|
|
442
|
+
for tracing a refusal back to the source file.
|
|
443
|
+
"""
|
|
444
|
+
stories, ambiguous = _read_folio(folio_path)
|
|
445
|
+
positions: Dict[int, int] = {}
|
|
446
|
+
|
|
447
|
+
for block in _iter_pfolio_blocks(pfolio_path):
|
|
448
|
+
story_id = block.story_id
|
|
449
|
+
position = positions.get(story_id, 0)
|
|
450
|
+
positions[story_id] = position + 1
|
|
451
|
+
|
|
452
|
+
def _refused(reason: str, detail: Optional[str]) -> Tuple[str, dict]:
|
|
453
|
+
return ("refused", {"story_id": story_id, "position": position,
|
|
454
|
+
"reason": reason, "detail": detail,
|
|
455
|
+
"row_no": block.header_row_no})
|
|
456
|
+
|
|
457
|
+
truth_value = _parse_truth_value(block.truth_raw)
|
|
458
|
+
if truth_value is None:
|
|
459
|
+
yield _refused("unparseable_truth_value", block.truth_raw)
|
|
460
|
+
continue
|
|
461
|
+
if story_id in ambiguous:
|
|
462
|
+
yield _refused("folio_story_ambiguous", ambiguous[story_id])
|
|
463
|
+
continue
|
|
464
|
+
story = stories.get(story_id)
|
|
465
|
+
if story is None:
|
|
466
|
+
yield _refused("story_not_in_folio", None)
|
|
467
|
+
continue
|
|
468
|
+
if position >= len(story.conclusions_nl):
|
|
469
|
+
yield _refused(
|
|
470
|
+
"position_out_of_range",
|
|
471
|
+
f"story {story_id} has only {len(story.conclusions_nl)} "
|
|
472
|
+
"FOLIO.csv conclusion(s)")
|
|
473
|
+
continue
|
|
474
|
+
folio_truth_value = story.truth_values[position]
|
|
475
|
+
if folio_truth_value != truth_value:
|
|
476
|
+
yield _refused(
|
|
477
|
+
"truth_value_mismatch",
|
|
478
|
+
f"P-FOLIO.csv={truth_value!r} FOLIO.csv={folio_truth_value!r}")
|
|
479
|
+
continue
|
|
480
|
+
|
|
481
|
+
example_id = f"pfolio:{story_id}:{position}"
|
|
482
|
+
proof_steps = [
|
|
483
|
+
{
|
|
484
|
+
"step_id": step["step_id"],
|
|
485
|
+
"premises_used": step["premises_used"],
|
|
486
|
+
"derivation": step["derivation"],
|
|
487
|
+
"derivation_corrected": step["derivation_corrected"],
|
|
488
|
+
"inference_rule": step["inference_rule"],
|
|
489
|
+
}
|
|
490
|
+
for step in block.steps
|
|
491
|
+
]
|
|
492
|
+
meta = {
|
|
493
|
+
"story_id": story_id,
|
|
494
|
+
"position": position,
|
|
495
|
+
"folio_comments": story.comments,
|
|
496
|
+
"folio_verified_by_prover": story.verified_by_prover,
|
|
497
|
+
"proof_steps": proof_steps,
|
|
498
|
+
}
|
|
499
|
+
yield ("ok", DatasetExample(
|
|
500
|
+
id=example_id,
|
|
501
|
+
nl_premises=story.premises_nl,
|
|
502
|
+
fol_premises=story.premises_fol,
|
|
503
|
+
nl_conclusion=story.conclusions_nl[position],
|
|
504
|
+
fol_conclusion=story.conclusions_fol[position],
|
|
505
|
+
label=folio_truth_value,
|
|
506
|
+
known_bad=example_id in known_bad_ids,
|
|
507
|
+
meta=meta,
|
|
508
|
+
))
|
|
509
|
+
|
|
510
|
+
|
|
511
|
+
def load_pfolio(pfolio_path: Union[str, Path], folio_path: Union[str, Path], *,
|
|
512
|
+
known_bad_ids: FrozenSet[str] = frozenset(),
|
|
513
|
+
on_refused: str = "raise") -> Iterator[DatasetExample]:
|
|
514
|
+
"""Stream :class:`~unicode_logic_kit.eval.datasets.DatasetExample` from a
|
|
515
|
+
local pair of P-FOLIO/FOLIO CSV files, joined per the module docstring.
|
|
516
|
+
|
|
517
|
+
Field mapping: ``nl_premises``/``fol_premises`` = the joined story's
|
|
518
|
+
``FOLIO.csv`` premises (repeated across every conclusion of that story,
|
|
519
|
+
the same "several consecutive examples share identical premises"
|
|
520
|
+
situation :mod:`~unicode_logic_kit.eval.datasets.folio`'s docstring
|
|
521
|
+
documents for its own story grouping); ``nl_conclusion``/
|
|
522
|
+
``fol_conclusion`` = that story's ``FOLIO.csv`` conclusion/FOL at the
|
|
523
|
+
block's position; ``label`` = the (agreeing) truth value, one of
|
|
524
|
+
:data:`PFOLIO_TRUTH_VALUES`. ``meta`` carries ``story_id``, ``position``
|
|
525
|
+
(both 0-based/int), ``folio_comments``/``folio_verified_by_prover``
|
|
526
|
+
(``FOLIO.csv``'s own free-text columns, kept at STORY granularity —
|
|
527
|
+
verified too sparse and inconsistently shaped to align per-conclusion,
|
|
528
|
+
see the module docstring), and ``proof_steps``: the block's derivation
|
|
529
|
+
rows verbatim, each ``{"step_id", "premises_used", "derivation",
|
|
530
|
+
"derivation_corrected", "inference_rule"}`` (``Derivation`` and
|
|
531
|
+
``Derivation - Corrected`` kept as two distinct keys, never merged;
|
|
532
|
+
``premises_used`` kept as one raw string, never parsed into indices —
|
|
533
|
+
see the module docstring for why).
|
|
534
|
+
|
|
535
|
+
Args:
|
|
536
|
+
pfolio_path: local path to ``P-FOLIO.csv``.
|
|
537
|
+
folio_path: local path to P-FOLIO's own bundled ``FOLIO.csv`` (NOT
|
|
538
|
+
the JSONL file :func:`~unicode_logic_kit.eval.datasets.folio.load_folio`
|
|
539
|
+
reads — see the module docstring).
|
|
540
|
+
known_bad_ids: ids (``f"pfolio:{story_id}:{position}"``) whose gold
|
|
541
|
+
annotation is known to be broken. Every yielded example with a
|
|
542
|
+
matching id gets ``known_bad=True``.
|
|
543
|
+
on_refused: what to do with a block this adapter cannot safely join
|
|
544
|
+
(see the module docstring's refusal reasons). ``"raise"``
|
|
545
|
+
(default): raise :class:`ValueError` naming the story, position
|
|
546
|
+
and reason at the first one reached — the loud, no-guessing
|
|
547
|
+
default. ``"skip"``: drop it from the yielded stream instead
|
|
548
|
+
(use :func:`pfolio_refusals` to see what was dropped and why —
|
|
549
|
+
nothing is silently lost either way, only excluded from this
|
|
550
|
+
call's own output).
|
|
551
|
+
|
|
552
|
+
Yields:
|
|
553
|
+
One :class:`~unicode_logic_kit.eval.datasets.DatasetExample` per
|
|
554
|
+
successfully-joined P-FOLIO block, in file order.
|
|
555
|
+
|
|
556
|
+
Raises:
|
|
557
|
+
ValueError: ``on_refused`` is not ``"raise"``/``"skip"``; either CSV
|
|
558
|
+
file has the wrong header or a malformed row (structural
|
|
559
|
+
corruption, refused unconditionally regardless of
|
|
560
|
+
``on_refused``); or, with ``on_refused="raise"``, the first
|
|
561
|
+
block that cannot be safely joined.
|
|
562
|
+
FileNotFoundError: either path does not exist.
|
|
563
|
+
"""
|
|
564
|
+
if on_refused not in ("raise", "skip"):
|
|
565
|
+
raise ValueError(
|
|
566
|
+
f"pfolio: on_refused must be 'raise' or 'skip', got {on_refused!r}")
|
|
567
|
+
|
|
568
|
+
for kind, payload in _iter_joined(pfolio_path, folio_path, known_bad_ids):
|
|
569
|
+
if kind == "ok":
|
|
570
|
+
yield payload
|
|
571
|
+
continue
|
|
572
|
+
if on_refused == "raise":
|
|
573
|
+
raise ValueError(
|
|
574
|
+
f"pfolio: story {payload['story_id']} conclusion #"
|
|
575
|
+
f"{payload['position']} (P-FOLIO.csv row {payload['row_no']}) "
|
|
576
|
+
f"refused ({payload['reason']}): {payload['detail']}")
|
|
577
|
+
# on_refused == "skip": drop it, recoverable via pfolio_refusals().
|
|
578
|
+
|
|
579
|
+
|
|
580
|
+
def pfolio_refusals(pfolio_path: Union[str, Path],
|
|
581
|
+
folio_path: Union[str, Path]) -> List[dict]:
|
|
582
|
+
"""Every block :func:`load_pfolio` would refuse, with why — a read-only
|
|
583
|
+
diagnostic pass, never raising for a refusal itself (only for the same
|
|
584
|
+
structural-corruption cases :func:`load_pfolio` always raises for).
|
|
585
|
+
|
|
586
|
+
Returns one ``{"story_id", "position", "reason", "detail", "row_no"}``
|
|
587
|
+
dict per refused block, in file order (see :func:`_iter_joined`'s
|
|
588
|
+
docstring for the field meanings, and the module docstring's "This
|
|
589
|
+
adapter's join, and what it refuses" section for what each ``reason``
|
|
590
|
+
means). Empty iff every block in ``pfolio_path`` joins cleanly.
|
|
591
|
+
"""
|
|
592
|
+
return [payload for kind, payload
|
|
593
|
+
in _iter_joined(pfolio_path, folio_path, frozenset())
|
|
594
|
+
if kind == "refused"]
|