unicode-logic-kit 0.31.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- unicode_logic_kit/__init__.py +385 -0
- unicode_logic_kit/__main__.py +520 -0
- unicode_logic_kit/_deadline.py +219 -0
- unicode_logic_kit/ace/__init__.py +126 -0
- unicode_logic_kit/ace/_align.py +135 -0
- unicode_logic_kit/ace/chem_lexicon.py +128 -0
- unicode_logic_kit/ace/drs_reader.py +570 -0
- unicode_logic_kit/ace/mapping.py +666 -0
- unicode_logic_kit/ace/reverse_modal.py +138 -0
- unicode_logic_kit/ace/runner.py +551 -0
- unicode_logic_kit/ace/translate.py +452 -0
- unicode_logic_kit/ace/verbalize.py +1070 -0
- unicode_logic_kit/api.py +1284 -0
- unicode_logic_kit/atp/__init__.py +177 -0
- unicode_logic_kit/atp/_ascii_names.py +113 -0
- unicode_logic_kit/atp/_html.py +72 -0
- unicode_logic_kit/atp/_substructural_input.py +228 -0
- unicode_logic_kit/atp/_tff_problem.py +715 -0
- unicode_logic_kit/atp/_tptp_problem.py +1111 -0
- unicode_logic_kit/atp/_writer_support.py +289 -0
- unicode_logic_kit/atp/clingo_backend.py +1180 -0
- unicode_logic_kit/atp/cvc5_backend.py +1385 -0
- unicode_logic_kit/atp/eprover_backend.py +732 -0
- unicode_logic_kit/atp/finite_domain.py +1055 -0
- unicode_logic_kit/atp/fitch.py +1547 -0
- unicode_logic_kit/atp/fitch_search.py +551 -0
- unicode_logic_kit/atp/hets_backend.py +339 -0
- unicode_logic_kit/atp/hybrid_down.py +120 -0
- unicode_logic_kit/atp/incremental.py +250 -0
- unicode_logic_kit/atp/kripke_enum.py +741 -0
- unicode_logic_kit/atp/lambek.py +436 -0
- unicode_logic_kit/atp/leo3_backend.py +332 -0
- unicode_logic_kit/atp/linear.py +738 -0
- unicode_logic_kit/atp/lj.py +705 -0
- unicode_logic_kit/atp/logic_backends.py +566 -0
- unicode_logic_kit/atp/ltl_tableau.py +1084 -0
- unicode_logic_kit/atp/minizinc_backend.py +1402 -0
- unicode_logic_kit/atp/modal_tableau.py +1382 -0
- unicode_logic_kit/atp/nanocop_backend.py +410 -0
- unicode_logic_kit/atp/portfolio.py +489 -0
- unicode_logic_kit/atp/protocol.py +1803 -0
- unicode_logic_kit/atp/prover9_entailment.py +1153 -0
- unicode_logic_kit/atp/resolution.py +1376 -0
- unicode_logic_kit/atp/resolution_check.py +1114 -0
- unicode_logic_kit/atp/sequent.py +1050 -0
- unicode_logic_kit/atp/tableau.py +921 -0
- unicode_logic_kit/atp/tableau_check.py +543 -0
- unicode_logic_kit/atp/tptp_ncl.py +811 -0
- unicode_logic_kit/atp/tptp_tff.py +1546 -0
- unicode_logic_kit/atp/tstp.py +1333 -0
- unicode_logic_kit/atp/tstp_check.py +1096 -0
- unicode_logic_kit/atp/twee_backend.py +236 -0
- unicode_logic_kit/atp/twee_check.py +711 -0
- unicode_logic_kit/atp/twee_entailment.py +953 -0
- unicode_logic_kit/atp/vampire_entailment.py +540 -0
- unicode_logic_kit/atp/z3_arith.py +470 -0
- unicode_logic_kit/atp/z3_equivalence.py +36 -0
- unicode_logic_kit/atp/z3_fuzzy.py +362 -0
- unicode_logic_kit/atp/z3_input.py +500 -0
- unicode_logic_kit/atp/z3_models.py +208 -0
- unicode_logic_kit/chem/__init__.py +88 -0
- unicode_logic_kit/chem/_naming.py +284 -0
- unicode_logic_kit/chem/cache.py +185 -0
- unicode_logic_kit/chem/interop.py +244 -0
- unicode_logic_kit/chem/mol.py +525 -0
- unicode_logic_kit/chem/signature.py +112 -0
- unicode_logic_kit/comorphism.py +497 -0
- unicode_logic_kit/dl/__init__.py +384 -0
- unicode_logic_kit/dl/classification.py +227 -0
- unicode_logic_kit/dl/concepts.py +632 -0
- unicode_logic_kit/dl/datatypes.py +818 -0
- unicode_logic_kit/dl/owl_functional.py +2433 -0
- unicode_logic_kit/dl/owl_manchester.py +1637 -0
- unicode_logic_kit/dl/owl_reasoner.py +790 -0
- unicode_logic_kit/dl/parser.py +391 -0
- unicode_logic_kit/dl/tableau.py +4048 -0
- unicode_logic_kit/dl/translate.py +2704 -0
- unicode_logic_kit/drt/__init__.py +94 -0
- unicode_logic_kit/drt/export.py +179 -0
- unicode_logic_kit/drt/nodes.py +506 -0
- unicode_logic_kit/drt/parser.py +965 -0
- unicode_logic_kit/drt/resolve.py +195 -0
- unicode_logic_kit/drt/reverse.py +175 -0
- unicode_logic_kit/eval/__init__.py +106 -0
- unicode_logic_kit/eval/batch.py +382 -0
- unicode_logic_kit/eval/canonical.py +663 -0
- unicode_logic_kit/eval/chem_batch.py +606 -0
- unicode_logic_kit/eval/converses.py +200 -0
- unicode_logic_kit/eval/datasets/__init__.py +136 -0
- unicode_logic_kit/eval/datasets/_base.py +263 -0
- unicode_logic_kit/eval/datasets/_proofwriter_proof.py +422 -0
- unicode_logic_kit/eval/datasets/c3po.py +678 -0
- unicode_logic_kit/eval/datasets/folio.py +158 -0
- unicode_logic_kit/eval/datasets/fracas.py +418 -0
- unicode_logic_kit/eval/datasets/groves.py +191 -0
- unicode_logic_kit/eval/datasets/logicbench.py +467 -0
- unicode_logic_kit/eval/datasets/logicnli.py +303 -0
- unicode_logic_kit/eval/datasets/malls.py +133 -0
- unicode_logic_kit/eval/datasets/pfolio.py +594 -0
- unicode_logic_kit/eval/datasets/pmb.py +242 -0
- unicode_logic_kit/eval/datasets/prontoqa.py +611 -0
- unicode_logic_kit/eval/datasets/proofwriter.py +1431 -0
- unicode_logic_kit/eval/datasets/proverqa.py +674 -0
- unicode_logic_kit/eval/datasets/willow.py +478 -0
- unicode_logic_kit/eval/equivalence.py +466 -0
- unicode_logic_kit/eval/exercise_gen.py +533 -0
- unicode_logic_kit/eval/explain.py +791 -0
- unicode_logic_kit/eval/generality.py +750 -0
- unicode_logic_kit/eval/metric_hf.py +458 -0
- unicode_logic_kit/eval/predicate_match.py +343 -0
- unicode_logic_kit/eval/theory_check.py +1170 -0
- unicode_logic_kit/eval/validate.py +306 -0
- unicode_logic_kit/fol/__init__.py +177 -0
- unicode_logic_kit/fol/_atom_keys.py +510 -0
- unicode_logic_kit/fol/_fol_nodes.py +3586 -0
- unicode_logic_kit/fol/_free_parameters.py +105 -0
- unicode_logic_kit/fol/_ho_nodes.py +448 -0
- unicode_logic_kit/fol/_hybrid_nodes.py +308 -0
- unicode_logic_kit/fol/_identifiers.py +1091 -0
- unicode_logic_kit/fol/_lambek_nodes.py +112 -0
- unicode_logic_kit/fol/_linear_nodes.py +352 -0
- unicode_logic_kit/fol/_modal_nodes.py +1467 -0
- unicode_logic_kit/fol/_msfl_nodes.py +2196 -0
- unicode_logic_kit/fol/_numeral_symbols.py +231 -0
- unicode_logic_kit/fol/_so_nodes.py +200 -0
- unicode_logic_kit/fol/_symbol_names.py +81 -0
- unicode_logic_kit/fol/_team_nodes.py +181 -0
- unicode_logic_kit/fol/_tptp_symbols.py +551 -0
- unicode_logic_kit/fol/_truth_constants.py +117 -0
- unicode_logic_kit/fol/casl_export.py +1135 -0
- unicode_logic_kit/fol/casl_import.py +929 -0
- unicode_logic_kit/fol/derivation.py +367 -0
- unicode_logic_kit/fol/dialect_detect.py +70 -0
- unicode_logic_kit/fol/dialect_repair.py +537 -0
- unicode_logic_kit/fol/frames.py +637 -0
- unicode_logic_kit/fol/grammars/terminals.lark +31 -0
- unicode_logic_kit/fol/lambda_tools.py +297 -0
- unicode_logic_kit/fol/latex_input.py +429 -0
- unicode_logic_kit/fol/modal_translation.py +944 -0
- unicode_logic_kit/fol/msflparser.py +1033 -0
- unicode_logic_kit/fol/naming.py +422 -0
- unicode_logic_kit/fol/nodes.py +241 -0
- unicode_logic_kit/fol/normalforms.py +492 -0
- unicode_logic_kit/fol/pal.py +287 -0
- unicode_logic_kit/fol/prolog_export.py +566 -0
- unicode_logic_kit/fol/prolog_input.py +505 -0
- unicode_logic_kit/fol/prover9_input.py +1325 -0
- unicode_logic_kit/fol/qml.py +1760 -0
- unicode_logic_kit/fol/qmltp_input.py +525 -0
- unicode_logic_kit/fol/sanitize.py +221 -0
- unicode_logic_kit/fol/serialize.py +79 -0
- unicode_logic_kit/fol/signature.py +1290 -0
- unicode_logic_kit/fol/simplify_check.py +544 -0
- unicode_logic_kit/fol/spans.py +594 -0
- unicode_logic_kit/fol/tptp_input.py +1503 -0
- unicode_logic_kit/fol/tptp_repair.py +941 -0
- unicode_logic_kit/fol/unification.py +157 -0
- unicode_logic_kit/fol/verbalize.py +263 -0
- unicode_logic_kit/hets/__init__.py +163 -0
- unicode_logic_kit/hets/bridge.py +142 -0
- unicode_logic_kit/hets/client.py +748 -0
- unicode_logic_kit/hets/docker.py +420 -0
- unicode_logic_kit/hets/dol.py +712 -0
- unicode_logic_kit/hets/haskell_json.py +355 -0
- unicode_logic_kit/hets/owl_backend.py +794 -0
- unicode_logic_kit/hets/owl_cli.py +598 -0
- unicode_logic_kit/hets/symbols.py +512 -0
- unicode_logic_kit/hol/__init__.py +140 -0
- unicode_logic_kit/hol/_ho_common.py +323 -0
- unicode_logic_kit/hol/_isabelle_binders.py +125 -0
- unicode_logic_kit/hol/classical.py +812 -0
- unicode_logic_kit/hol/deepshallow/__init__.py +45 -0
- unicode_logic_kit/hol/deepshallow/_common.py +177 -0
- unicode_logic_kit/hol/deepshallow/conditional.py +225 -0
- unicode_logic_kit/hol/deepshallow/intuitionistic.py +181 -0
- unicode_logic_kit/hol/deepshallow/modal.py +217 -0
- unicode_logic_kit/hol/deepshallow/qml.py +406 -0
- unicode_logic_kit/hol/deepshallow/relevant.py +206 -0
- unicode_logic_kit/hol/free.py +753 -0
- unicode_logic_kit/hol/goedel.py +336 -0
- unicode_logic_kit/hol/ho_modal.py +1743 -0
- unicode_logic_kit/hol/intuitionistic.py +403 -0
- unicode_logic_kit/hol/isabelle_conditional.py +593 -0
- unicode_logic_kit/hol/isabelle_modal.py +1908 -0
- unicode_logic_kit/hol/isabelle_relevant.py +412 -0
- unicode_logic_kit/hol/isabelle_runner.py +1147 -0
- unicode_logic_kit/hol/isabelle_substructural.py +884 -0
- unicode_logic_kit/hol/lean.py +1018 -0
- unicode_logic_kit/hol/manyvalued.py +921 -0
- unicode_logic_kit/hol/secondorder.py +687 -0
- unicode_logic_kit/hol/thf_modal.py +941 -0
- unicode_logic_kit/hol/thirdorder.py +397 -0
- unicode_logic_kit/ilp/__init__.py +89 -0
- unicode_logic_kit/ilp/readback.py +389 -0
- unicode_logic_kit/ilp/separation.py +153 -0
- unicode_logic_kit/ilp/task.py +730 -0
- unicode_logic_kit/logic.py +163 -0
- unicode_logic_kit/mcp/__init__.py +28 -0
- unicode_logic_kit/mcp/__main__.py +5 -0
- unicode_logic_kit/mcp/chem_tools.py +1031 -0
- unicode_logic_kit/mcp/server.py +2453 -0
- unicode_logic_kit/mcp/syntax_spec.py +681 -0
- unicode_logic_kit/prob/__init__.py +53 -0
- unicode_logic_kit/prob/_bdd.py +225 -0
- unicode_logic_kit/prob/_column_gen.py +668 -0
- unicode_logic_kit/prob/distribution.py +686 -0
- unicode_logic_kit/prob/nilsson.py +470 -0
- unicode_logic_kit/py.typed +0 -0
- unicode_logic_kit/semantics/__init__.py +137 -0
- unicode_logic_kit/semantics/_modal_reject.py +156 -0
- unicode_logic_kit/semantics/action_models.py +466 -0
- unicode_logic_kit/semantics/asp_models.py +1200 -0
- unicode_logic_kit/semantics/conditional.py +580 -0
- unicode_logic_kit/semantics/dynamic_epistemic.py +95 -0
- unicode_logic_kit/semantics/free_logic.py +913 -0
- unicode_logic_kit/semantics/fuzzy.py +384 -0
- unicode_logic_kit/semantics/fuzzy_kripke.py +442 -0
- unicode_logic_kit/semantics/intuitionistic.py +581 -0
- unicode_logic_kit/semantics/kripke.py +1139 -0
- unicode_logic_kit/semantics/manyvalued.py +580 -0
- unicode_logic_kit/semantics/matrix.py +342 -0
- unicode_logic_kit/semantics/model_eval.py +1135 -0
- unicode_logic_kit/semantics/modelfinder.py +1036 -0
- unicode_logic_kit/semantics/nonmonotonic.py +372 -0
- unicode_logic_kit/semantics/relevant.py +331 -0
- unicode_logic_kit/semantics/secondorder.py +657 -0
- unicode_logic_kit/semantics/structures.py +352 -0
- unicode_logic_kit/semantics/tarski.py +975 -0
- unicode_logic_kit/semantics/team.py +315 -0
- unicode_logic_kit/semantics/team_translation.py +416 -0
- unicode_logic_kit/semantics/thirdorder.py +358 -0
- unicode_logic_kit/semantics/tnorm.py +85 -0
- unicode_logic_kit/semantics/truthtable.py +201 -0
- unicode_logic_kit-0.31.0.dist-info/METADATA +333 -0
- unicode_logic_kit-0.31.0.dist-info/RECORD +237 -0
- unicode_logic_kit-0.31.0.dist-info/WHEEL +4 -0
- unicode_logic_kit-0.31.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,674 @@
|
|
|
1
|
+
"""Adapter for the ProverQA dataset (Qi et al., "Large Language Models Meet
|
|
2
|
+
Symbolic Provers for Logical Reasoning Evaluation", ICLR 2025; dataset built
|
|
3
|
+
by the ProverGen framework) — local JSONL only, no network access.
|
|
4
|
+
|
|
5
|
+
Source and verified schema
|
|
6
|
+
---------------------------
|
|
7
|
+
Source verified 2026-08-12 directly against
|
|
8
|
+
https://huggingface.co/datasets/opendatalab/ProverQA (repo file listing via
|
|
9
|
+
``https://huggingface.co/api/datasets/opendatalab/ProverQA``, record schema
|
|
10
|
+
via the raw files fetched from
|
|
11
|
+
``https://huggingface.co/datasets/opendatalab/ProverQA/resolve/main/<path>``,
|
|
12
|
+
cross-checked against the field descriptions and JSON example in the
|
|
13
|
+
repository's own ``README.md``). Code for the generation pipeline (not
|
|
14
|
+
consulted for this loader, provided for context only):
|
|
15
|
+
https://github.com/opendatalab/ProverGen.
|
|
16
|
+
|
|
17
|
+
The repository has exactly four data files, in TWO INCOMPATIBLE shapes:
|
|
18
|
+
|
|
19
|
+
* ``dev/easy.json``, ``dev/medium.json``, ``dev/hard.json`` — the paper's
|
|
20
|
+
1,500-instance evaluation benchmark (500 per difficulty tier: easy = 1-2
|
|
21
|
+
reasoning steps, medium = 3-5, hard = 6-9), each a JSON ARRAY of flat
|
|
22
|
+
objects. Verified by downloading and inspecting all 1,500 rows: every
|
|
23
|
+
single row has EXACTLY these 8 keys, no more, no fewer —
|
|
24
|
+
|
|
25
|
+
* ``"id"`` — ``int``, 0-based, unique WITHIN one tier file
|
|
26
|
+
(0..499 in each of the three files) but NOT globally unique — the same
|
|
27
|
+
id appears in all three tiers. A caller loading more than one tier must
|
|
28
|
+
pass a distinguishing ``tier`` to :func:`load_proverqa` (see below) or
|
|
29
|
+
the resulting ids collide.
|
|
30
|
+
* ``"options"`` — ``list[str]``, always exactly 3 entries in every
|
|
31
|
+
verified row: ``["A) True", "B) False", "C) Uncertain"]`` verbatim
|
|
32
|
+
(letter-and-text order never varies across the 1,500 rows checked).
|
|
33
|
+
* ``"answer"`` — ``str``, one of ``"A"``/``"B"``/``"C"``; verified
|
|
34
|
+
for all 1,500 rows that ``options[ord(answer) - ord("A")]`` starts with
|
|
35
|
+
``f"{answer})"`` — the letter always indexes its own option correctly.
|
|
36
|
+
* ``"question"`` — ``str``, the FULL interrogative prompt, e.g.
|
|
37
|
+
``"Based on the above information, is the following statement true,
|
|
38
|
+
false, or uncertain? Brecken has never experienced heartbreak."`` —
|
|
39
|
+
UNLIKE FOLIO's ``conclusion`` field, this is NOT a bare declarative
|
|
40
|
+
sentence; the bare statement is not separately provided upstream, and
|
|
41
|
+
this adapter does not attempt to extract it with a regex heuristic (that
|
|
42
|
+
would be an invented field, not a verified one).
|
|
43
|
+
* ``"reasoning"`` — ``str``, a free-text chain-of-thought trace.
|
|
44
|
+
* ``"context"`` — ``str``, the natural-language premises,
|
|
45
|
+
concatenated. Verified equal to ``" ".join(nl2fol.keys())`` for 1,497 of
|
|
46
|
+
the 1,500 rows; the 3 exceptions (e.g. ``dev/easy.json`` id 361) have a
|
|
47
|
+
premise sentence repeated VERBATIM TWICE in ``context`` but only once in
|
|
48
|
+
``nl2fol`` — ``nl2fol`` is a JSON object keyed by sentence text, so a
|
|
49
|
+
literal duplicate premise silently collapses to one entry. This is an
|
|
50
|
+
upstream data quirk (verified present in the raw file), not introduced
|
|
51
|
+
by this adapter; it means ``fol_premises`` can have fewer entries than
|
|
52
|
+
``context`` has sentences for a handful of rows.
|
|
53
|
+
* ``"nl2fol"`` — ``dict[str, str]``, NL premise sentence -> its
|
|
54
|
+
gold FOL translation, insertion-ordered (Python/JSON preserve object key
|
|
55
|
+
order, and ``dict.keys()``/``dict.values()`` iterate in matched,
|
|
56
|
+
corresponding order for the same dict) — this is the exact
|
|
57
|
+
``nl_premises``/``fol_premises`` pairing this adapter uses.
|
|
58
|
+
* ``"conclusion_fol"`` — ``str``, the gold FOL for the statement embedded
|
|
59
|
+
in ``question``.
|
|
60
|
+
|
|
61
|
+
* ``train/provergen-5000.json`` — the paper's 5,000-instance fine-tuning
|
|
62
|
+
split, a JSON array of flat objects. Verified by downloading and
|
|
63
|
+
inspecting all 5,000 rows: every row has EXACTLY 4 keys —
|
|
64
|
+
``"system"``/``"instruction"``/``"input"``/``"output"``, an
|
|
65
|
+
instruction-tuning prompt/response pair (the context, question, and
|
|
66
|
+
options are embedded as free NATURAL-LANGUAGE TEXT inside ``instruction``,
|
|
67
|
+
and the gold reasoning/answer letter as free text inside ``output``, as a
|
|
68
|
+
JSON-in-a-string). It carries **NO** structured ``id``/``nl2fol``/
|
|
69
|
+
``conclusion_fol``/``answer``/``options`` fields at all — no FOL
|
|
70
|
+
annotation of any kind is recoverable from this file without re-parsing
|
|
71
|
+
free text. **This loader does NOT support the train split** — only
|
|
72
|
+
``dev/easy.json``/``dev/medium.json``/``dev/hard.json`` carry the
|
|
73
|
+
structured FOL fields :class:`~unicode_logic_kit.eval.datasets.DatasetExample`
|
|
74
|
+
needs.
|
|
75
|
+
|
|
76
|
+
CRITICAL notation finding (parse rate is 0%, verified, not a bug)
|
|
77
|
+
-------------------------------------------------------------------
|
|
78
|
+
Every one of the 17,342 real FOL strings in the three ``dev/*.json`` files
|
|
79
|
+
(every ``nl2fol`` value plus every ``conclusion_fol``, across all 1,500 rows)
|
|
80
|
+
was run through :func:`unicode_logic_kit.api.parse_any` as part of verifying
|
|
81
|
+
this adapter. **0 of 17,342 parsed** under any dialect. The cause is a
|
|
82
|
+
NOTATION mismatch, not a data-quality defect in ProverQA: ProverQA's gold
|
|
83
|
+
FOL uses lower-case ``snake_case`` predicate identifiers with underscores
|
|
84
|
+
(e.g. ``has_experienced_heartbreak``) and Capitalized proper-noun constants
|
|
85
|
+
(e.g. ``Brecken``) — the OPPOSITE convention from this kit's own grammar
|
|
86
|
+
(``unicode_logic_kit/fol/grammars/terminals.lark``: ``PREDICATE`` must start
|
|
87
|
+
uppercase with no underscore, e.g. ``Cat``; the constant/function ``NAME``
|
|
88
|
+
terminal must start lowercase with no underscore, e.g. ``tom`` — the
|
|
89
|
+
convention FOLIO's and MALLS's gold data already happen to follow, which is
|
|
90
|
+
WHY those two adapters' fixtures mostly parse and this one's does not).
|
|
91
|
+
ProverQA's own README states its FOL was "validated through automated
|
|
92
|
+
symbolic provers (Prover9)" — a different tool with a different accepted
|
|
93
|
+
surface syntax than this kit's parser.
|
|
94
|
+
|
|
95
|
+
This adapter resolves the mismatch with a DEDICATED IMPORT-TIME GRAMMAR
|
|
96
|
+
rather than a lexical rewrite: :data:`_PROVERQA_GRAMMAR` parses the dataset's
|
|
97
|
+
own notation (operators verified against the real corpus), and the resulting
|
|
98
|
+
tree is converted into kit nodes with identifiers renamed into the kit's
|
|
99
|
+
convention (:class:`_NameConverter` — injective per namespace, constants
|
|
100
|
+
that would become variables refused; see the dialect comment block above the
|
|
101
|
+
grammar). With the default ``convert_fol=True``, ``fol_premises`` /
|
|
102
|
+
``fol_conclusion`` therefore hold KIT-notation strings that parse under
|
|
103
|
+
``api.parse_any``, while ``meta`` keeps the verbatim upstream strings
|
|
104
|
+
(``original_fol_premises``/``original_fol_conclusion``) and the
|
|
105
|
+
changed-names mapping (``fol_name_mapping``) — the gold data remains fully
|
|
106
|
+
recoverable and every rename is on the record. ``convert_fol=False``
|
|
107
|
+
restores the raw pass-through, whose 0% kit-parse rate is pinned by
|
|
108
|
+
``tests/test_datasets_proverqa.py``. :func:`solve_example` then decides a
|
|
109
|
+
converted example end-to-end (premises ⊨ conclusion → ``"A"``, premises ⊨
|
|
110
|
+
¬conclusion → ``"B"``, else ``"C"``) against the gold ``answer``.
|
|
111
|
+
|
|
112
|
+
Also observed, and left as-is (not fixed): ``dev/easy.json`` row ``id=7``
|
|
113
|
+
uses the predicate ``resolves_conflict_peacefully`` (singular "conflict") in
|
|
114
|
+
one ``nl2fol`` rule but ``resolves_conflicts_peacefully`` (plural
|
|
115
|
+
"conflicts") in ``conclusion_fol`` — an internal predicate-name
|
|
116
|
+
inconsistency verified present in the raw upstream file itself, not
|
|
117
|
+
introduced by this adapter. It is exactly the kind of defect
|
|
118
|
+
:func:`~unicode_logic_kit.eval.datasets.audit_examples` exists to surface (as
|
|
119
|
+
an ``arity_conflict``/mismatched-signature symptom) IF the formulas parsed
|
|
120
|
+
at all — here it is masked by the notation mismatch above, since neither
|
|
121
|
+
formula parses in the first place.
|
|
122
|
+
|
|
123
|
+
Upstream distribution format and this loader
|
|
124
|
+
-----------------------------------------------
|
|
125
|
+
Each ``dev/<tier>.json`` file is a JSON ARRAY (like MALLS's upstream
|
|
126
|
+
distribution), NOT JSONL. This loader, like
|
|
127
|
+
:func:`~unicode_logic_kit.eval.datasets.folio.load_folio` and
|
|
128
|
+
:func:`~unicode_logic_kit.eval.datasets.malls.load_malls`, reads local JSONL
|
|
129
|
+
(one JSON object per line) for a uniform, streaming-friendly adapter surface
|
|
130
|
+
across this subpackage — convert an upstream ``dev/<tier>.json`` file to
|
|
131
|
+
JSONL first (e.g. ``jq -c '.[]' dev/easy.json > proverqa_easy.jsonl``) before
|
|
132
|
+
calling :func:`load_proverqa`.
|
|
133
|
+
|
|
134
|
+
Because the tier (easy/medium/hard) is which FILE a row came from, not a
|
|
135
|
+
field inside the row, :func:`load_proverqa` takes an optional ``tier``
|
|
136
|
+
keyword purely so the loader can (a) namespace ids so multiple tiers can be
|
|
137
|
+
loaded into one corpus without id collisions (see the ``"id"`` bullet
|
|
138
|
+
above), and (b) record the tier in ``meta``. It is caller-supplied, not
|
|
139
|
+
inferred from the file content — passing the wrong ``tier`` for a given file
|
|
140
|
+
is a caller error this loader cannot detect.
|
|
141
|
+
|
|
142
|
+
Field mapping onto :class:`~unicode_logic_kit.eval.datasets.DatasetExample`
|
|
143
|
+
(a deliberate design choice for the fields the source has no direct
|
|
144
|
+
equivalent for, exactly like :mod:`~unicode_logic_kit.eval.datasets.malls`
|
|
145
|
+
documents its own mapping choices):
|
|
146
|
+
|
|
147
|
+
* ``id``: ``f"proverqa:{tier}:{record['id']}"`` when ``tier`` is given,
|
|
148
|
+
else ``f"proverqa:{record['id']}"`` (see the id-collision note above).
|
|
149
|
+
Falls back to a positional ``f"proverqa:{tier or 'untiered'}:pos{line_no}"``
|
|
150
|
+
ONLY if the record has no ``"id"`` key at all (defensive; never observed
|
|
151
|
+
in verified real data — every one of the 1,500 rows has it).
|
|
152
|
+
* ``nl_premises`` / ``fol_premises``: ``tuple(nl2fol.keys())`` /
|
|
153
|
+
``tuple(nl2fol.values())`` — see the ``"nl2fol"`` bullet above for why
|
|
154
|
+
this pairing is safe. ``()`` if ``"nl2fol"`` is absent or not a mapping.
|
|
155
|
+
* ``nl_conclusion``: the verbatim ``"question"`` string (the interrogative
|
|
156
|
+
prompt, NOT a bare declarative — see the ``"question"`` bullet above).
|
|
157
|
+
* ``fol_conclusion``: the verbatim ``"conclusion_fol"`` string.
|
|
158
|
+
* ``label``: the verbatim ``"answer"`` letter (``"A"``/``"B"``/``"C"``) —
|
|
159
|
+
kept as the dataset's own raw vocabulary rather than resolved to
|
|
160
|
+
``"True"``/``"False"``/``"Uncertain"`` text, so nothing is synthesised
|
|
161
|
+
that is not literally the ``"answer"`` field; a caller who wants the
|
|
162
|
+
semantic text can resolve it themselves from ``meta["options"]``.
|
|
163
|
+
* ``meta``: every other record key verbatim (``"options"``, ``"context"``,
|
|
164
|
+
``"reasoning"``, and any future/unrecognised key), plus ``"line_no"`` and
|
|
165
|
+
(when given) ``"tier"``.
|
|
166
|
+
|
|
167
|
+
License
|
|
168
|
+
-------
|
|
169
|
+
**UNSPECIFIED** by the authors — verified against both the Hugging Face
|
|
170
|
+
dataset card's metadata (``cardData`` has no ``license`` key at all) and the
|
|
171
|
+
README's own "License and Ethics" section, which states only that the
|
|
172
|
+
INPUT SOURCES used to build the dataset comply with their own licenses
|
|
173
|
+
("MIT for names, WordNet for keywords") — that is a statement about the
|
|
174
|
+
dataset's INGREDIENTS, not a license grant for the ProverQA data itself. No
|
|
175
|
+
CC/MIT/Apache/etc. license is declared for the dataset artifact. Treat as
|
|
176
|
+
all-rights-reserved / contact the authors before any redistribution beyond
|
|
177
|
+
the kind of small local research fixture this module's own test fixture is.
|
|
178
|
+
This loader itself never downloads or redistributes ProverQA data — it only
|
|
179
|
+
reads a LOCAL file the caller already obtained.
|
|
180
|
+
|
|
181
|
+
Citation (from the repository's ``README.md``)::
|
|
182
|
+
|
|
183
|
+
@inproceedings{qi2025large,
|
|
184
|
+
title={Large Language Models Meet Symbolic Provers for Logical
|
|
185
|
+
Reasoning Evaluation},
|
|
186
|
+
author={Chengwen Qi and Ren Ma and Bowen Li and He Du and
|
|
187
|
+
Binyuan Hui and Jinwang Wu and Yuanjun Laili and Conghui He},
|
|
188
|
+
booktitle={The Thirteenth International Conference on Learning
|
|
189
|
+
Representations},
|
|
190
|
+
year={2025},
|
|
191
|
+
url={https://openreview.net/forum?id=C25SgeXWjE}
|
|
192
|
+
}
|
|
193
|
+
|
|
194
|
+
This module never downloads anything — obtain and convert the data yourself
|
|
195
|
+
and pass its local JSONL path to :func:`load_proverqa`.
|
|
196
|
+
"""
|
|
197
|
+
|
|
198
|
+
import json
|
|
199
|
+
import re
|
|
200
|
+
from pathlib import Path
|
|
201
|
+
from typing import Dict, FrozenSet, Iterator, List, Optional, Sequence, Tuple, Union
|
|
202
|
+
|
|
203
|
+
from lark import Lark, Transformer
|
|
204
|
+
from lark.exceptions import LarkError, VisitError
|
|
205
|
+
|
|
206
|
+
from ...fol.nodes import (
|
|
207
|
+
Node, Atom, Not, And, Or, Xor, Implies, Iff, Quantifier,
|
|
208
|
+
Variable, Constant, free_variables,
|
|
209
|
+
)
|
|
210
|
+
from ._base import DatasetExample, _register_dataset_info
|
|
211
|
+
|
|
212
|
+
__all__ = [
|
|
213
|
+
"load_proverqa",
|
|
214
|
+
"parse_proverqa_formula", "convert_proverqa_formulas",
|
|
215
|
+
"solve_example",
|
|
216
|
+
]
|
|
217
|
+
|
|
218
|
+
_VALID_TIERS = ("easy", "medium", "hard")
|
|
219
|
+
|
|
220
|
+
_register_dataset_info(
|
|
221
|
+
"proverqa",
|
|
222
|
+
license=(
|
|
223
|
+
"UNSPECIFIED by the authors (no license key in the HF dataset card; "
|
|
224
|
+
"the README's 'License and Ethics' section only states that the "
|
|
225
|
+
"dataset's INPUT SOURCES — names/keywords — comply with their own "
|
|
226
|
+
"licenses, which is not a license grant for the ProverQA data "
|
|
227
|
+
"itself). Treat as all-rights-reserved until the authors clarify."
|
|
228
|
+
),
|
|
229
|
+
source_url="https://huggingface.co/datasets/opendatalab/ProverQA",
|
|
230
|
+
citation_hint=(
|
|
231
|
+
"Qi, Chengwen, et al. \"Large Language Models Meet Symbolic Provers "
|
|
232
|
+
"for Logical Reasoning Evaluation.\" ICLR 2025. "
|
|
233
|
+
"https://openreview.net/forum?id=C25SgeXWjE"
|
|
234
|
+
),
|
|
235
|
+
)
|
|
236
|
+
|
|
237
|
+
|
|
238
|
+
# --------------------------------------------------------------------------- #
|
|
239
|
+
# The ProverQA FOL dialect: an import-time grammar + conversion to kit ASTs
|
|
240
|
+
#
|
|
241
|
+
# ProverQA's gold FOL is honest first-order logic in a NOTATION this kit's own
|
|
242
|
+
# grammar refuses (snake_case predicates, Capitalized constants — see the
|
|
243
|
+
# module docstring's "CRITICAL notation finding"). Instead of lexically
|
|
244
|
+
# rewriting the strings, the loader parses them with THIS dedicated grammar
|
|
245
|
+
# (operators verified against the real corpus: ¬ ∧ ∨ ⊕ → ↔ ∀ ∃, standard
|
|
246
|
+
# precedence ¬ > ∧ > ∨ > ⊕ > → (right-assoc) > ↔; ProverQA parenthesises
|
|
247
|
+
# mixed nesting anyway) and converts the resulting tree into kit nodes,
|
|
248
|
+
# renaming identifiers into the kit's convention along the way:
|
|
249
|
+
#
|
|
250
|
+
# * predicate ``has_experienced_heartbreak`` → ``HasExperiencedHeartbreak``
|
|
251
|
+
# (underscore parts joined, each part's FIRST letter upcased, interior
|
|
252
|
+
# capitalisation preserved — minimal change, not a re-styling),
|
|
253
|
+
# * constant ``Brecken`` → ``brecken`` (first letter downcased; snake_case
|
|
254
|
+
# constants fold the same way with camelCase joints),
|
|
255
|
+
# * a term that lexes as a kit VARIABLE (``[a-z][0-9]*``) stays a variable.
|
|
256
|
+
#
|
|
257
|
+
# Two hard refusals keep the conversion honest: a constant whose converted
|
|
258
|
+
# form would lex as a VARIABLE (e.g. the constant ``A`` → ``a``) raises
|
|
259
|
+
# instead of silently changing quantification semantics, and the mapping is
|
|
260
|
+
# INJECTIVE per namespace across one whole example — two source names that
|
|
261
|
+
# would collapse onto one target (``p_a`` and ``pA`` → ``PA``) raise rather
|
|
262
|
+
# than merging distinct symbols. The loader stores the original strings and
|
|
263
|
+
# the (changed-entries-only) mapping in ``meta``, so nothing is hidden.
|
|
264
|
+
# --------------------------------------------------------------------------- #
|
|
265
|
+
|
|
266
|
+
_PROVERQA_GRAMMAR = r"""
|
|
267
|
+
?start: formula
|
|
268
|
+
?formula: iff
|
|
269
|
+
?iff: implies ("↔" implies)*
|
|
270
|
+
?implies: xor "→" implies -> implies
|
|
271
|
+
| xor
|
|
272
|
+
?xor: disj ("⊕" disj)*
|
|
273
|
+
?disj: conj ("∨" conj)*
|
|
274
|
+
?conj: unary ("∧" unary)*
|
|
275
|
+
?unary: "¬" unary -> neg
|
|
276
|
+
| "∀" IDENT unary -> forall
|
|
277
|
+
| "∃" IDENT unary -> exists
|
|
278
|
+
| atom
|
|
279
|
+
| "(" formula ")"
|
|
280
|
+
atom: IDENT "(" IDENT ("," IDENT)* ")"
|
|
281
|
+
IDENT: /[A-Za-z][A-Za-z0-9_]*/
|
|
282
|
+
%import common.WS
|
|
283
|
+
%ignore WS
|
|
284
|
+
"""
|
|
285
|
+
|
|
286
|
+
_KIT_PREDICATE_RE = re.compile(r"[A-Z][a-zA-Z0-9]*\Z")
|
|
287
|
+
_KIT_CONSTANT_RE = re.compile(r"[a-z][a-zA-Z0-9]*[a-zA-Z][a-zA-Z0-9]*\Z")
|
|
288
|
+
_KIT_VARIABLE_RE = re.compile(r"[a-z][0-9]*\Z")
|
|
289
|
+
|
|
290
|
+
|
|
291
|
+
class _NameConverter:
|
|
292
|
+
"""Kit-convention renaming with a per-namespace injectivity guarantee.
|
|
293
|
+
|
|
294
|
+
One instance spans ONE example (all premises + the conclusion), so a
|
|
295
|
+
predicate keeps the same converted name everywhere it occurs and two
|
|
296
|
+
distinct source names can never collapse onto one target. ``pred_map`` /
|
|
297
|
+
``const_map`` hold only the names that actually changed — exactly what
|
|
298
|
+
the loader records in ``meta["fol_name_mapping"]``.
|
|
299
|
+
"""
|
|
300
|
+
|
|
301
|
+
def __init__(self) -> None:
|
|
302
|
+
self.pred_map: Dict[str, str] = {}
|
|
303
|
+
self.const_map: Dict[str, str] = {}
|
|
304
|
+
self._final: Dict[str, Dict[str, str]] = {"predicate": {}, "constant": {}}
|
|
305
|
+
|
|
306
|
+
def _claim(self, namespace: str, source: str, target: str) -> str:
|
|
307
|
+
owner = self._final[namespace].setdefault(target, source)
|
|
308
|
+
if owner != source:
|
|
309
|
+
raise ValueError(
|
|
310
|
+
f"proverqa: {namespace} names {owner!r} and {source!r} would "
|
|
311
|
+
f"both convert to {target!r} — the renaming must stay "
|
|
312
|
+
"injective, so this example cannot be converted.")
|
|
313
|
+
return target
|
|
314
|
+
|
|
315
|
+
def predicate(self, name: str) -> str:
|
|
316
|
+
if _KIT_PREDICATE_RE.match(name):
|
|
317
|
+
return self._claim("predicate", name, name)
|
|
318
|
+
parts = [p for p in name.split("_") if p]
|
|
319
|
+
if not parts:
|
|
320
|
+
raise ValueError(f"proverqa: predicate {name!r} is underscores only.")
|
|
321
|
+
target = "".join(p[0].upper() + p[1:] for p in parts)
|
|
322
|
+
if not _KIT_PREDICATE_RE.match(target):
|
|
323
|
+
raise ValueError(
|
|
324
|
+
f"proverqa: predicate {name!r} does not convert to a legal "
|
|
325
|
+
f"kit predicate (got {target!r}).")
|
|
326
|
+
self._claim("predicate", name, target)
|
|
327
|
+
self.pred_map[name] = target
|
|
328
|
+
return target
|
|
329
|
+
|
|
330
|
+
def constant(self, name: str) -> str:
|
|
331
|
+
if _KIT_CONSTANT_RE.match(name):
|
|
332
|
+
return self._claim("constant", name, name)
|
|
333
|
+
parts = [p for p in name.split("_") if p]
|
|
334
|
+
if not parts:
|
|
335
|
+
raise ValueError(f"proverqa: constant {name!r} is underscores only.")
|
|
336
|
+
head = parts[0][0].lower() + parts[0][1:]
|
|
337
|
+
target = head + "".join(p[0].upper() + p[1:] for p in parts[1:])
|
|
338
|
+
if _KIT_VARIABLE_RE.match(target):
|
|
339
|
+
raise ValueError(
|
|
340
|
+
f"proverqa: constant {name!r} would convert to {target!r}, "
|
|
341
|
+
"which this kit lexes as a VARIABLE — refusing the silent "
|
|
342
|
+
"semantics change.")
|
|
343
|
+
if not _KIT_CONSTANT_RE.match(target):
|
|
344
|
+
raise ValueError(
|
|
345
|
+
f"proverqa: constant {name!r} does not convert to a legal "
|
|
346
|
+
f"kit constant (got {target!r}).")
|
|
347
|
+
self._claim("constant", name, target)
|
|
348
|
+
self.const_map[name] = target
|
|
349
|
+
return target
|
|
350
|
+
|
|
351
|
+
def mapping(self) -> Dict[str, Dict[str, str]]:
|
|
352
|
+
return {"predicates": dict(self.pred_map), "constants": dict(self.const_map)}
|
|
353
|
+
|
|
354
|
+
|
|
355
|
+
class _ProverQATransformer(Transformer):
|
|
356
|
+
"""Lark tree → kit :class:`~unicode_logic_kit.fol.nodes.Node`."""
|
|
357
|
+
|
|
358
|
+
def __init__(self, converter: _NameConverter) -> None:
|
|
359
|
+
super().__init__()
|
|
360
|
+
self._conv = converter
|
|
361
|
+
|
|
362
|
+
def _term(self, token) -> Node:
|
|
363
|
+
name = str(token)
|
|
364
|
+
if _KIT_VARIABLE_RE.match(name):
|
|
365
|
+
return Variable(name)
|
|
366
|
+
return Constant(self._conv.constant(name))
|
|
367
|
+
|
|
368
|
+
def _binder(self, token) -> Variable:
|
|
369
|
+
name = str(token)
|
|
370
|
+
if not _KIT_VARIABLE_RE.match(name):
|
|
371
|
+
raise ValueError(
|
|
372
|
+
f"proverqa: quantified variable {name!r} is not a kit "
|
|
373
|
+
"variable token ([a-z][0-9]*) — refusing to guess a rename "
|
|
374
|
+
"for a BOUND name.")
|
|
375
|
+
return Variable(name)
|
|
376
|
+
|
|
377
|
+
def atom(self, children) -> Node:
|
|
378
|
+
pred = self._conv.predicate(str(children[0]))
|
|
379
|
+
return Atom(pred, tuple(self._term(t) for t in children[1:]))
|
|
380
|
+
|
|
381
|
+
def neg(self, children) -> Node:
|
|
382
|
+
return Not(children[0])
|
|
383
|
+
|
|
384
|
+
def forall(self, children) -> Node:
|
|
385
|
+
return Quantifier("∀", self._binder(children[0]), children[1])
|
|
386
|
+
|
|
387
|
+
def exists(self, children) -> Node:
|
|
388
|
+
return Quantifier("∃", self._binder(children[0]), children[1])
|
|
389
|
+
|
|
390
|
+
def _fold(self, children, cls) -> Node:
|
|
391
|
+
node = children[0]
|
|
392
|
+
for right in children[1:]:
|
|
393
|
+
node = cls(node, right)
|
|
394
|
+
return node
|
|
395
|
+
|
|
396
|
+
def conj(self, children) -> Node:
|
|
397
|
+
return self._fold(children, And)
|
|
398
|
+
|
|
399
|
+
def disj(self, children) -> Node:
|
|
400
|
+
return self._fold(children, Or)
|
|
401
|
+
|
|
402
|
+
def xor(self, children) -> Node:
|
|
403
|
+
return self._fold(children, Xor)
|
|
404
|
+
|
|
405
|
+
def implies(self, children) -> Node:
|
|
406
|
+
if len(children) == 1:
|
|
407
|
+
return children[0]
|
|
408
|
+
return Implies(children[0], children[1])
|
|
409
|
+
|
|
410
|
+
def iff(self, children) -> Node:
|
|
411
|
+
return self._fold(children, Iff)
|
|
412
|
+
|
|
413
|
+
|
|
414
|
+
_proverqa_parser = Lark(_PROVERQA_GRAMMAR, parser="lalr")
|
|
415
|
+
|
|
416
|
+
|
|
417
|
+
def parse_proverqa_formula(text: str, *,
|
|
418
|
+
converter: Optional[_NameConverter] = None) -> Node:
|
|
419
|
+
"""Parse ONE formula in ProverQA's FOL notation into a kit AST.
|
|
420
|
+
|
|
421
|
+
Identifiers are converted into this kit's naming convention (see the
|
|
422
|
+
dialect comment block above). Pass a shared ``converter`` to keep the
|
|
423
|
+
renaming consistent and injective across several formulas of one example
|
|
424
|
+
— :func:`convert_proverqa_formulas` does exactly that.
|
|
425
|
+
|
|
426
|
+
Raises:
|
|
427
|
+
lark.exceptions.LarkError: ``text`` is not in the dialect.
|
|
428
|
+
ValueError: an identifier cannot be converted without a collision or
|
|
429
|
+
a constant→variable semantics change.
|
|
430
|
+
"""
|
|
431
|
+
conv = converter if converter is not None else _NameConverter()
|
|
432
|
+
tree = _proverqa_parser.parse(text)
|
|
433
|
+
try:
|
|
434
|
+
return _ProverQATransformer(conv).transform(tree)
|
|
435
|
+
except VisitError as exc:
|
|
436
|
+
# Lark wraps transformer exceptions; surface the converter's own
|
|
437
|
+
# ValueError unchanged so the documented error contract holds.
|
|
438
|
+
if isinstance(exc.orig_exc, ValueError):
|
|
439
|
+
raise exc.orig_exc from None
|
|
440
|
+
raise
|
|
441
|
+
|
|
442
|
+
|
|
443
|
+
def convert_proverqa_formulas(
|
|
444
|
+
texts: Sequence[str]) -> Tuple[Tuple[Node, ...], Dict[str, Dict[str, str]]]:
|
|
445
|
+
"""Convert several ProverQA formulas with ONE shared, injective renaming.
|
|
446
|
+
|
|
447
|
+
Returns ``(nodes, mapping)`` where ``mapping`` is
|
|
448
|
+
``{"predicates": {original: new}, "constants": {original: new}}`` holding
|
|
449
|
+
only the names that actually changed. Every returned node must be CLOSED
|
|
450
|
+
— a free variable (e.g. from a mis-scoped quantifier reading) raises
|
|
451
|
+
rather than yielding a silently defective formula.
|
|
452
|
+
"""
|
|
453
|
+
conv = _NameConverter()
|
|
454
|
+
nodes: List[Node] = []
|
|
455
|
+
for text in texts:
|
|
456
|
+
node = parse_proverqa_formula(text, converter=conv)
|
|
457
|
+
free = free_variables(node)
|
|
458
|
+
if free:
|
|
459
|
+
raise ValueError(
|
|
460
|
+
f"proverqa: converted formula has free variable(s) "
|
|
461
|
+
f"{sorted(free)!r} — refusing (source: {text!r}).")
|
|
462
|
+
nodes.append(node)
|
|
463
|
+
return tuple(nodes), conv.mapping()
|
|
464
|
+
|
|
465
|
+
|
|
466
|
+
def solve_example(example: DatasetExample, *, on_indefinite: str = "label",
|
|
467
|
+
**prove_kwargs) -> dict:
|
|
468
|
+
"""Decide one converted ProverQA example end-to-end via ``api.prove``.
|
|
469
|
+
|
|
470
|
+
ProverQA's three-way gold label maps onto classical entailment exactly:
|
|
471
|
+
``"A"`` (True) iff premises ⊨ conclusion, ``"B"`` (False) iff premises ⊨
|
|
472
|
+
¬conclusion, ``"C"`` (Uncertain) otherwise. This helper proves the
|
|
473
|
+
positive direction first and the negated one only when needed, and
|
|
474
|
+
returns ``{"predicted": letter, "verdict": ..., "verdict_negated": ...}``
|
|
475
|
+
(verdicts as dicts; ``verdict_negated`` is ``None`` when the positive
|
|
476
|
+
direction already settled it). Extra ``prove_kwargs`` go verbatim to
|
|
477
|
+
:func:`unicode_logic_kit.api.prove`, so the ATP is the caller's choice
|
|
478
|
+
(``backends=["vampire"]``, ``timeout=…``). It requires an example whose
|
|
479
|
+
formulas are in KIT notation — i.e. loaded with the default
|
|
480
|
+
``convert_fol=True`` and without a recorded
|
|
481
|
+
``meta["fol_conversion_error"]``; anything else raises ``ValueError``
|
|
482
|
+
rather than silently scoring garbage.
|
|
483
|
+
|
|
484
|
+
``on_indefinite`` controls how a NON-DEFINITIVE prover outcome (status
|
|
485
|
+
``unknown``/``error`` — timeout, hit bound, honest incompleteness) is
|
|
486
|
+
interpreted when neither direction was proved:
|
|
487
|
+
|
|
488
|
+
- ``"label"`` (default): predict ``"C"`` — correct whenever the chosen
|
|
489
|
+
prover is decisive on the fragment (z3 on ProverQA's dev tiers is).
|
|
490
|
+
- ``"abstain"``: ``"C"`` only when BOTH directions are definitively
|
|
491
|
+
REFUTED (underdetermination ESTABLISHED by countermodels); any
|
|
492
|
+
indefinite leg yields ``predicted=None``, so a prover timeout can
|
|
493
|
+
never be silently scored as a correct "Uncertain".
|
|
494
|
+
- ``"raise"``: like ``"abstain"`` but an indefinite leg raises
|
|
495
|
+
``ValueError``.
|
|
496
|
+
"""
|
|
497
|
+
from ... import api
|
|
498
|
+
|
|
499
|
+
if on_indefinite not in ("label", "abstain", "raise"):
|
|
500
|
+
raise ValueError(
|
|
501
|
+
f"proverqa: on_indefinite must be 'label', 'abstain' or 'raise', "
|
|
502
|
+
f"got {on_indefinite!r}")
|
|
503
|
+
if example.meta.get("fol_conversion_error"):
|
|
504
|
+
raise ValueError(
|
|
505
|
+
f"proverqa: example {example.id} carries a conversion error "
|
|
506
|
+
f"({example.meta['fol_conversion_error']}) — cannot solve it.")
|
|
507
|
+
if example.fol_conclusion is None:
|
|
508
|
+
raise ValueError(f"proverqa: example {example.id} has no conclusion.")
|
|
509
|
+
|
|
510
|
+
def _parse(text: str) -> Node:
|
|
511
|
+
parsed = api.parse_any(text)
|
|
512
|
+
if not parsed.ok:
|
|
513
|
+
raise ValueError(
|
|
514
|
+
f"proverqa: example {example.id}: {text!r} does not parse "
|
|
515
|
+
"under the kit grammar — was the example loaded with "
|
|
516
|
+
"convert_fol=False?")
|
|
517
|
+
return parsed.formula
|
|
518
|
+
|
|
519
|
+
premises = [_parse(p) for p in example.fol_premises]
|
|
520
|
+
conclusion = _parse(example.fol_conclusion)
|
|
521
|
+
|
|
522
|
+
verdict = api.prove(conclusion, premises, **prove_kwargs)
|
|
523
|
+
if verdict.status == "proved":
|
|
524
|
+
return {"predicted": "A", "verdict": verdict.to_dict(),
|
|
525
|
+
"verdict_negated": None}
|
|
526
|
+
negated = api.prove(Not(conclusion), premises, **prove_kwargs)
|
|
527
|
+
if negated.status == "proved":
|
|
528
|
+
predicted: Optional[str] = "B"
|
|
529
|
+
elif on_indefinite == "label":
|
|
530
|
+
predicted = "C"
|
|
531
|
+
elif verdict.status == "refuted" and negated.status == "refuted":
|
|
532
|
+
# Underdetermination ESTABLISHED (countermodels both ways): "C" is a
|
|
533
|
+
# definitive answer here, so abstain/raise modes still label it.
|
|
534
|
+
predicted = "C"
|
|
535
|
+
elif on_indefinite == "raise":
|
|
536
|
+
raise ValueError(
|
|
537
|
+
f"proverqa: example {example.id}: indefinite prover outcome "
|
|
538
|
+
f"(goal: {verdict.status}/{verdict.reason}, negated: "
|
|
539
|
+
f"{negated.status}/{negated.reason}) with on_indefinite='raise'.")
|
|
540
|
+
else: # "abstain"
|
|
541
|
+
predicted = None
|
|
542
|
+
return {"predicted": predicted, "verdict": verdict.to_dict(),
|
|
543
|
+
"verdict_negated": negated.to_dict()}
|
|
544
|
+
|
|
545
|
+
|
|
546
|
+
def _resolve_id(record: dict, line_no: int, tier: Optional[str]) -> str:
|
|
547
|
+
"""The record's own ``"id"`` when present (namespaced by ``tier`` if
|
|
548
|
+
given, to avoid the cross-tier collision documented in the module
|
|
549
|
+
docstring), else a positional fallback.
|
|
550
|
+
|
|
551
|
+
The positional fallback never fires against verified real data (every
|
|
552
|
+
one of the 1,500 checked rows carries ``"id"``); it exists only so a
|
|
553
|
+
malformed/hand-edited record still yields an addressable example instead
|
|
554
|
+
of crashing on a missing key.
|
|
555
|
+
"""
|
|
556
|
+
raw_id = record.get("id")
|
|
557
|
+
if raw_id is None:
|
|
558
|
+
raw_id = f"pos{line_no}"
|
|
559
|
+
if tier is not None:
|
|
560
|
+
return f"proverqa:{tier}:{raw_id}"
|
|
561
|
+
return f"proverqa:{raw_id}"
|
|
562
|
+
|
|
563
|
+
|
|
564
|
+
def _example_from_record(record: dict, line_no: int, tier: Optional[str],
|
|
565
|
+
known_bad_ids: FrozenSet[str],
|
|
566
|
+
convert_fol: bool) -> DatasetExample:
|
|
567
|
+
nl2fol = record.get("nl2fol") or {}
|
|
568
|
+
nl_premises = tuple(nl2fol.keys())
|
|
569
|
+
fol_premises = tuple(nl2fol.values())
|
|
570
|
+
question = record.get("question")
|
|
571
|
+
conclusion_fol = record.get("conclusion_fol")
|
|
572
|
+
answer = record.get("answer")
|
|
573
|
+
example_id = _resolve_id(record, line_no, tier)
|
|
574
|
+
|
|
575
|
+
meta = {
|
|
576
|
+
k: v for k, v in record.items()
|
|
577
|
+
if k not in ("id", "nl2fol", "conclusion_fol", "question", "answer")
|
|
578
|
+
}
|
|
579
|
+
meta["line_no"] = line_no
|
|
580
|
+
if tier is not None:
|
|
581
|
+
meta["tier"] = tier
|
|
582
|
+
|
|
583
|
+
if convert_fol:
|
|
584
|
+
texts = list(fol_premises)
|
|
585
|
+
if conclusion_fol is not None:
|
|
586
|
+
texts.append(conclusion_fol)
|
|
587
|
+
try:
|
|
588
|
+
nodes, mapping = convert_proverqa_formulas(texts)
|
|
589
|
+
except (LarkError, ValueError) as exc:
|
|
590
|
+
# Honest per-example fallback: the verbatim upstream strings stay
|
|
591
|
+
# in place and the failure is recorded, never swallowed.
|
|
592
|
+
meta["fol_conversion_error"] = f"{type(exc).__name__}: {exc}"
|
|
593
|
+
else:
|
|
594
|
+
meta["original_fol_premises"] = list(fol_premises)
|
|
595
|
+
meta["original_fol_conclusion"] = conclusion_fol
|
|
596
|
+
meta["fol_name_mapping"] = mapping
|
|
597
|
+
rendered = tuple(node.to_unicode_str() for node in nodes)
|
|
598
|
+
if conclusion_fol is not None:
|
|
599
|
+
fol_premises = rendered[:-1]
|
|
600
|
+
conclusion_fol = rendered[-1]
|
|
601
|
+
else:
|
|
602
|
+
fol_premises = rendered
|
|
603
|
+
|
|
604
|
+
return DatasetExample(
|
|
605
|
+
id=example_id,
|
|
606
|
+
nl_premises=nl_premises,
|
|
607
|
+
fol_premises=fol_premises,
|
|
608
|
+
nl_conclusion=question,
|
|
609
|
+
fol_conclusion=conclusion_fol,
|
|
610
|
+
label=answer,
|
|
611
|
+
known_bad=example_id in known_bad_ids,
|
|
612
|
+
meta=meta,
|
|
613
|
+
)
|
|
614
|
+
|
|
615
|
+
|
|
616
|
+
def load_proverqa(path: Union[str, Path], *, tier: Optional[str] = None,
|
|
617
|
+
known_bad_ids: FrozenSet[str] = frozenset(),
|
|
618
|
+
convert_fol: bool = True) -> Iterator[DatasetExample]:
|
|
619
|
+
"""Stream :class:`~unicode_logic_kit.eval.datasets.DatasetExample` from a
|
|
620
|
+
local ProverQA ``dev/<tier>.json``-derived JSONL file.
|
|
621
|
+
|
|
622
|
+
Args:
|
|
623
|
+
path: path to a local ``.jsonl`` file — one ``dev/<tier>.json``
|
|
624
|
+
record object per non-blank line (see module docstring for
|
|
625
|
+
converting the upstream JSON-array distribution to this format).
|
|
626
|
+
NEVER downloaded by this function. The upstream train split
|
|
627
|
+
(``train/provergen-5000.json``) is NOT supported — it carries no
|
|
628
|
+
structured FOL fields at all (see module docstring).
|
|
629
|
+
tier: ``"easy"``, ``"medium"``, ``"hard"``, or ``None`` (default).
|
|
630
|
+
Purely caller-supplied metadata (the source file itself does not
|
|
631
|
+
self-report which tier a row belongs to) used to namespace ids
|
|
632
|
+
(see module docstring for why that matters across tiers) and
|
|
633
|
+
recorded verbatim in ``meta["tier"]`` when given.
|
|
634
|
+
known_bad_ids: ids (see :func:`_resolve_id`) whose gold FOL is known
|
|
635
|
+
to be broken beyond the notation mismatch documented in the
|
|
636
|
+
module docstring (e.g. a curated finding on top of that). Every
|
|
637
|
+
yielded example with a matching id gets ``known_bad=True``.
|
|
638
|
+
Defaults to an empty set.
|
|
639
|
+
convert_fol: with the default ``True``, every example's gold FOL is
|
|
640
|
+
parsed with the dedicated ProverQA dialect grammar and re-emitted
|
|
641
|
+
in KIT notation (``fol_premises``/``fol_conclusion`` then parse
|
|
642
|
+
under ``api.parse_any``); the verbatim upstream strings and the
|
|
643
|
+
injective name mapping land in ``meta["original_fol_premises"]``
|
|
644
|
+
/ ``meta["original_fol_conclusion"]`` /
|
|
645
|
+
``meta["fol_name_mapping"]``, and a per-example conversion
|
|
646
|
+
failure is recorded in ``meta["fol_conversion_error"]`` with the
|
|
647
|
+
verbatim strings kept in place — never swallowed, never a crash
|
|
648
|
+
of the whole load. ``False`` restores the raw pass-through
|
|
649
|
+
(0% kit-parse rate, see module docstring).
|
|
650
|
+
|
|
651
|
+
Yields:
|
|
652
|
+
One :class:`~unicode_logic_kit.eval.datasets.DatasetExample` per
|
|
653
|
+
non-blank JSONL line, in file order.
|
|
654
|
+
|
|
655
|
+
Raises:
|
|
656
|
+
ValueError: ``tier`` is given but is not one of ``"easy"``,
|
|
657
|
+
``"medium"``, ``"hard"``.
|
|
658
|
+
FileNotFoundError: ``path`` does not exist.
|
|
659
|
+
json.JSONDecodeError: a non-blank line is not valid JSON — this is
|
|
660
|
+
NOT swallowed; a malformed dataset file is a loud failure, not a
|
|
661
|
+
silently-skipped row.
|
|
662
|
+
"""
|
|
663
|
+
if tier is not None and tier not in _VALID_TIERS:
|
|
664
|
+
raise ValueError(f"tier must be one of {_VALID_TIERS!r} or None, got {tier!r}")
|
|
665
|
+
|
|
666
|
+
path = Path(path)
|
|
667
|
+
with path.open("r", encoding="utf-8") as fh:
|
|
668
|
+
for line_no, raw_line in enumerate(fh):
|
|
669
|
+
line = raw_line.strip()
|
|
670
|
+
if not line:
|
|
671
|
+
continue
|
|
672
|
+
record = json.loads(line)
|
|
673
|
+
yield _example_from_record(record, line_no, tier, known_bad_ids,
|
|
674
|
+
convert_fol)
|