unicode-logic-kit 0.31.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- unicode_logic_kit/__init__.py +385 -0
- unicode_logic_kit/__main__.py +520 -0
- unicode_logic_kit/_deadline.py +219 -0
- unicode_logic_kit/ace/__init__.py +126 -0
- unicode_logic_kit/ace/_align.py +135 -0
- unicode_logic_kit/ace/chem_lexicon.py +128 -0
- unicode_logic_kit/ace/drs_reader.py +570 -0
- unicode_logic_kit/ace/mapping.py +666 -0
- unicode_logic_kit/ace/reverse_modal.py +138 -0
- unicode_logic_kit/ace/runner.py +551 -0
- unicode_logic_kit/ace/translate.py +452 -0
- unicode_logic_kit/ace/verbalize.py +1070 -0
- unicode_logic_kit/api.py +1284 -0
- unicode_logic_kit/atp/__init__.py +177 -0
- unicode_logic_kit/atp/_ascii_names.py +113 -0
- unicode_logic_kit/atp/_html.py +72 -0
- unicode_logic_kit/atp/_substructural_input.py +228 -0
- unicode_logic_kit/atp/_tff_problem.py +715 -0
- unicode_logic_kit/atp/_tptp_problem.py +1111 -0
- unicode_logic_kit/atp/_writer_support.py +289 -0
- unicode_logic_kit/atp/clingo_backend.py +1180 -0
- unicode_logic_kit/atp/cvc5_backend.py +1385 -0
- unicode_logic_kit/atp/eprover_backend.py +732 -0
- unicode_logic_kit/atp/finite_domain.py +1055 -0
- unicode_logic_kit/atp/fitch.py +1547 -0
- unicode_logic_kit/atp/fitch_search.py +551 -0
- unicode_logic_kit/atp/hets_backend.py +339 -0
- unicode_logic_kit/atp/hybrid_down.py +120 -0
- unicode_logic_kit/atp/incremental.py +250 -0
- unicode_logic_kit/atp/kripke_enum.py +741 -0
- unicode_logic_kit/atp/lambek.py +436 -0
- unicode_logic_kit/atp/leo3_backend.py +332 -0
- unicode_logic_kit/atp/linear.py +738 -0
- unicode_logic_kit/atp/lj.py +705 -0
- unicode_logic_kit/atp/logic_backends.py +566 -0
- unicode_logic_kit/atp/ltl_tableau.py +1084 -0
- unicode_logic_kit/atp/minizinc_backend.py +1402 -0
- unicode_logic_kit/atp/modal_tableau.py +1382 -0
- unicode_logic_kit/atp/nanocop_backend.py +410 -0
- unicode_logic_kit/atp/portfolio.py +489 -0
- unicode_logic_kit/atp/protocol.py +1803 -0
- unicode_logic_kit/atp/prover9_entailment.py +1153 -0
- unicode_logic_kit/atp/resolution.py +1376 -0
- unicode_logic_kit/atp/resolution_check.py +1114 -0
- unicode_logic_kit/atp/sequent.py +1050 -0
- unicode_logic_kit/atp/tableau.py +921 -0
- unicode_logic_kit/atp/tableau_check.py +543 -0
- unicode_logic_kit/atp/tptp_ncl.py +811 -0
- unicode_logic_kit/atp/tptp_tff.py +1546 -0
- unicode_logic_kit/atp/tstp.py +1333 -0
- unicode_logic_kit/atp/tstp_check.py +1096 -0
- unicode_logic_kit/atp/twee_backend.py +236 -0
- unicode_logic_kit/atp/twee_check.py +711 -0
- unicode_logic_kit/atp/twee_entailment.py +953 -0
- unicode_logic_kit/atp/vampire_entailment.py +540 -0
- unicode_logic_kit/atp/z3_arith.py +470 -0
- unicode_logic_kit/atp/z3_equivalence.py +36 -0
- unicode_logic_kit/atp/z3_fuzzy.py +362 -0
- unicode_logic_kit/atp/z3_input.py +500 -0
- unicode_logic_kit/atp/z3_models.py +208 -0
- unicode_logic_kit/chem/__init__.py +88 -0
- unicode_logic_kit/chem/_naming.py +284 -0
- unicode_logic_kit/chem/cache.py +185 -0
- unicode_logic_kit/chem/interop.py +244 -0
- unicode_logic_kit/chem/mol.py +525 -0
- unicode_logic_kit/chem/signature.py +112 -0
- unicode_logic_kit/comorphism.py +497 -0
- unicode_logic_kit/dl/__init__.py +384 -0
- unicode_logic_kit/dl/classification.py +227 -0
- unicode_logic_kit/dl/concepts.py +632 -0
- unicode_logic_kit/dl/datatypes.py +818 -0
- unicode_logic_kit/dl/owl_functional.py +2433 -0
- unicode_logic_kit/dl/owl_manchester.py +1637 -0
- unicode_logic_kit/dl/owl_reasoner.py +790 -0
- unicode_logic_kit/dl/parser.py +391 -0
- unicode_logic_kit/dl/tableau.py +4048 -0
- unicode_logic_kit/dl/translate.py +2704 -0
- unicode_logic_kit/drt/__init__.py +94 -0
- unicode_logic_kit/drt/export.py +179 -0
- unicode_logic_kit/drt/nodes.py +506 -0
- unicode_logic_kit/drt/parser.py +965 -0
- unicode_logic_kit/drt/resolve.py +195 -0
- unicode_logic_kit/drt/reverse.py +175 -0
- unicode_logic_kit/eval/__init__.py +106 -0
- unicode_logic_kit/eval/batch.py +382 -0
- unicode_logic_kit/eval/canonical.py +663 -0
- unicode_logic_kit/eval/chem_batch.py +606 -0
- unicode_logic_kit/eval/converses.py +200 -0
- unicode_logic_kit/eval/datasets/__init__.py +136 -0
- unicode_logic_kit/eval/datasets/_base.py +263 -0
- unicode_logic_kit/eval/datasets/_proofwriter_proof.py +422 -0
- unicode_logic_kit/eval/datasets/c3po.py +678 -0
- unicode_logic_kit/eval/datasets/folio.py +158 -0
- unicode_logic_kit/eval/datasets/fracas.py +418 -0
- unicode_logic_kit/eval/datasets/groves.py +191 -0
- unicode_logic_kit/eval/datasets/logicbench.py +467 -0
- unicode_logic_kit/eval/datasets/logicnli.py +303 -0
- unicode_logic_kit/eval/datasets/malls.py +133 -0
- unicode_logic_kit/eval/datasets/pfolio.py +594 -0
- unicode_logic_kit/eval/datasets/pmb.py +242 -0
- unicode_logic_kit/eval/datasets/prontoqa.py +611 -0
- unicode_logic_kit/eval/datasets/proofwriter.py +1431 -0
- unicode_logic_kit/eval/datasets/proverqa.py +674 -0
- unicode_logic_kit/eval/datasets/willow.py +478 -0
- unicode_logic_kit/eval/equivalence.py +466 -0
- unicode_logic_kit/eval/exercise_gen.py +533 -0
- unicode_logic_kit/eval/explain.py +791 -0
- unicode_logic_kit/eval/generality.py +750 -0
- unicode_logic_kit/eval/metric_hf.py +458 -0
- unicode_logic_kit/eval/predicate_match.py +343 -0
- unicode_logic_kit/eval/theory_check.py +1170 -0
- unicode_logic_kit/eval/validate.py +306 -0
- unicode_logic_kit/fol/__init__.py +177 -0
- unicode_logic_kit/fol/_atom_keys.py +510 -0
- unicode_logic_kit/fol/_fol_nodes.py +3586 -0
- unicode_logic_kit/fol/_free_parameters.py +105 -0
- unicode_logic_kit/fol/_ho_nodes.py +448 -0
- unicode_logic_kit/fol/_hybrid_nodes.py +308 -0
- unicode_logic_kit/fol/_identifiers.py +1091 -0
- unicode_logic_kit/fol/_lambek_nodes.py +112 -0
- unicode_logic_kit/fol/_linear_nodes.py +352 -0
- unicode_logic_kit/fol/_modal_nodes.py +1467 -0
- unicode_logic_kit/fol/_msfl_nodes.py +2196 -0
- unicode_logic_kit/fol/_numeral_symbols.py +231 -0
- unicode_logic_kit/fol/_so_nodes.py +200 -0
- unicode_logic_kit/fol/_symbol_names.py +81 -0
- unicode_logic_kit/fol/_team_nodes.py +181 -0
- unicode_logic_kit/fol/_tptp_symbols.py +551 -0
- unicode_logic_kit/fol/_truth_constants.py +117 -0
- unicode_logic_kit/fol/casl_export.py +1135 -0
- unicode_logic_kit/fol/casl_import.py +929 -0
- unicode_logic_kit/fol/derivation.py +367 -0
- unicode_logic_kit/fol/dialect_detect.py +70 -0
- unicode_logic_kit/fol/dialect_repair.py +537 -0
- unicode_logic_kit/fol/frames.py +637 -0
- unicode_logic_kit/fol/grammars/terminals.lark +31 -0
- unicode_logic_kit/fol/lambda_tools.py +297 -0
- unicode_logic_kit/fol/latex_input.py +429 -0
- unicode_logic_kit/fol/modal_translation.py +944 -0
- unicode_logic_kit/fol/msflparser.py +1033 -0
- unicode_logic_kit/fol/naming.py +422 -0
- unicode_logic_kit/fol/nodes.py +241 -0
- unicode_logic_kit/fol/normalforms.py +492 -0
- unicode_logic_kit/fol/pal.py +287 -0
- unicode_logic_kit/fol/prolog_export.py +566 -0
- unicode_logic_kit/fol/prolog_input.py +505 -0
- unicode_logic_kit/fol/prover9_input.py +1325 -0
- unicode_logic_kit/fol/qml.py +1760 -0
- unicode_logic_kit/fol/qmltp_input.py +525 -0
- unicode_logic_kit/fol/sanitize.py +221 -0
- unicode_logic_kit/fol/serialize.py +79 -0
- unicode_logic_kit/fol/signature.py +1290 -0
- unicode_logic_kit/fol/simplify_check.py +544 -0
- unicode_logic_kit/fol/spans.py +594 -0
- unicode_logic_kit/fol/tptp_input.py +1503 -0
- unicode_logic_kit/fol/tptp_repair.py +941 -0
- unicode_logic_kit/fol/unification.py +157 -0
- unicode_logic_kit/fol/verbalize.py +263 -0
- unicode_logic_kit/hets/__init__.py +163 -0
- unicode_logic_kit/hets/bridge.py +142 -0
- unicode_logic_kit/hets/client.py +748 -0
- unicode_logic_kit/hets/docker.py +420 -0
- unicode_logic_kit/hets/dol.py +712 -0
- unicode_logic_kit/hets/haskell_json.py +355 -0
- unicode_logic_kit/hets/owl_backend.py +794 -0
- unicode_logic_kit/hets/owl_cli.py +598 -0
- unicode_logic_kit/hets/symbols.py +512 -0
- unicode_logic_kit/hol/__init__.py +140 -0
- unicode_logic_kit/hol/_ho_common.py +323 -0
- unicode_logic_kit/hol/_isabelle_binders.py +125 -0
- unicode_logic_kit/hol/classical.py +812 -0
- unicode_logic_kit/hol/deepshallow/__init__.py +45 -0
- unicode_logic_kit/hol/deepshallow/_common.py +177 -0
- unicode_logic_kit/hol/deepshallow/conditional.py +225 -0
- unicode_logic_kit/hol/deepshallow/intuitionistic.py +181 -0
- unicode_logic_kit/hol/deepshallow/modal.py +217 -0
- unicode_logic_kit/hol/deepshallow/qml.py +406 -0
- unicode_logic_kit/hol/deepshallow/relevant.py +206 -0
- unicode_logic_kit/hol/free.py +753 -0
- unicode_logic_kit/hol/goedel.py +336 -0
- unicode_logic_kit/hol/ho_modal.py +1743 -0
- unicode_logic_kit/hol/intuitionistic.py +403 -0
- unicode_logic_kit/hol/isabelle_conditional.py +593 -0
- unicode_logic_kit/hol/isabelle_modal.py +1908 -0
- unicode_logic_kit/hol/isabelle_relevant.py +412 -0
- unicode_logic_kit/hol/isabelle_runner.py +1147 -0
- unicode_logic_kit/hol/isabelle_substructural.py +884 -0
- unicode_logic_kit/hol/lean.py +1018 -0
- unicode_logic_kit/hol/manyvalued.py +921 -0
- unicode_logic_kit/hol/secondorder.py +687 -0
- unicode_logic_kit/hol/thf_modal.py +941 -0
- unicode_logic_kit/hol/thirdorder.py +397 -0
- unicode_logic_kit/ilp/__init__.py +89 -0
- unicode_logic_kit/ilp/readback.py +389 -0
- unicode_logic_kit/ilp/separation.py +153 -0
- unicode_logic_kit/ilp/task.py +730 -0
- unicode_logic_kit/logic.py +163 -0
- unicode_logic_kit/mcp/__init__.py +28 -0
- unicode_logic_kit/mcp/__main__.py +5 -0
- unicode_logic_kit/mcp/chem_tools.py +1031 -0
- unicode_logic_kit/mcp/server.py +2453 -0
- unicode_logic_kit/mcp/syntax_spec.py +681 -0
- unicode_logic_kit/prob/__init__.py +53 -0
- unicode_logic_kit/prob/_bdd.py +225 -0
- unicode_logic_kit/prob/_column_gen.py +668 -0
- unicode_logic_kit/prob/distribution.py +686 -0
- unicode_logic_kit/prob/nilsson.py +470 -0
- unicode_logic_kit/py.typed +0 -0
- unicode_logic_kit/semantics/__init__.py +137 -0
- unicode_logic_kit/semantics/_modal_reject.py +156 -0
- unicode_logic_kit/semantics/action_models.py +466 -0
- unicode_logic_kit/semantics/asp_models.py +1200 -0
- unicode_logic_kit/semantics/conditional.py +580 -0
- unicode_logic_kit/semantics/dynamic_epistemic.py +95 -0
- unicode_logic_kit/semantics/free_logic.py +913 -0
- unicode_logic_kit/semantics/fuzzy.py +384 -0
- unicode_logic_kit/semantics/fuzzy_kripke.py +442 -0
- unicode_logic_kit/semantics/intuitionistic.py +581 -0
- unicode_logic_kit/semantics/kripke.py +1139 -0
- unicode_logic_kit/semantics/manyvalued.py +580 -0
- unicode_logic_kit/semantics/matrix.py +342 -0
- unicode_logic_kit/semantics/model_eval.py +1135 -0
- unicode_logic_kit/semantics/modelfinder.py +1036 -0
- unicode_logic_kit/semantics/nonmonotonic.py +372 -0
- unicode_logic_kit/semantics/relevant.py +331 -0
- unicode_logic_kit/semantics/secondorder.py +657 -0
- unicode_logic_kit/semantics/structures.py +352 -0
- unicode_logic_kit/semantics/tarski.py +975 -0
- unicode_logic_kit/semantics/team.py +315 -0
- unicode_logic_kit/semantics/team_translation.py +416 -0
- unicode_logic_kit/semantics/thirdorder.py +358 -0
- unicode_logic_kit/semantics/tnorm.py +85 -0
- unicode_logic_kit/semantics/truthtable.py +201 -0
- unicode_logic_kit-0.31.0.dist-info/METADATA +333 -0
- unicode_logic_kit-0.31.0.dist-info/RECORD +237 -0
- unicode_logic_kit-0.31.0.dist-info/WHEEL +4 -0
- unicode_logic_kit-0.31.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,1431 @@
|
|
|
1
|
+
"""Adapter for the ProofWriter dataset (Tafjord, Dalvi Mishra, Clark,
|
|
2
|
+
"ProofWriter: Generating Implications, Proofs, and Abductive Statements over
|
|
3
|
+
Natural Language", Findings of ACL 2021, arXiv:2012.13048) — local JSONL
|
|
4
|
+
only, no network access.
|
|
5
|
+
|
|
6
|
+
Source and verified schema
|
|
7
|
+
---------------------------
|
|
8
|
+
The primary distribution named by the paper, https://allenai.org/data/proofwriter,
|
|
9
|
+
does NOT serve a metadata page: fetching it returns a bare HTTP 307 redirect
|
|
10
|
+
straight to ``proofwriter-dataset-V2020.12.3.zip`` (~214 MB) on
|
|
11
|
+
``aristo-data-public.s3.amazonaws.com``, with no separate page describing its
|
|
12
|
+
internal JSONL field names, and the current https://allenai.org/data catalog
|
|
13
|
+
no longer lists a ProofWriter entry at all (both checked directly,
|
|
14
|
+
2026-08-12). Downloading and unpacking a 214 MB archive was out of scope for
|
|
15
|
+
schema verification here, so — per this adapter's instructions, which allow
|
|
16
|
+
falling back to "the best-documented HF mirror" when the original
|
|
17
|
+
distribution's internal structure cannot be confirmed via the rows API — the
|
|
18
|
+
schema below was verified instead against
|
|
19
|
+
https://huggingface.co/datasets/tasksource/proofwriter (unauthenticated,
|
|
20
|
+
11,898+ downloads, the most-used ProofWriter mirror on the Hub), directly via
|
|
21
|
+
the Hugging Face ``datasets-server`` ``/first-rows``, ``/statistics``, and
|
|
22
|
+
``/search`` APIs across all three of its splits (``train`` — 585,552 rows,
|
|
23
|
+
``validation`` — 85,468 rows, ``test`` — 174,476 rows), 2026-08-12.
|
|
24
|
+
:func:`load_proofwriter` reads that verified schema, NOT the original
|
|
25
|
+
AllenAI ZIP's internal format (unverified here) — see "Honesty" below for
|
|
26
|
+
what this means in practice.
|
|
27
|
+
|
|
28
|
+
Each verified row is a flat JSON object with exactly these keys (confirmed
|
|
29
|
+
via the dataset's declared ``dataset_info.features`` and by inspecting many
|
|
30
|
+
real fetched rows):
|
|
31
|
+
|
|
32
|
+
* ``"id"`` — ``str``. NOT a per-row unique id (see "Id resolution"
|
|
33
|
+
below) — it identifies one GENERATED THEORY, and every question asked
|
|
34
|
+
against that theory (2-6 in the fetched sample) repeats the same ``"id"``.
|
|
35
|
+
Its own substructure (e.g. ``"AttNeg-OWA-D0-1778"``) encodes, dash-separated:
|
|
36
|
+
a generation-template family (``AttNeg``/``AttNoneg``/``RelNeg``/``RelNoneg``
|
|
37
|
+
— attribute- or relation-typed facts, with or without negated facts), an
|
|
38
|
+
open/closed-world tag (``OWA`` in every single row observed here — see
|
|
39
|
+
"CWA vs OWA" below), a depth bucket (``D0`` etc., redundant with ``maxD``),
|
|
40
|
+
and a generator seed/index. This adapter does not parse that substring; it
|
|
41
|
+
is preserved verbatim in ``meta["theory_id"]``.
|
|
42
|
+
* ``"maxD"`` — ``int``, the maximum proof depth reachable from this
|
|
43
|
+
theory's rule set (0 in every fetched sample row; the dataset overall
|
|
44
|
+
ranges 0-10 per the mirror's column statistics).
|
|
45
|
+
* ``"NFact"`` — ``int``, number of atomic facts in ``"theory"``.
|
|
46
|
+
* ``"NRule"`` — ``int``, number of conditional rules in ``"theory"``.
|
|
47
|
+
* ``"theory"`` — ``str``, ALL of this theory's facts and rules concatenated
|
|
48
|
+
into one string, one sentence per fact/rule, each ending in ``"."`` and
|
|
49
|
+
separated by a single space — e.g. ``"Anne is smart. Dave is round. If
|
|
50
|
+
someone is cold then they are blue."``. Verified: in every fetched row,
|
|
51
|
+
``NFact + NRule`` equals exactly the number of ``". "``-delimited sentences
|
|
52
|
+
in ``"theory"`` (hand-checked below in the test suite), so this adapter's
|
|
53
|
+
``nl_premises`` sentence split (see "Field mapping") is not a guess.
|
|
54
|
+
* ``"question"`` — ``str``, one NL sentence being asked about (e.g. ``"Dave
|
|
55
|
+
is round."`` or its negation ``"Dave is not round."``).
|
|
56
|
+
* ``"answer"`` — ``str``, one of exactly ``{"True", "False", "Unknown"}``
|
|
57
|
+
(confirmed via the mirror's train-split column statistics: 158,805 /
|
|
58
|
+
158,805 / 267,942 rows respectively — no fourth value exists in this
|
|
59
|
+
mirror).
|
|
60
|
+
* ``"QDep"`` — ``int``, the depth of proof needed to answer this
|
|
61
|
+
specific question (0 in every "directly stated fact" question observed).
|
|
62
|
+
* ``"QLen"`` — ``float`` or ``null``. In every row fetched here, ``null``
|
|
63
|
+
exactly when ``"answer" == "Unknown"`` and ``1.0`` whenever the answer is
|
|
64
|
+
``"True"``/``"False"`` — an OBSERVED correlation from the sample fetched
|
|
65
|
+
for this verification, not a guarantee re-derived from a spec, so it is
|
|
66
|
+
not relied on by this loader beyond passing the raw value through in
|
|
67
|
+
``meta``.
|
|
68
|
+
* ``"allProofs"`` — ``str``, an opaque proof-forest annotation in the
|
|
69
|
+
dataset's own ``triple``/``rule``-reference notation (e.g. ``"@0: Anne is
|
|
70
|
+
smart.[(triple1)] ..."``), kept verbatim in ``meta`` — this adapter does
|
|
71
|
+
not parse it (it is not FOL, and parsing its internal proof-tree grammar
|
|
72
|
+
is out of scope here).
|
|
73
|
+
* ``"config"`` — ``str``, one of exactly 8 values confirmed via the
|
|
74
|
+
mirror's train-split column statistics: ``"depth-0"``, ``"depth-1"``,
|
|
75
|
+
``"depth-2"``, ``"depth-3"``, ``"depth-3ext"``, ``"depth-3ext-NatLang"``,
|
|
76
|
+
``"depth-5"``, ``"NatLang"`` (the paper's D0-D5 depth staircase, plus the
|
|
77
|
+
hand-authored "NatLang"/"birds-electricity"-style natural-language subsets
|
|
78
|
+
described in the paper; note ``"depth-4"`` does not appear as a distinct
|
|
79
|
+
``config`` value in this mirror).
|
|
80
|
+
|
|
81
|
+
Honesty: what this adapter does NOT give you
|
|
82
|
+
----------------------------------------------
|
|
83
|
+
* **No FOL annotation exists.** ProofWriter's ``"theory"``/``"question"``
|
|
84
|
+
are natural-language sentences ("If someone is red then they are kind.");
|
|
85
|
+
there is no gold first-order-logic formula anywhere in the source data.
|
|
86
|
+
Accordingly ``fol_premises`` is ALWAYS ``()`` and ``fol_conclusion`` is
|
|
87
|
+
ALWAYS ``None`` for every example this loader yields — never guessed,
|
|
88
|
+
never back-translated. A consequence: :func:`~unicode_logic_kit.eval.datasets.audit_examples`
|
|
89
|
+
run over ProofWriter examples is VACUOUSLY ``ok=True`` for all of them (no
|
|
90
|
+
FOL string means nothing to parse or fail ``check()`` on) — it is not a
|
|
91
|
+
meaningful signal for this dataset and callers should not read "0 defects"
|
|
92
|
+
as "ProofWriter's data is fine", only as "there is nothing here to audit".
|
|
93
|
+
* **CWA vs OWA, and why it is NOT classical FOL entailment.** ProofWriter
|
|
94
|
+
was released in TWO reasoning-assumption variants per the paper: **CWA**
|
|
95
|
+
(closed-world assumption — a fact not provable from the theory is assumed
|
|
96
|
+
FALSE, i.e. negation-as-failure) and **OWA** (open-world assumption — a
|
|
97
|
+
fact not provable either way is genuinely ``"Unknown"``, distinct from
|
|
98
|
+
``"False"``). The verified mirror used here contains **ONLY the OWA
|
|
99
|
+
variant** — every single ``"id"`` in the fetched sample, and a full-text
|
|
100
|
+
search for the substring ``"CWA"`` across the ENTIRE train split, returned
|
|
101
|
+
zero matches (``num_rows_total: 0``, confirmed 2026-08-12) — so this
|
|
102
|
+
adapter's field mapping and every claim above describes OWA data only. The
|
|
103
|
+
presence of the three-valued ``answer`` (``"Unknown"`` as a genuine third
|
|
104
|
+
value, not collapsed into ``"False"``) is itself the observable signature
|
|
105
|
+
of OWA rather than CWA.
|
|
106
|
+
|
|
107
|
+
How the two variants relate to classical FOL — stated precisely, because
|
|
108
|
+
a sloppy version of this claim is easy to make and wrong: **CWA is
|
|
109
|
+
ordinary two-valued FOL** — evaluation of the question in ONE canonical
|
|
110
|
+
model, the closed (minimal) model, where exactly the derivable atoms hold
|
|
111
|
+
and everything else is plainly false; there is no third value. **OWA's
|
|
112
|
+
three-way label is the ENTAILMENT split** — ``True`` iff the theory
|
|
113
|
+
entails the question, ``False`` iff it entails its negation, ``Unknown``
|
|
114
|
+
otherwise — and "Unknown" is a statement ABOUT entailment, not an FOL
|
|
115
|
+
truth value inside any model. Both are classical; they answer different
|
|
116
|
+
questions ("true in the closed model?" vs "true in every model?"), and
|
|
117
|
+
the labels visibly diverge exactly where a fact is underdetermined
|
|
118
|
+
(closed model: false; entailment: unknown). For the STRUCTURED route
|
|
119
|
+
below, both are decidable with any registered ATP:
|
|
120
|
+
``solve_structured_example(..., semantics="owa")`` runs the entailment
|
|
121
|
+
cascade, ``semantics="cwa"`` runs closed-model checking with the ATP as
|
|
122
|
+
the per-atom derivability oracle (exact for the definite, negation-free
|
|
123
|
+
theories; a rule with negation in its body would mean
|
|
124
|
+
negation-as-failure inside the theory and is refused). The FLAT
|
|
125
|
+
tasksource mirror THIS loader reads carries no representations, so for
|
|
126
|
+
it these labels remain data to report, not something this adapter
|
|
127
|
+
recomputes.
|
|
128
|
+
* **Only one split's worth of assumption is covered.** This loader's field
|
|
129
|
+
mapping was verified against OWA rows only (see above); if a caller
|
|
130
|
+
obtains genuine CWA-variant ProofWriter data (e.g. from the original
|
|
131
|
+
AllenAI ZIP, unverified here), the row shape is very likely structurally
|
|
132
|
+
identical (same ``id`` substring convention, same field names) but that
|
|
133
|
+
has NOT been independently confirmed by this adapter.
|
|
134
|
+
* **``"theory"`` is one blob string, not pre-split sentences.** ``nl_premises``
|
|
135
|
+
below is DERIVED by this adapter (splitting on ``". "``), not a field that
|
|
136
|
+
exists upstream — see "Field mapping".
|
|
137
|
+
|
|
138
|
+
License
|
|
139
|
+
-------
|
|
140
|
+
**UNVERIFIED** as of 2026-08-12. No ProofWriter-specific license text could
|
|
141
|
+
be confirmed from any live, unauthenticated source: the AllenAI dataset page
|
|
142
|
+
is a bare redirect straight to the ZIP with no accompanying license file
|
|
143
|
+
reachable without downloading it, the current AllenAI data catalog no longer
|
|
144
|
+
lists a ProofWriter entry to check, and the verified HF mirror
|
|
145
|
+
(tasksource/proofwriter)'s dataset card carries no ``license`` tag and reads
|
|
146
|
+
literally "More Information needed". The sibling AI2 RuleTaker CODE
|
|
147
|
+
repository (https://github.com/allenai/ruletaker, whose legacy JSONL example
|
|
148
|
+
format ProofWriter's ``theory``/``question`` sentences continue) is
|
|
149
|
+
Apache-2.0-licensed, but that governs that repository's CODE, not
|
|
150
|
+
ProofWriter's separately-hosted data ZIP, and is not treated here as a
|
|
151
|
+
substitute for a verified data license. Treat ProofWriter as all-rights-
|
|
152
|
+
reserved research data pending confirmation directly from the paper's
|
|
153
|
+
authors or AI2, and do not redistribute this loader's *inputs* (the JSONL
|
|
154
|
+
file itself) without resolving that.
|
|
155
|
+
|
|
156
|
+
Field mapping
|
|
157
|
+
--------------
|
|
158
|
+
ProofWriter has no premises/conclusion ENTAILMENT structure quite like
|
|
159
|
+
FOLIO's (no separate gold "this follows" formula), but its
|
|
160
|
+
theory-facts-and-rules / queried-question / True-False-Unknown shape maps
|
|
161
|
+
onto :class:`~unicode_logic_kit.eval.datasets.DatasetExample` as an entailment
|
|
162
|
+
example nonetheless:
|
|
163
|
+
|
|
164
|
+
* ``nl_premises`` — ``"theory"`` SPLIT into individual sentences on ``". "``
|
|
165
|
+
(each fragment re-terminated with ``"."`` if the split ate it) by this
|
|
166
|
+
module's :func:`_split_theory_sentences`. This is a LOCAL heuristic of
|
|
167
|
+
this adapter, not part of the verified upstream schema (upstream gives you
|
|
168
|
+
one string) — verified safe against every fetched sample row only insofar
|
|
169
|
+
as none of them contain a sentence-internal period, abbreviation, or
|
|
170
|
+
ellipsis (a synthetic, template-generated corpus, so this holds by
|
|
171
|
+
construction for the OWA rows checked). The RAW, unsplit ``"theory"``
|
|
172
|
+
string is ALSO kept verbatim in ``meta["theory"]`` so nothing is lost to
|
|
173
|
+
the split.
|
|
174
|
+
* ``fol_premises`` — ALWAYS ``()`` (no FOL exists; see "Honesty" above).
|
|
175
|
+
* ``nl_conclusion`` — ``"question"`` verbatim.
|
|
176
|
+
* ``fol_conclusion`` — ALWAYS ``None`` (no FOL exists; see "Honesty" above).
|
|
177
|
+
* ``label`` — ``"answer"`` verbatim (``"True"``/``"False"``/``"Unknown"``).
|
|
178
|
+
* ``meta`` — every other record key (``maxD``, ``NFact``, ``NRule``, ``QDep``,
|
|
179
|
+
``QLen``, ``allProofs``, ``config``), PLUS the raw ``"theory"`` string,
|
|
180
|
+
PLUS ``"theory_id"`` (renamed from the record's own ``"id"`` — see "Id
|
|
181
|
+
resolution"), PLUS ``"line_no"``.
|
|
182
|
+
|
|
183
|
+
Id resolution
|
|
184
|
+
--------------
|
|
185
|
+
ProofWriter's own ``"id"`` field identifies a GENERATED THEORY, not a single
|
|
186
|
+
row — exactly the same shape of problem as FOLIO's ``"story-id"`` (see
|
|
187
|
+
``folio.py``'s docstring): several consecutive rows in the verified sample
|
|
188
|
+
share one ``"id"`` while asking different questions about the same theory,
|
|
189
|
+
so using it directly as this adapter's per-example id would collide. This
|
|
190
|
+
loader therefore ALWAYS uses the positional id ``f"proofwriter:{line_no}"``
|
|
191
|
+
(0-based line number in the local file) and preserves the original,
|
|
192
|
+
non-unique ``"id"`` value verbatim in ``meta["theory_id"]`` for anyone who
|
|
193
|
+
wants to group rows back into their source theory.
|
|
194
|
+
|
|
195
|
+
This module never downloads anything — obtain a local JSONL file yourself
|
|
196
|
+
(e.g. by exporting the verified ``tasksource/proofwriter`` split with
|
|
197
|
+
``datasets.load_dataset("tasksource/proofwriter", split="train").to_json(path, orient="records", lines=True)``,
|
|
198
|
+
or via the HF ``rows``/``first-rows`` API) and pass its local path to
|
|
199
|
+
:func:`load_proofwriter`.
|
|
200
|
+
"""
|
|
201
|
+
|
|
202
|
+
import itertools
|
|
203
|
+
import json
|
|
204
|
+
import re
|
|
205
|
+
from pathlib import Path
|
|
206
|
+
from typing import Dict, FrozenSet, Iterator, List, Optional, Tuple, Union
|
|
207
|
+
|
|
208
|
+
from ...fol.nodes import (
|
|
209
|
+
Node, Atom, Not, And, Or, Xor, Implies, Iff, Quantifier,
|
|
210
|
+
Variable, Constant, substitute,
|
|
211
|
+
)
|
|
212
|
+
from ...fol._msfl_nodes import key_text
|
|
213
|
+
from ._base import DatasetExample, _register_dataset_info
|
|
214
|
+
from . import _proofwriter_proof as _proof
|
|
215
|
+
|
|
216
|
+
__all__ = [
|
|
217
|
+
"load_proofwriter",
|
|
218
|
+
"load_proofwriter_structured",
|
|
219
|
+
"parse_proofwriter_representation",
|
|
220
|
+
"solve_structured_example",
|
|
221
|
+
"check_gold_proof",
|
|
222
|
+
]
|
|
223
|
+
|
|
224
|
+
_register_dataset_info(
|
|
225
|
+
"proofwriter",
|
|
226
|
+
license=(
|
|
227
|
+
"UNVERIFIED as of 2026-08-12 -- no ProofWriter-specific license text "
|
|
228
|
+
"could be confirmed from any live, unauthenticated source (AllenAI's "
|
|
229
|
+
"dataset page redirects directly to a ZIP with no reachable license "
|
|
230
|
+
"file; the verified HF mirror's dataset card has no license tag); "
|
|
231
|
+
"treat as all-rights-reserved research data pending confirmation "
|
|
232
|
+
"from the paper's authors or AI2 -- see module docstring"
|
|
233
|
+
),
|
|
234
|
+
source_url="https://huggingface.co/datasets/tasksource/proofwriter",
|
|
235
|
+
citation_hint=(
|
|
236
|
+
"Tafjord, Oyvind, Bhavana Dalvi Mishra, and Peter Clark. \"ProofWriter: "
|
|
237
|
+
"Generating Implications, Proofs, and Abductive Statements over Natural "
|
|
238
|
+
"Language.\" Findings of the Association for Computational Linguistics: "
|
|
239
|
+
"ACL-IJCNLP 2021. arXiv:2012.13048."
|
|
240
|
+
),
|
|
241
|
+
)
|
|
242
|
+
|
|
243
|
+
|
|
244
|
+
def _split_theory_sentences(theory: Optional[str]) -> Tuple[str, ...]:
|
|
245
|
+
"""Split a raw ``"theory"`` blob into individual NL fact/rule sentences.
|
|
246
|
+
|
|
247
|
+
A LOCAL heuristic of this adapter (see the module docstring's "Field
|
|
248
|
+
mapping" section for why this is safe over the verified sample but is
|
|
249
|
+
not itself part of the upstream schema): splits on the literal
|
|
250
|
+
substring ``". "``, then re-appends a trailing ``"."`` to any fragment
|
|
251
|
+
the split consumed it from (every fragment except possibly the last).
|
|
252
|
+
``None`` or ``""`` (e.g. a record missing ``"theory"`` entirely) yields
|
|
253
|
+
``()`` rather than raising or fabricating a one-element tuple of empty
|
|
254
|
+
string.
|
|
255
|
+
"""
|
|
256
|
+
if not theory:
|
|
257
|
+
return ()
|
|
258
|
+
sentences = []
|
|
259
|
+
for part in theory.strip().split(". "):
|
|
260
|
+
part = part.strip()
|
|
261
|
+
if not part:
|
|
262
|
+
continue
|
|
263
|
+
if not part.endswith("."):
|
|
264
|
+
part = part + "."
|
|
265
|
+
sentences.append(part)
|
|
266
|
+
return tuple(sentences)
|
|
267
|
+
|
|
268
|
+
|
|
269
|
+
def _example_from_record(record: dict, line_no: int,
|
|
270
|
+
known_bad_ids: FrozenSet[str]) -> DatasetExample:
|
|
271
|
+
theory = record.get("theory")
|
|
272
|
+
question = record.get("question")
|
|
273
|
+
answer = record.get("answer")
|
|
274
|
+
example_id = f"proofwriter:{line_no}"
|
|
275
|
+
|
|
276
|
+
# Everything except question/answer stays in meta (including "theory"
|
|
277
|
+
# itself, verbatim -- see module docstring: nl_premises below is a
|
|
278
|
+
# DERIVED split, not a byte-identical copy). The record's own "id" is
|
|
279
|
+
# renamed to "theory_id" here: see "Id resolution" in the module
|
|
280
|
+
# docstring for why it is deliberately NOT used as this example's id.
|
|
281
|
+
meta = {k: v for k, v in record.items() if k not in ("question", "answer")}
|
|
282
|
+
meta["theory_id"] = meta.pop("id", None)
|
|
283
|
+
meta["line_no"] = line_no
|
|
284
|
+
|
|
285
|
+
return DatasetExample(
|
|
286
|
+
id=example_id,
|
|
287
|
+
nl_premises=_split_theory_sentences(theory),
|
|
288
|
+
fol_premises=(),
|
|
289
|
+
nl_conclusion=question,
|
|
290
|
+
fol_conclusion=None,
|
|
291
|
+
label=answer,
|
|
292
|
+
known_bad=example_id in known_bad_ids,
|
|
293
|
+
meta=meta,
|
|
294
|
+
)
|
|
295
|
+
|
|
296
|
+
|
|
297
|
+
def load_proofwriter(path: Union[str, Path], *,
|
|
298
|
+
known_bad_ids: FrozenSet[str] = frozenset()) -> Iterator[DatasetExample]:
|
|
299
|
+
"""Stream :class:`~unicode_logic_kit.eval.datasets.DatasetExample` from a
|
|
300
|
+
local ProofWriter JSONL file (verified ``tasksource/proofwriter`` schema
|
|
301
|
+
-- see module docstring).
|
|
302
|
+
|
|
303
|
+
Args:
|
|
304
|
+
path: path to a local ``.jsonl`` file — one
|
|
305
|
+
``{"id", "maxD", "NFact", "NRule", "theory", "question",
|
|
306
|
+
"answer", "QDep", "QLen", "allProofs", "config"}`` object per
|
|
307
|
+
non-blank line (see module docstring for how to produce this
|
|
308
|
+
from the verified HF mirror). NEVER downloaded by this function.
|
|
309
|
+
known_bad_ids: ids (the positional ``f"proofwriter:{line_no}"`` form
|
|
310
|
+
-- see "Id resolution" in the module docstring) whose ``answer``
|
|
311
|
+
is known to be broken (e.g. from a prior human review). Every
|
|
312
|
+
yielded example with a matching id gets ``known_bad=True``.
|
|
313
|
+
Defaults to an empty set.
|
|
314
|
+
|
|
315
|
+
Yields:
|
|
316
|
+
One :class:`~unicode_logic_kit.eval.datasets.DatasetExample` per
|
|
317
|
+
non-blank JSONL line, in file order. ``fol_premises`` is always
|
|
318
|
+
``()`` and ``fol_conclusion`` is always ``None`` (ProofWriter has no
|
|
319
|
+
FOL gold annotation -- see the module docstring's "Honesty"
|
|
320
|
+
section). Fields missing from a record (e.g. no ``"theory"``, no
|
|
321
|
+
``"answer"``) map to ``()``/``None`` rather than raising -- this
|
|
322
|
+
mirrors :mod:`~unicode_logic_kit.eval.datasets.folio`'s defensive
|
|
323
|
+
``dict.get`` style, documented, not an exception.
|
|
324
|
+
|
|
325
|
+
Raises:
|
|
326
|
+
FileNotFoundError: ``path`` does not exist.
|
|
327
|
+
json.JSONDecodeError: a non-blank line is not valid JSON -- this is
|
|
328
|
+
NOT swallowed; a malformed dataset file is a loud failure, not a
|
|
329
|
+
silently-skipped row.
|
|
330
|
+
"""
|
|
331
|
+
path = Path(path)
|
|
332
|
+
with path.open("r", encoding="utf-8") as fh:
|
|
333
|
+
for line_no, raw_line in enumerate(fh):
|
|
334
|
+
line = raw_line.strip()
|
|
335
|
+
if not line:
|
|
336
|
+
continue
|
|
337
|
+
record = json.loads(line)
|
|
338
|
+
yield _example_from_record(record, line_no, known_bad_ids)
|
|
339
|
+
|
|
340
|
+
|
|
341
|
+
# --------------------------------------------------------------------------- #
|
|
342
|
+
# Structured OWA distribution: deterministic FOL GENERATION from the dataset's
|
|
343
|
+
# own triple/rule representations
|
|
344
|
+
#
|
|
345
|
+
# The AllenAI ProofWriter release (mirrored with all metadata intact at
|
|
346
|
+
# https://huggingface.co/datasets/hitachi-nlp/proofwriter_processed_OWA —
|
|
347
|
+
# configs depth-0/1/2/3/3ext/5, NatLang, birds-electricity; schema verified
|
|
348
|
+
# against real depth-2 rows on 2026-08-12) carries, next to every NL
|
|
349
|
+
# sentence, the ORIGINAL RuleTaker-style symbolic representation:
|
|
350
|
+
#
|
|
351
|
+
# fact triple: ("Anne" "is" "white" "+") [attribute]
|
|
352
|
+
# ("bear" "chases" "dog" "+") [relation]
|
|
353
|
+
# rule: ((("something" "is" "young" "+"))
|
|
354
|
+
# -> ("something" "is" "white" "+"))
|
|
355
|
+
# question: ("Charlie" "is" "cold" "-") [negative polarity]
|
|
356
|
+
#
|
|
357
|
+
# Unlike the flat tasksource mirror load_proofwriter reads (NL text only —
|
|
358
|
+
# no FOL gold, as documented above), these representations admit a
|
|
359
|
+
# DETERMINISTIC, rule-based translation to FOL — no LLM, no heuristics on NL
|
|
360
|
+
# text. The convention implemented here (the standard reading of RuleTaker
|
|
361
|
+
# triples, Clark et al. 2020):
|
|
362
|
+
#
|
|
363
|
+
# * an attribute triple (E "is" A pol) becomes A'(e) — predicate = the
|
|
364
|
+
# attribute, capitalised (white → White); entity = a constant in kit
|
|
365
|
+
# casing (Anne → anne, multi-word "bald eagle" → baldEagle);
|
|
366
|
+
# * a relation triple (E V F pol) becomes V'(e, f) (chases → Chases);
|
|
367
|
+
# * polarity "-" (or "~") wraps the atom in ¬;
|
|
368
|
+
# * the placeholder words something/someone/somebody are RULE VARIABLES:
|
|
369
|
+
# every rule quantifies universally over each distinct placeholder it
|
|
370
|
+
# uses — ∀x (Young(x) → White(x)); a rule without placeholders stays a
|
|
371
|
+
# ground implication;
|
|
372
|
+
# * a rule ((c1 … cn) -> d) becomes (c1 ∧ … ∧ cn) → d under those
|
|
373
|
+
# quantifiers.
|
|
374
|
+
#
|
|
375
|
+
# HONESTY: the resulting fol_premises/fol_conclusion are GENERATED BY THIS
|
|
376
|
+
# KIT from the dataset's own symbolic annotations — ProofWriter itself ships
|
|
377
|
+
# no FOL strings; meta["fol_generated"] marks every such example and meta
|
|
378
|
+
# keeps the verbatim source representations. The OWA labels are the
|
|
379
|
+
# classical ENTAILMENT split (True iff premises ⊨ q, False iff premises ⊨
|
|
380
|
+
# ¬q, Unknown otherwise) — solve_structured_example(semantics="owa") decides
|
|
381
|
+
# exactly that, verified 24/24 on the fixture. The CWA labels are ordinary
|
|
382
|
+
# two-valued truth in the CLOSED (minimal) model — plain FOL model checking,
|
|
383
|
+
# no third value — and solve_structured_example(semantics="cwa") decides
|
|
384
|
+
# that via per-atom derivability with the caller's chosen ATP (see its
|
|
385
|
+
# docstring for the exact fragment contract). The verified hitachi-nlp
|
|
386
|
+
# mirror ships only OWA data; CWA gold labels live in the original AllenAI
|
|
387
|
+
# zip (proofwriter-dataset-V2020.12.3.zip), whose per-question schema is
|
|
388
|
+
# identical — the CWA route was verified against it (2026-08-12): first 100
|
|
389
|
+
# theories of CWA/depth-2/meta-dev.jsonl = 1078/1078 questions correct
|
|
390
|
+
# (29 AttNoneg / 31 AttNeg / 19 RelNoneg / 21 RelNeg; z3 cross-check on the
|
|
391
|
+
# definite ones). Several of those theories NEED local stratification:
|
|
392
|
+
# their predicate graphs cycle through negation while their ground graphs
|
|
393
|
+
# do not — predicate-level stratification would refuse them.
|
|
394
|
+
# --------------------------------------------------------------------------- #
|
|
395
|
+
|
|
396
|
+
_REPR_TOKEN_RE = re.compile(r'\(|\)|->|"[^"]*"')
|
|
397
|
+
|
|
398
|
+
#: RuleTaker's rule-variable placeholder words (subject/object position of a
|
|
399
|
+
#: rule triple). Every DISTINCT placeholder in one rule gets its own
|
|
400
|
+
#: universally quantified variable, in order of first appearance: x, y, z.
|
|
401
|
+
_PLACEHOLDER_WORDS = ("something", "someone", "somebody")
|
|
402
|
+
|
|
403
|
+
_KIT_PRED_RE = re.compile(r"[A-Z][a-zA-Z0-9]*\Z")
|
|
404
|
+
_KIT_CONST_RE = re.compile(r"[a-z][a-zA-Z0-9]*[a-zA-Z][a-zA-Z0-9]*\Z")
|
|
405
|
+
|
|
406
|
+
|
|
407
|
+
def _tokenise_representation(rep: str) -> List[str]:
|
|
408
|
+
tokens = _REPR_TOKEN_RE.findall(rep)
|
|
409
|
+
remainder = _REPR_TOKEN_RE.sub("", rep).strip()
|
|
410
|
+
if remainder:
|
|
411
|
+
raise ValueError(
|
|
412
|
+
f"proofwriter: unrecognised material {remainder!r} in "
|
|
413
|
+
f"representation {rep!r}")
|
|
414
|
+
return tokens
|
|
415
|
+
|
|
416
|
+
|
|
417
|
+
def _parse_sexpr(tokens: List[str], pos: int):
|
|
418
|
+
"""One S-expression starting at ``pos`` → (parsed, next_pos).
|
|
419
|
+
|
|
420
|
+
Strings stay strings (quotes stripped); ``(`` … ``)`` becomes a list;
|
|
421
|
+
``->`` stays the literal marker token.
|
|
422
|
+
"""
|
|
423
|
+
token = tokens[pos]
|
|
424
|
+
if token == "(":
|
|
425
|
+
items = []
|
|
426
|
+
pos += 1
|
|
427
|
+
while pos < len(tokens) and tokens[pos] != ")":
|
|
428
|
+
item, pos = _parse_sexpr(tokens, pos)
|
|
429
|
+
items.append(item)
|
|
430
|
+
if pos >= len(tokens):
|
|
431
|
+
raise ValueError("proofwriter: unbalanced '(' in representation")
|
|
432
|
+
return items, pos + 1
|
|
433
|
+
if token == ")":
|
|
434
|
+
raise ValueError("proofwriter: unbalanced ')' in representation")
|
|
435
|
+
if token == "->":
|
|
436
|
+
return "->", pos + 1
|
|
437
|
+
return token[1:-1], pos + 1 # strip the quotes
|
|
438
|
+
|
|
439
|
+
|
|
440
|
+
def _camel(words: List[str]) -> str:
|
|
441
|
+
return words[0] + "".join(w[0].upper() + w[1:] for w in words[1:] if w)
|
|
442
|
+
|
|
443
|
+
|
|
444
|
+
def _to_predicate_name(word: str) -> str:
|
|
445
|
+
parts = [p for p in re.split(r"[\s\-]+", word.strip()) if p]
|
|
446
|
+
if not parts:
|
|
447
|
+
raise ValueError("proofwriter: empty predicate word")
|
|
448
|
+
joined = _camel(parts)
|
|
449
|
+
name = joined[0].upper() + joined[1:]
|
|
450
|
+
if not _KIT_PRED_RE.match(name):
|
|
451
|
+
raise ValueError(
|
|
452
|
+
f"proofwriter: {word!r} does not map to a legal kit predicate "
|
|
453
|
+
f"(got {name!r})")
|
|
454
|
+
return name
|
|
455
|
+
|
|
456
|
+
|
|
457
|
+
def _to_constant_name(word: str) -> str:
|
|
458
|
+
parts = [p for p in re.split(r"[\s\-]+", word.strip()) if p]
|
|
459
|
+
if not parts:
|
|
460
|
+
raise ValueError("proofwriter: empty entity word")
|
|
461
|
+
joined = _camel(parts)
|
|
462
|
+
name = joined[0].lower() + joined[1:]
|
|
463
|
+
if not _KIT_CONST_RE.match(name):
|
|
464
|
+
raise ValueError(
|
|
465
|
+
f"proofwriter: entity {word!r} does not map to a legal kit "
|
|
466
|
+
f"constant (got {name!r} — single-letter entities would lex as "
|
|
467
|
+
"variables and are refused)")
|
|
468
|
+
return name
|
|
469
|
+
|
|
470
|
+
|
|
471
|
+
def _term_for(word: str, variables: "Dict[str, Variable]") -> Node:
|
|
472
|
+
lowered = word.strip().lower()
|
|
473
|
+
if lowered in _PLACEHOLDER_WORDS:
|
|
474
|
+
if lowered not in variables:
|
|
475
|
+
if len(variables) >= 3:
|
|
476
|
+
raise ValueError(
|
|
477
|
+
"proofwriter: more than three distinct placeholder "
|
|
478
|
+
"words in one rule — outside the documented convention")
|
|
479
|
+
variables[lowered] = Variable("xyz"[len(variables)])
|
|
480
|
+
return variables[lowered]
|
|
481
|
+
return Constant(_to_constant_name(word))
|
|
482
|
+
|
|
483
|
+
|
|
484
|
+
def _triple_to_node(parsed, variables: "Dict[str, Variable]") -> Node:
|
|
485
|
+
if (not isinstance(parsed, list) or len(parsed) != 4
|
|
486
|
+
or any(not isinstance(p, str) for p in parsed)):
|
|
487
|
+
raise ValueError(
|
|
488
|
+
f"proofwriter: not a (subject predicate object polarity) triple: "
|
|
489
|
+
f"{parsed!r}")
|
|
490
|
+
subject, predicate, obj, polarity = parsed
|
|
491
|
+
if polarity not in ("+", "-", "~"):
|
|
492
|
+
raise ValueError(f"proofwriter: unknown polarity {polarity!r}")
|
|
493
|
+
if predicate.strip().lower() == "is":
|
|
494
|
+
atom = Atom(_to_predicate_name(obj), (_term_for(subject, variables),))
|
|
495
|
+
else:
|
|
496
|
+
atom = Atom(_to_predicate_name(predicate),
|
|
497
|
+
(_term_for(subject, variables), _term_for(obj, variables)))
|
|
498
|
+
return Not(atom) if polarity in ("-", "~") else atom
|
|
499
|
+
|
|
500
|
+
|
|
501
|
+
def parse_proofwriter_representation(rep: str) -> Node:
|
|
502
|
+
"""One ProofWriter ``representation`` string → a closed kit formula.
|
|
503
|
+
|
|
504
|
+
Accepts both shapes the OWA distribution uses: a bare triple (fact or
|
|
505
|
+
question) and a rule ``((cond …) -> conclusion)``. See the comment block
|
|
506
|
+
above for the exact translation convention; every distinct placeholder
|
|
507
|
+
word in a rule is universally quantified, so the result is always closed.
|
|
508
|
+
|
|
509
|
+
Raises:
|
|
510
|
+
ValueError: the string is not a well-formed representation, or a
|
|
511
|
+
name in it has no legal kit rendering.
|
|
512
|
+
"""
|
|
513
|
+
tokens = _tokenise_representation(rep)
|
|
514
|
+
if not tokens:
|
|
515
|
+
raise ValueError("proofwriter: empty representation")
|
|
516
|
+
parsed, next_pos = _parse_sexpr(tokens, 0)
|
|
517
|
+
if next_pos != len(tokens):
|
|
518
|
+
raise ValueError(
|
|
519
|
+
f"proofwriter: trailing tokens after the representation: {rep!r}")
|
|
520
|
+
|
|
521
|
+
variables: "Dict[str, Variable]" = {}
|
|
522
|
+
if isinstance(parsed, list) and len(parsed) == 3 and parsed[1] == "->":
|
|
523
|
+
conditions, _, conclusion = parsed
|
|
524
|
+
if not isinstance(conditions, list) or not conditions:
|
|
525
|
+
raise ValueError(
|
|
526
|
+
f"proofwriter: rule without conditions: {rep!r}")
|
|
527
|
+
cond_nodes = [_triple_to_node(c, variables) for c in conditions]
|
|
528
|
+
body = cond_nodes[0]
|
|
529
|
+
for extra in cond_nodes[1:]:
|
|
530
|
+
body = And(body, extra)
|
|
531
|
+
node: Node = Implies(body, _triple_to_node(conclusion, variables))
|
|
532
|
+
for var in reversed(list(variables.values())):
|
|
533
|
+
node = Quantifier("∀", var, node)
|
|
534
|
+
return node
|
|
535
|
+
return _triple_to_node(parsed, variables)
|
|
536
|
+
|
|
537
|
+
|
|
538
|
+
def _named_entries(section: Optional[dict], prefix: str) -> "List[Tuple[str, dict]]":
|
|
539
|
+
"""Non-null ``triple<N>``/``rule<N>``/``Q<N>`` entries in numeric order."""
|
|
540
|
+
if not isinstance(section, dict):
|
|
541
|
+
return []
|
|
542
|
+
entries = []
|
|
543
|
+
for key, value in section.items():
|
|
544
|
+
if value is None or not isinstance(value, dict):
|
|
545
|
+
continue
|
|
546
|
+
match = re.fullmatch(re.escape(prefix) + r"(\d+)", key)
|
|
547
|
+
order = int(match.group(1)) if match else float("inf")
|
|
548
|
+
entries.append((order, key, value))
|
|
549
|
+
entries.sort(key=lambda item: (item[0], item[1]))
|
|
550
|
+
return [(key, value) for _, key, value in entries]
|
|
551
|
+
|
|
552
|
+
|
|
553
|
+
def load_proofwriter_structured(
|
|
554
|
+
path: Union[str, Path], *,
|
|
555
|
+
known_bad_ids: FrozenSet[str] = frozenset(),
|
|
556
|
+
convert_fol: bool = True) -> Iterator[DatasetExample]:
|
|
557
|
+
"""Stream ONE example PER QUESTION from a structured-OWA JSONL file.
|
|
558
|
+
|
|
559
|
+
Args:
|
|
560
|
+
path: local ``.jsonl`` file with one theory record per line in the
|
|
561
|
+
``hitachi-nlp/proofwriter_processed_OWA`` schema (``id``,
|
|
562
|
+
``theory``, ``triples``, ``rules``, ``questions``, …; obtain it
|
|
563
|
+
e.g. via ``load_dataset("hitachi-nlp/proofwriter_processed_OWA",
|
|
564
|
+
"depth-2", split="train").to_json(...)``). NEVER downloaded here.
|
|
565
|
+
known_bad_ids: example ids (``proofwriter:<row id>:<Qn>``) to flag
|
|
566
|
+
``known_bad=True``.
|
|
567
|
+
convert_fol: with the default ``True``, ``fol_premises`` /
|
|
568
|
+
``fol_conclusion`` are GENERATED from the record's own symbolic
|
|
569
|
+
representations via :func:`parse_proofwriter_representation`
|
|
570
|
+
(``meta["fol_generated"]`` is set, the verbatim representations
|
|
571
|
+
stay in ``meta``); a record whose representations cannot be
|
|
572
|
+
converted yields its questions with empty FOL fields plus
|
|
573
|
+
``meta["fol_conversion_error"]``. ``False`` skips generation
|
|
574
|
+
entirely (NL + label only, like :func:`load_proofwriter`).
|
|
575
|
+
|
|
576
|
+
Yields:
|
|
577
|
+
One :class:`DatasetExample` per non-null question, in file order and
|
|
578
|
+
numeric question order; ``nl_premises`` are the fact/rule sentence
|
|
579
|
+
texts, ``label`` is the OWA answer verbatim
|
|
580
|
+
(``"True"``/``"False"``/``"Unknown"``). ``meta["proofs"]`` (the
|
|
581
|
+
record's own, unparsed ``question["proofs"]`` string) and
|
|
582
|
+
``meta["strategy"]`` were already carried verbatim before this
|
|
583
|
+
docstring was written; ``meta["premise_keys"]`` — the source
|
|
584
|
+
record's ``"tripleN"``/``"ruleN"`` keys, parallel to
|
|
585
|
+
``meta["premise_representations"]`` — is additive, resolving those
|
|
586
|
+
proof references back to a premise index for
|
|
587
|
+
:func:`check_gold_proof`. Parse ``meta["proofs"]`` with
|
|
588
|
+
:func:`~._proofwriter_proof.parse_question_proof` and verify it
|
|
589
|
+
against this module's own forward-chaining fixpoint with
|
|
590
|
+
:func:`check_gold_proof`.
|
|
591
|
+
|
|
592
|
+
Raises:
|
|
593
|
+
FileNotFoundError / json.JSONDecodeError: as in the other loaders —
|
|
594
|
+
a missing or corrupt FILE fails loudly; per-record conversion
|
|
595
|
+
problems are recorded per example instead.
|
|
596
|
+
"""
|
|
597
|
+
path = Path(path)
|
|
598
|
+
with path.open("r", encoding="utf-8") as fh:
|
|
599
|
+
for line_no, raw_line in enumerate(fh):
|
|
600
|
+
line = raw_line.strip()
|
|
601
|
+
if not line:
|
|
602
|
+
continue
|
|
603
|
+
record = json.loads(line)
|
|
604
|
+
row_id = record.get("id", f"pos{line_no}")
|
|
605
|
+
|
|
606
|
+
triples = _named_entries(record.get("triples"), "triple")
|
|
607
|
+
rules = _named_entries(record.get("rules"), "rule")
|
|
608
|
+
premise_entries = triples + rules
|
|
609
|
+
nl_premises = tuple(entry.get("text", "")
|
|
610
|
+
for _, entry in premise_entries)
|
|
611
|
+
premise_reps = tuple(entry.get("representation", "")
|
|
612
|
+
for _, entry in premise_entries)
|
|
613
|
+
premise_keys = tuple(key for key, _ in premise_entries)
|
|
614
|
+
|
|
615
|
+
fol_premises: "Tuple[str, ...]" = ()
|
|
616
|
+
conversion_error: Optional[str] = None
|
|
617
|
+
if convert_fol:
|
|
618
|
+
try:
|
|
619
|
+
fol_premises = tuple(
|
|
620
|
+
parse_proofwriter_representation(rep).to_unicode_str()
|
|
621
|
+
for rep in premise_reps)
|
|
622
|
+
except ValueError as exc:
|
|
623
|
+
conversion_error = f"{type(exc).__name__}: {exc}"
|
|
624
|
+
|
|
625
|
+
for q_key, question in _named_entries(record.get("questions"), "Q"):
|
|
626
|
+
example_id = f"proofwriter:{row_id}:{q_key}"
|
|
627
|
+
q_rep = question.get("representation", "")
|
|
628
|
+
fol_conclusion: Optional[str] = None
|
|
629
|
+
q_error = conversion_error
|
|
630
|
+
if convert_fol and q_error is None:
|
|
631
|
+
try:
|
|
632
|
+
fol_conclusion = parse_proofwriter_representation(
|
|
633
|
+
q_rep).to_unicode_str()
|
|
634
|
+
except ValueError as exc:
|
|
635
|
+
q_error = f"{type(exc).__name__}: {exc}"
|
|
636
|
+
|
|
637
|
+
meta = {
|
|
638
|
+
"row_id": row_id,
|
|
639
|
+
"question_key": q_key,
|
|
640
|
+
"theory": record.get("theory"),
|
|
641
|
+
"premise_representations": list(premise_reps),
|
|
642
|
+
# Parallel to premise_representations/fol_premises: the
|
|
643
|
+
# source record's own "tripleN"/"ruleN" keys, in the SAME
|
|
644
|
+
# order -- added purely so a caller can resolve a
|
|
645
|
+
# question["proofs"] annotation's "tripleN"/"ruleN"
|
|
646
|
+
# references back to a premise index (see
|
|
647
|
+
# check_gold_proof). Every OTHER field here was already
|
|
648
|
+
# present before that addition; this key is new and
|
|
649
|
+
# purely additive, nothing existing was removed/renamed.
|
|
650
|
+
"premise_keys": list(premise_keys),
|
|
651
|
+
"question_representation": q_rep,
|
|
652
|
+
"QDep": question.get("QDep"),
|
|
653
|
+
"strategy": question.get("strategy"),
|
|
654
|
+
"proofs": question.get("proofs"),
|
|
655
|
+
"line_no": line_no,
|
|
656
|
+
}
|
|
657
|
+
if convert_fol and q_error is None:
|
|
658
|
+
meta["fol_generated"] = True
|
|
659
|
+
if q_error is not None:
|
|
660
|
+
meta["fol_conversion_error"] = q_error
|
|
661
|
+
|
|
662
|
+
yield DatasetExample(
|
|
663
|
+
id=example_id,
|
|
664
|
+
nl_premises=nl_premises,
|
|
665
|
+
fol_premises=fol_premises if q_error is None else (),
|
|
666
|
+
nl_conclusion=question.get("question"),
|
|
667
|
+
fol_conclusion=fol_conclusion,
|
|
668
|
+
label=question.get("answer"),
|
|
669
|
+
known_bad=example_id in known_bad_ids,
|
|
670
|
+
meta=meta,
|
|
671
|
+
)
|
|
672
|
+
|
|
673
|
+
|
|
674
|
+
def _collect_constants(nodes) -> "List[Constant]":
|
|
675
|
+
"""Every distinct :class:`Constant` in ``nodes``, in a fixed name order."""
|
|
676
|
+
by_name: "Dict[str, Constant]" = {}
|
|
677
|
+
for node in nodes:
|
|
678
|
+
for sub in node.walk():
|
|
679
|
+
if isinstance(sub, Constant):
|
|
680
|
+
by_name.setdefault(sub.name, sub)
|
|
681
|
+
return [by_name[name] for name in sorted(by_name)]
|
|
682
|
+
|
|
683
|
+
|
|
684
|
+
def _as_literal(node: Node):
|
|
685
|
+
"""``(atom, positive)`` for a literal node, else ``ValueError``."""
|
|
686
|
+
if isinstance(node, Atom):
|
|
687
|
+
return node, True
|
|
688
|
+
if isinstance(node, Not) and isinstance(node.formula, Atom):
|
|
689
|
+
return node.formula, False
|
|
690
|
+
raise ValueError(
|
|
691
|
+
f"proofwriter: {node.to_unicode_str()!r} is not a literal — the CWA "
|
|
692
|
+
"theory fragment is ground literals and (∀-quantified) rules "
|
|
693
|
+
"'literal ∧ … ∧ literal → literal'.")
|
|
694
|
+
|
|
695
|
+
|
|
696
|
+
def _as_rule(premise: Node):
|
|
697
|
+
"""Decompose one premise into ``(variables, body_literals, head_literal)``.
|
|
698
|
+
|
|
699
|
+
A ground literal comes back with an empty body (a fact). Anything outside
|
|
700
|
+
the fact/rule shape raises ``ValueError``.
|
|
701
|
+
"""
|
|
702
|
+
variables: "List[Variable]" = []
|
|
703
|
+
node = premise
|
|
704
|
+
while isinstance(node, Quantifier):
|
|
705
|
+
if node.type not in ("∀", "forall"):
|
|
706
|
+
raise ValueError(
|
|
707
|
+
f"proofwriter: CWA premises must be universally quantified, "
|
|
708
|
+
f"got {node.type!r} in {premise.to_unicode_str()!r}")
|
|
709
|
+
variables.append(node.variable)
|
|
710
|
+
node = node.formula
|
|
711
|
+
if isinstance(node, Implies):
|
|
712
|
+
body: "List[Node]" = []
|
|
713
|
+
stack = [node.left]
|
|
714
|
+
while stack:
|
|
715
|
+
item = stack.pop()
|
|
716
|
+
if isinstance(item, And):
|
|
717
|
+
stack.append(item.right)
|
|
718
|
+
stack.append(item.left)
|
|
719
|
+
else:
|
|
720
|
+
body.append(item)
|
|
721
|
+
body_literals = [_as_literal(b) for b in body]
|
|
722
|
+
head_literal = _as_literal(node.right)
|
|
723
|
+
return variables, body_literals, head_literal
|
|
724
|
+
if variables:
|
|
725
|
+
raise ValueError(
|
|
726
|
+
f"proofwriter: quantified premise without an implication is "
|
|
727
|
+
f"outside the CWA fragment: {premise.to_unicode_str()!r}")
|
|
728
|
+
return [], [], _as_literal(node)
|
|
729
|
+
|
|
730
|
+
|
|
731
|
+
def _ground_rule(rule, constants: "List[Constant]"):
|
|
732
|
+
"""One ``(variables, body, head)`` (see :func:`_as_rule`) → its list of
|
|
733
|
+
GROUND ``(body, head)`` instances over ``constants`` (``[]`` if the rule
|
|
734
|
+
is variable-free — a plain ground fact/implication grounds to itself)."""
|
|
735
|
+
variables, body, head = rule
|
|
736
|
+
if not variables:
|
|
737
|
+
return [(body, head)]
|
|
738
|
+
if not constants:
|
|
739
|
+
return []
|
|
740
|
+
groundings = []
|
|
741
|
+
for values in itertools.product(constants, repeat=len(variables)):
|
|
742
|
+
g_body = []
|
|
743
|
+
for atom, positive in body:
|
|
744
|
+
g_atom = atom
|
|
745
|
+
for var, value in zip(variables, values):
|
|
746
|
+
g_atom = substitute(g_atom, var, value)
|
|
747
|
+
g_body.append((g_atom, positive))
|
|
748
|
+
h_atom, h_positive = head
|
|
749
|
+
for var, value in zip(variables, values):
|
|
750
|
+
h_atom = substitute(h_atom, var, value)
|
|
751
|
+
groundings.append((g_body, (h_atom, h_positive)))
|
|
752
|
+
return groundings
|
|
753
|
+
|
|
754
|
+
|
|
755
|
+
def _closed_model(premises: "List[Node]", constants: "List[Constant]", *,
|
|
756
|
+
record_provenance: bool = False):
|
|
757
|
+
"""The theory's closed model, by STRATIFIED forward chaining.
|
|
758
|
+
|
|
759
|
+
Returns ``(true_atoms, has_naf)`` — the set of ground-atom keys
|
|
760
|
+
(``key_text``: the text of the atom with every constant written by its name,
|
|
761
|
+
which is not the text of the formula where the formula writes a constant in
|
|
762
|
+
quotes) true in the perfect model, and whether any rule
|
|
763
|
+
body used negation (i.e. the theory is beyond the definite fragment, so
|
|
764
|
+
membership is NOT classical entailment and must not be cross-checked
|
|
765
|
+
against a classical prover). With ``record_provenance=True`` a THIRD
|
|
766
|
+
element is returned, ``provenance`` (see below); with the default
|
|
767
|
+
``False`` the return shape is EXACTLY the 2-tuple above, unchanged —
|
|
768
|
+
every existing caller (:func:`solve_structured_example`) is untouched.
|
|
769
|
+
|
|
770
|
+
Semantics: negation in a rule body is NEGATION AS FAILURE, evaluated
|
|
771
|
+
stratum by stratum over the GROUND dependency graph — LOCAL
|
|
772
|
+
stratification: a ground atom that is tested negatively is fully
|
|
773
|
+
fixpointed in a strictly LOWER stratum before any ground rule reading
|
|
774
|
+
its absence may fire, which is exactly the perfect-model semantics of
|
|
775
|
+
locally stratified logic programs. Stratifying ground atoms rather than
|
|
776
|
+
predicate symbols is strictly more general and is required in practice:
|
|
777
|
+
real ProofWriter theories contain rules like ``¬Likes(mouse, dog) →
|
|
778
|
+
Likes(dog, rabbit)`` whose predicate graph has a negative self-loop but
|
|
779
|
+
whose ground graph is acyclic. Only a GROUND cycle through negation
|
|
780
|
+
(e.g. ``¬P(a) → P(a)``) has no local stratification (it would need
|
|
781
|
+
well-founded/stable-model semantics) and raises ``ValueError``. A
|
|
782
|
+
negative HEAD (a rule or fact concluding ``¬X``) derives no positive
|
|
783
|
+
atom; it is evaluated once against the COMPLETED perfect model and
|
|
784
|
+
tracked only for the final consistency check — a theory that derives
|
|
785
|
+
some atom both positively and negatively is inconsistent under CWA and
|
|
786
|
+
raises rather than answering arbitrarily.
|
|
787
|
+
|
|
788
|
+
``provenance`` (only computed when ``record_provenance=True``, for
|
|
789
|
+
:func:`check_gold_proof`): ``{signed_key: [(premise_index,
|
|
790
|
+
antecedent_keys), …]}`` — ``signed_key`` is the SIGNED rendering of a
|
|
791
|
+
derived atom (``"Foo(a)"`` for a positive derivation, ``"¬Foo(a)"`` for
|
|
792
|
+
a negative one, via a negative-headed rule or fact — real ProofWriter
|
|
793
|
+
theories have both, see the module's ``check_gold_proof`` docstring);
|
|
794
|
+
each ``premise_index`` is the position, in the ORIGINAL ``premises``
|
|
795
|
+
list, of the fact/rule whose grounding fired; ``antecedent_keys`` is the
|
|
796
|
+
ordered tuple of each required body literal's OWN SIGNED key (bare for a
|
|
797
|
+
positive requirement, ``"¬"``-prefixed for a negation-as-failure one) —
|
|
798
|
+
``()`` for a fact. Real ProofWriter theories DO use negation-as-failure
|
|
799
|
+
conditions (209 of 2401 real rules fetched from ``hitachi-nlp/
|
|
800
|
+
proofwriter_processed_OWA`` — depths 0/1/2/3/3ext/3ext-NatLang/5/
|
|
801
|
+
NatLang/birds-electricity — have one; ProofWriter's OWN grammar spells
|
|
802
|
+
this ``"~"``, distinct from the ``"-"`` a FACT or rule HEAD uses for a
|
|
803
|
+
flat negative assertion — see :func:`parse_proofwriter_representation`'s
|
|
804
|
+
``_triple_to_node``), and its own gold ``proofs`` annotation cites an
|
|
805
|
+
EXPLICIT derivation of ``¬X`` for such a condition whenever the theory
|
|
806
|
+
has one (rather than leaving it uncited as bare absence) — signing
|
|
807
|
+
``antecedent_keys`` is what lets :func:`check_gold_proof` look each one
|
|
808
|
+
up as a recursive ``signed_key`` regardless of polarity. A
|
|
809
|
+
``signed_key`` can have SEVERAL entries (several grounded rules/facts
|
|
810
|
+
derived it — the "OR-forest" a gold proof may cite any one branch of).
|
|
811
|
+
Signing ``antecedent_keys`` does NOT change ``true_atoms``/``stratum``
|
|
812
|
+
below: NAF satisfaction there is still checked by plain ABSENCE from
|
|
813
|
+
``true_atoms`` (``(key_text(atom) in true_atoms) == positive``),
|
|
814
|
+
exactly ProofWriter's own stratified-fixpoint semantics; ``provenance``
|
|
815
|
+
is a side record of what ALSO fired explicitly, not a different
|
|
816
|
+
satisfaction rule.
|
|
817
|
+
"""
|
|
818
|
+
parsed = [_as_rule(p) for p in premises]
|
|
819
|
+
has_naf = any(not positive for _, body, _ in parsed
|
|
820
|
+
for _, positive in body)
|
|
821
|
+
|
|
822
|
+
# Ground every rule over the finite constant set, remembering which
|
|
823
|
+
# ORIGINAL premise (index into `premises`) each grounding came from --
|
|
824
|
+
# needed only for `provenance`, but cheap enough to always compute.
|
|
825
|
+
ground_rules: "List[Tuple[list, tuple]]" = []
|
|
826
|
+
origin: "List[int]" = []
|
|
827
|
+
for premise_index, rule in enumerate(parsed):
|
|
828
|
+
for grounding in _ground_rule(rule, constants):
|
|
829
|
+
ground_rules.append(grounding)
|
|
830
|
+
origin.append(premise_index)
|
|
831
|
+
|
|
832
|
+
# LOCAL stratification: strata live on GROUND ATOMS. A positive body
|
|
833
|
+
# atom forces its head onto the same-or-higher stratum, a negated one
|
|
834
|
+
# onto a strictly higher stratum. The iteration is Bellman-Ford-like:
|
|
835
|
+
# each atom's stratum is bounded by #atoms on any locally stratifiable
|
|
836
|
+
# program, so non-convergence within #atoms+1 rounds means a ground
|
|
837
|
+
# cycle through negation. Negative-head rules derive nothing and are
|
|
838
|
+
# left out of the graph (they are evaluated against the finished model
|
|
839
|
+
# below).
|
|
840
|
+
stratum: "Dict[str, int]" = {}
|
|
841
|
+
for body, (head_atom, _) in ground_rules:
|
|
842
|
+
stratum.setdefault(key_text(head_atom), 0)
|
|
843
|
+
for atom, _ in body:
|
|
844
|
+
stratum.setdefault(key_text(atom), 0)
|
|
845
|
+
for _ in range(len(stratum) + 1):
|
|
846
|
+
changed = False
|
|
847
|
+
for body, (head_atom, head_positive) in ground_rules:
|
|
848
|
+
if not head_positive:
|
|
849
|
+
continue
|
|
850
|
+
head_key = key_text(head_atom)
|
|
851
|
+
for atom, positive in body:
|
|
852
|
+
required = (stratum[key_text(atom)]
|
|
853
|
+
+ (0 if positive else 1))
|
|
854
|
+
if stratum[head_key] < required:
|
|
855
|
+
stratum[head_key] = required
|
|
856
|
+
changed = True
|
|
857
|
+
if not changed:
|
|
858
|
+
break
|
|
859
|
+
else:
|
|
860
|
+
raise ValueError(
|
|
861
|
+
"proofwriter: the CWA theory has a GROUND cycle through negation "
|
|
862
|
+
"(not even locally stratifiable) — its negation-as-failure "
|
|
863
|
+
"semantics is not well-defined by stratified forward chaining; "
|
|
864
|
+
"refusing rather than guessing (well-founded semantics is out of "
|
|
865
|
+
"scope).")
|
|
866
|
+
|
|
867
|
+
provenance: "Optional[Dict[str, set]]" = {} if record_provenance else None
|
|
868
|
+
|
|
869
|
+
def _record(signed_key: str, gi: int, body) -> None:
|
|
870
|
+
if provenance is None:
|
|
871
|
+
return
|
|
872
|
+
# SIGNED per literal (bare for a positive requirement, "¬"-prefixed
|
|
873
|
+
# for a negation-as-failure one) -- a NAF antecedent's signed key is
|
|
874
|
+
# what a gold proof cites when it names an EXPLICIT negative
|
|
875
|
+
# derivation for it (see check_gold_proof's docstring on why real
|
|
876
|
+
# ProofWriter proofs prefer a constructive ¬X derivation over silent
|
|
877
|
+
# absence whenever one exists).
|
|
878
|
+
ante_keys = tuple(
|
|
879
|
+
key_text(atom) if positive else key_text(Not(atom))
|
|
880
|
+
for atom, positive in body)
|
|
881
|
+
provenance.setdefault(signed_key, set()).add((origin[gi], ante_keys))
|
|
882
|
+
|
|
883
|
+
true_atoms: set = set()
|
|
884
|
+
negative_atoms: set = set()
|
|
885
|
+
positive_rules = [
|
|
886
|
+
(gi, body, head) for gi, (body, head) in enumerate(ground_rules)
|
|
887
|
+
if head[1]
|
|
888
|
+
]
|
|
889
|
+
max_stratum = max(
|
|
890
|
+
(stratum[key_text(head[0])] for _, _, head in positive_rules),
|
|
891
|
+
default=0)
|
|
892
|
+
for level in range(max_stratum + 1):
|
|
893
|
+
level_rules = [
|
|
894
|
+
(gi, body, head) for gi, body, head in positive_rules
|
|
895
|
+
if stratum[key_text(head[0])] == level
|
|
896
|
+
]
|
|
897
|
+
changed = True
|
|
898
|
+
while changed:
|
|
899
|
+
changed = False
|
|
900
|
+
for gi, body, (head_atom, _) in level_rules:
|
|
901
|
+
fires = all(
|
|
902
|
+
(key_text(atom) in true_atoms) == positive
|
|
903
|
+
for atom, positive in body
|
|
904
|
+
)
|
|
905
|
+
if not fires:
|
|
906
|
+
continue
|
|
907
|
+
key = key_text(head_atom)
|
|
908
|
+
_record(key, gi, body)
|
|
909
|
+
if key not in true_atoms:
|
|
910
|
+
true_atoms.add(key)
|
|
911
|
+
changed = True
|
|
912
|
+
|
|
913
|
+
# Negative heads read the COMPLETED model (previously they fired at
|
|
914
|
+
# their head's level, silently missing body atoms derived later).
|
|
915
|
+
for gi, (body, (head_atom, head_positive)) in enumerate(ground_rules):
|
|
916
|
+
if head_positive:
|
|
917
|
+
continue
|
|
918
|
+
if all((key_text(atom) in true_atoms) == positive
|
|
919
|
+
for atom, positive in body):
|
|
920
|
+
negative_atoms.add(key_text(head_atom))
|
|
921
|
+
_record(key_text(Not(head_atom)), gi, body)
|
|
922
|
+
|
|
923
|
+
contradictions = sorted(true_atoms & negative_atoms)
|
|
924
|
+
if contradictions:
|
|
925
|
+
raise ValueError(
|
|
926
|
+
f"proofwriter: the CWA theory derives {contradictions[0]!r} both "
|
|
927
|
+
"positively and negatively — inconsistent under the closed-world "
|
|
928
|
+
"reading; refusing rather than answering arbitrarily.")
|
|
929
|
+
if record_provenance:
|
|
930
|
+
ordered_provenance = {
|
|
931
|
+
key: sorted(entries) for key, entries in provenance.items()
|
|
932
|
+
}
|
|
933
|
+
return frozenset(true_atoms), has_naf, ordered_provenance
|
|
934
|
+
return frozenset(true_atoms), has_naf
|
|
935
|
+
|
|
936
|
+
|
|
937
|
+
def _cwa_holds(formula: Node, atom_oracle, constants) -> bool:
|
|
938
|
+
"""Two-valued truth of ``formula`` in the CLOSED model.
|
|
939
|
+
|
|
940
|
+
The closed (minimal) model is represented by its atom valuation:
|
|
941
|
+
``atom_oracle(atom)`` answers "is this ground atom derivable from the
|
|
942
|
+
premises?" — everything not derivable is FALSE, which is exactly the
|
|
943
|
+
closed-world reading. Connectives are ordinary two-valued FOL on top of
|
|
944
|
+
that valuation, and quantifiers range over the theory's (finite) set of
|
|
945
|
+
constants — plain model checking, not entailment.
|
|
946
|
+
"""
|
|
947
|
+
if isinstance(formula, Atom):
|
|
948
|
+
return atom_oracle(formula)
|
|
949
|
+
if isinstance(formula, Not):
|
|
950
|
+
return not _cwa_holds(formula.formula, atom_oracle, constants)
|
|
951
|
+
if isinstance(formula, And):
|
|
952
|
+
return (_cwa_holds(formula.left, atom_oracle, constants)
|
|
953
|
+
and _cwa_holds(formula.right, atom_oracle, constants))
|
|
954
|
+
if isinstance(formula, Or):
|
|
955
|
+
return (_cwa_holds(formula.left, atom_oracle, constants)
|
|
956
|
+
or _cwa_holds(formula.right, atom_oracle, constants))
|
|
957
|
+
if isinstance(formula, Xor):
|
|
958
|
+
return (_cwa_holds(formula.left, atom_oracle, constants)
|
|
959
|
+
!= _cwa_holds(formula.right, atom_oracle, constants))
|
|
960
|
+
if isinstance(formula, Implies):
|
|
961
|
+
return ((not _cwa_holds(formula.left, atom_oracle, constants))
|
|
962
|
+
or _cwa_holds(formula.right, atom_oracle, constants))
|
|
963
|
+
if isinstance(formula, Iff):
|
|
964
|
+
return (_cwa_holds(formula.left, atom_oracle, constants)
|
|
965
|
+
== _cwa_holds(formula.right, atom_oracle, constants))
|
|
966
|
+
if isinstance(formula, Quantifier):
|
|
967
|
+
instances = (
|
|
968
|
+
_cwa_holds(substitute(formula.formula, formula.variable, c),
|
|
969
|
+
atom_oracle, constants)
|
|
970
|
+
for c in constants)
|
|
971
|
+
if formula.type in ("∀", "forall"):
|
|
972
|
+
return all(instances)
|
|
973
|
+
if formula.type in ("∃", "exists"):
|
|
974
|
+
return any(instances)
|
|
975
|
+
raise ValueError(
|
|
976
|
+
f"proofwriter: unknown quantifier {formula.type!r} in a CWA query")
|
|
977
|
+
raise ValueError(
|
|
978
|
+
f"proofwriter: node {type(formula).__name__} is outside the CWA "
|
|
979
|
+
"query fragment (atoms, ¬ ∧ ∨ ⊕ → ↔, ∀/∃ over the constants)")
|
|
980
|
+
|
|
981
|
+
|
|
982
|
+
def solve_structured_example(example: DatasetExample, *,
|
|
983
|
+
semantics: str = "owa",
|
|
984
|
+
on_indefinite: str = "label",
|
|
985
|
+
**prove_kwargs) -> dict:
|
|
986
|
+
"""Decide one generated-FOL ProofWriter example against its gold label.
|
|
987
|
+
|
|
988
|
+
The prover is the CALLER'S choice: every keyword in ``prove_kwargs`` goes
|
|
989
|
+
verbatim to :func:`unicode_logic_kit.api.prove` — e.g.
|
|
990
|
+
``backends=["vampire"]`` or ``backends=["z3"], timeout=5000`` — so any
|
|
991
|
+
registered ATP can drive either semantics.
|
|
992
|
+
|
|
993
|
+
``semantics="owa"`` (default) is the open-world three-way cascade over
|
|
994
|
+
ENTAILMENT: premises ⊨ q → ``"True"``; else premises ⊨ ¬q → ``"False"``;
|
|
995
|
+
else ``"Unknown"``. This maps the OWA labels exactly.
|
|
996
|
+
|
|
997
|
+
``on_indefinite`` controls how a NON-DEFINITIVE prover outcome (a
|
|
998
|
+
``Verdict`` with status ``unknown`` or ``error`` — timeouts, hit bounds,
|
|
999
|
+
honest incompleteness; ``Verdict.is_definitive`` is the exact test) is
|
|
1000
|
+
interpreted when the OWA cascade reaches its third arm. The distinction
|
|
1001
|
+
it preserves: ``"Unknown"`` can be ESTABLISHED (both directions
|
|
1002
|
+
definitively REFUTED — the prover found countermodels both ways, so the
|
|
1003
|
+
question is provably underdetermined) or merely DEFAULTED to (some leg
|
|
1004
|
+
timed out / gave up — the prover failed to tell).
|
|
1005
|
+
|
|
1006
|
+
- ``"label"`` (default): any not-proved outcome flows into the dataset's
|
|
1007
|
+
``"Unknown"`` label — the pragmatic scoring mode, correct whenever the
|
|
1008
|
+
chosen prover is decisive on the fragment (z3 on these ground/Horn
|
|
1009
|
+
theories is).
|
|
1010
|
+
- ``"abstain"``: ``"Unknown"`` only when BOTH legs are definitively
|
|
1011
|
+
REFUTED; if any leg is indefinite, ``predicted`` is ``None`` — so
|
|
1012
|
+
evaluation numbers cannot silently credit a prover timeout as a
|
|
1013
|
+
correct "Unknown" prediction. The verdict dicts show which leg failed
|
|
1014
|
+
and why (``status``/``reason``, e.g. ``timeout``).
|
|
1015
|
+
- ``"raise"``: like ``"abstain"``, but an indefinite leg raises
|
|
1016
|
+
``ValueError`` — for pipelines that must not contain holes.
|
|
1017
|
+
|
|
1018
|
+
Under ``semantics="cwa"`` the fixpoint decides every atom, so
|
|
1019
|
+
``on_indefinite`` has no effect there (the cross-check already records,
|
|
1020
|
+
and never alarms on, an indefinite prover verdict); the argument is
|
|
1021
|
+
still validated.
|
|
1022
|
+
|
|
1023
|
+
``semantics="cwa"`` is closed-world MODEL CHECKING — ordinary two-valued
|
|
1024
|
+
FOL evaluation in the closed model, which is COMPUTED exactly by
|
|
1025
|
+
stratified forward chaining over the grounded theory
|
|
1026
|
+
(:func:`_closed_model`): a ground atom is true iff it is in the perfect
|
|
1027
|
+
model, and negation/connectives/quantifiers in the QUERY are evaluated
|
|
1028
|
+
compositionally on top (¬q is true iff q is not in the model; ∀/∃ range
|
|
1029
|
+
over the theory's constants). There is no ``"Unknown"``: the result is
|
|
1030
|
+
``"True"`` or ``"False"``. Rules with NEGATED BODY literals are
|
|
1031
|
+
supported with their standard negation-as-failure reading via LOCAL
|
|
1032
|
+
stratification over the ground dependency graph (the negatively-tested
|
|
1033
|
+
ground atom is fully fixpointed in a lower stratum first); only a
|
|
1034
|
+
GROUND cycle through negation (no local stratification exists) or a
|
|
1035
|
+
theory that derives an atom both positively and negatively
|
|
1036
|
+
(inconsistent under CWA) raises ``ValueError``. On
|
|
1037
|
+
DEFINITE theories (no negated bodies) the least model coincides with
|
|
1038
|
+
classical entailment, so every queried atom is additionally
|
|
1039
|
+
CROSS-CHECKED against the caller's chosen ATP — a definitive
|
|
1040
|
+
disagreement (prover proves an atom the fixpoint excludes, or refutes
|
|
1041
|
+
one it contains) raises a soundness alarm instead of returning either
|
|
1042
|
+
answer; an honest prover UNKNOWN is recorded, never alarmed on.
|
|
1043
|
+
|
|
1044
|
+
Returns ``{"predicted": ..., "verdict": ..., "verdict_negated": ...}``;
|
|
1045
|
+
under ``"cwa"`` the two verdict slots are ``None`` and every per-atom
|
|
1046
|
+
oracle call is recorded in an additional ``"atom_calls"`` list
|
|
1047
|
+
(atom, derivable, full verdict dict). Requires an example produced with
|
|
1048
|
+
``convert_fol=True`` and without a recorded conversion error; anything
|
|
1049
|
+
else raises ``ValueError``.
|
|
1050
|
+
"""
|
|
1051
|
+
from ... import api
|
|
1052
|
+
|
|
1053
|
+
if semantics not in ("owa", "cwa"):
|
|
1054
|
+
raise ValueError(
|
|
1055
|
+
f"proofwriter: semantics must be 'owa' or 'cwa', got {semantics!r}")
|
|
1056
|
+
if on_indefinite not in ("label", "abstain", "raise"):
|
|
1057
|
+
raise ValueError(
|
|
1058
|
+
f"proofwriter: on_indefinite must be 'label', 'abstain' or "
|
|
1059
|
+
f"'raise', got {on_indefinite!r}")
|
|
1060
|
+
if example.meta.get("fol_conversion_error"):
|
|
1061
|
+
raise ValueError(
|
|
1062
|
+
f"proofwriter: example {example.id} carries a conversion error "
|
|
1063
|
+
f"({example.meta['fol_conversion_error']}) — cannot solve it.")
|
|
1064
|
+
if example.fol_conclusion is None:
|
|
1065
|
+
raise ValueError(
|
|
1066
|
+
f"proofwriter: example {example.id} has no generated conclusion "
|
|
1067
|
+
"— was it loaded with convert_fol=False?")
|
|
1068
|
+
|
|
1069
|
+
def _parse(text: str) -> Node:
|
|
1070
|
+
parsed = api.parse_any(text)
|
|
1071
|
+
if not parsed.ok:
|
|
1072
|
+
raise ValueError(
|
|
1073
|
+
f"proofwriter: example {example.id}: generated formula "
|
|
1074
|
+
f"{text!r} does not parse under the kit grammar")
|
|
1075
|
+
return parsed.formula
|
|
1076
|
+
|
|
1077
|
+
premises = [_parse(p) for p in example.fol_premises]
|
|
1078
|
+
conclusion = _parse(example.fol_conclusion)
|
|
1079
|
+
|
|
1080
|
+
if semantics == "owa":
|
|
1081
|
+
verdict = api.prove(conclusion, premises, **prove_kwargs)
|
|
1082
|
+
if verdict.status == "proved":
|
|
1083
|
+
return {"predicted": "True", "verdict": verdict.to_dict(),
|
|
1084
|
+
"verdict_negated": None}
|
|
1085
|
+
negated = api.prove(Not(conclusion), premises, **prove_kwargs)
|
|
1086
|
+
if negated.status == "proved":
|
|
1087
|
+
predicted: "Optional[str]" = "False"
|
|
1088
|
+
elif on_indefinite == "label":
|
|
1089
|
+
predicted = "Unknown"
|
|
1090
|
+
elif verdict.status == "refuted" and negated.status == "refuted":
|
|
1091
|
+
# Underdetermination ESTABLISHED: countermodels exist against
|
|
1092
|
+
# both directions — "Unknown" is a definitive answer here, not
|
|
1093
|
+
# a fallback, so abstain/raise modes still label it.
|
|
1094
|
+
predicted = "Unknown"
|
|
1095
|
+
elif on_indefinite == "raise":
|
|
1096
|
+
raise ValueError(
|
|
1097
|
+
f"proofwriter: example {example.id}: indefinite prover "
|
|
1098
|
+
f"outcome (goal: {verdict.status}/{verdict.reason}, negated: "
|
|
1099
|
+
f"{negated.status}/{negated.reason}) with "
|
|
1100
|
+
"on_indefinite='raise' — the cascade cannot honestly assign "
|
|
1101
|
+
"a label.")
|
|
1102
|
+
else: # "abstain"
|
|
1103
|
+
predicted = None
|
|
1104
|
+
return {"predicted": predicted, "verdict": verdict.to_dict(),
|
|
1105
|
+
"verdict_negated": negated.to_dict()}
|
|
1106
|
+
|
|
1107
|
+
constants = _collect_constants(premises + [conclusion])
|
|
1108
|
+
model, has_naf = _closed_model(premises, constants)
|
|
1109
|
+
atom_calls: "List[dict]" = []
|
|
1110
|
+
cache: "Dict[str, bool]" = {}
|
|
1111
|
+
|
|
1112
|
+
def atom_oracle(atom: Atom) -> bool:
|
|
1113
|
+
key = key_text(atom)
|
|
1114
|
+
if key not in cache:
|
|
1115
|
+
in_model = key in model
|
|
1116
|
+
record: dict = {"atom": key, "derivable": in_model}
|
|
1117
|
+
if not has_naf:
|
|
1118
|
+
# Definite theory: least model ⟺ classical entailment, so
|
|
1119
|
+
# the caller's ATP serves as an independent cross-check. Only
|
|
1120
|
+
# a DEFINITIVE disagreement is a soundness alarm; an honest
|
|
1121
|
+
# UNKNOWN (timeout, bound) is recorded, not alarmed on.
|
|
1122
|
+
verdict = api.prove(atom, premises, **prove_kwargs)
|
|
1123
|
+
record["verdict"] = verdict.to_dict()
|
|
1124
|
+
if ((verdict.status == "proved" and not in_model)
|
|
1125
|
+
or (verdict.status == "refuted" and in_model)):
|
|
1126
|
+
raise ValueError(
|
|
1127
|
+
f"proofwriter: soundness alarm on {example.id}: the "
|
|
1128
|
+
f"closed-model fixpoint says {key!r} is "
|
|
1129
|
+
f"{'in' if in_model else 'NOT in'} the least model, "
|
|
1130
|
+
f"but backend {verdict.backend!r} definitively says "
|
|
1131
|
+
"the opposite — on a definite theory these must "
|
|
1132
|
+
"coincide; refusing to answer.")
|
|
1133
|
+
cache[key] = in_model
|
|
1134
|
+
atom_calls.append(record)
|
|
1135
|
+
return cache[key]
|
|
1136
|
+
|
|
1137
|
+
holds = _cwa_holds(conclusion, atom_oracle, constants)
|
|
1138
|
+
return {"predicted": "True" if holds else "False",
|
|
1139
|
+
"verdict": None, "verdict_negated": None,
|
|
1140
|
+
"atom_calls": atom_calls}
|
|
1141
|
+
|
|
1142
|
+
|
|
1143
|
+
# --------------------------------------------------------------------------- #
|
|
1144
|
+
# check_gold_proof: verify a structured-route example's OWN question["proofs"]
|
|
1145
|
+
# annotation against the kit's own forward-chaining fixpoint — see
|
|
1146
|
+
# _proofwriter_proof.py for the annotation grammar this parses.
|
|
1147
|
+
# --------------------------------------------------------------------------- #
|
|
1148
|
+
|
|
1149
|
+
def _negate(node: Node) -> Node:
|
|
1150
|
+
"""Logical negation WITHOUT double-negating: ``¬X`` → ``X``, else ``¬``."""
|
|
1151
|
+
return node.formula if isinstance(node, Not) else Not(node)
|
|
1152
|
+
|
|
1153
|
+
|
|
1154
|
+
#: Every strategy tag observed across 390 real rows / 5452 questions fetched
|
|
1155
|
+
#: from ``hitachi-nlp/proofwriter_processed_OWA`` (every published config)
|
|
1156
|
+
#: plus the real AllenAI CWA fixture — see :func:`_target_for_strategy`.
|
|
1157
|
+
_KNOWN_STRATEGIES = frozenset({
|
|
1158
|
+
"proof", "inv-proof", "rconc", "inv-rconc", "random", "inv-random",
|
|
1159
|
+
})
|
|
1160
|
+
|
|
1161
|
+
|
|
1162
|
+
def _target_for_strategy(conclusion: Node, strategy: "Optional[str]",
|
|
1163
|
+
example_id: str) -> Node:
|
|
1164
|
+
"""The literal a gold ``proofs`` annotation is ABOUT, given the
|
|
1165
|
+
question's own strategy tag.
|
|
1166
|
+
|
|
1167
|
+
ProofWriter's ``"proof"``/``"rconc"``/``"random"`` strategies derive (or
|
|
1168
|
+
fail to derive) the question's conclusion EXACTLY as generated —
|
|
1169
|
+
whatever polarity the question itself has (a question can be phrased
|
|
1170
|
+
negatively, e.g. ``"The rabbit is not round."``, and be settled by
|
|
1171
|
+
directly citing a NEGATIVE fact — confirmed in real data: 96 of 1329
|
|
1172
|
+
real ``"proof"``-strategy questions fetched here are phrased negatively
|
|
1173
|
+
and their proof directly cites a negative-headed fact/rule). The
|
|
1174
|
+
``"inv-*"`` strategies derive the OPPOSITE of the question instead
|
|
1175
|
+
(that is what makes the label ``"False"``/the "inv-" failure a
|
|
1176
|
+
not-entailed positive/negative pair) — never a double negation, since
|
|
1177
|
+
the question's own polarity is stripped, not added to.
|
|
1178
|
+
"""
|
|
1179
|
+
if strategy not in _KNOWN_STRATEGIES:
|
|
1180
|
+
raise ValueError(
|
|
1181
|
+
f"proofwriter: example {example_id} has meta['strategy'] "
|
|
1182
|
+
f"{strategy!r}, not one of {sorted(_KNOWN_STRATEGIES)} — cannot "
|
|
1183
|
+
"tell which literal the gold proof is about")
|
|
1184
|
+
if strategy.startswith("inv-"):
|
|
1185
|
+
return _negate(conclusion)
|
|
1186
|
+
return conclusion
|
|
1187
|
+
|
|
1188
|
+
|
|
1189
|
+
def _rule_body_holds(rule_node: Node, target_key: str,
|
|
1190
|
+
true_atoms: "FrozenSet[str]",
|
|
1191
|
+
constants: "List[Constant]") -> "Optional[bool]":
|
|
1192
|
+
"""Is there SOME grounding of ``rule_node`` whose HEAD is ``target_key``
|
|
1193
|
+
(a SIGNED atom key — see :func:`_closed_model`'s ``provenance``) with a
|
|
1194
|
+
satisfied body?
|
|
1195
|
+
|
|
1196
|
+
This is an EXISTENTIAL question over every grounding whose head matches
|
|
1197
|
+
(a rule with a variable that occurs in the body but not the head, or
|
|
1198
|
+
otherwise sharing its head atom across more than one grounding, can have
|
|
1199
|
+
several) — all matching groundings are checked, not just the first one
|
|
1200
|
+
``itertools.product`` happens to visit, so a non-firing grounding never
|
|
1201
|
+
masks a later firing one.
|
|
1202
|
+
|
|
1203
|
+
Returns ``None`` when no grounding of ``rule_node`` concludes
|
|
1204
|
+
``target_key`` at all (the rule cannot structurally produce this atom,
|
|
1205
|
+
e.g. under any constant substitution its head is a different
|
|
1206
|
+
predicate/arguments) — a gold "deepest failure" witness naming such a
|
|
1207
|
+
rule cannot be confirmed or refuted this way, which
|
|
1208
|
+
:func:`check_gold_proof` surfaces rather than silently treating as
|
|
1209
|
+
either outcome.
|
|
1210
|
+
"""
|
|
1211
|
+
rule = _as_rule(rule_node)
|
|
1212
|
+
matches = []
|
|
1213
|
+
for g_body, (g_head_atom, g_head_positive) in _ground_rule(rule, constants):
|
|
1214
|
+
signed = (key_text(g_head_atom) if g_head_positive
|
|
1215
|
+
else key_text(Not(g_head_atom)))
|
|
1216
|
+
if signed != target_key:
|
|
1217
|
+
continue
|
|
1218
|
+
matches.append(g_body)
|
|
1219
|
+
if not matches:
|
|
1220
|
+
return None
|
|
1221
|
+
return any(all((key_text(atom) in true_atoms) == positive
|
|
1222
|
+
for atom, positive in g_body)
|
|
1223
|
+
for g_body in matches)
|
|
1224
|
+
|
|
1225
|
+
|
|
1226
|
+
def _match_gold_derivation(node: "_proof.ProofNode", target_key: str,
|
|
1227
|
+
key_to_index: "Dict[str, int]",
|
|
1228
|
+
provenance: "Dict[str, list]") -> bool:
|
|
1229
|
+
"""Does ``node`` (a :class:`~._proofwriter_proof.Leaf` /
|
|
1230
|
+
:class:`~._proofwriter_proof.Naf` / :class:`~._proofwriter_proof.Apply` /
|
|
1231
|
+
:class:`~._proofwriter_proof.Or`) explain how the fixpoint's OWN
|
|
1232
|
+
``provenance`` derived ``target_key``?
|
|
1233
|
+
|
|
1234
|
+
A :class:`~._proofwriter_proof.Leaf` matches iff the fact it cites fired
|
|
1235
|
+
with no antecedents for exactly ``target_key``; a
|
|
1236
|
+
:class:`~._proofwriter_proof.Naf` matches iff ``target_key`` is a
|
|
1237
|
+
NEGATIVE requirement (starts with ``"¬"``) whose bare positive form has
|
|
1238
|
+
NO provenance entry at all — genuine absence, matching negation-as-
|
|
1239
|
+
failure with no explicit ``¬X`` derivation to cite (see
|
|
1240
|
+
:class:`~._proofwriter_proof.Naf`); an :class:`~._proofwriter_proof.Apply`
|
|
1241
|
+
matches iff SOME provenance entry for ``target_key`` used the SAME rule
|
|
1242
|
+
with the SAME antecedent count, each antecedent recursively matching the
|
|
1243
|
+
corresponding gold sub-term (in order — see :func:`_closed_model`'s
|
|
1244
|
+
``provenance`` docstring on why body order is preserved and safe to rely
|
|
1245
|
+
on positionally); an :class:`~._proofwriter_proof.Or` matches iff ANY
|
|
1246
|
+
alternative does (the OR-forest offers several valid supports —
|
|
1247
|
+
matching any one is enough).
|
|
1248
|
+
"""
|
|
1249
|
+
if isinstance(node, _proof.Leaf):
|
|
1250
|
+
index = key_to_index.get(node.ref)
|
|
1251
|
+
if index is None:
|
|
1252
|
+
raise ValueError(
|
|
1253
|
+
f"proofwriter: gold proof cites unknown fact reference "
|
|
1254
|
+
f"{node.ref!r} — not among this example's premise_keys")
|
|
1255
|
+
return (index, ()) in provenance.get(target_key, ())
|
|
1256
|
+
if isinstance(node, _proof.Naf):
|
|
1257
|
+
if not target_key.startswith("¬"):
|
|
1258
|
+
raise ValueError(
|
|
1259
|
+
f"proofwriter: gold proof cites NAF (negation-as-failure) "
|
|
1260
|
+
f"for {target_key!r}, which is not itself a negative "
|
|
1261
|
+
"requirement — outside the documented grammar (NAF only "
|
|
1262
|
+
"ever justifies a '~'-polarity body condition)")
|
|
1263
|
+
return target_key[1:] not in provenance
|
|
1264
|
+
if isinstance(node, _proof.Or):
|
|
1265
|
+
return any(_match_gold_derivation(alt, target_key, key_to_index,
|
|
1266
|
+
provenance)
|
|
1267
|
+
for alt in node.alts)
|
|
1268
|
+
if isinstance(node, _proof.Apply):
|
|
1269
|
+
index = key_to_index.get(node.rule)
|
|
1270
|
+
if index is None:
|
|
1271
|
+
raise ValueError(
|
|
1272
|
+
f"proofwriter: gold proof cites unknown rule reference "
|
|
1273
|
+
f"{node.rule!r} — not among this example's premise_keys")
|
|
1274
|
+
parts = node.args.parts
|
|
1275
|
+
for origin_index, ante_keys in provenance.get(target_key, ()):
|
|
1276
|
+
if origin_index != index or len(ante_keys) != len(parts):
|
|
1277
|
+
continue
|
|
1278
|
+
if all(_match_gold_derivation(sub, ante_keys[i], key_to_index,
|
|
1279
|
+
provenance)
|
|
1280
|
+
for i, sub in enumerate(parts)):
|
|
1281
|
+
return True
|
|
1282
|
+
return False
|
|
1283
|
+
raise ValueError(
|
|
1284
|
+
f"proofwriter: gold proof node {type(node).__name__} is not a "
|
|
1285
|
+
"derivation node (Leaf/Naf/Apply/Or) — a FailWitness at this "
|
|
1286
|
+
"position is handled separately by check_gold_proof, never "
|
|
1287
|
+
"recursed into")
|
|
1288
|
+
|
|
1289
|
+
|
|
1290
|
+
def check_gold_proof(example: DatasetExample) -> dict:
|
|
1291
|
+
"""Verify a structured-route example's gold ``question["proofs"]``
|
|
1292
|
+
against the kit's OWN forward-chaining fixpoint (:func:`_closed_model`
|
|
1293
|
+
with ``record_provenance=True``) — a genuine "same derivation" check,
|
|
1294
|
+
because ProofWriter's own generator and this fixpoint are both doing
|
|
1295
|
+
forward chaining over the SAME ground theory (see the module docstring's
|
|
1296
|
+
CWA section). Unlike :func:`solve_structured_example`, no ATP is
|
|
1297
|
+
invoked — the whole point is a SECOND, independent route to the same
|
|
1298
|
+
ground theory's derivable atoms, so this takes no ``prove_kwargs``.
|
|
1299
|
+
Requires an ``example`` produced by :func:`load_proofwriter_structured`
|
|
1300
|
+
with ``convert_fol=True`` (so ``meta["premise_keys"]``/
|
|
1301
|
+
``meta["proofs"]``/``meta["strategy"]`` and the generated FOL are all
|
|
1302
|
+
present) and without a recorded conversion error.
|
|
1303
|
+
|
|
1304
|
+
Known, narrow disagreement (report, do not repair — see this module's
|
|
1305
|
+
"Independent verification" contract): a real rule body's ``"~"``
|
|
1306
|
+
(negation-as-failure) condition and a fact/rule head's ``"-"`` (strong
|
|
1307
|
+
negation) both lower to the SAME kit ``Not()`` in the generated FOL (see
|
|
1308
|
+
:func:`parse_proofwriter_representation`'s ``_triple_to_node`` —
|
|
1309
|
+
pre-existing, unrelated to this function), so when a ``"~"`` condition
|
|
1310
|
+
is genuinely UNDETERMINED under ProofWriter's own open-world reading
|
|
1311
|
+
(never asserted true OR false) rather than absent-under-closed-world,
|
|
1312
|
+
:func:`_closed_model`'s NAF-as-absence semantics (pre-existing,
|
|
1313
|
+
unrelated to this function, verified 1078/1078 against genuinely
|
|
1314
|
+
CWA-labelled data) can let a rule fire that ProofWriter's own OWA-
|
|
1315
|
+
consistent annotation says should not. Confirmed on 7 of 4550
|
|
1316
|
+
checkable real questions fetched here (0.15%) — every one traced to a
|
|
1317
|
+
rule using ``"~"``; :func:`check_gold_proof` correctly reports these as
|
|
1318
|
+
``ok=False`` rather than silently agreeing, which is the intended
|
|
1319
|
+
behaviour, not a bug in the parser or this checker.
|
|
1320
|
+
|
|
1321
|
+
Two shapes, per :mod:`._proofwriter_proof`'s grammar:
|
|
1322
|
+
|
|
1323
|
+
* A DERIVATION (``strategy`` ``"proof"``/``"inv-proof"``/``"rconc"``/
|
|
1324
|
+
``"random"``/``"inv-rconc"``/``"inv-random"`` whose ``proofs`` parses
|
|
1325
|
+
to a :class:`~._proofwriter_proof.Leaf`/:class:`~._proofwriter_proof.Apply`/
|
|
1326
|
+
:class:`~._proofwriter_proof.Or`): the target literal (see
|
|
1327
|
+
:func:`_target_for_strategy`) must be among the atoms the fixpoint's
|
|
1328
|
+
own provenance says were derived by exactly that named chain of
|
|
1329
|
+
facts/rules (:func:`_match_gold_derivation`).
|
|
1330
|
+
* A :class:`~._proofwriter_proof.FailWitness` (``"Unknown"`` answers):
|
|
1331
|
+
confirms the target literal is genuinely NOT derivable (absent from
|
|
1332
|
+
``provenance``), and — when the witness names a first candidate rule
|
|
1333
|
+
(``rule_chain[0]``) — that THAT rule's own grounding for this target
|
|
1334
|
+
has an unsatisfied body in the fixpoint's COMPLETED perfect model
|
|
1335
|
+
(:func:`_rule_body_holds`). ``true_atoms`` IS the theory's perfect
|
|
1336
|
+
model already (:func:`_closed_model` fully stratifies before
|
|
1337
|
+
returning), so this is the exact ground truth regardless of whether
|
|
1338
|
+
the rule's own body has a negation-as-failure condition — there is no
|
|
1339
|
+
"which round" ambiguity to approximate. **Explicitly NOT verified**:
|
|
1340
|
+
deeper links of a multi-rule failure chain (``rule_chain[1:]`` — real
|
|
1341
|
+
witnesses go up to 5 links deep, see
|
|
1342
|
+
:class:`~._proofwriter_proof.FailWitness`); only the first, named
|
|
1343
|
+
"could this have produced the target" candidate is checked.
|
|
1344
|
+
|
|
1345
|
+
Returns a dict with ``"kind"`` (``"derivation"`` or ``"fail_witness"``),
|
|
1346
|
+
``"ok"`` (bool), ``"atom"`` (the target's signed key) and ``"gold"``
|
|
1347
|
+
(the parsed :data:`~._proofwriter_proof.ProofNode`); a
|
|
1348
|
+
``"fail_witness"`` result additionally carries ``"derivable"`` and
|
|
1349
|
+
``"rule_body_holds"`` (``None`` when ``rule_chain`` is empty or its
|
|
1350
|
+
first rule cannot structurally conclude the target atom).
|
|
1351
|
+
|
|
1352
|
+
Raises:
|
|
1353
|
+
ValueError: the example is missing generated FOL, its
|
|
1354
|
+
``meta["proofs"]``/``meta["premise_keys"]``/``meta["strategy"]``
|
|
1355
|
+
(i.e. it was not produced by :func:`load_proofwriter_structured`
|
|
1356
|
+
with ``convert_fol=True``), the generated conclusion is not a
|
|
1357
|
+
(possibly negated) atom, or the gold annotation cites a
|
|
1358
|
+
``tripleN``/``ruleN`` reference outside this example's own
|
|
1359
|
+
premises — never silently ignored or guessed at.
|
|
1360
|
+
"""
|
|
1361
|
+
from ... import api
|
|
1362
|
+
|
|
1363
|
+
if example.meta.get("fol_conversion_error"):
|
|
1364
|
+
raise ValueError(
|
|
1365
|
+
f"proofwriter: example {example.id} carries a conversion error "
|
|
1366
|
+
f"({example.meta['fol_conversion_error']}) — cannot check its "
|
|
1367
|
+
"proof.")
|
|
1368
|
+
if example.fol_conclusion is None:
|
|
1369
|
+
raise ValueError(
|
|
1370
|
+
f"proofwriter: example {example.id} has no generated conclusion "
|
|
1371
|
+
"— was it loaded with convert_fol=False?")
|
|
1372
|
+
proofs_text = example.meta.get("proofs")
|
|
1373
|
+
if not proofs_text:
|
|
1374
|
+
raise ValueError(
|
|
1375
|
+
f"proofwriter: example {example.id} has no recorded "
|
|
1376
|
+
"meta['proofs'] annotation to check.")
|
|
1377
|
+
premise_keys = example.meta.get("premise_keys")
|
|
1378
|
+
if not premise_keys:
|
|
1379
|
+
raise ValueError(
|
|
1380
|
+
f"proofwriter: example {example.id} has no meta['premise_keys'] "
|
|
1381
|
+
"— only load_proofwriter_structured's output names its "
|
|
1382
|
+
"triples/rules.")
|
|
1383
|
+
|
|
1384
|
+
def _parse(text: str) -> Node:
|
|
1385
|
+
parsed = api.parse_any(text)
|
|
1386
|
+
if not parsed.ok:
|
|
1387
|
+
raise ValueError(
|
|
1388
|
+
f"proofwriter: example {example.id}: generated formula "
|
|
1389
|
+
f"{text!r} does not parse under the kit grammar")
|
|
1390
|
+
return parsed.formula
|
|
1391
|
+
|
|
1392
|
+
premises = [_parse(p) for p in example.fol_premises]
|
|
1393
|
+
conclusion = _parse(example.fol_conclusion)
|
|
1394
|
+
target = _target_for_strategy(conclusion, example.meta.get("strategy"),
|
|
1395
|
+
example.id)
|
|
1396
|
+
target_atom = target.formula if isinstance(target, Not) else target
|
|
1397
|
+
if not isinstance(target_atom, Atom):
|
|
1398
|
+
raise ValueError(
|
|
1399
|
+
f"proofwriter: example {example.id}: target "
|
|
1400
|
+
f"{target.to_unicode_str()!r} is not a (possibly negated) atom "
|
|
1401
|
+
"— cannot check its proof against a ground fixpoint")
|
|
1402
|
+
# The signed key of the target (``¬Foo(a)`` for a negated atom), the form the
|
|
1403
|
+
# fixpoint's provenance is filed under.
|
|
1404
|
+
target_key = key_text(target)
|
|
1405
|
+
|
|
1406
|
+
constants = _collect_constants(premises + [conclusion])
|
|
1407
|
+
true_atoms, _has_naf, provenance = _closed_model(
|
|
1408
|
+
premises, constants, record_provenance=True)
|
|
1409
|
+
key_to_index = {key: i for i, key in enumerate(premise_keys)}
|
|
1410
|
+
gold = _proof.parse_question_proof(proofs_text)
|
|
1411
|
+
|
|
1412
|
+
if isinstance(gold, _proof.FailWitness):
|
|
1413
|
+
derivable = target_key in provenance
|
|
1414
|
+
rule_body_holds: "Optional[bool]" = None
|
|
1415
|
+
if gold.rule_chain:
|
|
1416
|
+
rule_ref = gold.rule_chain[0]
|
|
1417
|
+
rule_index = key_to_index.get(rule_ref)
|
|
1418
|
+
if rule_index is None:
|
|
1419
|
+
raise ValueError(
|
|
1420
|
+
f"proofwriter: gold failure witness cites unknown rule "
|
|
1421
|
+
f"reference {rule_ref!r} — not among this example's "
|
|
1422
|
+
"premise_keys")
|
|
1423
|
+
rule_body_holds = _rule_body_holds(
|
|
1424
|
+
premises[rule_index], target_key, true_atoms, constants)
|
|
1425
|
+
ok = (not derivable) and (rule_body_holds is not True)
|
|
1426
|
+
return {"kind": "fail_witness", "ok": ok, "atom": target_key,
|
|
1427
|
+
"gold": gold, "derivable": derivable,
|
|
1428
|
+
"rule_body_holds": rule_body_holds}
|
|
1429
|
+
|
|
1430
|
+
ok = _match_gold_derivation(gold, target_key, key_to_index, provenance)
|
|
1431
|
+
return {"kind": "derivation", "ok": ok, "atom": target_key, "gold": gold}
|